mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-21 00:21:49 +00:00
Merge remote-tracking branch 'origin/main' into litellm_mcp_client_allowlist
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> # Conflicts: # litellm/proxy/_experimental/mcp_server/server.py
This commit is contained in:
commit
9917375d4b
124 changed files with 13279 additions and 1008 deletions
3
.github/workflows/osv-scan.yml
vendored
3
.github/workflows/osv-scan.yml
vendored
|
|
@ -41,4 +41,5 @@ jobs:
|
|||
"$RUNNER_TEMP/osv-scanner" scan source \
|
||||
--config osv-scanner.toml \
|
||||
-L uv.lock \
|
||||
-L ui/litellm-dashboard/package-lock.json
|
||||
-L ui/litellm-dashboard/package-lock.json \
|
||||
-L vscode-extension/package-lock.json
|
||||
|
|
|
|||
65
.github/workflows/test-vscode-extension.yml
vendored
Normal file
65
.github/workflows/test-vscode-extension.yml
vendored
Normal file
|
|
@ -0,0 +1,65 @@
|
|||
name: VS Code Extension
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
- litellm_internal_staging
|
||||
- litellm_oss_staging
|
||||
- "litellm_**"
|
||||
paths:
|
||||
- "vscode-extension/**"
|
||||
- ".github/workflows/test-vscode-extension.yml"
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- "vscode-extension/**"
|
||||
- ".github/workflows/test-vscode-extension.yml"
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.sha }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
jobs:
|
||||
vscode-extension:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
defaults:
|
||||
run:
|
||||
working-directory: vscode-extension
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
||||
with:
|
||||
fetch-depth: 1
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Node.js
|
||||
uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5.0.0
|
||||
with:
|
||||
node-version: "24"
|
||||
cache: npm
|
||||
cache-dependency-path: vscode-extension/package-lock.json
|
||||
|
||||
- name: Install dependencies
|
||||
run: npm ci
|
||||
|
||||
- name: Typecheck
|
||||
run: npm run typecheck
|
||||
|
||||
- name: Unit tests
|
||||
run: npm test
|
||||
|
||||
- name: Package extension
|
||||
run: npm run package
|
||||
|
||||
- name: Upload VSIX
|
||||
uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1
|
||||
with:
|
||||
name: litellm-vscode
|
||||
path: vscode-extension/*.vsix
|
||||
if-no-files-found: error
|
||||
|
|
@ -28,6 +28,7 @@
|
|||
"structured-outputs-2025-11-13": "structured-outputs-2025-11-13",
|
||||
"text_editor_20241022": null,
|
||||
"text_editor_20250124": null,
|
||||
"thinking-binding-controls-2026-08-01": "thinking-binding-controls-2026-08-01",
|
||||
"token-efficient-tools-2025-02-19": "token-efficient-tools-2025-02-19",
|
||||
"web-fetch-2025-09-10": "web-fetch-2025-09-10",
|
||||
"web-search-2025-03-05": "web-search-2025-03-05"
|
||||
|
|
@ -59,6 +60,7 @@
|
|||
"structured-outputs-2025-11-13": "structured-outputs-2025-11-13",
|
||||
"text_editor_20241022": null,
|
||||
"text_editor_20250124": null,
|
||||
"thinking-binding-controls-2026-08-01": null,
|
||||
"token-efficient-tools-2025-02-19": null,
|
||||
"web-fetch-2025-09-10": "web-fetch-2025-09-10",
|
||||
"web-search-2025-03-05": "web-search-2025-03-05"
|
||||
|
|
@ -90,6 +92,7 @@
|
|||
"structured-outputs-2025-11-13": "structured-outputs-2025-11-13",
|
||||
"text_editor_20241022": null,
|
||||
"text_editor_20250124": null,
|
||||
"thinking-binding-controls-2026-08-01": "thinking-binding-controls-2026-08-01",
|
||||
"token-efficient-tools-2025-02-19": null,
|
||||
"tool-search-tool-2025-10-19": null,
|
||||
"web-fetch-2025-09-10": null,
|
||||
|
|
@ -122,6 +125,7 @@
|
|||
"structured-outputs-2025-11-13": null,
|
||||
"text_editor_20241022": null,
|
||||
"text_editor_20250124": null,
|
||||
"thinking-binding-controls-2026-08-01": "thinking-binding-controls-2026-08-01",
|
||||
"token-efficient-tools-2025-02-19": null,
|
||||
"tool-search-tool-2025-10-19": "tool-search-tool-2025-10-19",
|
||||
"web-fetch-2025-09-10": null,
|
||||
|
|
@ -154,6 +158,7 @@
|
|||
"structured-outputs-2025-11-13": null,
|
||||
"text_editor_20241022": null,
|
||||
"text_editor_20250124": null,
|
||||
"thinking-binding-controls-2026-08-01": "thinking-binding-controls-2026-08-01",
|
||||
"token-efficient-tools-2025-02-19": null,
|
||||
"tool-search-tool-2025-10-19": "tool-search-tool-2025-10-19",
|
||||
"web-fetch-2025-09-10": null,
|
||||
|
|
@ -187,6 +192,7 @@
|
|||
"structured-outputs-2025-11-13": "structured-outputs-2025-11-13",
|
||||
"text_editor_20241022": null,
|
||||
"text_editor_20250124": null,
|
||||
"thinking-binding-controls-2026-08-01": "thinking-binding-controls-2026-08-01",
|
||||
"token-efficient-tools-2025-02-19": "token-efficient-tools-2025-02-19",
|
||||
"web-fetch-2025-09-10": "web-fetch-2025-09-10",
|
||||
"web-search-2025-03-05": "web-search-2025-03-05"
|
||||
|
|
|
|||
|
|
@ -183,6 +183,9 @@ MCP_TOOL_LISTING_TIMEOUT: Final = float(os.getenv("LITELLM_MCP_TOOL_LISTING_TIME
|
|||
MCP_METADATA_TIMEOUT: Final = float(os.getenv("LITELLM_MCP_METADATA_TIMEOUT", "10.0"))
|
||||
MCP_HEALTH_CHECK_TIMEOUT: Final = float(os.getenv("LITELLM_MCP_HEALTH_CHECK_TIMEOUT", "10.0"))
|
||||
MCP_TOOL_LISTING_MAX_PAGES: Final = 1000
|
||||
MCP_GATEWAY_SESSION_ID_PREFIX_LENGTH: Final = 8
|
||||
MCP_BYOK_CREDENTIAL_CACHE_TTL_SECONDS: Final = 60
|
||||
MCP_BYOK_CREDENTIAL_CACHE_MAX_SIZE: Final = 4096
|
||||
|
||||
# Allowlist of commands permitted for MCP stdio transport.
|
||||
# Prevents arbitrary command execution via /mcp-rest/test/* endpoints or server creation.
|
||||
|
|
@ -1657,6 +1660,11 @@ LITELLM_EXPIRED_UI_SESSION_KEY_CLEANUP_INTERVAL_SECONDS: Final = int(
|
|||
LITELLM_EXPIRED_UI_SESSION_KEY_CLEANUP_BATCH_SIZE: Final = int(
|
||||
os.getenv("LITELLM_EXPIRED_UI_SESSION_KEY_CLEANUP_BATCH_SIZE", 1000)
|
||||
)
|
||||
LOGIN_THROTTLE_CACHE_KEY_PREFIX: Final = "login_fail"
|
||||
LOGIN_THROTTLE_UNKNOWN_SOURCE: Final = "unknown"
|
||||
LOGIN_THROTTLE_MAX_TRACKED_COUNTERS: Final = 20_000
|
||||
LOGIN_THROTTLE_MAX_TRACKED_BLOCKS: Final = 10_000
|
||||
LOGIN_THROTTLE_NOT_BLOCKED: Final = (0, 0)
|
||||
LITELLM_PROXY_ADMIN_NAME: Final = "default_user_id"
|
||||
LITELLM_PROXY_BUDGET_NAME: Final = "litellm-proxy-budget"
|
||||
GLOBAL_PROXY_SPEND_CACHE_KEY: Final = f"{LITELLM_PROXY_ADMIN_NAME}:spend"
|
||||
|
|
@ -2049,6 +2057,7 @@ MCP_SPEND_LOG_MODEL_PREFIX: Final[str] = "MCP: "
|
|||
PTU_SENTINEL_API_KEY: Final[str] = "__ptu_flat_cost__"
|
||||
PTU_ROLLUP_JOB_ID: Final[str] = "ptu_flat_cost_rollup_job"
|
||||
PTU_ROLLUP_LOCK_TTL_SECONDS: Final[int] = 900
|
||||
USAGE_TOP_API_KEYS_LIMIT: Final[int] = int(os.getenv("USAGE_TOP_API_KEYS_LIMIT", "100"))
|
||||
# Furthest back the catch-up pass looks for unpriced PTU days when a deployment
|
||||
# declares no ptu_effective_from, bounding the scan for an open-ended window.
|
||||
PTU_ROLLUP_MAX_BACKFILL_DAYS: Final[int] = 90
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ import json
|
|||
import os
|
||||
import re
|
||||
import urllib.parse
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
from collections.abc import Callable, Mapping, MutableMapping, Sequence
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from datetime import datetime
|
||||
from functools import partial
|
||||
|
|
@ -33,7 +33,7 @@ from litellm.constants import (
|
|||
from litellm.litellm_core_utils.aws_partition import contains_bedrock_arn, get_aws_dns_suffix
|
||||
from litellm.litellm_core_utils.dd_tracing import tracer
|
||||
from litellm.secret_managers.main import get_secret, get_secret_str
|
||||
from litellm.types.llms.bedrock import AwsSessionTag
|
||||
from litellm.types.llms.bedrock import AWS_AUTH_PARAM_KEYS, AwsAuthParams, AwsSessionTag
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from botocore.awsrequest import AWSPreparedRequest
|
||||
|
|
@ -168,6 +168,14 @@ def build_web_identity_session_policy() -> WebIdentitySessionPolicy:
|
|||
)
|
||||
|
||||
|
||||
def pop_aws_auth_params(
|
||||
optional_params: MutableMapping[str, object], # mutable-ok: pops the aws_* keys out of the caller's mapping
|
||||
) -> AwsAuthParams:
|
||||
return AwsAuthParams.model_validate(
|
||||
MappingProxyType({key: optional_params.pop(key, None) for key in AWS_AUTH_PARAM_KEYS})
|
||||
)
|
||||
|
||||
|
||||
class BedrockRequestTarget(BaseModel):
|
||||
aws_region_name: str
|
||||
aws_bedrock_runtime_endpoint: str | None
|
||||
|
|
@ -501,6 +509,21 @@ class BaseAWSLLM(SignsRequestsWithAWS):
|
|||
else:
|
||||
return self._get_or_set_cached_credentials(args, self._auth_with_env_vars)
|
||||
|
||||
def resolve_credentials(self, auth_params: AwsAuthParams, aws_region_name: str | None) -> Credentials:
|
||||
return self.get_credentials(
|
||||
aws_access_key_id=auth_params.aws_access_key_id,
|
||||
aws_secret_access_key=auth_params.aws_secret_access_key,
|
||||
aws_session_token=auth_params.aws_session_token,
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=auth_params.aws_session_name,
|
||||
aws_profile_name=auth_params.aws_profile_name,
|
||||
aws_role_name=auth_params.aws_role_name,
|
||||
aws_web_identity_token=auth_params.aws_web_identity_token,
|
||||
aws_sts_endpoint=auth_params.aws_sts_endpoint,
|
||||
aws_external_id=auth_params.aws_external_id,
|
||||
aws_session_tags=_canonical_aws_session_tags(auth_params.aws_session_tags),
|
||||
)
|
||||
|
||||
def _get_aws_region_from_model_arn(self, model: str | None) -> str | None:
|
||||
try:
|
||||
# First check if the string contains the expected prefix
|
||||
|
|
@ -1515,23 +1538,10 @@ class BaseAWSLLM(SignsRequestsWithAWS):
|
|||
from botocore.credentials import Credentials
|
||||
except ImportError:
|
||||
raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
|
||||
## CREDENTIALS ##
|
||||
# pop aws_secret_access_key, aws_access_key_id, aws_region_name from kwargs, since completion calls fail with them
|
||||
aws_secret_access_key: Final = optional_params.pop("aws_secret_access_key", None)
|
||||
aws_access_key_id: Final = optional_params.pop("aws_access_key_id", None)
|
||||
aws_session_token: Final = optional_params.pop("aws_session_token", None)
|
||||
aws_region_name: Final = self._get_aws_region_name(optional_params, model)
|
||||
optional_params.pop("aws_region_name", None)
|
||||
aws_role_name: Final = optional_params.pop("aws_role_name", None)
|
||||
aws_session_name: Final = optional_params.pop("aws_session_name", None)
|
||||
aws_profile_name: Final = optional_params.pop("aws_profile_name", None)
|
||||
aws_web_identity_token: Final = optional_params.pop("aws_web_identity_token", None)
|
||||
aws_sts_endpoint: Final = optional_params.pop("aws_sts_endpoint", None)
|
||||
aws_bedrock_runtime_endpoint: Final = optional_params.pop(
|
||||
"aws_bedrock_runtime_endpoint", None
|
||||
) # https://bedrock-runtime.{region_name}.amazonaws.com
|
||||
aws_external_id: Final = optional_params.pop("aws_external_id", None)
|
||||
aws_session_tags: Final = optional_params.pop("aws_session_tags", None)
|
||||
auth_params: Final = pop_aws_auth_params(optional_params)
|
||||
aws_bedrock_runtime_endpoint: Final = optional_params.pop("aws_bedrock_runtime_endpoint", None)
|
||||
|
||||
if bearer_token is not None:
|
||||
return BearerRequestTarget(
|
||||
|
|
@ -1539,19 +1549,7 @@ class BaseAWSLLM(SignsRequestsWithAWS):
|
|||
aws_bedrock_runtime_endpoint=aws_bedrock_runtime_endpoint,
|
||||
)
|
||||
|
||||
credentials: Final[Credentials] = self.get_credentials(
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_session_token=aws_session_token,
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=aws_session_name,
|
||||
aws_profile_name=aws_profile_name,
|
||||
aws_role_name=aws_role_name,
|
||||
aws_web_identity_token=aws_web_identity_token,
|
||||
aws_sts_endpoint=aws_sts_endpoint,
|
||||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
)
|
||||
credentials: Final[Credentials] = self.resolve_credentials(auth_params, aws_region_name)
|
||||
return Boto3CredentialsInfo(
|
||||
credentials=credentials,
|
||||
aws_region_name=aws_region_name,
|
||||
|
|
@ -1685,33 +1683,9 @@ class BaseAWSLLM(SignsRequestsWithAWS):
|
|||
except ImportError:
|
||||
raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
|
||||
|
||||
## CREDENTIALS ##
|
||||
# pop aws_secret_access_key, aws_access_key_id, aws_session_token, aws_region_name from kwargs, since completion calls fail with them
|
||||
aws_secret_access_key: Final = optional_params.get("aws_secret_access_key", None)
|
||||
aws_access_key_id: Final = optional_params.get("aws_access_key_id", None)
|
||||
aws_session_token: Final = optional_params.get("aws_session_token", None)
|
||||
aws_role_name: Final = optional_params.get("aws_role_name", None)
|
||||
aws_session_name: Final = optional_params.get("aws_session_name", None)
|
||||
aws_profile_name: Final = optional_params.get("aws_profile_name", None)
|
||||
aws_web_identity_token: Final = optional_params.get("aws_web_identity_token", None)
|
||||
aws_sts_endpoint: Final = optional_params.get("aws_sts_endpoint", None)
|
||||
aws_external_id: Final = optional_params.get("aws_external_id", None)
|
||||
aws_session_tags: Final = optional_params.get("aws_session_tags", None)
|
||||
auth_params: Final = AwsAuthParams.model_validate(optional_params)
|
||||
aws_region_name: Final = self._get_aws_region_name(optional_params=optional_params, model=model)
|
||||
|
||||
credentials: Final[Credentials] = self.get_credentials(
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_session_token=aws_session_token,
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=aws_session_name,
|
||||
aws_profile_name=aws_profile_name,
|
||||
aws_role_name=aws_role_name,
|
||||
aws_web_identity_token=aws_web_identity_token,
|
||||
aws_sts_endpoint=aws_sts_endpoint,
|
||||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
)
|
||||
credentials: Final[Credentials] = self.resolve_credentials(auth_params, aws_region_name)
|
||||
|
||||
sigv4: Final = SigV4Auth(credentials, service_name, aws_region_name)
|
||||
headers = headers or {}
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ from openai.types.batch import BatchRequestCounts
|
|||
from openai.types.batch import Metadata as OpenAIBatchMetadata
|
||||
|
||||
from litellm.litellm_core_utils.aws_partition import get_aws_dns_suffix
|
||||
from litellm.types.llms.bedrock import AwsSessionTag
|
||||
from litellm.types.llms.bedrock import AwsAuthParams, AwsSessionTag
|
||||
from litellm.types.utils import LiteLLMBatch
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
|
@ -130,11 +130,10 @@ class BedrockBatchesHandler:
|
|||
|
||||
from litellm.llms.bedrock.batches.transformation import BedrockBatchesConfig
|
||||
|
||||
creds: Final = BedrockBatchesConfig().get_credentials(
|
||||
auth_params: Final = AwsAuthParams(
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_session_token=aws_session_token,
|
||||
aws_region_name=region,
|
||||
aws_session_name=aws_session_name,
|
||||
aws_profile_name=aws_profile_name,
|
||||
aws_role_name=aws_role_name,
|
||||
|
|
@ -143,6 +142,7 @@ class BedrockBatchesHandler:
|
|||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
)
|
||||
creds: Final = BedrockBatchesConfig().resolve_credentials(auth_params, region)
|
||||
|
||||
client: Final = boto3.client(
|
||||
"bedrock",
|
||||
|
|
@ -157,16 +157,7 @@ class BedrockBatchesHandler:
|
|||
batch_id=batch_id,
|
||||
aws_region_name=region,
|
||||
logging_obj=logging_obj,
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_session_token=aws_session_token,
|
||||
aws_session_name=aws_session_name,
|
||||
aws_profile_name=aws_profile_name,
|
||||
aws_role_name=aws_role_name,
|
||||
aws_web_identity_token=aws_web_identity_token,
|
||||
aws_sts_endpoint=aws_sts_endpoint,
|
||||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
**auth_params.model_dump(),
|
||||
)
|
||||
|
||||
try:
|
||||
|
|
@ -310,19 +301,7 @@ class BedrockBatchesHandler:
|
|||
# BaseAWSLLM) lazily to avoid a circular import at module load.
|
||||
from litellm.llms.bedrock.batches.transformation import BedrockBatchesConfig
|
||||
|
||||
creds: Final = BedrockBatchesConfig().get_credentials(
|
||||
aws_access_key_id=kwargs.get("aws_access_key_id"),
|
||||
aws_secret_access_key=kwargs.get("aws_secret_access_key"),
|
||||
aws_session_token=kwargs.get("aws_session_token"),
|
||||
aws_region_name=region,
|
||||
aws_session_name=kwargs.get("aws_session_name"),
|
||||
aws_profile_name=kwargs.get("aws_profile_name"),
|
||||
aws_role_name=kwargs.get("aws_role_name"),
|
||||
aws_web_identity_token=kwargs.get("aws_web_identity_token"),
|
||||
aws_sts_endpoint=kwargs.get("aws_sts_endpoint"),
|
||||
aws_external_id=kwargs.get("aws_external_id"),
|
||||
aws_session_tags=kwargs.get("aws_session_tags"),
|
||||
)
|
||||
creds: Final = BedrockBatchesConfig().resolve_credentials(AwsAuthParams.model_validate(kwargs), region)
|
||||
|
||||
client: Final = boto3.client(
|
||||
"bedrock",
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ from litellm.llms.custom_httpx.http_handler import (
|
|||
from litellm.types.utils import ModelResponse
|
||||
from litellm.utils import CustomStreamWrapper
|
||||
|
||||
from ..base_aws_llm import BaseAWSLLM, Credentials, bedrock_bearer_token, run_aws_signing
|
||||
from ..base_aws_llm import BaseAWSLLM, Credentials, bedrock_bearer_token, pop_aws_auth_params, run_aws_signing
|
||||
from ..common_utils import BedrockError, _get_all_bedrock_regions, error_response_text
|
||||
from .invoke_handler import AWSEventStreamDecoder, MockResponseIterator, make_call
|
||||
|
||||
|
|
@ -323,21 +323,8 @@ class BedrockConverseLLM(BaseAWSLLM):
|
|||
model_id=unencoded_model_id,
|
||||
)
|
||||
|
||||
## CREDENTIALS ##
|
||||
# pop aws_secret_access_key, aws_access_key_id, aws_region_name from kwargs, since completion calls fail with them
|
||||
aws_secret_access_key: Final = optional_params.pop("aws_secret_access_key", None)
|
||||
aws_access_key_id: Final = optional_params.pop("aws_access_key_id", None)
|
||||
aws_session_token: Final = optional_params.pop("aws_session_token", None)
|
||||
aws_role_name: Final = optional_params.pop("aws_role_name", None)
|
||||
aws_session_name: Final = optional_params.pop("aws_session_name", None)
|
||||
aws_profile_name: Final = optional_params.pop("aws_profile_name", None)
|
||||
aws_bedrock_runtime_endpoint: Final = optional_params.pop(
|
||||
"aws_bedrock_runtime_endpoint", None
|
||||
) # https://bedrock-runtime.{region_name}.amazonaws.com
|
||||
aws_web_identity_token: Final = optional_params.pop("aws_web_identity_token", None)
|
||||
aws_sts_endpoint: Final = optional_params.pop("aws_sts_endpoint", None)
|
||||
aws_external_id: Final = optional_params.pop("aws_external_id", None)
|
||||
aws_session_tags: Final = optional_params.pop("aws_session_tags", None)
|
||||
auth_params: Final = pop_aws_auth_params(optional_params)
|
||||
aws_bedrock_runtime_endpoint: Final = optional_params.pop("aws_bedrock_runtime_endpoint", None)
|
||||
optional_params.pop("aws_region_name", None)
|
||||
|
||||
litellm_params["aws_region_name"] = aws_region_name # [DO NOT DELETE] important for async calls
|
||||
|
|
@ -345,19 +332,7 @@ class BedrockConverseLLM(BaseAWSLLM):
|
|||
credentials: Final[Credentials | None] = (
|
||||
None
|
||||
if bedrock_bearer_token(api_key) is not None
|
||||
else self.get_credentials(
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_session_token=aws_session_token,
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=aws_session_name,
|
||||
aws_profile_name=aws_profile_name,
|
||||
aws_role_name=aws_role_name,
|
||||
aws_web_identity_token=aws_web_identity_token,
|
||||
aws_sts_endpoint=aws_sts_endpoint,
|
||||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
)
|
||||
else self.resolve_credentials(auth_params, aws_region_name)
|
||||
)
|
||||
|
||||
### SET RUNTIME ENDPOINT ###
|
||||
|
|
|
|||
|
|
@ -28,6 +28,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import (
|
|||
from litellm.llms.base_llm.base_utils import BaseLLMModelInfo, BaseTokenCounter
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.secret_managers.main import get_secret, get_secret_str
|
||||
from litellm.types.llms.bedrock import AWS_AUTH_PARAM_KEYS, AwsAuthParams
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.types.llms.openai import AllMessageValues
|
||||
|
|
@ -83,19 +84,7 @@ class BedrockError(BaseLLMException):
|
|||
)
|
||||
|
||||
|
||||
_BEDROCK_AWS_AUTH_PARAMETER_KEYS: Final[tuple[str, ...]] = (
|
||||
"aws_access_key_id",
|
||||
"aws_secret_access_key",
|
||||
"aws_session_token",
|
||||
"aws_region_name",
|
||||
"aws_session_name",
|
||||
"aws_profile_name",
|
||||
"aws_role_name",
|
||||
"aws_web_identity_token",
|
||||
"aws_sts_endpoint",
|
||||
"aws_external_id",
|
||||
"aws_session_tags",
|
||||
)
|
||||
_BEDROCK_AWS_AUTH_PARAMETER_KEYS: Final[tuple[str, ...]] = (*AWS_AUTH_PARAM_KEYS, "aws_region_name")
|
||||
|
||||
|
||||
def merge_bedrock_aws_request_params(
|
||||
|
|
@ -1669,20 +1658,9 @@ class CommonBatchFilesUtils:
|
|||
except ImportError:
|
||||
raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
|
||||
|
||||
# Get AWS credentials using existing methods
|
||||
aws_region_name: Final = self._base_aws._get_aws_region_name(optional_params=optional_params, model="")
|
||||
credentials: Final = self._base_aws.get_credentials(
|
||||
aws_access_key_id=optional_params.get("aws_access_key_id"),
|
||||
aws_secret_access_key=optional_params.get("aws_secret_access_key"),
|
||||
aws_session_token=optional_params.get("aws_session_token"),
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=optional_params.get("aws_session_name"),
|
||||
aws_profile_name=optional_params.get("aws_profile_name"),
|
||||
aws_role_name=optional_params.get("aws_role_name"),
|
||||
aws_web_identity_token=optional_params.get("aws_web_identity_token"),
|
||||
aws_sts_endpoint=optional_params.get("aws_sts_endpoint"),
|
||||
aws_external_id=optional_params.get("aws_external_id"),
|
||||
aws_session_tags=optional_params.get("aws_session_tags"),
|
||||
credentials: Final = self._base_aws.resolve_credentials(
|
||||
AwsAuthParams.model_validate(optional_params), aws_region_name
|
||||
)
|
||||
|
||||
# Prepare the request data
|
||||
|
|
|
|||
|
|
@ -26,7 +26,14 @@ from litellm.types.llms.bedrock import (
|
|||
)
|
||||
from litellm.types.utils import EmbeddingResponse, LlmProviders
|
||||
|
||||
from ..base_aws_llm import AWSPreparedRequest, BaseAWSLLM, Credentials, bedrock_bearer_token, run_aws_signing
|
||||
from ..base_aws_llm import (
|
||||
AWSPreparedRequest,
|
||||
BaseAWSLLM,
|
||||
Credentials,
|
||||
bedrock_bearer_token,
|
||||
pop_aws_auth_params,
|
||||
run_aws_signing,
|
||||
)
|
||||
from ..common_utils import BedrockError
|
||||
from .amazon_nova_transformation import AmazonNovaEmbeddingConfig
|
||||
from .amazon_titan_g1_transformation import AmazonTitanG1Config
|
||||
|
|
@ -75,19 +82,8 @@ class BedrockEmbedding(BaseAWSLLM):
|
|||
optional_params: dict,
|
||||
bearer_token: str | None = None,
|
||||
) -> tuple[Credentials | None, str]:
|
||||
## CREDENTIALS ##
|
||||
# pop aws_secret_access_key, aws_access_key_id, aws_session_token, aws_region_name from kwargs, since completion calls fail with them
|
||||
aws_secret_access_key: Final = optional_params.pop("aws_secret_access_key", None)
|
||||
aws_access_key_id: Final = optional_params.pop("aws_access_key_id", None)
|
||||
aws_session_token: Final = optional_params.pop("aws_session_token", None)
|
||||
auth_params: Final = pop_aws_auth_params(optional_params)
|
||||
aws_region_name = optional_params.pop("aws_region_name", None)
|
||||
aws_role_name: Final = optional_params.pop("aws_role_name", None)
|
||||
aws_session_name: Final = optional_params.pop("aws_session_name", None)
|
||||
aws_profile_name: Final = optional_params.pop("aws_profile_name", None)
|
||||
aws_web_identity_token: Final = optional_params.pop("aws_web_identity_token", None)
|
||||
aws_sts_endpoint: Final = optional_params.pop("aws_sts_endpoint", None)
|
||||
aws_external_id: Final = optional_params.pop("aws_external_id", None)
|
||||
aws_session_tags: Final = optional_params.pop("aws_session_tags", None)
|
||||
|
||||
### SET REGION NAME ###
|
||||
if aws_region_name is None:
|
||||
|
|
@ -105,21 +101,7 @@ class BedrockEmbedding(BaseAWSLLM):
|
|||
aws_region_name = "us-west-2"
|
||||
|
||||
credentials: Final[Credentials | None] = (
|
||||
None
|
||||
if bearer_token is not None
|
||||
else self.get_credentials(
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_session_token=aws_session_token,
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=aws_session_name,
|
||||
aws_profile_name=aws_profile_name,
|
||||
aws_role_name=aws_role_name,
|
||||
aws_web_identity_token=aws_web_identity_token,
|
||||
aws_sts_endpoint=aws_sts_endpoint,
|
||||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
)
|
||||
None if bearer_token is not None else self.resolve_credentials(auth_params, aws_region_name)
|
||||
)
|
||||
return credentials, aws_region_name
|
||||
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ from litellm.litellm_core_utils.cloud_storage_security import (
|
|||
validate_managed_cloud_file_id,
|
||||
)
|
||||
from litellm.llms.custom_httpx.http_handler import get_async_httpx_client
|
||||
from litellm.types.llms.bedrock import AwsAuthParams
|
||||
from litellm.types.llms.openai import (
|
||||
FileContentRequest,
|
||||
HttpxBinaryResponseContent,
|
||||
|
|
@ -101,19 +102,9 @@ class BedrockFilesHandler(BaseAWSLLM):
|
|||
allow_legacy_cloud_file_ids=should_allow_legacy_cloud_file_ids(optional_params),
|
||||
)
|
||||
|
||||
# Get AWS credentials
|
||||
aws_region_name: Final = self._get_aws_region_name(optional_params=optional_params, model="")
|
||||
credentials: Final[Credentials] = self.get_credentials(
|
||||
aws_access_key_id=optional_params.get("aws_access_key_id"),
|
||||
aws_secret_access_key=optional_params.get("aws_secret_access_key"),
|
||||
aws_session_token=optional_params.get("aws_session_token"),
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=optional_params.get("aws_session_name"),
|
||||
aws_profile_name=optional_params.get("aws_profile_name"),
|
||||
aws_role_name=optional_params.get("aws_role_name"),
|
||||
aws_web_identity_token=optional_params.get("aws_web_identity_token"),
|
||||
aws_sts_endpoint=optional_params.get("aws_sts_endpoint"),
|
||||
aws_external_id=optional_params.get("aws_external_id"),
|
||||
credentials: Final[Credentials] = self.resolve_credentials(
|
||||
AwsAuthParams.model_validate(optional_params), aws_region_name
|
||||
)
|
||||
|
||||
# Create S3 client
|
||||
|
|
|
|||
|
|
@ -46,7 +46,7 @@ from litellm.llms.base_llm.files.transformation import (
|
|||
BaseFilesConfig,
|
||||
LiteLLMLoggingObj,
|
||||
)
|
||||
from litellm.types.llms.bedrock import BedrockBatchRecordKind
|
||||
from litellm.types.llms.bedrock import AwsAuthParams, BedrockBatchRecordKind
|
||||
from litellm.types.llms.openai import (
|
||||
AllMessageValues,
|
||||
CreateFileRequest,
|
||||
|
|
@ -142,21 +142,10 @@ def _responses_request_adapter() -> TypeAdapter[ResponsesAPIOptionalRequestParam
|
|||
return TypeAdapter(ResponsesAPIOptionalRequestParams)
|
||||
|
||||
|
||||
class _BedrockS3RequestParams(BaseModel):
|
||||
class _BedrockS3RequestParams(AwsAuthParams):
|
||||
"""Typed view of the credential/region params the S3 GetObject path reads."""
|
||||
|
||||
model_config = ConfigDict(extra="ignore")
|
||||
|
||||
aws_access_key_id: str | None = None
|
||||
aws_secret_access_key: str | None = None
|
||||
aws_session_token: str | None = None
|
||||
aws_region_name: str | None = None
|
||||
aws_session_name: str | None = None
|
||||
aws_profile_name: str | None = None
|
||||
aws_role_name: str | None = None
|
||||
aws_web_identity_token: str | None = None
|
||||
aws_sts_endpoint: str | None = None
|
||||
aws_external_id: str | None = None
|
||||
s3_region_name: str | None = None
|
||||
s3_endpoint_url: str | None = None
|
||||
|
||||
|
|
@ -1157,20 +1146,8 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
except ImportError:
|
||||
raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
|
||||
|
||||
# Get AWS credentials using existing methods
|
||||
aws_region_name: Final = self._get_aws_region_name(optional_params=optional_params, model="")
|
||||
credentials: Final = self.get_credentials(
|
||||
aws_access_key_id=optional_params.get("aws_access_key_id"),
|
||||
aws_secret_access_key=optional_params.get("aws_secret_access_key"),
|
||||
aws_session_token=optional_params.get("aws_session_token"),
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=optional_params.get("aws_session_name"),
|
||||
aws_profile_name=optional_params.get("aws_profile_name"),
|
||||
aws_role_name=optional_params.get("aws_role_name"),
|
||||
aws_web_identity_token=optional_params.get("aws_web_identity_token"),
|
||||
aws_sts_endpoint=optional_params.get("aws_sts_endpoint"),
|
||||
aws_external_id=optional_params.get("aws_external_id"),
|
||||
)
|
||||
credentials: Final = self.resolve_credentials(AwsAuthParams.model_validate(optional_params), aws_region_name)
|
||||
|
||||
# Calculate SHA256 hash of the content (REQUIRED for S3)
|
||||
content_hash: Final = hashlib.sha256(content.encode("utf-8")).hexdigest()
|
||||
|
|
@ -1517,18 +1494,7 @@ class BedrockFilesConfig(BaseAWSLLM, BaseFilesConfig):
|
|||
except ImportError:
|
||||
raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
|
||||
|
||||
credentials: Final = self.get_credentials( # any-ok: boto3 Credentials is untyped
|
||||
aws_access_key_id=request_params.aws_access_key_id,
|
||||
aws_secret_access_key=request_params.aws_secret_access_key,
|
||||
aws_session_token=request_params.aws_session_token,
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=request_params.aws_session_name,
|
||||
aws_profile_name=request_params.aws_profile_name,
|
||||
aws_role_name=request_params.aws_role_name,
|
||||
aws_web_identity_token=request_params.aws_web_identity_token,
|
||||
aws_sts_endpoint=request_params.aws_sts_endpoint,
|
||||
aws_external_id=request_params.aws_external_id,
|
||||
)
|
||||
credentials: Final = self.resolve_credentials(request_params, aws_region_name)
|
||||
|
||||
empty_body_hash: Final = hashlib.sha256(b"").hexdigest()
|
||||
aws_request: Final = AWSRequest( # any-ok: botocore AWSRequest is untyped
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ from litellm.litellm_core_utils.aws_partition import get_aws_dns_suffix
|
|||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLogging
|
||||
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
|
||||
from litellm.litellm_core_utils.realtime_streaming import DefaultLoggedRealTimeEventTypes
|
||||
from litellm.types.llms.bedrock import AwsAuthParams
|
||||
from litellm.types.llms.openai import OpenAIRealtimeEvents
|
||||
from litellm.types.realtime import RealtimeResponseTransformInput
|
||||
|
||||
|
|
@ -257,6 +258,7 @@ class BedrockRealtime(BaseAWSLLM):
|
|||
aws_sts_endpoint: str | None = None,
|
||||
aws_bedrock_runtime_endpoint: str | None = None,
|
||||
aws_external_id: str | None = None,
|
||||
aws_session_tags: object = None,
|
||||
**kwargs: object,
|
||||
):
|
||||
"""
|
||||
|
|
@ -297,20 +299,20 @@ class BedrockRealtime(BaseAWSLLM):
|
|||
|
||||
verbose_proxy_logger.debug("Bedrock Realtime: Connecting to %s with model %s", endpoint_uri, model)
|
||||
|
||||
credentials: Final = await run_aws_signing(
|
||||
self.get_credentials,
|
||||
auth_params: Final = AwsAuthParams(
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_session_token=aws_session_token,
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=aws_session_name,
|
||||
aws_profile_name=aws_profile_name,
|
||||
aws_role_name=aws_role_name,
|
||||
aws_web_identity_token=aws_web_identity_token,
|
||||
aws_sts_endpoint=aws_sts_endpoint,
|
||||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
)
|
||||
if credentials is None:
|
||||
credentials: Final = await run_aws_signing(self.resolve_credentials, auth_params, aws_region_name)
|
||||
if credentials is None: # pyright: ignore[reportUnnecessaryComparison] # boto3.Session() env fallback yields None
|
||||
raise BedrockError(
|
||||
status_code=401,
|
||||
message=(
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@ from typing import Final
|
|||
import httpx
|
||||
|
||||
from litellm.litellm_core_utils.aws_partition import get_aws_dns_suffix
|
||||
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
|
||||
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM, pop_aws_auth_params
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
from litellm.utils import ModelResponse, get_secret
|
||||
|
||||
|
|
@ -23,20 +23,9 @@ class SagemakerChatHandler(BaseAWSLLM):
|
|||
from botocore.credentials import Credentials
|
||||
except ImportError:
|
||||
raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
|
||||
## CREDENTIALS ##
|
||||
# pop aws_secret_access_key, aws_access_key_id, aws_session_token, aws_region_name from kwargs, since completion calls fail with them
|
||||
aws_secret_access_key: Final = optional_params.pop("aws_secret_access_key", None)
|
||||
aws_access_key_id: Final = optional_params.pop("aws_access_key_id", None)
|
||||
aws_session_token: Final = optional_params.pop("aws_session_token", None)
|
||||
auth_params: Final = pop_aws_auth_params(optional_params)
|
||||
aws_region_name = optional_params.pop("aws_region_name", None)
|
||||
aws_role_name: Final = optional_params.pop("aws_role_name", None)
|
||||
aws_session_name: Final = optional_params.pop("aws_session_name", None)
|
||||
aws_profile_name: Final = optional_params.pop("aws_profile_name", None)
|
||||
optional_params.pop("aws_bedrock_runtime_endpoint", None) # https://bedrock-runtime.{region_name}.amazonaws.com
|
||||
aws_web_identity_token: Final = optional_params.pop("aws_web_identity_token", None)
|
||||
aws_sts_endpoint: Final = optional_params.pop("aws_sts_endpoint", None)
|
||||
aws_external_id: Final = optional_params.pop("aws_external_id", None)
|
||||
aws_session_tags: Final = optional_params.pop("aws_session_tags", None)
|
||||
optional_params.pop("aws_bedrock_runtime_endpoint", None)
|
||||
|
||||
### SET REGION NAME ###
|
||||
if aws_region_name is None:
|
||||
|
|
@ -53,19 +42,7 @@ class SagemakerChatHandler(BaseAWSLLM):
|
|||
if aws_region_name is None:
|
||||
aws_region_name = "us-west-2"
|
||||
|
||||
credentials: Final[Credentials] = self.get_credentials(
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_session_token=aws_session_token,
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=aws_session_name,
|
||||
aws_profile_name=aws_profile_name,
|
||||
aws_role_name=aws_role_name,
|
||||
aws_web_identity_token=aws_web_identity_token,
|
||||
aws_sts_endpoint=aws_sts_endpoint,
|
||||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
)
|
||||
credentials: Final[Credentials] = self.resolve_credentials(auth_params, aws_region_name)
|
||||
return credentials, aws_region_name
|
||||
|
||||
def _prepare_request(
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ from litellm._logging import verbose_logger
|
|||
from litellm.litellm_core_utils.asyncify import asyncify
|
||||
from litellm.litellm_core_utils.aws_partition import get_aws_dns_suffix
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
|
||||
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM, pop_aws_auth_params
|
||||
from litellm.llms.custom_httpx.http_handler import (
|
||||
_get_httpx_client,
|
||||
get_async_httpx_client,
|
||||
|
|
@ -46,20 +46,9 @@ class SagemakerLLM(BaseAWSLLM):
|
|||
from botocore.credentials import Credentials
|
||||
except ImportError:
|
||||
raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.")
|
||||
## CREDENTIALS ##
|
||||
# pop aws_secret_access_key, aws_access_key_id, aws_session_token, aws_region_name from kwargs, since completion calls fail with them
|
||||
aws_secret_access_key: Final = optional_params.pop("aws_secret_access_key", None)
|
||||
aws_access_key_id: Final = optional_params.pop("aws_access_key_id", None)
|
||||
aws_session_token: Final = optional_params.pop("aws_session_token", None)
|
||||
auth_params: Final = pop_aws_auth_params(optional_params)
|
||||
aws_region_name = optional_params.pop("aws_region_name", None)
|
||||
aws_role_name: Final = optional_params.pop("aws_role_name", None)
|
||||
aws_session_name: Final = optional_params.pop("aws_session_name", None)
|
||||
aws_profile_name: Final = optional_params.pop("aws_profile_name", None)
|
||||
optional_params.pop("aws_bedrock_runtime_endpoint", None) # https://bedrock-runtime.{region_name}.amazonaws.com
|
||||
aws_web_identity_token: Final = optional_params.pop("aws_web_identity_token", None)
|
||||
aws_sts_endpoint: Final = optional_params.pop("aws_sts_endpoint", None)
|
||||
aws_external_id: Final = optional_params.pop("aws_external_id", None)
|
||||
aws_session_tags: Final = optional_params.pop("aws_session_tags", None)
|
||||
optional_params.pop("aws_bedrock_runtime_endpoint", None)
|
||||
|
||||
### SET REGION NAME ###
|
||||
if aws_region_name is None:
|
||||
|
|
@ -76,19 +65,7 @@ class SagemakerLLM(BaseAWSLLM):
|
|||
if aws_region_name is None:
|
||||
aws_region_name = "us-west-2"
|
||||
|
||||
credentials: Final[Credentials] = self.get_credentials(
|
||||
aws_access_key_id=aws_access_key_id,
|
||||
aws_secret_access_key=aws_secret_access_key,
|
||||
aws_session_token=aws_session_token,
|
||||
aws_region_name=aws_region_name,
|
||||
aws_session_name=aws_session_name,
|
||||
aws_profile_name=aws_profile_name,
|
||||
aws_role_name=aws_role_name,
|
||||
aws_web_identity_token=aws_web_identity_token,
|
||||
aws_sts_endpoint=aws_sts_endpoint,
|
||||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
)
|
||||
credentials: Final[Credentials] = self.resolve_credentials(auth_params, aws_region_name)
|
||||
return credentials, aws_region_name
|
||||
|
||||
def _prepare_request(
|
||||
|
|
|
|||
|
|
@ -0,0 +1,38 @@
|
|||
"""Per-worker cache of stored BYOK credentials, keyed so peer workers can evict it over the auth cache pub/sub."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Final
|
||||
|
||||
from litellm.caching.in_memory_cache import InMemoryCache
|
||||
from litellm.constants import MCP_BYOK_CREDENTIAL_CACHE_MAX_SIZE, MCP_BYOK_CREDENTIAL_CACHE_TTL_SECONDS
|
||||
|
||||
_CACHE_KEY_PREFIX: Final = "mcp_byok_credential"
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class CachedByokCredential:
|
||||
credential: str | None
|
||||
|
||||
|
||||
byok_credential_cache: Final = InMemoryCache(
|
||||
max_size_in_memory=MCP_BYOK_CREDENTIAL_CACHE_MAX_SIZE,
|
||||
default_ttl=MCP_BYOK_CREDENTIAL_CACHE_TTL_SECONDS,
|
||||
)
|
||||
|
||||
|
||||
def byok_credential_cache_key(user_id: str, server_id: str) -> str:
|
||||
return f"{_CACHE_KEY_PREFIX}:{user_id}:{server_id}"
|
||||
|
||||
|
||||
def get_cached_byok_credential(user_id: str, server_id: str) -> CachedByokCredential | None:
|
||||
cached: Final = byok_credential_cache.get_cache( # pyright: ignore[reportUnknownMemberType, reportUnknownVariableType] # InMemoryCache is untyped
|
||||
byok_credential_cache_key(user_id, server_id)
|
||||
)
|
||||
return cached if isinstance(cached, CachedByokCredential) else None
|
||||
|
||||
|
||||
def cache_byok_credential(user_id: str, server_id: str, credential: str | None) -> None:
|
||||
byok_credential_cache.set_cache( # pyright: ignore[reportUnknownMemberType] # InMemoryCache is untyped
|
||||
byok_credential_cache_key(user_id, server_id),
|
||||
CachedByokCredential(credential=credential),
|
||||
)
|
||||
|
|
@ -865,7 +865,7 @@ async def byok_token(
|
|||
_invalidate_byok_cred_cache,
|
||||
)
|
||||
|
||||
_invalidate_byok_cred_cache(user_id, server_id)
|
||||
await _invalidate_byok_cred_cache(user_id, server_id)
|
||||
except Exception as exc:
|
||||
verbose_proxy_logger.error(
|
||||
"byok_token: failed to store user credential for user=%s server=%s: %s",
|
||||
|
|
|
|||
|
|
@ -24,6 +24,7 @@ from litellm.proxy._types import (
|
|||
MCPApprovalStatus,
|
||||
MCPEnvVar,
|
||||
MCPEnvVarScope,
|
||||
MCPServerUserCredentialListItem,
|
||||
MCPSubmissionsSummary,
|
||||
NewMCPServerRequest,
|
||||
SpecialMCPServerName,
|
||||
|
|
@ -1504,6 +1505,37 @@ async def get_user_oauth_credential(
|
|||
return _parse_oauth_payload(decoded)
|
||||
|
||||
|
||||
def _server_user_credential_item(
|
||||
row: "prisma_db_models.LiteLLM_MCPUserCredentials",
|
||||
) -> MCPServerUserCredentialListItem:
|
||||
oauth_payload: Final = _decode_oauth_payload(row.credential_b64)
|
||||
if oauth_payload is None:
|
||||
return MCPServerUserCredentialListItem(
|
||||
user_id=row.user_id,
|
||||
credential_type="byok",
|
||||
updated_at=row.updated_at.isoformat(),
|
||||
)
|
||||
return MCPServerUserCredentialListItem(
|
||||
user_id=row.user_id,
|
||||
credential_type="oauth2",
|
||||
expires_at=oauth_payload.get("expires_at"),
|
||||
connected_at=oauth_payload.get("connected_at"),
|
||||
updated_at=row.updated_at.isoformat(),
|
||||
)
|
||||
|
||||
|
||||
async def list_server_user_credentials(
|
||||
prisma_client: PrismaClient,
|
||||
server_id: str,
|
||||
) -> tuple[MCPServerUserCredentialListItem, ...]:
|
||||
"""Every user's stored credential for one server, typed but without the secret, for admins."""
|
||||
rows: Final = await _db_find_user_credential_rows(
|
||||
prisma_client,
|
||||
{"server_id": server_id}, # mutable-ok: prisma where-inputs must be plain dicts
|
||||
)
|
||||
return tuple(_server_user_credential_item(row) for row in rows)
|
||||
|
||||
|
||||
async def list_user_oauth_credentials(
|
||||
prisma_client: PrismaClient,
|
||||
user_id: str,
|
||||
|
|
|
|||
|
|
@ -295,12 +295,15 @@ class MCPPerUserTokenCache:
|
|||
)
|
||||
|
||||
async def delete(self, user_id: str, server_id: str) -> None:
|
||||
"""Invalidate the cached token (removes from both in-memory and Redis layers)."""
|
||||
"""Invalidate the cached token in Redis, here, and in every peer worker's in-memory layer."""
|
||||
try:
|
||||
from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import ( # noqa: PLC0415 # proxy import cycle
|
||||
evict_and_broadcast,
|
||||
)
|
||||
from litellm.proxy.proxy_server import user_api_key_cache # noqa: PLC0415
|
||||
|
||||
key: Final = self._cache_key(user_id, server_id)
|
||||
await user_api_key_cache.async_delete_cache(key)
|
||||
await evict_and_broadcast((key,), user_api_key_cache)
|
||||
except Exception as exc:
|
||||
verbose_logger.debug(
|
||||
"MCPPerUserTokenCache.delete failed for user=%s server=%s: %s",
|
||||
|
|
|
|||
|
|
@ -28,7 +28,10 @@ from starlette.types import Message, Receive, Scope, Send
|
|||
from typing_extensions import ReadOnly, TypedDict
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import MAXIMUM_TRACEBACK_LINES_TO_LOG
|
||||
from litellm.constants import (
|
||||
MAXIMUM_TRACEBACK_LINES_TO_LOG,
|
||||
MCP_GATEWAY_SESSION_ID_PREFIX_LENGTH,
|
||||
)
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.custom_httpx.http_handler import (
|
||||
get_async_httpx_client,
|
||||
|
|
@ -38,6 +41,12 @@ from litellm.proxy._experimental.mcp_server.auth.user_api_key_auth_mcp import (
|
|||
MCPRequestHandler,
|
||||
_is_mcp_admitted_user_subject,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.byok_credential_cache import (
|
||||
byok_credential_cache,
|
||||
byok_credential_cache_key,
|
||||
cache_byok_credential,
|
||||
get_cached_byok_credential,
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.client_allowlist import (
|
||||
MCPClientAllowlist,
|
||||
check_mcp_client_allowed,
|
||||
|
|
@ -87,6 +96,9 @@ from litellm.proxy._types import (
|
|||
UserAPIKeyAuth,
|
||||
)
|
||||
from litellm.proxy.auth.ip_address_utils import IPAddressUtils
|
||||
from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import (
|
||||
publish_auth_cache_invalidation,
|
||||
)
|
||||
from litellm.proxy.litellm_pre_call_utils import (
|
||||
LiteLLMProxyRequestSetup,
|
||||
get_chain_id_from_headers,
|
||||
|
|
@ -96,6 +108,7 @@ from litellm.types.mcp import (
|
|||
MCPGatewaySession,
|
||||
MCPGatewaySessionGroupCount,
|
||||
MCPGatewaySessionsResponse,
|
||||
MCPGatewaySessionsTerminateResponse,
|
||||
MCPSpecVersion,
|
||||
)
|
||||
from litellm.types.mcp_server.mcp_server_manager import MCPInfo, MCPServer
|
||||
|
|
@ -107,13 +120,6 @@ if TYPE_CHECKING:
|
|||
|
||||
from litellm.proxy._experimental.mcp_server.db import OAuthCredentialPayload
|
||||
|
||||
# Short-lived in-memory cache for BYOK credentials.
|
||||
# Keyed by (user_id, server_id); value is (credential_or_None, monotonic_timestamp).
|
||||
# Storing the credential value (not just a bool) means _get_byok_credential and
|
||||
# _check_byok_credential share a single DB round-trip per TTL window.
|
||||
_byok_cred_cache: Final[dict[tuple[str, str], tuple[str | None, float]]] = {}
|
||||
_BYOK_CRED_CACHE_TTL: Final = 60 # seconds
|
||||
_BYOK_CRED_CACHE_MAX_SIZE: Final = 4096 # cap to prevent unbounded growth
|
||||
_STATEFUL_SESSION_IDLE_TIMEOUT_SECONDS: Final = 30 * 60
|
||||
# Upper bound on concurrent stateful sessions a single caller may hold. Each
|
||||
# `initialize` creates a session that survives until the idle timeout, so
|
||||
|
|
@ -132,20 +138,11 @@ _MCP_TRANSPORT_SPAN_SCOPE_KEY: Final = "litellm_otel_transport_span"
|
|||
_MCP_DESTINATIONS_SCOPE_KEY: Final = "litellm_otel_request_destinations"
|
||||
|
||||
|
||||
def _invalidate_byok_cred_cache(user_id: str, server_id: str) -> None:
|
||||
"""Remove a (user_id, server_id) entry from the BYOK credential cache.
|
||||
|
||||
Call this after storing or deleting a credential so subsequent calls
|
||||
see the fresh value rather than a stale cached result.
|
||||
"""
|
||||
_byok_cred_cache.pop((user_id, server_id), None)
|
||||
|
||||
|
||||
def _write_byok_cred_cache(user_id: str, server_id: str, credential: str | None) -> None:
|
||||
"""Write a credential value to the cache, evicting all entries if at capacity."""
|
||||
if len(_byok_cred_cache) >= _BYOK_CRED_CACHE_MAX_SIZE:
|
||||
_byok_cred_cache.clear()
|
||||
_byok_cred_cache[(user_id, server_id)] = (credential, time.monotonic())
|
||||
async def _invalidate_byok_cred_cache(user_id: str, server_id: str) -> None:
|
||||
"""Drop a stored-or-deleted BYOK credential from this worker's cache and from every peer worker's."""
|
||||
cache_key: Final = byok_credential_cache_key(user_id, server_id)
|
||||
byok_credential_cache.delete_cache(cache_key)
|
||||
await publish_auth_cache_invalidation(cache_key=cache_key)
|
||||
|
||||
|
||||
# Check if MCP is available
|
||||
|
|
@ -623,6 +620,7 @@ if MCP_AVAILABLE:
|
|||
_stateful_session_locks: Final[dict[str, asyncio.Lock]] = {}
|
||||
_stateful_session_active_request_counts: Final[dict[str, int]] = {}
|
||||
_stateful_session_client_info: Final[dict[str, Implementation]] = {} # mutable-ok: cleared on session teardown
|
||||
_admin_terminated_session_ids: Final[dict[str, float]] = {} # mutable-ok: admin-closed id -> last replay
|
||||
|
||||
class _TerminableTransport(Protocol):
|
||||
async def terminate(self) -> None: ...
|
||||
|
|
@ -694,6 +692,7 @@ if MCP_AVAILABLE:
|
|||
for session_id in list(_stateful_session_auth_context_last_seen):
|
||||
if session_id not in _stateful_session_auth_contexts:
|
||||
_remove_stateful_session_tracking(session_id)
|
||||
_forget_expired_admin_terminated_session_ids(now)
|
||||
|
||||
async def _enforce_stateful_session_cap_for_owner(owner: str) -> bool:
|
||||
"""
|
||||
|
|
@ -2816,35 +2815,28 @@ if MCP_AVAILABLE:
|
|||
mcp_server: MCPServer,
|
||||
user_api_key_auth: UserAPIKeyAuth | None,
|
||||
) -> str | None:
|
||||
"""Retrieve the stored BYOK credential for a user+server pair.
|
||||
|
||||
Uses the shared _byok_cred_cache to avoid a DB round-trip on every
|
||||
tool call within the TTL window.
|
||||
"""
|
||||
"""Retrieve the stored BYOK credential for a user+server pair, served from the worker cache within its TTL."""
|
||||
if not mcp_server.is_byok:
|
||||
return None
|
||||
user_id: Final = (user_api_key_auth.user_id if user_api_key_auth else None) or ""
|
||||
if not user_id:
|
||||
return None
|
||||
|
||||
cache_key: Final = (user_id, mcp_server.server_id)
|
||||
cached: Final = _byok_cred_cache.get(cache_key)
|
||||
cached: Final = get_cached_byok_credential(user_id, mcp_server.server_id)
|
||||
if cached is not None:
|
||||
credential, ts = cached
|
||||
if time.monotonic() - ts < _BYOK_CRED_CACHE_TTL:
|
||||
return credential
|
||||
return cached.credential
|
||||
|
||||
from litellm.proxy._experimental.mcp_server.db import get_user_credential
|
||||
from litellm.proxy.proxy_server import prisma_client
|
||||
|
||||
if prisma_client is None:
|
||||
return None
|
||||
credential = await get_user_credential(
|
||||
credential: Final = await get_user_credential(
|
||||
prisma_client=prisma_client,
|
||||
user_id=user_id,
|
||||
server_id=mcp_server.server_id,
|
||||
)
|
||||
_write_byok_cred_cache(user_id, mcp_server.server_id, credential)
|
||||
cache_byok_credential(user_id, mcp_server.server_id, credential)
|
||||
return credential
|
||||
|
||||
async def _check_byok_credential(
|
||||
|
|
@ -2873,27 +2865,23 @@ if MCP_AVAILABLE:
|
|||
headers={"WWW-Authenticate": get_byok_www_authenticate()},
|
||||
)
|
||||
|
||||
# Check shared credential cache before hitting the DB.
|
||||
cache_key: Final = (user_id, mcp_server.server_id)
|
||||
cached: Final = _byok_cred_cache.get(cache_key)
|
||||
cached: Final = get_cached_byok_credential(user_id, mcp_server.server_id)
|
||||
if cached is not None:
|
||||
cached_cred, ts = cached
|
||||
if time.monotonic() - ts < _BYOK_CRED_CACHE_TTL:
|
||||
if cached_cred is None:
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail={
|
||||
"error": "byok_auth_required",
|
||||
"server_id": mcp_server.server_id,
|
||||
"server_name": mcp_server.server_name or mcp_server.name,
|
||||
"message": (
|
||||
"No stored credential found for this BYOK server. "
|
||||
"Complete the OAuth authorization flow to provide your API key."
|
||||
),
|
||||
},
|
||||
headers={"WWW-Authenticate": get_byok_www_authenticate()},
|
||||
)
|
||||
return
|
||||
if cached.credential is None:
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail={
|
||||
"error": "byok_auth_required",
|
||||
"server_id": mcp_server.server_id,
|
||||
"server_name": mcp_server.server_name or mcp_server.name,
|
||||
"message": (
|
||||
"No stored credential found for this BYOK server. "
|
||||
"Complete the OAuth authorization flow to provide your API key."
|
||||
),
|
||||
},
|
||||
headers={"WWW-Authenticate": get_byok_www_authenticate()},
|
||||
)
|
||||
return
|
||||
|
||||
from litellm.proxy._experimental.mcp_server.db import get_user_credential
|
||||
from litellm.proxy.proxy_server import prisma_client
|
||||
|
|
@ -2917,7 +2905,7 @@ if MCP_AVAILABLE:
|
|||
user_id=user_id,
|
||||
server_id=mcp_server.server_id,
|
||||
)
|
||||
_write_byok_cred_cache(user_id, mcp_server.server_id, credential)
|
||||
cache_byok_credential(user_id, mcp_server.server_id, credential)
|
||||
if credential is None:
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
|
|
@ -3871,7 +3859,7 @@ if MCP_AVAILABLE:
|
|||
client_info: Final = _stateful_session_client_info.get(session_id)
|
||||
key_auth: Final = auth_user.user_api_key_auth
|
||||
return MCPGatewaySession(
|
||||
session_id_prefix=session_id[:8],
|
||||
session_id_prefix=session_id[:MCP_GATEWAY_SESSION_ID_PREFIX_LENGTH],
|
||||
client_name=client_info.name if client_info is not None else None,
|
||||
client_version=client_info.version if client_info is not None else None,
|
||||
user_id=key_auth.user_id if key_auth is not None else None,
|
||||
|
|
@ -3906,6 +3894,72 @@ if MCP_AVAILABLE:
|
|||
sessions=sessions,
|
||||
)
|
||||
|
||||
def _session_matches_admin_selector(
|
||||
session_id: str,
|
||||
auth_user: MCPAuthenticatedUser,
|
||||
session_id_prefix: str | None,
|
||||
user_id: str | None,
|
||||
) -> bool:
|
||||
if session_id_prefix is not None and not session_id.startswith(session_id_prefix):
|
||||
return False
|
||||
if user_id is None:
|
||||
return True
|
||||
key_auth: Final = auth_user.user_api_key_auth
|
||||
return key_auth is not None and key_auth.user_id == user_id
|
||||
|
||||
def _forget_expired_admin_terminated_session_ids(now: float) -> None:
|
||||
for session_id in [
|
||||
session_id
|
||||
for session_id, last_replayed in _admin_terminated_session_ids.items()
|
||||
if now - last_replayed >= _STATEFUL_SESSION_IDLE_TIMEOUT_SECONDS
|
||||
]:
|
||||
del _admin_terminated_session_ids[session_id]
|
||||
|
||||
def _is_admin_terminated_session_id(session_id: str, now: float) -> bool:
|
||||
last_replayed: Final = _admin_terminated_session_ids.get(session_id)
|
||||
if last_replayed is None:
|
||||
return False
|
||||
if now - last_replayed >= _STATEFUL_SESSION_IDLE_TIMEOUT_SECONDS:
|
||||
del _admin_terminated_session_ids[session_id]
|
||||
return False
|
||||
_admin_terminated_session_ids[session_id] = now
|
||||
return True
|
||||
|
||||
async def terminate_mcp_gateway_sessions(
|
||||
*,
|
||||
session_id_prefix: str | None = None,
|
||||
user_id: str | None = None,
|
||||
) -> MCPGatewaySessionsTerminateResponse:
|
||||
"""Force-close every live stateful session on this worker matching the selector.
|
||||
|
||||
The transport is terminated (open streams close), all per-session
|
||||
tracking is dropped, and the id is remembered so a client that keeps
|
||||
sending it receives 404 and has to ``initialize`` again, which re-runs
|
||||
admission. Only sessions held by this worker process are affected.
|
||||
"""
|
||||
now: Final = time.monotonic()
|
||||
_forget_expired_admin_terminated_session_ids(now)
|
||||
server_instances: Final = _stateful_server_instances()
|
||||
targets: Final = tuple(
|
||||
(session_id, auth_user)
|
||||
for session_id, auth_user in tuple(_stateful_session_auth_contexts.items())
|
||||
if session_id in server_instances
|
||||
and _session_matches_admin_selector(session_id, auth_user, session_id_prefix, user_id)
|
||||
)
|
||||
terminated: Final = tuple(_gateway_session_for(session_id, auth_user, now) for session_id, auth_user in targets)
|
||||
for session_id, _ in targets:
|
||||
_admin_terminated_session_ids[session_id] = now
|
||||
transport = server_instances.pop(session_id, None)
|
||||
_remove_stateful_session_tracking(session_id)
|
||||
if transport is not None:
|
||||
await transport.terminate()
|
||||
verbose_logger.warning("MCP session '%s' terminated by an administrator.", session_id)
|
||||
return MCPGatewaySessionsTerminateResponse(
|
||||
worker_pid=os.getpid(),
|
||||
terminated_sessions=len(terminated),
|
||||
sessions=terminated,
|
||||
)
|
||||
|
||||
async def _read_request_body_for_routing(
|
||||
receive: Receive,
|
||||
) -> tuple[list[Message], bytes]:
|
||||
|
|
@ -4030,6 +4084,17 @@ if MCP_AVAILABLE:
|
|||
await success_response(scope, receive, send)
|
||||
return True
|
||||
|
||||
if _is_admin_terminated_session_id(_session_id, time.monotonic()):
|
||||
terminated_response: Final = JSONResponse(
|
||||
status_code=404,
|
||||
content={ # mutable-ok: JSONResponse content must be a plain dict
|
||||
"error": "Not Found",
|
||||
"details": "mcp-session-id was terminated by an administrator. Send initialize to start a new session.",
|
||||
},
|
||||
)
|
||||
await terminated_response(scope, receive, send)
|
||||
return True
|
||||
|
||||
# Non-DELETE: strip stale session ID to allow new session creation
|
||||
verbose_logger.warning(
|
||||
"MCP session ID '%s' not found in this worker's memory. "
|
||||
|
|
|
|||
|
|
@ -3050,6 +3050,18 @@
|
|||
},
|
||||
"DailySpendMetadata": {
|
||||
"properties": {
|
||||
"api_key_limit": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"description": "When set, api_keys and every api_key_breakdown list at most this many keys, ranked by spend. Totals and the model, provider, mcp and endpoint rollups still cover every key.",
|
||||
"title": "Api Key Limit"
|
||||
},
|
||||
"has_more": {
|
||||
"default": false,
|
||||
"title": "Has More",
|
||||
|
|
@ -3060,6 +3072,18 @@
|
|||
"title": "Page",
|
||||
"type": "integer"
|
||||
},
|
||||
"total_api_keys": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"description": "Distinct API keys matching the filters. When this exceeds api_key_limit, the per-key lists are truncated to the highest-spend keys.",
|
||||
"title": "Total Api Keys"
|
||||
},
|
||||
"total_api_requests": {
|
||||
"default": 0,
|
||||
"title": "Total Api Requests",
|
||||
|
|
@ -10030,7 +10054,7 @@
|
|||
},
|
||||
"unreachable_fallback": {
|
||||
"default": "fail_closed",
|
||||
"description": "Behavior when a guardrail endpoint is unreachable due to network errors. Implemented by guardrail='generic_guardrail_api', 'agent_365', 'akto', 'vigil_guard', 'repelloai', 'headroom', and 'compresr'. 'fail_closed' raises an error (default). 'fail_open' logs a critical error and allows the request to proceed.",
|
||||
"description": "Behavior when a guardrail endpoint is unreachable due to network errors. Implemented by guardrail='generic_guardrail_api', 'agent_365', 'akto', 'vigil_guard', 'repelloai', 'headroom', 'compresr', and 'typesafe'. 'fail_closed' raises an error (default). 'fail_open' logs a critical error and allows the request to proceed.",
|
||||
"enum": [
|
||||
"fail_closed",
|
||||
"fail_open"
|
||||
|
|
@ -27965,6 +27989,32 @@
|
|||
"title": "MCPGatewaySessionsResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"MCPGatewaySessionsTerminateResponse": {
|
||||
"description": "Stateful sessions an administrator force-closed on this proxy worker.",
|
||||
"properties": {
|
||||
"sessions": {
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/MCPGatewaySession"
|
||||
},
|
||||
"title": "Sessions",
|
||||
"type": "array"
|
||||
},
|
||||
"terminated_sessions": {
|
||||
"title": "Terminated Sessions",
|
||||
"type": "integer"
|
||||
},
|
||||
"worker_pid": {
|
||||
"title": "Worker Pid",
|
||||
"type": "integer"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"worker_pid",
|
||||
"terminated_sessions"
|
||||
],
|
||||
"title": "MCPGatewaySessionsTerminateResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"MCPOAuthUserCredentialRequest": {
|
||||
"description": "Stores a user's OAuth2 token for an OpenAPI MCP server.",
|
||||
"properties": {
|
||||
|
|
@ -28061,6 +28111,56 @@
|
|||
"title": "MCPOAuthUserCredentialStatus",
|
||||
"type": "object"
|
||||
},
|
||||
"MCPServerUserCredentialListItem": {
|
||||
"description": "One user's stored credential for an MCP server, as an admin sees it. Never carries the secret.",
|
||||
"properties": {
|
||||
"connected_at": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"title": "Connected At"
|
||||
},
|
||||
"credential_type": {
|
||||
"enum": [
|
||||
"oauth2",
|
||||
"byok"
|
||||
],
|
||||
"title": "Credential Type",
|
||||
"type": "string"
|
||||
},
|
||||
"expires_at": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"title": "Expires At"
|
||||
},
|
||||
"updated_at": {
|
||||
"title": "Updated At",
|
||||
"type": "string"
|
||||
},
|
||||
"user_id": {
|
||||
"title": "User Id",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"user_id",
|
||||
"credential_type",
|
||||
"updated_at"
|
||||
],
|
||||
"title": "MCPServerUserCredentialListItem",
|
||||
"type": "object"
|
||||
},
|
||||
"MCPSubmissionsSummary": {
|
||||
"properties": {
|
||||
"active": {
|
||||
|
|
@ -30237,7 +30337,7 @@
|
|||
},
|
||||
"/v1/mcp/server/{server_id}/oauth-user-credential": {
|
||||
"delete": {
|
||||
"description": "Revoke the calling user's stored OAuth2 token for an MCP server",
|
||||
"description": "Revoke the calling user's stored OAuth2 token for an MCP server. A proxy admin may pass user_id to revoke another user's stored token.",
|
||||
"operationId": "delete_mcp_oauth_user_credential_v1_mcp_server__server_id__oauth_user_credential_delete",
|
||||
"parameters": [
|
||||
{
|
||||
|
|
@ -30248,6 +30348,23 @@
|
|||
"title": "Server Id",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
{
|
||||
"in": "query",
|
||||
"name": "user_id",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"anyOf": [
|
||||
{
|
||||
"minLength": 1,
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"title": "User Id"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
|
|
@ -30447,7 +30564,7 @@
|
|||
},
|
||||
"/v1/mcp/server/{server_id}/user-credential": {
|
||||
"delete": {
|
||||
"description": "Delete the calling user's stored API key for a BYOK MCP server",
|
||||
"description": "Delete the calling user's stored API key for a BYOK MCP server. A proxy admin may pass user_id to revoke another user's stored key.",
|
||||
"operationId": "delete_mcp_user_credential_v1_mcp_server__server_id__user_credential_delete",
|
||||
"parameters": [
|
||||
{
|
||||
|
|
@ -30458,6 +30575,23 @@
|
|||
"title": "Server Id",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
{
|
||||
"in": "query",
|
||||
"name": "user_id",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"anyOf": [
|
||||
{
|
||||
"minLength": 1,
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"title": "User Id"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
|
|
@ -30549,6 +30683,58 @@
|
|||
]
|
||||
}
|
||||
},
|
||||
"/v1/mcp/server/{server_id}/user-credentials": {
|
||||
"get": {
|
||||
"description": "List every user's stored BYOK or OAuth2 credential for an MCP server (admin only, no secrets)",
|
||||
"operationId": "list_mcp_server_user_credentials_v1_mcp_server__server_id__user_credentials_get",
|
||||
"parameters": [
|
||||
{
|
||||
"in": "path",
|
||||
"name": "server_id",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"title": "Server Id",
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/MCPServerUserCredentialListItem"
|
||||
},
|
||||
"title": "Response List Mcp Server User Credentials V1 Mcp Server Server Id User Credentials Get",
|
||||
"type": "array"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Successful Response"
|
||||
},
|
||||
"422": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/HTTPValidationError"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Validation Error"
|
||||
}
|
||||
},
|
||||
"security": [
|
||||
{
|
||||
"APIKeyHeader": []
|
||||
}
|
||||
],
|
||||
"summary": "List Mcp Server User Credentials",
|
||||
"tags": [
|
||||
"mcp_management"
|
||||
]
|
||||
}
|
||||
},
|
||||
"/v1/mcp/server/{server_id}/user-env-vars": {
|
||||
"delete": {
|
||||
"description": "Clear the calling user's per-user MCP env var values for this server.",
|
||||
|
|
@ -30700,6 +30886,77 @@
|
|||
}
|
||||
},
|
||||
"/v1/mcp/sessions": {
|
||||
"delete": {
|
||||
"description": "Force-close live stateful MCP gateway sessions on this proxy worker, selected by session id prefix and/or by the LiteLLM user that opened them (proxy admin only).",
|
||||
"operationId": "delete_mcp_gateway_sessions_v1_mcp_sessions_delete",
|
||||
"parameters": [
|
||||
{
|
||||
"in": "query",
|
||||
"name": "session_id_prefix",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"anyOf": [
|
||||
{
|
||||
"minLength": 8,
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"title": "Session Id Prefix"
|
||||
}
|
||||
},
|
||||
{
|
||||
"in": "query",
|
||||
"name": "user_id",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"anyOf": [
|
||||
{
|
||||
"minLength": 1,
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"title": "User Id"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/MCPGatewaySessionsTerminateResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Successful Response"
|
||||
},
|
||||
"422": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/HTTPValidationError"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Validation Error"
|
||||
}
|
||||
},
|
||||
"security": [
|
||||
{
|
||||
"APIKeyHeader": []
|
||||
}
|
||||
],
|
||||
"summary": "Delete Mcp Gateway Sessions",
|
||||
"tags": [
|
||||
"mcp_management"
|
||||
]
|
||||
},
|
||||
"get": {
|
||||
"description": "Live stateful MCP gateway sessions on this proxy worker, grouped by AI client and by user.",
|
||||
"operationId": "get_mcp_gateway_sessions_v1_mcp_sessions_get",
|
||||
|
|
@ -38542,6 +38799,7 @@
|
|||
"type": "object"
|
||||
},
|
||||
"SCIMMultiValuedAttribute": {
|
||||
"additionalProperties": true,
|
||||
"properties": {
|
||||
"display": {
|
||||
"anyOf": [
|
||||
|
|
@ -38577,13 +38835,17 @@
|
|||
"title": "Type"
|
||||
},
|
||||
"value": {
|
||||
"title": "Value",
|
||||
"type": "string"
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"title": "Value"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"value"
|
||||
],
|
||||
"title": "SCIMMultiValuedAttribute",
|
||||
"type": "object"
|
||||
},
|
||||
|
|
|
|||
|
|
@ -662,6 +662,11 @@ class LiteLLMRoutes(enum.Enum):
|
|||
KeyManagementRoutes.AUTO_ROUTER_MANAGE.value,
|
||||
]
|
||||
|
||||
team_service_account_key_routes = (
|
||||
KeyManagementRoutes.KEY_GENERATE.value,
|
||||
KeyManagementRoutes.KEY_UPDATE.value,
|
||||
)
|
||||
|
||||
management_routes = (
|
||||
[
|
||||
# user
|
||||
|
|
@ -1721,6 +1726,16 @@ class MCPUserCredentialListItem(LiteLLMPydanticObjectBase):
|
|||
connected_at: str | None = None # ISO-8601
|
||||
|
||||
|
||||
class MCPServerUserCredentialListItem(LiteLLMPydanticObjectBase):
|
||||
"""One user's stored credential for an MCP server, as an admin sees it. Never carries the secret."""
|
||||
|
||||
user_id: str
|
||||
credential_type: Literal["oauth2", "byok"]
|
||||
expires_at: str | None = None
|
||||
connected_at: str | None = None
|
||||
updated_at: str
|
||||
|
||||
|
||||
class MCPUserEnvVarsRequest(LiteLLMPydanticObjectBase):
|
||||
"""Payload for storing the calling user's per-user env var values."""
|
||||
|
||||
|
|
@ -2763,6 +2778,25 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
|
|||
description="sends alerts if requests hang for 5min+",
|
||||
)
|
||||
ui_access_mode: Literal["admin_only", "all"] | None = Field("all", description="Control access to the Proxy UI")
|
||||
max_failed_login_attempts_per_source: int | None = Field(
|
||||
None,
|
||||
ge=1,
|
||||
description="Failed Admin UI sign-in attempts allowed from one source address, across every username, within `failed_login_window_seconds`. One more blocks that address for `failed_login_block_seconds`. Half this value, rounded down but at least 1, is the allowance for one username from that address; one more blocks that address for that username only, and its further failures stop counting toward the address limit, so a script stuck on one account does not block everyone behind a shared address. The per-address limit is only enforced when `trusted_proxy_ranges` is set: to the proxies in front of LiteLLM, or to an empty list when clients connect directly. Left unset, the peer address may be a shared ingress and only the per-username half runs. IPv6 addresses are grouped by /64. Set under `general_settings` in config.yaml. Defaults to 10",
|
||||
)
|
||||
max_failed_login_attempts_per_source_overrides: dict[str, int] | None = Field(
|
||||
None,
|
||||
description="Per-address overrides of `max_failed_login_attempts_per_source`, keyed by IP address or CIDR range, e.g. {'1.2.3.4': 200, '5.6.0.0/24': 500}. The most specific matching range wins (between equivalent keys such as '1.2.3.4' and '1.2.3.4/32', an exemption wins, then the higher limit), and the per-username allowance for that address follows as half the override. A value of 0 exempts the address from both limits. Set under `general_settings` in config.yaml",
|
||||
)
|
||||
failed_login_window_seconds: int | None = Field(
|
||||
None,
|
||||
ge=1,
|
||||
description="Fixed window in seconds over which failed Admin UI sign-in attempts are counted. The window starts at the first failure and is not extended by later ones. Set under `general_settings` in config.yaml. Defaults to 60",
|
||||
)
|
||||
failed_login_block_seconds: int | None = Field(
|
||||
None,
|
||||
ge=1,
|
||||
description="How long a blocked source address, or source address and username, stays blocked. Every attempt from a blocked key, right or wrong, is refused with 429 before the password is checked; the block is not extended by refused attempts. Set under `general_settings` in config.yaml. Defaults to 300",
|
||||
)
|
||||
allowed_routes: list | None = Field(None, description="Proxy API Endpoints you want users to be able to access")
|
||||
reject_clientside_metadata_tags: bool | None = Field(
|
||||
None,
|
||||
|
|
@ -2884,7 +2918,7 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
|
|||
)
|
||||
trusted_proxy_ranges: list[str] | None = Field(
|
||||
None,
|
||||
description="CIDR ranges of trusted reverse proxies allowed to provide identity headers for header-based auth paths such as enable_oauth2_proxy_auth and custom_ui_sso_sign_in_handler.",
|
||||
description="CIDR ranges of trusted reverse proxies allowed to provide identity headers for header-based auth paths such as enable_oauth2_proxy_auth and custom_ui_sso_sign_in_handler, and whose X-Forwarded-For is used to attribute Admin UI sign-in attempts to a source address. Set it to an empty list when clients connect directly, so the peer address is the source. Left unset, or containing an entry that is not an address or CIDR range, the per-source sign-in limit is off.",
|
||||
)
|
||||
store_model_in_db: bool | None = Field(
|
||||
None,
|
||||
|
|
@ -3287,6 +3321,15 @@ class UserAPIKeyAuth(LiteLLM_VerificationTokenView): # the expected response ob
|
|||
user_role=LitellmUserRoles.PROXY_ADMIN,
|
||||
)
|
||||
|
||||
@property
|
||||
def is_team_service_account(self) -> bool:
|
||||
return (
|
||||
self.user_id is None
|
||||
and self.team_id is not None
|
||||
and bool(self.metadata)
|
||||
and self.metadata.get("service_account_id") is not None
|
||||
)
|
||||
|
||||
|
||||
def user_api_key_has_admin_view(user_api_key_dict: UserAPIKeyAuth) -> bool:
|
||||
"""Return True if the caller's role grants unscoped read access to all
|
||||
|
|
|
|||
445
litellm/proxy/auth/login_throttle.py
Normal file
445
litellm/proxy/auth/login_throttle.py
Normal file
|
|
@ -0,0 +1,445 @@
|
|||
"""Failed-login accounting for the Admin UI sign-in path.
|
||||
|
||||
Wrong passwords are counted over a short window per source address and per source-and-username
|
||||
pair; too many in one window blocks that key for a fixed time. While a key is blocked every attempt
|
||||
from it, right or wrong, is refused with 429 before the password is checked. A blocked pair stops
|
||||
counting against its source, so one script stuck on one account does not block the whole office.
|
||||
Recovery is the master key over the API, which never passes through here, or waiting out the block.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import hashlib
|
||||
import ipaddress
|
||||
import math
|
||||
import time
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from functools import cache
|
||||
from typing import Final, Literal, NamedTuple, Protocol, TypeAlias
|
||||
|
||||
from fastapi import Request, status
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
from redis.exceptions import RedisError
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.caching.in_memory_cache import InMemoryCache
|
||||
from litellm.caching.redis_cache import RedisCache, RedisCircuitBreakerOpenError
|
||||
from litellm.constants import (
|
||||
EMPTY_MAPPING,
|
||||
LOGIN_THROTTLE_CACHE_KEY_PREFIX,
|
||||
LOGIN_THROTTLE_MAX_TRACKED_BLOCKS,
|
||||
LOGIN_THROTTLE_MAX_TRACKED_COUNTERS,
|
||||
LOGIN_THROTTLE_NOT_BLOCKED,
|
||||
LOGIN_THROTTLE_UNKNOWN_SOURCE,
|
||||
)
|
||||
from litellm.proxy._types import ProxyErrorTypes, ProxyException
|
||||
from litellm.proxy.auth.network import TrustedProxyConfig, resolve_client_ip
|
||||
from litellm.secret_managers.main import get_secret_bool
|
||||
|
||||
DEFAULT_MAX_FAILED_LOGIN_ATTEMPTS_PER_SOURCE: Final = 10
|
||||
DEFAULT_FAILED_LOGIN_WINDOW_SECONDS: Final = 60
|
||||
DEFAULT_FAILED_LOGIN_BLOCK_SECONDS: Final = 300
|
||||
|
||||
IPV6_SOURCE_PREFIX_LENGTH: Final = 64
|
||||
EXEMPT: Final = 0
|
||||
|
||||
SOURCE_LIMIT_KEY: Final = "max_failed_login_attempts_per_source"
|
||||
SOURCE_LIMIT_OVERRIDES_KEY: Final = "max_failed_login_attempts_per_source_overrides"
|
||||
WINDOW_KEY: Final = "failed_login_window_seconds"
|
||||
BLOCK_KEY: Final = "failed_login_block_seconds"
|
||||
TRUSTED_PROXY_RANGES_KEY: Final = "trusted_proxy_ranges"
|
||||
|
||||
_REDIS_FAILURES: Final = (RedisError, RedisCircuitBreakerOpenError, OSError, asyncio.TimeoutError)
|
||||
_LOCAL_BLOCK_EXPIRY: Final = TypeAdapter[float | None](float | None)
|
||||
_SOURCE_LIMIT_OVERRIDES: Final = TypeAdapter[Mapping[str, object]](Mapping[str, object])
|
||||
_RANGE_ENTRIES: Final = TypeAdapter[tuple[object, ...]](tuple[object, ...])
|
||||
|
||||
Scope: TypeAlias = Literal["user", "source"]
|
||||
|
||||
_BlockTtls: TypeAlias = tuple[int, int]
|
||||
_LUA_BLOCK_TTLS: Final = TypeAdapter[_BlockTtls](_BlockTtls)
|
||||
_Network: TypeAlias = ipaddress.IPv4Network | ipaddress.IPv6Network
|
||||
|
||||
|
||||
class LocalStore(Protocol):
|
||||
"""The per-worker store behind the counters and blocks; ``InMemoryCache`` satisfies it."""
|
||||
|
||||
def get_cache(self, key: str) -> object: ...
|
||||
|
||||
def set_cache(self, key: str, value: float, *, ttl: int) -> None: ...
|
||||
|
||||
def increment_cache(self, key: str, value: float, *, ttl: int) -> float: ...
|
||||
|
||||
def delete_cache(self, key: str) -> None: ...
|
||||
|
||||
|
||||
# KEYS: pair counter, pair block, source counter, source block (one cluster slot via the source hash tag)
|
||||
# ARGV: pair limit, source limit (0 = source scope off), window seconds, block seconds
|
||||
# Both scripts return {pair block TTL, source block TTL}; 0 or below means not blocked
|
||||
_BLOCK_TTLS_LUA: Final = "return {redis.call('TTL', KEYS[2]), redis.call('TTL', KEYS[4])}"
|
||||
_RECORD_FAILURE_LUA: Final = (
|
||||
"local function bump(count_key, block_key, limit) "
|
||||
"local blocked = redis.call('TTL', block_key) "
|
||||
"if blocked > 0 then return blocked end "
|
||||
"local count = redis.call('INCR', count_key) "
|
||||
"if redis.call('TTL', count_key) < 0 then redis.call('EXPIRE', count_key, ARGV[3]) end "
|
||||
"if count > limit then redis.call('SET', block_key, '1', 'EX', ARGV[4]) return tonumber(ARGV[4]) end "
|
||||
"return 0 end "
|
||||
"local user_block = bump(KEYS[1], KEYS[2], tonumber(ARGV[1])) "
|
||||
"local source_block = 0 "
|
||||
"if tonumber(ARGV[2]) > 0 and user_block == 0 then "
|
||||
"source_block = bump(KEYS[3], KEYS[4], tonumber(ARGV[2])) end "
|
||||
"return {user_block, source_block}"
|
||||
)
|
||||
|
||||
_COUNTERS: Final = InMemoryCache(
|
||||
max_size_in_memory=LOGIN_THROTTLE_MAX_TRACKED_COUNTERS, default_ttl=DEFAULT_FAILED_LOGIN_WINDOW_SECONDS
|
||||
)
|
||||
_BLOCKS: Final = InMemoryCache(
|
||||
max_size_in_memory=LOGIN_THROTTLE_MAX_TRACKED_BLOCKS, default_ttl=DEFAULT_FAILED_LOGIN_BLOCK_SECONDS
|
||||
)
|
||||
|
||||
|
||||
@cache
|
||||
def _rate_limit_disabled() -> bool:
|
||||
return get_secret_bool("LITELLM_DISABLE_LOGIN_RATE_LIMIT", default_value=False) is True
|
||||
|
||||
|
||||
@cache
|
||||
def warn_login_counters_are_per_worker(num_workers: str) -> None:
|
||||
verbose_proxy_logger.warning(
|
||||
"Running %s workers but Redis is not configured. Failed Admin UI sign-in attempts are counted "
|
||||
"per worker, so the effective limits are %s times the configured values. Configure Redis "
|
||||
"to share one count across workers.",
|
||||
num_workers,
|
||||
num_workers,
|
||||
)
|
||||
|
||||
|
||||
@cache
|
||||
def warn_source_login_limit_is_off() -> None:
|
||||
verbose_proxy_logger.warning(
|
||||
"%s is not set or not a valid list of ranges, so failed Admin UI sign-in attempts are limited per "
|
||||
"source address and username only. Set it to the address ranges of the proxies in front of LiteLLM, "
|
||||
"or to an empty list when clients connect directly, to also limit each source address across usernames.",
|
||||
TRUSTED_PROXY_RANGES_KEY,
|
||||
)
|
||||
|
||||
|
||||
def declared_proxy_ranges(settings: Mapping[str, object]) -> tuple[str, ...] | None:
|
||||
"""What the operator says fronts LiteLLM: the proxy ranges, an empty tuple for none, None when unsaid.
|
||||
|
||||
Only a declared topology makes the source address trustworthy enough to limit across usernames.
|
||||
An unset key, a value that is not a list of ranges, or a list with an entry that is not an address
|
||||
or range leaves it unknown and the source scope off.
|
||||
"""
|
||||
entries: Final = _configured_range_entries(settings.get(TRUSTED_PROXY_RANGES_KEY))
|
||||
if entries is None or any(_parse_network(entry, TRUSTED_PROXY_RANGES_KEY) is None for entry in entries):
|
||||
return None
|
||||
return entries
|
||||
|
||||
|
||||
def _configured_range_entries(raw_ranges: object) -> tuple[str, ...] | None:
|
||||
"""Every configured entry, blanks included, so a stray empty string fails validation like any other typo."""
|
||||
if raw_ranges is None:
|
||||
return None
|
||||
if isinstance(raw_ranges, str):
|
||||
return tuple(part.strip() for part in raw_ranges.split(","))
|
||||
try:
|
||||
return tuple(str(entry).strip() for entry in _RANGE_ENTRIES.validate_python(raw_ranges))
|
||||
except ValidationError:
|
||||
verbose_proxy_logger.warning(
|
||||
"Invalid %s value: expected a list of address ranges, got %s",
|
||||
TRUSTED_PROXY_RANGES_KEY,
|
||||
type(raw_ranges).__name__,
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _positive_int(raw: object, key: str, default: int) -> int:
|
||||
if raw is None:
|
||||
return default
|
||||
try:
|
||||
value: Final = int(str(raw))
|
||||
except (TypeError, ValueError):
|
||||
verbose_proxy_logger.warning("Invalid %s value %r; using %s", key, raw, default)
|
||||
return default
|
||||
if value < 1:
|
||||
verbose_proxy_logger.warning("Invalid %s value %s (must be >= 1); using %s", key, value, default)
|
||||
return default
|
||||
return value
|
||||
|
||||
|
||||
def _int_setting(settings: Mapping[str, object], key: str, default: int) -> int:
|
||||
return _positive_int(settings.get(key), key, default)
|
||||
|
||||
|
||||
def _override_limit(raw: object, default: int) -> int:
|
||||
"""A per-address override: a limit of 1 or more, or ``EXEMPT`` (0) to leave that address unlimited."""
|
||||
if str(raw).strip() == str(EXEMPT):
|
||||
return EXEMPT
|
||||
return _positive_int(raw, SOURCE_LIMIT_OVERRIDES_KEY, default)
|
||||
|
||||
|
||||
def _parse_address(client_ip: str) -> ipaddress.IPv4Address | ipaddress.IPv6Address | None:
|
||||
"""The address as it is limited and counted: an IPv4-mapped IPv6 address is its IPv4 address."""
|
||||
try:
|
||||
address: Final = ipaddress.ip_address(client_ip)
|
||||
except ValueError:
|
||||
return None
|
||||
if isinstance(address, ipaddress.IPv6Address) and address.ipv4_mapped is not None:
|
||||
return address.ipv4_mapped
|
||||
return address
|
||||
|
||||
|
||||
def _parse_network(raw_range: str, setting_name: str = SOURCE_LIMIT_OVERRIDES_KEY) -> _Network | None:
|
||||
try:
|
||||
return ipaddress.ip_network(raw_range.strip(), strict=False)
|
||||
except ValueError:
|
||||
verbose_proxy_logger.warning("Invalid address or range %r in %s; skipping", raw_range, setting_name)
|
||||
return None
|
||||
|
||||
|
||||
def _precedence(network: _Network, limit: int) -> tuple[int, bool, int]:
|
||||
"""Sort key for competing overrides: the longest prefix wins, then an exemption, then the higher limit."""
|
||||
return (network.prefixlen, limit == EXEMPT, limit)
|
||||
|
||||
|
||||
def _source_limit(settings: Mapping[str, object], client_ip: str) -> int:
|
||||
"""Failure allowance for this address: the most specific configured range containing it, else the default.
|
||||
|
||||
``EXEMPT`` (0) means the operator opted this address out of both limits. Between equivalent keys such as
|
||||
``1.2.3.4`` and ``1.2.3.4/32`` an exemption wins, then the higher limit.
|
||||
"""
|
||||
default: Final = _int_setting(settings, SOURCE_LIMIT_KEY, DEFAULT_MAX_FAILED_LOGIN_ATTEMPTS_PER_SOURCE)
|
||||
raw_overrides: Final = settings.get(SOURCE_LIMIT_OVERRIDES_KEY)
|
||||
if raw_overrides is None:
|
||||
return default
|
||||
try:
|
||||
overrides: Final = _SOURCE_LIMIT_OVERRIDES.validate_python(raw_overrides)
|
||||
except ValidationError:
|
||||
verbose_proxy_logger.warning(
|
||||
"Invalid %s value; expected a mapping of address or range to limit", SOURCE_LIMIT_OVERRIDES_KEY
|
||||
)
|
||||
return default
|
||||
address: Final = _parse_address(client_ip)
|
||||
if address is None:
|
||||
return default
|
||||
matches: Final = sorted(
|
||||
_precedence(network, _override_limit(raw_limit, default))
|
||||
for raw_range, raw_limit in overrides.items()
|
||||
if (network := _parse_network(raw_range)) is not None and address in network
|
||||
)
|
||||
return matches[-1][-1] if matches else default
|
||||
|
||||
|
||||
def user_limit_for(source_limit: int) -> int:
|
||||
"""Failures allowed for one username from one address: half the address allowance, rounded down, at least 1."""
|
||||
return max(source_limit // 2, 1)
|
||||
|
||||
|
||||
def source_group(client_ip: str) -> str:
|
||||
"""The bucket an address is counted in: IPv4 as is, IPv6 by its /64, so one prefix holder cannot rotate."""
|
||||
address: Final = _parse_address(client_ip)
|
||||
if address is None:
|
||||
return client_ip
|
||||
if isinstance(address, ipaddress.IPv6Address):
|
||||
return str(ipaddress.ip_network((address, IPV6_SOURCE_PREFIX_LENGTH), strict=False))
|
||||
return str(address)
|
||||
|
||||
|
||||
class _Keys(NamedTuple):
|
||||
pair_counter: str
|
||||
pair_block: str
|
||||
source_counter: str
|
||||
source_block: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Block:
|
||||
scope: Scope
|
||||
retry_after: int
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class LoginThrottle:
|
||||
"""Failed-login limits for one request's source address.
|
||||
|
||||
``source_limit`` is None when the source scope is off: ``trusted_proxy_ranges`` is unset, so the peer
|
||||
address may be a shared ingress. An empty list means clients connect directly and the peer is the source.
|
||||
``user_limit`` is derived from the address allowance either way, see ``user_limit_for``. An address whose
|
||||
override is ``EXEMPT`` gets a disabled throttle: nothing is counted or blocked for it.
|
||||
"""
|
||||
|
||||
client_ip: str
|
||||
source_limit: int | None
|
||||
user_limit: int
|
||||
window_seconds: int
|
||||
block_seconds: int
|
||||
counters: LocalStore
|
||||
blocks: LocalStore
|
||||
redis_cache: RedisCache | None = None
|
||||
enabled: bool = True
|
||||
|
||||
@classmethod
|
||||
def from_request(
|
||||
cls,
|
||||
request: Request,
|
||||
general_settings: Mapping[str, object] | None,
|
||||
redis_cache: RedisCache | None,
|
||||
) -> LoginThrottle:
|
||||
settings: Final[Mapping[str, object]] = general_settings if general_settings is not None else EMPTY_MAPPING
|
||||
proxies: Final = declared_proxy_ranges(settings)
|
||||
resolved, _ = resolve_client_ip(
|
||||
request, TrustedProxyConfig(use_forwarded_for=bool(proxies), trusted_proxy_cidrs=proxies or ())
|
||||
)
|
||||
source_limit: Final = _source_limit(settings, resolved or LOGIN_THROTTLE_UNKNOWN_SOURCE)
|
||||
exempt: Final = source_limit == EXEMPT
|
||||
return cls(
|
||||
client_ip=resolved or LOGIN_THROTTLE_UNKNOWN_SOURCE,
|
||||
source_limit=source_limit if proxies is not None and resolved is not None and not exempt else None,
|
||||
user_limit=user_limit_for(source_limit),
|
||||
window_seconds=_int_setting(settings, WINDOW_KEY, DEFAULT_FAILED_LOGIN_WINDOW_SECONDS),
|
||||
block_seconds=_int_setting(settings, BLOCK_KEY, DEFAULT_FAILED_LOGIN_BLOCK_SECONDS),
|
||||
counters=_COUNTERS,
|
||||
blocks=_BLOCKS,
|
||||
redis_cache=redis_cache,
|
||||
enabled=not exempt and not _rate_limit_disabled(),
|
||||
)
|
||||
|
||||
def _keys(self, username: str) -> _Keys:
|
||||
group: Final = source_group(self.client_ip)
|
||||
user: Final = hashlib.sha256(username.casefold().encode("utf-8")).hexdigest()
|
||||
return _Keys(
|
||||
pair_counter=f"{LOGIN_THROTTLE_CACHE_KEY_PREFIX}:{{{group}}}:user:{user}",
|
||||
pair_block=f"{LOGIN_THROTTLE_CACHE_KEY_PREFIX}:{{{group}}}:block:user:{user}",
|
||||
source_counter=f"{LOGIN_THROTTLE_CACHE_KEY_PREFIX}:{{{group}}}:source",
|
||||
source_block=f"{LOGIN_THROTTLE_CACHE_KEY_PREFIX}:{{{group}}}:block:source",
|
||||
)
|
||||
|
||||
async def attempt(self, username: str) -> LoginAttempt:
|
||||
"""Refuses a blocked key before any credential is looked at; otherwise hands back the attempt to settle."""
|
||||
if not self.enabled:
|
||||
return LoginAttempt(throttle=self, username=username)
|
||||
block: Final = await self._active_block(self._keys(username))
|
||||
if block is None:
|
||||
return LoginAttempt(throttle=self, username=username)
|
||||
verbose_proxy_logger.warning(
|
||||
"Admin UI sign-in refused: the %s is blocked for %s more seconds; username=%r source=%s",
|
||||
block.scope,
|
||||
block.retry_after,
|
||||
username,
|
||||
self.client_ip,
|
||||
)
|
||||
raise self.refused(block.retry_after)
|
||||
|
||||
async def _active_block(self, keys: _Keys) -> Block | None:
|
||||
local: Final = self._local_block_ttls(keys)
|
||||
shared: Final = await self._shared_block_ttls(keys)
|
||||
user_ttl: Final = max(local[0], shared[0])
|
||||
source_ttl: Final = max(local[1], shared[1])
|
||||
if self.source_limit is not None and source_ttl > 0:
|
||||
return Block(scope="source", retry_after=source_ttl)
|
||||
if user_ttl > 0:
|
||||
return Block(scope="user", retry_after=user_ttl)
|
||||
return None
|
||||
|
||||
async def _shared_block_ttls(self, keys: _Keys) -> _BlockTtls:
|
||||
if self.redis_cache is None:
|
||||
return LOGIN_THROTTLE_NOT_BLOCKED
|
||||
try:
|
||||
return _LUA_BLOCK_TTLS.validate_python(
|
||||
await self.redis_cache.async_register_script(_BLOCK_TTLS_LUA)(keys, ())
|
||||
)
|
||||
except _REDIS_FAILURES as err:
|
||||
self._warn_redis(err)
|
||||
return LOGIN_THROTTLE_NOT_BLOCKED
|
||||
|
||||
def _local_block_ttls(self, keys: _Keys) -> _BlockTtls:
|
||||
return self._local_block_ttl(keys.pair_block), self._local_block_ttl(keys.source_block)
|
||||
|
||||
def _local_block_ttl(self, block_key: str) -> int:
|
||||
expires_at: Final = _LOCAL_BLOCK_EXPIRY.validate_python(self.blocks.get_cache(block_key))
|
||||
if expires_at is None:
|
||||
return 0
|
||||
return max(math.ceil(expires_at - time.time()), 0)
|
||||
|
||||
async def record_failure(self, username: str) -> _BlockTtls:
|
||||
keys: Final = self._keys(username)
|
||||
source_limit: Final = self.source_limit or 0
|
||||
if self.redis_cache is not None:
|
||||
try:
|
||||
return _LUA_BLOCK_TTLS.validate_python(
|
||||
await self.redis_cache.async_register_script(_RECORD_FAILURE_LUA)(
|
||||
keys, (self.user_limit, source_limit, self.window_seconds, self.block_seconds)
|
||||
)
|
||||
)
|
||||
except _REDIS_FAILURES as err:
|
||||
self._warn_redis(err)
|
||||
user_block: Final = self._local_bump(keys.pair_counter, keys.pair_block, self.user_limit)
|
||||
if source_limit == 0 or user_block > 0:
|
||||
return user_block, 0
|
||||
return user_block, self._local_bump(keys.source_counter, keys.source_block, source_limit)
|
||||
|
||||
def _local_bump(self, count_key: str, block_key: str, limit: int) -> int:
|
||||
blocked: Final = self._local_block_ttl(block_key)
|
||||
if blocked > 0:
|
||||
return blocked
|
||||
count: Final = int(self.counters.increment_cache(count_key, 1, ttl=self.window_seconds))
|
||||
if count <= limit:
|
||||
return 0
|
||||
self.blocks.set_cache(block_key, time.time() + self.block_seconds, ttl=self.block_seconds)
|
||||
return self.block_seconds
|
||||
|
||||
async def clear_pair(self, username: str) -> None:
|
||||
pair_counter: Final = self._keys(username).pair_counter
|
||||
if self.redis_cache is not None:
|
||||
try:
|
||||
await self.redis_cache.async_delete_cache(pair_counter)
|
||||
except _REDIS_FAILURES as err:
|
||||
self._warn_redis(err)
|
||||
self.counters.delete_cache(pair_counter)
|
||||
|
||||
def _warn_redis(self, err: Exception) -> None:
|
||||
verbose_proxy_logger.warning(
|
||||
"Redis failed while counting Admin UI sign-in attempts; using this worker's own counters "
|
||||
"until it recovers: %s",
|
||||
err,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def refused(retry_after: int) -> ProxyException:
|
||||
return ProxyException(
|
||||
message="Too many failed sign-in attempts. Try again later.",
|
||||
type=ProxyErrorTypes.auth_error,
|
||||
param="username",
|
||||
code=status.HTTP_429_TOO_MANY_REQUESTS,
|
||||
headers={"Retry-After": str(retry_after)}, # mutable-ok: ProxyException writes into its headers dict
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class LoginAttempt:
|
||||
throttle: LoginThrottle
|
||||
username: str
|
||||
|
||||
async def succeeded(self) -> None:
|
||||
if not self.throttle.enabled:
|
||||
return
|
||||
await self.throttle.clear_pair(self.username)
|
||||
|
||||
async def failed(self) -> None:
|
||||
if not self.throttle.enabled:
|
||||
return
|
||||
user_block, source_block = await self.throttle.record_failure(self.username)
|
||||
if user_block == 0 and source_block == 0:
|
||||
return
|
||||
verbose_proxy_logger.warning(
|
||||
"Admin UI sign-in blocked for %s seconds after too many failures; scope=%s username=%r source=%s",
|
||||
user_block or source_block,
|
||||
"user" if user_block else "source",
|
||||
self.username,
|
||||
self.throttle.client_ip,
|
||||
)
|
||||
|
|
@ -27,6 +27,7 @@ from litellm.proxy._types import (
|
|||
UserAPIKeyAuth,
|
||||
)
|
||||
from litellm.proxy.auth.auth_utils import is_sso_provider_fully_configured
|
||||
from litellm.proxy.auth.login_throttle import LoginAttempt, LoginThrottle
|
||||
from litellm.proxy.management_endpoints.internal_user_endpoints import user_update
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
generate_key_helper_fn,
|
||||
|
|
@ -44,6 +45,11 @@ from litellm.repositories.user_repository import UserRepository
|
|||
from litellm.secret_managers.main import get_secret_bool
|
||||
from litellm.types.proxy.ui_sso import ReturnedUITokenObject
|
||||
|
||||
INVALID_UI_CREDENTIALS_MESSAGE: Final = (
|
||||
"Invalid credentials used to access UI. Check 'UI_USERNAME' and 'UI_PASSWORD', or the password set for your user"
|
||||
)
|
||||
INVALID_USER_PASSWORD_MESSAGE: Final = "Invalid credentials used to access UI. Check the password set for your user"
|
||||
|
||||
|
||||
async def _rehash_password_if_needed(user_id: str, password: str, stored: str) -> None:
|
||||
"""Rehash legacy password (SHA256) to scrypt on successful login."""
|
||||
|
|
@ -92,6 +98,21 @@ def _matches_env_credentials(username: str, password: str, master_key: str | Non
|
|||
)
|
||||
|
||||
|
||||
def _admin_credentials_match(
|
||||
username: str, password: str, master_key: str, general_settings: Mapping[str, object]
|
||||
) -> bool:
|
||||
return general_settings.get("disable_env_credential_login") is not True and _matches_env_credentials(
|
||||
username, password, master_key
|
||||
)
|
||||
|
||||
|
||||
def _invalid_credentials_message(general_settings: Mapping[str, object]) -> str:
|
||||
"""One rejection message for unknown usernames and wrong passwords alike, so neither can be enumerated."""
|
||||
if is_env_credential_login_enabled(general_settings):
|
||||
return INVALID_UI_CREDENTIALS_MESSAGE
|
||||
return INVALID_USER_PASSWORD_MESSAGE
|
||||
|
||||
|
||||
def is_env_credential_login_enabled(general_settings: Mapping[str, object]) -> bool:
|
||||
"""Whether a login with UI_USERNAME/UI_PASSWORD (or the master-key fallback) can succeed.
|
||||
|
||||
|
|
@ -137,6 +158,7 @@ async def authenticate_user(
|
|||
password: str,
|
||||
master_key: str | None,
|
||||
prisma_client: PrismaClient | None,
|
||||
throttle: LoginThrottle,
|
||||
general_settings: Mapping[str, object] = MappingProxyType({}),
|
||||
) -> LoginResult:
|
||||
"""
|
||||
|
|
@ -151,6 +173,7 @@ async def authenticate_user(
|
|||
password: Password from the login form
|
||||
master_key: Master key for the proxy (required)
|
||||
prisma_client: Prisma database client (optional)
|
||||
throttle: Failed sign-in accounting for this request's source address
|
||||
general_settings: Proxy general_settings, checked for
|
||||
`disable_password_login_when_sso_enabled` and
|
||||
`disable_env_credential_login`
|
||||
|
|
@ -163,9 +186,11 @@ async def authenticate_user(
|
|||
or if username/password login is disabled while SSO is configured
|
||||
|
||||
Recovery: an admin locked out of the UI by
|
||||
`disable_password_login_when_sso_enabled` can still administer the proxy over
|
||||
the API with the master key (Authorization: Bearer <master_key>), which never
|
||||
goes through this function. To restore UI username/password login, unset the
|
||||
`disable_password_login_when_sso_enabled`, or by the failed sign-in block in
|
||||
`throttle`, can still administer the proxy over the API with the master key
|
||||
(Authorization: Bearer <master_key>), which never goes through this function.
|
||||
No credential, the env admin credentials and the master key included, is
|
||||
exempt from the block. To restore UI username/password login, unset the
|
||||
setting in config.yaml (or the DB-persisted general_settings) and restart the
|
||||
proxy; this is a deliberate, auditable config change rather than a hidden
|
||||
bypass.
|
||||
|
|
@ -194,6 +219,19 @@ async def authenticate_user(
|
|||
code=500,
|
||||
)
|
||||
|
||||
attempt: Final = await throttle.attempt(username)
|
||||
return await _sign_in(username, password, master_key, prisma_client, attempt, general_settings)
|
||||
|
||||
|
||||
async def _sign_in(
|
||||
username: str,
|
||||
password: str,
|
||||
master_key: str,
|
||||
prisma_client: PrismaClient | None,
|
||||
attempt: LoginAttempt,
|
||||
general_settings: Mapping[str, object],
|
||||
) -> LoginResult:
|
||||
admin_credentials_match: Final = _admin_credentials_match(username, password, master_key, general_settings)
|
||||
# Check if we can find the `username` in the db. On the UI, users can enter username=their email
|
||||
_user_row: LiteLLM_UserTable | None = None
|
||||
user_role: (
|
||||
|
|
@ -219,20 +257,13 @@ async def authenticate_user(
|
|||
- Login with UI_USERNAME and UI_PASSWORD
|
||||
- Login with Invite Link `user_email` and `password` combination
|
||||
"""
|
||||
if general_settings.get("disable_env_credential_login") is not True and _matches_env_credentials(
|
||||
username, password, master_key
|
||||
):
|
||||
if admin_credentials_match:
|
||||
# Non SSO -> If user is using UI_USERNAME and UI_PASSWORD they are Proxy admin
|
||||
user_role = LitellmUserRoles.PROXY_ADMIN
|
||||
user_id = LITELLM_PROXY_ADMIN_NAME
|
||||
|
||||
# we want the key created to have PROXY_ADMIN_PERMISSIONS
|
||||
key_user_id = LITELLM_PROXY_ADMIN_NAME
|
||||
if (
|
||||
os.getenv("PROXY_ADMIN_ID", None) is not None and os.environ["PROXY_ADMIN_ID"] == user_id
|
||||
) or user_id == LITELLM_PROXY_ADMIN_NAME:
|
||||
# checks if user is admin
|
||||
key_user_id = os.getenv("PROXY_ADMIN_ID", LITELLM_PROXY_ADMIN_NAME)
|
||||
key_user_id: Final = os.getenv("PROXY_ADMIN_ID", LITELLM_PROXY_ADMIN_NAME)
|
||||
|
||||
# Admin is Authe'd in - generate key for the UI to access Proxy
|
||||
|
||||
|
|
@ -294,6 +325,8 @@ async def authenticate_user(
|
|||
|
||||
key = ExperimentalUIJWTToken.get_experimental_ui_login_jwt_auth_token(user_info)
|
||||
|
||||
await attempt.succeeded()
|
||||
|
||||
return LoginResult(
|
||||
user_id=user_id,
|
||||
key=key,
|
||||
|
|
@ -349,6 +382,8 @@ async def authenticate_user(
|
|||
|
||||
key = response["token"]
|
||||
|
||||
await attempt.succeeded()
|
||||
|
||||
return LoginResult(
|
||||
user_id=user_id,
|
||||
key=key,
|
||||
|
|
@ -357,20 +392,17 @@ async def authenticate_user(
|
|||
login_method="username_password",
|
||||
)
|
||||
else:
|
||||
await attempt.failed()
|
||||
raise ProxyException(
|
||||
message=f"Invalid credentials used to access UI.\nNot valid credentials for {username}",
|
||||
message=_invalid_credentials_message(general_settings),
|
||||
type=ProxyErrorTypes.auth_error,
|
||||
param="invalid_credentials",
|
||||
code=401,
|
||||
)
|
||||
else:
|
||||
env_credentials_hint: Final = (
|
||||
"\nCheck 'UI_USERNAME', 'UI_PASSWORD' in .env file"
|
||||
if is_env_credential_login_enabled(general_settings)
|
||||
else ""
|
||||
)
|
||||
await attempt.failed()
|
||||
raise ProxyException(
|
||||
message=f"Invalid credentials used to access UI.{env_credentials_hint}",
|
||||
message=_invalid_credentials_message(general_settings),
|
||||
type=ProxyErrorTypes.auth_error,
|
||||
param="invalid_credentials",
|
||||
code=401,
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import ipaddress
|
||||
from collections.abc import Sequence
|
||||
from typing import Any, Final
|
||||
|
||||
from fastapi import Request
|
||||
|
|
@ -19,7 +20,7 @@ class NetworkContext(BaseModel):
|
|||
|
||||
class TrustedProxyConfig(BaseModel):
|
||||
use_forwarded_for: bool = False
|
||||
trusted_proxy_cidrs: list[str] = Field(default_factory=list)
|
||||
trusted_proxy_cidrs: Sequence[str] = Field(default_factory=tuple)
|
||||
|
||||
|
||||
def normalize_cidr_ranges(configured_ranges: Any, *, setting_name: str = "trusted_proxy_cidrs") -> list[str]:
|
||||
|
|
@ -49,6 +50,12 @@ def parse_trusted_proxy_ranges(
|
|||
return networks
|
||||
|
||||
|
||||
def _unmapped(addr: ipaddress.IPv4Address | ipaddress.IPv6Address) -> ipaddress.IPv4Address | ipaddress.IPv6Address:
|
||||
if isinstance(addr, ipaddress.IPv6Address) and addr.ipv4_mapped is not None:
|
||||
return addr.ipv4_mapped
|
||||
return addr
|
||||
|
||||
|
||||
def ip_in_networks(client_ip: str | None, networks: list[TrustedProxyNetwork]) -> bool:
|
||||
if not client_ip or not networks:
|
||||
return False
|
||||
|
|
@ -56,7 +63,8 @@ def ip_in_networks(client_ip: str | None, networks: list[TrustedProxyNetwork]) -
|
|||
addr: Final = ipaddress.ip_address(client_ip.strip())
|
||||
except ValueError:
|
||||
return False
|
||||
return any(addr in network for network in networks)
|
||||
candidates: Final = (addr, _unmapped(addr))
|
||||
return any(candidate in network for candidate in candidates for network in networks)
|
||||
|
||||
|
||||
def _is_valid_ip(value: str) -> bool:
|
||||
|
|
|
|||
|
|
@ -326,7 +326,12 @@ class RouteChecks:
|
|||
pass
|
||||
elif route.startswith("/v1/mcp/") or route.startswith("/mcp-rest/"):
|
||||
pass # authN/authZ handled by api itself
|
||||
elif RouteChecks.check_passthrough_route_access(route=route, user_api_key_dict=valid_token):
|
||||
elif RouteChecks.check_passthrough_route_access(route=route, user_api_key_dict=valid_token) or (
|
||||
valid_token.is_team_service_account
|
||||
and RouteChecks.check_route_access(
|
||||
route=route, allowed_routes=LiteLLMRoutes.team_service_account_key_routes.value
|
||||
)
|
||||
):
|
||||
pass
|
||||
elif valid_token.allowed_routes is not None:
|
||||
# check if route is in allowed_routes (exact match or prefix match)
|
||||
|
|
|
|||
|
|
@ -87,6 +87,12 @@ class SettingsStore(MutableMapping[str, JsonValue]):
|
|||
)
|
||||
self._deleted_runtime_keys = self._deleted_runtime_keys | frozenset((key,))
|
||||
|
||||
def clear(self) -> None:
|
||||
self._deleted_runtime_keys = frozenset(key for key in self._keys() if not self.owned_by_config(key))
|
||||
self._runtime_values = MappingProxyType(
|
||||
{key: value for key, value in self._runtime_values.items() if self.owned_by_config(key)}
|
||||
)
|
||||
|
||||
def __iter__(self) -> Iterator[str]:
|
||||
return iter(
|
||||
key
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ if TYPE_CHECKING:
|
|||
from litellm.integrations.custom_guardrail import CustomGuardrail
|
||||
from litellm.router import Router
|
||||
|
||||
COMPRESSION_GUARDRAIL_PROVIDERS: Final = frozenset({"headroom", "compresr"})
|
||||
COMPRESSION_GUARDRAIL_PROVIDERS: Final = frozenset({"headroom", "compresr", "typesafe"})
|
||||
_NO_COMPRESSION: Final = "none"
|
||||
|
||||
# A ContextVar, not metadata: metadata reaches spend logs the caller can read, and a
|
||||
|
|
|
|||
|
|
@ -0,0 +1,74 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING, Final
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
from litellm.types.guardrails import (
|
||||
GuardrailEventHooks,
|
||||
Mode,
|
||||
SupportedGuardrailIntegrations,
|
||||
)
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.typesafe import (
|
||||
TypeSafeGuardrailOptionalParams,
|
||||
)
|
||||
|
||||
from .typesafe import TypeSafeGuardrail
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.types.guardrails import Guardrail, LitellmParams
|
||||
|
||||
|
||||
def _coerce_event_hook(
|
||||
mode: str | list[str] | Mode,
|
||||
) -> GuardrailEventHooks | list[GuardrailEventHooks] | Mode:
|
||||
if isinstance(mode, Mode):
|
||||
return mode
|
||||
if isinstance(mode, list):
|
||||
return [ # mutable-ok: CustomGuardrail event_hook contract wants a list
|
||||
GuardrailEventHooks(item) for item in mode
|
||||
]
|
||||
return GuardrailEventHooks(mode)
|
||||
|
||||
|
||||
def _optional_params(litellm_params: LitellmParams) -> TypeSafeGuardrailOptionalParams:
|
||||
value: Final = litellm_params.optional_params
|
||||
if isinstance(value, TypeSafeGuardrailOptionalParams):
|
||||
return value
|
||||
if isinstance(value, BaseModel):
|
||||
return TypeSafeGuardrailOptionalParams.model_validate(value.model_dump())
|
||||
return TypeSafeGuardrailOptionalParams()
|
||||
|
||||
|
||||
def initialize_guardrail(litellm_params: LitellmParams, guardrail: Guardrail) -> TypeSafeGuardrail:
|
||||
import litellm
|
||||
|
||||
optional_params: Final = _optional_params(litellm_params)
|
||||
|
||||
_callback: Final = TypeSafeGuardrail(
|
||||
api_base=litellm_params.api_base,
|
||||
api_key=litellm_params.api_key,
|
||||
model=litellm_params.model,
|
||||
relevance_threshold=optional_params.relevance_threshold,
|
||||
min_chars_to_evaluate=optional_params.min_chars_to_evaluate,
|
||||
max_result_chars_in_state=optional_params.max_result_chars_in_state,
|
||||
guardrail_name=guardrail["guardrail_name"],
|
||||
event_hook=_coerce_event_hook(litellm_params.mode),
|
||||
default_on=litellm_params.default_on or False,
|
||||
unreachable_fallback=(
|
||||
litellm_params.unreachable_fallback if "unreachable_fallback" in litellm_params.model_fields_set else None
|
||||
),
|
||||
)
|
||||
litellm.logging_callback_manager.add_litellm_callback( # pyright: ignore[reportUnknownMemberType] # callback manager is untyped
|
||||
_callback
|
||||
)
|
||||
return _callback
|
||||
|
||||
|
||||
guardrail_initializer_registry: Final = { # mutable-ok: guardrail_registry discovery checks isinstance(registry, dict)
|
||||
SupportedGuardrailIntegrations.TYPESAFE.value: initialize_guardrail,
|
||||
}
|
||||
|
||||
guardrail_class_registry: Final = { # mutable-ok: guardrail_registry discovery checks isinstance(registry, dict)
|
||||
SupportedGuardrailIntegrations.TYPESAFE.value: TypeSafeGuardrail,
|
||||
}
|
||||
416
litellm/proxy/guardrails/guardrail_hooks/typesafe/typesafe.py
Normal file
416
litellm/proxy/guardrails/guardrail_hooks/typesafe/typesafe.py
Normal file
|
|
@ -0,0 +1,416 @@
|
|||
"""TypeSafe (Jev) relevance-based compaction guardrail.
|
||||
|
||||
Instead of summarizing tool output, the guardrail asks TypeSafe's Jev model
|
||||
one yes/no question per completed tool exchange ("is this result still needed
|
||||
for the current task?") over ``POST {api_base}/v1/systemone`` and blanks the
|
||||
tool results Jev judges no longer relevant.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import time
|
||||
from collections.abc import Mapping, Sequence
|
||||
from typing import TYPE_CHECKING, Annotated, Final, Literal
|
||||
|
||||
import httpx
|
||||
from fastapi import HTTPException
|
||||
from httpx import Response as HttpxResponse
|
||||
from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, ValidationError
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.compression.compress import get_protected_indices
|
||||
from litellm.integrations.custom_guardrail import (
|
||||
CustomGuardrail,
|
||||
log_guardrail_information, # pyright: ignore[reportUnknownVariableType] # decorator is untyped in custom_guardrail
|
||||
)
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import group_tool_exchanges
|
||||
from litellm.llms.custom_httpx.http_handler import (
|
||||
AsyncHTTPHandler,
|
||||
get_async_httpx_client, # pyright: ignore[reportUnknownVariableType] # helper is untyped in http_handler
|
||||
httpxSpecialProvider,
|
||||
)
|
||||
from litellm.proxy.guardrails.guardrail_hooks.content_text import content_to_text
|
||||
from litellm.secret_managers.main import get_secret_str
|
||||
from litellm.types.guardrails import GuardrailEventHooks, Mode
|
||||
from litellm.types.utils import GenericGuardrailAPIInputs
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import (
|
||||
Logging as LiteLLMLoggingObj,
|
||||
)
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.typesafe import (
|
||||
TypeSafeGuardrailConfigModel,
|
||||
)
|
||||
|
||||
DEFAULT_API_BASE: Final = "https://api.typesafe.ai"
|
||||
DEFAULT_MODEL: Final = "jev-latest"
|
||||
DEFAULT_RELEVANCE_THRESHOLD: Final = 0.2
|
||||
DEFAULT_MIN_CHARS_TO_EVALUATE: Final = 200
|
||||
DEFAULT_MAX_RESULT_CHARS_IN_STATE: Final = 4000
|
||||
_MAX_EXCHANGES_EVALUATED: Final = 200
|
||||
_JEV_TIMEOUT_SECONDS: Final = 30.0
|
||||
DROPPED_RESULT_TEXT: Final = (
|
||||
"[Tool result removed by TypeSafe compaction: judged no longer relevant to the current task]"
|
||||
)
|
||||
_ELISION_MARKER: Final = "\n... [middle truncated] ...\n"
|
||||
|
||||
|
||||
_STR_OBJECT_DICT_ADAPTER: Final = TypeAdapter(dict[str, object])
|
||||
_OBJECT_LIST_ADAPTER: Final = TypeAdapter(list[object])
|
||||
|
||||
|
||||
def _as_str_object_dict(value: object) -> dict[str, object] | None:
|
||||
try:
|
||||
return _STR_OBJECT_DICT_ADAPTER.validate_python(value)
|
||||
except ValidationError:
|
||||
return None
|
||||
|
||||
|
||||
def _as_object_list(value: object) -> list[object] | None:
|
||||
try:
|
||||
return _OBJECT_LIST_ADAPTER.validate_python(value)
|
||||
except ValidationError:
|
||||
return None
|
||||
|
||||
|
||||
def _safe_response_text(response: HttpxResponse | None, limit: int = 500) -> str:
|
||||
if response is None:
|
||||
return ""
|
||||
try:
|
||||
text: Final = response.text
|
||||
except httpx.DecodingError:
|
||||
return "<undecodable response body>"
|
||||
return (text or "")[:limit]
|
||||
|
||||
|
||||
class _JevNoulAnswer(BaseModel):
|
||||
model_config = ConfigDict(frozen=True, allow_inf_nan=False)
|
||||
|
||||
type: Literal["noul"]
|
||||
noul: Annotated[float, Field(ge=0.0, le=1.0)]
|
||||
|
||||
|
||||
class _JevSystemOneResponse(BaseModel):
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
answers: Mapping[str, _JevNoulAnswer]
|
||||
|
||||
|
||||
_JEV_RESPONSE_ADAPTER: Final = TypeAdapter(_JevSystemOneResponse)
|
||||
|
||||
|
||||
def _truncate_for_state(text: str, max_chars: int) -> str:
|
||||
"""Keeps the head and tail within ``max_chars`` so Jev sees both ends of a long result."""
|
||||
if len(text) <= max_chars:
|
||||
return text
|
||||
if max_chars <= len(_ELISION_MARKER):
|
||||
return text[:max_chars]
|
||||
budget: Final = max_chars - len(_ELISION_MARKER)
|
||||
head: Final = budget // 2
|
||||
return text[:head] + _ELISION_MARKER + text[len(text) - (budget - head) :]
|
||||
|
||||
|
||||
def _question_instructions(question_id: str) -> str:
|
||||
return (
|
||||
f"Is tool exchange `{question_id}` in `tool_exchanges` still needed by the assistant to "
|
||||
"complete `task`? Answer yes if its result contains information the assistant has not yet "
|
||||
"fully used or will need again; answer no if it is off-topic, superseded, or already "
|
||||
"incorporated into later messages."
|
||||
)
|
||||
|
||||
|
||||
def _tool_call_entry(tool_call: object) -> dict[str, object] | None:
|
||||
parsed_call = _as_str_object_dict(tool_call)
|
||||
if parsed_call is None:
|
||||
return None
|
||||
function = _as_str_object_dict(parsed_call.get("function"))
|
||||
fn = function if function is not None else parsed_call
|
||||
return {"name": fn.get("name"), "arguments": fn.get("arguments")} # mutable-ok: serialized to JSON
|
||||
|
||||
|
||||
def _tool_call_entries(assistant_message: Mapping[str, object]) -> tuple[dict[str, object], ...]:
|
||||
tool_calls: Final = _as_object_list(assistant_message.get("tool_calls"))
|
||||
if tool_calls is None:
|
||||
return ()
|
||||
return tuple(entry for tool_call in tool_calls if (entry := _tool_call_entry(tool_call)) is not None)
|
||||
|
||||
|
||||
def _protected_indices(messages: Sequence[Mapping[str, object]]) -> frozenset[int]:
|
||||
"""``get_protected_indices`` expanded over whole tool exchanges, so the most recent exchange is never evaluated."""
|
||||
protected: Final = frozenset(get_protected_indices(messages))
|
||||
return protected | frozenset(
|
||||
index
|
||||
for group in group_tool_exchanges(messages)
|
||||
if any(member in protected for member in group)
|
||||
for index in group
|
||||
)
|
||||
|
||||
|
||||
class TypeSafeGuardrail(CustomGuardrail):
|
||||
def __init__(
|
||||
self,
|
||||
api_base: str | None = None,
|
||||
api_key: str | None = None,
|
||||
model: str | None = None,
|
||||
relevance_threshold: float | None = None,
|
||||
min_chars_to_evaluate: int | None = None,
|
||||
max_result_chars_in_state: int | None = None,
|
||||
unreachable_fallback: str | None = None,
|
||||
guardrail_name: str | None = None,
|
||||
event_hook: GuardrailEventHooks | list[GuardrailEventHooks] | Mode | None = None,
|
||||
default_on: bool = False,
|
||||
async_handler: AsyncHTTPHandler | None = None,
|
||||
) -> None:
|
||||
raw_api_base: Final = (api_base or get_secret_str("TYPESAFE_API_BASE") or DEFAULT_API_BASE).rstrip("/")
|
||||
self.typesafe_api_base = raw_api_base
|
||||
self.typesafe_api_key = api_key or get_secret_str("TYPESAFE_API_KEY")
|
||||
if not self.typesafe_api_key:
|
||||
raise ValueError(
|
||||
"TypeSafe guardrail requires an API key. Set `api_key` in the "
|
||||
"guardrail config or the TYPESAFE_API_KEY env var."
|
||||
)
|
||||
self.jev_model = model or DEFAULT_MODEL
|
||||
self.relevance_threshold = DEFAULT_RELEVANCE_THRESHOLD if relevance_threshold is None else relevance_threshold
|
||||
self.min_chars_to_evaluate = (
|
||||
DEFAULT_MIN_CHARS_TO_EVALUATE if min_chars_to_evaluate is None else min_chars_to_evaluate
|
||||
)
|
||||
self.max_result_chars_in_state = (
|
||||
DEFAULT_MAX_RESULT_CHARS_IN_STATE if max_result_chars_in_state is None else max_result_chars_in_state
|
||||
)
|
||||
self.unreachable_fallback: Literal["fail_closed", "fail_open"] = (
|
||||
"fail_closed" if unreachable_fallback == "fail_closed" else "fail_open"
|
||||
)
|
||||
self.async_handler: AsyncHTTPHandler = async_handler or get_async_httpx_client(
|
||||
llm_provider=httpxSpecialProvider.GuardrailCallback,
|
||||
)
|
||||
super().__init__( # pyright: ignore[reportUnknownMemberType] # CustomGuardrail.__init__ is untyped
|
||||
guardrail_name=guardrail_name,
|
||||
event_hook=event_hook,
|
||||
default_on=default_on,
|
||||
)
|
||||
|
||||
def _handle_failure(self, error: str, log_detail: dict[str, object]) -> None:
|
||||
"""fail_open logs and returns; fail_closed raises a generic 502 (upstream bodies stay in server logs)."""
|
||||
if self.unreachable_fallback == "fail_open":
|
||||
verbose_proxy_logger.warning(
|
||||
"TypeSafe: %s; fail_open configured, forwarding request uncompacted. detail=%s",
|
||||
error,
|
||||
log_detail,
|
||||
)
|
||||
return
|
||||
verbose_proxy_logger.error("TypeSafe: %s. detail=%s", error, log_detail)
|
||||
raise HTTPException(status_code=502, detail={"error": error}) # mutable-ok: FastAPI wants a dict detail
|
||||
|
||||
def _candidate_exchanges(self, messages: Sequence[dict[str, object]]) -> tuple[tuple[int, ...], ...]:
|
||||
"""Completed tool exchanges eligible for evaluation: unprotected, and long enough to be worth a call."""
|
||||
protected: Final = _protected_indices(messages)
|
||||
candidates: Final = tuple(
|
||||
group
|
||||
for group in group_tool_exchanges(messages)
|
||||
if len(group) >= 2
|
||||
and messages[group[0]].get("role") == "assistant"
|
||||
and not any(member in protected for member in group)
|
||||
and len(self._exchange_tool_text(messages, group)) >= self.min_chars_to_evaluate
|
||||
)
|
||||
return candidates[-_MAX_EXCHANGES_EVALUATED:]
|
||||
|
||||
@staticmethod
|
||||
def _exchange_tool_text(messages: Sequence[dict[str, object]], group: tuple[int, ...]) -> str:
|
||||
return "".join(
|
||||
content_to_text(messages[index].get("content"))
|
||||
for index in group[1:]
|
||||
if messages[index].get("role") in ("tool", "function")
|
||||
)
|
||||
|
||||
def _build_state(
|
||||
self, messages: Sequence[dict[str, object]], candidates: tuple[tuple[int, ...], ...]
|
||||
) -> dict[str, object]:
|
||||
task: Final = next(
|
||||
(
|
||||
content_to_text(messages[index].get("content"))
|
||||
for index in range(len(messages) - 1, -1, -1)
|
||||
if messages[index].get("role") == "user"
|
||||
),
|
||||
"",
|
||||
)
|
||||
system: Final = "\n\n".join(
|
||||
content_to_text(message.get("content")) for message in messages if message.get("role") == "system"
|
||||
)
|
||||
tool_exchanges: Final = { # mutable-ok: accumulated once, serialized to JSON
|
||||
f"e{ordinal}": { # mutable-ok: serialized to JSON
|
||||
"tool_calls": _tool_call_entries(messages[group[0]]),
|
||||
"result": _truncate_for_state(
|
||||
self._exchange_tool_text(messages, group), self.max_result_chars_in_state
|
||||
),
|
||||
}
|
||||
for ordinal, group in enumerate(candidates)
|
||||
}
|
||||
return {"task": task, "system": system, "tool_exchanges": tool_exchanges} # mutable-ok: serialized to JSON
|
||||
|
||||
async def _call_systemone(
|
||||
self, state: dict[str, object], question_ids: Sequence[str]
|
||||
) -> _JevSystemOneResponse | None:
|
||||
"""Returns the response, or None when the service failed and fail_open applies."""
|
||||
payload: Final[dict[str, object]] = { # mutable-ok: serialized to JSON by httpx
|
||||
"model": self.jev_model,
|
||||
"state": state,
|
||||
"questions": { # mutable-ok: serialized to JSON
|
||||
question_id: { # mutable-ok: serialized to JSON
|
||||
"type": "noul",
|
||||
"instructions": _question_instructions(question_id),
|
||||
}
|
||||
for question_id in question_ids
|
||||
},
|
||||
}
|
||||
try:
|
||||
raw_response: HttpxResponse = await self.async_handler.post( # pyright: ignore[reportUnknownMemberType] # AsyncHTTPHandler.post is untyped
|
||||
url=f"{self.typesafe_api_base}/v1/systemone",
|
||||
json=payload,
|
||||
headers={ # mutable-ok: httpx header contract is a dict
|
||||
"Authorization": f"Bearer {self.typesafe_api_key}",
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
timeout=_JEV_TIMEOUT_SECONDS,
|
||||
)
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception as e:
|
||||
detail: Final[dict[str, object]] = (
|
||||
{ # mutable-ok: log detail record
|
||||
"error_type": type(e).__name__,
|
||||
"detail": str(e),
|
||||
"status_code": e.response.status_code,
|
||||
"body": _safe_response_text(e.response),
|
||||
}
|
||||
if isinstance(e, httpx.HTTPStatusError)
|
||||
else {"error_type": type(e).__name__, "detail": str(e)} # mutable-ok: log detail record
|
||||
)
|
||||
self._handle_failure("TypeSafe evaluation service request failed", detail)
|
||||
return None
|
||||
if not 200 <= raw_response.status_code < 300:
|
||||
self._handle_failure(
|
||||
"TypeSafe evaluation service returned an error",
|
||||
{ # mutable-ok: log detail record
|
||||
"status_code": raw_response.status_code,
|
||||
"body": _safe_response_text(raw_response),
|
||||
},
|
||||
)
|
||||
return None
|
||||
try:
|
||||
body: Final[object] = raw_response.json() # pyright: ignore[reportAny] # httpx Response.json() is untyped
|
||||
except (ValueError, httpx.DecodingError, RecursionError):
|
||||
self._handle_failure(
|
||||
"TypeSafe evaluation service returned an unreadable response",
|
||||
{"body": _safe_response_text(raw_response)}, # mutable-ok: log detail record
|
||||
)
|
||||
return None
|
||||
try:
|
||||
return _JEV_RESPONSE_ADAPTER.validate_python(body)
|
||||
except ValidationError:
|
||||
self._handle_failure(
|
||||
"TypeSafe evaluation service returned unexpected response shape",
|
||||
{"body": _safe_response_text(raw_response)}, # mutable-ok: log detail record
|
||||
)
|
||||
return None
|
||||
|
||||
@log_guardrail_information
|
||||
async def apply_guardrail(
|
||||
self,
|
||||
inputs: GenericGuardrailAPIInputs,
|
||||
request_data: dict[str, object],
|
||||
input_type: Literal["request", "response"],
|
||||
logging_obj: LiteLLMLoggingObj | None = None,
|
||||
) -> GenericGuardrailAPIInputs:
|
||||
if input_type != "request":
|
||||
return inputs
|
||||
|
||||
structured_messages: Final = _as_object_list(inputs.get("structured_messages"))
|
||||
if not structured_messages:
|
||||
return inputs
|
||||
parsed_messages: Final = tuple(_as_str_object_dict(m) for m in structured_messages)
|
||||
if any(m is None for m in parsed_messages):
|
||||
return inputs
|
||||
messages: Final = tuple(m for m in parsed_messages if m is not None)
|
||||
|
||||
candidates: Final = self._candidate_exchanges(messages)
|
||||
if not candidates:
|
||||
verbose_proxy_logger.debug("TypeSafe: no completed tool exchanges eligible for evaluation")
|
||||
return inputs
|
||||
|
||||
question_ids: Final = tuple(f"e{ordinal}" for ordinal in range(len(candidates)))
|
||||
state: Final = self._build_state(messages, candidates)
|
||||
|
||||
start_time: Final = time.monotonic()
|
||||
response: Final = await self._call_systemone(state, question_ids)
|
||||
end_time: Final = time.monotonic()
|
||||
if response is None:
|
||||
self.add_standard_logging_guardrail_information_to_request_data( # pyright: ignore[reportUnknownMemberType] # untyped base helper
|
||||
guardrail_json_response={ # mutable-ok: must stay JSON-serializable for shared logging
|
||||
"error": "TypeSafe evaluation unavailable; request forwarded uncompacted",
|
||||
"model": self.jev_model,
|
||||
},
|
||||
request_data=request_data,
|
||||
guardrail_status="guardrail_failed_to_respond",
|
||||
guardrail_provider="typesafe",
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
duration=end_time - start_time,
|
||||
)
|
||||
return inputs
|
||||
|
||||
dropped_ordinals: Final = frozenset(
|
||||
ordinal
|
||||
for ordinal in range(len(candidates))
|
||||
if (answer := response.answers.get(f"e{ordinal}")) is not None and answer.noul < self.relevance_threshold
|
||||
)
|
||||
dropped_tool_indices: Final[frozenset[int]] = frozenset(
|
||||
index
|
||||
for ordinal in dropped_ordinals
|
||||
for index in candidates[ordinal][1:]
|
||||
if messages[index].get("role") in ("tool", "function")
|
||||
)
|
||||
if not dropped_tool_indices:
|
||||
verbose_proxy_logger.debug("TypeSafe: all evaluated exchanges still relevant; request unchanged")
|
||||
return inputs
|
||||
|
||||
compacted_messages: Final = [ # mutable-ok: structured_messages contract is a list of dicts
|
||||
{**message, "content": DROPPED_RESULT_TEXT} # mutable-ok: JSON message row
|
||||
if index in dropped_tool_indices
|
||||
else message
|
||||
for index, message in enumerate(messages)
|
||||
]
|
||||
chars_removed: Final = sum(
|
||||
len(content_to_text(messages[index].get("content"))) - len(DROPPED_RESULT_TEXT)
|
||||
for index in dropped_tool_indices
|
||||
)
|
||||
exchanges_dropped: Final = len(dropped_ordinals)
|
||||
verbose_proxy_logger.info(
|
||||
"TypeSafe: evaluated %s tool exchange(s), dropped %s, ~%s chars removed",
|
||||
len(candidates),
|
||||
exchanges_dropped,
|
||||
chars_removed,
|
||||
)
|
||||
self.add_standard_logging_guardrail_information_to_request_data( # pyright: ignore[reportUnknownMemberType] # untyped base helper
|
||||
guardrail_json_response={ # mutable-ok: must stay JSON-serializable for shared logging
|
||||
"exchanges_evaluated": len(candidates),
|
||||
"exchanges_dropped": exchanges_dropped,
|
||||
"chars_removed": chars_removed,
|
||||
"model": self.jev_model,
|
||||
},
|
||||
request_data=request_data,
|
||||
guardrail_status="success",
|
||||
guardrail_provider="typesafe",
|
||||
start_time=start_time,
|
||||
end_time=end_time,
|
||||
duration=end_time - start_time,
|
||||
)
|
||||
return {**inputs, "structured_messages": compacted_messages} # pyright: ignore[reportReturnType] # mutable-ok: inputs protocol is a plain dict # plain dicts satisfy AllMessageValues at runtime
|
||||
|
||||
@staticmethod
|
||||
def get_config_model() -> type[TypeSafeGuardrailConfigModel] | None:
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.typesafe import (
|
||||
TypeSafeGuardrailConfigModel,
|
||||
)
|
||||
|
||||
return TypeSafeGuardrailConfigModel
|
||||
|
|
@ -9,7 +9,7 @@ from fastapi import HTTPException, status
|
|||
from typing_extensions import ReadOnly, TypedDict
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.constants import PTU_SENTINEL_API_KEY
|
||||
from litellm.constants import PTU_SENTINEL_API_KEY, USAGE_TOP_API_KEYS_LIMIT
|
||||
from litellm.proxy._types import CommonProxyErrors
|
||||
from litellm.proxy.spend_tracking.key_metadata_recovery import (
|
||||
attach_user_emails,
|
||||
|
|
@ -146,15 +146,9 @@ class _AggregatedSpendData(TypedDict):
|
|||
totals: SpendMetrics
|
||||
|
||||
|
||||
class _GroupingSetsRow(SimpleNamespace):
|
||||
class _RollupMetricsRow(SimpleNamespace):
|
||||
date: str
|
||||
api_key: str | None
|
||||
model: str | None
|
||||
model_group: str | None
|
||||
custom_llm_provider: str | None
|
||||
mcp_namespaced_tool_name: str | None
|
||||
endpoint: str | None
|
||||
group_level: int
|
||||
spend: float | None
|
||||
prompt_tokens: int | None
|
||||
completion_tokens: int | None
|
||||
|
|
@ -172,12 +166,46 @@ class _GroupingSetsRow(SimpleNamespace):
|
|||
timed_requests: int | None
|
||||
|
||||
|
||||
class _EntityRollupRow(_GroupingSetsRow):
|
||||
class _GroupingSetsRow(_RollupMetricsRow):
|
||||
model: str | None
|
||||
model_group: str | None
|
||||
custom_llm_provider: str | None
|
||||
mcp_namespaced_tool_name: str | None
|
||||
endpoint: str | None
|
||||
group_level: int
|
||||
distinct_api_keys: int | None
|
||||
|
||||
|
||||
class _EntityRollupRow(_RollupMetricsRow):
|
||||
entity_id: str | None
|
||||
api_key_rolled: int
|
||||
|
||||
|
||||
def _reported_flat_cost(record: DailySpendRecord | _GroupingSetsRow) -> float:
|
||||
class _AggregatedQueryKwargs(TypedDict):
|
||||
table_name: ReadOnly[str]
|
||||
entity_id_field: ReadOnly[str]
|
||||
entity_id: ReadOnly[str | list[str] | None]
|
||||
start_date: ReadOnly[str]
|
||||
end_date: ReadOnly[str]
|
||||
model: ReadOnly[str | None]
|
||||
api_key: ReadOnly[str | list[str] | None]
|
||||
exclude_entity_ids: ReadOnly[list[str] | None]
|
||||
timezone_offset_minutes: ReadOnly[int | None]
|
||||
include_current_utc_day: ReadOnly[bool]
|
||||
|
||||
|
||||
_SqlQuery = tuple[str, list[str]]
|
||||
|
||||
|
||||
async def _query_raw_optional(
|
||||
prisma_client: PrismaClient, query: _SqlQuery | None
|
||||
) -> list[dict[str, object]] | None: # mutable-ok: prisma query_raw return shape
|
||||
if query is None:
|
||||
return None
|
||||
return await prisma_client.db.query_raw(query[0], *query[1])
|
||||
|
||||
|
||||
def _reported_flat_cost(record: DailySpendRecord | _RollupMetricsRow) -> float:
|
||||
"""Flat cost a daily row reports, which is zero unless PTU cost attribution is enabled.
|
||||
|
||||
Both read paths funnel through here: the paginated path reads the ``ptu_flat_cost``
|
||||
|
|
@ -699,71 +727,8 @@ def _ptu_flat_cost_select(table_name: str) -> str:
|
|||
return "0::float AS ptu_flat_cost"
|
||||
|
||||
|
||||
def _build_aggregated_sql_query(
|
||||
*,
|
||||
table_name: str,
|
||||
entity_id_field: str,
|
||||
entity_id: str | list[str] | None, # mutable-ok: filter union shared with the paginated path
|
||||
start_date: str,
|
||||
end_date: str,
|
||||
model: str | None,
|
||||
api_key: str | list[str] | None, # mutable-ok: filter union shared with the paginated path
|
||||
exclude_entity_ids: list[str] | None = None, # mutable-ok: filter union shared with the paginated path
|
||||
timezone_offset_minutes: int | None = None,
|
||||
include_current_utc_day: bool = False,
|
||||
) -> tuple[str, list[str]]: # mutable-ok: SQL text plus its ordered $N params
|
||||
"""Build a parameterized SQL GROUP BY query for aggregated daily activity.
|
||||
|
||||
Groups by (date, api_key, model, model_group, custom_llm_provider,
|
||||
mcp_namespaced_tool_name, endpoint) with SUMs on all metric columns.
|
||||
The entity_id column is intentionally omitted from GROUP BY to collapse
|
||||
rows across entities — this is where the biggest row reduction comes from.
|
||||
|
||||
Returns:
|
||||
Tuple of (sql_query, params_list) ready for prisma_client.db.query_raw().
|
||||
"""
|
||||
pg_table: Final = _PRISMA_TO_PG_TABLE.get(table_name)
|
||||
if pg_table is None:
|
||||
raise ValueError(f"Unknown table name: {table_name}")
|
||||
|
||||
adjusted_start, adjusted_end = _adjust_dates_for_timezone(
|
||||
start_date, end_date, timezone_offset_minutes, include_current_utc_day
|
||||
)
|
||||
|
||||
where_clause, sql_params = _build_aggregated_where_clause(
|
||||
entity_id_field=entity_id_field,
|
||||
entity_id=entity_id,
|
||||
adjusted_start=adjusted_start,
|
||||
adjusted_end=adjusted_end,
|
||||
model=model,
|
||||
api_key=api_key,
|
||||
exclude_entity_ids=exclude_entity_ids,
|
||||
)
|
||||
|
||||
# Postgres computes every rollup level the response needs — per-date
|
||||
# totals, per-(date, model), per-(date, model, api_key), per-provider,
|
||||
# etc. — in a single pass via GROUPING SETS. The GROUPING() bitmask
|
||||
# encodes which level a row belongs to so Python can dispatch rows
|
||||
# straight into their buckets without re-summing. The leaf grouping
|
||||
# is omitted on purpose: nothing in the response shape needs it once
|
||||
# all the rollups are present.
|
||||
#
|
||||
# TODO: drop the successful_requests/failed_requests aggregates (and the
|
||||
# total_successful_requests metadata they feed) once the admin UI reads SGR
|
||||
# only from LiteLLM_DailyGatewayRequests. The remaining spend, token and
|
||||
# api_requests rollups are still served from here.
|
||||
sql_query: Final = f"""
|
||||
SELECT
|
||||
date,
|
||||
api_key,
|
||||
model,
|
||||
COALESCE(NULLIF(model_group, ''), model) AS model_group,
|
||||
custom_llm_provider,
|
||||
mcp_namespaced_tool_name,
|
||||
endpoint,
|
||||
GROUPING(date, api_key, model, COALESCE(NULLIF(model_group, ''), model),
|
||||
custom_llm_provider, mcp_namespaced_tool_name,
|
||||
endpoint) AS group_level,
|
||||
def _rollup_metric_select(table_name: str) -> str:
|
||||
return f"""
|
||||
SUM(spend)::float AS spend,
|
||||
{_ptu_flat_cost_select(table_name)},
|
||||
SUM(prompt_tokens)::bigint AS prompt_tokens,
|
||||
|
|
@ -779,27 +744,113 @@ def _build_aggregated_sql_query(
|
|||
SUM(successful_requests)::bigint AS successful_requests,
|
||||
SUM(failed_requests)::bigint AS failed_requests,
|
||||
SUM(total_response_time_ms)::bigint AS total_response_time_ms,
|
||||
SUM(timed_requests)::bigint AS timed_requests
|
||||
SUM(timed_requests)::bigint AS timed_requests"""
|
||||
|
||||
|
||||
_MODEL_GROUP_EXPR: Final = "COALESCE(NULLIF(model_group, ''), model)"
|
||||
|
||||
|
||||
def _build_aggregated_sql_query(
|
||||
*,
|
||||
table_name: str,
|
||||
entity_id_field: str,
|
||||
entity_id: str | list[str] | None, # mutable-ok: filter union shared with the paginated path
|
||||
start_date: str,
|
||||
end_date: str,
|
||||
model: str | None,
|
||||
api_key: str | list[str] | None, # mutable-ok: filter union shared with the paginated path
|
||||
exclude_entity_ids: list[str] | None = None, # mutable-ok: filter union shared with the paginated path
|
||||
timezone_offset_minutes: int | None = None,
|
||||
include_current_utc_day: bool = False,
|
||||
) -> tuple[str, list[str]]: # mutable-ok: SQL text plus its ordered $N params
|
||||
"""Build the GROUPING SETS query for aggregated daily activity.
|
||||
|
||||
Returns:
|
||||
Tuple of (sql_query, params_list) ready for prisma_client.db.query_raw().
|
||||
"""
|
||||
pg_table: Final = _PRISMA_TO_PG_TABLE.get(table_name)
|
||||
if pg_table is None:
|
||||
raise ValueError(f"Unknown table name: {table_name}")
|
||||
|
||||
adjusted_start, adjusted_end = _adjust_dates_for_timezone(
|
||||
start_date, end_date, timezone_offset_minutes, include_current_utc_day
|
||||
)
|
||||
|
||||
where_clause, where_params = _build_aggregated_where_clause(
|
||||
entity_id_field=entity_id_field,
|
||||
entity_id=entity_id,
|
||||
adjusted_start=adjusted_start,
|
||||
adjusted_end=adjusted_end,
|
||||
model=model,
|
||||
api_key=api_key,
|
||||
exclude_entity_ids=exclude_entity_ids,
|
||||
)
|
||||
sentinel_param: Final = f"${len(where_params) + 1}"
|
||||
metric_select: Final = _rollup_metric_select(table_name)
|
||||
|
||||
# TODO: drop the successful_requests/failed_requests aggregates (and the
|
||||
# total_successful_requests metadata they feed) once the admin UI reads SGR
|
||||
# only from LiteLLM_DailyGatewayRequests. The remaining spend, token and
|
||||
# api_requests rollups are still served from here.
|
||||
sql_query: Final = f"""
|
||||
(SELECT
|
||||
date,
|
||||
NULL::text AS api_key,
|
||||
model,
|
||||
{_MODEL_GROUP_EXPR} AS model_group,
|
||||
custom_llm_provider,
|
||||
mcp_namespaced_tool_name,
|
||||
endpoint,
|
||||
(GROUPING(date) << 6) | {_API_KEY_ROLLED_UP_BIT}
|
||||
| GROUPING(model, {_MODEL_GROUP_EXPR},
|
||||
custom_llm_provider, mcp_namespaced_tool_name,
|
||||
endpoint) AS group_level,
|
||||
NULL::bigint AS distinct_api_keys,{metric_select}
|
||||
FROM "{pg_table}"
|
||||
WHERE {where_clause}
|
||||
GROUP BY GROUPING SETS (
|
||||
(date),
|
||||
(date, api_key),
|
||||
(date, model),
|
||||
(date, model, api_key),
|
||||
(date, COALESCE(NULLIF(model_group, ''), model)),
|
||||
(date, COALESCE(NULLIF(model_group, ''), model), api_key),
|
||||
(date, {_MODEL_GROUP_EXPR}),
|
||||
(date, custom_llm_provider),
|
||||
(date, custom_llm_provider, api_key),
|
||||
(date, mcp_namespaced_tool_name),
|
||||
(date, mcp_namespaced_tool_name, api_key),
|
||||
(date, endpoint),
|
||||
(date, endpoint, api_key),
|
||||
()
|
||||
))
|
||||
UNION ALL
|
||||
(WITH top_api_keys AS (
|
||||
SELECT api_key, COUNT(*) OVER () AS distinct_api_keys
|
||||
FROM "{pg_table}"
|
||||
WHERE {where_clause} AND api_key <> {sentinel_param}
|
||||
GROUP BY api_key
|
||||
ORDER BY SUM(spend) DESC, api_key
|
||||
LIMIT {USAGE_TOP_API_KEYS_LIMIT}
|
||||
)
|
||||
SELECT
|
||||
date,
|
||||
api_key,
|
||||
model,
|
||||
{_MODEL_GROUP_EXPR} AS model_group,
|
||||
custom_llm_provider,
|
||||
mcp_namespaced_tool_name,
|
||||
endpoint,
|
||||
GROUPING(date, api_key, model, {_MODEL_GROUP_EXPR},
|
||||
custom_llm_provider, mcp_namespaced_tool_name,
|
||||
endpoint) AS group_level,
|
||||
MAX(top_api_keys.distinct_api_keys) AS distinct_api_keys,{metric_select}
|
||||
FROM "{pg_table}" JOIN top_api_keys USING (api_key)
|
||||
WHERE {where_clause}
|
||||
GROUP BY GROUPING SETS (
|
||||
(date, api_key),
|
||||
(date, model, api_key),
|
||||
(date, {_MODEL_GROUP_EXPR}, api_key),
|
||||
(date, custom_llm_provider, api_key),
|
||||
(date, mcp_namespaced_tool_name, api_key),
|
||||
(date, endpoint, api_key)
|
||||
))
|
||||
"""
|
||||
|
||||
return sql_query, sql_params
|
||||
return sql_query, [*where_params, PTU_SENTINEL_API_KEY]
|
||||
|
||||
|
||||
def _build_entity_rollup_sql_query(
|
||||
|
|
@ -844,23 +895,7 @@ def _build_entity_rollup_sql_query(
|
|||
"{entity_id_field}" AS entity_id,
|
||||
date,
|
||||
api_key,
|
||||
GROUPING(api_key) AS api_key_rolled,
|
||||
SUM(spend)::float AS spend,
|
||||
{_ptu_flat_cost_select(table_name)},
|
||||
SUM(prompt_tokens)::bigint AS prompt_tokens,
|
||||
SUM(completion_tokens)::bigint AS completion_tokens,
|
||||
SUM(cache_read_input_tokens)::bigint AS cache_read_input_tokens,
|
||||
SUM(cache_creation_input_tokens)::bigint AS cache_creation_input_tokens,
|
||||
SUM(compression_saved_tokens)::bigint AS compression_saved_tokens,
|
||||
SUM(compression_savings_spend)::float AS compression_savings_spend,
|
||||
SUM(prompt_caching_savings_spend)::float AS prompt_caching_savings_spend,
|
||||
SUM(gateway_injected_caching_savings_spend)::float AS gateway_injected_caching_savings_spend,
|
||||
SUM(autorouter_savings_spend)::float AS autorouter_savings_spend,
|
||||
SUM(api_requests)::bigint AS api_requests,
|
||||
SUM(successful_requests)::bigint AS successful_requests,
|
||||
SUM(failed_requests)::bigint AS failed_requests,
|
||||
SUM(total_response_time_ms)::bigint AS total_response_time_ms,
|
||||
SUM(timed_requests)::bigint AS timed_requests
|
||||
GROUPING(api_key) AS api_key_rolled,{_rollup_metric_select(table_name)}
|
||||
FROM "{pg_table}"
|
||||
WHERE {where_clause}
|
||||
GROUP BY GROUPING SETS (
|
||||
|
|
@ -962,6 +997,7 @@ async def _aggregate_spend_records(
|
|||
# current grouping set's key), 0 when the column is part of the key.
|
||||
_GROUP_GRAND_TOTAL: Final = 127 # 0b1111111 — all rolled up
|
||||
_GROUP_DATE: Final = 63 # 0b0111111 — only date kept
|
||||
_API_KEY_ROLLED_UP_BIT: Final = 32 # 0b0100000
|
||||
_GROUP_DATE_API_KEY: Final = 31 # 0b0011111
|
||||
_GROUP_DATE_MODEL: Final = 47 # 0b0101111
|
||||
_GROUP_DATE_MODEL_API_KEY: Final = 15 # 0b0001111
|
||||
|
|
@ -975,7 +1011,7 @@ _GROUP_DATE_ENDPOINT: Final = 62 # 0b0111110
|
|||
_GROUP_DATE_ENDPOINT_API_KEY: Final = 30 # 0b0011110
|
||||
|
||||
|
||||
def _record_to_spend_metrics(record: _GroupingSetsRow) -> SpendMetrics:
|
||||
def _record_to_spend_metrics(record: _RollupMetricsRow) -> SpendMetrics:
|
||||
"""Build a SpendMetrics directly from one already-aggregated rollup row.
|
||||
|
||||
SUM() over zero rows is SQL NULL, so rollup rows (notably the grand-total
|
||||
|
|
@ -1329,10 +1365,6 @@ async def get_daily_activity_aggregated(
|
|||
) -> SpendAnalyticsPaginatedResponse:
|
||||
"""Aggregated variant that returns the full result set (no pagination).
|
||||
|
||||
Uses SQL GROUP BY to aggregate rows in the database rather than fetching
|
||||
all individual rows into Python. This collapses rows across entities
|
||||
(users/teams/orgs), reducing ~150k rows to ~2-3k grouped rows.
|
||||
|
||||
include_entity_breakdown runs a small companion rollup query and folds
|
||||
`breakdown.entities` onto the response, as entity-scoped views like Team Usage need.
|
||||
|
||||
|
|
@ -1351,7 +1383,7 @@ async def get_daily_activity_aggregated(
|
|||
)
|
||||
|
||||
try:
|
||||
sql_query, sql_params = _build_aggregated_sql_query(
|
||||
query_kwargs: Final = _AggregatedQueryKwargs(
|
||||
table_name=table_name,
|
||||
entity_id_field=entity_id_field,
|
||||
entity_id=entity_id,
|
||||
|
|
@ -1363,36 +1395,16 @@ async def get_daily_activity_aggregated(
|
|||
timezone_offset_minutes=timezone_offset_minutes,
|
||||
include_current_utc_day=include_current_utc_day,
|
||||
)
|
||||
sql_query, sql_params = _build_aggregated_sql_query(**query_kwargs)
|
||||
entity_query: Final = _build_entity_rollup_sql_query(**query_kwargs) if include_entity_breakdown else None
|
||||
|
||||
entity_query: Final = (
|
||||
_build_entity_rollup_sql_query(
|
||||
table_name=table_name,
|
||||
entity_id_field=entity_id_field,
|
||||
entity_id=entity_id,
|
||||
start_date=start_date,
|
||||
end_date=end_date,
|
||||
model=model,
|
||||
api_key=api_key,
|
||||
exclude_entity_ids=exclude_entity_ids,
|
||||
timezone_offset_minutes=timezone_offset_minutes,
|
||||
include_current_utc_day=include_current_utc_day,
|
||||
)
|
||||
if include_entity_breakdown
|
||||
else None
|
||||
raw_rows, raw_entity_rows = await asyncio.gather(
|
||||
prisma_client.db.query_raw(sql_query, *sql_params),
|
||||
_query_raw_optional(prisma_client, entity_query),
|
||||
)
|
||||
|
||||
# Execute the GROUPING SETS query (one row per rollup level), alongside
|
||||
# the per-entity companion rollup when the caller wants entities.
|
||||
raw_rows, raw_entity_rows = (
|
||||
await asyncio.gather(
|
||||
prisma_client.db.query_raw(sql_query, *sql_params),
|
||||
prisma_client.db.query_raw(entity_query[0], *entity_query[1]),
|
||||
)
|
||||
if entity_query is not None
|
||||
else (await prisma_client.db.query_raw(sql_query, *sql_params), None)
|
||||
)
|
||||
|
||||
records: Final = [_GroupingSetsRow(**row) for row in (raw_rows or [])]
|
||||
records: Final = [_GroupingSetsRow(**row) for row in (raw_rows or ())]
|
||||
total_api_keys: Final = next((r.distinct_api_keys for r in records if r.distinct_api_keys is not None), 0)
|
||||
|
||||
# The grouping-sets dispatcher places each row directly in its bucket
|
||||
# using the row's GROUPING() bitmask. No Python-side summing needed.
|
||||
|
|
@ -1446,6 +1458,8 @@ async def get_daily_activity_aggregated(
|
|||
page=1,
|
||||
total_pages=1,
|
||||
has_more=False,
|
||||
api_key_limit=USAGE_TOP_API_KEYS_LIMIT,
|
||||
total_api_keys=total_api_keys,
|
||||
),
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -510,6 +510,16 @@ def _get_user_in_team(team_table: LiteLLM_TeamTableCachedObj, user_id: str | Non
|
|||
return None
|
||||
|
||||
|
||||
def _get_caller_team_role(
|
||||
team_table: LiteLLM_TeamTableCachedObj,
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
) -> Literal["admin", "user"] | None:
|
||||
if user_api_key_dict.is_team_service_account and user_api_key_dict.team_id == team_table.team_id:
|
||||
return "user"
|
||||
member: Final = _get_user_in_team(team_table=team_table, user_id=user_api_key_dict.user_id)
|
||||
return None if member is None else member.role
|
||||
|
||||
|
||||
def _calculate_key_rotation_time(rotation_interval: str) -> datetime:
|
||||
"""
|
||||
Helper function to calculate the next rotation time for a key based on the rotation interval.
|
||||
|
|
@ -604,7 +614,7 @@ def _team_key_operation_team_member_check(
|
|||
detail=f"User={assigned_user_id} not assigned to team={team_table.team_id}",
|
||||
)
|
||||
|
||||
team_member_object: Final = _get_user_in_team(team_table=team_table, user_id=user_api_key_dict.user_id)
|
||||
caller_team_role: Final = _get_caller_team_role(team_table=team_table, user_api_key_dict=user_api_key_dict)
|
||||
|
||||
is_admin: Final = (
|
||||
user_api_key_dict.user_role is not None and user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value
|
||||
|
|
@ -612,22 +622,22 @@ def _team_key_operation_team_member_check(
|
|||
|
||||
if is_admin:
|
||||
return True
|
||||
elif team_member_object is None:
|
||||
elif caller_team_role is None:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"User={user_api_key_dict.user_id} not assigned to team={team_table.team_id}",
|
||||
)
|
||||
elif (
|
||||
"allowed_team_member_roles" in team_key_generation
|
||||
and team_member_object.role not in team_key_generation["allowed_team_member_roles"]
|
||||
and caller_team_role not in team_key_generation["allowed_team_member_roles"]
|
||||
):
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"Team member role {team_member_object.role} not in allowed_team_member_roles={team_key_generation['allowed_team_member_roles']}",
|
||||
detail=f"Team member role {caller_team_role} not in allowed_team_member_roles={team_key_generation['allowed_team_member_roles']}",
|
||||
)
|
||||
|
||||
TeamMemberPermissionChecks.does_team_member_have_permissions_for_endpoint(
|
||||
team_member_object=team_member_object,
|
||||
team_member_role=caller_team_role,
|
||||
team_table=team_table,
|
||||
route=route,
|
||||
)
|
||||
|
|
@ -748,6 +758,12 @@ def key_generation_check(
|
|||
Check if admin has restricted key creation to certain roles for teams or individuals
|
||||
"""
|
||||
|
||||
if user_api_key_dict.is_team_service_account and data.team_id != user_api_key_dict.team_id:
|
||||
raise HTTPException(
|
||||
status_code=403,
|
||||
detail=f"Service account keys can only create keys for their own team. team_id={user_api_key_dict.team_id}",
|
||||
)
|
||||
|
||||
## check if key is for team or individual
|
||||
is_team_key: Final = _is_team_key(data=data)
|
||||
_is_admin: Final = (
|
||||
|
|
@ -2233,6 +2249,14 @@ async def generate_service_account_key_fn(
|
|||
prisma_client=prisma_client,
|
||||
)
|
||||
|
||||
if data.metadata is None or data.metadata.get("service_account_id") is None:
|
||||
service_account_id: Final = data.key_alias or str(uuid.uuid4())
|
||||
stamped_metadata: Final = { # mutable-ok: GenerateKeyRequest.metadata is a plain dict field
|
||||
**(data.metadata or MappingProxyType({})),
|
||||
"service_account_id": service_account_id,
|
||||
}
|
||||
data.metadata = stamped_metadata # rebind-ok: the request carries the stamp so it persists on the key
|
||||
|
||||
verbose_proxy_logger.debug("entered /key/generate")
|
||||
|
||||
custom_key_generate_hook: Final[Callable[..., Awaitable[Mapping[str, object]]] | None] = _custom_key_generate_hook(
|
||||
|
|
@ -3891,8 +3915,10 @@ async def validate_key_team_change(
|
|||
detail=f"Key={key.token} has a rpm_limit={key.rpm_limit} which is greater than the team's rpm_limit={team.rpm_limit}.",
|
||||
)
|
||||
|
||||
team_table: Final = cast(LiteLLM_TeamTableCachedObj, team)
|
||||
|
||||
# Check if the key's user_id is a member of the team
|
||||
member_object: Final = _get_user_in_team(team_table=cast(LiteLLM_TeamTableCachedObj, team), user_id=key.user_id)
|
||||
member_object: Final = _get_user_in_team(team_table=team_table, user_id=key.user_id)
|
||||
if key.user_id is not None:
|
||||
if not member_object:
|
||||
raise HTTPException(
|
||||
|
|
@ -3908,8 +3934,8 @@ async def validate_key_team_change(
|
|||
team_obj=team,
|
||||
)
|
||||
or TeamMemberPermissionChecks.does_team_member_have_permissions_for_endpoint(
|
||||
team_member_object=member_object,
|
||||
team_table=cast(LiteLLM_TeamTableCachedObj, team),
|
||||
team_member_role=None if member_object is None else member_object.role,
|
||||
team_table=team_table,
|
||||
route=KeyManagementRoutes.KEY_UPDATE.value,
|
||||
)
|
||||
):
|
||||
|
|
|
|||
|
|
@ -47,7 +47,7 @@ except ImportError:
|
|||
import litellm
|
||||
from litellm._logging import verbose_logger, verbose_proxy_logger
|
||||
from litellm._uuid import uuid
|
||||
from litellm.constants import LITELLM_PROXY_ADMIN_NAME
|
||||
from litellm.constants import LITELLM_PROXY_ADMIN_NAME, MCP_GATEWAY_SESSION_ID_PREFIX_LENGTH
|
||||
from litellm.proxy._experimental.mcp_server.utils import (
|
||||
LITELLM_MCP_SERVER_DESCRIPTION,
|
||||
LITELLM_MCP_SERVER_NAME,
|
||||
|
|
@ -145,6 +145,7 @@ if MCP_AVAILABLE:
|
|||
get_user_env_vars,
|
||||
get_user_env_vars_bulk,
|
||||
get_user_oauth_credential,
|
||||
list_server_user_credentials,
|
||||
list_user_oauth_credentials,
|
||||
mcp_oauth_token_identity,
|
||||
merge_user_env_vars,
|
||||
|
|
@ -180,6 +181,7 @@ if MCP_AVAILABLE:
|
|||
MCPApprovalStatus,
|
||||
MCPOAuthUserCredentialRequest,
|
||||
MCPOAuthUserCredentialStatus,
|
||||
MCPServerUserCredentialListItem,
|
||||
MCPSubmissionsSummary,
|
||||
MCPTransport,
|
||||
MCPUserCredentialListItem,
|
||||
|
|
@ -221,6 +223,7 @@ if MCP_AVAILABLE:
|
|||
MCPAuth,
|
||||
MCPCredentials,
|
||||
MCPGatewaySessionsResponse,
|
||||
MCPGatewaySessionsTerminateResponse,
|
||||
normalize_upstream_header_name,
|
||||
)
|
||||
from litellm.types.mcp_server.mcp_server_manager import MCPServer
|
||||
|
|
@ -662,6 +665,31 @@ if MCP_AVAILABLE:
|
|||
"""
|
||||
return user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN
|
||||
|
||||
def _resolve_credential_target_user_id(user_api_key_dict: UserAPIKeyAuth, requested_user_id: str | None) -> str:
|
||||
"""The user whose stored MCP credential a request acts on.
|
||||
|
||||
Defaults to the caller. Naming another user is a revocation and needs
|
||||
``PROXY_ADMIN``; a read-only admin or a regular user gets 403.
|
||||
"""
|
||||
caller_user_id: Final = user_api_key_dict.user_id or ""
|
||||
if requested_user_id is not None and requested_user_id != caller_user_id:
|
||||
if not _user_is_full_admin(user_api_key_dict):
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_403_FORBIDDEN,
|
||||
detail={ # mutable-ok: FastAPI HTTPException detail requires a plain dict
|
||||
"error": "Proxy admin access required to revoke another user's MCP credential.",
|
||||
},
|
||||
)
|
||||
return requested_user_id
|
||||
if not caller_user_id:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_400_BAD_REQUEST,
|
||||
detail={
|
||||
"error": "User ID not found in token"
|
||||
}, # mutable-ok: FastAPI HTTPException detail requires a plain dict
|
||||
)
|
||||
return caller_user_id
|
||||
|
||||
def _is_restricted_virtual_key_request(user_api_key_dict: UserAPIKeyAuth) -> bool:
|
||||
"""Best-effort detection for route-restricted virtual keys.
|
||||
|
||||
|
|
@ -1373,6 +1401,41 @@ if MCP_AVAILABLE:
|
|||
|
||||
return get_mcp_gateway_sessions_report()
|
||||
|
||||
@router.delete(
|
||||
"/sessions",
|
||||
description=(
|
||||
"Force-close live stateful MCP gateway sessions on this proxy worker, selected by session id prefix "
|
||||
"and/or by the LiteLLM user that opened them (proxy admin only)."
|
||||
),
|
||||
dependencies=(Depends(user_api_key_auth),),
|
||||
response_model=MCPGatewaySessionsTerminateResponse,
|
||||
)
|
||||
@management_endpoint_wrapper
|
||||
async def delete_mcp_gateway_sessions(
|
||||
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
|
||||
session_id_prefix: Annotated[str | None, Query(min_length=MCP_GATEWAY_SESSION_ID_PREFIX_LENGTH)] = None,
|
||||
user_id: Annotated[str | None, Query(min_length=1)] = None,
|
||||
) -> MCPGatewaySessionsTerminateResponse:
|
||||
if not _user_is_full_admin(user_api_key_dict):
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_403_FORBIDDEN,
|
||||
detail={ # mutable-ok: FastAPI HTTPException detail requires a plain dict
|
||||
"error": "Proxy admin access required to terminate MCP gateway sessions.",
|
||||
},
|
||||
)
|
||||
if session_id_prefix is None and user_id is None:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_400_BAD_REQUEST,
|
||||
detail={ # mutable-ok: FastAPI HTTPException detail requires a plain dict
|
||||
"error": "Provide session_id_prefix and/or user_id to select the sessions to terminate.",
|
||||
},
|
||||
)
|
||||
from litellm.proxy._experimental.mcp_server.server import (
|
||||
terminate_mcp_gateway_sessions,
|
||||
)
|
||||
|
||||
return await terminate_mcp_gateway_sessions(session_id_prefix=session_id_prefix, user_id=user_id)
|
||||
|
||||
@router.get(
|
||||
"/server/submissions",
|
||||
description="Returns all MCP servers submitted by non-admin users (admin review queue). Mirrors GET /guardrails/submissions.",
|
||||
|
|
@ -2254,14 +2317,17 @@ if MCP_AVAILABLE:
|
|||
_invalidate_byok_cred_cache,
|
||||
)
|
||||
|
||||
_invalidate_byok_cred_cache(user_id, server_id)
|
||||
await _invalidate_byok_cred_cache(user_id, server_id)
|
||||
return MCPUserCredentialResponse(server_id=server_id, has_credential=True)
|
||||
# save=False: credential not persisted
|
||||
return MCPUserCredentialResponse(server_id=server_id, has_credential=False)
|
||||
|
||||
@router.delete(
|
||||
"/server/{server_id}/user-credential",
|
||||
description="Delete the calling user's stored API key for a BYOK MCP server",
|
||||
description=(
|
||||
"Delete the calling user's stored API key for a BYOK MCP server. "
|
||||
"A proxy admin may pass user_id to revoke another user's stored key."
|
||||
),
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
response_model=MCPUserCredentialResponse,
|
||||
)
|
||||
|
|
@ -2269,24 +2335,20 @@ if MCP_AVAILABLE:
|
|||
async def delete_mcp_user_credential(
|
||||
server_id: str,
|
||||
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
|
||||
user_id: Annotated[str | None, Query(min_length=1)] = None,
|
||||
):
|
||||
"""Remove the calling user's BYOK credential."""
|
||||
"""Remove the target user's BYOK credential (the caller unless an admin names another user)."""
|
||||
prisma_client: Final = get_prisma_client_or_throw("Database not connected. Connect a database to your proxy")
|
||||
user_id: Final = user_api_key_dict.user_id or ""
|
||||
if not user_id:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_400_BAD_REQUEST,
|
||||
detail={"error": "User ID not found in token"},
|
||||
)
|
||||
target_user_id: Final = _resolve_credential_target_user_id(user_api_key_dict, user_id)
|
||||
try:
|
||||
await delete_user_credential(prisma_client, user_id, server_id)
|
||||
await delete_user_credential(prisma_client, target_user_id, server_id)
|
||||
except RecordNotFoundError:
|
||||
pass # Already deleted or didn't exist
|
||||
from litellm.proxy._experimental.mcp_server.server import (
|
||||
_invalidate_byok_cred_cache,
|
||||
)
|
||||
|
||||
_invalidate_byok_cred_cache(user_id, server_id)
|
||||
await _invalidate_byok_cred_cache(target_user_id, server_id)
|
||||
return MCPUserCredentialResponse(server_id=server_id, has_credential=False)
|
||||
|
||||
# ── OAuth2 user-credential endpoints ──────────────────────────────────────
|
||||
|
|
@ -2362,7 +2424,10 @@ if MCP_AVAILABLE:
|
|||
|
||||
@router.delete(
|
||||
"/server/{server_id}/oauth-user-credential",
|
||||
description="Revoke the calling user's stored OAuth2 token for an MCP server",
|
||||
description=(
|
||||
"Revoke the calling user's stored OAuth2 token for an MCP server. "
|
||||
"A proxy admin may pass user_id to revoke another user's stored token."
|
||||
),
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
response_model=MCPOAuthUserCredentialStatus,
|
||||
)
|
||||
|
|
@ -2370,29 +2435,25 @@ if MCP_AVAILABLE:
|
|||
async def delete_mcp_oauth_user_credential(
|
||||
server_id: str,
|
||||
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
|
||||
user_id: Annotated[str | None, Query(min_length=1)] = None,
|
||||
):
|
||||
"""Revoke/delete the user's OAuth2 credential."""
|
||||
"""Revoke the target user's OAuth2 credential (the caller unless an admin names another user)."""
|
||||
prisma_client: Final = get_prisma_client_or_throw("Database not connected. Connect a database to your proxy")
|
||||
user_id: Final = user_api_key_dict.user_id or ""
|
||||
if not user_id:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_400_BAD_REQUEST,
|
||||
detail={"error": "User ID not found in token"},
|
||||
)
|
||||
target_user_id: Final = _resolve_credential_target_user_id(user_api_key_dict, user_id)
|
||||
# Only delete if the stored credential is actually an OAuth2 token.
|
||||
# This prevents accidentally deleting a BYOK credential if one exists
|
||||
# for the same (user_id, server_id) pair.
|
||||
cred_to_delete: Final = await get_user_oauth_credential(prisma_client, user_id, server_id)
|
||||
cred_to_delete: Final = await get_user_oauth_credential(prisma_client, target_user_id, server_id)
|
||||
if cred_to_delete is not None:
|
||||
try:
|
||||
await delete_user_credential(prisma_client, user_id, server_id)
|
||||
await delete_user_credential(prisma_client, target_user_id, server_id)
|
||||
except RecordNotFoundError:
|
||||
pass # Already gone — treat as a successful delete
|
||||
from litellm.proxy._experimental.mcp_server.mcp_server_manager import ( # noqa: PLC0415
|
||||
global_mcp_server_manager,
|
||||
)
|
||||
|
||||
await global_mcp_server_manager.invalidate_user_oauth_token_cache(user_id, server_id)
|
||||
await global_mcp_server_manager.invalidate_user_oauth_token_cache(target_user_id, server_id)
|
||||
return MCPOAuthUserCredentialStatus(
|
||||
server_id=server_id,
|
||||
has_credential=False,
|
||||
|
|
@ -2481,6 +2542,30 @@ if MCP_AVAILABLE:
|
|||
)
|
||||
return items
|
||||
|
||||
@router.get(
|
||||
"/server/{server_id}/user-credentials",
|
||||
description="List every user's stored BYOK or OAuth2 credential for an MCP server (admin only, no secrets)",
|
||||
dependencies=(Depends(user_api_key_auth),),
|
||||
response_model=list[MCPServerUserCredentialListItem],
|
||||
)
|
||||
@management_endpoint_wrapper
|
||||
async def list_mcp_server_user_credentials(
|
||||
server_id: str,
|
||||
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
|
||||
) -> tuple[MCPServerUserCredentialListItem, ...]:
|
||||
if user_api_key_dict.user_role not in (
|
||||
LitellmUserRoles.PROXY_ADMIN,
|
||||
LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY,
|
||||
):
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_403_FORBIDDEN,
|
||||
detail={ # mutable-ok: FastAPI HTTPException detail requires a plain dict
|
||||
"error": "Admin access required to view MCP server user credentials.",
|
||||
},
|
||||
)
|
||||
prisma_client: Final = get_prisma_client_or_throw("Database not connected. Connect a database to your proxy")
|
||||
return await list_server_user_credentials(prisma_client, server_id)
|
||||
|
||||
# ── Per-user MCP env var endpoints ────────────────────────────────────────
|
||||
|
||||
async def _authorize_and_fetch_mcp_server(
|
||||
|
|
|
|||
|
|
@ -2107,7 +2107,7 @@ def _handle_multi_valued_attribute_update(path: str, op_type: str, value: object
|
|||
except ValidationError:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail={"error": f"Invalid value for {base}: expected a list of objects with a 'value' sub-attribute"},
|
||||
detail={"error": f"Invalid value for {base}: expected a list of objects or strings"},
|
||||
)
|
||||
|
||||
dumped: Final = [attr.model_dump(exclude_none=True) for attr in attrs]
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
from typing import Final
|
||||
from typing import Final, Literal
|
||||
|
||||
from litellm.proxy._types import (
|
||||
KeyManagementRoutes,
|
||||
|
|
@ -6,7 +6,6 @@ from litellm.proxy._types import (
|
|||
LiteLLM_VerificationToken,
|
||||
LiteLLMRoutes,
|
||||
LitellmUserRoles,
|
||||
Member,
|
||||
ProxyErrorTypes,
|
||||
ProxyException,
|
||||
UserAPIKeyAuth,
|
||||
|
|
@ -27,7 +26,6 @@ DEFAULT_TEAM_MEMBER_PERMISSIONS: Final = BASELINE_TEAM_MEMBER_PERMISSIONS
|
|||
class TeamMemberPermissionChecks:
|
||||
@staticmethod
|
||||
def get_permissions_for_team_member(
|
||||
team_member_object: Member,
|
||||
team_table: LiteLLM_TeamTableCachedObj,
|
||||
) -> list[KeyManagementRoutes]:
|
||||
"""
|
||||
|
|
@ -67,7 +65,7 @@ class TeamMemberPermissionChecks:
|
|||
Main handler for checking if a team member can update a key
|
||||
"""
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
_get_user_in_team,
|
||||
_get_caller_team_role,
|
||||
)
|
||||
|
||||
# 1. Don't execute these checks if the user role is proxy admin
|
||||
|
|
@ -87,12 +85,11 @@ class TeamMemberPermissionChecks:
|
|||
check_db_only=True,
|
||||
)
|
||||
|
||||
# 4. Extract `Member` object from `team_table`
|
||||
key_assigned_user_in_team: Final = _get_user_in_team(team_table=team_table, user_id=user_api_key_dict.user_id)
|
||||
caller_team_role: Final = _get_caller_team_role(team_table=team_table, user_api_key_dict=user_api_key_dict)
|
||||
|
||||
# 5. Check if the team member has permissions for the endpoint
|
||||
# 4. Check if the team member has permissions for the endpoint
|
||||
has_permission: Final = TeamMemberPermissionChecks.does_team_member_have_permissions_for_endpoint(
|
||||
team_member_object=key_assigned_user_in_team,
|
||||
team_member_role=caller_team_role,
|
||||
team_table=team_table,
|
||||
route=route,
|
||||
)
|
||||
|
|
@ -106,7 +103,7 @@ class TeamMemberPermissionChecks:
|
|||
|
||||
@staticmethod
|
||||
def does_team_member_have_permissions_for_endpoint(
|
||||
team_member_object: Member | None,
|
||||
team_member_role: Literal["admin", "user"] | None,
|
||||
team_table: LiteLLM_TeamTableCachedObj,
|
||||
route: str,
|
||||
) -> bool | None:
|
||||
|
|
@ -116,13 +113,12 @@ class TeamMemberPermissionChecks:
|
|||
|
||||
# permission checks only run for non-admin users
|
||||
# Non-Admin user trying to access information about a team's key
|
||||
if team_member_object is None:
|
||||
if team_member_role is None:
|
||||
return False
|
||||
if team_member_object.role == "admin":
|
||||
if team_member_role == "admin":
|
||||
return True
|
||||
|
||||
_team_member_permissions: Final = TeamMemberPermissionChecks.get_permissions_for_team_member(
|
||||
team_member_object=team_member_object,
|
||||
team_table=team_table,
|
||||
)
|
||||
team_member_permissions = TeamMemberPermissionChecks._get_list_of_route_enum_as_str(_team_member_permissions)
|
||||
|
|
@ -156,7 +152,7 @@ class TeamMemberPermissionChecks:
|
|||
from fastapi import HTTPException
|
||||
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
_get_user_in_team,
|
||||
_get_caller_team_role,
|
||||
)
|
||||
|
||||
# No-op when the request does not assign any access groups.
|
||||
|
|
@ -177,20 +173,19 @@ class TeamMemberPermissionChecks:
|
|||
),
|
||||
)
|
||||
|
||||
team_member_object: Final = _get_user_in_team(team_table=team_table, user_id=user_api_key_dict.user_id)
|
||||
caller_team_role: Final = _get_caller_team_role(team_table=team_table, user_api_key_dict=user_api_key_dict)
|
||||
|
||||
# Team admins always bypass (consistent with other member-permission checks).
|
||||
if team_member_object is not None and team_member_object.role == "admin":
|
||||
if caller_team_role == "admin":
|
||||
return
|
||||
|
||||
permissions: Final = (
|
||||
TeamMemberPermissionChecks._get_list_of_route_enum_as_str(
|
||||
TeamMemberPermissionChecks.get_permissions_for_team_member(
|
||||
team_member_object=team_member_object,
|
||||
team_table=team_table,
|
||||
)
|
||||
)
|
||||
if team_member_object is not None
|
||||
if caller_team_role is not None
|
||||
else []
|
||||
)
|
||||
|
||||
|
|
@ -214,7 +209,7 @@ class TeamMemberPermissionChecks:
|
|||
Returns True if the user belongs to the team that the key is assigned to
|
||||
"""
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
_get_user_in_team,
|
||||
_get_caller_team_role,
|
||||
)
|
||||
from litellm.proxy.proxy_server import prisma_client, user_api_key_cache
|
||||
|
||||
|
|
@ -228,9 +223,8 @@ class TeamMemberPermissionChecks:
|
|||
check_db_only=True,
|
||||
)
|
||||
|
||||
# 4. Extract `Member` object from `team_table`
|
||||
team_member_object: Final = _get_user_in_team(team_table=team_table, user_id=user_api_key_dict.user_id)
|
||||
return team_member_object is not None
|
||||
caller_team_role: Final = _get_caller_team_role(team_table=team_table, user_api_key_dict=user_api_key_dict)
|
||||
return caller_team_role is not None
|
||||
|
||||
@staticmethod
|
||||
def get_all_available_team_member_permissions() -> list[str]:
|
||||
|
|
|
|||
|
|
@ -1411,6 +1411,8 @@ def run_server(
|
|||
# DO NOT DELETE - enables global variables to work across files
|
||||
from litellm.proxy.proxy_server import app
|
||||
|
||||
os.environ["NUM_WORKERS"] = str(num_workers)
|
||||
|
||||
# Auto-create PROMETHEUS_MULTIPROC_DIR for multi-worker setups
|
||||
prometheus_multiproc_dir: Final = ProxyInitializationHelpers._maybe_setup_prometheus_multiproc_dir(
|
||||
num_workers=num_workers,
|
||||
|
|
|
|||
|
|
@ -145,6 +145,7 @@ from litellm.router_utils.auto_router_tuning_baseline import (
|
|||
snapshot_tuning_baselines,
|
||||
tuning_limit_violation,
|
||||
)
|
||||
from litellm.router_utils.routing_groups import parse_routing_groups
|
||||
from litellm.types.caching import RedisPipelineIncrementOperation
|
||||
from litellm.types.utils import (
|
||||
ModelResponse,
|
||||
|
|
@ -308,6 +309,7 @@ from litellm.litellm_core_utils.sensitive_data_masker import (
|
|||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
|
||||
from litellm.llms.openai_like.model_info import MODEL_INFO_REFRESH_SECONDS
|
||||
from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
|
||||
from litellm.proxy._experimental.mcp_server.byok_credential_cache import byok_credential_cache
|
||||
from litellm.proxy._lazy_features import attach_lazy_features, reserve_lazy_slot
|
||||
from litellm.proxy._types import *
|
||||
from litellm.proxy.analytics_endpoints.analytics_endpoints import (
|
||||
|
|
@ -329,6 +331,12 @@ from litellm.proxy.auth.fallback_budget import router_fallback_budget_check
|
|||
from litellm.proxy.auth.fallback_model_access import router_fallback_access_check
|
||||
from litellm.proxy.auth.handle_jwt import JWTHandler
|
||||
from litellm.proxy.auth.litellm_license import AUTO_ROUTER_LICENSE_REMEDY, LicenseCheck
|
||||
from litellm.proxy.auth.login_throttle import (
|
||||
LoginThrottle,
|
||||
declared_proxy_ranges,
|
||||
warn_login_counters_are_per_worker,
|
||||
warn_source_login_limit_is_off,
|
||||
)
|
||||
from litellm.proxy.auth.model_checks import (
|
||||
expand_wildcard_deployments_for_model_info,
|
||||
get_all_fallbacks,
|
||||
|
|
@ -780,6 +788,7 @@ from litellm.types.router import (
|
|||
ClassifierPlugin,
|
||||
DeploymentTypedDict,
|
||||
RouterGeneralSettings,
|
||||
RoutingGroup,
|
||||
RoutingPlugin,
|
||||
SearchToolTypedDict,
|
||||
updateDeployment,
|
||||
|
|
@ -825,6 +834,7 @@ from fastapi.openapi.docs import get_swagger_ui_html
|
|||
from fastapi.openapi.utils import get_openapi
|
||||
from fastapi.responses import (
|
||||
FileResponse,
|
||||
HTMLResponse,
|
||||
JSONResponse,
|
||||
ORJSONResponse,
|
||||
RedirectResponse,
|
||||
|
|
@ -6048,6 +6058,12 @@ class ProxyConfig:
|
|||
general_settings = config.get("general_settings", {})
|
||||
if general_settings is None:
|
||||
general_settings = {}
|
||||
|
||||
if os.getenv("NUM_WORKERS", "1") != "1" and redis_usage_cache is None:
|
||||
warn_login_counters_are_per_worker(os.getenv("NUM_WORKERS", "1"))
|
||||
if declared_proxy_ranges(general_settings) is None:
|
||||
warn_source_login_limit_is_off()
|
||||
|
||||
_bg_hc_model_groups: Final = parse_background_health_check_model_groups(general_settings)
|
||||
_enable_hc_routing = False
|
||||
_hc_staleness = None
|
||||
|
|
@ -7097,7 +7113,21 @@ class ProxyConfig:
|
|||
self.router_settings.apply_db_row("router_settings", db_values)
|
||||
combined_router_settings: Final = self.router_settings.resolved()
|
||||
if combined_router_settings:
|
||||
llm_router.update_settings(**combined_router_settings)
|
||||
self._apply_router_settings(llm_router, combined_router_settings)
|
||||
|
||||
@staticmethod
|
||||
def _apply_router_settings(llm_router: Router, router_settings: Mapping[str, object]) -> None:
|
||||
llm_router.update_settings(**{k: v for k, v in router_settings.items() if k != "routing_groups"})
|
||||
if "routing_groups" not in router_settings:
|
||||
return
|
||||
try:
|
||||
llm_router.update_settings(routing_groups=router_settings["routing_groups"])
|
||||
except (TypeError, ValueError) as invalid_groups:
|
||||
verbose_proxy_logger.error(
|
||||
"Ignoring invalid router_settings.routing_groups from config/DB, all other router settings still "
|
||||
"apply. Fix the routing groups in the Admin UI to load them: %s",
|
||||
invalid_groups,
|
||||
)
|
||||
|
||||
async def _reschedule_spend_log_cleanup_job(self):
|
||||
"""
|
||||
|
|
@ -7521,7 +7551,7 @@ class ProxyConfig:
|
|||
subscriber: Final = AuthCacheInvalidationSubscriber(
|
||||
redis_cache=redis_cache,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
additional_in_memory_caches=(spend_counter_cache.in_memory_cache,),
|
||||
additional_in_memory_caches=(spend_counter_cache.in_memory_cache, byok_credential_cache),
|
||||
)
|
||||
self.auth_cache_invalidation_subscriber = subscriber
|
||||
subscriber.start()
|
||||
|
|
@ -15885,8 +15915,6 @@ async def fallback_login(request: Request):
|
|||
else:
|
||||
redirect_url += "/sso/callback"
|
||||
|
||||
from fastapi.responses import HTMLResponse
|
||||
|
||||
hide_default_credentials_hint: Final = should_hide_default_credentials_hint(general_settings)
|
||||
return HTMLResponse(
|
||||
content=build_ui_login_form(
|
||||
|
|
@ -15908,13 +15936,27 @@ async def login(request: Request):
|
|||
password: Final = str(form.get("password"))
|
||||
|
||||
# Authenticate user and get login result
|
||||
login_result: Final = await authenticate_user(
|
||||
username=username,
|
||||
password=password,
|
||||
master_key=master_key,
|
||||
prisma_client=prisma_client,
|
||||
general_settings=general_settings,
|
||||
)
|
||||
try:
|
||||
login_result: Final = await authenticate_user(
|
||||
username=username,
|
||||
password=password,
|
||||
master_key=master_key,
|
||||
prisma_client=prisma_client,
|
||||
throttle=LoginThrottle.from_request(request, general_settings, redis_usage_cache),
|
||||
general_settings=general_settings,
|
||||
)
|
||||
except ProxyException as exc:
|
||||
if int(exc.code) != status.HTTP_429_TOO_MANY_REQUESTS:
|
||||
raise
|
||||
retry_after: Final = exc.headers.get("Retry-After", "30")
|
||||
return HTMLResponse(
|
||||
content=(
|
||||
"<html><body><h1>Too many sign-in attempts</h1>"
|
||||
f"<p>Try again in about {retry_after} seconds</p></body></html>"
|
||||
),
|
||||
status_code=status.HTTP_429_TOO_MANY_REQUESTS,
|
||||
headers=exc.headers,
|
||||
)
|
||||
|
||||
# Create UI token object
|
||||
returned_ui_token_object: Final = create_ui_token_object(
|
||||
|
|
@ -15993,6 +16035,7 @@ async def login_v2(request: Request):
|
|||
password=password,
|
||||
master_key=master_key,
|
||||
prisma_client=prisma_client,
|
||||
throttle=LoginThrottle.from_request(request, general_settings, redis_usage_cache),
|
||||
general_settings=general_settings,
|
||||
)
|
||||
|
||||
|
|
@ -16064,6 +16107,7 @@ async def login_v3(request: Request):
|
|||
password=password,
|
||||
master_key=master_key,
|
||||
prisma_client=prisma_client,
|
||||
throttle=LoginThrottle.from_request(request, general_settings, redis_usage_cache),
|
||||
general_settings=general_settings,
|
||||
)
|
||||
|
||||
|
|
@ -16940,6 +16984,12 @@ async def update_config(
|
|||
)
|
||||
},
|
||||
)
|
||||
try:
|
||||
parse_routing_groups(
|
||||
TypeAdapter(list[RoutingGroup] | None).validate_python(raw_router_settings.get("routing_groups"))
|
||||
)
|
||||
except (ValidationError, ValueError) as invalid_groups:
|
||||
raise HTTPException(status_code=400, detail={"error": str(invalid_groups)})
|
||||
|
||||
if prisma_client is None:
|
||||
raise Exception("No DB Connected")
|
||||
|
|
|
|||
|
|
@ -481,6 +481,7 @@ async def _arealtime(
|
|||
aws_sts_endpoint: Final = kwargs.get("aws_sts_endpoint")
|
||||
aws_bedrock_runtime_endpoint: Final = kwargs.get("aws_bedrock_runtime_endpoint")
|
||||
aws_external_id: Final = kwargs.get("aws_external_id")
|
||||
aws_session_tags: Final = kwargs.get("aws_session_tags")
|
||||
|
||||
await bedrock_realtime.async_realtime(
|
||||
model=model,
|
||||
|
|
@ -500,6 +501,7 @@ async def _arealtime(
|
|||
aws_sts_endpoint=aws_sts_endpoint,
|
||||
aws_bedrock_runtime_endpoint=aws_bedrock_runtime_endpoint,
|
||||
aws_external_id=aws_external_id,
|
||||
aws_session_tags=aws_session_tags,
|
||||
)
|
||||
elif _custom_llm_provider == "xai":
|
||||
api_base = (
|
||||
|
|
|
|||
|
|
@ -228,6 +228,7 @@ from litellm.router_utils.router_callbacks.track_deployment_metrics import (
|
|||
increment_deployment_failures_for_current_minute,
|
||||
increment_deployment_successes_for_current_minute,
|
||||
)
|
||||
from litellm.router_utils.routing_groups import parse_routing_groups, validate_routing_strategy
|
||||
from litellm.scheduler import FlowItem, Scheduler
|
||||
from litellm.types.llms.openai import (
|
||||
AllMessageValues,
|
||||
|
|
@ -1285,20 +1286,9 @@ class Router:
|
|||
return strategy.value
|
||||
return strategy
|
||||
|
||||
def _validate_routing_strategy(self, routing_strategy: RoutingStrategy | str | None) -> None:
|
||||
# See: https://github.com/BerriAI/litellm/issues/11330
|
||||
valid_strategy_strings: Final = ["simple-shuffle", "lar1"] + [s.value for s in RoutingStrategy]
|
||||
if routing_strategy is None:
|
||||
return
|
||||
is_valid_string: Final = isinstance(routing_strategy, str) and routing_strategy in valid_strategy_strings
|
||||
is_valid_enum: Final = isinstance(routing_strategy, RoutingStrategy)
|
||||
if not is_valid_string and not is_valid_enum:
|
||||
raise ValueError(
|
||||
f"Invalid routing_strategy: '{routing_strategy}'. "
|
||||
f"Valid options: {valid_strategy_strings}. "
|
||||
f"Check 'router_settings.routing_strategy' in your config.yaml "
|
||||
f"or the 'routing_strategy' parameter if using the Router SDK directly."
|
||||
)
|
||||
@staticmethod
|
||||
def _validate_routing_strategy(routing_strategy: RoutingStrategy | str | None) -> None:
|
||||
validate_routing_strategy(routing_strategy)
|
||||
|
||||
def _build_strategy_selector(
|
||||
self,
|
||||
|
|
@ -1315,11 +1305,6 @@ class Router:
|
|||
match self._normalize_strategy(strategy):
|
||||
case RoutingStrategy.LEAST_BUSY.value:
|
||||
selector = LeastBusyLoggingHandler(router_cache=self.cache)
|
||||
if register_callbacks:
|
||||
if isinstance(litellm.input_callback, list):
|
||||
litellm.logging_callback_manager.add_litellm_input_callback(selector)
|
||||
else:
|
||||
litellm.input_callback = [selector]
|
||||
case RoutingStrategy.USAGE_BASED_ROUTING.value:
|
||||
selector = LowestTPMLoggingHandler(
|
||||
router_cache=self.cache,
|
||||
|
|
@ -1343,11 +1328,21 @@ class Router:
|
|||
case _:
|
||||
pass
|
||||
|
||||
if selector is not None and register_callbacks and isinstance(litellm.callbacks, list):
|
||||
litellm.logging_callback_manager.add_litellm_callback(selector)
|
||||
if selector is not None and register_callbacks:
|
||||
self._register_router_selector(selector)
|
||||
|
||||
return selector
|
||||
|
||||
@staticmethod
|
||||
def _register_router_selector(selector: RouterStrategySelector) -> None:
|
||||
if isinstance(selector, LeastBusyLoggingHandler):
|
||||
if isinstance(litellm.input_callback, list):
|
||||
litellm.logging_callback_manager.add_litellm_input_callback(selector)
|
||||
else:
|
||||
litellm.input_callback = [selector]
|
||||
if isinstance(litellm.callbacks, list):
|
||||
litellm.logging_callback_manager.add_litellm_callback(selector)
|
||||
|
||||
def _unregister_router_selectors(self, selectors: Sequence[object]) -> None:
|
||||
"""
|
||||
Drop router-owned strategy selectors from litellm's global callback
|
||||
|
|
@ -1442,71 +1437,61 @@ class Router:
|
|||
`"default"` group, whose selectors are the `self.<strategy>_logger`
|
||||
attributes set up in `routing_strategy_init`.
|
||||
"""
|
||||
group_selectors: Final[Mapping[str, Mapping[str, RouterStrategySelector]]] = getattr(
|
||||
self, "_group_selectors", {}
|
||||
)
|
||||
self._unregister_router_selectors([sel for selectors in group_selectors.values() for sel in selectors.values()])
|
||||
|
||||
self._routing_groups: dict[str, RoutingGroup] = {}
|
||||
self._model_to_group: dict[str, str] = {}
|
||||
self._group_selectors: dict[str, dict[str, RouterStrategySelector]] = {}
|
||||
self._invalidate_model_group_info_cache()
|
||||
self._invalidate_access_groups_cache()
|
||||
|
||||
if not groups_input:
|
||||
self._replace_routing_groups(())
|
||||
return
|
||||
|
||||
known_model_names: Final = {m.get("model_name") for m in (self.model_list or []) if m.get("model_name")}
|
||||
known_model_names: Final = frozenset(m["model_name"] for m in (self.model_list or ()) if m.get("model_name"))
|
||||
groups: Final = parse_routing_groups(groups_input, known_model_names=known_model_names)
|
||||
|
||||
seen_group_names: Final[set] = set()
|
||||
for raw in groups_input:
|
||||
group = raw if isinstance(raw, RoutingGroup) else RoutingGroup(**raw)
|
||||
|
||||
if not group.group_name:
|
||||
raise ValueError("routing_groups: group_name must be non-empty.")
|
||||
if group.group_name == "default":
|
||||
raise ValueError("routing_groups: 'default' is reserved for the implicit fallback group.")
|
||||
if group.group_name in known_model_names or group.group_name in (self.model_group_alias or {}):
|
||||
alias_names: Final = frozenset(self.model_group_alias or ())
|
||||
for group in groups:
|
||||
if group.group_name in known_model_names or group.group_name in alias_names:
|
||||
verbose_router_logger.warning(
|
||||
"routing_groups: group_name '%s' is shadowed by an existing model_name or model_group_alias; "
|
||||
"the group's strategy still applies to its members, but the name is not callable until renamed.",
|
||||
group.group_name,
|
||||
)
|
||||
if group.group_name in seen_group_names:
|
||||
raise ValueError(
|
||||
f"routing_groups: group names must be unique, duplicate group_name '{group.group_name}'."
|
||||
)
|
||||
seen_group_names.add(group.group_name)
|
||||
|
||||
self._validate_routing_strategy(group.routing_strategy)
|
||||
|
||||
for model_name in group.models:
|
||||
if model_name in self._model_to_group:
|
||||
raise ValueError(
|
||||
f"routing_groups: model_name '{model_name}' appears in "
|
||||
f"both '{self._model_to_group[model_name]}' and "
|
||||
f"'{group.group_name}'. Each model may belong to at most one group."
|
||||
)
|
||||
if known_model_names and model_name not in known_model_names:
|
||||
verbose_router_logger.warning(
|
||||
"routing_groups: model_name '%s' (group '%s') is not in model_list; "
|
||||
"the group entry will only take effect once a deployment with that "
|
||||
"model_name is added.",
|
||||
model_name,
|
||||
group.group_name,
|
||||
)
|
||||
self._model_to_group[model_name] = group.group_name
|
||||
|
||||
self._routing_groups[group.group_name] = group
|
||||
|
||||
strategy_value = self._normalize_strategy(group.routing_strategy) or ""
|
||||
group_selector = self._build_strategy_selector(
|
||||
strategy=group.routing_strategy,
|
||||
routing_strategy_args=group.routing_strategy_args or {},
|
||||
built: Final = tuple(
|
||||
(
|
||||
group,
|
||||
self._build_strategy_selector(
|
||||
strategy=group.routing_strategy,
|
||||
routing_strategy_args=group.routing_strategy_args or {},
|
||||
register_callbacks=False,
|
||||
),
|
||||
)
|
||||
self._group_selectors[group.group_name] = (
|
||||
{strategy_value: group_selector} if group_selector is not None else {}
|
||||
for group in groups
|
||||
)
|
||||
self._replace_routing_groups(built)
|
||||
|
||||
def _replace_routing_groups(
|
||||
self,
|
||||
built: tuple[tuple[RoutingGroup, RouterStrategySelector | None], ...],
|
||||
) -> None:
|
||||
previous_selectors: Final[Mapping[str, Mapping[str, RouterStrategySelector]]] = getattr(
|
||||
self, "_group_selectors", {}
|
||||
)
|
||||
self._unregister_router_selectors(
|
||||
tuple(sel for selectors in previous_selectors.values() for sel in selectors.values())
|
||||
)
|
||||
for _, selector in built:
|
||||
if selector is not None:
|
||||
self._register_router_selector(selector)
|
||||
|
||||
self._routing_groups: dict[str, RoutingGroup] = {group.group_name: group for group, _ in built}
|
||||
self._model_to_group: dict[str, str] = {
|
||||
model_name: group.group_name for group, _ in built for model_name in group.models
|
||||
}
|
||||
self._group_selectors: dict[str, dict[str, RouterStrategySelector]] = {
|
||||
group.group_name: (
|
||||
{} if selector is None else {self._normalize_strategy(group.routing_strategy) or "": selector}
|
||||
)
|
||||
for group, selector in built
|
||||
}
|
||||
self._invalidate_model_group_info_cache()
|
||||
self._invalidate_access_groups_cache()
|
||||
|
||||
def get_routing_group(self, model_name: str) -> RoutingGroup | None:
|
||||
"""
|
||||
|
|
@ -11980,7 +11965,6 @@ class Router:
|
|||
_casted_value = int(kwargs[var])
|
||||
setattr(self, var, _casted_value)
|
||||
elif var == "routing_groups":
|
||||
self._routing_groups_input = kwargs[var]
|
||||
rebuild_routing_groups = True
|
||||
elif var == "optional_pre_call_checks":
|
||||
self.set_optional_pre_call_checks(kwargs[var])
|
||||
|
|
@ -12021,7 +12005,9 @@ class Router:
|
|||
self._apply_updated_routing_strategy_args()
|
||||
|
||||
if rebuild_routing_groups:
|
||||
self._init_routing_groups(self._routing_groups_input)
|
||||
routing_groups_input: Final = kwargs.get("routing_groups", self._routing_groups_input)
|
||||
self._init_routing_groups(routing_groups_input)
|
||||
self._routing_groups_input = routing_groups_input
|
||||
verbose_router_logger.debug("Updated Router settings: %s", self.get_settings())
|
||||
|
||||
def _get_client(self, deployment, kwargs, client_type=None):
|
||||
|
|
|
|||
78
litellm/router_utils/routing_groups.py
Normal file
78
litellm/router_utils/routing_groups.py
Normal file
|
|
@ -0,0 +1,78 @@
|
|||
from collections.abc import Sequence
|
||||
from typing import Final
|
||||
|
||||
from litellm._logging import verbose_router_logger
|
||||
from litellm.types.router import RoutingGroup, RoutingStrategy
|
||||
|
||||
|
||||
def validate_routing_strategy(routing_strategy: RoutingStrategy | str | None) -> None:
|
||||
if routing_strategy is None:
|
||||
return
|
||||
|
||||
valid_strategy_strings: Final = ("simple-shuffle", "lar1", *(s.value for s in RoutingStrategy))
|
||||
is_valid_string: Final = isinstance(routing_strategy, str) and routing_strategy in valid_strategy_strings
|
||||
is_valid_enum: Final = isinstance(routing_strategy, RoutingStrategy)
|
||||
if not is_valid_string and not is_valid_enum:
|
||||
raise ValueError(
|
||||
f"Invalid routing_strategy: '{routing_strategy}'. "
|
||||
f"Valid options: {list(valid_strategy_strings)}. "
|
||||
f"Check 'router_settings.routing_strategy' in your config.yaml "
|
||||
f"or the 'routing_strategy' parameter if using the Router SDK directly."
|
||||
)
|
||||
|
||||
|
||||
def parse_routing_groups(
|
||||
groups_input: Sequence[RoutingGroup | dict] | None,
|
||||
known_model_names: frozenset[str] = frozenset(),
|
||||
) -> tuple[RoutingGroup, ...]:
|
||||
if not groups_input:
|
||||
return ()
|
||||
|
||||
groups: Final = tuple(raw if isinstance(raw, RoutingGroup) else RoutingGroup(**raw) for raw in groups_input)
|
||||
|
||||
if any(not group.group_name for group in groups):
|
||||
raise ValueError("routing_groups: group_name must be non-empty.")
|
||||
|
||||
if any(group.group_name == "default" for group in groups):
|
||||
raise ValueError("routing_groups: 'default' is reserved for the implicit fallback group.")
|
||||
|
||||
names: Final = tuple(group.group_name for group in groups)
|
||||
duplicate_names: Final = frozenset(name for name in names if names.count(name) > 1)
|
||||
if duplicate_names:
|
||||
raise ValueError(f"routing_groups: group names must be unique, duplicate group_name '{min(duplicate_names)}'.")
|
||||
|
||||
for group in groups:
|
||||
validate_routing_strategy(group.routing_strategy)
|
||||
|
||||
owners_by_model: Final = tuple(
|
||||
(model_name, tuple(group.group_name for group in groups if model_name in group.models))
|
||||
for model_name in dict.fromkeys(model_name for group in groups for model_name in group.models)
|
||||
)
|
||||
conflicts: Final = tuple(
|
||||
f"model_name '{model_name}' appears in {' and '.join(repr(owner) for owner in owners)}"
|
||||
for model_name, owners in owners_by_model
|
||||
if len(owners) > 1
|
||||
)
|
||||
if conflicts:
|
||||
raise ValueError(f"routing_groups: {'; '.join(conflicts)}. Each model may belong to at most one group.")
|
||||
|
||||
unknown_models: Final = (
|
||||
tuple(
|
||||
(model_name, group.group_name)
|
||||
for group in groups
|
||||
for model_name in group.models
|
||||
if model_name not in known_model_names
|
||||
)
|
||||
if known_model_names
|
||||
else ()
|
||||
)
|
||||
for model_name, group_name in unknown_models:
|
||||
verbose_router_logger.warning(
|
||||
"routing_groups: model_name '%s' (group '%s') is not in model_list; "
|
||||
"the group entry will only take effect once a deployment with that "
|
||||
"model_name is added.",
|
||||
model_name,
|
||||
group_name,
|
||||
)
|
||||
|
||||
return groups
|
||||
|
|
@ -62,6 +62,9 @@ from litellm.types.proxy.guardrails.guardrail_hooks.singulr import (
|
|||
from litellm.types.proxy.guardrails.guardrail_hooks.tool_permission import (
|
||||
ToolPermissionGuardrailConfigModel,
|
||||
)
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.typesafe import (
|
||||
TypeSafeGuardrailConfigModel,
|
||||
)
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.vigil_guard import (
|
||||
VigilGuardGuardrailConfigModel,
|
||||
)
|
||||
|
|
@ -138,6 +141,7 @@ class SupportedGuardrailIntegrations(Enum):
|
|||
SINGULR = "singulr"
|
||||
HEADROOM = "headroom"
|
||||
COMPRESR = "compresr"
|
||||
TYPESAFE = "typesafe"
|
||||
STRAIKER = "straiker"
|
||||
ALICE = "alice"
|
||||
AGENT_365 = "agent_365"
|
||||
|
|
@ -1055,7 +1059,7 @@ class BaseLitellmParams(ContentFilterConfigModel): # works for new and patch up
|
|||
default="fail_closed",
|
||||
description=(
|
||||
"Behavior when a guardrail endpoint is unreachable due to network errors. "
|
||||
"Implemented by guardrail='generic_guardrail_api', 'agent_365', 'akto', 'vigil_guard', 'repelloai', 'headroom', and 'compresr'. "
|
||||
"Implemented by guardrail='generic_guardrail_api', 'agent_365', 'akto', 'vigil_guard', 'repelloai', 'headroom', 'compresr', and 'typesafe'. "
|
||||
"'fail_closed' raises an error (default). 'fail_open' logs a critical error and allows the request to proceed."
|
||||
),
|
||||
)
|
||||
|
|
@ -1171,6 +1175,7 @@ class LitellmParams( # pyright: ignore[reportIncompatibleVariableOverride] # o
|
|||
LakeraV2GuardrailConfigModel,
|
||||
HeadroomGuardrailConfigModel,
|
||||
CompresrGuardrailConfigModel,
|
||||
TypeSafeGuardrailConfigModel,
|
||||
RepelloAIGuardrailConfigModel,
|
||||
LassoGuardrailConfigModel,
|
||||
DeepKeepGuardrailConfigModel,
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ from collections.abc import Sequence
|
|||
from enum import Enum
|
||||
from typing import TYPE_CHECKING, Any, Final, Literal, TypeAlias
|
||||
|
||||
from pydantic import BaseModel, ConfigDict
|
||||
from typing_extensions import ReadOnly, Required, TypedDict, override
|
||||
|
||||
from .openai import ChatCompletionToolCallChunk
|
||||
|
|
@ -1112,6 +1113,26 @@ class AwsSessionTag(TypedDict):
|
|||
Value: str # writable-ok: boto3's STS stubs type assume_role Tags as writable TagTypeDef, which rejects ReadOnly
|
||||
|
||||
|
||||
class AwsAuthParams(BaseModel):
|
||||
"""Every credential-shaped aws_* param BaseAWSLLM.get_credentials accepts; region is resolved separately."""
|
||||
|
||||
model_config = ConfigDict(frozen=True, extra="ignore")
|
||||
|
||||
aws_access_key_id: str | None = None
|
||||
aws_secret_access_key: str | None = None
|
||||
aws_session_token: str | None = None
|
||||
aws_session_name: str | None = None
|
||||
aws_profile_name: str | None = None
|
||||
aws_role_name: str | None = None
|
||||
aws_web_identity_token: str | None = None
|
||||
aws_sts_endpoint: str | None = None
|
||||
aws_external_id: str | None = None
|
||||
aws_session_tags: object = None
|
||||
|
||||
|
||||
AWS_AUTH_PARAM_KEYS: Final[tuple[str, ...]] = tuple(AwsAuthParams.model_fields)
|
||||
|
||||
|
||||
class BedrockCreateBatchRequest(TypedDict, total=False):
|
||||
"""
|
||||
Request structure for creating a Bedrock batch inference job.
|
||||
|
|
|
|||
|
|
@ -464,3 +464,11 @@ class MCPGatewaySessionsResponse(BaseModel):
|
|||
by_client: list[MCPGatewaySessionGroupCount] = Field(default_factory=list)
|
||||
by_user: list[MCPGatewaySessionGroupCount] = Field(default_factory=list)
|
||||
sessions: list[MCPGatewaySession] = Field(default_factory=list)
|
||||
|
||||
|
||||
class MCPGatewaySessionsTerminateResponse(BaseModel):
|
||||
"""Stateful sessions an administrator force-closed on this proxy worker."""
|
||||
|
||||
worker_pid: int
|
||||
terminated_sessions: int
|
||||
sessions: list[MCPGatewaySession] = Field(default_factory=list)
|
||||
|
|
|
|||
63
litellm/types/proxy/guardrails/guardrail_hooks/typesafe.py
Normal file
63
litellm/types/proxy/guardrails/guardrail_hooks/typesafe.py
Normal file
|
|
@ -0,0 +1,63 @@
|
|||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from .base import GuardrailConfigModel
|
||||
|
||||
|
||||
class TypeSafeGuardrailOptionalParams(BaseModel):
|
||||
"""Optional tuning knobs for the TypeSafe (Jev) compaction guardrail."""
|
||||
|
||||
relevance_threshold: float | None = Field(
|
||||
default=None,
|
||||
ge=0.0,
|
||||
le=1.0,
|
||||
description=(
|
||||
"Relevance cutoff in [0, 1]. A completed tool exchange is dropped when Jev "
|
||||
"scores the probability that it is still needed below this value. Defaults to 0.2."
|
||||
),
|
||||
)
|
||||
min_chars_to_evaluate: int | None = Field(
|
||||
default=None,
|
||||
ge=0,
|
||||
description=(
|
||||
"Skip tool exchanges whose combined tool-result text is shorter than this many characters. Defaults to 200."
|
||||
),
|
||||
)
|
||||
max_result_chars_in_state: int | None = Field(
|
||||
default=None,
|
||||
ge=1,
|
||||
description=(
|
||||
"Tool result text is truncated to this many characters when sent to the Jev evaluator, "
|
||||
"keeping the head and tail. Defaults to 4000."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
class TypeSafeGuardrailConfigModel(GuardrailConfigModel[TypeSafeGuardrailOptionalParams]):
|
||||
api_key: str | None = Field(
|
||||
default=None,
|
||||
description="TypeSafe API key, sent as a Bearer token. Falls back to the TYPESAFE_API_KEY env var.",
|
||||
)
|
||||
api_base: str | None = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"Base URL of the TypeSafe API. Falls back to the TYPESAFE_API_BASE env var, then https://api.typesafe.ai."
|
||||
),
|
||||
)
|
||||
model: str | None = Field(
|
||||
default=None,
|
||||
description="TypeSafe evaluation model (not the LLM). Defaults to 'jev-latest'.",
|
||||
)
|
||||
unreachable_fallback: Literal["fail_closed", "fail_open"] = Field(
|
||||
default="fail_open",
|
||||
description=(
|
||||
"Behavior when the TypeSafe evaluation service is unreachable or errors. "
|
||||
"'fail_open' (default) forwards the request uncompacted. 'fail_closed' "
|
||||
"raises an error instead."
|
||||
),
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def ui_friendly_name() -> str:
|
||||
return "TypeSafe (Jev) Compaction"
|
||||
|
|
@ -100,6 +100,16 @@ class DailySpendMetadata(BaseModel):
|
|||
page: int = Field(default=1)
|
||||
total_pages: int = Field(default=1)
|
||||
has_more: bool = Field(default=False)
|
||||
api_key_limit: int | None = Field(
|
||||
default=None,
|
||||
description="When set, api_keys and every api_key_breakdown list at most this many keys, "
|
||||
"ranked by spend. Totals and the model, provider, mcp and endpoint rollups still cover every key.",
|
||||
)
|
||||
total_api_keys: int | None = Field(
|
||||
default=None,
|
||||
description="Distinct API keys matching the filters. When this exceeds api_key_limit, the per-key "
|
||||
"lists are truncated to the highest-spend keys.",
|
||||
)
|
||||
|
||||
|
||||
class SpendAnalyticsPaginatedResponse(BaseModel):
|
||||
|
|
|
|||
|
|
@ -61,7 +61,9 @@ class SCIMUserGroup(BaseModel):
|
|||
|
||||
|
||||
class SCIMMultiValuedAttribute(BaseModel):
|
||||
value: str
|
||||
model_config = ConfigDict(extra="allow")
|
||||
|
||||
value: str | None = None
|
||||
display: str | None = None
|
||||
type: str | None = None
|
||||
primary: bool | None = None
|
||||
|
|
|
|||
|
|
@ -51,9 +51,7 @@ try:
|
|||
if general_settings_section:
|
||||
# Extract the table rows, which contain the documented keys
|
||||
table_content = general_settings_section.group(1)
|
||||
doc_key_pattern = re.compile(
|
||||
r"\|\s*([^\|]+?)\s*\|"
|
||||
) # Capture the key from each row of the table
|
||||
doc_key_pattern = re.compile(r"^\|\s*([^\|]+?)\s*\|", re.MULTILINE)
|
||||
documented_keys.update(doc_key_pattern.findall(table_content))
|
||||
except Exception as e:
|
||||
raise Exception(
|
||||
|
|
|
|||
|
|
@ -185,19 +185,19 @@ class DummyCredentials:
|
|||
],
|
||||
)
|
||||
@pytest.mark.parametrize(
|
||||
"param_name, param_value",
|
||||
"param_name, param_value, expected_credentials_value",
|
||||
[
|
||||
("aws_session_token", "dummy_session_token"),
|
||||
("aws_session_name", "dummy_session_name"),
|
||||
("aws_profile_name", "dummy_profile_name"),
|
||||
("aws_role_name", "dummy_role_name"),
|
||||
("aws_web_identity_token", "dummy_web_identity_token"),
|
||||
("aws_sts_endpoint", "dummy_sts_endpoint"),
|
||||
("aws_external_id", "dummy_external_id"),
|
||||
("aws_session_tags", [{"Key": "team", "Value": "genai"}]),
|
||||
("aws_session_token", "dummy_session_token", "dummy_session_token"),
|
||||
("aws_session_name", "dummy_session_name", "dummy_session_name"),
|
||||
("aws_profile_name", "dummy_profile_name", "dummy_profile_name"),
|
||||
("aws_role_name", "dummy_role_name", "dummy_role_name"),
|
||||
("aws_web_identity_token", "dummy_web_identity_token", "dummy_web_identity_token"),
|
||||
("aws_sts_endpoint", "dummy_sts_endpoint", "dummy_sts_endpoint"),
|
||||
("aws_external_id", "dummy_external_id", "dummy_external_id"),
|
||||
("aws_session_tags", [{"Key": "team", "Value": "genai"}], ({"Key": "team", "Value": "genai"},)),
|
||||
],
|
||||
)
|
||||
def test_dynamic_aws_params_propagation(model, param_name, param_value):
|
||||
def test_dynamic_aws_params_propagation(model, param_name, param_value, expected_credentials_value):
|
||||
"""
|
||||
When passed to litellm.completion, each dynamic AWS authentication parameter
|
||||
should propagate down to the get_credentials() call in BaseAWSLLM.
|
||||
|
|
@ -282,6 +282,4 @@ def test_dynamic_aws_params_propagation(model, param_name, param_value):
|
|||
)
|
||||
|
||||
# We now assert that get_credentials() was called with the dynamic param.
|
||||
assert (
|
||||
dummy_get_credentials.called_kwargs.get(param_name) == param_value
|
||||
)
|
||||
assert dummy_get_credentials.called_kwargs.get(param_name) == expected_credentials_value
|
||||
|
|
|
|||
|
|
@ -331,9 +331,7 @@ class TestMCPPerUserTokenCache:
|
|||
with patch("litellm.proxy.proxy_server.user_api_key_cache", mock_dual_cache):
|
||||
await cache.delete("alice", "slack-test")
|
||||
|
||||
mock_dual_cache.async_delete_cache.assert_called_once_with(
|
||||
"mcp:per_user_token:alice:slack-test"
|
||||
)
|
||||
mock_dual_cache.async_delete_cache.assert_called_once_with(key="mcp:per_user_token:alice:slack-test")
|
||||
mock_dual_cache.async_set_cache.assert_not_called()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
|
|||
|
|
@ -2629,6 +2629,100 @@ def test_sign_s3_request_without_body_assumes_role_with_external_id(monkeypatch)
|
|||
assert "ASIAFILESGETROLE" in authorization
|
||||
|
||||
|
||||
class _SessionTagGatedSTSClient:
|
||||
"""Mimics a trust policy with an aws:RequestTag condition: assume_role only succeeds with the expected tags."""
|
||||
|
||||
def __init__(self, expected_tags, access_key_id):
|
||||
self.expected_tags = expected_tags
|
||||
self.access_key_id = access_key_id
|
||||
|
||||
def get_caller_identity(self):
|
||||
return {"Arn": "arn:aws:iam::111111111111:user/litellm-proxy-pod"}
|
||||
|
||||
def assume_role(self, **params):
|
||||
import datetime
|
||||
|
||||
from botocore.exceptions import ClientError
|
||||
|
||||
if list(params.get("Tags") or ()) != self.expected_tags:
|
||||
raise ClientError(
|
||||
{"Error": {"Code": "AccessDenied", "Message": "is not authorized to perform: sts:TagSession"}},
|
||||
"AssumeRole",
|
||||
)
|
||||
return {
|
||||
"Credentials": {
|
||||
"AccessKeyId": self.access_key_id,
|
||||
"SecretAccessKey": "assumed-secret",
|
||||
"SessionToken": "assumed-session-token",
|
||||
"Expiration": datetime.datetime.now(datetime.timezone.utc) + datetime.timedelta(minutes=30),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
def test_sign_s3_request_assumes_role_with_session_tags():
|
||||
"""The deployment's aws_session_tags must reach STS when signing the S3 upload, not only on chat calls."""
|
||||
from unittest.mock import patch
|
||||
|
||||
import boto3
|
||||
|
||||
from litellm.llms.bedrock.files.transformation import BedrockFilesConfig
|
||||
|
||||
expected_tags = [{"Key": "team", "Value": "genai"}]
|
||||
optional_params = {
|
||||
"aws_region_name": "us-east-1",
|
||||
"aws_access_key_id": "AKIAFILESPUTCALLER",
|
||||
"aws_secret_access_key": "pod-caller-secret",
|
||||
"aws_role_name": "arn:aws:iam::999999999999:role/litellm-files-put-role",
|
||||
"aws_session_name": "litellm-files-put-session",
|
||||
"aws_session_tags": [{"Key": "team", "Value": "genai"}],
|
||||
}
|
||||
|
||||
with patch.object(boto3, "client", return_value=_SessionTagGatedSTSClient(expected_tags, "ASIAFILESPUTTAGGED")):
|
||||
signed_headers, _signed_body = BedrockFilesConfig()._sign_s3_request(
|
||||
content='{"custom_id": "req-1"}',
|
||||
api_base="https://s3.us-east-1.amazonaws.com/safe-bucket/litellm-bedrock-files-model-id-abc.jsonl",
|
||||
optional_params=optional_params,
|
||||
)
|
||||
|
||||
authorization = {key.lower(): value for key, value in signed_headers.items()}["authorization"]
|
||||
assert "ASIAFILESPUTTAGGED" in authorization
|
||||
|
||||
|
||||
def test_sign_s3_request_without_body_assumes_role_with_session_tags():
|
||||
"""The deployment's aws_session_tags must reach STS when signing the S3 download too."""
|
||||
from unittest.mock import patch
|
||||
|
||||
import boto3
|
||||
|
||||
from litellm.llms.bedrock.files.transformation import (
|
||||
BedrockFilesConfig,
|
||||
_BedrockS3RequestParams,
|
||||
)
|
||||
|
||||
expected_tags = [{"Key": "team", "Value": "genai"}]
|
||||
request_params = _BedrockS3RequestParams.model_validate(
|
||||
{
|
||||
"aws_region_name": "us-east-1",
|
||||
"aws_access_key_id": "AKIAFILESGETCALLER",
|
||||
"aws_secret_access_key": "pod-caller-secret",
|
||||
"aws_role_name": "arn:aws:iam::999999999999:role/litellm-files-get-role",
|
||||
"aws_session_name": "litellm-files-get-session",
|
||||
"aws_session_tags": [{"Key": "team", "Value": "genai"}],
|
||||
}
|
||||
)
|
||||
|
||||
with patch.object(boto3, "client", return_value=_SessionTagGatedSTSClient(expected_tags, "ASIAFILESGETTAGGED")):
|
||||
signed_headers = BedrockFilesConfig()._sign_s3_request_without_body(
|
||||
method="GET",
|
||||
api_base="https://s3.us-east-1.amazonaws.com/safe-bucket/litellm-bedrock-files-model-id-abc.jsonl",
|
||||
aws_region_name="us-east-1",
|
||||
request_params=request_params,
|
||||
)
|
||||
|
||||
authorization = {key.lower(): value for key, value in signed_headers.items()}["authorization"]
|
||||
assert "ASIAFILESGETTAGGED" in authorization
|
||||
|
||||
|
||||
def _s3_signature_for(method: str, url: str, headers: Mapping[str, str]) -> str:
|
||||
sent = {name.lower(): value for name, value in headers.items()}
|
||||
signed_names = sent["authorization"].split("SignedHeaders=")[1].split(",")[0].split(";")
|
||||
|
|
|
|||
|
|
@ -855,6 +855,7 @@ class TestBedrockRealtimeAwsAuth:
|
|||
aws_role_name="arn:aws:iam::123456789012:role/nova-sonic",
|
||||
aws_session_name="realtime-session",
|
||||
aws_external_id="realtime-external-id",
|
||||
aws_session_tags=[{"Key": "team", "Value": "realtime"}],
|
||||
)
|
||||
|
||||
assert handler.get_credentials_kwargs == {
|
||||
|
|
@ -868,6 +869,7 @@ class TestBedrockRealtimeAwsAuth:
|
|||
"aws_web_identity_token": None,
|
||||
"aws_sts_endpoint": None,
|
||||
"aws_external_id": "realtime-external-id",
|
||||
"aws_session_tags": ({"Key": "team", "Value": "realtime"},),
|
||||
}
|
||||
resolver = stub_aws_sdk_client["config_kwargs"]["aws_credentials_identity_resolver"]
|
||||
assert isinstance(resolver, FakeStaticCredentialsResolver)
|
||||
|
|
|
|||
|
|
@ -10,6 +10,7 @@ from fastapi.testclient import TestClient
|
|||
|
||||
|
||||
|
||||
from collections.abc import Callable
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any, Dict, Optional
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
|
@ -3555,3 +3556,148 @@ def test_run_aws_signing_leaves_the_default_executor_free_for_other_providers():
|
|||
other_provider, signing_thread = asyncio.run(scenario())
|
||||
assert other_provider != signing_thread
|
||||
assert signing_thread.startswith("aws-signing")
|
||||
|
||||
|
||||
def _recording_boto3_client(recorded: dict[str, dict[str, object]]) -> Callable[..., MagicMock]:
|
||||
"""boto3.client replacement that records the STS client kwargs and the assume-role params."""
|
||||
|
||||
def _client(service_name: str, **client_kwargs: object) -> MagicMock:
|
||||
recorded["client_kwargs"] = client_kwargs
|
||||
sts = MagicMock()
|
||||
|
||||
def _assume(**params: object) -> dict[str, object]:
|
||||
recorded["assume_role"] = params
|
||||
return {
|
||||
"Credentials": {
|
||||
"AccessKeyId": "ASIAASSUMED",
|
||||
"SecretAccessKey": "assumed-secret",
|
||||
"SessionToken": "assumed-token",
|
||||
"Expiration": datetime.now(timezone.utc) + timedelta(minutes=30),
|
||||
}
|
||||
}
|
||||
|
||||
def _assume_web_identity(**params: object) -> dict[str, object]:
|
||||
recorded["assume_role_with_web_identity"] = params
|
||||
return {
|
||||
"Credentials": {
|
||||
"AccessKeyId": "ASIAWEBIDENTITY",
|
||||
"SecretAccessKey": "assumed-secret",
|
||||
"SessionToken": "assumed-token",
|
||||
"Expiration": datetime.now(timezone.utc) + timedelta(minutes=30),
|
||||
},
|
||||
"PackedPolicySize": 10,
|
||||
}
|
||||
|
||||
sts.assume_role.side_effect = _assume
|
||||
sts.assume_role_with_web_identity.side_effect = _assume_web_identity
|
||||
return sts
|
||||
|
||||
return _client
|
||||
|
||||
|
||||
def test_resolve_credentials_forwards_static_keys_role_session_and_external_id():
|
||||
"""Every field the role-assumption route reads must reach STS, so a dropped struct field fails here."""
|
||||
from litellm.types.llms.bedrock import AwsAuthParams
|
||||
|
||||
auth_params = AwsAuthParams(
|
||||
aws_access_key_id="AKIACALLER",
|
||||
aws_secret_access_key="caller-secret",
|
||||
aws_session_token="caller-token",
|
||||
aws_role_name="arn:aws:iam::123456789012:role/litellm-target",
|
||||
aws_session_name="litellm-session",
|
||||
aws_external_id="litellm-external-id",
|
||||
aws_sts_endpoint="https://custom-sts.example",
|
||||
aws_session_tags=[{"Key": "team", "Value": "genai"}, {"Key": "cost-center", "Value": "42"}],
|
||||
)
|
||||
recorded: dict[str, dict[str, object]] = {}
|
||||
|
||||
with (
|
||||
patch.dict(os.environ, _os_environ_without_aws_keys(), clear=True),
|
||||
patch("boto3.client", side_effect=_recording_boto3_client(recorded)),
|
||||
):
|
||||
credentials = BaseAWSLLM().resolve_credentials(auth_params, "us-east-1")
|
||||
|
||||
assert recorded["client_kwargs"]["aws_access_key_id"] == "AKIACALLER"
|
||||
assert recorded["client_kwargs"]["aws_secret_access_key"] == "caller-secret"
|
||||
assert recorded["client_kwargs"]["aws_session_token"] == "caller-token"
|
||||
assert recorded["client_kwargs"]["endpoint_url"] == "https://custom-sts.example"
|
||||
assert recorded["assume_role"]["RoleArn"] == "arn:aws:iam::123456789012:role/litellm-target"
|
||||
assert recorded["assume_role"]["RoleSessionName"] == "litellm-session"
|
||||
assert recorded["assume_role"]["ExternalId"] == "litellm-external-id"
|
||||
assert recorded["assume_role"]["Tags"] == (
|
||||
{"Key": "cost-center", "Value": "42"},
|
||||
{"Key": "team", "Value": "genai"},
|
||||
)
|
||||
assert credentials.access_key == "ASIAASSUMED"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"malformed_tags",
|
||||
[
|
||||
"team=genai",
|
||||
{"team": "genai"},
|
||||
[{"key": "team", "value": "genai"}],
|
||||
[{"Key": "team"}],
|
||||
],
|
||||
)
|
||||
def test_resolve_credentials_rejects_malformed_session_tags(malformed_tags):
|
||||
"""A struct built from raw config must surface the friendly session-tag error before STS is called."""
|
||||
from litellm.types.llms.bedrock import AwsAuthParams
|
||||
|
||||
auth_params = AwsAuthParams(
|
||||
aws_role_name="arn:aws:iam::123456789012:role/litellm-target",
|
||||
aws_session_name="litellm-session",
|
||||
aws_session_tags=malformed_tags,
|
||||
)
|
||||
recorded: dict[str, dict[str, object]] = {}
|
||||
|
||||
with (
|
||||
patch.dict(os.environ, _os_environ_without_aws_keys(), clear=True),
|
||||
patch("boto3.client", side_effect=_recording_boto3_client(recorded)),
|
||||
):
|
||||
with pytest.raises(ValueError, match="Invalid 'aws_session_tags' value"):
|
||||
BaseAWSLLM().resolve_credentials(auth_params, "us-east-1")
|
||||
|
||||
assert "assume_role" not in recorded
|
||||
|
||||
|
||||
def test_resolve_credentials_forwards_web_identity_token():
|
||||
"""A struct carrying a web-identity token must take the web-identity route, not plain role assumption."""
|
||||
from litellm.types.llms.bedrock import AwsAuthParams
|
||||
|
||||
auth_params = AwsAuthParams(
|
||||
aws_web_identity_token="unresolvable-oidc-token",
|
||||
aws_role_name="arn:aws:iam::123456789012:role/litellm-wif",
|
||||
aws_session_name="litellm-wif-session",
|
||||
)
|
||||
recorded: dict[str, dict[str, object]] = {}
|
||||
|
||||
with (
|
||||
patch.dict(os.environ, _os_environ_without_aws_keys(), clear=True),
|
||||
patch("boto3.client", side_effect=_recording_boto3_client(recorded)),
|
||||
):
|
||||
with pytest.raises(AwsAuthError) as exc:
|
||||
BaseAWSLLM().resolve_credentials(auth_params, "us-east-1")
|
||||
|
||||
assert exc.value.status_code == 401
|
||||
assert "assume_role" not in recorded
|
||||
|
||||
|
||||
def test_resolve_credentials_forwards_profile_name():
|
||||
"""The profile route must receive the struct's profile name rather than the ambient session."""
|
||||
from litellm.types.llms.bedrock import AwsAuthParams
|
||||
|
||||
auth_params = AwsAuthParams(aws_profile_name="litellm-qa-profile")
|
||||
session_instance = MagicMock()
|
||||
session_instance.get_credentials.return_value = Credentials(
|
||||
access_key="AKIAPROFILE", secret_key="profile-secret", token=None
|
||||
)
|
||||
|
||||
with (
|
||||
patch.dict(os.environ, _os_environ_without_aws_keys(), clear=True),
|
||||
patch("boto3.Session", return_value=session_instance) as mock_session_cls,
|
||||
):
|
||||
credentials = BaseAWSLLM().resolve_credentials(auth_params, "us-east-1")
|
||||
|
||||
assert mock_session_cls.call_args.kwargs["profile_name"] == "litellm-qa-profile"
|
||||
assert credentials.access_key == "AKIAPROFILE"
|
||||
|
|
|
|||
|
|
@ -0,0 +1,57 @@
|
|||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.proxy._experimental.mcp_server.byok_credential_cache import (
|
||||
CachedByokCredential,
|
||||
byok_credential_cache,
|
||||
byok_credential_cache_key,
|
||||
cache_byok_credential,
|
||||
get_cached_byok_credential,
|
||||
)
|
||||
from litellm.proxy.common_utils.auth_cache_invalidation_pubsub import AuthCacheInvalidationSubscriber
|
||||
from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache
|
||||
|
||||
|
||||
class _FakeRedisCache:
|
||||
namespace = None
|
||||
|
||||
def init_async_client(self) -> object:
|
||||
return object()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _empty_cache():
|
||||
byok_credential_cache.flush_cache()
|
||||
yield
|
||||
byok_credential_cache.flush_cache()
|
||||
|
||||
|
||||
def test_a_cached_negative_lookup_is_distinguishable_from_a_miss():
|
||||
assert get_cached_byok_credential("u-1", "srv-1") is None
|
||||
cache_byok_credential("u-1", "srv-1", None)
|
||||
assert get_cached_byok_credential("u-1", "srv-1") == CachedByokCredential(credential=None)
|
||||
cache_byok_credential("u-1", "srv-1", "sk-stored")
|
||||
assert get_cached_byok_credential("u-1", "srv-1") == CachedByokCredential(credential="sk-stored")
|
||||
assert get_cached_byok_credential("u-1", "srv-2") is None
|
||||
|
||||
|
||||
def test_peer_worker_invalidation_message_evicts_the_cached_credential():
|
||||
"""The key a mutating worker broadcasts must be the key every other worker caches under."""
|
||||
cache_byok_credential("mallory", "srv-byok", "sk-revoked")
|
||||
cache_byok_credential("alice", "srv-byok", "sk-kept")
|
||||
subscriber = AuthCacheInvalidationSubscriber(
|
||||
redis_cache=_FakeRedisCache(), # pyright: ignore[reportArgumentType] # subscriber is never started; only its message handler runs
|
||||
user_api_key_cache=UserApiKeyCache(),
|
||||
additional_in_memory_caches=(byok_credential_cache,),
|
||||
)
|
||||
|
||||
subscriber._apply_message( # pyright: ignore[reportPrivateUsage] # exercising the real cross-worker message handler
|
||||
{
|
||||
"type": "message",
|
||||
"data": json.dumps({"cache_key": byok_credential_cache_key("mallory", "srv-byok")}).encode(),
|
||||
}
|
||||
)
|
||||
|
||||
assert get_cached_byok_credential("mallory", "srv-byok") is None
|
||||
assert get_cached_byok_credential("alice", "srv-byok") == CachedByokCredential(credential="sk-kept")
|
||||
|
|
@ -592,7 +592,7 @@ async def test_check_byok_credential_missing_credential(monkeypatch):
|
|||
|
||||
monkeypatch.delenv("PROXY_BASE_URL", raising=False)
|
||||
monkeypatch.delenv("SERVER_ROOT_PATH", raising=False)
|
||||
monkeypatch.setattr(server_module, "_byok_cred_cache", {})
|
||||
server_module.byok_credential_cache.flush_cache()
|
||||
mock_prisma = MagicMock()
|
||||
|
||||
with (
|
||||
|
|
@ -628,7 +628,7 @@ async def test_execute_byok_tool_missing_credential_advertises_api_key_flow(monk
|
|||
from litellm.types.mcp_server.mcp_server_manager import MCPServer
|
||||
|
||||
monkeypatch.setenv("PROXY_BASE_URL", "https://gateway.example.com/proxy")
|
||||
monkeypatch.setattr(mcp_module, "_byok_cred_cache", {})
|
||||
mcp_module.byok_credential_cache.flush_cache()
|
||||
server = MCPServer(server_id="byok-discovery", name="byok-discovery", transport=MCPTransport.http, is_byok=True)
|
||||
prisma = MagicMock()
|
||||
prisma.db.litellm_mcpusercredentials.find_unique = AsyncMock(return_value=None)
|
||||
|
|
@ -677,6 +677,40 @@ async def test_check_byok_credential_has_credential():
|
|||
await _check_byok_credential(server, user_auth)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_invalidate_byok_cred_cache_evicts_locally_and_broadcasts_the_same_key():
|
||||
"""A revoked credential must stop being served here and on every peer worker within the TTL."""
|
||||
from litellm.proxy._experimental.mcp_server import server as server_module
|
||||
from litellm.proxy._experimental.mcp_server.byok_credential_cache import byok_credential_cache_key
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.types.mcp_server.mcp_server_manager import MCPServer
|
||||
|
||||
server = MCPServer(server_id="byok-revoke", name="byok-server", transport=MCPTransport.http, is_byok=True)
|
||||
user_auth = UserAPIKeyAuth(user_id="mallory", api_key="sk-test")
|
||||
server_module.byok_credential_cache.flush_cache()
|
||||
db_lookup = AsyncMock(side_effect=["sk-before-revoke", None])
|
||||
publish = AsyncMock()
|
||||
|
||||
with (
|
||||
patch( # test-quality-ok: the DB row lookup is the only seam below the credential resolver; no Prisma fake exists
|
||||
"litellm.proxy._experimental.mcp_server.db.get_user_credential", new=db_lookup
|
||||
),
|
||||
patch( # test-quality-ok: the resolver reads the module-level prisma_client singleton; the suite's only seam
|
||||
"litellm.proxy.proxy_server.prisma_client", MagicMock()
|
||||
),
|
||||
patch.object( # test-quality-ok: the redis publisher is module-level; asserting the broadcast without a redis
|
||||
server_module, "publish_auth_cache_invalidation", new=publish
|
||||
),
|
||||
):
|
||||
assert await server_module._get_byok_credential(server, user_auth) == "sk-before-revoke"
|
||||
assert await server_module._get_byok_credential(server, user_auth) == "sk-before-revoke"
|
||||
await server_module._invalidate_byok_cred_cache("mallory", "byok-revoke")
|
||||
assert await server_module._get_byok_credential(server, user_auth) is None
|
||||
|
||||
assert db_lookup.await_count == 2
|
||||
publish.assert_awaited_once_with(cache_key=byok_credential_cache_key("mallory", "byok-revoke"))
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_check_byok_credential_db_unavailable_fails_closed():
|
||||
"""BYOK server with no prisma_client → 503, not silent pass.
|
||||
|
|
|
|||
|
|
@ -212,6 +212,53 @@ async def test_purge_user_oauth_credentials_for_server_invalidates_each_user():
|
|||
assert set(invalidations) == {("alice", "srv-1"), ("bob", "srv-1")}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_list_server_user_credentials_types_each_row_without_leaking_the_secret():
|
||||
"""The admin view of one server's stored credentials names the user and the kind of
|
||||
credential (OAuth2 vs BYOK) and echoes OAuth expiry, but never the token or key itself."""
|
||||
from litellm.proxy._experimental.mcp_server.db import list_server_user_credentials
|
||||
|
||||
oauth_row = _legacy_row(
|
||||
json.dumps(
|
||||
{
|
||||
"type": "oauth2",
|
||||
"access_token": "tok-alice",
|
||||
"expires_at": "2026-12-31T00:00:00+00:00",
|
||||
"connected_at": "2026-01-01T00:00:00+00:00",
|
||||
}
|
||||
)
|
||||
)
|
||||
oauth_row.user_id = "alice"
|
||||
oauth_row.updated_at = datetime(2026, 1, 1, tzinfo=timezone.utc)
|
||||
byok_row = _byok_row("carol")
|
||||
byok_row.updated_at = datetime(2026, 2, 1, tzinfo=timezone.utc)
|
||||
prisma = MagicMock()
|
||||
prisma.db.litellm_mcpusercredentials.find_many = AsyncMock(return_value=[oauth_row, byok_row])
|
||||
|
||||
items = await list_server_user_credentials(prisma, "srv-1")
|
||||
|
||||
prisma.db.litellm_mcpusercredentials.find_many.assert_awaited_once_with(where={"server_id": "srv-1"})
|
||||
assert [item.model_dump() for item in items] == [
|
||||
{
|
||||
"user_id": "alice",
|
||||
"credential_type": "oauth2",
|
||||
"expires_at": "2026-12-31T00:00:00+00:00",
|
||||
"connected_at": "2026-01-01T00:00:00+00:00",
|
||||
"updated_at": "2026-01-01T00:00:00+00:00",
|
||||
},
|
||||
{
|
||||
"user_id": "carol",
|
||||
"credential_type": "byok",
|
||||
"expires_at": None,
|
||||
"connected_at": None,
|
||||
"updated_at": "2026-02-01T00:00:00+00:00",
|
||||
},
|
||||
]
|
||||
serialized = "".join(item.model_dump_json() for item in items)
|
||||
assert "tok-alice" not in serialized
|
||||
assert "sk-byok-carol" not in serialized
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_purge_user_oauth_credentials_for_server_spares_byok_rows():
|
||||
"""Regression: the purge used to delete_many on server_id alone, wiping BYOK API keys that share
|
||||
|
|
|
|||
|
|
@ -3088,6 +3088,255 @@ def test_remove_stateful_session_tracking_drops_client_info():
|
|||
assert session_id not in mcp_server._stateful_session_client_info
|
||||
|
||||
|
||||
def _admin_terminate_fixture(mcp_server):
|
||||
def auth_user(user_id: str):
|
||||
return mcp_server.MCPAuthenticatedUser(
|
||||
user_api_key_auth=UserAPIKeyAuth(api_key=f"key-{user_id}", user_id=user_id),
|
||||
)
|
||||
|
||||
contexts = {
|
||||
"alice-session-1": auth_user("alice"),
|
||||
"alice-session-2": auth_user("alice"),
|
||||
"bob-session-1": auth_user("bob"),
|
||||
"anon-session-1": mcp_server.MCPAuthenticatedUser(user_api_key_auth=None),
|
||||
"gone-session-1": auth_user("alice"),
|
||||
}
|
||||
transports = {
|
||||
session_id: MagicMock(terminate=AsyncMock())
|
||||
for session_id in ("alice-session-1", "alice-session-2", "bob-session-1", "anon-session-1")
|
||||
}
|
||||
return contexts, transports
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_terminate_mcp_gateway_sessions_by_user_closes_every_live_session_of_that_user():
|
||||
try:
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
from litellm.proxy._experimental.mcp_server.server import session_manager_stateful
|
||||
except ImportError:
|
||||
pytest.skip("MCP server not available")
|
||||
|
||||
contexts, transports = _admin_terminate_fixture(mcp_server)
|
||||
live_transports = dict(transports)
|
||||
last_seen = {session_id: 100.0 for session_id in contexts}
|
||||
locks = {session_id: asyncio.Lock() for session_id in contexts}
|
||||
|
||||
with (
|
||||
patch.object( # test-quality-ok: the transport registry is a module-level singleton; the suite's only seam
|
||||
session_manager_stateful, "_server_instances", live_transports
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_auth_contexts, contexts, clear=True
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_auth_context_last_seen, last_seen, clear=True
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_locks, locks, clear=True
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_owners, {session_id: "owner" for session_id in contexts}, clear=True
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_active_request_counts, {}, clear=True
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_client_info, {}, clear=True
|
||||
),
|
||||
):
|
||||
result = await mcp_server.terminate_mcp_gateway_sessions(user_id="alice")
|
||||
|
||||
assert set(live_transports) == {"bob-session-1", "anon-session-1"}
|
||||
assert set(mcp_server._stateful_session_auth_contexts) == {"bob-session-1", "anon-session-1", "gone-session-1"}
|
||||
assert set(mcp_server._stateful_session_locks) == {"bob-session-1", "anon-session-1", "gone-session-1"}
|
||||
assert set(mcp_server._stateful_session_owners) == {"bob-session-1", "anon-session-1", "gone-session-1"}
|
||||
assert set(mcp_server._stateful_session_auth_context_last_seen) == {
|
||||
"bob-session-1",
|
||||
"anon-session-1",
|
||||
"gone-session-1",
|
||||
}
|
||||
|
||||
transports["alice-session-1"].terminate.assert_awaited_once()
|
||||
transports["alice-session-2"].terminate.assert_awaited_once()
|
||||
transports["bob-session-1"].terminate.assert_not_awaited()
|
||||
transports["anon-session-1"].terminate.assert_not_awaited()
|
||||
assert result.terminated_sessions == 2
|
||||
assert sorted(session.session_id_prefix for session in result.sessions) == ["alice-se", "alice-se"]
|
||||
assert {session.user_id for session in result.sessions} == {"alice"}
|
||||
assert "key-alice" not in result.model_dump_json()
|
||||
assert "alice-session-1" not in result.model_dump_json()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_terminate_mcp_gateway_sessions_prefix_and_user_must_both_match():
|
||||
try:
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
from litellm.proxy._experimental.mcp_server.server import session_manager_stateful
|
||||
except ImportError:
|
||||
pytest.skip("MCP server not available")
|
||||
|
||||
contexts, transports = _admin_terminate_fixture(mcp_server)
|
||||
live_transports = dict(transports)
|
||||
|
||||
with (
|
||||
patch.object( # test-quality-ok: the transport registry is a module-level singleton; the suite's only seam
|
||||
session_manager_stateful, "_server_instances", live_transports
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_auth_contexts, contexts, clear=True
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_client_info, {}, clear=True
|
||||
),
|
||||
):
|
||||
mismatch = await mcp_server.terminate_mcp_gateway_sessions(session_id_prefix="alice-session-1", user_id="bob")
|
||||
assert mismatch.terminated_sessions == 0
|
||||
assert set(live_transports) == set(transports)
|
||||
|
||||
stale = await mcp_server.terminate_mcp_gateway_sessions(session_id_prefix="gone-session-1")
|
||||
assert stale.terminated_sessions == 0
|
||||
|
||||
exact = await mcp_server.terminate_mcp_gateway_sessions(session_id_prefix="alice-session-1", user_id="alice")
|
||||
assert exact.terminated_sessions == 1
|
||||
assert set(live_transports) == {"alice-session-2", "bob-session-1", "anon-session-1"}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_admin_terminated_session_id_gets_404_instead_of_a_fresh_stateless_session():
|
||||
"""Once an admin closes a session, a client replaying its id must not be silently upgraded to a
|
||||
new stateless session by the stale-header path; it gets 404 and has to initialize again."""
|
||||
try:
|
||||
from starlette.types import Scope
|
||||
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
from litellm.proxy._experimental.mcp_server.server import session_manager_stateful
|
||||
except ImportError:
|
||||
pytest.skip("MCP server not available")
|
||||
|
||||
session_id = "admin-closed-session-1"
|
||||
live_transports = {session_id: MagicMock(terminate=AsyncMock())}
|
||||
contexts = {
|
||||
session_id: mcp_server.MCPAuthenticatedUser(
|
||||
user_api_key_auth=UserAPIKeyAuth(api_key="key-alice", user_id="alice"),
|
||||
)
|
||||
}
|
||||
|
||||
def scope_with_session_header() -> Scope:
|
||||
return {
|
||||
"type": "http",
|
||||
"method": "POST",
|
||||
"headers": [(b"content-type", b"application/json"), (b"mcp-session-id", session_id.encode())],
|
||||
}
|
||||
|
||||
try:
|
||||
with (
|
||||
patch.object( # test-quality-ok: the transport registry is a module-level singleton; the suite's only seam
|
||||
session_manager_stateful, "_server_instances", live_transports
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_auth_contexts, contexts, clear=True
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_client_info, {}, clear=True
|
||||
),
|
||||
):
|
||||
await mcp_server.terminate_mcp_gateway_sessions(session_id_prefix=session_id)
|
||||
|
||||
terminated_scope = scope_with_session_header()
|
||||
send = AsyncMock()
|
||||
handled = await mcp_server._handle_stale_mcp_session(
|
||||
terminated_scope, AsyncMock(), send, session_manager_stateful
|
||||
)
|
||||
|
||||
assert handled is True
|
||||
statuses = [m["status"] for (m,), _ in send.await_args_list if m["type"] == "http.response.start"]
|
||||
assert statuses == [404]
|
||||
assert [k for k, _ in terminated_scope["headers"]] == [b"content-type", b"mcp-session-id"]
|
||||
|
||||
unknown_scope = scope_with_session_header()
|
||||
unknown_scope["headers"][1] = (b"mcp-session-id", b"never-seen-session")
|
||||
assert (
|
||||
await mcp_server._handle_stale_mcp_session(
|
||||
unknown_scope, AsyncMock(), AsyncMock(), session_manager_stateful
|
||||
)
|
||||
is False
|
||||
)
|
||||
assert [k for k, _ in unknown_scope["headers"]] == [b"content-type"]
|
||||
finally:
|
||||
mcp_server._admin_terminated_session_ids.clear()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_admin_terminated_session_id_stays_refused_while_replayed_and_is_forgotten_like_an_idle_session():
|
||||
"""The refusal window slides on every replay, so a client that keeps retrying is never silently
|
||||
upgraded to a stateless session no matter how many other sessions an admin closes later; an id
|
||||
nobody has replayed for a full idle timeout is dropped from the table by the idle sweep."""
|
||||
try:
|
||||
from starlette.types import Scope
|
||||
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
from litellm.proxy._experimental.mcp_server.server import session_manager_stateful
|
||||
except ImportError:
|
||||
pytest.skip("MCP server not available")
|
||||
|
||||
idle_timeout = mcp_server._STATEFUL_SESSION_IDLE_TIMEOUT_SECONDS
|
||||
retrying_id, silent_id = "admin-closed-retrying", "admin-closed-silent"
|
||||
contexts = {
|
||||
session_id: mcp_server.MCPAuthenticatedUser(
|
||||
user_api_key_auth=UserAPIKeyAuth(api_key="key-alice", user_id="alice"),
|
||||
)
|
||||
for session_id in (retrying_id, silent_id)
|
||||
}
|
||||
live_transports = {session_id: MagicMock(terminate=AsyncMock()) for session_id in contexts}
|
||||
|
||||
async def replay(session_id: str, now: float) -> tuple[bool, list[bytes]]:
|
||||
scope: Scope = {
|
||||
"type": "http",
|
||||
"method": "POST",
|
||||
"headers": [(b"content-type", b"application/json"), (b"mcp-session-id", session_id.encode())],
|
||||
}
|
||||
with patch.object( # test-quality-ok: the stale-session handler reads the clock directly; no injectable now
|
||||
mcp_server.time, "monotonic", return_value=now
|
||||
):
|
||||
handled = await mcp_server._handle_stale_mcp_session(
|
||||
scope, AsyncMock(), AsyncMock(), session_manager_stateful
|
||||
)
|
||||
return handled, [k for k, _ in scope["headers"]]
|
||||
|
||||
try:
|
||||
with (
|
||||
patch.object( # test-quality-ok: the transport registry is a module-level singleton; the suite's only seam
|
||||
session_manager_stateful, "_server_instances", live_transports
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_auth_contexts, contexts, clear=True
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_client_info, {}, clear=True
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_auth_context_last_seen, {}, clear=True
|
||||
),
|
||||
):
|
||||
with patch.object( # test-quality-ok: termination stamps the tombstone from the clock directly; no injectable now
|
||||
mcp_server.time, "monotonic", return_value=1000.0
|
||||
):
|
||||
closed = await mcp_server.terminate_mcp_gateway_sessions(user_id="alice")
|
||||
assert closed.terminated_sessions == 2
|
||||
|
||||
for elapsed in (idle_timeout - 1, 2 * idle_timeout - 2, 3 * idle_timeout - 3):
|
||||
assert await replay(retrying_id, 1000.0 + elapsed) == (True, [b"content-type", b"mcp-session-id"])
|
||||
|
||||
await mcp_server._purge_expired_stateful_session_auth_contexts(now=1000.0 + idle_timeout)
|
||||
assert set(mcp_server._admin_terminated_session_ids) == {retrying_id}
|
||||
|
||||
assert await replay(silent_id, 1000.0 + idle_timeout) == (False, [b"content-type"])
|
||||
assert await replay(retrying_id, 1000.0 + 4 * idle_timeout) == (False, [b"content-type"])
|
||||
assert mcp_server._admin_terminated_session_ids == {}
|
||||
finally:
|
||||
mcp_server._admin_terminated_session_ids.clear()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_initialize_request_with_existing_session_tracks_new_session():
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -395,6 +395,34 @@ async def test_invalidate_clears_every_identity_for_a_server():
|
|||
assert mock_client.post.call_count == 3
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_per_user_token_delete_evicts_locally_and_broadcasts_to_peer_workers():
|
||||
"""Revoking a user's OAuth token must not leave peer workers serving it from their in-memory layer."""
|
||||
from litellm.proxy import proxy_server
|
||||
from litellm.proxy._experimental.mcp_server.oauth2_token_cache import MCPPerUserTokenCache
|
||||
from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache
|
||||
|
||||
local_cache = UserApiKeyCache()
|
||||
publish = AsyncMock()
|
||||
token_cache = MCPPerUserTokenCache()
|
||||
key = token_cache._cache_key("mallory", "srv-oauth") # pyright: ignore[reportPrivateUsage] # asserting the broadcast names the stored key
|
||||
local_cache.in_memory_cache.set_cache(key, "encrypted-token")
|
||||
|
||||
with (
|
||||
patch.object( # test-quality-ok: the token cache reads the module-level user_api_key_cache singleton; the suite's only seam
|
||||
proxy_server, "user_api_key_cache", local_cache
|
||||
),
|
||||
patch( # test-quality-ok: the redis publisher is module-level; asserting the broadcast without a redis
|
||||
"litellm.proxy.common_utils.auth_cache_invalidation_pubsub.publish_auth_cache_invalidation",
|
||||
new=publish,
|
||||
),
|
||||
):
|
||||
await token_cache.delete("mallory", "srv-oauth")
|
||||
|
||||
assert local_cache.in_memory_cache.get_cache(key) is None
|
||||
publish.assert_awaited_once_with(cache_key=key)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_m2m_mint_uses_admin_entered_token_url_when_issuer_yield_empties_resolved():
|
||||
"""A pinned issuer empties the resolved token_url while configured_token_url keeps the
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -57,6 +57,21 @@ def test_xff_honored_from_trusted_peer():
|
|||
assert via_proxy is True
|
||||
|
||||
|
||||
def test_ipv4_mapped_peer_and_hop_match_ipv4_trusted_ranges():
|
||||
request = make_request(headers={"x-forwarded-for": "203.0.113.9, ::ffff:10.0.0.5"}, client=("::ffff:10.0.0.1", 1))
|
||||
ip, via_proxy = resolve_client_ip(request, TRUSTED)
|
||||
assert ip == "203.0.113.9"
|
||||
assert via_proxy is True
|
||||
|
||||
|
||||
def test_ipv4_mapped_peer_still_matches_mapped_notation_trusted_range():
|
||||
config = TrustedProxyConfig(use_forwarded_for=True, trusted_proxy_cidrs=["::ffff:10.0.0.0/104"])
|
||||
request = make_request(headers={"x-forwarded-for": "203.0.113.9"}, client=("::ffff:10.0.0.1", 1))
|
||||
ip, via_proxy = resolve_client_ip(request, config)
|
||||
assert ip == "203.0.113.9"
|
||||
assert via_proxy is True
|
||||
|
||||
|
||||
def test_spoofed_xff_from_untrusted_peer_is_ignored():
|
||||
request = make_request(
|
||||
headers={"x-forwarded-for": "203.0.113.9"}, client=("8.8.8.8", 1)
|
||||
|
|
|
|||
|
|
@ -3967,3 +3967,74 @@ def test_auto_router_session_read_grant_rejects_other_methods_paths_and_scopes(
|
|||
RouteChecks.should_call_route(route, valid_token, request)
|
||||
|
||||
assert error.value.status_code == 403
|
||||
|
||||
|
||||
@pytest.mark.parametrize("route", ["/key/generate", "/key/update"])
|
||||
def test_team_service_account_key_allowed_key_management_routes(route):
|
||||
"""A service account key (user_id=None, team_id set, metadata.service_account_id)
|
||||
can reach key-management routes; team scoping is enforced in the handlers."""
|
||||
valid_token = UserAPIKeyAuth(
|
||||
api_key="sk",
|
||||
team_id="t1",
|
||||
user_id=None,
|
||||
metadata={"service_account_id": "ci"},
|
||||
)
|
||||
request = MagicMock(spec=Request)
|
||||
request.query_params = {}
|
||||
|
||||
result = RouteChecks.non_proxy_admin_allowed_routes_check(
|
||||
user_obj=None,
|
||||
_user_role=None,
|
||||
route=route,
|
||||
request=request,
|
||||
valid_token=valid_token,
|
||||
request_data={},
|
||||
)
|
||||
assert result is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("route", ["/team/new", "/spend/logs", "/key/delete", "/key/regenerate"])
|
||||
def test_team_service_account_key_rejected_outside_generate_and_update(route):
|
||||
"""The service account carve-out covers only /key/generate and /key/update; other
|
||||
key-management routes lack team scoping for a userless caller and stay denied."""
|
||||
valid_token = UserAPIKeyAuth(
|
||||
api_key="sk",
|
||||
team_id="t1",
|
||||
user_id=None,
|
||||
metadata={"service_account_id": "ci"},
|
||||
)
|
||||
request = MagicMock(spec=Request)
|
||||
request.query_params = {}
|
||||
|
||||
with pytest.raises(Exception, match="Only proxy admin can be used to generate, delete, update"):
|
||||
RouteChecks.non_proxy_admin_allowed_routes_check(
|
||||
user_obj=None,
|
||||
_user_role=None,
|
||||
route=route,
|
||||
request=request,
|
||||
valid_token=valid_token,
|
||||
request_data={},
|
||||
)
|
||||
|
||||
|
||||
def test_team_key_without_service_account_marker_still_rejected():
|
||||
"""A team key without metadata.service_account_id is not a service account
|
||||
and still cannot reach key-management routes."""
|
||||
valid_token = UserAPIKeyAuth(
|
||||
api_key="sk",
|
||||
team_id="t1",
|
||||
user_id=None,
|
||||
metadata={},
|
||||
)
|
||||
request = MagicMock(spec=Request)
|
||||
request.query_params = {}
|
||||
|
||||
with pytest.raises(Exception, match="Only proxy admin can be used to generate, delete, update"):
|
||||
RouteChecks.non_proxy_admin_allowed_routes_check(
|
||||
user_obj=None,
|
||||
_user_role=None,
|
||||
route="/key/generate",
|
||||
request=request,
|
||||
valid_token=valid_token,
|
||||
request_data={},
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -168,6 +169,54 @@ def test_settings_store_refuses_a_runtime_write_to_a_config_owned_key() -> None:
|
|||
assert store.source("max_parallel_requests") == "config"
|
||||
|
||||
|
||||
@pytest.mark.timeout(10)
|
||||
def test_settings_store_clear_removes_every_key_the_config_file_does_not_own() -> None:
|
||||
store: Final = SettingsStore("general_settings")
|
||||
store.load_yaml({"master_key": "os.environ/MASTER_KEY"})
|
||||
store.apply_db_row("general_settings", {"max_parallel_requests": 3, "alerting": ["slack"]})
|
||||
store.apply_runtime_values({"master_key": "sk-resolved", "alerting": ["slack"]})
|
||||
store["allow_requests_on_db_unavailable"] = True
|
||||
del store["alerting"]
|
||||
|
||||
store.clear()
|
||||
|
||||
assert dict(store) == {"master_key": "sk-resolved"}
|
||||
assert "alerting" not in store
|
||||
with pytest.raises(KeyError):
|
||||
store["max_parallel_requests"]
|
||||
|
||||
|
||||
@pytest.mark.timeout(10)
|
||||
def test_settings_store_clear_then_refill_matches_a_plain_dict() -> None:
|
||||
refilled: Final[dict[str, JsonValue]] = {"alerting": ["email"], "max_parallel_requests": 11}
|
||||
store: Final = SettingsStore("general_settings")
|
||||
store.update({"max_parallel_requests": 3, "alerting": ["slack"]})
|
||||
|
||||
store.clear()
|
||||
store.update(refilled)
|
||||
|
||||
assert dict(store) == refilled
|
||||
assert tuple(store) == tuple(refilled)
|
||||
assert len(store) == len(refilled)
|
||||
|
||||
|
||||
@pytest.mark.timeout(10)
|
||||
@pytest.mark.parametrize("clear", (False, True))
|
||||
def test_settings_store_survives_a_patch_dict_round_trip_when_the_config_file_owns_a_key(clear: bool) -> None:
|
||||
store: Final = SettingsStore("general_settings")
|
||||
store.load_yaml({"master_key": "os.environ/MASTER_KEY"})
|
||||
store.apply_db_row("general_settings", {"max_parallel_requests": 3})
|
||||
store.apply_runtime_values({"master_key": "sk-resolved", "max_parallel_requests": 3})
|
||||
before: Final = dict(store)
|
||||
|
||||
with patch.dict(store, {"allow_requests_on_db_unavailable": True}, clear=clear):
|
||||
assert store["allow_requests_on_db_unavailable"] is True
|
||||
assert store["master_key"] == "sk-resolved"
|
||||
assert ("max_parallel_requests" in store) is not clear
|
||||
|
||||
assert dict(store) == before
|
||||
|
||||
|
||||
def test_settings_store_reports_the_config_owned_keys_a_write_would_change() -> None:
|
||||
store: Final = SettingsStore("general_settings")
|
||||
store.load_yaml({"max_parallel_requests": 3, "ui_access_mode": "admin_only"})
|
||||
|
|
|
|||
|
|
@ -0,0 +1,409 @@
|
|||
"""
|
||||
Unit tests for the TypeSafe (Jev) compaction guardrail.
|
||||
|
||||
Tests cover:
|
||||
- exchanges scored below relevance_threshold have their tool rows blanked while
|
||||
assistant tool-call rows and kept exchanges pass through verbatim, without
|
||||
mutating the caller's message list
|
||||
- protected rows (system, last user, and the last tool exchange via the
|
||||
last-assistant rule) are never sent to Jev even when long
|
||||
- exchanges under min_chars_to_evaluate are skipped
|
||||
- request shape: POST {api_base}/v1/systemone with Bearer auth, one noul
|
||||
question per candidate keyed e<i>, task = last user text, results truncated
|
||||
to max_result_chars_in_state
|
||||
- identity return when there are no candidates or nothing is dropped
|
||||
- fail_open forwards uncompacted on service failure; fail_closed raises
|
||||
- response input_type passthrough and initialize_guardrail wiring
|
||||
"""
|
||||
|
||||
from unittest.mock import AsyncMock, MagicMock, PropertyMock
|
||||
|
||||
import pytest
|
||||
from fastapi import HTTPException
|
||||
|
||||
from litellm.proxy.guardrails.guardrail_hooks.typesafe import (
|
||||
TypeSafeGuardrail,
|
||||
guardrail_class_registry,
|
||||
guardrail_initializer_registry,
|
||||
initialize_guardrail,
|
||||
)
|
||||
from litellm.proxy.guardrails.guardrail_hooks.typesafe.typesafe import DROPPED_RESULT_TEXT
|
||||
from litellm.types.guardrails import SupportedGuardrailIntegrations
|
||||
from litellm.types.utils import GenericGuardrailAPIInputs
|
||||
|
||||
FAKE_API_BASE = "https://typesafe.example.com"
|
||||
FAKE_API_KEY = "ts_test-key"
|
||||
|
||||
SYSTEM_TEXT = "You are a research assistant."
|
||||
USER_TEXT = "Which 2026 EV has the longest range?"
|
||||
TOOL_OUTPUT_LONG = "Result: EV range comparison. " * 40
|
||||
TOOL_OUTPUT_SHORT = "short"
|
||||
|
||||
|
||||
def _exchange(call_id: str, tool_text: str, name: str = "web_search") -> list[dict[str, object]]:
|
||||
return [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": call_id,
|
||||
"type": "function",
|
||||
"function": {"name": name, "arguments": '{"query": "ev"}'},
|
||||
}
|
||||
],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": call_id, "name": name, "content": tool_text},
|
||||
]
|
||||
|
||||
|
||||
def _messages(*, tail: list[dict[str, object]] | None = None) -> list[dict[str, object]]:
|
||||
base = [
|
||||
{"role": "system", "content": SYSTEM_TEXT},
|
||||
{"role": "user", "content": USER_TEXT},
|
||||
]
|
||||
return base + (tail or [])
|
||||
|
||||
|
||||
def _make_guardrail(
|
||||
handler: MagicMock | None = None,
|
||||
*,
|
||||
max_result_chars_in_state: int | None = None,
|
||||
unreachable_fallback: str | None = None,
|
||||
) -> TypeSafeGuardrail:
|
||||
return TypeSafeGuardrail(
|
||||
api_base=FAKE_API_BASE,
|
||||
api_key=FAKE_API_KEY,
|
||||
guardrail_name="typesafe",
|
||||
default_on=True,
|
||||
async_handler=handler or _make_handler({"e0": 0.9}),
|
||||
max_result_chars_in_state=max_result_chars_in_state,
|
||||
unreachable_fallback=unreachable_fallback,
|
||||
)
|
||||
|
||||
|
||||
def _make_handler(answers: dict[str, float], status: int = 200) -> MagicMock:
|
||||
response = MagicMock()
|
||||
response.status_code = status
|
||||
response.json.return_value = {
|
||||
"model": "jev-1.13.0",
|
||||
"answers": {qid: {"type": "noul", "noul": score} for qid, score in answers.items()},
|
||||
"usage": {"input_tokens": 10, "output_tokens": 1},
|
||||
}
|
||||
response.text = ""
|
||||
handler = MagicMock()
|
||||
handler.post = AsyncMock(return_value=response)
|
||||
return handler
|
||||
|
||||
|
||||
def _inputs(messages: list[dict[str, object]]) -> GenericGuardrailAPIInputs:
|
||||
return GenericGuardrailAPIInputs(structured_messages=messages)
|
||||
|
||||
|
||||
async def _apply(
|
||||
guardrail: TypeSafeGuardrail, messages: list[dict[str, object]], input_type: str = "request"
|
||||
) -> GenericGuardrailAPIInputs:
|
||||
return await guardrail.apply_guardrail(
|
||||
inputs=_inputs(messages),
|
||||
request_data={},
|
||||
input_type=input_type, # pyright: ignore[reportArgumentType] # test uses the same literal domain
|
||||
logging_obj=None,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_low_noul_exchange_blanked_high_kept_and_input_not_mutated():
|
||||
handler = _make_handler({"e0": 0.1, "e1": 0.95})
|
||||
guardrail = _make_guardrail(handler)
|
||||
messages = _messages(
|
||||
tail=[
|
||||
*_exchange("call_1", TOOL_OUTPUT_LONG),
|
||||
*_exchange("call_2", TOOL_OUTPUT_LONG),
|
||||
{"role": "assistant", "content": "still thinking"},
|
||||
]
|
||||
)
|
||||
snapshot = [dict(m) for m in messages]
|
||||
|
||||
result = await _apply(guardrail, messages)
|
||||
out = result["structured_messages"]
|
||||
|
||||
assert out[3]["content"] == DROPPED_RESULT_TEXT
|
||||
assert out[3]["tool_call_id"] == "call_1"
|
||||
assert out[3]["role"] == "tool"
|
||||
assert out[5]["content"] == TOOL_OUTPUT_LONG
|
||||
assert out[2] == messages[2]
|
||||
assert out[4] == messages[4]
|
||||
assert out[6]["content"] == "still thinking"
|
||||
assert messages == snapshot
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_last_exchange_and_protected_rows_never_evaluated():
|
||||
handler = _make_handler({"e0": 0.05})
|
||||
guardrail = _make_guardrail(handler)
|
||||
messages = _messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), *_exchange("call_2", TOOL_OUTPUT_LONG)])
|
||||
|
||||
result = await _apply(guardrail, messages)
|
||||
|
||||
payload = handler.post.call_args.kwargs["json"]
|
||||
assert list(payload["questions"]) == ["e0"]
|
||||
assert list(payload["state"]["tool_exchanges"]) == ["e0"]
|
||||
assert payload["state"]["task"] == USER_TEXT
|
||||
assert payload["state"]["system"] == SYSTEM_TEXT
|
||||
out = result["structured_messages"]
|
||||
assert out[3]["content"] == DROPPED_RESULT_TEXT
|
||||
assert out[5]["content"] == TOOL_OUTPUT_LONG
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_short_exchange_not_sent():
|
||||
handler = _make_handler({"e0": 0.9})
|
||||
guardrail = _make_guardrail(handler)
|
||||
messages = _messages(
|
||||
tail=[
|
||||
*_exchange("call_1", TOOL_OUTPUT_SHORT),
|
||||
*_exchange("call_2", TOOL_OUTPUT_LONG),
|
||||
{"role": "assistant", "content": "done"},
|
||||
]
|
||||
)
|
||||
result = await _apply(guardrail, messages)
|
||||
payload = handler.post.call_args.kwargs["json"]
|
||||
assert list(payload["questions"]) == ["e0"]
|
||||
exchange = payload["state"]["tool_exchanges"]["e0"]
|
||||
assert exchange["result"] == TOOL_OUTPUT_LONG
|
||||
assert result is not None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_request_body_shape_and_truncation():
|
||||
handler = _make_handler({"e0": 0.9})
|
||||
guardrail = _make_guardrail(handler, max_result_chars_in_state=50)
|
||||
messages = _messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "done"}])
|
||||
await _apply(guardrail, messages)
|
||||
|
||||
kwargs = handler.post.call_args.kwargs
|
||||
assert kwargs["url"].endswith("/v1/systemone")
|
||||
assert kwargs["url"].startswith(FAKE_API_BASE)
|
||||
assert kwargs["headers"]["Authorization"] == f"Bearer {FAKE_API_KEY}"
|
||||
assert kwargs["headers"]["Content-Type"] == "application/json"
|
||||
payload = kwargs["json"]
|
||||
assert payload["model"] == "jev-latest"
|
||||
assert list(payload["questions"]) == ["e0"]
|
||||
assert payload["questions"]["e0"]["type"] == "noul"
|
||||
assert "e0" in payload["questions"]["e0"]["instructions"]
|
||||
assert payload["state"]["task"] == USER_TEXT
|
||||
exchange = payload["state"]["tool_exchanges"]["e0"]
|
||||
assert len(exchange["result"]) == 50
|
||||
assert exchange["result"].startswith(TOOL_OUTPUT_LONG[:10])
|
||||
assert exchange["result"].endswith(TOOL_OUTPUT_LONG[-11:])
|
||||
assert list(exchange["tool_calls"]) == [{"name": "web_search", "arguments": '{"query": "ev"}'}]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_no_candidates_returns_identity_and_skips_http():
|
||||
handler = _make_handler({})
|
||||
guardrail = _make_guardrail(handler)
|
||||
inputs = _inputs(_messages(tail=[{"role": "assistant", "content": "plain answer"}]))
|
||||
result = await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="request", logging_obj=None)
|
||||
assert result is inputs
|
||||
handler.post.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_all_above_threshold_returns_identity():
|
||||
handler = _make_handler({"e0": 0.9})
|
||||
guardrail = _make_guardrail(handler)
|
||||
inputs = _inputs(_messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "x"}]))
|
||||
result = await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="request", logging_obj=None)
|
||||
assert result is inputs
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_fail_open_returns_inputs_on_exception():
|
||||
handler = MagicMock()
|
||||
handler.post = AsyncMock(side_effect=Exception("connection refused"))
|
||||
guardrail = _make_guardrail(handler, unreachable_fallback="fail_open")
|
||||
inputs = _inputs(_messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "x"}]))
|
||||
result = await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="request", logging_obj=None)
|
||||
assert result is inputs
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_fail_closed_raises_http_exception():
|
||||
handler = MagicMock()
|
||||
handler.post = AsyncMock(side_effect=Exception("connection refused"))
|
||||
guardrail = _make_guardrail(handler, unreachable_fallback="fail_closed")
|
||||
inputs = _inputs(_messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "x"}]))
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="request", logging_obj=None)
|
||||
assert exc_info.value.status_code == 502
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_fail_open_on_non_2xx():
|
||||
handler = _make_handler({"e0": 0.9}, status=500)
|
||||
guardrail = _make_guardrail(handler)
|
||||
inputs = _inputs(_messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "x"}]))
|
||||
result = await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="request", logging_obj=None)
|
||||
assert result is inputs
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_response_input_type_passthrough():
|
||||
handler = _make_handler({"e0": 0.05})
|
||||
guardrail = _make_guardrail(handler)
|
||||
inputs = _inputs(_messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG)]))
|
||||
result = await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="response", logging_obj=None)
|
||||
assert result is inputs
|
||||
handler.post.assert_not_called()
|
||||
|
||||
|
||||
def test_initialize_guardrail_applies_optional_params_and_registry_keys():
|
||||
from litellm.types.guardrails import LitellmParams
|
||||
|
||||
litellm_params = LitellmParams(
|
||||
guardrail="typesafe",
|
||||
mode="pre_call",
|
||||
api_key=FAKE_API_KEY,
|
||||
api_base=FAKE_API_BASE,
|
||||
optional_params={
|
||||
"relevance_threshold": 0.5,
|
||||
"min_chars_to_evaluate": 10,
|
||||
"max_result_chars_in_state": 100,
|
||||
},
|
||||
)
|
||||
callback = initialize_guardrail(litellm_params, {"guardrail_name": "jev-compaction"})
|
||||
assert isinstance(callback, TypeSafeGuardrail)
|
||||
assert callback.relevance_threshold == 0.5
|
||||
assert callback.min_chars_to_evaluate == 10
|
||||
assert callback.max_result_chars_in_state == 100
|
||||
assert callback.unreachable_fallback == "fail_open"
|
||||
assert guardrail_initializer_registry[SupportedGuardrailIntegrations.TYPESAFE.value] is initialize_guardrail
|
||||
assert guardrail_class_registry[SupportedGuardrailIntegrations.TYPESAFE.value] is TypeSafeGuardrail
|
||||
|
||||
|
||||
def test_missing_api_key_raises(monkeypatch):
|
||||
monkeypatch.delenv("TYPESAFE_API_KEY", raising=False)
|
||||
with pytest.raises(ValueError, match="requires an API key"):
|
||||
TypeSafeGuardrail(api_key=None)
|
||||
|
||||
|
||||
def test_get_config_model_and_ui_name():
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.typesafe import (
|
||||
TypeSafeGuardrailConfigModel,
|
||||
)
|
||||
|
||||
assert TypeSafeGuardrail.get_config_model() is TypeSafeGuardrailConfigModel
|
||||
assert TypeSafeGuardrailConfigModel.ui_friendly_name() == "TypeSafe (Jev) Compaction"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_non_list_and_non_dict_messages_return_identity():
|
||||
guardrail = _make_guardrail()
|
||||
not_a_list = GenericGuardrailAPIInputs(structured_messages={"role": "user"})
|
||||
assert (
|
||||
await guardrail.apply_guardrail(inputs=not_a_list, request_data={}, input_type="request", logging_obj=None)
|
||||
is not_a_list
|
||||
)
|
||||
with_bad_row = _inputs(_messages(tail=[["not", "a", "dict"]]))
|
||||
assert (
|
||||
await guardrail.apply_guardrail(inputs=with_bad_row, request_data={}, input_type="request", logging_obj=None)
|
||||
is with_bad_row
|
||||
)
|
||||
|
||||
|
||||
def test_odd_tool_call_shapes_yield_no_entries():
|
||||
from litellm.proxy.guardrails.guardrail_hooks.typesafe.typesafe import _tool_call_entries
|
||||
|
||||
assert _tool_call_entries({"tool_calls": "not-a-list"}) == ()
|
||||
assert _tool_call_entries({"tool_calls": None}) == ()
|
||||
assert list(_tool_call_entries({"tool_calls": [42]})) == []
|
||||
entries = _tool_call_entries({"tool_calls": [{"function": {"name": "web_search", "arguments": "{}"}}]})
|
||||
assert list(entries) == [{"name": "web_search", "arguments": "{}"}]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_short_max_chars_uses_prefix_slice():
|
||||
handler = _make_handler({"e0": 0.9})
|
||||
guardrail = _make_guardrail(handler, max_result_chars_in_state=5)
|
||||
await _apply(
|
||||
guardrail, _messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "x"}])
|
||||
)
|
||||
result = handler.post.call_args.kwargs["json"]["state"]["tool_exchanges"]["e0"]["result"]
|
||||
assert result == TOOL_OUTPUT_LONG[:5]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_unreadable_json_body_fails_open():
|
||||
handler = MagicMock()
|
||||
response = MagicMock()
|
||||
response.status_code = 200
|
||||
response.text = "not json"
|
||||
response.json.side_effect = ValueError("no json")
|
||||
handler.post = AsyncMock(return_value=response)
|
||||
guardrail = _make_guardrail(handler)
|
||||
inputs = _inputs(_messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "x"}]))
|
||||
result = await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="request", logging_obj=None)
|
||||
assert result is inputs
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_malformed_answers_shape_fails_open():
|
||||
handler = MagicMock()
|
||||
response = MagicMock()
|
||||
response.status_code = 200
|
||||
response.text = '{"answers": "oops"}'
|
||||
response.json.return_value = {"answers": "oops"}
|
||||
handler.post = AsyncMock(return_value=response)
|
||||
guardrail = _make_guardrail(handler)
|
||||
inputs = _inputs(_messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "x"}]))
|
||||
result = await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="request", logging_obj=None)
|
||||
assert result is inputs
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_http_status_error_includes_status_and_undecodable_body():
|
||||
import httpx
|
||||
|
||||
response = MagicMock()
|
||||
response.status_code = 503
|
||||
type(response).text = PropertyMock(side_effect=httpx.DecodingError("bad codec"))
|
||||
handler = MagicMock()
|
||||
handler.post = AsyncMock(side_effect=httpx.HTTPStatusError("unavailable", request=MagicMock(), response=response))
|
||||
guardrail = _make_guardrail(handler)
|
||||
inputs = _inputs(_messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "x"}]))
|
||||
result = await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="request", logging_obj=None)
|
||||
assert result is inputs
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cancelled_jev_call_propagates():
|
||||
import asyncio
|
||||
|
||||
handler = MagicMock()
|
||||
handler.post = AsyncMock(side_effect=asyncio.CancelledError())
|
||||
guardrail = _make_guardrail(handler)
|
||||
inputs = _inputs(_messages(tail=[*_exchange("call_1", TOOL_OUTPUT_LONG), {"role": "assistant", "content": "x"}]))
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await guardrail.apply_guardrail(inputs=inputs, request_data={}, input_type="request", logging_obj=None)
|
||||
|
||||
|
||||
def test_optional_params_defaults_and_event_hook_coercion():
|
||||
from litellm.proxy.guardrails.guardrail_hooks.typesafe import _coerce_event_hook, _optional_params
|
||||
from litellm.types.guardrails import GuardrailEventHooks, LitellmParams
|
||||
|
||||
assert _coerce_event_hook("pre_call") is GuardrailEventHooks.pre_call
|
||||
assert _coerce_event_hook(["pre_call", "post_call"]) == [
|
||||
GuardrailEventHooks.pre_call,
|
||||
GuardrailEventHooks.post_call,
|
||||
]
|
||||
litellm_params = LitellmParams(guardrail="typesafe", mode="pre_call", api_key=FAKE_API_KEY)
|
||||
params = _optional_params(litellm_params)
|
||||
assert params.relevance_threshold is None
|
||||
|
||||
|
||||
def test_typesafe_initializer_discoverable_via_hook_registries():
|
||||
from litellm.proxy.guardrails.guardrail_registry import get_guardrail_initializer_from_hooks
|
||||
|
||||
initializers = get_guardrail_initializer_from_hooks()
|
||||
assert initializers["typesafe"] is initialize_guardrail
|
||||
|
|
@ -422,7 +422,7 @@ def test_apply_patch_ops_invalid_entitlements_value_raises_400():
|
|||
patch_ops = SCIMPatchOp(
|
||||
Operations=[
|
||||
SCIMPatchOperation(
|
||||
op="replace", path="entitlements", value=[{"display": "no value"}]
|
||||
op="replace", path="entitlements", value=[42]
|
||||
)
|
||||
]
|
||||
)
|
||||
|
|
@ -433,6 +433,22 @@ def test_apply_patch_ops_invalid_entitlements_value_raises_400():
|
|||
assert exc_info.value.status_code == 400
|
||||
|
||||
|
||||
def test_apply_patch_ops_replace_entitlements_without_value_member_is_stored_as_sent():
|
||||
patch_ops = SCIMPatchOp(
|
||||
Operations=[
|
||||
SCIMPatchOperation(
|
||||
op="replace", path="entitlements", value=[{"groups": ["S0506MKA55L"]}]
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
update_data, _ = _apply_patch_ops(
|
||||
existing_user=_user_with_metadata({}), patch_ops=patch_ops
|
||||
)
|
||||
|
||||
assert update_data["metadata"]["scim_entitlements"] == [{"groups": ["S0506MKA55L"]}]
|
||||
|
||||
|
||||
def test_apply_patch_ops_add_without_value_raises_400_naming_value_member():
|
||||
patch_ops = SCIMPatchOp(
|
||||
Operations=[SCIMPatchOperation(op="add", path="entitlements")]
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import json
|
||||
import logging
|
||||
import time
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
|
|
@ -1303,6 +1304,75 @@ async def test_update_user_success(mocker):
|
|||
assert call_args[1]["data"]["teams"] == ["new-team"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_update_user_put_with_valueless_entitlements_deactivates_user(scim_test_client, mocker):
|
||||
existing_user = mocker.MagicMock()
|
||||
existing_user.teams = []
|
||||
existing_user.metadata = {"scim_active": True}
|
||||
|
||||
updated_user = {
|
||||
"user_id": "suspend-me",
|
||||
"user_email": "suspend@example.com",
|
||||
"user_alias": None,
|
||||
"teams": [],
|
||||
"metadata": "{}",
|
||||
}
|
||||
response_scim_user = SCIMUser(
|
||||
schemas=["urn:ietf:params:scim:schemas:core:2.0:User"],
|
||||
id="suspend-me",
|
||||
userName="suspend-me",
|
||||
active=False,
|
||||
)
|
||||
|
||||
mock_prisma_client = mocker.MagicMock()
|
||||
mock_prisma_client.db = mocker.MagicMock()
|
||||
mock_prisma_client.db.litellm_usertable = mocker.MagicMock()
|
||||
mock_prisma_client.db.litellm_usertable.update = AsyncMock(return_value=updated_user)
|
||||
|
||||
mocker.patch( # test-quality-ok: endpoint collaborators are module-level, not injectable
|
||||
"litellm.proxy.management_endpoints.scim.scim_v2._get_prisma_client_or_raise_exception",
|
||||
AsyncMock(return_value=mock_prisma_client),
|
||||
)
|
||||
mocker.patch( # test-quality-ok: endpoint collaborators are module-level, not injectable
|
||||
"litellm.proxy.management_endpoints.scim.scim_v2._check_user_exists",
|
||||
AsyncMock(return_value=existing_user),
|
||||
)
|
||||
mocker.patch( # test-quality-ok: endpoint collaborators are module-level, not injectable
|
||||
"litellm.proxy.management_endpoints.scim.scim_v2._handle_team_membership_changes",
|
||||
AsyncMock(),
|
||||
)
|
||||
set_keys_blocked_mock = mocker.patch( # test-quality-ok: endpoint collaborators are module-level, not injectable
|
||||
"litellm.proxy.management_endpoints.scim.scim_v2._set_user_keys_blocked",
|
||||
AsyncMock(return_value=1),
|
||||
)
|
||||
mocker.patch( # test-quality-ok: endpoint collaborators are module-level, not injectable
|
||||
"litellm.proxy.management_endpoints.scim.scim_v2.ScimTransformations.transform_litellm_user_to_scim_user",
|
||||
AsyncMock(return_value=response_scim_user),
|
||||
)
|
||||
|
||||
async with scim_test_client as client:
|
||||
response = await client.put(
|
||||
"/scim/v2/Users/suspend-me",
|
||||
json={
|
||||
"schemas": ["urn:ietf:params:scim:schemas:core:2.0:User"],
|
||||
"userName": "suspend-me",
|
||||
"emails": [{"value": "suspend@example.com", "primary": True}],
|
||||
"entitlements": [{"groups": ["S0506MKA55L", "S0506MKA56M"]}],
|
||||
"roles": [{"display": "Viewer"}],
|
||||
"active": False,
|
||||
},
|
||||
)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
assert response.json()["active"] is False
|
||||
|
||||
written_metadata = json.loads(mock_prisma_client.db.litellm_usertable.update.call_args.kwargs["data"]["metadata"])
|
||||
assert written_metadata["scim_active"] is False
|
||||
assert written_metadata["scim_entitlements"] == [{"groups": ["S0506MKA55L", "S0506MKA56M"]}]
|
||||
assert written_metadata["scim_roles"] == [{"display": "Viewer"}]
|
||||
set_keys_blocked_mock.assert_awaited_once_with(user_id="suspend-me", blocked=True)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("groups", [None, []], ids=["groups-omitted", "groups-empty"])
|
||||
async def test_update_user_without_groups_preserves_memberships_and_role(mocker, monkeypatch, groups):
|
||||
|
|
|
|||
|
|
@ -1,13 +1,19 @@
|
|||
import re
|
||||
from collections.abc import Sequence
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from types import SimpleNamespace
|
||||
from typing import Final
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import psycopg
|
||||
import pytest
|
||||
from psycopg.rows import dict_row
|
||||
from pytest_postgresql import factories
|
||||
|
||||
from litellm.proxy.spend_tracking.ptu_feature_flag import PTU_COST_ATTRIBUTION_ENV_VAR
|
||||
|
||||
|
||||
from litellm.constants import PTU_SENTINEL_API_KEY, USAGE_TOP_API_KEYS_LIMIT
|
||||
from litellm.proxy.management_endpoints.common_daily_activity import (
|
||||
_adjust_dates_for_timezone,
|
||||
_build_aggregated_sql_query,
|
||||
|
|
@ -169,6 +175,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": "/v1/chat/completions",
|
||||
"api_key": None,
|
||||
"group_level": 62,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 15.0,
|
||||
"prompt_tokens": 150,
|
||||
"completion_tokens": 75,
|
||||
|
|
@ -181,31 +188,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": "/v1/embeddings",
|
||||
"api_key": None,
|
||||
"group_level": 62,
|
||||
"spend": 3.0,
|
||||
"prompt_tokens": 30,
|
||||
"completion_tokens": 0,
|
||||
"api_requests": 1,
|
||||
"successful_requests": 1,
|
||||
},
|
||||
# (date, endpoint, api_key) — populates the per-key sub-bucket
|
||||
{
|
||||
**base,
|
||||
"date": "2024-01-01",
|
||||
"endpoint": "/v1/chat/completions",
|
||||
"api_key": "key-1",
|
||||
"group_level": 30,
|
||||
"spend": 15.0,
|
||||
"prompt_tokens": 150,
|
||||
"completion_tokens": 75,
|
||||
"api_requests": 2,
|
||||
"successful_requests": 2,
|
||||
},
|
||||
{
|
||||
**base,
|
||||
"date": "2024-01-01",
|
||||
"endpoint": "/v1/embeddings",
|
||||
"api_key": "key-2",
|
||||
"group_level": 30,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 3.0,
|
||||
"prompt_tokens": 30,
|
||||
"completion_tokens": 0,
|
||||
|
|
@ -219,6 +202,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": None,
|
||||
"api_key": None,
|
||||
"group_level": 63,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 18.0,
|
||||
"prompt_tokens": 180,
|
||||
"completion_tokens": 75,
|
||||
|
|
@ -232,12 +216,40 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown():
|
|||
"endpoint": None,
|
||||
"api_key": None,
|
||||
"group_level": 127,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 18.0,
|
||||
"prompt_tokens": 180,
|
||||
"completion_tokens": 75,
|
||||
"api_requests": 3,
|
||||
"successful_requests": 3,
|
||||
},
|
||||
# (date, endpoint, api_key) — populates the per-key sub-bucket
|
||||
{
|
||||
**base,
|
||||
"date": "2024-01-01",
|
||||
"endpoint": "/v1/chat/completions",
|
||||
"api_key": "key-1",
|
||||
"group_level": 30,
|
||||
"distinct_api_keys": 2,
|
||||
"spend": 15.0,
|
||||
"prompt_tokens": 150,
|
||||
"completion_tokens": 75,
|
||||
"api_requests": 2,
|
||||
"successful_requests": 2,
|
||||
},
|
||||
{
|
||||
**base,
|
||||
"date": "2024-01-01",
|
||||
"endpoint": "/v1/embeddings",
|
||||
"api_key": "key-2",
|
||||
"group_level": 30,
|
||||
"distinct_api_keys": 2,
|
||||
"spend": 3.0,
|
||||
"prompt_tokens": 30,
|
||||
"completion_tokens": 0,
|
||||
"api_requests": 1,
|
||||
"successful_requests": 1,
|
||||
},
|
||||
]
|
||||
|
||||
mock_prisma.db.query_raw = AsyncMock(return_value=mock_rows)
|
||||
|
|
@ -474,9 +486,7 @@ async def test_get_api_key_metadata_recovers_double_hashed_key_via_reverse_hash(
|
|||
return_value=[SimpleNamespace(user_id="alice", user_email="alice@example.com")]
|
||||
)
|
||||
mock_prisma.db.query_raw = AsyncMock(
|
||||
return_value=[
|
||||
{"digest": double_hashed, "key_alias": "batch-worker", "team_id": "team-1", "user_id": "alice"}
|
||||
]
|
||||
return_value=[{"digest": double_hashed, "key_alias": "batch-worker", "team_id": "team-1", "user_id": "alice"}]
|
||||
)
|
||||
|
||||
result = await get_api_key_metadata(
|
||||
|
|
@ -835,6 +845,7 @@ async def test_aggregated_activity_preserves_metadata_for_deleted_keys():
|
|||
"endpoint": "/v1/chat/completions",
|
||||
"api_key": None,
|
||||
"group_level": 62,
|
||||
"distinct_api_keys": None,
|
||||
"spend": 10.0,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 50,
|
||||
|
|
@ -847,6 +858,7 @@ async def test_aggregated_activity_preserves_metadata_for_deleted_keys():
|
|||
"endpoint": "/v1/chat/completions",
|
||||
"api_key": "deleted-key-hash",
|
||||
"group_level": 30,
|
||||
"distinct_api_keys": 1,
|
||||
"spend": 10.0,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 50,
|
||||
|
|
@ -1230,42 +1242,11 @@ class TestBuildAggregatedSqlQuery:
|
|||
"user-1",
|
||||
"bedrock/global.anthropic.claude-opus-4-8",
|
||||
"sk-test",
|
||||
PTU_SENTINEL_API_KEY,
|
||||
]
|
||||
assert "model = $4" in sql
|
||||
assert "api_key = $5" in sql
|
||||
|
||||
def test_model_group_rollups_fall_back_to_model_name(self):
|
||||
"""Aggregated model_groups rollups must fall back to model for group-less rows.
|
||||
|
||||
The (date, model_group) grouping level cannot recover the model column
|
||||
after the fact (it is rolled up), so the fallback has to happen in SQL;
|
||||
without it, group-less rows silently vanish from the model_groups
|
||||
breakdown that the usage UI now renders by default. Group-less rows are
|
||||
stored as empty strings, not NULL (spend_tracking_utils defaults
|
||||
model_group to ""), so a plain COALESCE is not enough: the fallback must
|
||||
be NULLIF-wrapped to catch both
|
||||
"""
|
||||
sql, _ = _build_aggregated_sql_query(
|
||||
table_name="litellm_dailyuserspend",
|
||||
entity_id_field="user_id",
|
||||
entity_id=None,
|
||||
start_date="2026-07-01",
|
||||
end_date="2026-07-01",
|
||||
model=None,
|
||||
api_key=None,
|
||||
)
|
||||
|
||||
normalized = " ".join(sql.split())
|
||||
fallback = "COALESCE(NULLIF(model_group, ''), model)"
|
||||
assert f"{fallback} AS model_group" in normalized
|
||||
assert (
|
||||
f"GROUPING(date, api_key, model, {fallback}, "
|
||||
"custom_llm_provider, mcp_namespaced_tool_name, endpoint) AS group_level" in normalized
|
||||
)
|
||||
assert f"(date, {fallback}), (date, {fallback}, api_key)," in normalized
|
||||
assert "(date, model_group)" not in normalized
|
||||
assert "COALESCE(model_group, model)" not in normalized
|
||||
|
||||
|
||||
class TestAggregatedEmptyEntityFilter:
|
||||
_BUILDERS: Final = (_build_aggregated_sql_query, _build_entity_rollup_sql_query)
|
||||
|
|
@ -1285,7 +1266,8 @@ class TestAggregatedEmptyEntityFilter:
|
|||
normalized = " ".join(sql.split())
|
||||
assert "IN ()" not in normalized
|
||||
assert '"team_id" IN' not in normalized
|
||||
assert params == ["2026-08-01", "2026-08-19"]
|
||||
sentinel_params = [PTU_SENTINEL_API_KEY] if build is _build_aggregated_sql_query else []
|
||||
assert params == ["2026-08-01", "2026-08-19", *sentinel_params]
|
||||
|
||||
@pytest.mark.parametrize("build", _BUILDERS)
|
||||
def test_empty_entity_list_matches_nothing_rather_than_everything(self, build):
|
||||
|
|
@ -1316,7 +1298,8 @@ class TestAggregatedEmptyEntityFilter:
|
|||
normalized = " ".join(sql.split())
|
||||
assert '"team_id" IN ($3, $4)' in normalized
|
||||
assert "FALSE" not in normalized
|
||||
assert params == ["2026-08-01", "2026-08-19", "team-alpha", "team-beta"]
|
||||
sentinel_params = [PTU_SENTINEL_API_KEY] if build is _build_aggregated_sql_query else []
|
||||
assert params == ["2026-08-01", "2026-08-19", "team-alpha", "team-beta", *sentinel_params]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -1341,6 +1324,7 @@ async def test_get_daily_activity_aggregated_empty_result_set():
|
|||
"mcp_namespaced_tool_name": None,
|
||||
"endpoint": None,
|
||||
"group_level": 127,
|
||||
"distinct_api_keys": None,
|
||||
"spend": None,
|
||||
"prompt_tokens": None,
|
||||
"completion_tokens": None,
|
||||
|
|
@ -1385,6 +1369,305 @@ async def test_get_daily_activity_aggregated_empty_result_set():
|
|||
assert result.metadata.total_compression_saved_tokens == 0
|
||||
|
||||
|
||||
_aggregated_postgresql_proc: Final = factories.postgresql_proc()
|
||||
_aggregated_postgresql: Final = factories.postgresql("_aggregated_postgresql_proc")
|
||||
|
||||
_DAILY_USER_SPEND_DDL: Final = """
|
||||
CREATE TABLE "LiteLLM_DailyUserSpend" (
|
||||
id TEXT PRIMARY KEY,
|
||||
user_id TEXT,
|
||||
date TEXT NOT NULL,
|
||||
api_key TEXT NOT NULL,
|
||||
model TEXT,
|
||||
model_group TEXT,
|
||||
custom_llm_provider TEXT,
|
||||
mcp_namespaced_tool_name TEXT,
|
||||
endpoint TEXT,
|
||||
prompt_tokens BIGINT DEFAULT 0,
|
||||
completion_tokens BIGINT DEFAULT 0,
|
||||
cache_read_input_tokens BIGINT DEFAULT 0,
|
||||
cache_creation_input_tokens BIGINT DEFAULT 0,
|
||||
compression_saved_tokens BIGINT DEFAULT 0,
|
||||
compression_savings_spend DOUBLE PRECISION DEFAULT 0,
|
||||
prompt_caching_savings_spend DOUBLE PRECISION DEFAULT 0,
|
||||
gateway_injected_caching_savings_spend DOUBLE PRECISION DEFAULT 0,
|
||||
autorouter_savings_spend DOUBLE PRECISION DEFAULT 0,
|
||||
spend DOUBLE PRECISION DEFAULT 0,
|
||||
api_requests BIGINT DEFAULT 0,
|
||||
successful_requests BIGINT DEFAULT 0,
|
||||
failed_requests BIGINT DEFAULT 0,
|
||||
total_response_time_ms BIGINT DEFAULT 0,
|
||||
timed_requests BIGINT DEFAULT 0
|
||||
)
|
||||
"""
|
||||
|
||||
|
||||
def _seed_daily_user_spend(conn: psycopg.Connection, rows: Sequence[tuple[object, ...]]) -> None:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(_DAILY_USER_SPEND_DDL)
|
||||
cur.executemany(
|
||||
"""
|
||||
INSERT INTO "LiteLLM_DailyUserSpend"
|
||||
(id, user_id, date, api_key, model, model_group, custom_llm_provider,
|
||||
endpoint, prompt_tokens, spend, api_requests, successful_requests)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
""",
|
||||
rows,
|
||||
)
|
||||
conn.commit()
|
||||
|
||||
|
||||
def _psycopg_query_raw(conn: psycopg.Connection, row_counts: list[int]):
|
||||
"""Run the proxy's $N-parameterized SQL through psycopg, recording each result size."""
|
||||
|
||||
async def query_raw(sql: str, *params: str) -> list[dict[str, object]]:
|
||||
converted: Final = re.sub(r"\$(\d+)", r"%(p\1)s", sql)
|
||||
with conn.cursor(row_factory=dict_row) as cur:
|
||||
cur.execute(
|
||||
converted, # pyright: ignore[reportArgumentType] # psycopg stubs want a literal-typed query
|
||||
{f"p{i}": v for i, v in enumerate(params, start=1)},
|
||||
)
|
||||
rows: Final = cur.fetchall()
|
||||
row_counts.append(len(rows))
|
||||
return rows
|
||||
|
||||
return query_raw
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_daily_activity_aggregated_bounds_api_key_rollups(
|
||||
_aggregated_postgresql: psycopg.Connection,
|
||||
):
|
||||
"""Run the GROUPING SETS statement against real Postgres with more keys than the cap.
|
||||
|
||||
key-004 and key-005 tie on spend exactly at the USAGE_TOP_API_KEYS_LIMIT
|
||||
cutoff; the api_key tiebreaker must keep key-004 and drop key-005. The PTU
|
||||
sentinel outspends every key but must not take a slot. Excluded keys and the
|
||||
sentinel still count toward the totals and the model rollup, which come from
|
||||
the key-free arm.
|
||||
"""
|
||||
n_keys: Final = USAGE_TOP_API_KEYS_LIMIT + 5
|
||||
key_rows: Final = [
|
||||
(
|
||||
f"row-{i:03d}",
|
||||
f"user-{i:03d}",
|
||||
"2026-06-01",
|
||||
f"key-{i:03d}",
|
||||
"gpt-5",
|
||||
"",
|
||||
"openai",
|
||||
"/v1/chat/completions",
|
||||
10,
|
||||
6.0 if i == 4 else float(i + 1),
|
||||
1,
|
||||
1,
|
||||
)
|
||||
for i in range(n_keys)
|
||||
]
|
||||
sentinel_row: Final = (
|
||||
"row-ptu",
|
||||
None,
|
||||
"2026-06-01",
|
||||
PTU_SENTINEL_API_KEY,
|
||||
"gpt-5",
|
||||
"",
|
||||
"azure",
|
||||
None,
|
||||
0,
|
||||
1000.0,
|
||||
0,
|
||||
0,
|
||||
)
|
||||
_seed_daily_user_spend(_aggregated_postgresql, [*key_rows, sentinel_row])
|
||||
key_spend: Final = sum(6.0 if i == 4 else float(i + 1) for i in range(n_keys))
|
||||
|
||||
row_counts: Final[list[int]] = [] # mutable-ok: out-param for the query_raw shim
|
||||
mock_prisma = MagicMock()
|
||||
mock_prisma.db = MagicMock()
|
||||
mock_prisma.db.query_raw = _psycopg_query_raw(_aggregated_postgresql, row_counts)
|
||||
mock_prisma.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[])
|
||||
mock_prisma.db.litellm_deletedverificationtoken.find_many = AsyncMock(return_value=[])
|
||||
|
||||
result = await get_daily_activity_aggregated(
|
||||
prisma_client=mock_prisma,
|
||||
table_name="litellm_dailyuserspend",
|
||||
entity_id_field="user_id",
|
||||
entity_id=None,
|
||||
entity_metadata_field=None,
|
||||
start_date="2026-06-01",
|
||||
end_date="2026-06-01",
|
||||
model=None,
|
||||
api_key=None,
|
||||
)
|
||||
|
||||
# Key-free arm: (), (date), (date, model), (date, model_group), two providers,
|
||||
# one mcp NULL bucket, endpoint plus its NULL bucket = 9 rows regardless of key count.
|
||||
# Per-key arm: six per-key grouping sets, each capped at the limit.
|
||||
assert row_counts == [9 + 6 * USAGE_TOP_API_KEYS_LIMIT]
|
||||
|
||||
assert result.metadata.total_spend == pytest.approx(key_spend + 1000.0)
|
||||
assert result.metadata.total_api_requests == n_keys
|
||||
assert result.metadata.api_key_limit == USAGE_TOP_API_KEYS_LIMIT
|
||||
assert result.metadata.total_api_keys == n_keys
|
||||
|
||||
expected_top: Final = {f"key-{i:03d}" for i in range(6, n_keys)} | {"key-004"}
|
||||
day: Final = result.results[0]
|
||||
assert day.metrics.spend == pytest.approx(key_spend + 1000.0)
|
||||
assert set(day.breakdown.api_keys) == expected_top
|
||||
assert day.breakdown.api_keys["key-004"].metrics.spend == 6.0
|
||||
assert "key-005" not in day.breakdown.api_keys
|
||||
assert PTU_SENTINEL_API_KEY not in day.breakdown.api_keys
|
||||
|
||||
assert day.breakdown.models["gpt-5"].metrics.spend == pytest.approx(key_spend + 1000.0)
|
||||
assert set(day.breakdown.models["gpt-5"].api_key_breakdown) == expected_top
|
||||
assert day.breakdown.providers["openai"].metrics.spend == pytest.approx(key_spend)
|
||||
assert set(day.breakdown.providers["openai"].api_key_breakdown) == expected_top
|
||||
assert day.breakdown.endpoints["/v1/chat/completions"].metrics.api_requests == n_keys
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_daily_activity_aggregated_explicit_api_key_filter_scopes_both_arms(
|
||||
_aggregated_postgresql: psycopg.Connection,
|
||||
):
|
||||
"""An explicit api_key filter must scope the key-free totals and the per-key
|
||||
rollups to that key alone, so the two arms never disagree."""
|
||||
rows: Final = [
|
||||
(
|
||||
f"row-{i}",
|
||||
f"user-{i}",
|
||||
"2026-06-01",
|
||||
f"key-{i}",
|
||||
"gpt-5",
|
||||
"",
|
||||
"openai",
|
||||
"/v1/chat/completions",
|
||||
10,
|
||||
float(i + 1),
|
||||
1,
|
||||
1,
|
||||
)
|
||||
for i in range(3)
|
||||
]
|
||||
_seed_daily_user_spend(_aggregated_postgresql, rows)
|
||||
|
||||
row_counts: Final[list[int]] = [] # mutable-ok: out-param for the query_raw shim
|
||||
mock_prisma = MagicMock()
|
||||
mock_prisma.db = MagicMock()
|
||||
mock_prisma.db.query_raw = _psycopg_query_raw(_aggregated_postgresql, row_counts)
|
||||
mock_prisma.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[])
|
||||
mock_prisma.db.litellm_deletedverificationtoken.find_many = AsyncMock(return_value=[])
|
||||
|
||||
result = await get_daily_activity_aggregated(
|
||||
prisma_client=mock_prisma,
|
||||
table_name="litellm_dailyuserspend",
|
||||
entity_id_field="user_id",
|
||||
entity_id=None,
|
||||
entity_metadata_field=None,
|
||||
start_date="2026-06-01",
|
||||
end_date="2026-06-01",
|
||||
model=None,
|
||||
api_key="key-1",
|
||||
)
|
||||
|
||||
assert result.metadata.total_spend == 2.0
|
||||
assert result.metadata.total_api_keys == 1
|
||||
day: Final = result.results[0]
|
||||
assert set(day.breakdown.api_keys) == {"key-1"}
|
||||
assert day.breakdown.api_keys["key-1"].metrics.spend == 2.0
|
||||
assert day.breakdown.models["gpt-5"].metrics.spend == 2.0
|
||||
assert set(day.breakdown.models["gpt-5"].api_key_breakdown) == {"key-1"}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_daily_activity_aggregated_reports_exact_limit_key_count_as_complete(
|
||||
_aggregated_postgresql: psycopg.Connection,
|
||||
):
|
||||
"""With exactly USAGE_TOP_API_KEYS_LIMIT keys nothing is dropped, and the
|
||||
response must say so: total_api_keys equals the limit rather than exceeding it."""
|
||||
rows: Final = [
|
||||
(
|
||||
f"row-{i:03d}",
|
||||
f"user-{i:03d}",
|
||||
"2026-06-01",
|
||||
f"key-{i:03d}",
|
||||
"gpt-5",
|
||||
"",
|
||||
"openai",
|
||||
"/v1/chat/completions",
|
||||
10,
|
||||
float(i + 1),
|
||||
1,
|
||||
1,
|
||||
)
|
||||
for i in range(USAGE_TOP_API_KEYS_LIMIT)
|
||||
]
|
||||
_seed_daily_user_spend(_aggregated_postgresql, rows)
|
||||
|
||||
row_counts: Final[list[int]] = [] # mutable-ok: out-param for the query_raw shim
|
||||
mock_prisma = MagicMock()
|
||||
mock_prisma.db = MagicMock()
|
||||
mock_prisma.db.query_raw = _psycopg_query_raw(_aggregated_postgresql, row_counts)
|
||||
mock_prisma.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[])
|
||||
mock_prisma.db.litellm_deletedverificationtoken.find_many = AsyncMock(return_value=[])
|
||||
|
||||
result = await get_daily_activity_aggregated(
|
||||
prisma_client=mock_prisma,
|
||||
table_name="litellm_dailyuserspend",
|
||||
entity_id_field="user_id",
|
||||
entity_id=None,
|
||||
entity_metadata_field=None,
|
||||
start_date="2026-06-01",
|
||||
end_date="2026-06-01",
|
||||
model=None,
|
||||
api_key=None,
|
||||
)
|
||||
|
||||
assert result.metadata.total_api_keys == USAGE_TOP_API_KEYS_LIMIT
|
||||
assert result.metadata.api_key_limit == USAGE_TOP_API_KEYS_LIMIT
|
||||
assert len(result.results[0].breakdown.api_keys) == USAGE_TOP_API_KEYS_LIMIT
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_daily_activity_aggregated_model_group_rollups_fall_back_to_model_name(
|
||||
_aggregated_postgresql: psycopg.Connection,
|
||||
):
|
||||
"""Rows stored with an empty or NULL model_group must land in the model_groups
|
||||
breakdown under their model name instead of vanishing from the usage UI."""
|
||||
rows: Final = [
|
||||
("row-0", "user-0", "2026-06-01", "key-0", "gpt-5", "gpt-5-eu", "openai", "/v1/chat/completions", 10, 7.0, 1, 1),
|
||||
("row-1", "user-1", "2026-06-01", "key-1", "gpt-5", "", "openai", "/v1/chat/completions", 10, 3.0, 1, 1),
|
||||
("row-2", "user-2", "2026-06-01", "key-2", "claude-x", None, "anthropic", "/v1/messages", 10, 2.0, 1, 1),
|
||||
]
|
||||
_seed_daily_user_spend(_aggregated_postgresql, rows)
|
||||
|
||||
mock_prisma = MagicMock()
|
||||
mock_prisma.db = MagicMock()
|
||||
mock_prisma.db.query_raw = _psycopg_query_raw(_aggregated_postgresql, [])
|
||||
mock_prisma.db.litellm_verificationtoken.find_many = AsyncMock(return_value=[])
|
||||
mock_prisma.db.litellm_deletedverificationtoken.find_many = AsyncMock(return_value=[])
|
||||
|
||||
result = await get_daily_activity_aggregated(
|
||||
prisma_client=mock_prisma,
|
||||
table_name="litellm_dailyuserspend",
|
||||
entity_id_field="user_id",
|
||||
entity_id=None,
|
||||
entity_metadata_field=None,
|
||||
start_date="2026-06-01",
|
||||
end_date="2026-06-01",
|
||||
model=None,
|
||||
api_key=None,
|
||||
)
|
||||
|
||||
breakdown: Final = result.results[0].breakdown
|
||||
assert set(breakdown.model_groups) == {"gpt-5-eu", "gpt-5", "claude-x"}
|
||||
assert breakdown.model_groups["gpt-5-eu"].metrics.spend == 7.0
|
||||
assert breakdown.model_groups["gpt-5"].metrics.spend == 3.0
|
||||
assert breakdown.model_groups["claude-x"].metrics.spend == 2.0
|
||||
assert set(breakdown.model_groups["gpt-5"].api_key_breakdown) == {"key-1"}
|
||||
assert set(breakdown.models) == {"gpt-5", "claude-x"}
|
||||
assert breakdown.models["gpt-5"].metrics.spend == 10.0
|
||||
|
||||
|
||||
def _no_spend_record():
|
||||
"""A rollup row for a key with no spend, where SUM() returns NULL (None)."""
|
||||
return SimpleNamespace(
|
||||
|
|
@ -2170,7 +2453,7 @@ def test_entity_rollup_sql_query_and_api_key_list_filter():
|
|||
api_key=[],
|
||||
)
|
||||
assert "FALSE" in empty_sql
|
||||
assert empty_params == ["2024-01-01", "2024-01-31"]
|
||||
assert empty_params == ["2024-01-01", "2024-01-31", PTU_SENTINEL_API_KEY]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -2204,10 +2487,10 @@ async def test_get_daily_activity_aggregated_with_entity_breakdown():
|
|||
"successful_requests": 0,
|
||||
}
|
||||
main_rows = [
|
||||
{**base, "date": None, "group_level": 127, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "group_level": 63, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "model": "gpt-4o", "group_level": 47, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "api_key": "key-1", "group_level": 31, "spend": 12.0},
|
||||
{**base, "date": None, "group_level": 127, "distinct_api_keys": None, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "group_level": 63, "distinct_api_keys": None, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "model": "gpt-4o", "group_level": 47, "distinct_api_keys": None, "spend": 18.0},
|
||||
{**base, "date": "2024-01-01", "api_key": "key-1", "group_level": 31, "distinct_api_keys": 1, "spend": 12.0},
|
||||
]
|
||||
entity_base = {
|
||||
key: value
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ import inspect
|
|||
|
||||
from litellm.proxy._types import (
|
||||
GenerateKeyRequest,
|
||||
KeyManagementRoutes,
|
||||
NewUserRequest,
|
||||
LiteLLM_BudgetTable,
|
||||
LiteLLM_ObjectPermissionBase,
|
||||
|
|
@ -3297,7 +3298,7 @@ async def test_validate_key_team_change_with_member_permissions():
|
|||
|
||||
# Verify the permission check was called with correct parameters
|
||||
mock_has_perms.assert_called_once_with(
|
||||
team_member_object=mock_member_object,
|
||||
team_member_role=mock_member_object.role,
|
||||
team_table=mock_team,
|
||||
route=KeyManagementRoutes.KEY_UPDATE.value,
|
||||
)
|
||||
|
|
@ -19937,3 +19938,130 @@ async def test_bulk_update_team_keys_runs_custom_key_policy_per_key(monkeypatch)
|
|||
assert [policy_request.operation for policy_request in received] == ["update", "update"]
|
||||
assert [policy_request.effective_key.max_budget for policy_request in received] == [50.0, 50.0]
|
||||
assert [policy_request.effective_key.team_id for policy_request in received] == ["team-abc", "team-abc"]
|
||||
|
||||
|
||||
class TestServiceAccountKeyGenerationCheck:
|
||||
"""Service account keys (user_id=None, team_id set, metadata.service_account_id)
|
||||
may only create keys for their own team."""
|
||||
|
||||
def _service_account_token(self, team_id: str) -> UserAPIKeyAuth:
|
||||
return UserAPIKeyAuth(
|
||||
api_key="sk-sa",
|
||||
user_id=None,
|
||||
team_id=team_id,
|
||||
metadata={"service_account_id": "sa-1"},
|
||||
)
|
||||
|
||||
def test_other_team_denied(self):
|
||||
data = GenerateKeyRequest(team_id="team-b")
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
key_generation_check(
|
||||
team_table=None,
|
||||
user_api_key_dict=self._service_account_token(team_id="team-a"),
|
||||
data=data,
|
||||
route=KeyManagementRoutes.KEY_GENERATE,
|
||||
)
|
||||
assert exc_info.value.status_code == 403
|
||||
|
||||
def test_personal_key_denied(self):
|
||||
"""team_id=None would mint a personal key; service accounts may only
|
||||
create keys for their own team."""
|
||||
data = GenerateKeyRequest()
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
key_generation_check(
|
||||
team_table=None,
|
||||
user_api_key_dict=self._service_account_token(team_id="team-a"),
|
||||
data=data,
|
||||
route=KeyManagementRoutes.KEY_GENERATE,
|
||||
)
|
||||
assert exc_info.value.status_code == 403
|
||||
|
||||
def test_own_team_with_permission_allowed(self):
|
||||
team_table = LiteLLM_TeamTableCachedObj(
|
||||
team_id="team-a",
|
||||
members_with_roles=[],
|
||||
team_member_permissions=["/key/generate"],
|
||||
)
|
||||
data = GenerateKeyRequest(team_id="team-a")
|
||||
assert (
|
||||
key_generation_check(
|
||||
team_table=team_table,
|
||||
user_api_key_dict=self._service_account_token(team_id="team-a"),
|
||||
data=data,
|
||||
route=KeyManagementRoutes.KEY_GENERATE,
|
||||
)
|
||||
is True
|
||||
)
|
||||
|
||||
def test_own_team_without_permission_denied(self):
|
||||
team_table = LiteLLM_TeamTableCachedObj(
|
||||
team_id="team-a",
|
||||
members_with_roles=[],
|
||||
team_member_permissions=["/key/info"],
|
||||
)
|
||||
data = GenerateKeyRequest(team_id="team-a")
|
||||
with pytest.raises(ProxyException) as exc_info:
|
||||
key_generation_check(
|
||||
team_table=team_table,
|
||||
user_api_key_dict=self._service_account_token(team_id="team-a"),
|
||||
data=data,
|
||||
route=KeyManagementRoutes.KEY_GENERATE,
|
||||
)
|
||||
assert str(exc_info.value.code) == "401"
|
||||
|
||||
|
||||
def _stub_service_account_generation(monkeypatch):
|
||||
"""Stub the DB lookups generate_service_account_key_fn needs so the test
|
||||
exercises only the service_account_id stamping and user_id clearing."""
|
||||
from litellm.proxy import proxy_server
|
||||
from litellm.proxy.management_endpoints import key_management_endpoints as kme
|
||||
|
||||
mock_helper = AsyncMock(return_value=MagicMock())
|
||||
monkeypatch.setattr(proxy_server, "prisma_client", MagicMock())
|
||||
monkeypatch.setattr(kme, "validate_team_id_used_in_service_account_request", AsyncMock())
|
||||
monkeypatch.setattr(kme, "_common_key_generation_helper", mock_helper)
|
||||
return mock_helper
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_generate_service_account_key_stamps_service_account_id(monkeypatch):
|
||||
"""generate_service_account_key_fn must stamp metadata.service_account_id
|
||||
(key_alias fallback) so the key is identifiable as a service account by
|
||||
is_team_service_account and check_if_token_is_service_account."""
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
generate_service_account_key_fn,
|
||||
)
|
||||
|
||||
mock_helper = _stub_service_account_generation(monkeypatch)
|
||||
data = GenerateKeyRequest(team_id="team-a", key_alias="sa-alias")
|
||||
|
||||
await generate_service_account_key_fn(
|
||||
data=data,
|
||||
user_api_key_dict=UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN, api_key="sk-1"),
|
||||
litellm_changed_by=None,
|
||||
)
|
||||
|
||||
assert data.metadata is not None
|
||||
assert data.metadata["service_account_id"] == "sa-alias"
|
||||
assert data.user_id is None
|
||||
mock_helper.assert_awaited_once()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_generate_service_account_key_generates_uuid_when_no_alias(monkeypatch):
|
||||
"""Without key_alias, service_account_id falls back to a generated uuid."""
|
||||
from litellm.proxy.management_endpoints.key_management_endpoints import (
|
||||
generate_service_account_key_fn,
|
||||
)
|
||||
|
||||
_stub_service_account_generation(monkeypatch)
|
||||
data = GenerateKeyRequest(team_id="team-a")
|
||||
|
||||
await generate_service_account_key_fn(
|
||||
data=data,
|
||||
user_api_key_dict=UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN, api_key="sk-1"),
|
||||
litellm_changed_by=None,
|
||||
)
|
||||
|
||||
assert data.metadata is not None
|
||||
assert data.metadata["service_account_id"]
|
||||
|
|
|
|||
|
|
@ -24,6 +24,7 @@ from litellm.proxy._types import (
|
|||
LiteLLM_MCPServerTable,
|
||||
LitellmUserRoles,
|
||||
MCPTransport,
|
||||
MCPUserCredentialResponse,
|
||||
NewMCPServerRequest,
|
||||
UpdateMCPServerRequest,
|
||||
UserAPIKeyAuth,
|
||||
|
|
@ -5136,6 +5137,266 @@ async def test_delete_mcp_oauth_user_credential_invalidates_when_record_already_
|
|||
assert result.has_credential is False
|
||||
|
||||
|
||||
def _make_admin_auth(role: LitellmUserRoles = LitellmUserRoles.PROXY_ADMIN) -> "UserAPIKeyAuth":
|
||||
return UserAPIKeyAuth(api_key="sk-admin", user_id="admin-user", user_role=role)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_admin_revokes_another_users_byok_credential():
|
||||
"""A proxy admin naming user_id deletes and cache-invalidates that user's stored key, not their own."""
|
||||
if not mgmt_endpoints.MCP_AVAILABLE:
|
||||
pytest.skip("MCP module not installed")
|
||||
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
delete_mcp_user_credential,
|
||||
)
|
||||
|
||||
delete_mock = AsyncMock(return_value=None)
|
||||
invalidate_mock = AsyncMock()
|
||||
with (
|
||||
patch( # test-quality-ok: endpoint test stubs the Prisma client lookup
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.get_prisma_client_or_throw",
|
||||
return_value=_make_prisma_client(),
|
||||
),
|
||||
patch( # test-quality-ok: endpoint test stubs the credential row delete
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.delete_user_credential",
|
||||
new=delete_mock,
|
||||
),
|
||||
patch.object( # test-quality-ok: the cache invalidator is module scoped; the suite's only seam
|
||||
mcp_server, "_invalidate_byok_cred_cache", new=invalidate_mock
|
||||
),
|
||||
):
|
||||
result = await delete_mcp_user_credential(
|
||||
server_id="srv-byok-admin",
|
||||
user_api_key_dict=_make_admin_auth(),
|
||||
user_id="mallory",
|
||||
)
|
||||
|
||||
delete_mock.assert_awaited_once()
|
||||
assert delete_mock.await_args.args[1:] == ("mallory", "srv-byok-admin")
|
||||
invalidate_mock.assert_awaited_once_with("mallory", "srv-byok-admin")
|
||||
assert result.has_credential is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("role", [LitellmUserRoles.INTERNAL_USER, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY])
|
||||
async def test_non_full_admin_cannot_revoke_another_users_byok_credential(role):
|
||||
if not mgmt_endpoints.MCP_AVAILABLE:
|
||||
pytest.skip("MCP module not installed")
|
||||
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
delete_mcp_user_credential,
|
||||
)
|
||||
|
||||
delete_mock = AsyncMock(return_value=None)
|
||||
with (
|
||||
patch( # test-quality-ok: endpoint test stubs the Prisma client lookup
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.get_prisma_client_or_throw",
|
||||
return_value=_make_prisma_client(),
|
||||
),
|
||||
patch( # test-quality-ok: endpoint test stubs the credential row delete
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.delete_user_credential",
|
||||
new=delete_mock,
|
||||
),
|
||||
):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await delete_mcp_user_credential(
|
||||
server_id="srv-byok-forbidden",
|
||||
user_api_key_dict=_make_admin_auth(role),
|
||||
user_id="mallory",
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 403
|
||||
delete_mock.assert_not_awaited()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_user_naming_themselves_still_deletes_own_byok_credential():
|
||||
if not mgmt_endpoints.MCP_AVAILABLE:
|
||||
pytest.skip("MCP module not installed")
|
||||
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
delete_mcp_user_credential,
|
||||
)
|
||||
|
||||
deleted_rows: list[tuple[str, str]] = [] # mutable-ok: test-local recorder for the fake delete boundary
|
||||
|
||||
async def _fake_delete_user_credential(_prisma_client: object, user_id: str, server_id: str) -> None:
|
||||
deleted_rows.append((user_id, server_id))
|
||||
|
||||
with (
|
||||
patch( # test-quality-ok: endpoint test stubs the Prisma client lookup
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.get_prisma_client_or_throw",
|
||||
return_value=_make_prisma_client(),
|
||||
),
|
||||
patch( # test-quality-ok: endpoint test stubs the credential row delete
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.delete_user_credential",
|
||||
new=_fake_delete_user_credential,
|
||||
),
|
||||
patch.object( # test-quality-ok: the cache invalidator is module scoped; the suite's only seam
|
||||
mcp_server, "_invalidate_byok_cred_cache", new=AsyncMock()
|
||||
),
|
||||
):
|
||||
result = await delete_mcp_user_credential(
|
||||
server_id="srv-byok-self",
|
||||
user_api_key_dict=_make_user_auth("user-self"),
|
||||
user_id="user-self",
|
||||
)
|
||||
|
||||
assert deleted_rows == [("user-self", "srv-byok-self")]
|
||||
assert result == MCPUserCredentialResponse(server_id="srv-byok-self", has_credential=False)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_admin_revokes_another_users_oauth_credential():
|
||||
"""A proxy admin naming user_id reads, deletes, and cache-invalidates that user's OAuth token."""
|
||||
if not mgmt_endpoints.MCP_AVAILABLE:
|
||||
pytest.skip("MCP module not installed")
|
||||
|
||||
from litellm.proxy._experimental.mcp_server import mcp_server_manager as manager_module
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
delete_mcp_oauth_user_credential,
|
||||
)
|
||||
|
||||
get_mock = AsyncMock(return_value={"type": "oauth2", "access_token": "mallory-tok"})
|
||||
delete_mock = AsyncMock(return_value=None)
|
||||
invalidate_mock = AsyncMock(return_value=None)
|
||||
with (
|
||||
patch( # test-quality-ok: endpoint test stubs the Prisma client lookup
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.get_prisma_client_or_throw",
|
||||
return_value=_make_prisma_client(),
|
||||
),
|
||||
patch( # test-quality-ok: endpoint test stubs the stored OAuth token read
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.get_user_oauth_credential",
|
||||
new=get_mock,
|
||||
),
|
||||
patch( # test-quality-ok: endpoint test stubs the credential row delete
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.delete_user_credential",
|
||||
new=delete_mock,
|
||||
),
|
||||
patch.object( # test-quality-ok: the OAuth cache lives on the global manager; the suite's only seam
|
||||
manager_module.global_mcp_server_manager,
|
||||
"invalidate_user_oauth_token_cache",
|
||||
new=invalidate_mock,
|
||||
),
|
||||
):
|
||||
result = await delete_mcp_oauth_user_credential(
|
||||
server_id="srv-oauth-admin",
|
||||
user_api_key_dict=_make_admin_auth(),
|
||||
user_id="mallory",
|
||||
)
|
||||
|
||||
assert get_mock.await_args.args[1:] == ("mallory", "srv-oauth-admin")
|
||||
assert delete_mock.await_args.args[1:] == ("mallory", "srv-oauth-admin")
|
||||
invalidate_mock.assert_awaited_once_with("mallory", "srv-oauth-admin")
|
||||
assert result.has_credential is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("role", [LitellmUserRoles.INTERNAL_USER, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY])
|
||||
async def test_non_full_admin_cannot_revoke_another_users_oauth_credential(role):
|
||||
if not mgmt_endpoints.MCP_AVAILABLE:
|
||||
pytest.skip("MCP module not installed")
|
||||
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
delete_mcp_oauth_user_credential,
|
||||
)
|
||||
|
||||
get_mock = AsyncMock(return_value={"type": "oauth2", "access_token": "mallory-tok"})
|
||||
delete_mock = AsyncMock(return_value=None)
|
||||
with (
|
||||
patch( # test-quality-ok: endpoint test stubs the Prisma client lookup
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.get_prisma_client_or_throw",
|
||||
return_value=_make_prisma_client(),
|
||||
),
|
||||
patch( # test-quality-ok: endpoint test stubs the stored OAuth token read
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.get_user_oauth_credential",
|
||||
new=get_mock,
|
||||
),
|
||||
patch( # test-quality-ok: endpoint test stubs the credential row delete
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.delete_user_credential",
|
||||
new=delete_mock,
|
||||
),
|
||||
):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await delete_mcp_oauth_user_credential(
|
||||
server_id="srv-oauth-forbidden",
|
||||
user_api_key_dict=_make_admin_auth(role),
|
||||
user_id="mallory",
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 403
|
||||
get_mock.assert_not_awaited()
|
||||
delete_mock.assert_not_awaited()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("role", [LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY])
|
||||
async def test_admin_lists_every_users_credential_for_a_server(role):
|
||||
if not mgmt_endpoints.MCP_AVAILABLE:
|
||||
pytest.skip("MCP module not installed")
|
||||
|
||||
from litellm.proxy._types import MCPServerUserCredentialListItem
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
list_mcp_server_user_credentials,
|
||||
)
|
||||
|
||||
items = (
|
||||
MCPServerUserCredentialListItem(user_id="alice", credential_type="byok", updated_at="2026-01-01T00:00:00"),
|
||||
MCPServerUserCredentialListItem(user_id="bob", credential_type="oauth2", updated_at="2026-01-02T00:00:00"),
|
||||
)
|
||||
list_mock = AsyncMock(return_value=items)
|
||||
with (
|
||||
patch( # test-quality-ok: endpoint test stubs the Prisma client lookup
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.get_prisma_client_or_throw",
|
||||
return_value=_make_prisma_client(),
|
||||
),
|
||||
patch( # test-quality-ok: endpoint test stubs the credential row listing
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.list_server_user_credentials",
|
||||
new=list_mock,
|
||||
),
|
||||
):
|
||||
result = await list_mcp_server_user_credentials(
|
||||
server_id="srv-list-admin",
|
||||
user_api_key_dict=_make_admin_auth(role),
|
||||
)
|
||||
|
||||
assert list_mock.await_args.args[1:] == ("srv-list-admin",)
|
||||
assert [(item.user_id, item.credential_type) for item in result] == [("alice", "byok"), ("bob", "oauth2")]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_non_admin_cannot_list_a_servers_user_credentials():
|
||||
if not mgmt_endpoints.MCP_AVAILABLE:
|
||||
pytest.skip("MCP module not installed")
|
||||
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
list_mcp_server_user_credentials,
|
||||
)
|
||||
|
||||
list_mock = AsyncMock(return_value=())
|
||||
with (
|
||||
patch( # test-quality-ok: endpoint test stubs the Prisma client lookup
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.get_prisma_client_or_throw",
|
||||
return_value=_make_prisma_client(),
|
||||
),
|
||||
patch( # test-quality-ok: endpoint test stubs the credential row listing
|
||||
"litellm.proxy.management_endpoints.mcp_management_endpoints.list_server_user_credentials",
|
||||
new=list_mock,
|
||||
),
|
||||
):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await list_mcp_server_user_credentials(
|
||||
server_id="srv-list-forbidden",
|
||||
user_api_key_dict=_make_user_auth("user-plain"),
|
||||
)
|
||||
|
||||
assert exc_info.value.status_code == 403
|
||||
list_mock.assert_not_awaited()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_list_mcp_user_credentials_batch_server_fetch():
|
||||
"""list_mcp_user_credentials uses a single batch DB call, not N+1 queries."""
|
||||
|
|
@ -7321,3 +7582,146 @@ class TestGetMCPGatewaySessions:
|
|||
assert [(group.label, group.count) for group in result.by_client] == [("cursor", 1)]
|
||||
assert [(group.label, group.count) for group in result.by_user] == [("alice", 1)]
|
||||
assert "sk-live-secret" not in result.model_dump_json()
|
||||
|
||||
|
||||
class TestDeleteMCPGatewaySessions:
|
||||
@pytest.fixture(autouse=True)
|
||||
def _forget_admin_terminated_ids(self):
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
|
||||
yield
|
||||
mcp_server._admin_terminated_session_ids.clear()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("role", [LitellmUserRoles.INTERNAL_USER, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY])
|
||||
async def test_non_full_admin_forbidden_before_any_session_is_touched(self, role):
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
delete_mcp_gateway_sessions,
|
||||
)
|
||||
|
||||
session_id = "gateway-terminate-forbidden-1"
|
||||
transport = MagicMock(terminate=AsyncMock())
|
||||
auth_user = mcp_server.MCPAuthenticatedUser(
|
||||
user_api_key_auth=UserAPIKeyAuth(api_key="sk-live", user_id="alice"),
|
||||
)
|
||||
with (
|
||||
patch.object( # test-quality-ok: the transport registry is a module-level singleton; the suite's only seam
|
||||
mcp_server.session_manager_stateful, "_server_instances", {session_id: transport}
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_auth_contexts, {session_id: auth_user}, clear=True
|
||||
),
|
||||
):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await delete_mcp_gateway_sessions(
|
||||
user_api_key_dict=generate_mock_user_api_key_auth(user_role=role),
|
||||
session_id_prefix=session_id,
|
||||
user_id=None,
|
||||
)
|
||||
assert exc_info.value.status_code == 403
|
||||
transport.terminate.assert_not_awaited()
|
||||
assert session_id in mcp_server._stateful_session_auth_contexts
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_requires_a_selector(self):
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
delete_mcp_gateway_sessions,
|
||||
)
|
||||
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
await delete_mcp_gateway_sessions(
|
||||
user_api_key_dict=generate_mock_user_api_key_auth(user_role=LitellmUserRoles.PROXY_ADMIN),
|
||||
session_id_prefix=None,
|
||||
user_id=None,
|
||||
)
|
||||
assert exc_info.value.status_code == 400
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_admin_terminates_only_the_selected_session(self):
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
delete_mcp_gateway_sessions,
|
||||
)
|
||||
from litellm.types.mcp import MCPGatewaySessionsTerminateResponse
|
||||
|
||||
target_id = "11111111-target-session"
|
||||
other_id = "22222222-other-session"
|
||||
target_transport = MagicMock(terminate=AsyncMock())
|
||||
other_transport = MagicMock(terminate=AsyncMock())
|
||||
transports = {target_id: target_transport, other_id: other_transport}
|
||||
contexts = {
|
||||
target_id: mcp_server.MCPAuthenticatedUser(
|
||||
user_api_key_auth=UserAPIKeyAuth(api_key="sk-live-target", user_id="alice"),
|
||||
),
|
||||
other_id: mcp_server.MCPAuthenticatedUser(
|
||||
user_api_key_auth=UserAPIKeyAuth(api_key="sk-live-other", user_id="bob"),
|
||||
),
|
||||
}
|
||||
with (
|
||||
patch.object( # test-quality-ok: the transport registry is a module-level singleton; the suite's only seam
|
||||
mcp_server.session_manager_stateful, "_server_instances", transports
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_auth_contexts, contexts, clear=True
|
||||
),
|
||||
):
|
||||
result = await delete_mcp_gateway_sessions(
|
||||
user_api_key_dict=generate_mock_user_api_key_auth(user_role=LitellmUserRoles.PROXY_ADMIN),
|
||||
session_id_prefix=target_id[:8],
|
||||
user_id=None,
|
||||
)
|
||||
assert target_id not in transports
|
||||
assert other_id in transports
|
||||
assert target_id not in mcp_server._stateful_session_auth_contexts
|
||||
assert other_id in mcp_server._stateful_session_auth_contexts
|
||||
|
||||
target_transport.terminate.assert_awaited_once()
|
||||
other_transport.terminate.assert_not_awaited()
|
||||
assert isinstance(result, MCPGatewaySessionsTerminateResponse)
|
||||
assert result.terminated_sessions == 1
|
||||
assert [(s.session_id_prefix, s.user_id) for s in result.sessions] == [(target_id[:8], "alice")]
|
||||
assert target_id not in result.model_dump_json()
|
||||
assert "sk-live-target" not in result.model_dump_json()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_admin_terminates_every_session_of_the_selected_user(self):
|
||||
from litellm.proxy._experimental.mcp_server import server as mcp_server
|
||||
from litellm.proxy.management_endpoints.mcp_management_endpoints import (
|
||||
delete_mcp_gateway_sessions,
|
||||
)
|
||||
|
||||
def auth_user(user_id: str):
|
||||
return mcp_server.MCPAuthenticatedUser(
|
||||
user_api_key_auth=UserAPIKeyAuth(api_key=f"sk-live-{user_id}", user_id=user_id),
|
||||
)
|
||||
|
||||
transports = {
|
||||
"bob-session-1": MagicMock(terminate=AsyncMock()),
|
||||
"bob-session-2": MagicMock(terminate=AsyncMock()),
|
||||
"alice-session-1": MagicMock(terminate=AsyncMock()),
|
||||
}
|
||||
contexts = {
|
||||
"bob-session-1": auth_user("bob"),
|
||||
"bob-session-2": auth_user("bob"),
|
||||
"alice-session-1": auth_user("alice"),
|
||||
}
|
||||
with (
|
||||
patch.object( # test-quality-ok: the transport registry is a module-level singleton; the suite's only seam
|
||||
mcp_server.session_manager_stateful, "_server_instances", transports
|
||||
),
|
||||
patch.dict( # test-quality-ok: the session tables are module-level singletons; the suite's only seam
|
||||
mcp_server._stateful_session_auth_contexts, contexts, clear=True
|
||||
),
|
||||
):
|
||||
result = await delete_mcp_gateway_sessions(
|
||||
user_api_key_dict=generate_mock_user_api_key_auth(user_role=LitellmUserRoles.PROXY_ADMIN),
|
||||
session_id_prefix=None,
|
||||
user_id="bob",
|
||||
)
|
||||
assert set(transports) == {"alice-session-1"}
|
||||
assert set(mcp_server._stateful_session_auth_contexts) == {"alice-session-1"}
|
||||
|
||||
assert result.terminated_sessions == 2
|
||||
assert {s.user_id for s in result.sessions} == {"bob"}
|
||||
assert "sk-live-bob" not in result.model_dump_json()
|
||||
|
|
|
|||
|
|
@ -3,7 +3,12 @@ from unittest.mock import MagicMock
|
|||
import pytest
|
||||
|
||||
|
||||
from litellm.proxy._types import KeyManagementRoutes, Member, ProxyException
|
||||
from litellm.proxy._types import (
|
||||
KeyManagementRoutes,
|
||||
Member,
|
||||
ProxyException,
|
||||
UserAPIKeyAuth,
|
||||
)
|
||||
from litellm.proxy.management_helpers.team_member_permission_checks import (
|
||||
BASELINE_TEAM_MEMBER_PERMISSIONS,
|
||||
TeamMemberPermissionChecks,
|
||||
|
|
@ -21,22 +26,16 @@ class TestGetPermissionsForTeamMember:
|
|||
def test_none_permissions_returns_defaults(self):
|
||||
"""When team_member_permissions is None, return DEFAULT_TEAM_MEMBER_PERMISSIONS."""
|
||||
team = _make_team_table(None)
|
||||
member = MagicMock(spec=Member)
|
||||
|
||||
result = TeamMemberPermissionChecks.get_permissions_for_team_member(
|
||||
team_member_object=member, team_table=team
|
||||
)
|
||||
result = TeamMemberPermissionChecks.get_permissions_for_team_member(team_table=team)
|
||||
|
||||
assert set(result) == set(BASELINE_TEAM_MEMBER_PERMISSIONS)
|
||||
|
||||
def test_empty_list_includes_baseline(self):
|
||||
"""When team_member_permissions is [], baseline permissions are still included."""
|
||||
team = _make_team_table([])
|
||||
member = MagicMock(spec=Member)
|
||||
|
||||
result = TeamMemberPermissionChecks.get_permissions_for_team_member(
|
||||
team_member_object=member, team_table=team
|
||||
)
|
||||
result = TeamMemberPermissionChecks.get_permissions_for_team_member(team_table=team)
|
||||
|
||||
assert KeyManagementRoutes.KEY_INFO in result
|
||||
assert KeyManagementRoutes.KEY_HEALTH in result
|
||||
|
|
@ -44,11 +43,8 @@ class TestGetPermissionsForTeamMember:
|
|||
def test_explicit_permissions_include_baseline(self):
|
||||
"""When explicit permissions are set, baseline is always included."""
|
||||
team = _make_team_table(["/key/generate", "/key/delete"])
|
||||
member = MagicMock(spec=Member)
|
||||
|
||||
result = TeamMemberPermissionChecks.get_permissions_for_team_member(
|
||||
team_member_object=member, team_table=team
|
||||
)
|
||||
result = TeamMemberPermissionChecks.get_permissions_for_team_member(team_table=team)
|
||||
|
||||
assert KeyManagementRoutes.KEY_GENERATE in result
|
||||
assert KeyManagementRoutes.KEY_DELETE in result
|
||||
|
|
@ -58,11 +54,8 @@ class TestGetPermissionsForTeamMember:
|
|||
def test_explicit_permissions_with_baseline_no_duplicates(self):
|
||||
"""When explicit permissions already include baseline, no duplicates."""
|
||||
team = _make_team_table(["/key/info", "/key/generate"])
|
||||
member = MagicMock(spec=Member)
|
||||
|
||||
result = TeamMemberPermissionChecks.get_permissions_for_team_member(
|
||||
team_member_object=member, team_table=team
|
||||
)
|
||||
result = TeamMemberPermissionChecks.get_permissions_for_team_member(team_table=team)
|
||||
|
||||
# Using set ensures no duplicates from the implementation
|
||||
assert KeyManagementRoutes.KEY_INFO in result
|
||||
|
|
@ -402,3 +395,148 @@ class TestEnforceMemberCanAssignAccessGroups:
|
|||
team_table=self._team(["/key/generate", self.AG_PERMISSION]),
|
||||
access_group_ids=["ag-1"],
|
||||
)
|
||||
|
||||
|
||||
class TestDoesTeamMemberHavePermissionsForEndpoint:
|
||||
def _team(self, team_member_permissions, team_id="team-a"):
|
||||
team = MagicMock()
|
||||
team.team_id = team_id
|
||||
team.team_member_permissions = team_member_permissions
|
||||
return team
|
||||
|
||||
def test_none_role_returns_false(self):
|
||||
"""A caller with no team membership is denied."""
|
||||
result = TeamMemberPermissionChecks.does_team_member_have_permissions_for_endpoint(
|
||||
team_member_role=None,
|
||||
team_table=self._team(["/key/update"]),
|
||||
route=KeyManagementRoutes.KEY_UPDATE.value,
|
||||
)
|
||||
assert result is False
|
||||
|
||||
def test_admin_role_always_allowed(self):
|
||||
"""Team admins bypass the member permission list."""
|
||||
result = TeamMemberPermissionChecks.does_team_member_have_permissions_for_endpoint(
|
||||
team_member_role="admin",
|
||||
team_table=self._team([]),
|
||||
route=KeyManagementRoutes.KEY_UPDATE.value,
|
||||
)
|
||||
assert result is True
|
||||
|
||||
def test_user_role_with_permission_allowed(self):
|
||||
result = TeamMemberPermissionChecks.does_team_member_have_permissions_for_endpoint(
|
||||
team_member_role="user",
|
||||
team_table=self._team(["/key/update"]),
|
||||
route=KeyManagementRoutes.KEY_UPDATE.value,
|
||||
)
|
||||
assert result is True
|
||||
|
||||
def test_user_role_without_permission_raises(self):
|
||||
with pytest.raises(ProxyException) as exc:
|
||||
TeamMemberPermissionChecks.does_team_member_have_permissions_for_endpoint(
|
||||
team_member_role="user",
|
||||
team_table=self._team(["/key/generate"]),
|
||||
route=KeyManagementRoutes.KEY_UPDATE.value,
|
||||
)
|
||||
assert str(exc.value.code) == "401"
|
||||
assert exc.value.type == "team_member_permission_error"
|
||||
|
||||
|
||||
class TestCanTeamMemberExecuteKeyManagementEndpointServiceAccount:
|
||||
def _service_account_token(self, team_id: str) -> UserAPIKeyAuth:
|
||||
return UserAPIKeyAuth(
|
||||
api_key="sk-test",
|
||||
user_id=None,
|
||||
team_id=team_id,
|
||||
metadata={"service_account_id": "sa-1"},
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_service_account_same_team_with_permission(self, monkeypatch):
|
||||
"""A service account key can manage keys in its own team when the
|
||||
team grants the route via team_member_permissions."""
|
||||
from litellm.proxy.management_helpers import (
|
||||
team_member_permission_checks as module,
|
||||
)
|
||||
|
||||
async def _mock_get_team_object(**kwargs):
|
||||
team = MagicMock()
|
||||
team.team_id = "team-a"
|
||||
team.members_with_roles = []
|
||||
team.team_member_permissions = ["/key/update"]
|
||||
return team
|
||||
|
||||
monkeypatch.setattr(module, "get_team_object", _mock_get_team_object)
|
||||
|
||||
existing_key_row = MagicMock()
|
||||
existing_key_row.team_id = "team-a"
|
||||
|
||||
result = await TeamMemberPermissionChecks.can_team_member_execute_key_management_endpoint(
|
||||
user_api_key_dict=self._service_account_token(team_id="team-a"),
|
||||
route=KeyManagementRoutes.KEY_UPDATE,
|
||||
prisma_client=MagicMock(),
|
||||
user_api_key_cache=MagicMock(),
|
||||
existing_key_row=existing_key_row,
|
||||
)
|
||||
assert result is None
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_service_account_same_team_without_permission(self, monkeypatch):
|
||||
"""A service account key is denied when the team's
|
||||
team_member_permissions does not include the route."""
|
||||
from litellm.proxy.management_helpers import (
|
||||
team_member_permission_checks as module,
|
||||
)
|
||||
|
||||
async def _mock_get_team_object(**kwargs):
|
||||
team = MagicMock()
|
||||
team.team_id = "team-a"
|
||||
team.members_with_roles = []
|
||||
team.team_member_permissions = ["/key/generate"]
|
||||
return team
|
||||
|
||||
monkeypatch.setattr(module, "get_team_object", _mock_get_team_object)
|
||||
|
||||
existing_key_row = MagicMock()
|
||||
existing_key_row.team_id = "team-a"
|
||||
|
||||
with pytest.raises(ProxyException) as exc:
|
||||
await TeamMemberPermissionChecks.can_team_member_execute_key_management_endpoint(
|
||||
user_api_key_dict=self._service_account_token(team_id="team-a"),
|
||||
route=KeyManagementRoutes.KEY_UPDATE,
|
||||
prisma_client=MagicMock(),
|
||||
user_api_key_cache=MagicMock(),
|
||||
existing_key_row=existing_key_row,
|
||||
)
|
||||
assert str(exc.value.code) == "401"
|
||||
assert exc.value.type == "team_member_permission_error"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_service_account_different_team_denied(self, monkeypatch):
|
||||
"""A service account key cannot manage keys in another team, even if
|
||||
that team grants the route to its members."""
|
||||
from litellm.proxy.management_helpers import (
|
||||
team_member_permission_checks as module,
|
||||
)
|
||||
|
||||
async def _mock_get_team_object(**kwargs):
|
||||
team = MagicMock()
|
||||
team.team_id = "team-b"
|
||||
team.members_with_roles = []
|
||||
team.team_member_permissions = ["/key/update"]
|
||||
return team
|
||||
|
||||
monkeypatch.setattr(module, "get_team_object", _mock_get_team_object)
|
||||
|
||||
existing_key_row = MagicMock()
|
||||
existing_key_row.team_id = "team-b"
|
||||
|
||||
with pytest.raises(ProxyException) as exc:
|
||||
await TeamMemberPermissionChecks.can_team_member_execute_key_management_endpoint(
|
||||
user_api_key_dict=self._service_account_token(team_id="team-a"),
|
||||
route=KeyManagementRoutes.KEY_UPDATE,
|
||||
prisma_client=MagicMock(),
|
||||
user_api_key_cache=MagicMock(),
|
||||
existing_key_row=existing_key_row,
|
||||
)
|
||||
assert str(exc.value.code) == "401"
|
||||
assert exc.value.type == "team_member_permission_error"
|
||||
|
|
|
|||
|
|
@ -511,3 +511,27 @@ def make_key(
|
|||
max_budget=max_budget,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def reset_login_throttle(monkeypatch):
|
||||
"""Clear the Admin UI failed-login counters between tests.
|
||||
|
||||
`client` is session scoped and the counters live in shared module stores with a 300s block
|
||||
window, so without this a failed sign-in test could block unrelated tests later.
|
||||
Only the throttle's own keys are removed, so other cache entries remain untouched.
|
||||
"""
|
||||
from litellm.constants import LOGIN_THROTTLE_CACHE_KEY_PREFIX
|
||||
from litellm.proxy import proxy_server as ps
|
||||
from litellm.proxy.auth.login_throttle import _BLOCKS, _COUNTERS
|
||||
|
||||
def _drop_throttle_keys() -> None:
|
||||
for store in (_COUNTERS, _BLOCKS):
|
||||
for key in tuple(store.cache_dict) + tuple(store.ttl_dict):
|
||||
if key.startswith(LOGIN_THROTTLE_CACHE_KEY_PREFIX):
|
||||
store.delete_cache(key)
|
||||
|
||||
monkeypatch.setattr(ps, "redis_usage_cache", None)
|
||||
_drop_throttle_keys()
|
||||
yield _drop_throttle_keys
|
||||
_drop_throttle_keys()
|
||||
|
|
|
|||
|
|
@ -12,8 +12,6 @@ from __future__ import annotations
|
|||
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from .conftest import normalize
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
|
@ -29,7 +27,7 @@ def _install_login_mocks(monkeypatch, raise_on_auth: bool = False) -> None:
|
|||
"""
|
||||
from litellm.proxy import proxy_server as ps
|
||||
|
||||
async def _fake_auth(username, password, master_key, prisma_client, general_settings=None):
|
||||
async def _fake_auth(username, password, master_key, prisma_client, throttle=None, general_settings=None):
|
||||
if raise_on_auth:
|
||||
raise Exception("boom-auth-failure")
|
||||
fake = MagicMock()
|
||||
|
|
@ -471,3 +469,222 @@ def test_login_form_ignores_open_redirect_return_to(client, monkeypatch):
|
|||
location = response.headers.get("location", "")
|
||||
assert "evil.example.com" not in location
|
||||
assert "/ui" in location # dashboard fallback
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Failed-login accounting across the login routes (LIT-5285)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _install_real_auth(monkeypatch, **settings):
|
||||
"""Run the real authenticate_user so the throttle inside it is exercised.
|
||||
|
||||
prisma_client stays None, so every guess falls through to the credential rejection.
|
||||
"""
|
||||
from litellm.proxy import proxy_server as ps
|
||||
|
||||
monkeypatch.setenv("UI_USERNAME", "admin")
|
||||
monkeypatch.setenv("UI_PASSWORD", "right-password")
|
||||
monkeypatch.setattr(ps, "master_key", "sk-test-master")
|
||||
monkeypatch.setattr(ps, "prisma_client", None)
|
||||
monkeypatch.setattr(ps, "premium_user", False)
|
||||
monkeypatch.setattr(ps, "general_settings", dict(settings))
|
||||
|
||||
|
||||
def _form_login(client, username="admin", password="wrong"):
|
||||
return client.post("/login", data={"username": username, "password": password}, follow_redirects=False).status_code
|
||||
|
||||
|
||||
def _json_login(client, path, username="admin", password="wrong"):
|
||||
return client.post(path, json={"username": username, "password": password}).status_code
|
||||
|
||||
|
||||
def _db_user(monkeypatch, email: str):
|
||||
"""A database user with a stored hash, faked so the route reaches the known-user branch without Postgres."""
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
from litellm.proxy import proxy_server as ps
|
||||
|
||||
user = MagicMock()
|
||||
user.user_id = "u-1"
|
||||
user.user_email = email
|
||||
user.user_role = "internal_user"
|
||||
user.password = "scrypt:stored"
|
||||
repo = MagicMock()
|
||||
repo.return_value.table.find_first = AsyncMock(return_value=user)
|
||||
monkeypatch.setattr(ps, "prisma_client", MagicMock())
|
||||
monkeypatch.setattr("litellm.proxy.auth.login_utils.UserRepository", repo)
|
||||
monkeypatch.setattr("litellm.proxy.auth.login_utils._rehash_password_if_needed", AsyncMock())
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.auth.login_utils.verify_password", lambda given, stored: given == "right-db-password"
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.auth.login_utils.generate_key_helper_fn", AsyncMock(return_value={"token": "sk-ui"})
|
||||
)
|
||||
monkeypatch.setenv("DATABASE_URL", "postgresql://stub")
|
||||
|
||||
|
||||
def test_budget_is_shared_across_every_login_endpoint(client, monkeypatch, reset_login_throttle):
|
||||
"""The endpoint is not part of the key, so spending the budget on one route blocks the rest.
|
||||
|
||||
Partitioning the counter per endpoint would silently triple the real allowance.
|
||||
"""
|
||||
_install_real_auth(
|
||||
monkeypatch,
|
||||
max_failed_login_attempts_per_source=20,
|
||||
control_plane_url="https://cp.example.com",
|
||||
)
|
||||
|
||||
assert [_form_login(client) for _ in range(5)] == [401] * 5
|
||||
assert [_json_login(client, "/v2/login") for _ in range(5)] == [401] * 5
|
||||
|
||||
assert _json_login(client, "/v3/login") == 401, "the eleventh failure crosses the limit and installs the block"
|
||||
assert _json_login(client, "/v3/login") == 429, "the twelfth attempt must be refused on a third route"
|
||||
|
||||
|
||||
def test_budget_is_shared_across_username_casing(client, monkeypatch, reset_login_throttle):
|
||||
"""The database lookup is case-insensitive, so casing must not partition the counter."""
|
||||
_install_real_auth(monkeypatch, max_failed_login_attempts_per_source=6)
|
||||
|
||||
assert [_json_login(client, "/v2/login", username="admin@corp.com") for _ in range(2)] == [401] * 2
|
||||
assert [_json_login(client, "/v2/login", username="ADMIN@corp.com") for _ in range(2)] == [401] * 2
|
||||
|
||||
assert _json_login(client, "/v2/login", username="Admin@corp.com") == 429
|
||||
|
||||
|
||||
def test_a_refused_attempt_carries_retry_after(client, monkeypatch, reset_login_throttle):
|
||||
"""The 429 tells the caller how long the block has left."""
|
||||
_install_real_auth(monkeypatch, max_failed_login_attempts_per_source=2, failed_login_block_seconds=77)
|
||||
|
||||
assert [_json_login(client, "/v2/login") for _ in range(2)] == [401, 401]
|
||||
|
||||
refused = client.post("/v2/login", json={"username": "admin", "password": "wrong"})
|
||||
assert refused.status_code == 429
|
||||
assert refused.headers.get("retry-after") == "77"
|
||||
|
||||
|
||||
def test_the_form_returns_a_human_readable_lockout_page(client, monkeypatch, reset_login_throttle):
|
||||
"""The no-JavaScript form must render a wait page when its POST is throttled."""
|
||||
_install_real_auth(monkeypatch, max_failed_login_attempts_per_source=2, failed_login_block_seconds=77)
|
||||
|
||||
assert [_form_login(client) for _ in range(2)] == [401, 401]
|
||||
|
||||
refused = client.post("/login", data={"username": "admin", "password": "wrong"})
|
||||
assert refused.status_code == 429
|
||||
assert refused.headers.get("content-type", "").startswith("text/html")
|
||||
assert "Try again in about 77 seconds" in refused.text
|
||||
assert refused.headers.get("retry-after") == "77"
|
||||
|
||||
|
||||
def test_a_second_username_from_the_same_source_still_gets_through(client, monkeypatch, reset_login_throttle):
|
||||
"""The pair block is per username, so one account's block cannot take the office down with it."""
|
||||
_install_real_auth(monkeypatch, max_failed_login_attempts_per_source=2)
|
||||
|
||||
assert [_json_login(client, "/v2/login", username="admin") for _ in range(3)] == [401, 401, 429]
|
||||
|
||||
assert _json_login(client, "/v2/login", username="someone-else@example.com") == 401
|
||||
|
||||
|
||||
def test_a_spray_across_usernames_is_blocked_on_the_source_when_the_source_is_attributable(
|
||||
client, monkeypatch, reset_login_throttle
|
||||
):
|
||||
"""A fresh username per guess keeps every pair at one, so the address is what stops it."""
|
||||
_install_real_auth(monkeypatch, trusted_proxy_ranges=["10.0.0.0/8"], max_failed_login_attempts_per_source=4)
|
||||
|
||||
sprayed = [_json_login(client, "/v2/login", username=f"sprayed-{i}@corp.com") for i in range(5)]
|
||||
assert sprayed == [401] * 5
|
||||
|
||||
assert _json_login(client, "/v2/login", username="sprayed-6@corp.com") == 429
|
||||
|
||||
|
||||
def test_a_spray_across_usernames_is_not_blocked_without_trusted_proxy_ranges(
|
||||
client, monkeypatch, reset_login_throttle
|
||||
):
|
||||
"""Without a configured proxy range the peer address is whoever fronts the proxy, shared by every
|
||||
client, so a source-wide block would block them all and the source scope stays off."""
|
||||
_install_real_auth(monkeypatch, max_failed_login_attempts_per_source=4)
|
||||
|
||||
sprayed = [_json_login(client, "/v2/login", username=f"sprayed-{i}@corp.com") for i in range(8)]
|
||||
assert sprayed == [401] * 8
|
||||
|
||||
|
||||
def test_a_spray_across_usernames_is_blocked_on_the_source_with_an_empty_trusted_proxy_ranges(
|
||||
client, monkeypatch, reset_login_throttle
|
||||
):
|
||||
"""An explicit empty list says nothing fronts the proxy, so the peer address is the client and the
|
||||
source scope is on. A forwarded header from an untrusted peer is ignored rather than trusted."""
|
||||
_install_real_auth(monkeypatch, trusted_proxy_ranges=[], max_failed_login_attempts_per_source=4)
|
||||
|
||||
sprayed = [
|
||||
client.post(
|
||||
"/v2/login",
|
||||
json={"username": f"sprayed-{i}@corp.com", "password": "wrong"},
|
||||
headers={"x-forwarded-for": f"203.0.113.{i}"},
|
||||
).status_code
|
||||
for i in range(5)
|
||||
]
|
||||
assert sprayed == [401] * 5
|
||||
|
||||
assert _json_login(client, "/v2/login", username="sprayed-6@corp.com") == 429
|
||||
|
||||
|
||||
def test_the_configured_admin_password_is_refused_while_blocked(client, monkeypatch, reset_login_throttle):
|
||||
"""The env credentials get no bypass: a bypass would make them the one password worth guessing without
|
||||
limit. An operator who is blocked administers the proxy with the master key over the API meanwhile."""
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
_install_real_auth(monkeypatch, max_failed_login_attempts_per_source=2)
|
||||
monkeypatch.setenv("DATABASE_URL", "postgresql://stub")
|
||||
|
||||
assert [_json_login(client, "/v2/login") for _ in range(3)] == [401, 401, 429]
|
||||
|
||||
with (
|
||||
patch( # test-quality-ok: the admin sign-in upserts the admin row; faked so no DB is needed
|
||||
"litellm.proxy.auth.login_utils.user_update", new=AsyncMock()
|
||||
),
|
||||
patch( # test-quality-ok: success mints a UI key and persists the user; faked so no DB is needed
|
||||
"litellm.proxy.auth.login_utils.generate_key_helper_fn", new=AsyncMock(return_value={"token": "sk-ui"})
|
||||
),
|
||||
):
|
||||
assert _json_login(client, "/v2/login", password="right-password") == 429
|
||||
reset_login_throttle()
|
||||
assert _json_login(client, "/v2/login", password="right-password") == 200
|
||||
|
||||
|
||||
def test_the_master_key_as_a_bearer_token_still_works_while_the_ui_password_is_blocked(
|
||||
client, monkeypatch, reset_login_throttle
|
||||
):
|
||||
"""Lockout recovery: the API path with the master key never enters the sign-in throttle."""
|
||||
_install_real_auth(monkeypatch, max_failed_login_attempts_per_source=2)
|
||||
|
||||
assert [_json_login(client, "/v2/login") for _ in range(3)] == [401, 401, 429]
|
||||
|
||||
assert client.get("/models", headers={"Authorization": "Bearer sk-not-the-master"}).status_code >= 400
|
||||
assert client.get("/models", headers={"Authorization": "Bearer sk-test-master"}).status_code == 200
|
||||
assert _json_login(client, "/v2/login", password="right-password") == 429, "the UI block is unaffected"
|
||||
|
||||
|
||||
def test_a_database_users_correct_password_is_refused_while_blocked(client, monkeypatch, reset_login_throttle):
|
||||
"""The block is hard: while it lasts, nothing from that source signs in as that user, right password or not,
|
||||
and the block is not extended by the refused attempts."""
|
||||
_install_real_auth(monkeypatch, max_failed_login_attempts_per_source=2, failed_login_block_seconds=64)
|
||||
_db_user(monkeypatch, "user@corp.com")
|
||||
|
||||
assert [_json_login(client, "/v2/login", username="user@corp.com") for _ in range(3)] == [401, 401, 429]
|
||||
|
||||
refused = client.post("/v2/login", json={"username": "user@corp.com", "password": "right-db-password"})
|
||||
assert refused.status_code == 429
|
||||
assert refused.headers.get("retry-after") == "64"
|
||||
|
||||
reset_login_throttle()
|
||||
assert _json_login(client, "/v2/login", username="user@corp.com", password="right-db-password") == 200
|
||||
|
||||
|
||||
def test_sign_in_succeeds_again_once_the_block_is_cleared(client, monkeypatch, reset_login_throttle):
|
||||
"""A cleared store lets the same username straight back to a plain credential check."""
|
||||
_install_real_auth(monkeypatch, max_failed_login_attempts_per_source=2)
|
||||
|
||||
assert [_json_login(client, "/v2/login") for _ in range(3)] == [401, 401, 429]
|
||||
|
||||
reset_login_throttle()
|
||||
assert _json_login(client, "/v2/login") == 401
|
||||
|
|
|
|||
|
|
@ -26,7 +26,6 @@ from fastapi.encoders import jsonable_encoder
|
|||
from fastapi.staticfiles import StaticFiles
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
import litellm
|
||||
import litellm.proxy.proxy_server as proxy_server_module
|
||||
from litellm.caching.caching import RedisCache
|
||||
|
|
@ -41,6 +40,7 @@ from litellm.proxy._types import (
|
|||
TokenCountRequest,
|
||||
UserAPIKeyAuth,
|
||||
)
|
||||
from litellm.proxy.auth.login_throttle import LoginThrottle
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.hooks.parallel_request_limiter_v3 import RequestRateLimiterStash
|
||||
from litellm.proxy.proxy_server import app, initialize, openai_exception_handler
|
||||
|
|
@ -139,13 +139,14 @@ def test_login_v2_returns_redirect_url_and_sets_cookie(monkeypatch):
|
|||
}
|
||||
assert response.cookies.get("token") == "signed-token"
|
||||
|
||||
mock_authenticate_user.assert_awaited_once_with(
|
||||
username="alice",
|
||||
password="secret",
|
||||
master_key="test-master-key",
|
||||
prisma_client=mock_prisma_client,
|
||||
general_settings={},
|
||||
)
|
||||
mock_authenticate_user.assert_awaited_once()
|
||||
auth_kwargs = mock_authenticate_user.call_args.kwargs
|
||||
assert auth_kwargs["username"] == "alice"
|
||||
assert auth_kwargs["password"] == "secret"
|
||||
assert auth_kwargs["master_key"] == "test-master-key"
|
||||
assert auth_kwargs["prisma_client"] is mock_prisma_client
|
||||
assert auth_kwargs["general_settings"] == {}
|
||||
assert isinstance(auth_kwargs["throttle"], LoginThrottle), "the endpoint must thread a throttle through"
|
||||
mock_create_ui_token_object.assert_called_once_with(
|
||||
login_result=mock_login_result,
|
||||
general_settings={},
|
||||
|
|
@ -3410,6 +3411,60 @@ async def test_load_config_user_url_validation_handles_null_and_string_false(tmp
|
|||
assert litellm.user_url_validation is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_load_config_warns_per_worker_login_counters_without_general_settings(tmp_path, monkeypatch, caplog):
|
||||
"""Regression: the failed-login throttle is on by default, so a multi-worker proxy with no
|
||||
Redis must hear that its counters are per worker even when the config has no general_settings."""
|
||||
import logging
|
||||
|
||||
import litellm.proxy.proxy_server as proxy_server
|
||||
from litellm.proxy.auth.login_throttle import warn_login_counters_are_per_worker
|
||||
from litellm.proxy.proxy_server import ProxyConfig
|
||||
|
||||
for redis_var in ("REDIS_HOST", "REDIS_URL", "REDIS_CLUSTER_NODES", "REDIS_SENTINEL_NODES"):
|
||||
monkeypatch.delenv(redis_var, raising=False)
|
||||
monkeypatch.setenv("NUM_WORKERS", "4")
|
||||
monkeypatch.setattr(proxy_server, "redis_usage_cache", None)
|
||||
warn_login_counters_are_per_worker.cache_clear()
|
||||
config_file = tmp_path / "config.yaml"
|
||||
config_file.write_text("model_list: []\n")
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"):
|
||||
await ProxyConfig().load_config(router=MagicMock(), config_file_path=str(config_file))
|
||||
|
||||
assert "Running 4 workers but Redis is not configured" in caplog.text
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_load_config_warns_that_the_source_login_limit_is_off_without_trusted_proxy_ranges(
|
||||
tmp_path, monkeypatch, caplog
|
||||
):
|
||||
"""The per-source failed-login limit is skipped when the source cannot be attributed, and the
|
||||
operator must be told so at startup. Both a configured range and an explicit empty list (no
|
||||
proxies, the peer is the source) silence it, since both keep the limit on."""
|
||||
import logging
|
||||
|
||||
from litellm.proxy.auth.login_throttle import warn_source_login_limit_is_off
|
||||
from litellm.proxy.proxy_server import ProxyConfig
|
||||
|
||||
monkeypatch.setenv("NUM_WORKERS", "1")
|
||||
warn_source_login_limit_is_off.cache_clear()
|
||||
config_file = tmp_path / "config.yaml"
|
||||
config_file.write_text("model_list: []\n")
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"):
|
||||
await ProxyConfig().load_config(router=MagicMock(), config_file_path=str(config_file))
|
||||
assert "trusted_proxy_ranges is not set" in caplog.text
|
||||
|
||||
for configured in ("['10.0.0.0/8']", "[]"):
|
||||
caplog.clear()
|
||||
warn_source_login_limit_is_off.cache_clear()
|
||||
config_file.write_text(f"model_list: []\ngeneral_settings:\n trusted_proxy_ranges: {configured}\n")
|
||||
with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"):
|
||||
await ProxyConfig().load_config(router=MagicMock(), config_file_path=str(config_file))
|
||||
assert "trusted_proxy_ranges is not set" not in caplog.text, configured
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_load_environment_variables_direct_and_os_environ():
|
||||
"""
|
||||
|
|
@ -4944,6 +4999,69 @@ async def test_add_router_settings_from_db_config_merge_logic():
|
|||
assert combined_settings["retry_delay"] == 2
|
||||
|
||||
|
||||
def _routing_groups_router():
|
||||
from litellm import Router
|
||||
|
||||
return Router(
|
||||
model_list=[
|
||||
{"model_name": "m1", "litellm_params": {"model": "openai/gpt-4o", "api_key": "sk-test"}},
|
||||
{"model_name": "m2", "litellm_params": {"model": "openai/gpt-4o-mini", "api_key": "sk-test"}},
|
||||
],
|
||||
routing_groups=[{"group_name": "g1", "models": ["m1"], "routing_strategy": "latency-based-routing"}],
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_invalid_db_routing_groups_do_not_abort_other_router_settings():
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
from litellm.proxy.proxy_server import ProxyConfig
|
||||
|
||||
router = _routing_groups_router()
|
||||
mock_db_config = MagicMock()
|
||||
mock_db_config.param_value = {
|
||||
"num_retries": 7,
|
||||
"routing_groups": [
|
||||
{"group_name": "g1", "models": ["m1"], "routing_strategy": "latency-based-routing"},
|
||||
{"group_name": "g2", "models": ["m1"], "routing_strategy": "least-busy"},
|
||||
],
|
||||
}
|
||||
mock_prisma_client = MagicMock()
|
||||
mock_prisma_client.db.litellm_config.find_first = AsyncMock(return_value=mock_db_config)
|
||||
|
||||
await ProxyConfig()._add_router_settings_from_db_config(
|
||||
config_data={}, llm_router=router, prisma_client=mock_prisma_client
|
||||
)
|
||||
|
||||
assert router.num_retries == 7
|
||||
assert router._model_to_group == {"m1": "g1"}
|
||||
assert router._get_routing_context("m1", None)[0] == "latency-based-routing"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_valid_db_routing_groups_still_replace_router_groups():
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
from litellm.proxy.proxy_server import ProxyConfig
|
||||
|
||||
router = _routing_groups_router()
|
||||
mock_db_config = MagicMock()
|
||||
mock_db_config.param_value = {
|
||||
"num_retries": 7,
|
||||
"routing_groups": [{"group_name": "g2", "models": ["m2"], "routing_strategy": "least-busy"}],
|
||||
}
|
||||
mock_prisma_client = MagicMock()
|
||||
mock_prisma_client.db.litellm_config.find_first = AsyncMock(return_value=mock_db_config)
|
||||
|
||||
await ProxyConfig()._add_router_settings_from_db_config(
|
||||
config_data={}, llm_router=router, prisma_client=mock_prisma_client
|
||||
)
|
||||
|
||||
assert router.num_retries == 7
|
||||
assert router._model_to_group == {"m2": "g2"}
|
||||
assert router._get_routing_context("m2", None)[0] == "least-busy"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_add_router_settings_from_db_config_empty_db_lists_do_not_clobber_config_fallbacks():
|
||||
"""
|
||||
|
|
@ -9275,6 +9393,50 @@ def test_update_config_writes_only_sent_section(_update_config_setup):
|
|||
restore()
|
||||
|
||||
|
||||
def test_update_config_rejects_overlapping_routing_groups_before_writing(_update_config_setup):
|
||||
existing_groups = [{"group_name": "g1", "models": ["m1"], "routing_strategy": "least-busy"}]
|
||||
client, prisma, restore = _update_config_setup(
|
||||
initial_rows={"router_settings": {"num_retries": 2, "routing_groups": existing_groups}}
|
||||
)
|
||||
try:
|
||||
resp = client.post(
|
||||
"/config/update",
|
||||
json={
|
||||
"router_settings": {
|
||||
"routing_groups": [
|
||||
*existing_groups,
|
||||
{"group_name": "g2", "models": ["m1"], "routing_strategy": "latency-based-routing"},
|
||||
]
|
||||
}
|
||||
},
|
||||
)
|
||||
assert resp.status_code == 400
|
||||
assert "'m1' appears in 'g1' and 'g2'" in resp.text
|
||||
assert prisma.db.litellm_config.upsert_calls == []
|
||||
assert prisma.db.litellm_config.rows["router_settings"]["routing_groups"] == existing_groups
|
||||
finally:
|
||||
restore()
|
||||
|
||||
|
||||
def test_update_config_accepts_disjoint_routing_groups(_update_config_setup):
|
||||
client, prisma, restore = _update_config_setup(initial_rows={"router_settings": {"num_retries": 2}})
|
||||
groups = [
|
||||
{"group_name": "g1", "models": ["m1"], "routing_strategy": "least-busy"},
|
||||
{"group_name": "g2", "models": ["m2"], "routing_strategy": "latency-based-routing"},
|
||||
]
|
||||
try:
|
||||
resp = client.post("/config/update", json={"router_settings": {"routing_groups": groups}})
|
||||
assert resp.status_code == 200
|
||||
stored = prisma.db.litellm_config.rows["router_settings"]
|
||||
assert stored["num_retries"] == 2
|
||||
assert [(g["group_name"], g["models"]) for g in stored["routing_groups"]] == [
|
||||
("g1", ["m1"]),
|
||||
("g2", ["m2"]),
|
||||
]
|
||||
finally:
|
||||
restore()
|
||||
|
||||
|
||||
def test_update_config_env_var_round_trip_not_double_encrypted(_update_config_setup, monkeypatch):
|
||||
"""Endpoint-level regression for the /config/update double-encryption bug.
|
||||
|
||||
|
|
@ -13845,6 +14007,35 @@ async def test_authoritative_floor_spend_keeps_a_reset_marker_written_during_the
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_login_throttle_settings_are_not_hot_applied_from_the_database():
|
||||
"""LIT-5285: a stored sign-in limit does not take effect on a live worker.
|
||||
|
||||
_update_general_settings copies an allowlist of keys out of the DB row on every config
|
||||
poll. Adding these to it would let a stored value outrank config.yaml without a restart,
|
||||
so an operator locked out by a bad value could not fix it by editing YAML and restarting.
|
||||
"""
|
||||
import litellm.proxy.proxy_server as ps
|
||||
from litellm.proxy.proxy_server import ProxyConfig
|
||||
|
||||
original = dict(ps.general_settings)
|
||||
try:
|
||||
ps.general_settings.clear()
|
||||
await ProxyConfig()._update_general_settings(
|
||||
db_general_settings={
|
||||
"max_failed_login_attempts_per_source": 999,
|
||||
"failed_login_window_seconds": 1,
|
||||
"failed_login_block_seconds": 1,
|
||||
}
|
||||
)
|
||||
assert "max_failed_login_attempts_per_source" not in ps.general_settings
|
||||
assert "failed_login_window_seconds" not in ps.general_settings
|
||||
assert "failed_login_block_seconds" not in ps.general_settings
|
||||
finally:
|
||||
ps.general_settings.clear()
|
||||
ps.general_settings.update(original)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_load_config_router_authorizes_fallback_targets_against_the_calling_key(tmp_path):
|
||||
from litellm.proxy.auth.fallback_model_access import router_fallback_access_check
|
||||
|
|
@ -14196,3 +14387,74 @@ async def test_token_counter_loads_a_custom_tokenizer_once_per_identifier_revisi
|
|||
]
|
||||
finally:
|
||||
litellm.utils._select_custom_tokenizer_helper.cache_clear()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_auth_cache_invalidation_subscriber_evicts_byok_credentials_cached_by_this_worker():
|
||||
"""A peer worker's BYOK revocation broadcast must reach this worker's BYOK credential cache."""
|
||||
from redis.asyncio import Redis
|
||||
|
||||
from litellm.proxy._experimental.mcp_server.byok_credential_cache import (
|
||||
byok_credential_cache,
|
||||
byok_credential_cache_key,
|
||||
cache_byok_credential,
|
||||
get_cached_byok_credential,
|
||||
)
|
||||
from litellm.proxy.common_utils.user_api_key_cache import UserApiKeyCache
|
||||
|
||||
class _QueuePubSub:
|
||||
def __init__(self, messages: list[object]) -> None:
|
||||
self.queue: asyncio.Queue[object] = asyncio.Queue()
|
||||
for message in messages:
|
||||
self.queue.put_nowait(message)
|
||||
|
||||
async def subscribe(self, *channels: str) -> None:
|
||||
return None
|
||||
|
||||
async def get_message(self, *, ignore_subscribe_messages: bool, timeout: float) -> object | None:
|
||||
try:
|
||||
return await asyncio.wait_for(self.queue.get(), timeout)
|
||||
except asyncio.TimeoutError:
|
||||
return None
|
||||
|
||||
async def aclose(self) -> None:
|
||||
return None
|
||||
|
||||
class _PubSubRedisClient(Redis):
|
||||
def __init__(self, pubsub: _QueuePubSub) -> None:
|
||||
self._scripted_pubsub = pubsub
|
||||
|
||||
def pubsub(self) -> _QueuePubSub:
|
||||
return self._scripted_pubsub
|
||||
|
||||
class _FakeRedisCache:
|
||||
namespace = None
|
||||
|
||||
def __init__(self, client: object) -> None:
|
||||
self._client = client
|
||||
|
||||
def init_async_client(self) -> object:
|
||||
return self._client
|
||||
|
||||
byok_credential_cache.flush_cache()
|
||||
cache_byok_credential("mallory", "srv-byok", "sk-revoked-elsewhere")
|
||||
message: Final = {
|
||||
"type": "message",
|
||||
"data": json.dumps({"cache_key": byok_credential_cache_key("mallory", "srv-byok")}).encode(),
|
||||
}
|
||||
proxy_config: Final = proxy_server_module.ProxyConfig()
|
||||
proxy_config.start_auth_cache_invalidation_subscriber(
|
||||
redis_cache=_FakeRedisCache(_PubSubRedisClient(_QueuePubSub([message]))), # pyright: ignore[reportArgumentType] # fake pub/sub capable redis; no live redis in this unit test
|
||||
user_api_key_cache=UserApiKeyCache(),
|
||||
)
|
||||
try:
|
||||
for _ in range(200):
|
||||
if get_cached_byok_credential("mallory", "srv-byok") is None:
|
||||
break
|
||||
await asyncio.sleep(0.01)
|
||||
evicted: Final = get_cached_byok_credential("mallory", "srv-byok") is None
|
||||
finally:
|
||||
await proxy_config.stop_auth_cache_invalidation_subscriber()
|
||||
byok_credential_cache.flush_cache()
|
||||
|
||||
assert evicted, "the subscriber does not evict the BYOK credential cache on a peer worker's broadcast"
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ from collections.abc import Callable
|
|||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from pydantic import ValidationError
|
||||
|
||||
import litellm
|
||||
from litellm import Router
|
||||
|
|
@ -806,6 +807,165 @@ def test_strategy_reinit_unregisters_override_selectors():
|
|||
assert router._get_override_strategy_selector("latency-based-routing") is router.lowestlatency_logger
|
||||
|
||||
|
||||
def _single_latency_group():
|
||||
return [{"group_name": "g1", "models": ["filtered-model"], "routing_strategy": "latency-based-routing"}]
|
||||
|
||||
|
||||
def _assert_still_routes_with_original_group(router, selector):
|
||||
assert list(router._routing_groups) == ["g1"]
|
||||
assert router._model_to_group == {"filtered-model": "g1"}
|
||||
assert router._group_selectors["g1"]["latency-based-routing"] is selector
|
||||
assert router._get_routing_context("filtered-model", None) == ("latency-based-routing", selector)
|
||||
assert sum(1 for cb in litellm.callbacks if cb is selector) == 1
|
||||
|
||||
|
||||
def test_failed_routing_groups_update_keeps_previous_groups(monkeypatch):
|
||||
monkeypatch.setattr(litellm, "callbacks", [])
|
||||
monkeypatch.setattr(litellm, "input_callback", [])
|
||||
router = _build_router(routing_groups=_single_latency_group())
|
||||
selector = router._group_selectors["g1"]["latency-based-routing"]
|
||||
|
||||
with pytest.raises(ValueError, match="appears in"):
|
||||
router.update_settings(
|
||||
routing_groups=[
|
||||
*_single_latency_group(),
|
||||
{"group_name": "g2", "models": ["filtered-model"], "routing_strategy": "least-busy"},
|
||||
],
|
||||
)
|
||||
|
||||
_assert_still_routes_with_original_group(router, selector)
|
||||
assert sum(1 for cb in litellm.callbacks if type(cb) is not type(selector)) == 0
|
||||
assert litellm.input_callback == []
|
||||
|
||||
|
||||
def test_failed_routing_groups_update_does_not_poison_later_strategy_changes(monkeypatch):
|
||||
monkeypatch.setattr(litellm, "callbacks", [])
|
||||
monkeypatch.setattr(litellm, "input_callback", [])
|
||||
router = _build_router(routing_groups=_single_latency_group())
|
||||
|
||||
with pytest.raises(ValueError, match="appears in"):
|
||||
router.update_settings(
|
||||
routing_groups=[
|
||||
*_single_latency_group(),
|
||||
{"group_name": "g2", "models": ["filtered-model"], "routing_strategy": "least-busy"},
|
||||
],
|
||||
)
|
||||
|
||||
router.update_settings(routing_strategy="least-busy")
|
||||
|
||||
assert list(router._routing_groups) == ["g1"]
|
||||
assert [g["group_name"] for g in router.get_settings()["routing_groups"]] == ["g1"]
|
||||
|
||||
|
||||
def test_overlap_error_names_every_conflicting_model():
|
||||
with pytest.raises(ValueError, match="appears in") as exc_info:
|
||||
_build_router(
|
||||
routing_groups=[
|
||||
{
|
||||
"group_name": "g1",
|
||||
"models": ["filtered-model", "other-model"],
|
||||
"routing_strategy": "latency-based-routing",
|
||||
},
|
||||
{
|
||||
"group_name": "g2",
|
||||
"models": ["filtered-model", "other-model"],
|
||||
"routing_strategy": "least-busy",
|
||||
},
|
||||
],
|
||||
)
|
||||
message = str(exc_info.value)
|
||||
assert "'filtered-model' appears in 'g1' and 'g2'" in message
|
||||
assert "'other-model' appears in 'g1' and 'g2'" in message
|
||||
|
||||
|
||||
def test_invalid_group_strategy_keeps_previous_groups(monkeypatch):
|
||||
monkeypatch.setattr(litellm, "callbacks", [])
|
||||
monkeypatch.setattr(litellm, "input_callback", [])
|
||||
router = _build_router(routing_groups=_single_latency_group())
|
||||
selector = router._group_selectors["g1"]["latency-based-routing"]
|
||||
|
||||
with pytest.raises(ValueError, match="Invalid routing_strategy"):
|
||||
router.update_settings(
|
||||
routing_groups=[
|
||||
{"group_name": "g2", "models": ["other-model"], "routing_strategy": "not-a-real-strategy"},
|
||||
],
|
||||
)
|
||||
|
||||
_assert_still_routes_with_original_group(router, selector)
|
||||
|
||||
|
||||
def test_unbuildable_group_selector_keeps_previous_groups(monkeypatch):
|
||||
monkeypatch.setattr(litellm, "callbacks", [])
|
||||
monkeypatch.setattr(litellm, "input_callback", [])
|
||||
router = _build_router(routing_groups=_single_latency_group())
|
||||
selector = router._group_selectors["g1"]["latency-based-routing"]
|
||||
|
||||
with pytest.raises(ValidationError, match="ttl"):
|
||||
router.update_settings(
|
||||
routing_groups=[
|
||||
{"group_name": "g0", "models": ["other-model"], "routing_strategy": "least-busy"},
|
||||
*_single_latency_group(),
|
||||
{
|
||||
"group_name": "g2",
|
||||
"models": ["other-model-2"],
|
||||
"routing_strategy": "latency-based-routing",
|
||||
"routing_strategy_args": {"ttl": "not-a-number"},
|
||||
},
|
||||
],
|
||||
)
|
||||
|
||||
_assert_still_routes_with_original_group(router, selector)
|
||||
assert litellm.callbacks == [selector]
|
||||
assert litellm.input_callback == []
|
||||
|
||||
|
||||
def test_register_router_selector_wires_only_the_hooks_the_strategy_needs(monkeypatch):
|
||||
monkeypatch.setattr(litellm, "callbacks", [])
|
||||
monkeypatch.setattr(litellm, "input_callback", [])
|
||||
router = _build_router()
|
||||
least_busy = router._build_strategy_selector(
|
||||
strategy="least-busy", routing_strategy_args={}, register_callbacks=False
|
||||
)
|
||||
latency = router._build_strategy_selector(
|
||||
strategy="latency-based-routing", routing_strategy_args={}, register_callbacks=False
|
||||
)
|
||||
assert least_busy is not None and latency is not None
|
||||
assert litellm.callbacks == [] and litellm.input_callback == []
|
||||
|
||||
router._register_router_selector(least_busy)
|
||||
router._register_router_selector(latency)
|
||||
|
||||
assert [cb for cb in litellm.callbacks if cb is least_busy or cb is latency] == [least_busy, latency]
|
||||
assert litellm.input_callback == [least_busy]
|
||||
|
||||
|
||||
def test_replace_routing_groups_swaps_state_and_callbacks_in_one_step(monkeypatch):
|
||||
monkeypatch.setattr(litellm, "callbacks", [])
|
||||
monkeypatch.setattr(litellm, "input_callback", [])
|
||||
router = _build_router(routing_groups=_single_latency_group())
|
||||
old_selector = router._group_selectors["g1"]["latency-based-routing"]
|
||||
new_selector = router._build_strategy_selector(
|
||||
strategy="least-busy", routing_strategy_args={}, register_callbacks=False
|
||||
)
|
||||
assert new_selector is not None
|
||||
|
||||
router._replace_routing_groups(
|
||||
(
|
||||
(RoutingGroup(group_name="g2", models=["other-model"], routing_strategy="least-busy"), new_selector),
|
||||
(RoutingGroup(group_name="g3", models=["other-model-2"], routing_strategy="simple-shuffle"), None),
|
||||
)
|
||||
)
|
||||
|
||||
assert list(router._routing_groups) == ["g2", "g3"]
|
||||
assert router._model_to_group == {"other-model": "g2", "other-model-2": "g3"}
|
||||
assert router._group_selectors == {"g2": {"least-busy": new_selector}, "g3": {}}
|
||||
assert router._get_routing_context("other-model", None) == ("least-busy", new_selector)
|
||||
assert router._get_routing_context("filtered-model", None)[0] == router.routing_strategy
|
||||
assert all(cb is not old_selector for cb in litellm.callbacks)
|
||||
assert sum(1 for cb in litellm.callbacks if cb is new_selector) == 1
|
||||
assert litellm.input_callback == [new_selector]
|
||||
|
||||
|
||||
def test_override_selectors_are_not_registered_process_wide(monkeypatch):
|
||||
monkeypatch.setattr(litellm, "callbacks", [])
|
||||
monkeypatch.setattr(litellm, "input_callback", [])
|
||||
|
|
|
|||
|
|
@ -426,6 +426,22 @@ class TestAnthropicBetaHeadersFiltering:
|
|||
|
||||
assert filtered == ["fine-grained-tool-streaming-2025-05-14"]
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"provider", ["anthropic", "bedrock", "bedrock_converse", "vertex_ai", "databricks"]
|
||||
)
|
||||
def test_thinking_binding_controls_forwarded(self, provider):
|
||||
"""`thinking.block_binding` (preserved thinking, Claude Fable 5.1) is only
|
||||
accepted alongside thinking-binding-controls-2026-08-01. The body field is
|
||||
forwarded untouched, so stripping the header (previously unknown, hence
|
||||
dropped) makes Bedrock and Vertex reject the request with
|
||||
"thinking.adaptive.block_binding: Extra inputs are not permitted"."""
|
||||
filtered = filter_and_transform_beta_headers(
|
||||
beta_headers=["thinking-binding-controls-2026-08-01"],
|
||||
provider=provider,
|
||||
)
|
||||
|
||||
assert filtered == ["thinking-binding-controls-2026-08-01"]
|
||||
|
||||
def test_null_value_headers_filtered(self):
|
||||
"""Test that headers with null values are always filtered out."""
|
||||
for provider in [
|
||||
|
|
|
|||
46
tests/test_litellm/types/llms/test_types_llms_bedrock.py
Normal file
46
tests/test_litellm/types/llms/test_types_llms_bedrock.py
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
import pytest
|
||||
from pydantic import ValidationError
|
||||
|
||||
from litellm.types.llms.bedrock import AWS_AUTH_PARAM_KEYS, AwsAuthParams
|
||||
|
||||
|
||||
def test_model_validate_keeps_auth_params_and_ignores_request_params():
|
||||
auth_params = AwsAuthParams.model_validate(
|
||||
{
|
||||
"aws_role_name": "arn:aws:iam::999999999999:role/litellm-role",
|
||||
"aws_session_name": "litellm-session",
|
||||
"aws_external_id": "litellm-external-id",
|
||||
"aws_region_name": "us-west-2",
|
||||
"aws_bedrock_runtime_endpoint": "https://bedrock.example.com",
|
||||
"model": "anthropic.claude-haiku-4-5-20251001-v1:0",
|
||||
"temperature": 0.1,
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
}
|
||||
)
|
||||
|
||||
assert auth_params.aws_role_name == "arn:aws:iam::999999999999:role/litellm-role"
|
||||
assert auth_params.aws_session_name == "litellm-session"
|
||||
assert auth_params.aws_external_id == "litellm-external-id"
|
||||
assert auth_params.aws_access_key_id is None
|
||||
assert set(auth_params.model_dump()) == set(AWS_AUTH_PARAM_KEYS)
|
||||
assert not set(AWS_AUTH_PARAM_KEYS) & {"aws_region_name", "aws_bedrock_runtime_endpoint", "model", "temperature"}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("field", "value"),
|
||||
[
|
||||
("aws_role_name", 1234),
|
||||
("aws_session_name", ["litellm-session"]),
|
||||
("aws_external_id", {"id": "x"}),
|
||||
],
|
||||
)
|
||||
def test_model_validate_rejects_non_string_credentials(field, value):
|
||||
with pytest.raises(ValidationError):
|
||||
AwsAuthParams.model_validate({field: value})
|
||||
|
||||
|
||||
def test_frozen_struct_rejects_field_assignment():
|
||||
auth_params = AwsAuthParams(aws_role_name="arn:aws:iam::999999999999:role/litellm-role")
|
||||
|
||||
with pytest.raises(ValidationError):
|
||||
auth_params.aws_role_name = "arn:aws:iam::999999999999:role/other-role"
|
||||
|
|
@ -178,4 +178,28 @@ describe("CacheLeakageCard", () => {
|
|||
screen.queryByText("Data is still loading; rows and totals will update as the rest of the range arrives."),
|
||||
).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("says which keys are missing from the key ranking when the proxy capped the per-key lists", () => {
|
||||
const day = dayWithKeys("2026-07-12", {
|
||||
"hash-leaky": key("leaky-key", { prompt_tokens: 10000, cache_read_input_tokens: 0 }),
|
||||
});
|
||||
renderWith([day], { apiKeyTruncation: { limit: 100, total: 3000 } });
|
||||
|
||||
expect(screen.getByRole("note")).toHaveTextContent(
|
||||
"Only the 100 highest-spend keys of 3,000 are loaded, so a lower-spend key that leaks more is not listed here.",
|
||||
);
|
||||
|
||||
fireEvent.click(screen.getByRole("tab", { name: "By model" }));
|
||||
|
||||
expect(screen.queryByRole("note")).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("keeps the key ranking note off when every key was loaded", () => {
|
||||
const day = dayWithKeys("2026-07-12", {
|
||||
"hash-leaky": key("leaky-key", { prompt_tokens: 10000, cache_read_input_tokens: 0 }),
|
||||
});
|
||||
renderWith([day]);
|
||||
|
||||
expect(screen.queryByRole("note")).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -81,7 +81,7 @@ const SortableHead = ({
|
|||
};
|
||||
|
||||
const CacheLeakageCard: React.FC<CacheLeakageCardProps> = ({ activity }) => {
|
||||
const { dateValue, onDateChange, results, loading, isFetchingMore } = activity;
|
||||
const { dateValue, onDateChange, results, loading, isFetchingMore, apiKeyTruncation } = activity;
|
||||
const [dimension, setDimension] = useState<CacheLeakageDimension>("key");
|
||||
const [sort, setSort] = useState<SortState>({ column: "potentialSavings", dir: "desc" });
|
||||
const leakage = useMemo(() => computeCacheLeakage(results, dimension), [results, dimension]);
|
||||
|
|
@ -123,6 +123,13 @@ const CacheLeakageCard: React.FC<CacheLeakageCardProps> = ({ activity }) => {
|
|||
</Tabs>
|
||||
</CardHeader>
|
||||
<CardContent>
|
||||
{dimension === "key" && apiKeyTruncation !== undefined && (
|
||||
<p className="mb-2 text-sm text-muted-foreground" role="note">
|
||||
Only the {apiKeyTruncation.limit.toLocaleString()} highest-spend keys of{" "}
|
||||
{apiKeyTruncation.total.toLocaleString()} are loaded, so a lower-spend key that leaks more is not listed
|
||||
here. Raise USAGE_TOP_API_KEYS_LIMIT on the proxy to load more keys.
|
||||
</p>
|
||||
)}
|
||||
{rows.length > 0 && isFetchingMore && (
|
||||
<p className="mb-2 text-sm text-muted-foreground">
|
||||
Data is still loading; rows and totals will update as the rest of the range arrives.
|
||||
|
|
|
|||
|
|
@ -4,12 +4,13 @@ import { describe, expect, it, vi } from "vitest";
|
|||
const mockUsePaginatedDailyActivity = vi.fn();
|
||||
|
||||
const mockCancel = vi.fn();
|
||||
let mockMetadata: Record<string, number> = {};
|
||||
|
||||
vi.mock("@/app/(dashboard)/usage/_components/hooks/usePaginatedDailyActivity", () => ({
|
||||
usePaginatedDailyActivity: (args: unknown) => {
|
||||
mockUsePaginatedDailyActivity(args);
|
||||
return {
|
||||
data: { results: [] },
|
||||
data: { results: [], metadata: mockMetadata },
|
||||
loading: false,
|
||||
isFetchingMore: false,
|
||||
progress: { currentPage: 4, totalPages: 9 },
|
||||
|
|
@ -80,4 +81,18 @@ describe("useDailyActivityRange", () => {
|
|||
|
||||
expect(mockUsePaginatedDailyActivity).toHaveBeenLastCalledWith(expect.objectContaining({ enabled: false }));
|
||||
});
|
||||
|
||||
it("reports how many keys the proxy left out of the per-key lists", () => {
|
||||
mockMetadata = { api_key_limit: 100, total_api_keys: 3000 };
|
||||
const { result } = renderHook(() => useDailyActivityRange("test-token", "u1", "proxy_admin"));
|
||||
|
||||
expect(result.current.apiKeyTruncation).toEqual({ limit: 100, total: 3000 });
|
||||
});
|
||||
|
||||
it("reports no key truncation when every key fit under the proxy limit", () => {
|
||||
mockMetadata = { api_key_limit: 100, total_api_keys: 100 };
|
||||
const { result } = renderHook(() => useDailyActivityRange("test-token", "u1", "proxy_admin"));
|
||||
|
||||
expect(result.current.apiKeyTruncation).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import { useMemo, useState } from "react";
|
||||
|
||||
import { userDailyActivityAggregatedCall, userDailyActivityCall } from "@/components/networking";
|
||||
import { ApiKeyTruncation, getApiKeyTruncation } from "@/components/EntityUsageExport/exportBlockedReason";
|
||||
import { DailyData } from "@/components/UsagePage/types";
|
||||
import { spendScopeUserId } from "@/utils/roles";
|
||||
import { usePaginatedDailyActivity } from "@/app/(dashboard)/usage/_components/hooks/usePaginatedDailyActivity";
|
||||
|
|
@ -22,6 +23,7 @@ export interface DailyActivityRange {
|
|||
cancelled: boolean;
|
||||
failed: boolean;
|
||||
cancel: () => void;
|
||||
apiKeyTruncation?: ApiKeyTruncation;
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -78,6 +80,7 @@ export const useScopedDailyActivityRange = (
|
|||
cancelled,
|
||||
failed,
|
||||
cancel,
|
||||
apiKeyTruncation: getApiKeyTruncation(data.metadata?.api_key_limit, data.metadata?.total_api_keys),
|
||||
};
|
||||
};
|
||||
|
||||
|
|
|
|||
|
|
@ -1,13 +1,15 @@
|
|||
import React from "react";
|
||||
import { render, screen, within } from "@testing-library/react";
|
||||
import userEvent from "@testing-library/user-event";
|
||||
import { describe, it, expect, vi, beforeEach } from "vitest";
|
||||
import { QueryClient, QueryClientProvider } from "@tanstack/react-query";
|
||||
import { MCPGatewaySessionsTab, formatIdleSeconds } from "./MCPGatewaySessionsTab";
|
||||
import { MCPGatewaySessionsTab, describeTerminateResult, formatIdleSeconds } from "./MCPGatewaySessionsTab";
|
||||
import * as networking from "@/components/networking";
|
||||
import type { MCPGatewaySessionsResponse } from "@/components/mcp_tools/types";
|
||||
import type { MCPGatewaySessionsResponse, MCPGatewaySessionsTerminateResponse } from "@/components/mcp_tools/types";
|
||||
|
||||
vi.mock("@/components/networking", () => ({
|
||||
fetchMCPGatewaySessions: vi.fn(),
|
||||
terminateMCPGatewaySessions: vi.fn(),
|
||||
}));
|
||||
|
||||
const REPORT: MCPGatewaySessionsResponse = {
|
||||
|
|
@ -64,11 +66,11 @@ const REPORT: MCPGatewaySessionsResponse = {
|
|||
],
|
||||
};
|
||||
|
||||
const renderTab = () => {
|
||||
const renderTab = ({ canTerminate = false }: { canTerminate?: boolean } = {}) => {
|
||||
const queryClient = new QueryClient({ defaultOptions: { queries: { retry: false, gcTime: 0 } } });
|
||||
return render(
|
||||
<QueryClientProvider client={queryClient}>
|
||||
<MCPGatewaySessionsTab accessToken="token" />
|
||||
<MCPGatewaySessionsTab accessToken="token" canTerminate={canTerminate} />
|
||||
</QueryClientProvider>,
|
||||
);
|
||||
};
|
||||
|
|
@ -83,6 +85,17 @@ describe("formatIdleSeconds", () => {
|
|||
});
|
||||
});
|
||||
|
||||
describe("describeTerminateResult", () => {
|
||||
it("pluralizes the session count and names the worker", () => {
|
||||
expect(describeTerminateResult({ worker_pid: 9, terminated_sessions: 1, sessions: [] })).toBe(
|
||||
"Disconnected 1 session on worker pid 9.",
|
||||
);
|
||||
expect(describeTerminateResult({ worker_pid: 9, terminated_sessions: 0, sessions: [] })).toBe(
|
||||
"Disconnected 0 sessions on worker pid 9.",
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe("MCPGatewaySessionsTab", () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
|
|
@ -135,4 +148,73 @@ describe("MCPGatewaySessionsTab", () => {
|
|||
expect(alert).toHaveTextContent("Could not load live connections");
|
||||
expect(alert).toHaveTextContent("Admin access required");
|
||||
});
|
||||
|
||||
it("hides every disconnect control from a read-only admin", async () => {
|
||||
vi.mocked(networking.fetchMCPGatewaySessions).mockResolvedValue(REPORT);
|
||||
renderTab({ canTerminate: false });
|
||||
|
||||
await screen.findByRole("region", { name: "Live sessions" });
|
||||
expect(screen.queryByRole("button", { name: /^Disconnect/ })).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("disconnects one session by its displayed prefix after confirmation and refetches", async () => {
|
||||
const user = userEvent.setup();
|
||||
const terminated: MCPGatewaySessionsTerminateResponse = {
|
||||
worker_pid: 4242,
|
||||
terminated_sessions: 1,
|
||||
sessions: [REPORT.sessions[1]],
|
||||
};
|
||||
vi.mocked(networking.fetchMCPGatewaySessions).mockResolvedValue(REPORT);
|
||||
vi.mocked(networking.terminateMCPGatewaySessions).mockResolvedValue(terminated);
|
||||
renderTab({ canTerminate: true });
|
||||
|
||||
await user.click(await screen.findByRole("button", { name: "Disconnect session bbbb2222" }));
|
||||
expect(networking.terminateMCPGatewaySessions).not.toHaveBeenCalled();
|
||||
const dialog = await screen.findByRole("alertdialog");
|
||||
expect(dialog).toHaveTextContent("session bbbb2222");
|
||||
await user.click(within(dialog).getByRole("button", { name: "Disconnect" }));
|
||||
|
||||
const status = await screen.findByText("Disconnected 1 session on worker pid 4242.", { exact: false });
|
||||
expect(status).toBeInTheDocument();
|
||||
expect(networking.terminateMCPGatewaySessions).toHaveBeenCalledWith("token", { session_id_prefix: "bbbb2222" });
|
||||
expect(networking.fetchMCPGatewaySessions).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it("disconnects every session of a user from the by-user table", async () => {
|
||||
const user = userEvent.setup();
|
||||
vi.mocked(networking.fetchMCPGatewaySessions).mockResolvedValue(REPORT);
|
||||
vi.mocked(networking.terminateMCPGatewaySessions).mockResolvedValue({
|
||||
worker_pid: 4242,
|
||||
terminated_sessions: 2,
|
||||
sessions: [REPORT.sessions[0], REPORT.sessions[1]],
|
||||
});
|
||||
renderTab({ canTerminate: true });
|
||||
|
||||
const byUser = await screen.findByRole("region", { name: "Sessions by user" });
|
||||
expect(within(byUser).queryByRole("button", { name: /\(unknown\)/ })).not.toBeInTheDocument();
|
||||
await user.click(within(byUser).getByRole("button", { name: "Disconnect all sessions for user alice" }));
|
||||
const dialog = await screen.findByRole("alertdialog");
|
||||
expect(dialog).toHaveTextContent("every live session opened by user alice");
|
||||
await user.click(within(dialog).getByRole("button", { name: "Disconnect" }));
|
||||
|
||||
expect(await screen.findByText(/Disconnected 2 sessions on worker pid 4242\./)).toBeInTheDocument();
|
||||
expect(networking.terminateMCPGatewaySessions).toHaveBeenCalledWith("token", { user_id: "alice" });
|
||||
});
|
||||
|
||||
it("shows the API error when a disconnect is refused", async () => {
|
||||
const user = userEvent.setup();
|
||||
vi.mocked(networking.fetchMCPGatewaySessions).mockResolvedValue(REPORT);
|
||||
vi.mocked(networking.terminateMCPGatewaySessions).mockRejectedValue(
|
||||
new Error("Proxy admin access required to terminate MCP gateway sessions."),
|
||||
);
|
||||
renderTab({ canTerminate: true });
|
||||
|
||||
await user.click(await screen.findByRole("button", { name: "Disconnect session aaaa1111" }));
|
||||
await user.click(within(await screen.findByRole("alertdialog")).getByRole("button", { name: "Disconnect" }));
|
||||
|
||||
const alert = await screen.findByRole("alert");
|
||||
expect(alert).toHaveTextContent("Could not disconnect");
|
||||
expect(alert).toHaveTextContent("Proxy admin access required to terminate MCP gateway sessions.");
|
||||
expect(screen.getByRole("region", { name: "Live sessions" })).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1,14 +1,27 @@
|
|||
"use client";
|
||||
|
||||
import React from "react";
|
||||
import { useQuery } from "@tanstack/react-query";
|
||||
import { RefreshCw } from "lucide-react";
|
||||
import React, { useState } from "react";
|
||||
import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query";
|
||||
import { RefreshCw, Unplug } from "lucide-react";
|
||||
import { Alert, AlertDescription, AlertTitle } from "@/components/ui/alert";
|
||||
import {
|
||||
AlertDialog,
|
||||
AlertDialogContent,
|
||||
AlertDialogDescription,
|
||||
AlertDialogFooter,
|
||||
AlertDialogHeader,
|
||||
AlertDialogTitle,
|
||||
} from "@/components/ui/alert-dialog";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui/table";
|
||||
import { UiLoadingSpinner } from "@/components/ui/ui-loading-spinner";
|
||||
import { fetchMCPGatewaySessions } from "@/components/networking";
|
||||
import type { MCPGatewaySessionGroupCount, MCPGatewaySessionsResponse } from "@/components/mcp_tools/types";
|
||||
import { fetchMCPGatewaySessions, terminateMCPGatewaySessions } from "@/components/networking";
|
||||
import type {
|
||||
MCPGatewaySessionGroupCount,
|
||||
MCPGatewaySessionSelector,
|
||||
MCPGatewaySessionsResponse,
|
||||
MCPGatewaySessionsTerminateResponse,
|
||||
} from "@/components/mcp_tools/types";
|
||||
import { createQueryKeys } from "@/app/(dashboard)/hooks/common/queryKeysFactory";
|
||||
|
||||
const mcpGatewaySessionKeys = createQueryKeys("mcpGatewaySessions");
|
||||
|
|
@ -28,6 +41,16 @@ function groupLabel(label: string | null): string {
|
|||
return label === "" ? '""' : label;
|
||||
}
|
||||
|
||||
export function describeSelector(selector: MCPGatewaySessionSelector): string {
|
||||
if (selector.user_id !== undefined) return `every live session opened by user ${groupLabel(selector.user_id)}`;
|
||||
return `session ${selector.session_id_prefix}`;
|
||||
}
|
||||
|
||||
export function describeTerminateResult(result: MCPGatewaySessionsTerminateResponse): string {
|
||||
const noun = result.terminated_sessions === 1 ? "session" : "sessions";
|
||||
return `Disconnected ${result.terminated_sessions} ${noun} on worker pid ${result.worker_pid}.`;
|
||||
}
|
||||
|
||||
function StatCard({ label, value }: { label: string; value: number }) {
|
||||
return (
|
||||
<div className="bg-card border border-border rounded-lg px-4 py-3">
|
||||
|
|
@ -37,14 +60,37 @@ function StatCard({ label, value }: { label: string; value: number }) {
|
|||
);
|
||||
}
|
||||
|
||||
function DisconnectUserButton({
|
||||
userId,
|
||||
onDisconnectUser,
|
||||
}: {
|
||||
userId: string | null;
|
||||
onDisconnectUser: (userId: string) => void;
|
||||
}) {
|
||||
if (userId === null || userId === "") return null;
|
||||
return (
|
||||
<Button
|
||||
variant="outline"
|
||||
size="sm"
|
||||
onClick={() => onDisconnectUser(userId)}
|
||||
aria-label={`Disconnect all sessions for user ${groupLabel(userId)}`}
|
||||
>
|
||||
<Unplug className="size-4" />
|
||||
Disconnect all
|
||||
</Button>
|
||||
);
|
||||
}
|
||||
|
||||
function GroupCountTable({
|
||||
title,
|
||||
groups,
|
||||
labelHeader,
|
||||
onDisconnectUser,
|
||||
}: {
|
||||
title: string;
|
||||
groups: MCPGatewaySessionGroupCount[];
|
||||
labelHeader: string;
|
||||
onDisconnectUser?: (userId: string) => void;
|
||||
}) {
|
||||
return (
|
||||
<section aria-label={title} className="rounded-lg border border-border bg-card">
|
||||
|
|
@ -54,6 +100,7 @@ function GroupCountTable({
|
|||
<TableRow>
|
||||
<TableHead>{labelHeader}</TableHead>
|
||||
<TableHead className="text-right">Sessions</TableHead>
|
||||
{onDisconnectUser ? <TableHead className="text-right">Actions</TableHead> : null}
|
||||
</TableRow>
|
||||
</TableHeader>
|
||||
<TableBody>
|
||||
|
|
@ -61,6 +108,11 @@ function GroupCountTable({
|
|||
<TableRow key={group.label ?? "__unknown__"}>
|
||||
<TableCell className="font-mono text-xs">{groupLabel(group.label)}</TableCell>
|
||||
<TableCell className="text-right">{group.count}</TableCell>
|
||||
{onDisconnectUser ? (
|
||||
<TableCell className="text-right">
|
||||
<DisconnectUserButton userId={group.label} onDisconnectUser={onDisconnectUser} />
|
||||
</TableCell>
|
||||
) : null}
|
||||
</TableRow>
|
||||
))}
|
||||
</TableBody>
|
||||
|
|
@ -73,10 +125,12 @@ function SessionsBody({
|
|||
data,
|
||||
error,
|
||||
isLoading,
|
||||
onDisconnect,
|
||||
}: {
|
||||
data: MCPGatewaySessionsResponse | undefined;
|
||||
error: Error | null;
|
||||
isLoading: boolean;
|
||||
onDisconnect: ((selector: MCPGatewaySessionSelector) => void) | null;
|
||||
}) {
|
||||
if (isLoading) {
|
||||
return (
|
||||
|
|
@ -117,7 +171,12 @@ function SessionsBody({
|
|||
</div>
|
||||
<div className="grid grid-cols-1 gap-4 lg:grid-cols-2">
|
||||
<GroupCountTable title="Sessions by AI client" labelHeader="Client" groups={data.by_client} />
|
||||
<GroupCountTable title="Sessions by user" labelHeader="User" groups={data.by_user} />
|
||||
<GroupCountTable
|
||||
title="Sessions by user"
|
||||
labelHeader="User"
|
||||
groups={data.by_user}
|
||||
onDisconnectUser={onDisconnect ? (userId) => onDisconnect({ user_id: userId }) : undefined}
|
||||
/>
|
||||
</div>
|
||||
<section aria-label="Live sessions" className="rounded-lg border border-border bg-card">
|
||||
<h3 className="border-b border-border px-4 py-2 text-sm font-semibold text-foreground">
|
||||
|
|
@ -134,11 +193,12 @@ function SessionsBody({
|
|||
<TableHead>Client IP</TableHead>
|
||||
<TableHead className="text-right">Idle</TableHead>
|
||||
<TableHead className="text-right">In flight</TableHead>
|
||||
{onDisconnect ? <TableHead className="text-right">Actions</TableHead> : null}
|
||||
</TableRow>
|
||||
</TableHeader>
|
||||
<TableBody>
|
||||
{data.sessions.map((session) => (
|
||||
<TableRow key={session.session_id_prefix}>
|
||||
{data.sessions.map((session, index) => (
|
||||
<TableRow key={`${session.session_id_prefix}-${index}`}>
|
||||
<TableCell className="font-mono text-xs">{session.session_id_prefix}</TableCell>
|
||||
<TableCell>
|
||||
{session.client_name === null ? (
|
||||
|
|
@ -169,6 +229,19 @@ function SessionsBody({
|
|||
<TableCell className="font-mono text-xs">{session.client_ip || "-"}</TableCell>
|
||||
<TableCell className="text-right text-xs">{formatIdleSeconds(session.idle_seconds)}</TableCell>
|
||||
<TableCell className="text-right text-xs">{session.in_flight_requests}</TableCell>
|
||||
{onDisconnect ? (
|
||||
<TableCell className="text-right">
|
||||
<Button
|
||||
variant="outline"
|
||||
size="sm"
|
||||
onClick={() => onDisconnect({ session_id_prefix: session.session_id_prefix })}
|
||||
aria-label={`Disconnect session ${session.session_id_prefix}`}
|
||||
>
|
||||
<Unplug className="size-4" />
|
||||
Disconnect
|
||||
</Button>
|
||||
</TableCell>
|
||||
) : null}
|
||||
</TableRow>
|
||||
))}
|
||||
</TableBody>
|
||||
|
|
@ -180,9 +253,12 @@ function SessionsBody({
|
|||
|
||||
interface MCPGatewaySessionsTabProps {
|
||||
accessToken: string | null;
|
||||
canTerminate: boolean;
|
||||
}
|
||||
|
||||
export function MCPGatewaySessionsTab({ accessToken }: MCPGatewaySessionsTabProps) {
|
||||
export function MCPGatewaySessionsTab({ accessToken, canTerminate }: MCPGatewaySessionsTabProps) {
|
||||
const queryClient = useQueryClient();
|
||||
const [pendingSelector, setPendingSelector] = useState<MCPGatewaySessionSelector | null>(null);
|
||||
const queryOptions = {
|
||||
queryKey: mcpGatewaySessionKeys.lists(),
|
||||
queryFn: () => fetchMCPGatewaySessions(accessToken!),
|
||||
|
|
@ -190,6 +266,15 @@ export function MCPGatewaySessionsTab({ accessToken }: MCPGatewaySessionsTabProp
|
|||
refetchInterval: REFETCH_INTERVAL_MS,
|
||||
};
|
||||
const { data, error, isLoading, isFetching, refetch } = useQuery<MCPGatewaySessionsResponse, Error>(queryOptions);
|
||||
const terminate = useMutation<MCPGatewaySessionsTerminateResponse, Error, MCPGatewaySessionSelector>({
|
||||
mutationFn: (selector) => terminateMCPGatewaySessions(accessToken!, selector),
|
||||
onSettled: () => queryClient.invalidateQueries({ queryKey: mcpGatewaySessionKeys.lists() }),
|
||||
});
|
||||
const confirmDisconnect = () => {
|
||||
if (pendingSelector === null) return;
|
||||
terminate.mutate(pendingSelector);
|
||||
setPendingSelector(null);
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="mt-4 space-y-4" data-testid="mcp-gateway-sessions-tab">
|
||||
|
|
@ -214,7 +299,48 @@ export function MCPGatewaySessionsTab({ accessToken }: MCPGatewaySessionsTabProp
|
|||
</Button>
|
||||
</div>
|
||||
|
||||
<SessionsBody data={data} error={error} isLoading={isLoading} />
|
||||
{terminate.isError ? (
|
||||
<Alert variant="destructive">
|
||||
<AlertTitle>Could not disconnect</AlertTitle>
|
||||
<AlertDescription>{terminate.error.message}</AlertDescription>
|
||||
</Alert>
|
||||
) : null}
|
||||
{terminate.isSuccess ? (
|
||||
<Alert>
|
||||
<AlertTitle>Disconnected</AlertTitle>
|
||||
<AlertDescription>
|
||||
{describeTerminateResult(terminate.data)} Clients holding those sessions must send a new initialize request,
|
||||
which re-runs authentication. Sessions on other proxy workers are not affected.
|
||||
</AlertDescription>
|
||||
</Alert>
|
||||
) : null}
|
||||
|
||||
<SessionsBody
|
||||
data={data}
|
||||
error={error}
|
||||
isLoading={isLoading}
|
||||
onDisconnect={canTerminate ? setPendingSelector : null}
|
||||
/>
|
||||
|
||||
<AlertDialog open={pendingSelector !== null} onOpenChange={(open) => !open && setPendingSelector(null)}>
|
||||
<AlertDialogContent>
|
||||
<AlertDialogHeader>
|
||||
<AlertDialogTitle>Disconnect MCP session</AlertDialogTitle>
|
||||
<AlertDialogDescription>
|
||||
{pendingSelector ? `This force-closes ${describeSelector(pendingSelector)} on this proxy worker. ` : ""}
|
||||
In-flight requests fail and the client must initialize again before it can call tools.
|
||||
</AlertDialogDescription>
|
||||
</AlertDialogHeader>
|
||||
<AlertDialogFooter>
|
||||
<Button variant="outline" onClick={() => setPendingSelector(null)}>
|
||||
Cancel
|
||||
</Button>
|
||||
<Button variant="destructive" onClick={confirmDisconnect} disabled={terminate.isPending}>
|
||||
Disconnect
|
||||
</Button>
|
||||
</AlertDialogFooter>
|
||||
</AlertDialogContent>
|
||||
</AlertDialog>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,102 @@
|
|||
import React from "react";
|
||||
import { render, screen, within } from "@testing-library/react";
|
||||
import userEvent from "@testing-library/user-event";
|
||||
import { describe, it, expect, vi, beforeEach } from "vitest";
|
||||
import { QueryClient, QueryClientProvider } from "@tanstack/react-query";
|
||||
import { MCPServerUserCredentialsPanel } from "./MCPServerUserCredentialsPanel";
|
||||
import * as networking from "@/components/networking";
|
||||
import type { MCPServerUserCredentialListItem } from "@/components/mcp_tools/types";
|
||||
|
||||
vi.mock("@/components/networking", () => ({
|
||||
fetchMCPServerUserCredentials: vi.fn(),
|
||||
revokeMCPServerUserCredential: vi.fn(),
|
||||
}));
|
||||
|
||||
const ITEMS: MCPServerUserCredentialListItem[] = [
|
||||
{
|
||||
user_id: "alice",
|
||||
credential_type: "oauth2",
|
||||
expires_at: "2026-12-31T00:00:00+00:00",
|
||||
connected_at: "2026-01-01T00:00:00+00:00",
|
||||
updated_at: "2026-01-01T00:00:00+00:00",
|
||||
},
|
||||
{
|
||||
user_id: "carol",
|
||||
credential_type: "byok",
|
||||
expires_at: null,
|
||||
connected_at: null,
|
||||
updated_at: "2026-02-01T00:00:00+00:00",
|
||||
},
|
||||
];
|
||||
|
||||
const renderPanel = ({ canRevoke = false }: { canRevoke?: boolean } = {}) => {
|
||||
const queryClient = new QueryClient({ defaultOptions: { queries: { retry: false, gcTime: 0 } } });
|
||||
return render(
|
||||
<QueryClientProvider client={queryClient}>
|
||||
<MCPServerUserCredentialsPanel serverId="srv-1" accessToken="token" canRevoke={canRevoke} />
|
||||
</QueryClientProvider>,
|
||||
);
|
||||
};
|
||||
|
||||
describe("MCPServerUserCredentialsPanel", () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("lists each user's credential type without a revoke control for a read-only admin", async () => {
|
||||
vi.mocked(networking.fetchMCPServerUserCredentials).mockResolvedValue(ITEMS);
|
||||
renderPanel({ canRevoke: false });
|
||||
|
||||
const table = await screen.findByRole("region", { name: "Stored user credentials" });
|
||||
expect(within(table).getByRole("row", { name: /alice/ })).toHaveTextContent("OAuth2");
|
||||
expect(within(table).getByRole("row", { name: /carol/ })).toHaveTextContent("BYOK API key");
|
||||
expect(screen.queryByRole("button", { name: /^Revoke credential/ })).not.toBeInTheDocument();
|
||||
expect(networking.fetchMCPServerUserCredentials).toHaveBeenCalledWith("token", "srv-1");
|
||||
});
|
||||
|
||||
it("revokes the selected user's credential through the route for its type and refetches", async () => {
|
||||
const user = userEvent.setup();
|
||||
vi.mocked(networking.fetchMCPServerUserCredentials).mockResolvedValueOnce(ITEMS).mockResolvedValueOnce([ITEMS[1]]);
|
||||
vi.mocked(networking.revokeMCPServerUserCredential).mockResolvedValue(undefined);
|
||||
renderPanel({ canRevoke: true });
|
||||
|
||||
await user.click(await screen.findByRole("button", { name: "Revoke credential for user alice" }));
|
||||
expect(networking.revokeMCPServerUserCredential).not.toHaveBeenCalled();
|
||||
const dialog = await screen.findByRole("alertdialog");
|
||||
expect(dialog).toHaveTextContent("OAuth2 credential stored for user alice");
|
||||
await user.click(within(dialog).getByRole("button", { name: "Revoke" }));
|
||||
|
||||
expect(await screen.findByText(/OAuth2 credential for user alice was deleted/)).toBeInTheDocument();
|
||||
expect(networking.revokeMCPServerUserCredential).toHaveBeenCalledWith("token", "srv-1", "alice", "oauth2");
|
||||
const table = await screen.findByRole("region", { name: "Stored user credentials" });
|
||||
expect(within(table).queryByRole("row", { name: /alice/ })).not.toBeInTheDocument();
|
||||
expect(within(table).getByRole("row", { name: /carol/ })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("shows the API error when a revoke is refused and keeps the list", async () => {
|
||||
const user = userEvent.setup();
|
||||
vi.mocked(networking.fetchMCPServerUserCredentials).mockResolvedValue(ITEMS);
|
||||
vi.mocked(networking.revokeMCPServerUserCredential).mockRejectedValue(
|
||||
new Error("Proxy admin access required to revoke another user's MCP credential."),
|
||||
);
|
||||
renderPanel({ canRevoke: true });
|
||||
|
||||
await user.click(await screen.findByRole("button", { name: "Revoke credential for user carol" }));
|
||||
await user.click(within(await screen.findByRole("alertdialog")).getByRole("button", { name: "Revoke" }));
|
||||
|
||||
const alert = await screen.findByRole("alert");
|
||||
expect(alert).toHaveTextContent("Could not revoke credential");
|
||||
expect(alert).toHaveTextContent("Proxy admin access required to revoke another user's MCP credential.");
|
||||
expect(networking.revokeMCPServerUserCredential).toHaveBeenCalledWith("token", "srv-1", "carol", "byok");
|
||||
expect(screen.getByRole("region", { name: "Stored user credentials" })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("shows the API error when the list cannot be loaded", async () => {
|
||||
vi.mocked(networking.fetchMCPServerUserCredentials).mockRejectedValue(new Error("Admin access required"));
|
||||
renderPanel();
|
||||
|
||||
const alert = await screen.findByRole("alert");
|
||||
expect(alert).toHaveTextContent("Could not load user credentials");
|
||||
expect(alert).toHaveTextContent("Admin access required");
|
||||
});
|
||||
});
|
||||
|
|
@ -0,0 +1,212 @@
|
|||
"use client";
|
||||
|
||||
import React, { useState } from "react";
|
||||
import { useMutation, useQuery, useQueryClient } from "@tanstack/react-query";
|
||||
import { RefreshCw, ShieldOff } from "lucide-react";
|
||||
import { Alert, AlertDescription, AlertTitle } from "@/components/ui/alert";
|
||||
import {
|
||||
AlertDialog,
|
||||
AlertDialogContent,
|
||||
AlertDialogDescription,
|
||||
AlertDialogFooter,
|
||||
AlertDialogHeader,
|
||||
AlertDialogTitle,
|
||||
} from "@/components/ui/alert-dialog";
|
||||
import { Badge } from "@/components/ui/badge";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui/table";
|
||||
import { UiLoadingSpinner } from "@/components/ui/ui-loading-spinner";
|
||||
import { fetchMCPServerUserCredentials, revokeMCPServerUserCredential } from "@/components/networking";
|
||||
import type { MCPServerUserCredentialListItem } from "@/components/mcp_tools/types";
|
||||
import { createQueryKeys } from "@/app/(dashboard)/hooks/common/queryKeysFactory";
|
||||
|
||||
const mcpServerUserCredentialKeys = createQueryKeys("mcpServerUserCredentials");
|
||||
|
||||
export function credentialTypeLabel(credentialType: MCPServerUserCredentialListItem["credential_type"]): string {
|
||||
return credentialType === "oauth2" ? "OAuth2" : "BYOK API key";
|
||||
}
|
||||
|
||||
export function formatTimestamp(value: string | null): string {
|
||||
if (value === null) return "-";
|
||||
const parsed = new Date(value);
|
||||
return Number.isNaN(parsed.getTime()) ? value : parsed.toLocaleString();
|
||||
}
|
||||
|
||||
function CredentialsBody({
|
||||
items,
|
||||
error,
|
||||
isLoading,
|
||||
onRevoke,
|
||||
}: {
|
||||
items: MCPServerUserCredentialListItem[] | undefined;
|
||||
error: Error | null;
|
||||
isLoading: boolean;
|
||||
onRevoke: ((item: MCPServerUserCredentialListItem) => void) | null;
|
||||
}) {
|
||||
if (isLoading) {
|
||||
return (
|
||||
<div
|
||||
role="status"
|
||||
className="flex items-center justify-center gap-3 rounded-lg border border-dashed border-border bg-card p-12"
|
||||
>
|
||||
<UiLoadingSpinner className="size-6 text-muted-foreground" />
|
||||
<p className="text-sm text-muted-foreground">Loading user credentials...</p>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
if (error) {
|
||||
return (
|
||||
<Alert variant="destructive">
|
||||
<AlertTitle>Could not load user credentials</AlertTitle>
|
||||
<AlertDescription>{error.message}</AlertDescription>
|
||||
</Alert>
|
||||
);
|
||||
}
|
||||
if (!items) return null;
|
||||
if (items.length === 0) {
|
||||
return (
|
||||
<div className="rounded-lg border border-dashed border-border bg-card p-12 text-center">
|
||||
<p className="text-sm text-muted-foreground">No user has a stored credential for this server.</p>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<section aria-label="Stored user credentials" className="rounded-lg border border-border bg-card">
|
||||
<Table>
|
||||
<TableHeader>
|
||||
<TableRow>
|
||||
<TableHead>User</TableHead>
|
||||
<TableHead>Type</TableHead>
|
||||
<TableHead>Connected</TableHead>
|
||||
<TableHead>Expires</TableHead>
|
||||
<TableHead>Updated</TableHead>
|
||||
{onRevoke ? <TableHead className="text-right">Actions</TableHead> : null}
|
||||
</TableRow>
|
||||
</TableHeader>
|
||||
<TableBody>
|
||||
{items.map((item) => (
|
||||
<TableRow key={item.user_id}>
|
||||
<TableCell className="font-mono text-xs">{item.user_id}</TableCell>
|
||||
<TableCell>
|
||||
<Badge variant="secondary">{credentialTypeLabel(item.credential_type)}</Badge>
|
||||
</TableCell>
|
||||
<TableCell className="text-xs">{formatTimestamp(item.connected_at)}</TableCell>
|
||||
<TableCell className="text-xs">{formatTimestamp(item.expires_at)}</TableCell>
|
||||
<TableCell className="text-xs">{formatTimestamp(item.updated_at)}</TableCell>
|
||||
{onRevoke ? (
|
||||
<TableCell className="text-right">
|
||||
<Button
|
||||
variant="outline"
|
||||
size="sm"
|
||||
onClick={() => onRevoke(item)}
|
||||
aria-label={`Revoke credential for user ${item.user_id}`}
|
||||
>
|
||||
<ShieldOff className="size-4" />
|
||||
Revoke
|
||||
</Button>
|
||||
</TableCell>
|
||||
) : null}
|
||||
</TableRow>
|
||||
))}
|
||||
</TableBody>
|
||||
</Table>
|
||||
</section>
|
||||
);
|
||||
}
|
||||
|
||||
interface MCPServerUserCredentialsPanelProps {
|
||||
serverId: string;
|
||||
accessToken: string | null;
|
||||
canRevoke: boolean;
|
||||
}
|
||||
|
||||
export function MCPServerUserCredentialsPanel({
|
||||
serverId,
|
||||
accessToken,
|
||||
canRevoke,
|
||||
}: MCPServerUserCredentialsPanelProps) {
|
||||
const queryClient = useQueryClient();
|
||||
const [pendingItem, setPendingItem] = useState<MCPServerUserCredentialListItem | null>(null);
|
||||
const queryKey = mcpServerUserCredentialKeys.detail(serverId);
|
||||
const { data, error, isLoading, isFetching, refetch } = useQuery<MCPServerUserCredentialListItem[], Error>({
|
||||
queryKey,
|
||||
queryFn: () => fetchMCPServerUserCredentials(accessToken!, serverId),
|
||||
enabled: !!accessToken,
|
||||
});
|
||||
const revoke = useMutation<void, Error, MCPServerUserCredentialListItem>({
|
||||
mutationFn: (item) => revokeMCPServerUserCredential(accessToken!, serverId, item.user_id, item.credential_type),
|
||||
onSettled: () => queryClient.invalidateQueries({ queryKey }),
|
||||
});
|
||||
const confirmRevoke = () => {
|
||||
if (pendingItem === null) return;
|
||||
revoke.mutate(pendingItem);
|
||||
setPendingItem(null);
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="space-y-4" data-testid="mcp-server-user-credentials-panel">
|
||||
<div className="flex flex-wrap items-start justify-between gap-3">
|
||||
<div>
|
||||
<h2 className="text-lg font-medium">User Credentials</h2>
|
||||
<p className="text-sm text-muted-foreground">
|
||||
Per-user OAuth2 tokens and BYOK API keys stored for this server. Revoking one deletes it from the database
|
||||
and clears the cached copy, so the user must connect again before the gateway will call this server for
|
||||
them.
|
||||
</p>
|
||||
</div>
|
||||
<Button
|
||||
variant="outline"
|
||||
size="sm"
|
||||
onClick={() => refetch()}
|
||||
disabled={isFetching}
|
||||
aria-label="Refresh user credentials"
|
||||
>
|
||||
<RefreshCw className={`size-4 ${isFetching ? "animate-spin" : ""}`} />
|
||||
Refresh
|
||||
</Button>
|
||||
</div>
|
||||
|
||||
{revoke.isError ? (
|
||||
<Alert variant="destructive">
|
||||
<AlertTitle>Could not revoke credential</AlertTitle>
|
||||
<AlertDescription>{revoke.error.message}</AlertDescription>
|
||||
</Alert>
|
||||
) : null}
|
||||
{revoke.isSuccess ? (
|
||||
<Alert>
|
||||
<AlertTitle>Credential revoked</AlertTitle>
|
||||
<AlertDescription>
|
||||
The stored {credentialTypeLabel(revoke.variables.credential_type)} credential for user{" "}
|
||||
{revoke.variables.user_id} was deleted.
|
||||
</AlertDescription>
|
||||
</Alert>
|
||||
) : null}
|
||||
|
||||
<CredentialsBody items={data} error={error} isLoading={isLoading} onRevoke={canRevoke ? setPendingItem : null} />
|
||||
|
||||
<AlertDialog open={pendingItem !== null} onOpenChange={(open) => !open && setPendingItem(null)}>
|
||||
<AlertDialogContent>
|
||||
<AlertDialogHeader>
|
||||
<AlertDialogTitle>Revoke stored credential</AlertDialogTitle>
|
||||
<AlertDialogDescription>
|
||||
{pendingItem
|
||||
? `This deletes the ${credentialTypeLabel(pendingItem.credential_type)} credential stored for user ${pendingItem.user_id}. `
|
||||
: ""}
|
||||
Their next MCP request to this server fails until they connect again.
|
||||
</AlertDialogDescription>
|
||||
</AlertDialogHeader>
|
||||
<AlertDialogFooter>
|
||||
<Button variant="outline" onClick={() => setPendingItem(null)}>
|
||||
Cancel
|
||||
</Button>
|
||||
<Button variant="destructive" onClick={confirmRevoke} disabled={revoke.isPending}>
|
||||
Revoke
|
||||
</Button>
|
||||
</AlertDialogFooter>
|
||||
</AlertDialogContent>
|
||||
</AlertDialog>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
export default MCPServerUserCredentialsPanel;
|
||||
|
|
@ -1,7 +1,9 @@
|
|||
import { render, screen } from "@testing-library/react";
|
||||
import { render, screen, within } from "@testing-library/react";
|
||||
import userEvent from "@testing-library/user-event";
|
||||
import { describe, it, expect, vi, beforeEach } from "vitest";
|
||||
import { QueryClient, QueryClientProvider } from "@tanstack/react-query";
|
||||
import { MCPServerView } from "./mcp_server_view";
|
||||
import * as networking from "@/components/networking";
|
||||
import type { MCPServer } from "@/components/mcp_tools/types";
|
||||
|
||||
vi.mock(".", () => ({
|
||||
|
|
@ -13,6 +15,12 @@ vi.mock("./mcp_server_edit", () => ({
|
|||
EDIT_OAUTH_UI_STATE_KEY: "litellm-mcp-oauth-edit-state",
|
||||
}));
|
||||
|
||||
vi.mock("@/components/networking", async (importOriginal) => ({
|
||||
...(await importOriginal<typeof import("@/components/networking")>()),
|
||||
fetchMCPServerUserCredentials: vi.fn(),
|
||||
revokeMCPServerUserCredential: vi.fn(),
|
||||
}));
|
||||
|
||||
const baseServer = {
|
||||
server_id: "srv-1",
|
||||
server_name: "demo server",
|
||||
|
|
@ -25,19 +33,38 @@ const baseServer = {
|
|||
|
||||
const renderView = (overrides: Partial<MCPServer> = {}, props: Record<string, unknown> = {}) =>
|
||||
render(
|
||||
<MCPServerView
|
||||
mcpServer={{ ...baseServer, ...overrides } as MCPServer}
|
||||
onBack={vi.fn()}
|
||||
isProxyAdmin
|
||||
isEditing={false}
|
||||
accessToken="tok"
|
||||
userRole="Admin"
|
||||
userID="u1"
|
||||
availableAccessGroups={[]}
|
||||
{...props}
|
||||
/>,
|
||||
<QueryClientProvider client={new QueryClient({ defaultOptions: { queries: { retry: false, gcTime: 0 } } })}>
|
||||
<MCPServerView
|
||||
mcpServer={{ ...baseServer, ...overrides } as MCPServer}
|
||||
onBack={vi.fn()}
|
||||
isProxyAdmin
|
||||
isEditing={false}
|
||||
accessToken="tok"
|
||||
userRole="Admin"
|
||||
userID="u1"
|
||||
availableAccessGroups={[]}
|
||||
{...props}
|
||||
/>
|
||||
</QueryClientProvider>,
|
||||
);
|
||||
|
||||
const openUserCredentials = async (props: Record<string, unknown>) => {
|
||||
vi.mocked(networking.fetchMCPServerUserCredentials).mockResolvedValue([
|
||||
{
|
||||
user_id: "alice",
|
||||
credential_type: "byok",
|
||||
expires_at: null,
|
||||
connected_at: null,
|
||||
updated_at: "2026-01-01T00:00:00+00:00",
|
||||
},
|
||||
]);
|
||||
renderView({}, props);
|
||||
await userEvent.click(screen.getByRole("tab", { name: "User Credentials" }));
|
||||
return within(await screen.findByRole("region", { name: "Stored user credentials" })).getByRole("row", {
|
||||
name: /alice/,
|
||||
});
|
||||
};
|
||||
|
||||
describe("MCPServerView", () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks();
|
||||
|
|
@ -149,4 +176,15 @@ describe("MCPServerView", () => {
|
|||
|
||||
expect(await screen.findByText("All tools enabled")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("lets a full admin revoke a stored user credential", async () => {
|
||||
const row = await openUserCredentials({});
|
||||
expect(within(row).getByRole("button", { name: "Revoke credential for user alice" })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("shows stored credentials to a view-only admin session without a revoke control", async () => {
|
||||
const row = await openUserCredentials({ isViewOnly: true });
|
||||
expect(row).toHaveTextContent("BYOK API key");
|
||||
expect(within(row).queryByRole("button", { name: /^Revoke credential/ })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -9,7 +9,9 @@ import { MCPServer, handleTransport, handleAuth } from "@/components/mcp_tools/t
|
|||
// TODO: Move Tools viewer from index file
|
||||
import { MCPToolsViewer } from ".";
|
||||
import MCPServerEdit, { EDIT_OAUTH_UI_STATE_KEY } from "./mcp_server_edit";
|
||||
import { MCPServerUserCredentialsPanel } from "./MCPServerUserCredentialsPanel";
|
||||
import { getSecureItem } from "@/utils/secureStorage";
|
||||
import { isProxyAdminRole, isProxyAdminTierRole } from "@/utils/roles";
|
||||
import MCPServerCostDisplay from "./mcp_server_cost_display";
|
||||
import { getMaskedAndFullUrl } from "./utils";
|
||||
import { copyToClipboard as utilCopyToClipboard } from "@/utils/dataUtils";
|
||||
|
|
@ -23,6 +25,7 @@ interface MCPServerViewProps {
|
|||
accessToken: string | null;
|
||||
userRole: string | null;
|
||||
userID: string | null;
|
||||
isViewOnly?: boolean;
|
||||
availableAccessGroups: string[];
|
||||
initialTabIndex?: number;
|
||||
}
|
||||
|
|
@ -53,6 +56,7 @@ export const MCPServerView: React.FC<MCPServerViewProps> = ({
|
|||
accessToken,
|
||||
userRole,
|
||||
userID,
|
||||
isViewOnly = false,
|
||||
availableAccessGroups,
|
||||
initialTabIndex = 0,
|
||||
}) => {
|
||||
|
|
@ -63,6 +67,8 @@ export const MCPServerView: React.FC<MCPServerViewProps> = ({
|
|||
const [showFullUrl, setShowFullUrl] = useState(false);
|
||||
const [copiedStates, setCopiedStates] = useState<Record<string, boolean>>({});
|
||||
const [selectedTabIndex, setSelectedTabIndex] = useState(returningFromEditOAuth ? 2 : initialTabIndex);
|
||||
const canViewUserCredentials = userRole !== null && isProxyAdminTierRole(userRole);
|
||||
const canRevokeUserCredentials = userRole !== null && isProxyAdminRole(userRole) && !isViewOnly;
|
||||
|
||||
const handleSuccess = (updated: MCPServer) => {
|
||||
setEditing(false);
|
||||
|
|
@ -142,6 +148,11 @@ export const MCPServerView: React.FC<MCPServerViewProps> = ({
|
|||
Settings
|
||||
</TabsTrigger>
|
||||
)}
|
||||
{canViewUserCredentials && (
|
||||
<TabsTrigger value="3" className="flex-none rounded-none px-4 py-2">
|
||||
User Credentials
|
||||
</TabsTrigger>
|
||||
)}
|
||||
</TabsList>
|
||||
|
||||
{/* Overview Panel */}
|
||||
|
|
@ -387,6 +398,18 @@ export const MCPServerView: React.FC<MCPServerViewProps> = ({
|
|||
)}
|
||||
</Card>
|
||||
</TabsContent>
|
||||
|
||||
{canViewUserCredentials && (
|
||||
<TabsContent value="3">
|
||||
<Card className="p-6">
|
||||
<MCPServerUserCredentialsPanel
|
||||
serverId={mcpServer.server_id}
|
||||
accessToken={accessToken}
|
||||
canRevoke={canRevokeUserCredentials}
|
||||
/>
|
||||
</Card>
|
||||
</TabsContent>
|
||||
)}
|
||||
</Tabs>
|
||||
</div>
|
||||
);
|
||||
|
|
|
|||
|
|
@ -17,6 +17,8 @@ vi.mock("@/components/networking", () => ({
|
|||
updateConfigFieldSetting: vi.fn().mockResolvedValue(undefined),
|
||||
deleteConfigFieldSetting: vi.fn().mockResolvedValue(undefined),
|
||||
listMCPUserEnvVarStatus: vi.fn().mockResolvedValue([]),
|
||||
fetchMCPGatewaySessions: vi.fn(),
|
||||
terminateMCPGatewaySessions: vi.fn(),
|
||||
}));
|
||||
|
||||
const createQueryClient = () =>
|
||||
|
|
@ -400,4 +402,50 @@ describe("MCPServers", () => {
|
|||
// The server list refresh must NOT trigger a second health check
|
||||
expect(networking.fetchMCPServerHealth).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
const liveSessionsReport = {
|
||||
worker_pid: 4242,
|
||||
total_sessions: 1,
|
||||
by_client: [{ label: "claude-code", count: 1 }],
|
||||
by_user: [{ label: "alice", count: 1 }],
|
||||
sessions: [
|
||||
{
|
||||
session_id_prefix: "aaaa1111",
|
||||
client_name: "claude-code",
|
||||
client_version: "1.0.0",
|
||||
user_id: "alice",
|
||||
user_email: "alice@example.com",
|
||||
key_alias: "alice-key",
|
||||
team_id: null,
|
||||
team_alias: null,
|
||||
client_ip: "10.0.0.1",
|
||||
idle_seconds: 5,
|
||||
in_flight_requests: 0,
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
const openLiveConnections = async (props: { isViewOnly?: boolean }) => {
|
||||
vi.mocked(networking.fetchMCPServers).mockResolvedValue([]);
|
||||
vi.mocked(networking.fetchMCPGatewaySessions).mockResolvedValue(liveSessionsReport);
|
||||
render(
|
||||
<QueryClientProvider client={createQueryClient()}>
|
||||
<MCPServers {...defaultProps} {...props} />
|
||||
</QueryClientProvider>,
|
||||
);
|
||||
await userEvent.click(await screen.findByRole("tab", { name: "Live Connections" }));
|
||||
return within(await screen.findByRole("region", { name: "Live sessions" })).getByRole("row", { name: /aaaa1111/ });
|
||||
};
|
||||
|
||||
it("lets a full admin disconnect a live session", async () => {
|
||||
const row = await openLiveConnections({ isViewOnly: false });
|
||||
expect(within(row).getByRole("button", { name: "Disconnect session aaaa1111" })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("shows live sessions to a view-only admin session without any disconnect control", async () => {
|
||||
const row = await openLiveConnections({ isViewOnly: true });
|
||||
expect(row).toHaveTextContent("alice@example.com");
|
||||
expect(within(row).queryByRole("button", { name: /^Disconnect/ })).not.toBeInTheDocument();
|
||||
expect(screen.queryByRole("button", { name: /^Disconnect all/ })).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
import { isAdminRole, isProxyAdminTierRole } from "@/utils/roles";
|
||||
import { isAdminRole, isProxyAdminRole, isProxyAdminTierRole } from "@/utils/roles";
|
||||
import { CircleHelp, Search } from "lucide-react";
|
||||
import { Badge } from "@/components/ui/badge";
|
||||
import { Button } from "@/components/ui/button";
|
||||
|
|
@ -109,7 +109,7 @@ const readToolsOAuthServerId = (): string | null => {
|
|||
}
|
||||
};
|
||||
|
||||
const MCPServers: React.FC<MCPServerProps> = ({ accessToken, userRole, userID }) => {
|
||||
const MCPServers: React.FC<MCPServerProps> = ({ accessToken, userRole, userID, isViewOnly = false }) => {
|
||||
const { data: mcpServers, isLoading: isLoadingServers, refetch } = useMCPServers();
|
||||
|
||||
// Fetch health status for all servers
|
||||
|
|
@ -578,6 +578,7 @@ const MCPServers: React.FC<MCPServerProps> = ({ accessToken, userRole, userID })
|
|||
accessToken={accessToken}
|
||||
userID={userID}
|
||||
userRole={userRole}
|
||||
isViewOnly={isViewOnly}
|
||||
availableAccessGroups={uniqueMcpAccessGroups}
|
||||
initialTabIndex={selectedServerId === toolsTabServerId ? 1 : 0}
|
||||
/>
|
||||
|
|
@ -755,7 +756,10 @@ const MCPServers: React.FC<MCPServerProps> = ({ accessToken, userRole, userID })
|
|||
)}
|
||||
{isProxyAdminTierRole(userRole) && (
|
||||
<TabsContent value="connections">
|
||||
<MCPGatewaySessionsTab accessToken={accessToken} />
|
||||
<MCPGatewaySessionsTab
|
||||
accessToken={accessToken}
|
||||
canTerminate={isProxyAdminRole(userRole) && !isViewOnly}
|
||||
/>
|
||||
</TabsContent>
|
||||
)}
|
||||
</Tabs>
|
||||
|
|
|
|||
|
|
@ -4,6 +4,6 @@ import { MCPServers } from "./_components";
|
|||
import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized";
|
||||
|
||||
export default function McpServers() {
|
||||
const { accessToken, userRole, userId } = useAuthorized();
|
||||
return <MCPServers accessToken={accessToken} userRole={userRole} userID={userId} />;
|
||||
const { accessToken, userRole, userId, isViewOnly } = useAuthorized();
|
||||
return <MCPServers accessToken={accessToken} userRole={userRole} userID={userId} isViewOnly={isViewOnly} />;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -569,6 +569,23 @@ describe("EntityUsage", () => {
|
|||
expect(screen.getAllByText("Activity Metrics")[1]).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("tells the team view how many keys the proxy left out of the per-key lists", async () => {
|
||||
mockTeamDailyActivityAggregatedCall.mockResolvedValue({
|
||||
...mockSpendData,
|
||||
metadata: { ...mockSpendData.metadata, api_key_limit: 100, total_api_keys: 3000 },
|
||||
});
|
||||
render(<EntityUsage {...defaultProps} entityType="team" />);
|
||||
|
||||
await waitFor(() => {
|
||||
expect(mockTeamDailyActivityAggregatedCall).toHaveBeenCalled();
|
||||
});
|
||||
act(() => {
|
||||
fireEvent.click(screen.getByText("Key Activity"));
|
||||
});
|
||||
|
||||
expect(await screen.findByRole("note")).toHaveTextContent("Only the 100 highest-spend keys of 3,000 are loaded");
|
||||
});
|
||||
|
||||
// An inactive tab panel is marked aria-selected="false" by one tab library and hidden by the
|
||||
// other, so treat either as "not on screen" and the assertion holds whichever one is rendering.
|
||||
const isShowing = (element: HTMLElement): boolean => {
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ import TeamMultiSelect from "@/components/common_components/team_multi_select";
|
|||
import UserDropdown from "@/components/common_components/UserDropdown";
|
||||
import { ActivityMetrics, processActivityData } from "@/components/activity_metrics";
|
||||
import { UsageExportHeader } from "@/components/EntityUsageExport";
|
||||
import { getExportBlockedReason } from "@/components/EntityUsageExport/exportBlockedReason";
|
||||
import { getApiKeyTruncation, getExportBlockedReason } from "@/components/EntityUsageExport/exportBlockedReason";
|
||||
import type { EntityType } from "@/components/EntityUsageExport/types";
|
||||
import {
|
||||
agentDailyActivityCall,
|
||||
|
|
@ -71,6 +71,8 @@ interface EntitySpendData {
|
|||
total_successful_requests: number;
|
||||
total_failed_requests: number;
|
||||
total_tokens: number;
|
||||
api_key_limit?: number | null;
|
||||
total_api_keys?: number | null;
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -160,6 +162,7 @@ const EntityUsage: React.FC<EntityUsageProps> = ({
|
|||
});
|
||||
|
||||
const spendData = spendDataRaw as unknown as EntitySpendData;
|
||||
const apiKeyTruncation = getApiKeyTruncation(spendData.metadata?.api_key_limit, spendData.metadata?.total_api_keys);
|
||||
|
||||
const {
|
||||
data: agentSpendDataRaw,
|
||||
|
|
@ -659,12 +662,18 @@ const EntityUsage: React.FC<EntityUsageProps> = ({
|
|||
{
|
||||
key: "keys",
|
||||
label: "Key Activity",
|
||||
content: <KeyActivityPanel keyMetrics={keyMetrics} hidePromptCachingMetrics={entityType === "agent"} />,
|
||||
content: (
|
||||
<KeyActivityPanel
|
||||
keyMetrics={keyMetrics}
|
||||
hidePromptCachingMetrics={entityType === "agent"}
|
||||
apiKeyTruncation={apiKeyTruncation}
|
||||
/>
|
||||
),
|
||||
},
|
||||
{ key: "endpoints", label: "Endpoint Activity", content: <EndpointUsage userSpendData={spendData} /> },
|
||||
];
|
||||
|
||||
const spendFetchState = { coversRange, cancelled, failed };
|
||||
const spendFetchState = { coversRange, cancelled, failed, apiKeyTruncation };
|
||||
|
||||
return (
|
||||
<div style={{ width: "100%" }} className="relative">
|
||||
|
|
|
|||
|
|
@ -30,7 +30,7 @@ import { ActivityMetrics, processActivityData } from "@/components/activity_metr
|
|||
import CloudZeroExportModal from "@/components/cloudzero_export_modal";
|
||||
import UserDropdown from "@/components/common_components/UserDropdown";
|
||||
import EntityUsageExportModal from "@/components/EntityUsageExport";
|
||||
import { getExportBlockedReason } from "@/components/EntityUsageExport/exportBlockedReason";
|
||||
import { getApiKeyTruncation, getExportBlockedReason } from "@/components/EntityUsageExport/exportBlockedReason";
|
||||
import KeyActivityPanel from "@/components/UsagePage/components/KeyActivityPanel";
|
||||
import { Team } from "@/components/key_team_helpers/key_list";
|
||||
import {
|
||||
|
|
@ -256,6 +256,10 @@ const UsagePage: React.FC<UsagePageProps> = ({ teams, organizations }) => {
|
|||
coversRange: activeAggregated !== null || paginatedResult.coversRange,
|
||||
cancelled: paginatedResult.cancelled,
|
||||
failed: paginatedResult.failed,
|
||||
apiKeyTruncation: getApiKeyTruncation(
|
||||
userSpendData.metadata?.api_key_limit,
|
||||
userSpendData.metadata?.total_api_keys,
|
||||
),
|
||||
};
|
||||
const exportBlockedReason = getExportBlockedReason(spendFetchState);
|
||||
|
||||
|
|
@ -904,7 +908,7 @@ const UsagePage: React.FC<UsagePageProps> = ({ teams, organizations }) => {
|
|||
<ActivityMetrics modelMetrics={modelMetrics} />
|
||||
</TabsContent>
|
||||
<TabsContent value="keys" keepMounted>
|
||||
<KeyActivityPanel keyMetrics={keyMetrics} />
|
||||
<KeyActivityPanel keyMetrics={keyMetrics} apiKeyTruncation={spendFetchState.apiKeyTruncation} />
|
||||
</TabsContent>
|
||||
<TabsContent value="mcp" keepMounted>
|
||||
<ActivityMetrics modelMetrics={mcpServerMetrics} />
|
||||
|
|
|
|||
|
|
@ -1,11 +1,12 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
|
||||
import { getExportBlockedReason, type UsageFetchState } from "./exportBlockedReason";
|
||||
import { getApiKeyTruncation, getExportBlockedReason, type UsageFetchState } from "./exportBlockedReason";
|
||||
|
||||
const state = (overrides: Partial<UsageFetchState> = {}): UsageFetchState => ({
|
||||
coversRange: true,
|
||||
cancelled: false,
|
||||
failed: false,
|
||||
apiKeyTruncation: undefined,
|
||||
...overrides,
|
||||
});
|
||||
|
||||
|
|
@ -31,4 +32,27 @@ describe("getExportBlockedReason", () => {
|
|||
expect(reason).toMatch(/failed to load/i);
|
||||
expect(reason).not.toMatch(/stopped/i);
|
||||
});
|
||||
|
||||
it("blocks when the aggregated endpoint dropped keys, since a per-team CSV would miss them", () => {
|
||||
const reason = getExportBlockedReason(state({ apiKeyTruncation: { limit: 100, total: 3000 } }));
|
||||
|
||||
expect(reason).toMatch(/100 highest-spend keys of 3000/);
|
||||
expect(reason).toMatch(/USAGE_TOP_API_KEYS_LIMIT/);
|
||||
});
|
||||
});
|
||||
|
||||
describe("getApiKeyTruncation", () => {
|
||||
it("reports truncation once the proxy saw more keys than it returned", () => {
|
||||
expect(getApiKeyTruncation(100, 101)).toEqual({ limit: 100, total: 101 });
|
||||
});
|
||||
|
||||
it("stays quiet when exactly the cap exists, since every key is on screen", () => {
|
||||
expect(getApiKeyTruncation(100, 100)).toBeUndefined();
|
||||
expect(getApiKeyTruncation(100, 7)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("stays quiet when the response carries no cap, as the paginated fallback does", () => {
|
||||
expect(getApiKeyTruncation(undefined, undefined)).toBeUndefined();
|
||||
expect(getApiKeyTruncation(100, null)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1,13 +1,31 @@
|
|||
export interface ApiKeyTruncation {
|
||||
limit: number;
|
||||
total: number;
|
||||
}
|
||||
|
||||
export interface UsageFetchState {
|
||||
coversRange: boolean;
|
||||
cancelled: boolean;
|
||||
failed: boolean;
|
||||
apiKeyTruncation: ApiKeyTruncation | undefined;
|
||||
}
|
||||
|
||||
export const getExportBlockedReason = ({ coversRange, cancelled, failed }: UsageFetchState): string | undefined => {
|
||||
export const getApiKeyTruncation = (apiKeyLimit: unknown, totalApiKeys: unknown): ApiKeyTruncation | undefined => {
|
||||
if (typeof apiKeyLimit !== "number" || typeof totalApiKeys !== "number") return undefined;
|
||||
return totalApiKeys > apiKeyLimit ? { limit: apiKeyLimit, total: totalApiKeys } : undefined;
|
||||
};
|
||||
|
||||
export const getExportBlockedReason = ({
|
||||
coversRange,
|
||||
cancelled,
|
||||
failed,
|
||||
apiKeyTruncation,
|
||||
}: UsageFetchState): string | undefined => {
|
||||
if (failed) return "Some spend data failed to load, so an export would under-report. Reload the page to try again.";
|
||||
if (cancelled)
|
||||
return "Loading was stopped before the whole range arrived, so an export would under-report. Reload the page to load it all.";
|
||||
if (!coversRange) return "Spend data is still loading, so an export would under-report. Wait for it to finish.";
|
||||
if (apiKeyTruncation !== undefined)
|
||||
return `Only the ${apiKeyTruncation.limit} highest-spend keys of ${apiKeyTruncation.total} were loaded, so a per-team export would under-report. Raise USAGE_TOP_API_KEYS_LIMIT on the proxy to load more keys.`;
|
||||
return undefined;
|
||||
};
|
||||
|
|
|
|||
|
|
@ -68,4 +68,14 @@ describe("KeyActivityPanel", () => {
|
|||
expect(screen.getByLabelText("Search keys")).toHaveValue("");
|
||||
expect(screen.getByTestId("rendered-keys")).toHaveTextContent("hash-alicehash-bob");
|
||||
});
|
||||
|
||||
it("says how many keys the proxy left out when only the top spenders were loaded", () => {
|
||||
render(<KeyActivityPanel keyMetrics={keyMetrics} apiKeyTruncation={{ limit: 2, total: 3000 }} />);
|
||||
expect(screen.getByRole("note")).toHaveTextContent("Only the 2 highest-spend keys of 3,000 are loaded");
|
||||
});
|
||||
|
||||
it("shows no truncation note when every key is loaded", () => {
|
||||
render(<KeyActivityPanel keyMetrics={keyMetrics} />);
|
||||
expect(screen.queryByRole("note")).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ import { Search, X } from "lucide-react";
|
|||
import React, { useMemo, useState } from "react";
|
||||
|
||||
import { ActivityMetrics } from "@/components/activity_metrics";
|
||||
import type { ApiKeyTruncation } from "@/components/EntityUsageExport/exportBlockedReason";
|
||||
import { InputGroup, InputGroupAddon, InputGroupButton, InputGroupInput } from "@/components/ui/input-group";
|
||||
|
||||
import { filterKeyActivity } from "../keyActivityFilter";
|
||||
|
|
@ -10,9 +11,14 @@ import type { ModelActivityData } from "../types";
|
|||
interface KeyActivityPanelProps {
|
||||
keyMetrics: Record<string, ModelActivityData>;
|
||||
hidePromptCachingMetrics?: boolean;
|
||||
apiKeyTruncation?: ApiKeyTruncation;
|
||||
}
|
||||
|
||||
const KeyActivityPanel: React.FC<KeyActivityPanelProps> = ({ keyMetrics, hidePromptCachingMetrics = false }) => {
|
||||
const KeyActivityPanel: React.FC<KeyActivityPanelProps> = ({
|
||||
keyMetrics,
|
||||
hidePromptCachingMetrics = false,
|
||||
apiKeyTruncation,
|
||||
}) => {
|
||||
const [query, setQuery] = useState("");
|
||||
const filtered = useMemo(() => filterKeyActivity(keyMetrics, query), [keyMetrics, query]);
|
||||
const totalKeys = Object.keys(keyMetrics).length;
|
||||
|
|
@ -43,6 +49,12 @@ const KeyActivityPanel: React.FC<KeyActivityPanelProps> = ({ keyMetrics, hidePro
|
|||
<span className="text-sm text-muted-foreground">
|
||||
Showing {shownKeys.toLocaleString()} of {totalKeys.toLocaleString()} keys
|
||||
</span>
|
||||
{apiKeyTruncation !== undefined && (
|
||||
<span className="text-sm text-muted-foreground" role="note">
|
||||
Only the {apiKeyTruncation.limit.toLocaleString()} highest-spend keys of{" "}
|
||||
{apiKeyTruncation.total.toLocaleString()} are loaded
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
{isFiltering && totalKeys > 0 && shownKeys === 0 ? (
|
||||
<p className="rounded-lg border p-6 text-center text-sm text-muted-foreground">
|
||||
|
|
|
|||
|
|
@ -14,7 +14,7 @@ export const NO_COMPRESSION = "none";
|
|||
|
||||
/** Guardrail providers that compress prompts, mirroring COMPRESSION_GUARDRAIL_PROVIDERS in
|
||||
* litellm/proxy/guardrails/auto_router_compression.py. Both are selectable per hop. */
|
||||
export const COMPRESSION_GUARDRAIL_PROVIDERS: readonly string[] = ["headroom", "compresr"];
|
||||
export const COMPRESSION_GUARDRAIL_PROVIDERS: readonly string[] = ["headroom", "compresr", "typesafe"];
|
||||
|
||||
export const isCompressionGuardrailProvider = (provider: unknown): boolean =>
|
||||
typeof provider === "string" && COMPRESSION_GUARDRAIL_PROVIDERS.includes(provider.toLowerCase());
|
||||
|
|
|
|||
|
|
@ -517,6 +517,7 @@ export interface MCPServerProps {
|
|||
accessToken: string | null;
|
||||
userRole: string | null;
|
||||
userID: string | null;
|
||||
isViewOnly?: boolean;
|
||||
}
|
||||
|
||||
export interface MCPToolsetTool {
|
||||
|
|
@ -587,3 +588,23 @@ export interface MCPGatewaySessionsResponse {
|
|||
by_user: MCPGatewaySessionGroupCount[];
|
||||
sessions: MCPGatewaySession[];
|
||||
}
|
||||
|
||||
export interface MCPGatewaySessionsTerminateResponse {
|
||||
worker_pid: number;
|
||||
terminated_sessions: number;
|
||||
sessions: MCPGatewaySession[];
|
||||
}
|
||||
|
||||
export type MCPGatewaySessionSelector =
|
||||
| { session_id_prefix: string; user_id?: undefined }
|
||||
| { user_id: string; session_id_prefix?: undefined };
|
||||
|
||||
export type MCPServerUserCredentialType = "oauth2" | "byok";
|
||||
|
||||
export interface MCPServerUserCredentialListItem {
|
||||
user_id: string;
|
||||
credential_type: MCPServerUserCredentialType;
|
||||
expires_at: string | null;
|
||||
connected_at: string | null;
|
||||
updated_at: string;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -97,7 +97,14 @@ import type { ModelBudgetUsage, ModelMaxBudget } from "./key_team_helpers/ModelM
|
|||
import type { ObjectPermission } from "./object_permission_types";
|
||||
import type { components } from "@/lib/http/schema";
|
||||
import { jsonFields } from "./common_components/check_openapi_schema";
|
||||
import type { MCPGatewaySessionsResponse, MCPUserEnvVarsStatus } from "./mcp_tools/types";
|
||||
import type {
|
||||
MCPGatewaySessionSelector,
|
||||
MCPGatewaySessionsResponse,
|
||||
MCPGatewaySessionsTerminateResponse,
|
||||
MCPServerUserCredentialListItem,
|
||||
MCPServerUserCredentialType,
|
||||
MCPUserEnvVarsStatus,
|
||||
} from "./mcp_tools/types";
|
||||
import type {
|
||||
CoordinationRedisSettings,
|
||||
CoordinationRedisSettingsResponse,
|
||||
|
|
@ -4976,6 +4983,33 @@ export const fetchMCPSubmissions = async (accessToken: string) => {
|
|||
export const fetchMCPGatewaySessions = async (accessToken: string): Promise<MCPGatewaySessionsResponse> =>
|
||||
apiClient.get<MCPGatewaySessionsResponse>(`/v1/mcp/sessions`, { accessToken });
|
||||
|
||||
export const terminateMCPGatewaySessions = async (
|
||||
accessToken: string,
|
||||
selector: MCPGatewaySessionSelector,
|
||||
): Promise<MCPGatewaySessionsTerminateResponse> =>
|
||||
apiClient.delete<MCPGatewaySessionsTerminateResponse>(`/v1/mcp/sessions`, { accessToken, query: { ...selector } });
|
||||
|
||||
export const fetchMCPServerUserCredentials = async (
|
||||
accessToken: string,
|
||||
serverId: string,
|
||||
): Promise<MCPServerUserCredentialListItem[]> =>
|
||||
apiClient.get<MCPServerUserCredentialListItem[]>(`/v1/mcp/server/${encodeURIComponent(serverId)}/user-credentials`, {
|
||||
accessToken,
|
||||
});
|
||||
|
||||
export const revokeMCPServerUserCredential = async (
|
||||
accessToken: string,
|
||||
serverId: string,
|
||||
userId: string,
|
||||
credentialType: MCPServerUserCredentialType,
|
||||
): Promise<void> => {
|
||||
const route = credentialType === "oauth2" ? "oauth-user-credential" : "user-credential";
|
||||
await apiClient.delete(`/v1/mcp/server/${encodeURIComponent(serverId)}/${route}`, {
|
||||
accessToken,
|
||||
query: { user_id: userId },
|
||||
});
|
||||
};
|
||||
|
||||
export const approveMCPServer = async (accessToken: string, serverId: string) => {
|
||||
try {
|
||||
const url = (proxyBaseUrl ? `${proxyBaseUrl}` : "") + `/v1/mcp/server/${encodeURIComponent(serverId)}/approve`;
|
||||
|
|
|
|||
|
|
@ -59,6 +59,7 @@ const renderModal = (overrides: Partial<React.ComponentProps<typeof RoutingGroup
|
|||
strategyDescriptions={STRATEGY_DESCRIPTIONS}
|
||||
modelOptions={MODEL_OPTIONS}
|
||||
existingGroupNames={["already-taken", "other-group"]}
|
||||
groupNameByModel={{}}
|
||||
onClose={onClose}
|
||||
onSubmit={onSubmit}
|
||||
{...overrides}
|
||||
|
|
@ -307,6 +308,19 @@ describe("RoutingGroupModal", () => {
|
|||
expect(onSubmit.mock.calls[0][0]).toStrictEqual(expected);
|
||||
});
|
||||
|
||||
it("blocks a model another group already claims", async () => {
|
||||
const user = userEvent.setup();
|
||||
const { onSubmit } = renderModal({ groupNameByModel: { "gpt-4o": "cheap" } });
|
||||
|
||||
await typeName(user, "security");
|
||||
await pickModels(user, "gpt-4o");
|
||||
await pickStrategy(user, "latency-based-routing");
|
||||
await save(user, "Create Group");
|
||||
|
||||
expect(await screen.findByText(/Already claimed: gpt-4o/)).toBeInTheDocument();
|
||||
expect(onSubmit).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("describes the selected strategy", async () => {
|
||||
renderModal();
|
||||
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ import {
|
|||
toRoutingGroupFormValues,
|
||||
} from "./routingGroupPayload";
|
||||
import type { RoutingGroup } from "./types";
|
||||
import { modelConflictError } from "./modelOwnership";
|
||||
import { Dialog, DialogContent, DialogFooter, DialogHeader, DialogTitle } from "@/components/ui/dialog";
|
||||
import { Button } from "@/components/ui/button";
|
||||
|
||||
|
|
@ -40,6 +41,7 @@ interface RoutingGroupModalProps {
|
|||
strategyDescriptions: Record<string, string>;
|
||||
modelOptions: string[];
|
||||
existingGroupNames: string[];
|
||||
groupNameByModel: Record<string, string>;
|
||||
onClose: () => void;
|
||||
onSubmit: (group: RoutingGroup) => Promise<void> | void;
|
||||
saving?: boolean;
|
||||
|
|
@ -57,6 +59,7 @@ const RoutingGroupModal: React.FC<RoutingGroupModalProps> = ({
|
|||
strategyDescriptions,
|
||||
modelOptions,
|
||||
existingGroupNames,
|
||||
groupNameByModel,
|
||||
onClose,
|
||||
onSubmit,
|
||||
saving,
|
||||
|
|
@ -77,12 +80,20 @@ const RoutingGroupModal: React.FC<RoutingGroupModalProps> = ({
|
|||
.min(1, "Group name is required")
|
||||
.max(GROUP_NAME_MAX_LENGTH, `Must be ${GROUP_NAME_MAX_LENGTH} characters or fewer`)
|
||||
.refine((value) => !reservedNames.has(value.toLowerCase()), "A group with this name already exists"),
|
||||
models: z.array(z.string()).min(1, "Select at least one model"),
|
||||
models: z
|
||||
.array(z.string())
|
||||
.min(1, "Select at least one model")
|
||||
.superRefine((models, ctx) => {
|
||||
const conflict = modelConflictError(models, groupNameByModel);
|
||||
if (conflict !== null) {
|
||||
ctx.addIssue({ code: "custom", message: conflict });
|
||||
}
|
||||
}),
|
||||
routing_strategy: z.string().min(1, "Strategy is required"),
|
||||
routing_strategy_args: z.string(),
|
||||
};
|
||||
return z.object(shape);
|
||||
}, [reservedNames]);
|
||||
}, [reservedNames, groupNameByModel]);
|
||||
|
||||
const form = useZodForm(schema, { defaultValues: toRoutingGroupFormValues(initialValue, availableStrategies) });
|
||||
|
||||
|
|
@ -124,7 +135,7 @@ const RoutingGroupModal: React.FC<RoutingGroupModalProps> = ({
|
|||
control={form.control}
|
||||
name="models"
|
||||
label="Models"
|
||||
description="Models from your model list that this group routes between."
|
||||
description="Models from your model list that this group routes between. A model can only be in one group."
|
||||
>
|
||||
{({ id, value, onChange, "aria-invalid": ariaInvalid, "aria-describedby": ariaDescribedBy }) => (
|
||||
<Combobox multiple items={modelOptions} value={value} onValueChange={onChange}>
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ import RoutingGroupsTable from "./RoutingGroupsTable";
|
|||
import RoutingGroupModal from "./RoutingGroupModal";
|
||||
import { toast } from "@/lib/toast";
|
||||
import type { RoutingGroup } from "./types";
|
||||
import { groupNameByModel } from "./modelOwnership";
|
||||
import { Dialog, DialogContent, DialogFooter, DialogHeader, DialogTitle } from "@/components/ui/dialog";
|
||||
|
||||
const RoutingGroups: React.FC = () => {
|
||||
|
|
@ -30,7 +31,7 @@ const RoutingGroups: React.FC = () => {
|
|||
const [editingGroup, setEditingGroup] = useState<RoutingGroup | null>(null);
|
||||
const [deletingGroup, setDeletingGroup] = useState<RoutingGroup | null>(null);
|
||||
|
||||
const groups = data?.routingGroups ?? [];
|
||||
const groups = useMemo(() => data?.routingGroups ?? [], [data?.routingGroups]);
|
||||
|
||||
const filteredGroups = useMemo(() => {
|
||||
const q = searchQuery.trim().toLowerCase();
|
||||
|
|
@ -51,6 +52,11 @@ const RoutingGroups: React.FC = () => {
|
|||
|
||||
const strategyDescriptions = routerFields?.routing_strategy_descriptions ?? {};
|
||||
|
||||
const ownerByModel = useMemo(
|
||||
() => groupNameByModel(groups, drawerMode === "edit" ? editingGroup?.group_name : undefined),
|
||||
[groups, drawerMode, editingGroup],
|
||||
);
|
||||
|
||||
const modelOptions = useMemo<string[]>(() => {
|
||||
const records = (modelHub?.data ?? []) as Array<{ model_group?: string }>;
|
||||
const names = records.map((r) => r.model_group).filter((n): n is string => Boolean(n));
|
||||
|
|
@ -160,6 +166,7 @@ const RoutingGroups: React.FC = () => {
|
|||
strategyDescriptions={strategyDescriptions}
|
||||
modelOptions={modelOptions}
|
||||
existingGroupNames={groups.map((g) => g.group_name)}
|
||||
groupNameByModel={ownerByModel}
|
||||
onClose={() => setDrawerOpen(false)}
|
||||
onSubmit={handleSubmit}
|
||||
saving={saveMutation.isPending}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,33 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
|
||||
import { groupNameByModel, modelConflictError } from "./modelOwnership";
|
||||
import type { RoutingGroup } from "./types";
|
||||
|
||||
const groups: RoutingGroup[] = [
|
||||
{ group_name: "cheap", models: ["m1", "m2"], routing_strategy: "latency-based-routing" },
|
||||
{ group_name: "security", models: ["m3"], routing_strategy: "least-busy" },
|
||||
];
|
||||
|
||||
describe("groupNameByModel", () => {
|
||||
it("maps every claimed model to its owning group", () => {
|
||||
expect(groupNameByModel(groups)).toEqual({ m1: "cheap", m2: "cheap", m3: "security" });
|
||||
});
|
||||
|
||||
it("excludes the group being edited so its own models stay selectable", () => {
|
||||
expect(groupNameByModel(groups, "cheap")).toEqual({ m3: "security" });
|
||||
});
|
||||
});
|
||||
|
||||
describe("modelConflictError", () => {
|
||||
it("passes models that no other group claims", () => {
|
||||
expect(modelConflictError(["m4"], groupNameByModel(groups, "cheap"))).toBeNull();
|
||||
expect(modelConflictError(undefined, groupNameByModel(groups))).toBeNull();
|
||||
});
|
||||
|
||||
it("names every model already claimed by another group", () => {
|
||||
const error = modelConflictError(["m1", "m3", "m4"], groupNameByModel(groups));
|
||||
expect(error).toBe(
|
||||
'Each model may belong to at most one group. Already claimed: m1 (in "cheap"), m3 (in "security")',
|
||||
);
|
||||
});
|
||||
});
|
||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue