diff --git a/litellm/anthropic_beta_headers_manager.py b/litellm/anthropic_beta_headers_manager.py index 7e7099a53b0..8b7fa1c9347 100644 --- a/litellm/anthropic_beta_headers_manager.py +++ b/litellm/anthropic_beta_headers_manager.py @@ -29,7 +29,7 @@ from typing import Final import httpx -from litellm.litellm_core_utils.litellm_logging import verbose_logger +from litellm._logging import verbose_logger # Cache for the loaded configuration _BETA_HEADERS_CONFIG: dict | None = None diff --git a/litellm/integrations/SlackAlerting/slack_alerting.py b/litellm/integrations/SlackAlerting/slack_alerting.py index 4c2b722ef90..34ba857ecb6 100644 --- a/litellm/integrations/SlackAlerting/slack_alerting.py +++ b/litellm/integrations/SlackAlerting/slack_alerting.py @@ -39,7 +39,6 @@ from litellm.llms.custom_httpx.http_handler import ( httpxSpecialProvider, ) from litellm.proxy._types import ( - AlertType, CallInfo, InvitationModel, InvitationNew, @@ -52,6 +51,7 @@ from litellm.repositories.table_repositories import InvitationLinkRepository from litellm.repositories.team_repository import TeamRepository from litellm.repositories.user_repository import UserRepository from litellm.types.integrations.slack_alerting import * +from litellm.types.integrations.slack_alerting import AlertType from litellm.types.proxy.model_deprecation import ( DEFAULT_DEPRECATION_CHECK_INTERVAL_SECONDS, DEPRECATION_IDLE_POLL_SECONDS, diff --git a/litellm/integrations/SlackAlerting/utils.py b/litellm/integrations/SlackAlerting/utils.py index eb3a7f80f72..da256e5143f 100644 --- a/litellm/integrations/SlackAlerting/utils.py +++ b/litellm/integrations/SlackAlerting/utils.py @@ -8,8 +8,8 @@ from typing import TYPE_CHECKING, Any, Final import litellm from litellm.integrations.custom_logger import CustomLogger -from litellm.proxy._types import AlertType from litellm.secret_managers.main import get_secret +from litellm.types.integrations.slack_alerting import AlertType if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging as _Logging diff --git a/litellm/integrations/langtrace.py b/litellm/integrations/langtrace.py index 53f5d2a0318..eeff76e7063 100644 --- a/litellm/integrations/langtrace.py +++ b/litellm/integrations/langtrace.py @@ -1,7 +1,7 @@ import json from typing import TYPE_CHECKING, Any, Final -from litellm.proxy._types import SpanAttributes +from litellm.types.integrations.otel_span_attributes import SpanAttributes if TYPE_CHECKING: from opentelemetry.trace import Span as _Span diff --git a/litellm/integrations/opentelemetry.py b/litellm/integrations/opentelemetry.py index 956e754944a..290cba90fd7 100644 --- a/litellm/integrations/opentelemetry.py +++ b/litellm/integrations/opentelemetry.py @@ -14,7 +14,9 @@ import litellm from litellm._logging import verbose_logger from litellm.integrations._types.open_inference import ( OpenInferenceSpanKindValues, - SpanAttributes, +) +from litellm.integrations._types.open_inference import ( + SpanAttributes as OpenInferenceSpanAttributes, ) from litellm.integrations.custom_logger import CustomLogger from litellm.integrations.langtrace import LANGTRACE_TRACE_PATH @@ -38,6 +40,7 @@ from litellm.litellm_core_utils.service_tier_utils import ( get_served_service_tier, ) from litellm.secret_managers.main import get_secret_bool, str_to_bool +from litellm.types.integrations.otel_span_attributes import SpanAttributes from litellm.types.services import ServiceLoggerPayload from litellm.types.utils import ( ChatCompletionMessageToolCall, @@ -2055,7 +2058,7 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger): self.safe_set_attribute( span=guardrail_span, - key=SpanAttributes.OPENINFERENCE_SPAN_KIND, + key=OpenInferenceSpanAttributes.OPENINFERENCE_SPAN_KIND, value=OpenInferenceSpanKindValues.GUARDRAIL.value, ) @@ -2306,8 +2309,6 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger): def set_tools_attributes(self, span: Span, tools): import json - from litellm.proxy._types import SpanAttributes - if not tools: return @@ -2357,8 +2358,6 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger): def _tool_calls_kv_pair( tool_calls: list[ChatCompletionMessageToolCall], ) -> dict[str, object]: - from litellm.proxy._types import SpanAttributes - kv_pairs: Final[dict[str, object]] = {} for idx, tool_call in enumerate(tool_calls): _function = tool_call.get("function") @@ -2394,8 +2393,6 @@ class OpenTelemetry(OTELGenAISemconvMixin, CustomLogger): set_weave_otel_attributes(span, kwargs, response_obj) return - from litellm.proxy._types import SpanAttributes - optional_params: Final = kwargs.get("optional_params", {}) litellm_params: Final = kwargs.get("litellm_params", {}) or {} standard_logging_payload: Final[StandardLoggingPayload | None] = kwargs.get("standard_logging_object") diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index 6da3807ab15..476e154ed0f 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -39,7 +39,6 @@ from litellm.llms.anthropic.wif import ( ) from litellm.llms.base_llm.base_utils import BaseLLMModelInfo, BaseTokenCounter from litellm.llms.base_llm.chat.transformation import BaseLLMException -from litellm.proxy._types import SpecialHeaders from litellm.types.llms.anthropic import ( ANTHROPIC_HOSTED_TOOLS, ANTHROPIC_MID_CONVERSATION_OUTPUT_CONFIG_BETA_HEADER, @@ -53,6 +52,7 @@ from litellm.types.llms.anthropic import ( AnthropicThinkingParam, ) from litellm.types.llms.openai import AllMessageValues +from litellm.types.proxy.auth.special_headers import SpecialHeaders from litellm.types.proxy.model_listing import ModelInfoResponse from litellm.types.utils import LlmProviders diff --git a/litellm/llms/anthropic/pass_through/messages/transformation.py b/litellm/llms/anthropic/pass_through/messages/transformation.py index bc61b133f4e..1cf5d2fcc6e 100644 --- a/litellm/llms/anthropic/pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/pass_through/messages/transformation.py @@ -3,9 +3,9 @@ from typing import Any, ClassVar, Final import httpx +from litellm._logging import verbose_logger from litellm.exceptions import AuthenticationError from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj -from litellm.litellm_core_utils.litellm_logging import verbose_logger from litellm.llms.base_llm.anthropic_messages.transformation import ( BaseAnthropicMessagesConfig, ) diff --git a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py index 6234ca3a9c3..4baa77f0ed7 100644 --- a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py @@ -5,6 +5,7 @@ from typing import TYPE_CHECKING, Any, Final, cast import httpx import litellm +from litellm._logging import verbose_logger from litellm.anthropic_beta_headers_manager import filter_and_transform_beta_headers from litellm.constants import ( BEDROCK_MIN_THINKING_BUDGET_TOKENS, @@ -12,7 +13,6 @@ from litellm.constants import ( DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET, ) -from litellm.litellm_core_utils.litellm_logging import verbose_logger from litellm.llms.anthropic.chat.transformation import ( DROP_UNSUPPORTED_OUTPUT_CONFIG_WARNING, AnthropicConfig, diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index c21c88c3505..c0f2472a6bd 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -34,6 +34,9 @@ from litellm.types.agents import AgentCaller, AgentResponse from litellm.types.integrations.compression_interception import ( CompressionSavingsMetadata, ) +from litellm.types.integrations.otel_span_attributes import ( + SpanAttributes as SpanAttributes, # noqa: PLC0414 # public re-export +) from litellm.types.integrations.slack_alerting import AlertType from litellm.types.llms.openai import ( AllMessageValues, @@ -52,6 +55,9 @@ from litellm.types.mcp import ( ) from litellm.types.mcp_server.mcp_server_manager import MCPInfo from litellm.types.proxy.agent_identity import ManagedAgentContext +from litellm.types.proxy.auth.special_headers import ( + SpecialHeaders as SpecialHeaders, # noqa: PLC0414 # public re-export +) from litellm.types.proxy.carried_budget_state import ( OrgBudgetSnapshot, TeamBudgetSnapshot, @@ -59,7 +65,7 @@ from litellm.types.proxy.carried_budget_state import ( ) from litellm.types.proxy.control_plane_endpoints import WorkerRegistryEntry from litellm.types.proxy.spend_capture_rate import SpendCaptureRateCheckSettings -from litellm.types.router import RouterErrors, UpdateRouterConfig +from litellm.types.router import AllowedModelRegion, RouterErrors, UpdateRouterConfig from litellm.types.router_weights import validate_router_settings_dict from litellm.types.secret_managers.main import KeyManagementSystem from litellm.types.utils import ( @@ -2064,9 +2070,6 @@ class DeleteUserRequest(LiteLLMPydanticObjectBase): user_ids: list[str] # required -AllowedModelRegion = Literal["eu", "us"] - - class BudgetNewRequest(LiteLLMPydanticObjectBase): budget_id: str | None = Field(default=None, description="The unique budget id.") max_budget: float | None = Field( @@ -4299,66 +4302,6 @@ class SpendLogsPayload(TypedDict): litellm_call_id: ReadOnly[str | None] -class SpanAttributes(str, enum.Enum): - # Note: We've taken this from opentelemetry-semantic-conventions-ai - # I chose to not add a new dependency to litellm for this - - # Semantic Conventions for LLM requests, this needs to be removed after - # OpenTelemetry Semantic Conventions support Gen AI. - # Issue at https://github.com/open-telemetry/opentelemetry-python/issues/3868 - # Refer to https://github.com/open-telemetry/semantic-conventions/blob/main/docs/gen-ai/llm-spans.md - - LLM_SYSTEM = "gen_ai.system" - LLM_REQUEST_MODEL = "gen_ai.request.model" - LLM_REQUEST_MAX_TOKENS = "gen_ai.request.max_tokens" - LLM_REQUEST_TEMPERATURE = "gen_ai.request.temperature" - LLM_REQUEST_TOP_P = "gen_ai.request.top_p" - LLM_PROMPTS = "gen_ai.prompt" - LLM_COMPLETIONS = "gen_ai.completion" - LLM_RESPONSE_MODEL = "gen_ai.response.model" - LLM_USAGE_COMPLETION_TOKENS = "gen_ai.usage.completion_tokens" - LLM_USAGE_PROMPT_TOKENS = "gen_ai.usage.prompt_tokens" - - # OTEL 1.38 attributes - GEN_AI_INPUT_MESSAGES = "gen_ai.input.messages" - GEN_AI_OUTPUT_MESSAGES = "gen_ai.output.messages" - GEN_AI_USAGE_INPUT_TOKENS = "gen_ai.usage.input_tokens" - GEN_AI_USAGE_OUTPUT_TOKENS = "gen_ai.usage.output_tokens" - GEN_AI_USAGE_TOTAL_TOKENS = "gen_ai.usage.total_tokens" - GEN_AI_OPERATION_NAME = "gen_ai.operation.name" - GEN_AI_REQUEST_ID = "gen_ai.request.id" - GEN_AI_SYSTEM_INSTRUCTIONS = "gen_ai.system_instructions" - GEN_AI_RESPONSE_FINISH_REASONS = "gen_ai.response.finish_reasons" - - LLM_TOKEN_TYPE = "gen_ai.token.type" - # To be added - # LLM_RESPONSE_FINISH_REASON = "gen_ai.response.finish_reasons" - # LLM_RESPONSE_ID = "gen_ai.response.id" - - # LLM - LLM_REQUEST_TYPE = "llm.request.type" - LLM_USAGE_TOTAL_TOKENS = "llm.usage.total_tokens" - LLM_USAGE_TOKEN_TYPE = "llm.usage.token_type" - LLM_USER = "llm.user" - LLM_HEADERS = "llm.headers" - LLM_TOP_K = "llm.top_k" - LLM_IS_STREAMING = "llm.is_streaming" - LLM_FREQUENCY_PENALTY = "llm.frequency_penalty" - LLM_PRESENCE_PENALTY = "llm.presence_penalty" - LLM_CHAT_STOP_SEQUENCES = "llm.chat.stop_sequences" - LLM_REQUEST_FUNCTIONS = "llm.request.functions" - LLM_REQUEST_REPETITION_PENALTY = "llm.request.repetition_penalty" - LLM_RESPONSE_FINISH_REASON = "llm.response.finish_reason" - LLM_RESPONSE_STOP_REASON = "llm.response.stop_reason" - LLM_CONTENT_COMPLETION_CHUNK = "llm.content.completion.chunk" - - # OpenAI - LLM_OPENAI_RESPONSE_SYSTEM_FINGERPRINT = "gen_ai.openai.system_fingerprint" - LLM_OPENAI_API_BASE = "gen_ai.openai.api_base" - LLM_OPENAI_API_VERSION = "gen_ai.openai.api_version" - LLM_OPENAI_API_TYPE = "gen_ai.openai.api_type" - - class ManagementEndpointLoggingPayload(LiteLLMPydanticObjectBase): route: str request_data: dict @@ -4993,42 +4936,6 @@ class JWTKeyMappingResponse(LiteLLMPydanticObjectBase): updated_by: str | None = None -class SpecialHeaders(enum.Enum): - """Used by user_api_key_auth.py to get litellm key""" - - openai_authorization = "Authorization" - azure_authorization = "API-Key" - anthropic_authorization = "x-api-key" - google_ai_studio_authorization = "x-goog-api-key" - azure_apim_authorization = "Ocp-Apim-Subscription-Key" - custom_litellm_api_key = "x-litellm-api-key" - mcp_auth = "x-mcp-auth" - mcp_servers = "x-mcp-servers" - mcp_access_groups = "x-mcp-access-groups" - - @classmethod - def litellm_credential_header_names(cls) -> "frozenset[str]": - """Lowercased header names user_api_key_auth accepts as a litellm key. - - Every header here authenticates the caller, so any code that forwards a - request onward (e.g. the plugin reverse proxy) must strip all of them to - avoid leaking the caller's litellm credential downstream. The static - custom-key header (general_settings.litellm_key_header_name) is runtime - config and must be added on top of this set by the caller. - """ - return frozenset( - header.value.lower() - for header in ( - cls.openai_authorization, - cls.azure_authorization, - cls.anthropic_authorization, - cls.google_ai_studio_authorization, - cls.azure_apim_authorization, - cls.custom_litellm_api_key, - ) - ) - - class LitellmDataForBackendLLMCall(TypedDict, total=False): headers: dict organization: str diff --git a/litellm/secret_managers/aws_secret_manager.py b/litellm/secret_managers/aws_secret_manager.py index b3a9fbf31c8..b95557789be 100644 --- a/litellm/secret_managers/aws_secret_manager.py +++ b/litellm/secret_managers/aws_secret_manager.py @@ -17,7 +17,7 @@ from typing import Any, Final from pydantic import TypeAdapter import litellm -from litellm.proxy._types import KeyManagementSystem +from litellm.types.secret_managers.main import KeyManagementSystem _PARSED_LITERAL: Final = TypeAdapter(object) diff --git a/litellm/secret_managers/aws_secret_manager_v2.py b/litellm/secret_managers/aws_secret_manager_v2.py index 80fe3f38c03..8b78980eb73 100644 --- a/litellm/secret_managers/aws_secret_manager_v2.py +++ b/litellm/secret_managers/aws_secret_manager_v2.py @@ -30,11 +30,10 @@ from litellm.llms.custom_httpx.http_handler import ( _get_httpx_client, get_async_httpx_client, ) -from litellm.proxy._types import KeyManagementSystem from litellm.rust_bridge.secret_manager import resolve_native_provider_reader from litellm.secret_managers.main import get_secret_str from litellm.types.llms.custom_http import httpxSpecialProvider -from litellm.types.secret_managers.main import KeyManagementSettings +from litellm.types.secret_managers.main import KeyManagementSettings, KeyManagementSystem from .base_secret_manager import BaseSecretManager diff --git a/litellm/secret_managers/cyberark_secret_manager.py b/litellm/secret_managers/cyberark_secret_manager.py index f8e17488167..b869db51c85 100644 --- a/litellm/secret_managers/cyberark_secret_manager.py +++ b/litellm/secret_managers/cyberark_secret_manager.py @@ -16,8 +16,8 @@ from litellm.llms.custom_httpx.http_handler import ( get_async_httpx_client, httpxSpecialProvider, ) -from litellm.proxy._types import KeyManagementSystem from litellm.rust_bridge.secret_manager import resolve_native_provider_reader, resolve_native_provider_writer +from litellm.types.secret_managers.main import KeyManagementSystem from .base_secret_manager import BaseSecretManager, raise_if_unsafe_secret_name from .main import str_to_bool diff --git a/litellm/secret_managers/google_kms.py b/litellm/secret_managers/google_kms.py index 86d69be3294..50d0695284b 100644 --- a/litellm/secret_managers/google_kms.py +++ b/litellm/secret_managers/google_kms.py @@ -12,7 +12,7 @@ import os from typing import Final import litellm -from litellm.proxy._types import KeyManagementSystem +from litellm.types.secret_managers.main import KeyManagementSystem def validate_environment(): diff --git a/litellm/secret_managers/google_secret_manager.py b/litellm/secret_managers/google_secret_manager.py index 913fd4d5204..f8de9b25e96 100644 --- a/litellm/secret_managers/google_secret_manager.py +++ b/litellm/secret_managers/google_secret_manager.py @@ -8,8 +8,9 @@ from litellm.caching.caching import InMemoryCache from litellm.constants import SECRET_MANAGER_REFRESH_INTERVAL from litellm.integrations.gcs_bucket.gcs_bucket_base import GCSBucketBase from litellm.llms.custom_httpx.http_handler import _get_httpx_client -from litellm.proxy._types import CommonProxyErrors, KeyManagementSystem +from litellm.proxy._types import CommonProxyErrors from litellm.rust_bridge.secret_manager import resolve_native_provider_reader +from litellm.types.secret_managers.main import KeyManagementSystem class GoogleSecretManager(GCSBucketBase): diff --git a/litellm/secret_managers/hashicorp_secret_manager.py b/litellm/secret_managers/hashicorp_secret_manager.py index 27523892d61..d130481f8fc 100644 --- a/litellm/secret_managers/hashicorp_secret_manager.py +++ b/litellm/secret_managers/hashicorp_secret_manager.py @@ -15,8 +15,8 @@ from litellm.llms.custom_httpx.http_handler import ( get_async_httpx_client, httpxSpecialProvider, ) -from litellm.proxy._types import KeyManagementSystem from litellm.rust_bridge.secret_manager import resolve_native_provider_reader, resolve_native_provider_writer +from litellm.types.secret_managers.main import KeyManagementSystem from .base_secret_manager import BaseSecretManager, raise_if_unsafe_secret_name diff --git a/litellm/types/integrations/otel_span_attributes.py b/litellm/types/integrations/otel_span_attributes.py new file mode 100644 index 00000000000..455eaa94c77 --- /dev/null +++ b/litellm/types/integrations/otel_span_attributes.py @@ -0,0 +1,61 @@ +import enum + + +class SpanAttributes(str, enum.Enum): + # Note: We've taken this from opentelemetry-semantic-conventions-ai + # I chose to not add a new dependency to litellm for this + + # Semantic Conventions for LLM requests, this needs to be removed after + # OpenTelemetry Semantic Conventions support Gen AI. + # Issue at https://github.com/open-telemetry/opentelemetry-python/issues/3868 + # Refer to https://github.com/open-telemetry/semantic-conventions/blob/main/docs/gen-ai/llm-spans.md + + LLM_SYSTEM = "gen_ai.system" + LLM_REQUEST_MODEL = "gen_ai.request.model" + LLM_REQUEST_MAX_TOKENS = "gen_ai.request.max_tokens" + LLM_REQUEST_TEMPERATURE = "gen_ai.request.temperature" + LLM_REQUEST_TOP_P = "gen_ai.request.top_p" + LLM_PROMPTS = "gen_ai.prompt" + LLM_COMPLETIONS = "gen_ai.completion" + LLM_RESPONSE_MODEL = "gen_ai.response.model" + LLM_USAGE_COMPLETION_TOKENS = "gen_ai.usage.completion_tokens" + LLM_USAGE_PROMPT_TOKENS = "gen_ai.usage.prompt_tokens" + + # OTEL 1.38 attributes + GEN_AI_INPUT_MESSAGES = "gen_ai.input.messages" + GEN_AI_OUTPUT_MESSAGES = "gen_ai.output.messages" + GEN_AI_USAGE_INPUT_TOKENS = "gen_ai.usage.input_tokens" + GEN_AI_USAGE_OUTPUT_TOKENS = "gen_ai.usage.output_tokens" + GEN_AI_USAGE_TOTAL_TOKENS = "gen_ai.usage.total_tokens" + GEN_AI_OPERATION_NAME = "gen_ai.operation.name" + GEN_AI_REQUEST_ID = "gen_ai.request.id" + GEN_AI_SYSTEM_INSTRUCTIONS = "gen_ai.system_instructions" + GEN_AI_RESPONSE_FINISH_REASONS = "gen_ai.response.finish_reasons" + + LLM_TOKEN_TYPE = "gen_ai.token.type" + # To be added + # LLM_RESPONSE_FINISH_REASON = "gen_ai.response.finish_reasons" + # LLM_RESPONSE_ID = "gen_ai.response.id" + + # LLM + LLM_REQUEST_TYPE = "llm.request.type" + LLM_USAGE_TOTAL_TOKENS = "llm.usage.total_tokens" + LLM_USAGE_TOKEN_TYPE = "llm.usage.token_type" + LLM_USER = "llm.user" + LLM_HEADERS = "llm.headers" + LLM_TOP_K = "llm.top_k" + LLM_IS_STREAMING = "llm.is_streaming" + LLM_FREQUENCY_PENALTY = "llm.frequency_penalty" + LLM_PRESENCE_PENALTY = "llm.presence_penalty" + LLM_CHAT_STOP_SEQUENCES = "llm.chat.stop_sequences" + LLM_REQUEST_FUNCTIONS = "llm.request.functions" + LLM_REQUEST_REPETITION_PENALTY = "llm.request.repetition_penalty" + LLM_RESPONSE_FINISH_REASON = "llm.response.finish_reason" + LLM_RESPONSE_STOP_REASON = "llm.response.stop_reason" + LLM_CONTENT_COMPLETION_CHUNK = "llm.content.completion.chunk" + + # OpenAI + LLM_OPENAI_RESPONSE_SYSTEM_FINGERPRINT = "gen_ai.openai.system_fingerprint" + LLM_OPENAI_API_BASE = "gen_ai.openai.api_base" + LLM_OPENAI_API_VERSION = "gen_ai.openai.api_version" + LLM_OPENAI_API_TYPE = "gen_ai.openai.api_type" diff --git a/litellm/types/proxy/auth/special_headers.py b/litellm/types/proxy/auth/special_headers.py new file mode 100644 index 00000000000..7449e47534e --- /dev/null +++ b/litellm/types/proxy/auth/special_headers.py @@ -0,0 +1,37 @@ +import enum + + +class SpecialHeaders(enum.Enum): + """Used by user_api_key_auth.py to get litellm key""" + + openai_authorization = "Authorization" + azure_authorization = "API-Key" + anthropic_authorization = "x-api-key" + google_ai_studio_authorization = "x-goog-api-key" + azure_apim_authorization = "Ocp-Apim-Subscription-Key" + custom_litellm_api_key = "x-litellm-api-key" + mcp_auth = "x-mcp-auth" + mcp_servers = "x-mcp-servers" + mcp_access_groups = "x-mcp-access-groups" + + @classmethod + def litellm_credential_header_names(cls) -> "frozenset[str]": + """Lowercased header names user_api_key_auth accepts as a litellm key. + + Every header here authenticates the caller, so any code that forwards a + request onward (e.g. the plugin reverse proxy) must strip all of them to + avoid leaking the caller's litellm credential downstream. The static + custom-key header (general_settings.litellm_key_header_name) is runtime + config and must be added on top of this set by the caller. + """ + return frozenset( + header.value.lower() + for header in ( + cls.openai_authorization, + cls.azure_authorization, + cls.anthropic_authorization, + cls.google_ai_studio_authorization, + cls.azure_apim_authorization, + cls.custom_litellm_api_key, + ) + ) diff --git a/litellm/types/router.py b/litellm/types/router.py index e131a7184d4..6c52120c3ae 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -6,7 +6,18 @@ import datetime import enum from collections.abc import Container, Mapping, Sequence from dataclasses import dataclass -from typing import TYPE_CHECKING, Annotated, Any, ClassVar, Final, Generic, Literal, TypeVar, get_type_hints +from typing import ( + TYPE_CHECKING, + Annotated, + Any, + ClassVar, + Final, + Generic, + Literal, + TypeAlias, + TypeVar, + get_type_hints, +) from zoneinfo import ZoneInfo, ZoneInfoNotFoundError import httpx @@ -39,6 +50,8 @@ from .utils import ( server_owned_wif_litellm_params as _server_owned_wif_litellm_params, ) +AllowedModelRegion: TypeAlias = Literal["eu", "us"] + class ConfigurableClientsideParamsCustomAuth(TypedDict): api_base: str diff --git a/litellm/utils.py b/litellm/utils.py index a7d7447d7f3..a2dac2fcf8d 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -432,7 +432,6 @@ if TYPE_CHECKING: ) from litellm.llms.cohere.common_utils import CohereModelInfo from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler - from litellm.proxy._types import AllowedModelRegion from litellm.router_utils.get_retry_from_policy import ( get_num_retries_from_retry_policy, reset_retry_policy, @@ -447,7 +446,7 @@ if TYPE_CHECKING: ChatCompletionToolCallFunctionChunk, ) from litellm.types.rerank import RerankResponse - from litellm.types.router import LiteLLM_Params + from litellm.types.router import AllowedModelRegion, LiteLLM_Params from litellm.llms.base_llm.chat.transformation import BaseConfig from litellm.llms.base_llm.completion.transformation import BaseTextCompletionConfig