diff --git a/litellm/integrations/shadow_eval_logger.py b/litellm/integrations/shadow_eval_logger.py index 7f5d9fedd3d..fceba9e391c 100644 --- a/litellm/integrations/shadow_eval_logger.py +++ b/litellm/integrations/shadow_eval_logger.py @@ -30,7 +30,6 @@ from litellm.constants import INTERNAL_CALL_ORIGIN_METADATA_KEY from litellm.integrations.custom_logger import CustomLogger from litellm.integrations.websearch_interception.tools import is_web_search_tool_responses from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs, independent_snapshot -from litellm.litellm_core_utils.internal_call_metadata import sanitized_forwardable_call_metadata from litellm.litellm_core_utils.llm_judge import ( default_router_provider, extract_text_from_content, @@ -620,50 +619,30 @@ def _record_funnel_event(job_id: str, stage: "ShadowEvalFunnelStage") -> None: verbose_logger.debug("shadow_eval: funnel increment failed for %s: %s", job_id, e) -async def _key_or_team_is_over_budget(metadata: Mapping[str, object]) -> bool: - """Whether the shadowed key or its team is over budget, decided by the same owners - the request path uses, so counter keys and thresholds can never drift from auth's. - - Advisory and fail-open: real traffic on an over-budget key is already rejected at - auth (so nothing reaches the success hook), and this gate only closes the race - where the key crosses its budget while a request is in flight. - """ +async def _admit_creator(creator_user_id: str | None, models: Sequence[str]) -> Mapping[str, object] | None: + """The call metadata the job's creator pays under, or None when the creator cannot pay + for ``models`` or that cannot be verified, so the sample is withheld before any spend.""" + if creator_user_id is None: + return None try: - from litellm.exceptions import BudgetExceededError - from litellm.proxy._types import UserAPIKeyAuth - from litellm.proxy.auth.auth_checks import ( - _team_max_budget_check, - _virtual_key_max_budget_check, - get_team_object, + from litellm.proxy.auth.evaluation_principal import ( + evaluation_principal, + principal_call_metadata, + principal_can_pay_for, ) - from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache - except ImportError: - return False - auth: Final = metadata.get("user_api_key_auth") - if not isinstance(auth, UserAPIKeyAuth): - return False - try: - await _virtual_key_max_budget_check(valid_token=auth, proxy_logging_obj=proxy_logging_obj) - if auth.team_id: - team: Final = await get_team_object( - team_id=auth.team_id, - prisma_client=prisma_client, - user_api_key_cache=user_api_key_cache, - check_cache_only=True, - ) - await _team_max_budget_check(team_object=team, valid_token=auth, proxy_logging_obj=proxy_logging_obj) - except BudgetExceededError: - return True - except Exception as e: # noqa: BLE001 # advisory gate: a failed read must not block sampling - verbose_logger.debug("shadow_eval: budget read failed: %s", e) - return False + principal: Final = await evaluation_principal(creator_user_id) + if not await principal_can_pay_for(principal, models): + return None + return principal_call_metadata(principal) + except Exception as e: # noqa: BLE001 # an unverifiable creator budget withholds the sample rather than spending + verbose_logger.warning("shadow_eval: creator %s budget unverifiable, sample withheld: %s", creator_user_id, e) + return None def _forwarded_team_id(metadata: Mapping[str, object]) -> str | None: - """The shadowed key's team, the identity the judge call already carries in its metadata - and the router already selects deployments with. Read here too so the arm choice, which - happens before the router sees the call, is made under the same team.""" + """The shadowed key's team, the scope start-time validation resolved the judge under, + so the judge's router-or-SDK arm is chosen the way the job was validated.""" team_id: Final = metadata.get("user_api_key_team_id") return team_id if isinstance(team_id, str) and team_id else None @@ -749,6 +728,7 @@ class ActiveShadowEvalJob(BaseModel): judge_model: str max_turns: int max_budget: float | None = None + created_by: str | None = None ends_at: datetime attempts: int = 0 spend: float = 0.0 @@ -833,6 +813,7 @@ class ShadowEvalLogger(CustomLogger): job_spend_reader: Callable[[str, float, float], Awaitable[float]] | None = None, job_spend_writer: Callable[[str, float], Awaitable[None]] | None = None, funnel_recorder: Callable[[str, "ShadowEvalFunnelStage"], None] | None = None, + creator_admission: Callable[[str | None, Sequence[str]], Awaitable[Mapping[str, object] | None]] | None = None, ) -> None: """Providers are callables so the proxy's lazily-initialized globals are resolved at call time, not at logger construction. The spend reader and writer wrap the @@ -843,6 +824,7 @@ class ShadowEvalLogger(CustomLogger): self._read_job_spend = job_spend_reader or _job_spend_from_counter self._write_job_spend = job_spend_writer or _add_job_spend_to_counter self._record_funnel = funnel_recorder or _record_funnel_event + self._admit_creator = creator_admission or _admit_creator self._inflight_shadow_tasks: int = 0 # Starts per job since the last cache fill, never decremented within a # generation; the refill absorbs written rows and resets. @@ -1055,8 +1037,9 @@ class ShadowEvalLogger(CustomLogger): ) -> None: """Budget gates once per sampled request, then every router arm in turn: shadow call -> blind judge -> one attempt row stamped with the arm. The gates that - decline to spend on an admitted sample (no DB to record into, an over-budget key, - an unverifiable or exhausted eval budget) count the REQUEST withheld before any + decline to spend on an admitted sample (no DB to record into, an unverifiable or + exhausted eval budget, a creator who cannot pay for every model the sample would + call) count the REQUEST withheld before any arm runs, so funnel counters stay per-request and a leg's eligible traffic still reconciles as not_sampled + unjudgeable + shed + withheld + sampled requests, where each sampled request writes one attempt row per arm. A budget crossed @@ -1069,9 +1052,6 @@ class ShadowEvalLogger(CustomLogger): if prisma is None: self._record_funnel(job.id, "withheld") return - if await _key_or_team_is_over_budget(parent_metadata): - self._record_funnel(job.id, "withheld") - return if job.max_budget is not None: try: spend: Final = await self._read_job_spend(_job_spend_counter_key(job.id), job.spend, job.max_budget) @@ -1082,6 +1062,12 @@ class ShadowEvalLogger(CustomLogger): if spend >= job.max_budget: self._record_funnel(job.id, "withheld") return + creator_metadata: Final = await self._admit_creator( + job.created_by, (*(job.arm_target(arm) for arm in job.arm_router_names), job.judge_model) + ) + if creator_metadata is None: + self._record_funnel(job.id, "withheld") + return for arm_router in job.arm_router_names: await self._run_shadow_arm( prisma=prisma, @@ -1097,6 +1083,7 @@ class ShadowEvalLogger(CustomLogger): control_tier=control_tier, shadow_params=shadow_params, parent_metadata=parent_metadata, + creator_metadata=creator_metadata, ) async def _run_shadow_arm( @@ -1114,12 +1101,13 @@ class ShadowEvalLogger(CustomLogger): control_tier: str | None, shadow_params: Mapping[str, object], parent_metadata: Mapping[str, object], + creator_metadata: Mapping[str, object], ) -> None: """One arm's pipeline: shadow call -> blind judge -> one attempt row, every exit recording this arm's outcome, so one arm's fault never silences a sibling arm.""" try: shadow: Final = await self._call_router_shadow( - job.arm_target(arm_router), messages, shadow_params, parent_metadata + job.arm_target(arm_router), messages, shadow_params, creator_metadata ) except Exception as e: # noqa: BLE001 # detached task: nothing billed yet, record and never raise verbose_logger.debug("shadow_eval: pipeline failed for %s: %s", request_id, e) @@ -1161,6 +1149,7 @@ class ShadowEvalLogger(CustomLogger): shadow_text=shadow.text, tools=shadow_params.get("tools"), parent_metadata=parent_metadata, + creator_metadata=creator_metadata, ) if isinstance(verdict, _CallFailure): await self._record_attempt( @@ -1268,18 +1257,19 @@ class ShadowEvalLogger(CustomLogger): target_model: str, messages: Sequence[Mapping[str, object]], shadow_params: Mapping[str, object], - parent_metadata: Mapping[str, object], + creator_metadata: Mapping[str, object], ) -> "_ShadowResponse | _CallFailure": """Send the prompt through the arm nobody was served: the auto-router under - evaluation, or a reverse job's fixed baseline model. The metadata carries the - shadowed key's identity (spend attribution) and receives a routing decision - write-back, which a plain baseline model simply never makes.""" + evaluation, or a reverse job's fixed baseline model. The call runs as the job's + creator and receives a routing decision write-back, which a plain baseline model + simply never makes.""" router: Final = self._router_provider() if router is None: return _CallFailure("no router configured on this pod") - shadow_metadata: Final[dict[str, object]] = ( # mutable-ok: router writes its routing decision back - sanitized_forwardable_call_metadata(parent_metadata, SHADOW_EVAL_ROUTER_CALL_ORIGIN) - ) + shadow_metadata: Final[dict[str, object]] = { # mutable-ok: router writes its routing decision back + **creator_metadata, + INTERNAL_CALL_ORIGIN_METADATA_KEY: SHADOW_EVAL_ROUTER_CALL_ORIGIN, + } try: response: Final = await router.acompletion( model=target_model, @@ -1321,6 +1311,7 @@ class ShadowEvalLogger(CustomLogger): shadow_text: str, tools: object, parent_metadata: Mapping[str, object], + creator_metadata: Mapping[str, object], ) -> "_JudgeVerdict | _CallFailure": """Blind pairwise judge with A/B labels randomized to cancel position bias. Both arms were offered the same tools, so the judge is shown their definitions too: a @@ -1334,7 +1325,7 @@ class ShadowEvalLogger(CustomLogger): for m in messages if m.get("content") is not None ) - judge_metadata: Final = sanitized_forwardable_call_metadata(parent_metadata, SHADOW_EVAL_JUDGE_CALL_ORIGIN) + judge_metadata: Final = {**creator_metadata, INTERNAL_CALL_ORIGIN_METADATA_KEY: SHADOW_EVAL_JUDGE_CALL_ORIGIN} judge_messages: Final = [ {"role": "system", "content": PAIRWISE_JUDGE_SYSTEM_PROMPT}, { diff --git a/litellm/litellm_core_utils/internal_call_metadata.py b/litellm/litellm_core_utils/internal_call_metadata.py index d844cbae367..250513c8160 100644 --- a/litellm/litellm_core_utils/internal_call_metadata.py +++ b/litellm/litellm_core_utils/internal_call_metadata.py @@ -1,9 +1,9 @@ """Metadata a request forwards to the internal LLM sub-calls it triggers. -Internal features (the auto-router's classifier and embeddings, shadow eval's shadow and -judge calls) bill real provider spend that nobody typed a prompt for. That spend must land -on the same key/team/org/user as the request that caused it, so the sub-call carries the -caller's identity metadata, minus two things that must never be forwarded as-is: +Internal features (the auto-router's classifier, embeddings and context compaction) bill +real provider spend that nobody typed a prompt for. That spend must land on the same +key/team/org/user as the request that caused it, so the sub-call carries the caller's +identity metadata, minus two things that must never be forwarded as-is: * ``user_api_key_budget_reservation`` (and the reservation nested inside ``user_api_key_auth``) belongs to the parent completion. If a sub-call's cost callback @@ -161,8 +161,8 @@ def sanitized_forwardable_call_metadata( ) -> dict[str, object]: # mutable-ok: SDK metadata kwarg """Just the caller's identity, stamped with the sub-call's origin. - For sub-calls detached from the parent request (shadow eval), which outlive it and - must not inherit per-request state such as its routing decision or logging payload. + For sub-calls that must not inherit the parent's per-request state, such as its + routing decision or logging payload. """ identity: Final = {k: v for k, v in parent_metadata.items() if k in FORWARDABLE_IDENTITY_METADATA_KEYS} return _sanitized(identity) | {INTERNAL_CALL_ORIGIN_METADATA_KEY: call_origin} diff --git a/litellm/proxy/auth/evaluation_principal.py b/litellm/proxy/auth/evaluation_principal.py new file mode 100644 index 00000000000..6613c0f75e1 --- /dev/null +++ b/litellm/proxy/auth/evaluation_principal.py @@ -0,0 +1,88 @@ +"""The principal a shadow evaluation's own LLM calls run as: the admin who created the job. + +Evaluation calls are spend the creator chose to incur, so they are attributed, routed and +budget-checked as that admin's own requests, through the same owners an authenticated +request uses. Nothing the sampled request carried decides who pays for them. +""" + +from collections.abc import Mapping, Sequence +from types import MappingProxyType +from typing import Final + +from pydantic import TypeAdapter + +from litellm._logging import verbose_proxy_logger +from litellm.exceptions import BudgetExceededError +from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth +from litellm.proxy.auth.auth_checks import effective_user_role, get_user_object +from litellm.proxy.auth.fallback_budget import is_token_within_budget_for_model +from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup +from litellm.types.proxy.auth.auth_checks import UserNotFoundError + +_MODEL_BUDGETS: Final[TypeAdapter[Mapping[str, object] | None]] = TypeAdapter(Mapping[str, object] | None) + + +async def evaluation_principal(creator_user_id: str) -> UserAPIKeyAuth: + """The creator as a principal carrying their current budgets. + + A creator with no user row is the master key's admin, which has no personal budget to + enforce: the spend is still attributed to that id. Any other read failure propagates, so + the caller withholds work it cannot verify the creator can pay for. + """ + from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache + + try: + user: Final = await get_user_object( + user_id=creator_user_id, + prisma_client=prisma_client, + user_api_key_cache=user_api_key_cache, + user_id_upsert=False, + proxy_logging_obj=proxy_logging_obj, + ) + except UserNotFoundError: + return UserAPIKeyAuth(user_id=creator_user_id, user_role=LitellmUserRoles.PROXY_ADMIN) + if user is None: + return UserAPIKeyAuth(user_id=creator_user_id, user_role=LitellmUserRoles.PROXY_ADMIN) + return UserAPIKeyAuth( + user_id=user.user_id, + user_role=effective_user_role(user.user_role), + user_email=user.user_email, + user_spend=user.spend, + user_max_budget=user.max_budget, + user_model_max_budget=_MODEL_BUDGETS.validate_python(user.model_max_budget), + ) + + +async def principal_can_pay_for(principal: UserAPIKeyAuth, models: Sequence[str]) -> bool: + """Whether the principal is within its total budget and every per-model budget for + ``models``, read through the owners the request path enforces them with.""" + from litellm.proxy.proxy_server import llm_router, model_max_budget_limiter + + if llm_router is None: + return False + try: + for model in models: + if not await is_token_within_budget_for_model(model=model, valid_token=principal, llm_router=llm_router): + return False + if principal.user_id is not None and principal.user_model_max_budget: + await model_max_budget_limiter.is_user_within_model_budget( + user_id=principal.user_id, user_model_max_budget=principal.user_model_max_budget, model=model + ) + except BudgetExceededError as e: + verbose_proxy_logger.debug("shadow_eval: creator %s is over budget: %s", principal.user_id, e) + return False + return True + + +def principal_call_metadata(principal: UserAPIKeyAuth) -> Mapping[str, object]: + """Request metadata naming ``principal`` exactly as auth stamps it on a request the + principal sent, so every spend writer, limiter and router filter reads one identity.""" + stamped: Final[dict[str, object]] = {} # mutable-ok: the stampers write into a request dict + request: Final = {"metadata": stamped} + LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata( + data=request, user_api_key_dict=principal, _metadata_variable_name="metadata" + ) + LiteLLMProxyRequestSetup.add_budget_metadata_to_request_metadata( + data=request, user_api_key_dict=principal, _metadata_variable_name="metadata" + ) + return MappingProxyType(stamped) diff --git a/litellm/proxy/litellm_pre_call_utils.py b/litellm/proxy/litellm_pre_call_utils.py index 152438d0573..4dd0a12e262 100644 --- a/litellm/proxy/litellm_pre_call_utils.py +++ b/litellm/proxy/litellm_pre_call_utils.py @@ -1650,6 +1650,27 @@ class LiteLLMProxyRequestSetup: ) return user_api_key_logged_metadata + @staticmethod + def add_budget_metadata_to_request_metadata( + data: dict, user_api_key_dict: UserAPIKeyAuth, _metadata_variable_name: str + ) -> None: + """The principal's budgets and spend: prometheus reports them and the per-model budget + limiter reads the ``*_model_max_budget`` entries to charge spend.""" + data[_metadata_variable_name]["user_api_key_team_max_budget"] = user_api_key_dict.team_max_budget + data[_metadata_variable_name]["user_api_key_team_spend"] = user_api_key_dict.team_spend + data[_metadata_variable_name]["user_api_key_team_model_max_budget"] = user_api_key_dict.team_model_max_budget + data[_metadata_variable_name]["user_api_key_request_route"] = user_api_key_dict.request_route + data[_metadata_variable_name]["user_api_key_spend"] = user_api_key_dict.spend + data[_metadata_variable_name]["user_api_key_max_budget"] = user_api_key_dict.max_budget + data[_metadata_variable_name]["user_api_key_model_max_budget"] = user_api_key_dict.model_max_budget + data[_metadata_variable_name]["user_api_key_end_user_model_max_budget"] = ( + user_api_key_dict.end_user_model_max_budget + ) + data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend + data[_metadata_variable_name]["user_api_key_user_max_budget"] = user_api_key_dict.user_max_budget + data[_metadata_variable_name]["user_api_key_user_model_max_budget"] = user_api_key_dict.user_model_max_budget + data[_metadata_variable_name].update(carried_budget_metadata(user_api_key_dict)) + @staticmethod def add_user_api_key_auth_to_request_metadata( data: dict, @@ -2405,28 +2426,10 @@ async def add_litellm_data_to_request( _metadata_variable_name=_metadata_variable_name, ) - # Team spend, budget - used by prometheus.py - data[_metadata_variable_name]["user_api_key_team_max_budget"] = user_api_key_dict.team_max_budget - data[_metadata_variable_name]["user_api_key_team_spend"] = user_api_key_dict.team_spend - data[_metadata_variable_name]["user_api_key_team_model_max_budget"] = user_api_key_dict.team_model_max_budget - data[_metadata_variable_name]["user_api_key_request_route"] = user_api_key_dict.request_route - - # API Key spend, budget - used by prometheus.py - data[_metadata_variable_name]["user_api_key_spend"] = user_api_key_dict.spend - data[_metadata_variable_name]["user_api_key_max_budget"] = user_api_key_dict.max_budget - data[_metadata_variable_name]["user_api_key_model_max_budget"] = user_api_key_dict.model_max_budget - data[_metadata_variable_name]["user_api_key_end_user_model_max_budget"] = ( - user_api_key_dict.end_user_model_max_budget + LiteLLMProxyRequestSetup.add_budget_metadata_to_request_metadata( + data=data, user_api_key_dict=user_api_key_dict, _metadata_variable_name=_metadata_variable_name ) - # User spend, budget - used by prometheus.py - # Follow same pattern as team and API key budgets - data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend - data[_metadata_variable_name]["user_api_key_user_max_budget"] = user_api_key_dict.user_max_budget - user_model_budget: Final = user_api_key_dict.user_model_max_budget - data[_metadata_variable_name]["user_api_key_user_model_max_budget"] = user_model_budget # rebind-ok: out-param - data[_metadata_variable_name].update(carried_budget_metadata(user_api_key_dict)) - data[_metadata_variable_name]["user_api_key_metadata"] = strip_callback_config(user_api_key_dict.metadata) data[_metadata_variable_name]["user_api_key_team_metadata"] = strip_callback_config(user_api_key_dict.team_metadata) data[_metadata_variable_name]["user_api_key_object_permission_id"] = getattr( diff --git a/litellm/proxy/management_endpoints/auto_router_endpoints.py b/litellm/proxy/management_endpoints/auto_router_endpoints.py index 33fb069afbd..850ccf9d691 100644 --- a/litellm/proxy/management_endpoints/auto_router_endpoints.py +++ b/litellm/proxy/management_endpoints/auto_router_endpoints.py @@ -1684,8 +1684,9 @@ async def start_shadow_eval( eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the job's window ends, or the job is stopped, so one target running out of budget does not end sampling for the others; sampling changes propagate to pods within about 10 - seconds. Shadow and judge calls bill to the sampled request's own identity but are - excluded from request counts and auto-router adoption metrics. + seconds. Shadow and judge calls run as the admin who started the job: their spend is + that admin's, the admin's total and per-model budgets withhold samples once exhausted, + and they are excluded from request counts and auto-router adoption metrics. """ from litellm.proxy.proxy_server import llm_router, prisma_client diff --git a/tests/unit/integrations/test_shadow_eval_logger.py b/tests/unit/integrations/test_shadow_eval_logger.py index e3f059a7941..baf7ef86fcd 100644 --- a/tests/unit/integrations/test_shadow_eval_logger.py +++ b/tests/unit/integrations/test_shadow_eval_logger.py @@ -2,8 +2,9 @@ the detached pipeline's single attempt-row write, and the cache-first job lookup.""" import asyncio -from collections.abc import Mapping +from collections.abc import Mapping, Sequence from datetime import datetime, timedelta, timezone +from types import MappingProxyType from typing import Final, Literal from unittest.mock import AsyncMock, MagicMock @@ -55,6 +56,7 @@ def _job(**overrides) -> ActiveShadowEvalJob: max_turns=200, ends_at=datetime.now(timezone.utc) + timedelta(days=1), attempts=0, + created_by="eval-admin", ) return ActiveShadowEvalJob(**{**defaults, **overrides}) @@ -92,6 +94,7 @@ def _job_record(job: ActiveShadowEvalJob, target_type="key", target_id="key-hash judge_model=job.judge_model, max_turns=job.max_turns, max_budget=job.max_budget, + created_by=job.created_by, ends_at=job.ends_at, ).items(): setattr(record, field, value) @@ -230,10 +233,28 @@ def _spend_counter(store=None): return counter, read, write -def _logger(router=None, prisma=None, jobs=(), counter_store=None, jobs_by_target=None) -> ShadowEvalLogger: +CREATOR_METADATA: Final = MappingProxyType({"user_api_key_user_id": "eval-admin", "user_api_key_hash": None}) + + +def _creator_admission(admitted: bool = True): + """Stands in for the proxy's creator budget owners: answers with the creator's call + metadata, or None for a creator who cannot pay, and records what it was asked.""" + asked: Final[list[tuple[str | None, tuple[str, ...]]]] = [] + + async def admit(creator_user_id: str | None, models: Sequence[str]) -> Mapping[str, object] | None: + asked.append((creator_user_id, tuple(models))) + return CREATOR_METADATA if admitted else None + + return admit, asked + + +def _logger( + router=None, prisma=None, jobs=(), counter_store=None, jobs_by_target=None, creator_admitted=True +) -> ShadowEvalLogger: cache = InMemoryCache(max_size_in_memory=4, default_ttl=60) counter, read, write = _spend_counter(counter_store) funnel_events = [] + admit, asked = _creator_admission(creator_admitted) logger = ShadowEvalLogger( router_provider=lambda: router, prisma_provider=lambda: prisma, @@ -241,9 +262,11 @@ def _logger(router=None, prisma=None, jobs=(), counter_store=None, jobs_by_targe job_spend_reader=read, job_spend_writer=write, funnel_recorder=lambda job_id, stage: funnel_events.append((job_id, stage)), + creator_admission=admit, ) logger._test_counter = counter logger._test_funnel = funnel_events + logger._test_admission_asks = asked seeded = jobs_by_target if jobs_by_target is not None else ({("key", "key-hash"): tuple(jobs)} if jobs else None) if seeded is not None: cache.set_cache("shadow_eval:active_jobs", seeded) @@ -1225,24 +1248,35 @@ class TestSuccessHookSkipChain: assert prisma.db.litellm_shadowevalattempt.create.await_count == 1 - async def test_v1_messages_surface_forwards_identity_from_litellm_metadata(self): - """/v1/messages stores identity in litellm_params.litellm_metadata, so the hook - resolves the bucket through the shared helper; every surface forwards the same - identity to the shadow and judge calls.""" + @pytest.mark.parametrize( + "call_type,bucket", [("acompletion", "metadata"), ("anthropic_messages", "litellm_metadata")] + ) + async def test_eval_calls_run_as_the_creator_whatever_the_sampled_caller_was(self, call_type: str, bucket: str): + """Every surface's sampled request carries its caller's identity in a different bucket; + none of it reaches the shadow or judge call, which bill to the job's creator.""" prisma = _prisma() router = _router() logger = _logger(router=router, prisma=prisma, jobs=(_job(),)) - hook_kwargs = _success_kwargs() + hook_kwargs = _success_kwargs(call_type=call_type) hook_kwargs["litellm_params"] = { - "litellm_metadata": {"user_api_key_hash": "key-hash", "user_api_key_team_id": "team-1"} + bucket: { + "user_api_key_hash": "key-hash", + "user_api_key_team_id": "team-1", + "user_api_key_user_id": "sampled-user", + "user_api_key_end_user_id": "sampled-end-user", + }, + "proxy_server_request": {"body": {"messages": [{"role": "user", "content": "hi"}], "max_tokens": 9}}, } await logger.async_log_success_event(hook_kwargs, RESPONSE, None, None) await _drain(logger) - shadow_call = router.acompletion.call_args_list[0].kwargs - assert shadow_call["metadata"]["user_api_key_hash"] == "key-hash" - assert shadow_call["metadata"]["user_api_key_team_id"] == "team-1" + assert logger._test_admission_asks == [("eval-admin", ("my-router", "judge-model"))] + assert router.acompletion.call_count == 2 + for call in router.acompletion.call_args_list: + metadata = call.kwargs["metadata"] + assert metadata["user_api_key_user_id"] == "eval-admin" + assert {k: v for k, v in metadata.items() if v in ("key-hash", "team-1", "sampled-user", "sampled-end-user")} == {} async def test_redacted_requests_are_never_shadowed(self): """Redaction rewrites the logged messages before callbacks run, so this hook only @@ -1513,24 +1547,23 @@ class TestShadowPipeline: router.acompletion.assert_not_called() assert logger._test_funnel == [("job-1", "withheld")] - async def test_over_budget_key_skips_before_any_call(self, monkeypatch: pytest.MonkeyPatch): - """The gate delegates to the auth path's own budget owner, so an over-budget - verdict there (BudgetExceededError) skips the shadow before any provider call.""" - from litellm.exceptions import BudgetExceededError - from litellm.proxy._types import UserAPIKeyAuth - from litellm.proxy.auth import auth_checks - - monkeypatch.setattr( - auth_checks, - "_virtual_key_max_budget_check", - AsyncMock(side_effect=BudgetExceededError(current_cost=11.0, max_budget=10.0)), - ) + @pytest.mark.parametrize( + "job", + [ + _job(), + _job(router_names=("my-router", "router-b")), + _job(direction="reverse", baseline_model="baseline-model"), + ], + ) + async def test_a_creator_who_cannot_pay_withholds_the_sample_before_any_call(self, job: ActiveShadowEvalJob): + """Admission is asked once per sample about every model the sample would call, each + arm's target plus the judge, and a creator over any of those budgets spends nothing.""" router = _router() prisma = _prisma() - logger = _logger(router=router, prisma=prisma) + logger = _logger(router=router, prisma=prisma, creator_admitted=False) await logger._run_shadow_eval( - job=_job(), + job=job, request_id="req-1", messages=({"role": "user", "content": "hi"},), real_text="real answer", @@ -1540,12 +1573,15 @@ class TestShadowPipeline: real_cache_hit=False, control_tier=None, shadow_params={}, - parent_metadata={"user_api_key_auth": UserAPIKeyAuth(api_key="sk-abc", max_budget=10.0)}, + parent_metadata={"user_api_key_hash": "key-hash"}, ) + assert logger._test_admission_asks == [ + ("eval-admin", (*(job.arm_target(arm) for arm in job.arm_router_names), "judge-model")) + ] router.acompletion.assert_not_called() prisma.db.litellm_shadowevalattempt.create.assert_not_called() - assert logger._test_funnel == [("job-1", "withheld")] + assert logger._test_funnel == [(job.id, "withheld")] @pytest.mark.parametrize( "router_factory,expected_error,expected_cost,expected_shadow_cost", @@ -1965,7 +2001,7 @@ class TestShadowPipeline: assert row["judge_cost"] == 0.0 assert logger._test_counter["spend:shadow_eval:job-1"] == 0.007 - async def test_sub_calls_carry_identity_and_origin_but_never_parent_request_state(self): + async def test_sub_calls_carry_the_creator_and_origin_but_never_parent_request_state(self): prisma = _prisma() router = _router() logger = _logger(router=router, prisma=prisma) @@ -1995,8 +2031,9 @@ class TestShadowPipeline: for call in (shadow_call, judge_call): assert call["num_retries"] == 0 assert call["fallbacks"] == [] - assert call["metadata"]["user_api_key_hash"] == "key-hash" - assert call["metadata"]["user_api_key_team_id"] == "team-1" + assert call["metadata"]["user_api_key_user_id"] == "eval-admin" + assert call["metadata"]["user_api_key_hash"] is None + assert "user_api_key_team_id" not in call["metadata"] assert "user_api_key_budget_reservation" not in call["metadata"] assert shadow_call["metadata"][INTERNAL_CALL_ORIGIN_METADATA_KEY] == SHADOW_EVAL_ROUTER_CALL_ORIGIN assert judge_call["metadata"][INTERNAL_CALL_ORIGIN_METADATA_KEY] == SHADOW_EVAL_JUDGE_CALL_ORIGIN diff --git a/tests/unit/proxy/auth/test_evaluation_principal.py b/tests/unit/proxy/auth/test_evaluation_principal.py new file mode 100644 index 00000000000..1e93611d467 --- /dev/null +++ b/tests/unit/proxy/auth/test_evaluation_principal.py @@ -0,0 +1,127 @@ +"""The shadow-eval creator principal: who it is, what its stamped calls charge, and +whether the request path's own budget owners let it pay.""" + +from typing import Final +from unittest.mock import AsyncMock + +import pytest + +from litellm import Router +from litellm.caching.dual_cache import DualCache +from litellm.proxy import proxy_server +from litellm.proxy._types import LitellmUserRoles +from litellm.proxy.auth import evaluation_principal as module +from litellm.proxy.auth.evaluation_principal import ( + evaluation_principal, + principal_call_metadata, + principal_can_pay_for, +) +from litellm.proxy.hooks.model_max_budget_limiter import _PROXY_VirtualKeyModelMaxBudgetLimiter +from litellm.models.user import LiteLLM_UserTable +from litellm.types.proxy.auth.auth_checks import UserNotFoundError + +PAID_MODEL: Final = { + "model_name": "eval-judge", + "litellm_params": { + "model": "openai/gpt-4o", + "api_key": "k", + "input_cost_per_token": 0.000001, + "output_cost_per_token": 0.000002, + }, + "model_info": {"id": "eval-judge-id"}, +} + + +def _creator(**overrides) -> LiteLLM_UserTable: + fields = {"user_id": "eval-admin", "user_role": "proxy_admin", "spend": 0.0, "max_budget": 10.0} + return LiteLLM_UserTable(**{**fields, **overrides}) + + +@pytest.fixture +def proxy(monkeypatch: pytest.MonkeyPatch) -> _PROXY_VirtualKeyModelMaxBudgetLimiter: + limiter: Final = _PROXY_VirtualKeyModelMaxBudgetLimiter(dual_cache=DualCache()) + + async def counter_spend(counter_key: str, fallback_spend: float, max_budget: float | None = None) -> float: + return fallback_spend + + monkeypatch.setattr(proxy_server, "llm_router", Router(model_list=[PAID_MODEL]), raising=False) + monkeypatch.setattr(proxy_server, "model_max_budget_limiter", limiter, raising=False) + monkeypatch.setattr(proxy_server, "get_current_spend", counter_spend, raising=False) + monkeypatch.setattr(proxy_server, "general_settings", {}, raising=False) + monkeypatch.setattr(proxy_server, "prisma_client", object(), raising=False) + return limiter + + +@pytest.mark.asyncio +async def test_spend_charged_under_the_stamp_exhausts_the_budget_the_gate_reads(proxy, monkeypatch): + """The workflow end to end: an evaluation call stamped as the creator is charged by the + real per-model limiter, and the next admission reads that same counter and refuses.""" + monkeypatch.setattr( + module, + "get_user_object", + AsyncMock( + return_value=_creator(model_max_budget={"eval-judge": {"budget_limit": 0.001, "time_period": "1d"}}) + ), + ) + principal = await evaluation_principal("eval-admin") + assert await principal_can_pay_for(principal, ("eval-judge",)) is True + + stamped = principal_call_metadata(principal) + await proxy.async_log_success_event( + { + "standard_logging_object": { + "call_type": "acompletion", + "response_cost": 0.002, + "model": "openai/gpt-4o", + "model_group": "eval-judge", + "metadata": {k: stamped.get(k) for k in ("user_api_key_hash", "user_api_key_user_id")}, + }, + "litellm_params": {"metadata": dict(stamped)}, + }, + response_obj=None, + start_time=None, + end_time=None, + ) + + assert stamped["user_api_key_user_id"] == "eval-admin" + assert await principal_can_pay_for(principal, ("eval-judge",)) is False + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "creator,models,expected", + [ + (_creator(spend=10.0), ("eval-judge",), False), + (_creator(spend=10.0), (), True), + (_creator(spend=9.0), ("eval-judge",), True), + (_creator(spend=999.0, max_budget=None), ("eval-judge",), True), + ], +) +async def test_total_budget_gates_every_paid_model( + proxy: _PROXY_VirtualKeyModelMaxBudgetLimiter, + monkeypatch: pytest.MonkeyPatch, + creator: LiteLLM_UserTable, + models: tuple[str, ...], + expected: bool, +): + monkeypatch.setattr(module, "get_user_object", AsyncMock(return_value=creator)) + assert await principal_can_pay_for(await evaluation_principal("eval-admin"), models) is expected + + +@pytest.mark.asyncio +async def test_a_creator_without_a_user_row_is_the_unbudgeted_master_key_admin(proxy, monkeypatch): + monkeypatch.setattr(module, "get_user_object", AsyncMock(side_effect=UserNotFoundError(user_id="default_user_id"))) + + principal = await evaluation_principal("default_user_id") + + assert (principal.user_id, principal.user_role) == ("default_user_id", LitellmUserRoles.PROXY_ADMIN) + assert await principal_can_pay_for(principal, ("eval-judge",)) is True + assert principal_call_metadata(principal)["user_api_key_user_id"] == "default_user_id" + + +@pytest.mark.asyncio +async def test_an_unreadable_creator_row_propagates_so_the_caller_withholds(proxy, monkeypatch): + monkeypatch.setattr(module, "get_user_object", AsyncMock(side_effect=RuntimeError("db down"))) + + with pytest.raises(RuntimeError, match="db down"): + await evaluation_principal("eval-admin") diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalStartForm.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalStartForm.tsx index 06e332d2cfc..4bee9aab52f 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalStartForm.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalStartForm.tsx @@ -38,9 +38,9 @@ const DIRECTION_OPTIONS: readonly { value: ShadowEvalDirection; label: string }[ const START_FORM_DESCRIPTION: Record = { forward: - "Duplicates a sampled slice of the selected targets' traffic (keys, teams, or users) through the auto-router and has an LLM judge compare both answers blind. Each target gets its own spend budget. The router's answers are never served to users; judge calls bill to the sampled traffic's own identity.", + "Duplicates a sampled slice of the selected targets' traffic (keys, teams, or users) through the auto-router and has an LLM judge compare both answers blind. Each target gets its own spend budget. The router's answers are never served to users; shadow and judge calls bill to you and count against your budgets.", reverse: - "Duplicates a sampled slice of the traffic the auto-router already serves against a fixed baseline model and has an LLM judge compare both answers blind. Each target gets its own spend budget. The baseline's answers are never served to users; judge calls bill to the sampled traffic's own identity.", + "Duplicates a sampled slice of the traffic the auto-router already serves against a fixed baseline model and has an LLM judge compare both answers blind. Each target gets its own spend budget. The baseline's answers are never served to users; shadow and judge calls bill to you and count against your budgets.", }; const DURATION_OPTIONS = [ diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index ca6d3bdf269..6d9ac9877f0 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -1475,8 +1475,9 @@ export interface paths { * eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the * job's window ends, or the job is stopped, so one target running out of budget does * not end sampling for the others; sampling changes propagate to pods within about 10 - * seconds. Shadow and judge calls bill to the sampled request's own identity but are - * excluded from request counts and auto-router adoption metrics. + * seconds. Shadow and judge calls run as the admin who started the job: their spend is + * that admin's, the admin's total and per-model budgets withhold samples once exhausted, + * and they are excluded from request counts and auto-router adoption metrics. */ post: operations["start_shadow_eval_auto_router_shadow_eval_start_post"]; delete?: never;