mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
Merge c05014629c into 6532dcb73b
This commit is contained in:
commit
a36693f63d
9 changed files with 363 additions and 115 deletions
|
|
@ -30,7 +30,6 @@ from litellm.constants import INTERNAL_CALL_ORIGIN_METADATA_KEY
|
|||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.integrations.websearch_interception.tools import is_web_search_tool_responses
|
||||
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs, independent_snapshot
|
||||
from litellm.litellm_core_utils.internal_call_metadata import sanitized_forwardable_call_metadata
|
||||
from litellm.litellm_core_utils.llm_judge import (
|
||||
default_router_provider,
|
||||
extract_text_from_content,
|
||||
|
|
@ -620,50 +619,30 @@ def _record_funnel_event(job_id: str, stage: "ShadowEvalFunnelStage") -> None:
|
|||
verbose_logger.debug("shadow_eval: funnel increment failed for %s: %s", job_id, e)
|
||||
|
||||
|
||||
async def _key_or_team_is_over_budget(metadata: Mapping[str, object]) -> bool:
|
||||
"""Whether the shadowed key or its team is over budget, decided by the same owners
|
||||
the request path uses, so counter keys and thresholds can never drift from auth's.
|
||||
|
||||
Advisory and fail-open: real traffic on an over-budget key is already rejected at
|
||||
auth (so nothing reaches the success hook), and this gate only closes the race
|
||||
where the key crosses its budget while a request is in flight.
|
||||
"""
|
||||
async def _admit_creator(creator_user_id: str | None, models: Sequence[str]) -> Mapping[str, object] | None:
|
||||
"""The call metadata the job's creator pays under, or None when the creator cannot pay
|
||||
for ``models`` or that cannot be verified, so the sample is withheld before any spend."""
|
||||
if creator_user_id is None:
|
||||
return None
|
||||
try:
|
||||
from litellm.exceptions import BudgetExceededError
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth.auth_checks import (
|
||||
_team_max_budget_check,
|
||||
_virtual_key_max_budget_check,
|
||||
get_team_object,
|
||||
from litellm.proxy.auth.evaluation_principal import (
|
||||
evaluation_principal,
|
||||
principal_call_metadata,
|
||||
principal_can_pay_for,
|
||||
)
|
||||
from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache
|
||||
except ImportError:
|
||||
return False
|
||||
|
||||
auth: Final = metadata.get("user_api_key_auth")
|
||||
if not isinstance(auth, UserAPIKeyAuth):
|
||||
return False
|
||||
try:
|
||||
await _virtual_key_max_budget_check(valid_token=auth, proxy_logging_obj=proxy_logging_obj)
|
||||
if auth.team_id:
|
||||
team: Final = await get_team_object(
|
||||
team_id=auth.team_id,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
check_cache_only=True,
|
||||
)
|
||||
await _team_max_budget_check(team_object=team, valid_token=auth, proxy_logging_obj=proxy_logging_obj)
|
||||
except BudgetExceededError:
|
||||
return True
|
||||
except Exception as e: # noqa: BLE001 # advisory gate: a failed read must not block sampling
|
||||
verbose_logger.debug("shadow_eval: budget read failed: %s", e)
|
||||
return False
|
||||
principal: Final = await evaluation_principal(creator_user_id)
|
||||
if not await principal_can_pay_for(principal, models):
|
||||
return None
|
||||
return principal_call_metadata(principal)
|
||||
except Exception as e: # noqa: BLE001 # an unverifiable creator budget withholds the sample rather than spending
|
||||
verbose_logger.warning("shadow_eval: creator %s budget unverifiable, sample withheld: %s", creator_user_id, e)
|
||||
return None
|
||||
|
||||
|
||||
def _forwarded_team_id(metadata: Mapping[str, object]) -> str | None:
|
||||
"""The shadowed key's team, the identity the judge call already carries in its metadata
|
||||
and the router already selects deployments with. Read here too so the arm choice, which
|
||||
happens before the router sees the call, is made under the same team."""
|
||||
"""The shadowed key's team, the scope start-time validation resolved the judge under,
|
||||
so the judge's router-or-SDK arm is chosen the way the job was validated."""
|
||||
team_id: Final = metadata.get("user_api_key_team_id")
|
||||
return team_id if isinstance(team_id, str) and team_id else None
|
||||
|
||||
|
|
@ -749,6 +728,7 @@ class ActiveShadowEvalJob(BaseModel):
|
|||
judge_model: str
|
||||
max_turns: int
|
||||
max_budget: float | None = None
|
||||
created_by: str | None = None
|
||||
ends_at: datetime
|
||||
attempts: int = 0
|
||||
spend: float = 0.0
|
||||
|
|
@ -833,6 +813,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
job_spend_reader: Callable[[str, float, float], Awaitable[float]] | None = None,
|
||||
job_spend_writer: Callable[[str, float], Awaitable[None]] | None = None,
|
||||
funnel_recorder: Callable[[str, "ShadowEvalFunnelStage"], None] | None = None,
|
||||
creator_admission: Callable[[str | None, Sequence[str]], Awaitable[Mapping[str, object] | None]] | None = None,
|
||||
) -> None:
|
||||
"""Providers are callables so the proxy's lazily-initialized globals are resolved
|
||||
at call time, not at logger construction. The spend reader and writer wrap the
|
||||
|
|
@ -843,6 +824,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
self._read_job_spend = job_spend_reader or _job_spend_from_counter
|
||||
self._write_job_spend = job_spend_writer or _add_job_spend_to_counter
|
||||
self._record_funnel = funnel_recorder or _record_funnel_event
|
||||
self._admit_creator = creator_admission or _admit_creator
|
||||
self._inflight_shadow_tasks: int = 0
|
||||
# Starts per job since the last cache fill, never decremented within a
|
||||
# generation; the refill absorbs written rows and resets.
|
||||
|
|
@ -1055,8 +1037,9 @@ class ShadowEvalLogger(CustomLogger):
|
|||
) -> None:
|
||||
"""Budget gates once per sampled request, then every router arm in turn: shadow
|
||||
call -> blind judge -> one attempt row stamped with the arm. The gates that
|
||||
decline to spend on an admitted sample (no DB to record into, an over-budget key,
|
||||
an unverifiable or exhausted eval budget) count the REQUEST withheld before any
|
||||
decline to spend on an admitted sample (no DB to record into, an unverifiable or
|
||||
exhausted eval budget, a creator who cannot pay for every model the sample would
|
||||
call) count the REQUEST withheld before any
|
||||
arm runs, so funnel counters stay per-request and a leg's eligible traffic still
|
||||
reconciles as not_sampled + unjudgeable + shed + withheld + sampled requests,
|
||||
where each sampled request writes one attempt row per arm. A budget crossed
|
||||
|
|
@ -1069,9 +1052,6 @@ class ShadowEvalLogger(CustomLogger):
|
|||
if prisma is None:
|
||||
self._record_funnel(job.id, "withheld")
|
||||
return
|
||||
if await _key_or_team_is_over_budget(parent_metadata):
|
||||
self._record_funnel(job.id, "withheld")
|
||||
return
|
||||
if job.max_budget is not None:
|
||||
try:
|
||||
spend: Final = await self._read_job_spend(_job_spend_counter_key(job.id), job.spend, job.max_budget)
|
||||
|
|
@ -1082,6 +1062,12 @@ class ShadowEvalLogger(CustomLogger):
|
|||
if spend >= job.max_budget:
|
||||
self._record_funnel(job.id, "withheld")
|
||||
return
|
||||
creator_metadata: Final = await self._admit_creator(
|
||||
job.created_by, (*(job.arm_target(arm) for arm in job.arm_router_names), job.judge_model)
|
||||
)
|
||||
if creator_metadata is None:
|
||||
self._record_funnel(job.id, "withheld")
|
||||
return
|
||||
for arm_router in job.arm_router_names:
|
||||
await self._run_shadow_arm(
|
||||
prisma=prisma,
|
||||
|
|
@ -1097,6 +1083,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
control_tier=control_tier,
|
||||
shadow_params=shadow_params,
|
||||
parent_metadata=parent_metadata,
|
||||
creator_metadata=creator_metadata,
|
||||
)
|
||||
|
||||
async def _run_shadow_arm(
|
||||
|
|
@ -1114,12 +1101,13 @@ class ShadowEvalLogger(CustomLogger):
|
|||
control_tier: str | None,
|
||||
shadow_params: Mapping[str, object],
|
||||
parent_metadata: Mapping[str, object],
|
||||
creator_metadata: Mapping[str, object],
|
||||
) -> None:
|
||||
"""One arm's pipeline: shadow call -> blind judge -> one attempt row, every exit
|
||||
recording this arm's outcome, so one arm's fault never silences a sibling arm."""
|
||||
try:
|
||||
shadow: Final = await self._call_router_shadow(
|
||||
job.arm_target(arm_router), messages, shadow_params, parent_metadata
|
||||
job.arm_target(arm_router), messages, shadow_params, creator_metadata
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 # detached task: nothing billed yet, record and never raise
|
||||
verbose_logger.debug("shadow_eval: pipeline failed for %s: %s", request_id, e)
|
||||
|
|
@ -1161,6 +1149,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
shadow_text=shadow.text,
|
||||
tools=shadow_params.get("tools"),
|
||||
parent_metadata=parent_metadata,
|
||||
creator_metadata=creator_metadata,
|
||||
)
|
||||
if isinstance(verdict, _CallFailure):
|
||||
await self._record_attempt(
|
||||
|
|
@ -1268,18 +1257,19 @@ class ShadowEvalLogger(CustomLogger):
|
|||
target_model: str,
|
||||
messages: Sequence[Mapping[str, object]],
|
||||
shadow_params: Mapping[str, object],
|
||||
parent_metadata: Mapping[str, object],
|
||||
creator_metadata: Mapping[str, object],
|
||||
) -> "_ShadowResponse | _CallFailure":
|
||||
"""Send the prompt through the arm nobody was served: the auto-router under
|
||||
evaluation, or a reverse job's fixed baseline model. The metadata carries the
|
||||
shadowed key's identity (spend attribution) and receives a routing decision
|
||||
write-back, which a plain baseline model simply never makes."""
|
||||
evaluation, or a reverse job's fixed baseline model. The call runs as the job's
|
||||
creator and receives a routing decision write-back, which a plain baseline model
|
||||
simply never makes."""
|
||||
router: Final = self._router_provider()
|
||||
if router is None:
|
||||
return _CallFailure("no router configured on this pod")
|
||||
shadow_metadata: Final[dict[str, object]] = ( # mutable-ok: router writes its routing decision back
|
||||
sanitized_forwardable_call_metadata(parent_metadata, SHADOW_EVAL_ROUTER_CALL_ORIGIN)
|
||||
)
|
||||
shadow_metadata: Final[dict[str, object]] = { # mutable-ok: router writes its routing decision back
|
||||
**creator_metadata,
|
||||
INTERNAL_CALL_ORIGIN_METADATA_KEY: SHADOW_EVAL_ROUTER_CALL_ORIGIN,
|
||||
}
|
||||
try:
|
||||
response: Final = await router.acompletion(
|
||||
model=target_model,
|
||||
|
|
@ -1321,6 +1311,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
shadow_text: str,
|
||||
tools: object,
|
||||
parent_metadata: Mapping[str, object],
|
||||
creator_metadata: Mapping[str, object],
|
||||
) -> "_JudgeVerdict | _CallFailure":
|
||||
"""Blind pairwise judge with A/B labels randomized to cancel position bias. Both
|
||||
arms were offered the same tools, so the judge is shown their definitions too: a
|
||||
|
|
@ -1334,7 +1325,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
for m in messages
|
||||
if m.get("content") is not None
|
||||
)
|
||||
judge_metadata: Final = sanitized_forwardable_call_metadata(parent_metadata, SHADOW_EVAL_JUDGE_CALL_ORIGIN)
|
||||
judge_metadata: Final = {**creator_metadata, INTERNAL_CALL_ORIGIN_METADATA_KEY: SHADOW_EVAL_JUDGE_CALL_ORIGIN}
|
||||
judge_messages: Final = [
|
||||
{"role": "system", "content": PAIRWISE_JUDGE_SYSTEM_PROMPT},
|
||||
{
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
"""Metadata a request forwards to the internal LLM sub-calls it triggers.
|
||||
|
||||
Internal features (the auto-router's classifier and embeddings, shadow eval's shadow and
|
||||
judge calls) bill real provider spend that nobody typed a prompt for. That spend must land
|
||||
on the same key/team/org/user as the request that caused it, so the sub-call carries the
|
||||
caller's identity metadata, minus two things that must never be forwarded as-is:
|
||||
Internal features (the auto-router's classifier, embeddings and context compaction) bill
|
||||
real provider spend that nobody typed a prompt for. That spend must land on the same
|
||||
key/team/org/user as the request that caused it, so the sub-call carries the caller's
|
||||
identity metadata, minus two things that must never be forwarded as-is:
|
||||
|
||||
* ``user_api_key_budget_reservation`` (and the reservation nested inside
|
||||
``user_api_key_auth``) belongs to the parent completion. If a sub-call's cost callback
|
||||
|
|
@ -161,8 +161,8 @@ def sanitized_forwardable_call_metadata(
|
|||
) -> dict[str, object]: # mutable-ok: SDK metadata kwarg
|
||||
"""Just the caller's identity, stamped with the sub-call's origin.
|
||||
|
||||
For sub-calls detached from the parent request (shadow eval), which outlive it and
|
||||
must not inherit per-request state such as its routing decision or logging payload.
|
||||
For sub-calls that must not inherit the parent's per-request state, such as its
|
||||
routing decision or logging payload.
|
||||
"""
|
||||
identity: Final = {k: v for k, v in parent_metadata.items() if k in FORWARDABLE_IDENTITY_METADATA_KEYS}
|
||||
return _sanitized(identity) | {INTERNAL_CALL_ORIGIN_METADATA_KEY: call_origin}
|
||||
|
|
|
|||
88
litellm/proxy/auth/evaluation_principal.py
Normal file
88
litellm/proxy/auth/evaluation_principal.py
Normal file
|
|
@ -0,0 +1,88 @@
|
|||
"""The principal a shadow evaluation's own LLM calls run as: the admin who created the job.
|
||||
|
||||
Evaluation calls are spend the creator chose to incur, so they are attributed, routed and
|
||||
budget-checked as that admin's own requests, through the same owners an authenticated
|
||||
request uses. Nothing the sampled request carried decides who pays for them.
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
from types import MappingProxyType
|
||||
from typing import Final
|
||||
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.exceptions import BudgetExceededError
|
||||
from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
|
||||
from litellm.proxy.auth.auth_checks import effective_user_role, get_user_object
|
||||
from litellm.proxy.auth.fallback_budget import is_token_within_budget_for_model
|
||||
from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup
|
||||
from litellm.types.proxy.auth.auth_checks import UserNotFoundError
|
||||
|
||||
_MODEL_BUDGETS: Final[TypeAdapter[Mapping[str, object] | None]] = TypeAdapter(Mapping[str, object] | None)
|
||||
|
||||
|
||||
async def evaluation_principal(creator_user_id: str) -> UserAPIKeyAuth:
|
||||
"""The creator as a principal carrying their current budgets.
|
||||
|
||||
A creator with no user row is the master key's admin, which has no personal budget to
|
||||
enforce: the spend is still attributed to that id. Any other read failure propagates, so
|
||||
the caller withholds work it cannot verify the creator can pay for.
|
||||
"""
|
||||
from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache
|
||||
|
||||
try:
|
||||
user: Final = await get_user_object(
|
||||
user_id=creator_user_id,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
user_id_upsert=False,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
)
|
||||
except UserNotFoundError:
|
||||
return UserAPIKeyAuth(user_id=creator_user_id, user_role=LitellmUserRoles.PROXY_ADMIN)
|
||||
if user is None:
|
||||
return UserAPIKeyAuth(user_id=creator_user_id, user_role=LitellmUserRoles.PROXY_ADMIN)
|
||||
return UserAPIKeyAuth(
|
||||
user_id=user.user_id,
|
||||
user_role=effective_user_role(user.user_role),
|
||||
user_email=user.user_email,
|
||||
user_spend=user.spend,
|
||||
user_max_budget=user.max_budget,
|
||||
user_model_max_budget=_MODEL_BUDGETS.validate_python(user.model_max_budget),
|
||||
)
|
||||
|
||||
|
||||
async def principal_can_pay_for(principal: UserAPIKeyAuth, models: Sequence[str]) -> bool:
|
||||
"""Whether the principal is within its total budget and every per-model budget for
|
||||
``models``, read through the owners the request path enforces them with."""
|
||||
from litellm.proxy.proxy_server import llm_router, model_max_budget_limiter
|
||||
|
||||
if llm_router is None:
|
||||
return False
|
||||
try:
|
||||
for model in models:
|
||||
if not await is_token_within_budget_for_model(model=model, valid_token=principal, llm_router=llm_router):
|
||||
return False
|
||||
if principal.user_id is not None and principal.user_model_max_budget:
|
||||
await model_max_budget_limiter.is_user_within_model_budget(
|
||||
user_id=principal.user_id, user_model_max_budget=principal.user_model_max_budget, model=model
|
||||
)
|
||||
except BudgetExceededError as e:
|
||||
verbose_proxy_logger.debug("shadow_eval: creator %s is over budget: %s", principal.user_id, e)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def principal_call_metadata(principal: UserAPIKeyAuth) -> Mapping[str, object]:
|
||||
"""Request metadata naming ``principal`` exactly as auth stamps it on a request the
|
||||
principal sent, so every spend writer, limiter and router filter reads one identity."""
|
||||
stamped: Final[dict[str, object]] = {} # mutable-ok: the stampers write into a request dict
|
||||
request: Final = {"metadata": stamped}
|
||||
LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata(
|
||||
data=request, user_api_key_dict=principal, _metadata_variable_name="metadata"
|
||||
)
|
||||
LiteLLMProxyRequestSetup.add_budget_metadata_to_request_metadata(
|
||||
data=request, user_api_key_dict=principal, _metadata_variable_name="metadata"
|
||||
)
|
||||
return MappingProxyType(stamped)
|
||||
|
|
@ -1650,6 +1650,27 @@ class LiteLLMProxyRequestSetup:
|
|||
)
|
||||
return user_api_key_logged_metadata
|
||||
|
||||
@staticmethod
|
||||
def add_budget_metadata_to_request_metadata(
|
||||
data: dict, user_api_key_dict: UserAPIKeyAuth, _metadata_variable_name: str
|
||||
) -> None:
|
||||
"""The principal's budgets and spend: prometheus reports them and the per-model budget
|
||||
limiter reads the ``*_model_max_budget`` entries to charge spend."""
|
||||
data[_metadata_variable_name]["user_api_key_team_max_budget"] = user_api_key_dict.team_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_team_spend"] = user_api_key_dict.team_spend
|
||||
data[_metadata_variable_name]["user_api_key_team_model_max_budget"] = user_api_key_dict.team_model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_request_route"] = user_api_key_dict.request_route
|
||||
data[_metadata_variable_name]["user_api_key_spend"] = user_api_key_dict.spend
|
||||
data[_metadata_variable_name]["user_api_key_max_budget"] = user_api_key_dict.max_budget
|
||||
data[_metadata_variable_name]["user_api_key_model_max_budget"] = user_api_key_dict.model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_end_user_model_max_budget"] = (
|
||||
user_api_key_dict.end_user_model_max_budget
|
||||
)
|
||||
data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend
|
||||
data[_metadata_variable_name]["user_api_key_user_max_budget"] = user_api_key_dict.user_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_user_model_max_budget"] = user_api_key_dict.user_model_max_budget
|
||||
data[_metadata_variable_name].update(carried_budget_metadata(user_api_key_dict))
|
||||
|
||||
@staticmethod
|
||||
def add_user_api_key_auth_to_request_metadata(
|
||||
data: dict,
|
||||
|
|
@ -2405,28 +2426,10 @@ async def add_litellm_data_to_request(
|
|||
_metadata_variable_name=_metadata_variable_name,
|
||||
)
|
||||
|
||||
# Team spend, budget - used by prometheus.py
|
||||
data[_metadata_variable_name]["user_api_key_team_max_budget"] = user_api_key_dict.team_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_team_spend"] = user_api_key_dict.team_spend
|
||||
data[_metadata_variable_name]["user_api_key_team_model_max_budget"] = user_api_key_dict.team_model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_request_route"] = user_api_key_dict.request_route
|
||||
|
||||
# API Key spend, budget - used by prometheus.py
|
||||
data[_metadata_variable_name]["user_api_key_spend"] = user_api_key_dict.spend
|
||||
data[_metadata_variable_name]["user_api_key_max_budget"] = user_api_key_dict.max_budget
|
||||
data[_metadata_variable_name]["user_api_key_model_max_budget"] = user_api_key_dict.model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_end_user_model_max_budget"] = (
|
||||
user_api_key_dict.end_user_model_max_budget
|
||||
LiteLLMProxyRequestSetup.add_budget_metadata_to_request_metadata(
|
||||
data=data, user_api_key_dict=user_api_key_dict, _metadata_variable_name=_metadata_variable_name
|
||||
)
|
||||
|
||||
# User spend, budget - used by prometheus.py
|
||||
# Follow same pattern as team and API key budgets
|
||||
data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend
|
||||
data[_metadata_variable_name]["user_api_key_user_max_budget"] = user_api_key_dict.user_max_budget
|
||||
user_model_budget: Final = user_api_key_dict.user_model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_user_model_max_budget"] = user_model_budget # rebind-ok: out-param
|
||||
data[_metadata_variable_name].update(carried_budget_metadata(user_api_key_dict))
|
||||
|
||||
data[_metadata_variable_name]["user_api_key_metadata"] = strip_callback_config(user_api_key_dict.metadata)
|
||||
data[_metadata_variable_name]["user_api_key_team_metadata"] = strip_callback_config(user_api_key_dict.team_metadata)
|
||||
data[_metadata_variable_name]["user_api_key_object_permission_id"] = getattr(
|
||||
|
|
|
|||
|
|
@ -1684,8 +1684,9 @@ async def start_shadow_eval(
|
|||
eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the
|
||||
job's window ends, or the job is stopped, so one target running out of budget does
|
||||
not end sampling for the others; sampling changes propagate to pods within about 10
|
||||
seconds. Shadow and judge calls bill to the sampled request's own identity but are
|
||||
excluded from request counts and auto-router adoption metrics.
|
||||
seconds. Shadow and judge calls run as the admin who started the job: their spend is
|
||||
that admin's, the admin's total and per-model budgets withhold samples once exhausted,
|
||||
and they are excluded from request counts and auto-router adoption metrics.
|
||||
"""
|
||||
from litellm.proxy.proxy_server import llm_router, prisma_client
|
||||
|
||||
|
|
|
|||
|
|
@ -2,8 +2,9 @@
|
|||
the detached pipeline's single attempt-row write, and the cache-first job lookup."""
|
||||
|
||||
import asyncio
|
||||
from collections.abc import Mapping
|
||||
from collections.abc import Mapping, Sequence
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from types import MappingProxyType
|
||||
from typing import Final, Literal
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
|
|
@ -55,6 +56,7 @@ def _job(**overrides) -> ActiveShadowEvalJob:
|
|||
max_turns=200,
|
||||
ends_at=datetime.now(timezone.utc) + timedelta(days=1),
|
||||
attempts=0,
|
||||
created_by="eval-admin",
|
||||
)
|
||||
return ActiveShadowEvalJob(**{**defaults, **overrides})
|
||||
|
||||
|
|
@ -92,6 +94,7 @@ def _job_record(job: ActiveShadowEvalJob, target_type="key", target_id="key-hash
|
|||
judge_model=job.judge_model,
|
||||
max_turns=job.max_turns,
|
||||
max_budget=job.max_budget,
|
||||
created_by=job.created_by,
|
||||
ends_at=job.ends_at,
|
||||
).items():
|
||||
setattr(record, field, value)
|
||||
|
|
@ -230,10 +233,28 @@ def _spend_counter(store=None):
|
|||
return counter, read, write
|
||||
|
||||
|
||||
def _logger(router=None, prisma=None, jobs=(), counter_store=None, jobs_by_target=None) -> ShadowEvalLogger:
|
||||
CREATOR_METADATA: Final = MappingProxyType({"user_api_key_user_id": "eval-admin", "user_api_key_hash": None})
|
||||
|
||||
|
||||
def _creator_admission(admitted: bool = True):
|
||||
"""Stands in for the proxy's creator budget owners: answers with the creator's call
|
||||
metadata, or None for a creator who cannot pay, and records what it was asked."""
|
||||
asked: Final[list[tuple[str | None, tuple[str, ...]]]] = []
|
||||
|
||||
async def admit(creator_user_id: str | None, models: Sequence[str]) -> Mapping[str, object] | None:
|
||||
asked.append((creator_user_id, tuple(models)))
|
||||
return CREATOR_METADATA if admitted else None
|
||||
|
||||
return admit, asked
|
||||
|
||||
|
||||
def _logger(
|
||||
router=None, prisma=None, jobs=(), counter_store=None, jobs_by_target=None, creator_admitted=True
|
||||
) -> ShadowEvalLogger:
|
||||
cache = InMemoryCache(max_size_in_memory=4, default_ttl=60)
|
||||
counter, read, write = _spend_counter(counter_store)
|
||||
funnel_events = []
|
||||
admit, asked = _creator_admission(creator_admitted)
|
||||
logger = ShadowEvalLogger(
|
||||
router_provider=lambda: router,
|
||||
prisma_provider=lambda: prisma,
|
||||
|
|
@ -241,9 +262,11 @@ def _logger(router=None, prisma=None, jobs=(), counter_store=None, jobs_by_targe
|
|||
job_spend_reader=read,
|
||||
job_spend_writer=write,
|
||||
funnel_recorder=lambda job_id, stage: funnel_events.append((job_id, stage)),
|
||||
creator_admission=admit,
|
||||
)
|
||||
logger._test_counter = counter
|
||||
logger._test_funnel = funnel_events
|
||||
logger._test_admission_asks = asked
|
||||
seeded = jobs_by_target if jobs_by_target is not None else ({("key", "key-hash"): tuple(jobs)} if jobs else None)
|
||||
if seeded is not None:
|
||||
cache.set_cache("shadow_eval:active_jobs", seeded)
|
||||
|
|
@ -1225,24 +1248,35 @@ class TestSuccessHookSkipChain:
|
|||
|
||||
assert prisma.db.litellm_shadowevalattempt.create.await_count == 1
|
||||
|
||||
async def test_v1_messages_surface_forwards_identity_from_litellm_metadata(self):
|
||||
"""/v1/messages stores identity in litellm_params.litellm_metadata, so the hook
|
||||
resolves the bucket through the shared helper; every surface forwards the same
|
||||
identity to the shadow and judge calls."""
|
||||
@pytest.mark.parametrize(
|
||||
"call_type,bucket", [("acompletion", "metadata"), ("anthropic_messages", "litellm_metadata")]
|
||||
)
|
||||
async def test_eval_calls_run_as_the_creator_whatever_the_sampled_caller_was(self, call_type: str, bucket: str):
|
||||
"""Every surface's sampled request carries its caller's identity in a different bucket;
|
||||
none of it reaches the shadow or judge call, which bill to the job's creator."""
|
||||
prisma = _prisma()
|
||||
router = _router()
|
||||
logger = _logger(router=router, prisma=prisma, jobs=(_job(),))
|
||||
|
||||
hook_kwargs = _success_kwargs()
|
||||
hook_kwargs = _success_kwargs(call_type=call_type)
|
||||
hook_kwargs["litellm_params"] = {
|
||||
"litellm_metadata": {"user_api_key_hash": "key-hash", "user_api_key_team_id": "team-1"}
|
||||
bucket: {
|
||||
"user_api_key_hash": "key-hash",
|
||||
"user_api_key_team_id": "team-1",
|
||||
"user_api_key_user_id": "sampled-user",
|
||||
"user_api_key_end_user_id": "sampled-end-user",
|
||||
},
|
||||
"proxy_server_request": {"body": {"messages": [{"role": "user", "content": "hi"}], "max_tokens": 9}},
|
||||
}
|
||||
await logger.async_log_success_event(hook_kwargs, RESPONSE, None, None)
|
||||
await _drain(logger)
|
||||
|
||||
shadow_call = router.acompletion.call_args_list[0].kwargs
|
||||
assert shadow_call["metadata"]["user_api_key_hash"] == "key-hash"
|
||||
assert shadow_call["metadata"]["user_api_key_team_id"] == "team-1"
|
||||
assert logger._test_admission_asks == [("eval-admin", ("my-router", "judge-model"))]
|
||||
assert router.acompletion.call_count == 2
|
||||
for call in router.acompletion.call_args_list:
|
||||
metadata = call.kwargs["metadata"]
|
||||
assert metadata["user_api_key_user_id"] == "eval-admin"
|
||||
assert {k: v for k, v in metadata.items() if v in ("key-hash", "team-1", "sampled-user", "sampled-end-user")} == {}
|
||||
|
||||
async def test_redacted_requests_are_never_shadowed(self):
|
||||
"""Redaction rewrites the logged messages before callbacks run, so this hook only
|
||||
|
|
@ -1513,24 +1547,23 @@ class TestShadowPipeline:
|
|||
router.acompletion.assert_not_called()
|
||||
assert logger._test_funnel == [("job-1", "withheld")]
|
||||
|
||||
async def test_over_budget_key_skips_before_any_call(self, monkeypatch: pytest.MonkeyPatch):
|
||||
"""The gate delegates to the auth path's own budget owner, so an over-budget
|
||||
verdict there (BudgetExceededError) skips the shadow before any provider call."""
|
||||
from litellm.exceptions import BudgetExceededError
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth import auth_checks
|
||||
|
||||
monkeypatch.setattr(
|
||||
auth_checks,
|
||||
"_virtual_key_max_budget_check",
|
||||
AsyncMock(side_effect=BudgetExceededError(current_cost=11.0, max_budget=10.0)),
|
||||
)
|
||||
@pytest.mark.parametrize(
|
||||
"job",
|
||||
[
|
||||
_job(),
|
||||
_job(router_names=("my-router", "router-b")),
|
||||
_job(direction="reverse", baseline_model="baseline-model"),
|
||||
],
|
||||
)
|
||||
async def test_a_creator_who_cannot_pay_withholds_the_sample_before_any_call(self, job: ActiveShadowEvalJob):
|
||||
"""Admission is asked once per sample about every model the sample would call, each
|
||||
arm's target plus the judge, and a creator over any of those budgets spends nothing."""
|
||||
router = _router()
|
||||
prisma = _prisma()
|
||||
logger = _logger(router=router, prisma=prisma)
|
||||
logger = _logger(router=router, prisma=prisma, creator_admitted=False)
|
||||
|
||||
await logger._run_shadow_eval(
|
||||
job=_job(),
|
||||
job=job,
|
||||
request_id="req-1",
|
||||
messages=({"role": "user", "content": "hi"},),
|
||||
real_text="real answer",
|
||||
|
|
@ -1540,12 +1573,15 @@ class TestShadowPipeline:
|
|||
real_cache_hit=False,
|
||||
control_tier=None,
|
||||
shadow_params={},
|
||||
parent_metadata={"user_api_key_auth": UserAPIKeyAuth(api_key="sk-abc", max_budget=10.0)},
|
||||
parent_metadata={"user_api_key_hash": "key-hash"},
|
||||
)
|
||||
|
||||
assert logger._test_admission_asks == [
|
||||
("eval-admin", (*(job.arm_target(arm) for arm in job.arm_router_names), "judge-model"))
|
||||
]
|
||||
router.acompletion.assert_not_called()
|
||||
prisma.db.litellm_shadowevalattempt.create.assert_not_called()
|
||||
assert logger._test_funnel == [("job-1", "withheld")]
|
||||
assert logger._test_funnel == [(job.id, "withheld")]
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"router_factory,expected_error,expected_cost,expected_shadow_cost",
|
||||
|
|
@ -1965,7 +2001,7 @@ class TestShadowPipeline:
|
|||
assert row["judge_cost"] == 0.0
|
||||
assert logger._test_counter["spend:shadow_eval:job-1"] == 0.007
|
||||
|
||||
async def test_sub_calls_carry_identity_and_origin_but_never_parent_request_state(self):
|
||||
async def test_sub_calls_carry_the_creator_and_origin_but_never_parent_request_state(self):
|
||||
prisma = _prisma()
|
||||
router = _router()
|
||||
logger = _logger(router=router, prisma=prisma)
|
||||
|
|
@ -1995,8 +2031,9 @@ class TestShadowPipeline:
|
|||
for call in (shadow_call, judge_call):
|
||||
assert call["num_retries"] == 0
|
||||
assert call["fallbacks"] == []
|
||||
assert call["metadata"]["user_api_key_hash"] == "key-hash"
|
||||
assert call["metadata"]["user_api_key_team_id"] == "team-1"
|
||||
assert call["metadata"]["user_api_key_user_id"] == "eval-admin"
|
||||
assert call["metadata"]["user_api_key_hash"] is None
|
||||
assert "user_api_key_team_id" not in call["metadata"]
|
||||
assert "user_api_key_budget_reservation" not in call["metadata"]
|
||||
assert shadow_call["metadata"][INTERNAL_CALL_ORIGIN_METADATA_KEY] == SHADOW_EVAL_ROUTER_CALL_ORIGIN
|
||||
assert judge_call["metadata"][INTERNAL_CALL_ORIGIN_METADATA_KEY] == SHADOW_EVAL_JUDGE_CALL_ORIGIN
|
||||
|
|
|
|||
127
tests/unit/proxy/auth/test_evaluation_principal.py
Normal file
127
tests/unit/proxy/auth/test_evaluation_principal.py
Normal file
|
|
@ -0,0 +1,127 @@
|
|||
"""The shadow-eval creator principal: who it is, what its stamped calls charge, and
|
||||
whether the request path's own budget owners let it pay."""
|
||||
|
||||
from typing import Final
|
||||
from unittest.mock import AsyncMock
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm import Router
|
||||
from litellm.caching.dual_cache import DualCache
|
||||
from litellm.proxy import proxy_server
|
||||
from litellm.proxy._types import LitellmUserRoles
|
||||
from litellm.proxy.auth import evaluation_principal as module
|
||||
from litellm.proxy.auth.evaluation_principal import (
|
||||
evaluation_principal,
|
||||
principal_call_metadata,
|
||||
principal_can_pay_for,
|
||||
)
|
||||
from litellm.proxy.hooks.model_max_budget_limiter import _PROXY_VirtualKeyModelMaxBudgetLimiter
|
||||
from litellm.models.user import LiteLLM_UserTable
|
||||
from litellm.types.proxy.auth.auth_checks import UserNotFoundError
|
||||
|
||||
PAID_MODEL: Final = {
|
||||
"model_name": "eval-judge",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o",
|
||||
"api_key": "k",
|
||||
"input_cost_per_token": 0.000001,
|
||||
"output_cost_per_token": 0.000002,
|
||||
},
|
||||
"model_info": {"id": "eval-judge-id"},
|
||||
}
|
||||
|
||||
|
||||
def _creator(**overrides) -> LiteLLM_UserTable:
|
||||
fields = {"user_id": "eval-admin", "user_role": "proxy_admin", "spend": 0.0, "max_budget": 10.0}
|
||||
return LiteLLM_UserTable(**{**fields, **overrides})
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def proxy(monkeypatch: pytest.MonkeyPatch) -> _PROXY_VirtualKeyModelMaxBudgetLimiter:
|
||||
limiter: Final = _PROXY_VirtualKeyModelMaxBudgetLimiter(dual_cache=DualCache())
|
||||
|
||||
async def counter_spend(counter_key: str, fallback_spend: float, max_budget: float | None = None) -> float:
|
||||
return fallback_spend
|
||||
|
||||
monkeypatch.setattr(proxy_server, "llm_router", Router(model_list=[PAID_MODEL]), raising=False)
|
||||
monkeypatch.setattr(proxy_server, "model_max_budget_limiter", limiter, raising=False)
|
||||
monkeypatch.setattr(proxy_server, "get_current_spend", counter_spend, raising=False)
|
||||
monkeypatch.setattr(proxy_server, "general_settings", {}, raising=False)
|
||||
monkeypatch.setattr(proxy_server, "prisma_client", object(), raising=False)
|
||||
return limiter
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_spend_charged_under_the_stamp_exhausts_the_budget_the_gate_reads(proxy, monkeypatch):
|
||||
"""The workflow end to end: an evaluation call stamped as the creator is charged by the
|
||||
real per-model limiter, and the next admission reads that same counter and refuses."""
|
||||
monkeypatch.setattr(
|
||||
module,
|
||||
"get_user_object",
|
||||
AsyncMock(
|
||||
return_value=_creator(model_max_budget={"eval-judge": {"budget_limit": 0.001, "time_period": "1d"}})
|
||||
),
|
||||
)
|
||||
principal = await evaluation_principal("eval-admin")
|
||||
assert await principal_can_pay_for(principal, ("eval-judge",)) is True
|
||||
|
||||
stamped = principal_call_metadata(principal)
|
||||
await proxy.async_log_success_event(
|
||||
{
|
||||
"standard_logging_object": {
|
||||
"call_type": "acompletion",
|
||||
"response_cost": 0.002,
|
||||
"model": "openai/gpt-4o",
|
||||
"model_group": "eval-judge",
|
||||
"metadata": {k: stamped.get(k) for k in ("user_api_key_hash", "user_api_key_user_id")},
|
||||
},
|
||||
"litellm_params": {"metadata": dict(stamped)},
|
||||
},
|
||||
response_obj=None,
|
||||
start_time=None,
|
||||
end_time=None,
|
||||
)
|
||||
|
||||
assert stamped["user_api_key_user_id"] == "eval-admin"
|
||||
assert await principal_can_pay_for(principal, ("eval-judge",)) is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"creator,models,expected",
|
||||
[
|
||||
(_creator(spend=10.0), ("eval-judge",), False),
|
||||
(_creator(spend=10.0), (), True),
|
||||
(_creator(spend=9.0), ("eval-judge",), True),
|
||||
(_creator(spend=999.0, max_budget=None), ("eval-judge",), True),
|
||||
],
|
||||
)
|
||||
async def test_total_budget_gates_every_paid_model(
|
||||
proxy: _PROXY_VirtualKeyModelMaxBudgetLimiter,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
creator: LiteLLM_UserTable,
|
||||
models: tuple[str, ...],
|
||||
expected: bool,
|
||||
):
|
||||
monkeypatch.setattr(module, "get_user_object", AsyncMock(return_value=creator))
|
||||
assert await principal_can_pay_for(await evaluation_principal("eval-admin"), models) is expected
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_creator_without_a_user_row_is_the_unbudgeted_master_key_admin(proxy, monkeypatch):
|
||||
monkeypatch.setattr(module, "get_user_object", AsyncMock(side_effect=UserNotFoundError(user_id="default_user_id")))
|
||||
|
||||
principal = await evaluation_principal("default_user_id")
|
||||
|
||||
assert (principal.user_id, principal.user_role) == ("default_user_id", LitellmUserRoles.PROXY_ADMIN)
|
||||
assert await principal_can_pay_for(principal, ("eval-judge",)) is True
|
||||
assert principal_call_metadata(principal)["user_api_key_user_id"] == "default_user_id"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_an_unreadable_creator_row_propagates_so_the_caller_withholds(proxy, monkeypatch):
|
||||
monkeypatch.setattr(module, "get_user_object", AsyncMock(side_effect=RuntimeError("db down")))
|
||||
|
||||
with pytest.raises(RuntimeError, match="db down"):
|
||||
await evaluation_principal("eval-admin")
|
||||
|
|
@ -38,9 +38,9 @@ const DIRECTION_OPTIONS: readonly { value: ShadowEvalDirection; label: string }[
|
|||
|
||||
const START_FORM_DESCRIPTION: Record<ShadowEvalDirection, string> = {
|
||||
forward:
|
||||
"Duplicates a sampled slice of the selected targets' traffic (keys, teams, or users) through the auto-router and has an LLM judge compare both answers blind. Each target gets its own spend budget. The router's answers are never served to users; judge calls bill to the sampled traffic's own identity.",
|
||||
"Duplicates a sampled slice of the selected targets' traffic (keys, teams, or users) through the auto-router and has an LLM judge compare both answers blind. Each target gets its own spend budget. The router's answers are never served to users; shadow and judge calls bill to you and count against your budgets.",
|
||||
reverse:
|
||||
"Duplicates a sampled slice of the traffic the auto-router already serves against a fixed baseline model and has an LLM judge compare both answers blind. Each target gets its own spend budget. The baseline's answers are never served to users; judge calls bill to the sampled traffic's own identity.",
|
||||
"Duplicates a sampled slice of the traffic the auto-router already serves against a fixed baseline model and has an LLM judge compare both answers blind. Each target gets its own spend budget. The baseline's answers are never served to users; shadow and judge calls bill to you and count against your budgets.",
|
||||
};
|
||||
|
||||
const DURATION_OPTIONS = [
|
||||
|
|
|
|||
5
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
5
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -1475,8 +1475,9 @@ export interface paths {
|
|||
* eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the
|
||||
* job's window ends, or the job is stopped, so one target running out of budget does
|
||||
* not end sampling for the others; sampling changes propagate to pods within about 10
|
||||
* seconds. Shadow and judge calls bill to the sampled request's own identity but are
|
||||
* excluded from request counts and auto-router adoption metrics.
|
||||
* seconds. Shadow and judge calls run as the admin who started the job: their spend is
|
||||
* that admin's, the admin's total and per-model budgets withhold samples once exhausted,
|
||||
* and they are excluded from request counts and auto-router adoption metrics.
|
||||
*/
|
||||
post: operations["start_shadow_eval_auto_router_shadow_eval_start_post"];
|
||||
delete?: never;
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue