mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(shadow-eval): run evaluation calls as the job creator
Shadow and judge calls copied the sampled request's key, team and user identity, so the sampled user paid for an admin's evaluation and the admin's own budgets never gated it Each sample now resolves the job's creator once, checks their total and per-model budgets through the request path's own owners, and stamps both calls with the creator's identity through the proxy's metadata stamper. A creator who cannot pay, or whose budget cannot be read, withholds the sample before any provider call Resolves LIT-9199
This commit is contained in:
parent
1ef0fe9790
commit
c05014629c
9 changed files with 363 additions and 115 deletions
|
|
@ -30,7 +30,6 @@ from litellm.constants import INTERNAL_CALL_ORIGIN_METADATA_KEY
|
|||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.integrations.websearch_interception.tools import is_web_search_tool_responses
|
||||
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs, independent_snapshot
|
||||
from litellm.litellm_core_utils.internal_call_metadata import sanitized_forwardable_call_metadata
|
||||
from litellm.litellm_core_utils.llm_judge import (
|
||||
default_router_provider,
|
||||
extract_text_from_content,
|
||||
|
|
@ -620,50 +619,30 @@ def _record_funnel_event(job_id: str, stage: "ShadowEvalFunnelStage") -> None:
|
|||
verbose_logger.debug("shadow_eval: funnel increment failed for %s: %s", job_id, e)
|
||||
|
||||
|
||||
async def _key_or_team_is_over_budget(metadata: Mapping[str, object]) -> bool:
|
||||
"""Whether the shadowed key or its team is over budget, decided by the same owners
|
||||
the request path uses, so counter keys and thresholds can never drift from auth's.
|
||||
|
||||
Advisory and fail-open: real traffic on an over-budget key is already rejected at
|
||||
auth (so nothing reaches the success hook), and this gate only closes the race
|
||||
where the key crosses its budget while a request is in flight.
|
||||
"""
|
||||
async def _admit_creator(creator_user_id: str | None, models: Sequence[str]) -> Mapping[str, object] | None:
|
||||
"""The call metadata the job's creator pays under, or None when the creator cannot pay
|
||||
for ``models`` or that cannot be verified, so the sample is withheld before any spend."""
|
||||
if creator_user_id is None:
|
||||
return None
|
||||
try:
|
||||
from litellm.exceptions import BudgetExceededError
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth.auth_checks import (
|
||||
_team_max_budget_check,
|
||||
_virtual_key_max_budget_check,
|
||||
get_team_object,
|
||||
from litellm.proxy.auth.evaluation_principal import (
|
||||
evaluation_principal,
|
||||
principal_call_metadata,
|
||||
principal_can_pay_for,
|
||||
)
|
||||
from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache
|
||||
except ImportError:
|
||||
return False
|
||||
|
||||
auth: Final = metadata.get("user_api_key_auth")
|
||||
if not isinstance(auth, UserAPIKeyAuth):
|
||||
return False
|
||||
try:
|
||||
await _virtual_key_max_budget_check(valid_token=auth, proxy_logging_obj=proxy_logging_obj)
|
||||
if auth.team_id:
|
||||
team: Final = await get_team_object(
|
||||
team_id=auth.team_id,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
check_cache_only=True,
|
||||
)
|
||||
await _team_max_budget_check(team_object=team, valid_token=auth, proxy_logging_obj=proxy_logging_obj)
|
||||
except BudgetExceededError:
|
||||
return True
|
||||
except Exception as e: # noqa: BLE001 # advisory gate: a failed read must not block sampling
|
||||
verbose_logger.debug("shadow_eval: budget read failed: %s", e)
|
||||
return False
|
||||
principal: Final = await evaluation_principal(creator_user_id)
|
||||
if not await principal_can_pay_for(principal, models):
|
||||
return None
|
||||
return principal_call_metadata(principal)
|
||||
except Exception as e: # noqa: BLE001 # an unverifiable creator budget withholds the sample rather than spending
|
||||
verbose_logger.warning("shadow_eval: creator %s budget unverifiable, sample withheld: %s", creator_user_id, e)
|
||||
return None
|
||||
|
||||
|
||||
def _forwarded_team_id(metadata: Mapping[str, object]) -> str | None:
|
||||
"""The shadowed key's team, the identity the judge call already carries in its metadata
|
||||
and the router already selects deployments with. Read here too so the arm choice, which
|
||||
happens before the router sees the call, is made under the same team."""
|
||||
"""The shadowed key's team, the scope start-time validation resolved the judge under,
|
||||
so the judge's router-or-SDK arm is chosen the way the job was validated."""
|
||||
team_id: Final = metadata.get("user_api_key_team_id")
|
||||
return team_id if isinstance(team_id, str) and team_id else None
|
||||
|
||||
|
|
@ -749,6 +728,7 @@ class ActiveShadowEvalJob(BaseModel):
|
|||
judge_model: str
|
||||
max_turns: int
|
||||
max_budget: float | None = None
|
||||
created_by: str | None = None
|
||||
ends_at: datetime
|
||||
attempts: int = 0
|
||||
spend: float = 0.0
|
||||
|
|
@ -833,6 +813,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
job_spend_reader: Callable[[str, float, float], Awaitable[float]] | None = None,
|
||||
job_spend_writer: Callable[[str, float], Awaitable[None]] | None = None,
|
||||
funnel_recorder: Callable[[str, "ShadowEvalFunnelStage"], None] | None = None,
|
||||
creator_admission: Callable[[str | None, Sequence[str]], Awaitable[Mapping[str, object] | None]] | None = None,
|
||||
) -> None:
|
||||
"""Providers are callables so the proxy's lazily-initialized globals are resolved
|
||||
at call time, not at logger construction. The spend reader and writer wrap the
|
||||
|
|
@ -843,6 +824,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
self._read_job_spend = job_spend_reader or _job_spend_from_counter
|
||||
self._write_job_spend = job_spend_writer or _add_job_spend_to_counter
|
||||
self._record_funnel = funnel_recorder or _record_funnel_event
|
||||
self._admit_creator = creator_admission or _admit_creator
|
||||
self._inflight_shadow_tasks: int = 0
|
||||
# Starts per job since the last cache fill, never decremented within a
|
||||
# generation; the refill absorbs written rows and resets.
|
||||
|
|
@ -1055,8 +1037,9 @@ class ShadowEvalLogger(CustomLogger):
|
|||
) -> None:
|
||||
"""Budget gates once per sampled request, then every router arm in turn: shadow
|
||||
call -> blind judge -> one attempt row stamped with the arm. The gates that
|
||||
decline to spend on an admitted sample (no DB to record into, an over-budget key,
|
||||
an unverifiable or exhausted eval budget) count the REQUEST withheld before any
|
||||
decline to spend on an admitted sample (no DB to record into, an unverifiable or
|
||||
exhausted eval budget, a creator who cannot pay for every model the sample would
|
||||
call) count the REQUEST withheld before any
|
||||
arm runs, so funnel counters stay per-request and a leg's eligible traffic still
|
||||
reconciles as not_sampled + unjudgeable + shed + withheld + sampled requests,
|
||||
where each sampled request writes one attempt row per arm. A budget crossed
|
||||
|
|
@ -1069,9 +1052,6 @@ class ShadowEvalLogger(CustomLogger):
|
|||
if prisma is None:
|
||||
self._record_funnel(job.id, "withheld")
|
||||
return
|
||||
if await _key_or_team_is_over_budget(parent_metadata):
|
||||
self._record_funnel(job.id, "withheld")
|
||||
return
|
||||
if job.max_budget is not None:
|
||||
try:
|
||||
spend: Final = await self._read_job_spend(_job_spend_counter_key(job.id), job.spend, job.max_budget)
|
||||
|
|
@ -1082,6 +1062,12 @@ class ShadowEvalLogger(CustomLogger):
|
|||
if spend >= job.max_budget:
|
||||
self._record_funnel(job.id, "withheld")
|
||||
return
|
||||
creator_metadata: Final = await self._admit_creator(
|
||||
job.created_by, (*(job.arm_target(arm) for arm in job.arm_router_names), job.judge_model)
|
||||
)
|
||||
if creator_metadata is None:
|
||||
self._record_funnel(job.id, "withheld")
|
||||
return
|
||||
for arm_router in job.arm_router_names:
|
||||
await self._run_shadow_arm(
|
||||
prisma=prisma,
|
||||
|
|
@ -1097,6 +1083,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
control_tier=control_tier,
|
||||
shadow_params=shadow_params,
|
||||
parent_metadata=parent_metadata,
|
||||
creator_metadata=creator_metadata,
|
||||
)
|
||||
|
||||
async def _run_shadow_arm(
|
||||
|
|
@ -1114,12 +1101,13 @@ class ShadowEvalLogger(CustomLogger):
|
|||
control_tier: str | None,
|
||||
shadow_params: Mapping[str, object],
|
||||
parent_metadata: Mapping[str, object],
|
||||
creator_metadata: Mapping[str, object],
|
||||
) -> None:
|
||||
"""One arm's pipeline: shadow call -> blind judge -> one attempt row, every exit
|
||||
recording this arm's outcome, so one arm's fault never silences a sibling arm."""
|
||||
try:
|
||||
shadow: Final = await self._call_router_shadow(
|
||||
job.arm_target(arm_router), messages, shadow_params, parent_metadata
|
||||
job.arm_target(arm_router), messages, shadow_params, creator_metadata
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 # detached task: nothing billed yet, record and never raise
|
||||
verbose_logger.debug("shadow_eval: pipeline failed for %s: %s", request_id, e)
|
||||
|
|
@ -1161,6 +1149,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
shadow_text=shadow.text,
|
||||
tools=shadow_params.get("tools"),
|
||||
parent_metadata=parent_metadata,
|
||||
creator_metadata=creator_metadata,
|
||||
)
|
||||
if isinstance(verdict, _CallFailure):
|
||||
await self._record_attempt(
|
||||
|
|
@ -1268,18 +1257,19 @@ class ShadowEvalLogger(CustomLogger):
|
|||
target_model: str,
|
||||
messages: Sequence[Mapping[str, object]],
|
||||
shadow_params: Mapping[str, object],
|
||||
parent_metadata: Mapping[str, object],
|
||||
creator_metadata: Mapping[str, object],
|
||||
) -> "_ShadowResponse | _CallFailure":
|
||||
"""Send the prompt through the arm nobody was served: the auto-router under
|
||||
evaluation, or a reverse job's fixed baseline model. The metadata carries the
|
||||
shadowed key's identity (spend attribution) and receives a routing decision
|
||||
write-back, which a plain baseline model simply never makes."""
|
||||
evaluation, or a reverse job's fixed baseline model. The call runs as the job's
|
||||
creator and receives a routing decision write-back, which a plain baseline model
|
||||
simply never makes."""
|
||||
router: Final = self._router_provider()
|
||||
if router is None:
|
||||
return _CallFailure("no router configured on this pod")
|
||||
shadow_metadata: Final[dict[str, object]] = ( # mutable-ok: router writes its routing decision back
|
||||
sanitized_forwardable_call_metadata(parent_metadata, SHADOW_EVAL_ROUTER_CALL_ORIGIN)
|
||||
)
|
||||
shadow_metadata: Final[dict[str, object]] = { # mutable-ok: router writes its routing decision back
|
||||
**creator_metadata,
|
||||
INTERNAL_CALL_ORIGIN_METADATA_KEY: SHADOW_EVAL_ROUTER_CALL_ORIGIN,
|
||||
}
|
||||
try:
|
||||
response: Final = await router.acompletion(
|
||||
model=target_model,
|
||||
|
|
@ -1321,6 +1311,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
shadow_text: str,
|
||||
tools: object,
|
||||
parent_metadata: Mapping[str, object],
|
||||
creator_metadata: Mapping[str, object],
|
||||
) -> "_JudgeVerdict | _CallFailure":
|
||||
"""Blind pairwise judge with A/B labels randomized to cancel position bias. Both
|
||||
arms were offered the same tools, so the judge is shown their definitions too: a
|
||||
|
|
@ -1334,7 +1325,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
for m in messages
|
||||
if m.get("content") is not None
|
||||
)
|
||||
judge_metadata: Final = sanitized_forwardable_call_metadata(parent_metadata, SHADOW_EVAL_JUDGE_CALL_ORIGIN)
|
||||
judge_metadata: Final = {**creator_metadata, INTERNAL_CALL_ORIGIN_METADATA_KEY: SHADOW_EVAL_JUDGE_CALL_ORIGIN}
|
||||
judge_messages: Final = [
|
||||
{"role": "system", "content": PAIRWISE_JUDGE_SYSTEM_PROMPT},
|
||||
{
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
"""Metadata a request forwards to the internal LLM sub-calls it triggers.
|
||||
|
||||
Internal features (the auto-router's classifier and embeddings, shadow eval's shadow and
|
||||
judge calls) bill real provider spend that nobody typed a prompt for. That spend must land
|
||||
on the same key/team/org/user as the request that caused it, so the sub-call carries the
|
||||
caller's identity metadata, minus two things that must never be forwarded as-is:
|
||||
Internal features (the auto-router's classifier, embeddings and context compaction) bill
|
||||
real provider spend that nobody typed a prompt for. That spend must land on the same
|
||||
key/team/org/user as the request that caused it, so the sub-call carries the caller's
|
||||
identity metadata, minus two things that must never be forwarded as-is:
|
||||
|
||||
* ``user_api_key_budget_reservation`` (and the reservation nested inside
|
||||
``user_api_key_auth``) belongs to the parent completion. If a sub-call's cost callback
|
||||
|
|
@ -161,8 +161,8 @@ def sanitized_forwardable_call_metadata(
|
|||
) -> dict[str, object]: # mutable-ok: SDK metadata kwarg
|
||||
"""Just the caller's identity, stamped with the sub-call's origin.
|
||||
|
||||
For sub-calls detached from the parent request (shadow eval), which outlive it and
|
||||
must not inherit per-request state such as its routing decision or logging payload.
|
||||
For sub-calls that must not inherit the parent's per-request state, such as its
|
||||
routing decision or logging payload.
|
||||
"""
|
||||
identity: Final = {k: v for k, v in parent_metadata.items() if k in FORWARDABLE_IDENTITY_METADATA_KEYS}
|
||||
return _sanitized(identity) | {INTERNAL_CALL_ORIGIN_METADATA_KEY: call_origin}
|
||||
|
|
|
|||
88
litellm/proxy/auth/evaluation_principal.py
Normal file
88
litellm/proxy/auth/evaluation_principal.py
Normal file
|
|
@ -0,0 +1,88 @@
|
|||
"""The principal a shadow evaluation's own LLM calls run as: the admin who created the job.
|
||||
|
||||
Evaluation calls are spend the creator chose to incur, so they are attributed, routed and
|
||||
budget-checked as that admin's own requests, through the same owners an authenticated
|
||||
request uses. Nothing the sampled request carried decides who pays for them.
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
from types import MappingProxyType
|
||||
from typing import Final
|
||||
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.exceptions import BudgetExceededError
|
||||
from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
|
||||
from litellm.proxy.auth.auth_checks import effective_user_role, get_user_object
|
||||
from litellm.proxy.auth.fallback_budget import is_token_within_budget_for_model
|
||||
from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup
|
||||
from litellm.types.proxy.auth.auth_checks import UserNotFoundError
|
||||
|
||||
_MODEL_BUDGETS: Final[TypeAdapter[Mapping[str, object] | None]] = TypeAdapter(Mapping[str, object] | None)
|
||||
|
||||
|
||||
async def evaluation_principal(creator_user_id: str) -> UserAPIKeyAuth:
|
||||
"""The creator as a principal carrying their current budgets.
|
||||
|
||||
A creator with no user row is the master key's admin, which has no personal budget to
|
||||
enforce: the spend is still attributed to that id. Any other read failure propagates, so
|
||||
the caller withholds work it cannot verify the creator can pay for.
|
||||
"""
|
||||
from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache
|
||||
|
||||
try:
|
||||
user: Final = await get_user_object(
|
||||
user_id=creator_user_id,
|
||||
prisma_client=prisma_client,
|
||||
user_api_key_cache=user_api_key_cache,
|
||||
user_id_upsert=False,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
)
|
||||
except UserNotFoundError:
|
||||
return UserAPIKeyAuth(user_id=creator_user_id, user_role=LitellmUserRoles.PROXY_ADMIN)
|
||||
if user is None:
|
||||
return UserAPIKeyAuth(user_id=creator_user_id, user_role=LitellmUserRoles.PROXY_ADMIN)
|
||||
return UserAPIKeyAuth(
|
||||
user_id=user.user_id,
|
||||
user_role=effective_user_role(user.user_role),
|
||||
user_email=user.user_email,
|
||||
user_spend=user.spend,
|
||||
user_max_budget=user.max_budget,
|
||||
user_model_max_budget=_MODEL_BUDGETS.validate_python(user.model_max_budget),
|
||||
)
|
||||
|
||||
|
||||
async def principal_can_pay_for(principal: UserAPIKeyAuth, models: Sequence[str]) -> bool:
|
||||
"""Whether the principal is within its total budget and every per-model budget for
|
||||
``models``, read through the owners the request path enforces them with."""
|
||||
from litellm.proxy.proxy_server import llm_router, model_max_budget_limiter
|
||||
|
||||
if llm_router is None:
|
||||
return False
|
||||
try:
|
||||
for model in models:
|
||||
if not await is_token_within_budget_for_model(model=model, valid_token=principal, llm_router=llm_router):
|
||||
return False
|
||||
if principal.user_id is not None and principal.user_model_max_budget:
|
||||
await model_max_budget_limiter.is_user_within_model_budget(
|
||||
user_id=principal.user_id, user_model_max_budget=principal.user_model_max_budget, model=model
|
||||
)
|
||||
except BudgetExceededError as e:
|
||||
verbose_proxy_logger.debug("shadow_eval: creator %s is over budget: %s", principal.user_id, e)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def principal_call_metadata(principal: UserAPIKeyAuth) -> Mapping[str, object]:
|
||||
"""Request metadata naming ``principal`` exactly as auth stamps it on a request the
|
||||
principal sent, so every spend writer, limiter and router filter reads one identity."""
|
||||
stamped: Final[dict[str, object]] = {} # mutable-ok: the stampers write into a request dict
|
||||
request: Final = {"metadata": stamped}
|
||||
LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata(
|
||||
data=request, user_api_key_dict=principal, _metadata_variable_name="metadata"
|
||||
)
|
||||
LiteLLMProxyRequestSetup.add_budget_metadata_to_request_metadata(
|
||||
data=request, user_api_key_dict=principal, _metadata_variable_name="metadata"
|
||||
)
|
||||
return MappingProxyType(stamped)
|
||||
|
|
@ -1650,6 +1650,27 @@ class LiteLLMProxyRequestSetup:
|
|||
)
|
||||
return user_api_key_logged_metadata
|
||||
|
||||
@staticmethod
|
||||
def add_budget_metadata_to_request_metadata(
|
||||
data: dict, user_api_key_dict: UserAPIKeyAuth, _metadata_variable_name: str
|
||||
) -> None:
|
||||
"""The principal's budgets and spend: prometheus reports them and the per-model budget
|
||||
limiter reads the ``*_model_max_budget`` entries to charge spend."""
|
||||
data[_metadata_variable_name]["user_api_key_team_max_budget"] = user_api_key_dict.team_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_team_spend"] = user_api_key_dict.team_spend
|
||||
data[_metadata_variable_name]["user_api_key_team_model_max_budget"] = user_api_key_dict.team_model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_request_route"] = user_api_key_dict.request_route
|
||||
data[_metadata_variable_name]["user_api_key_spend"] = user_api_key_dict.spend
|
||||
data[_metadata_variable_name]["user_api_key_max_budget"] = user_api_key_dict.max_budget
|
||||
data[_metadata_variable_name]["user_api_key_model_max_budget"] = user_api_key_dict.model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_end_user_model_max_budget"] = (
|
||||
user_api_key_dict.end_user_model_max_budget
|
||||
)
|
||||
data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend
|
||||
data[_metadata_variable_name]["user_api_key_user_max_budget"] = user_api_key_dict.user_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_user_model_max_budget"] = user_api_key_dict.user_model_max_budget
|
||||
data[_metadata_variable_name].update(carried_budget_metadata(user_api_key_dict))
|
||||
|
||||
@staticmethod
|
||||
def add_user_api_key_auth_to_request_metadata(
|
||||
data: dict,
|
||||
|
|
@ -2405,28 +2426,10 @@ async def add_litellm_data_to_request(
|
|||
_metadata_variable_name=_metadata_variable_name,
|
||||
)
|
||||
|
||||
# Team spend, budget - used by prometheus.py
|
||||
data[_metadata_variable_name]["user_api_key_team_max_budget"] = user_api_key_dict.team_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_team_spend"] = user_api_key_dict.team_spend
|
||||
data[_metadata_variable_name]["user_api_key_team_model_max_budget"] = user_api_key_dict.team_model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_request_route"] = user_api_key_dict.request_route
|
||||
|
||||
# API Key spend, budget - used by prometheus.py
|
||||
data[_metadata_variable_name]["user_api_key_spend"] = user_api_key_dict.spend
|
||||
data[_metadata_variable_name]["user_api_key_max_budget"] = user_api_key_dict.max_budget
|
||||
data[_metadata_variable_name]["user_api_key_model_max_budget"] = user_api_key_dict.model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_end_user_model_max_budget"] = (
|
||||
user_api_key_dict.end_user_model_max_budget
|
||||
LiteLLMProxyRequestSetup.add_budget_metadata_to_request_metadata(
|
||||
data=data, user_api_key_dict=user_api_key_dict, _metadata_variable_name=_metadata_variable_name
|
||||
)
|
||||
|
||||
# User spend, budget - used by prometheus.py
|
||||
# Follow same pattern as team and API key budgets
|
||||
data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend
|
||||
data[_metadata_variable_name]["user_api_key_user_max_budget"] = user_api_key_dict.user_max_budget
|
||||
user_model_budget: Final = user_api_key_dict.user_model_max_budget
|
||||
data[_metadata_variable_name]["user_api_key_user_model_max_budget"] = user_model_budget # rebind-ok: out-param
|
||||
data[_metadata_variable_name].update(carried_budget_metadata(user_api_key_dict))
|
||||
|
||||
data[_metadata_variable_name]["user_api_key_metadata"] = strip_callback_config(user_api_key_dict.metadata)
|
||||
data[_metadata_variable_name]["user_api_key_team_metadata"] = strip_callback_config(user_api_key_dict.team_metadata)
|
||||
data[_metadata_variable_name]["user_api_key_object_permission_id"] = getattr(
|
||||
|
|
|
|||
|
|
@ -1684,8 +1684,9 @@ async def start_shadow_eval(
|
|||
eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the
|
||||
job's window ends, or the job is stopped, so one target running out of budget does
|
||||
not end sampling for the others; sampling changes propagate to pods within about 10
|
||||
seconds. Shadow and judge calls bill to the sampled request's own identity but are
|
||||
excluded from request counts and auto-router adoption metrics.
|
||||
seconds. Shadow and judge calls run as the admin who started the job: their spend is
|
||||
that admin's, the admin's total and per-model budgets withhold samples once exhausted,
|
||||
and they are excluded from request counts and auto-router adoption metrics.
|
||||
"""
|
||||
from litellm.proxy.proxy_server import llm_router, prisma_client
|
||||
|
||||
|
|
|
|||
|
|
@ -2,8 +2,9 @@
|
|||
the detached pipeline's single attempt-row write, and the cache-first job lookup."""
|
||||
|
||||
import asyncio
|
||||
from collections.abc import Mapping
|
||||
from collections.abc import Mapping, Sequence
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from types import MappingProxyType
|
||||
from typing import Final, Literal
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
|
|
@ -55,6 +56,7 @@ def _job(**overrides) -> ActiveShadowEvalJob:
|
|||
max_turns=200,
|
||||
ends_at=datetime.now(timezone.utc) + timedelta(days=1),
|
||||
attempts=0,
|
||||
created_by="eval-admin",
|
||||
)
|
||||
return ActiveShadowEvalJob(**{**defaults, **overrides})
|
||||
|
||||
|
|
@ -92,6 +94,7 @@ def _job_record(job: ActiveShadowEvalJob, target_type="key", target_id="key-hash
|
|||
judge_model=job.judge_model,
|
||||
max_turns=job.max_turns,
|
||||
max_budget=job.max_budget,
|
||||
created_by=job.created_by,
|
||||
ends_at=job.ends_at,
|
||||
).items():
|
||||
setattr(record, field, value)
|
||||
|
|
@ -230,10 +233,28 @@ def _spend_counter(store=None):
|
|||
return counter, read, write
|
||||
|
||||
|
||||
def _logger(router=None, prisma=None, jobs=(), counter_store=None, jobs_by_target=None) -> ShadowEvalLogger:
|
||||
CREATOR_METADATA: Final = MappingProxyType({"user_api_key_user_id": "eval-admin", "user_api_key_hash": None})
|
||||
|
||||
|
||||
def _creator_admission(admitted: bool = True):
|
||||
"""Stands in for the proxy's creator budget owners: answers with the creator's call
|
||||
metadata, or None for a creator who cannot pay, and records what it was asked."""
|
||||
asked: Final[list[tuple[str | None, tuple[str, ...]]]] = []
|
||||
|
||||
async def admit(creator_user_id: str | None, models: Sequence[str]) -> Mapping[str, object] | None:
|
||||
asked.append((creator_user_id, tuple(models)))
|
||||
return CREATOR_METADATA if admitted else None
|
||||
|
||||
return admit, asked
|
||||
|
||||
|
||||
def _logger(
|
||||
router=None, prisma=None, jobs=(), counter_store=None, jobs_by_target=None, creator_admitted=True
|
||||
) -> ShadowEvalLogger:
|
||||
cache = InMemoryCache(max_size_in_memory=4, default_ttl=60)
|
||||
counter, read, write = _spend_counter(counter_store)
|
||||
funnel_events = []
|
||||
admit, asked = _creator_admission(creator_admitted)
|
||||
logger = ShadowEvalLogger(
|
||||
router_provider=lambda: router,
|
||||
prisma_provider=lambda: prisma,
|
||||
|
|
@ -241,9 +262,11 @@ def _logger(router=None, prisma=None, jobs=(), counter_store=None, jobs_by_targe
|
|||
job_spend_reader=read,
|
||||
job_spend_writer=write,
|
||||
funnel_recorder=lambda job_id, stage: funnel_events.append((job_id, stage)),
|
||||
creator_admission=admit,
|
||||
)
|
||||
logger._test_counter = counter
|
||||
logger._test_funnel = funnel_events
|
||||
logger._test_admission_asks = asked
|
||||
seeded = jobs_by_target if jobs_by_target is not None else ({("key", "key-hash"): tuple(jobs)} if jobs else None)
|
||||
if seeded is not None:
|
||||
cache.set_cache("shadow_eval:active_jobs", seeded)
|
||||
|
|
@ -1225,24 +1248,35 @@ class TestSuccessHookSkipChain:
|
|||
|
||||
assert prisma.db.litellm_shadowevalattempt.create.await_count == 1
|
||||
|
||||
async def test_v1_messages_surface_forwards_identity_from_litellm_metadata(self):
|
||||
"""/v1/messages stores identity in litellm_params.litellm_metadata, so the hook
|
||||
resolves the bucket through the shared helper; every surface forwards the same
|
||||
identity to the shadow and judge calls."""
|
||||
@pytest.mark.parametrize(
|
||||
"call_type,bucket", [("acompletion", "metadata"), ("anthropic_messages", "litellm_metadata")]
|
||||
)
|
||||
async def test_eval_calls_run_as_the_creator_whatever_the_sampled_caller_was(self, call_type: str, bucket: str):
|
||||
"""Every surface's sampled request carries its caller's identity in a different bucket;
|
||||
none of it reaches the shadow or judge call, which bill to the job's creator."""
|
||||
prisma = _prisma()
|
||||
router = _router()
|
||||
logger = _logger(router=router, prisma=prisma, jobs=(_job(),))
|
||||
|
||||
hook_kwargs = _success_kwargs()
|
||||
hook_kwargs = _success_kwargs(call_type=call_type)
|
||||
hook_kwargs["litellm_params"] = {
|
||||
"litellm_metadata": {"user_api_key_hash": "key-hash", "user_api_key_team_id": "team-1"}
|
||||
bucket: {
|
||||
"user_api_key_hash": "key-hash",
|
||||
"user_api_key_team_id": "team-1",
|
||||
"user_api_key_user_id": "sampled-user",
|
||||
"user_api_key_end_user_id": "sampled-end-user",
|
||||
},
|
||||
"proxy_server_request": {"body": {"messages": [{"role": "user", "content": "hi"}], "max_tokens": 9}},
|
||||
}
|
||||
await logger.async_log_success_event(hook_kwargs, RESPONSE, None, None)
|
||||
await _drain(logger)
|
||||
|
||||
shadow_call = router.acompletion.call_args_list[0].kwargs
|
||||
assert shadow_call["metadata"]["user_api_key_hash"] == "key-hash"
|
||||
assert shadow_call["metadata"]["user_api_key_team_id"] == "team-1"
|
||||
assert logger._test_admission_asks == [("eval-admin", ("my-router", "judge-model"))]
|
||||
assert router.acompletion.call_count == 2
|
||||
for call in router.acompletion.call_args_list:
|
||||
metadata = call.kwargs["metadata"]
|
||||
assert metadata["user_api_key_user_id"] == "eval-admin"
|
||||
assert {k: v for k, v in metadata.items() if v in ("key-hash", "team-1", "sampled-user", "sampled-end-user")} == {}
|
||||
|
||||
async def test_redacted_requests_are_never_shadowed(self):
|
||||
"""Redaction rewrites the logged messages before callbacks run, so this hook only
|
||||
|
|
@ -1513,24 +1547,23 @@ class TestShadowPipeline:
|
|||
router.acompletion.assert_not_called()
|
||||
assert logger._test_funnel == [("job-1", "withheld")]
|
||||
|
||||
async def test_over_budget_key_skips_before_any_call(self, monkeypatch: pytest.MonkeyPatch):
|
||||
"""The gate delegates to the auth path's own budget owner, so an over-budget
|
||||
verdict there (BudgetExceededError) skips the shadow before any provider call."""
|
||||
from litellm.exceptions import BudgetExceededError
|
||||
from litellm.proxy._types import UserAPIKeyAuth
|
||||
from litellm.proxy.auth import auth_checks
|
||||
|
||||
monkeypatch.setattr(
|
||||
auth_checks,
|
||||
"_virtual_key_max_budget_check",
|
||||
AsyncMock(side_effect=BudgetExceededError(current_cost=11.0, max_budget=10.0)),
|
||||
)
|
||||
@pytest.mark.parametrize(
|
||||
"job",
|
||||
[
|
||||
_job(),
|
||||
_job(router_names=("my-router", "router-b")),
|
||||
_job(direction="reverse", baseline_model="baseline-model"),
|
||||
],
|
||||
)
|
||||
async def test_a_creator_who_cannot_pay_withholds_the_sample_before_any_call(self, job: ActiveShadowEvalJob):
|
||||
"""Admission is asked once per sample about every model the sample would call, each
|
||||
arm's target plus the judge, and a creator over any of those budgets spends nothing."""
|
||||
router = _router()
|
||||
prisma = _prisma()
|
||||
logger = _logger(router=router, prisma=prisma)
|
||||
logger = _logger(router=router, prisma=prisma, creator_admitted=False)
|
||||
|
||||
await logger._run_shadow_eval(
|
||||
job=_job(),
|
||||
job=job,
|
||||
request_id="req-1",
|
||||
messages=({"role": "user", "content": "hi"},),
|
||||
real_text="real answer",
|
||||
|
|
@ -1540,12 +1573,15 @@ class TestShadowPipeline:
|
|||
real_cache_hit=False,
|
||||
control_tier=None,
|
||||
shadow_params={},
|
||||
parent_metadata={"user_api_key_auth": UserAPIKeyAuth(api_key="sk-abc", max_budget=10.0)},
|
||||
parent_metadata={"user_api_key_hash": "key-hash"},
|
||||
)
|
||||
|
||||
assert logger._test_admission_asks == [
|
||||
("eval-admin", (*(job.arm_target(arm) for arm in job.arm_router_names), "judge-model"))
|
||||
]
|
||||
router.acompletion.assert_not_called()
|
||||
prisma.db.litellm_shadowevalattempt.create.assert_not_called()
|
||||
assert logger._test_funnel == [("job-1", "withheld")]
|
||||
assert logger._test_funnel == [(job.id, "withheld")]
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"router_factory,expected_error,expected_cost,expected_shadow_cost",
|
||||
|
|
@ -1965,7 +2001,7 @@ class TestShadowPipeline:
|
|||
assert row["judge_cost"] == 0.0
|
||||
assert logger._test_counter["spend:shadow_eval:job-1"] == 0.007
|
||||
|
||||
async def test_sub_calls_carry_identity_and_origin_but_never_parent_request_state(self):
|
||||
async def test_sub_calls_carry_the_creator_and_origin_but_never_parent_request_state(self):
|
||||
prisma = _prisma()
|
||||
router = _router()
|
||||
logger = _logger(router=router, prisma=prisma)
|
||||
|
|
@ -1995,8 +2031,9 @@ class TestShadowPipeline:
|
|||
for call in (shadow_call, judge_call):
|
||||
assert call["num_retries"] == 0
|
||||
assert call["fallbacks"] == []
|
||||
assert call["metadata"]["user_api_key_hash"] == "key-hash"
|
||||
assert call["metadata"]["user_api_key_team_id"] == "team-1"
|
||||
assert call["metadata"]["user_api_key_user_id"] == "eval-admin"
|
||||
assert call["metadata"]["user_api_key_hash"] is None
|
||||
assert "user_api_key_team_id" not in call["metadata"]
|
||||
assert "user_api_key_budget_reservation" not in call["metadata"]
|
||||
assert shadow_call["metadata"][INTERNAL_CALL_ORIGIN_METADATA_KEY] == SHADOW_EVAL_ROUTER_CALL_ORIGIN
|
||||
assert judge_call["metadata"][INTERNAL_CALL_ORIGIN_METADATA_KEY] == SHADOW_EVAL_JUDGE_CALL_ORIGIN
|
||||
|
|
|
|||
127
tests/unit/proxy/auth/test_evaluation_principal.py
Normal file
127
tests/unit/proxy/auth/test_evaluation_principal.py
Normal file
|
|
@ -0,0 +1,127 @@
|
|||
"""The shadow-eval creator principal: who it is, what its stamped calls charge, and
|
||||
whether the request path's own budget owners let it pay."""
|
||||
|
||||
from typing import Final
|
||||
from unittest.mock import AsyncMock
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm import Router
|
||||
from litellm.caching.dual_cache import DualCache
|
||||
from litellm.proxy import proxy_server
|
||||
from litellm.proxy._types import LitellmUserRoles
|
||||
from litellm.proxy.auth import evaluation_principal as module
|
||||
from litellm.proxy.auth.evaluation_principal import (
|
||||
evaluation_principal,
|
||||
principal_call_metadata,
|
||||
principal_can_pay_for,
|
||||
)
|
||||
from litellm.proxy.hooks.model_max_budget_limiter import _PROXY_VirtualKeyModelMaxBudgetLimiter
|
||||
from litellm.models.user import LiteLLM_UserTable
|
||||
from litellm.types.proxy.auth.auth_checks import UserNotFoundError
|
||||
|
||||
PAID_MODEL: Final = {
|
||||
"model_name": "eval-judge",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o",
|
||||
"api_key": "k",
|
||||
"input_cost_per_token": 0.000001,
|
||||
"output_cost_per_token": 0.000002,
|
||||
},
|
||||
"model_info": {"id": "eval-judge-id"},
|
||||
}
|
||||
|
||||
|
||||
def _creator(**overrides) -> LiteLLM_UserTable:
|
||||
fields = {"user_id": "eval-admin", "user_role": "proxy_admin", "spend": 0.0, "max_budget": 10.0}
|
||||
return LiteLLM_UserTable(**{**fields, **overrides})
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def proxy(monkeypatch: pytest.MonkeyPatch) -> _PROXY_VirtualKeyModelMaxBudgetLimiter:
|
||||
limiter: Final = _PROXY_VirtualKeyModelMaxBudgetLimiter(dual_cache=DualCache())
|
||||
|
||||
async def counter_spend(counter_key: str, fallback_spend: float, max_budget: float | None = None) -> float:
|
||||
return fallback_spend
|
||||
|
||||
monkeypatch.setattr(proxy_server, "llm_router", Router(model_list=[PAID_MODEL]), raising=False)
|
||||
monkeypatch.setattr(proxy_server, "model_max_budget_limiter", limiter, raising=False)
|
||||
monkeypatch.setattr(proxy_server, "get_current_spend", counter_spend, raising=False)
|
||||
monkeypatch.setattr(proxy_server, "general_settings", {}, raising=False)
|
||||
monkeypatch.setattr(proxy_server, "prisma_client", object(), raising=False)
|
||||
return limiter
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_spend_charged_under_the_stamp_exhausts_the_budget_the_gate_reads(proxy, monkeypatch):
|
||||
"""The workflow end to end: an evaluation call stamped as the creator is charged by the
|
||||
real per-model limiter, and the next admission reads that same counter and refuses."""
|
||||
monkeypatch.setattr(
|
||||
module,
|
||||
"get_user_object",
|
||||
AsyncMock(
|
||||
return_value=_creator(model_max_budget={"eval-judge": {"budget_limit": 0.001, "time_period": "1d"}})
|
||||
),
|
||||
)
|
||||
principal = await evaluation_principal("eval-admin")
|
||||
assert await principal_can_pay_for(principal, ("eval-judge",)) is True
|
||||
|
||||
stamped = principal_call_metadata(principal)
|
||||
await proxy.async_log_success_event(
|
||||
{
|
||||
"standard_logging_object": {
|
||||
"call_type": "acompletion",
|
||||
"response_cost": 0.002,
|
||||
"model": "openai/gpt-4o",
|
||||
"model_group": "eval-judge",
|
||||
"metadata": {k: stamped.get(k) for k in ("user_api_key_hash", "user_api_key_user_id")},
|
||||
},
|
||||
"litellm_params": {"metadata": dict(stamped)},
|
||||
},
|
||||
response_obj=None,
|
||||
start_time=None,
|
||||
end_time=None,
|
||||
)
|
||||
|
||||
assert stamped["user_api_key_user_id"] == "eval-admin"
|
||||
assert await principal_can_pay_for(principal, ("eval-judge",)) is False
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"creator,models,expected",
|
||||
[
|
||||
(_creator(spend=10.0), ("eval-judge",), False),
|
||||
(_creator(spend=10.0), (), True),
|
||||
(_creator(spend=9.0), ("eval-judge",), True),
|
||||
(_creator(spend=999.0, max_budget=None), ("eval-judge",), True),
|
||||
],
|
||||
)
|
||||
async def test_total_budget_gates_every_paid_model(
|
||||
proxy: _PROXY_VirtualKeyModelMaxBudgetLimiter,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
creator: LiteLLM_UserTable,
|
||||
models: tuple[str, ...],
|
||||
expected: bool,
|
||||
):
|
||||
monkeypatch.setattr(module, "get_user_object", AsyncMock(return_value=creator))
|
||||
assert await principal_can_pay_for(await evaluation_principal("eval-admin"), models) is expected
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_creator_without_a_user_row_is_the_unbudgeted_master_key_admin(proxy, monkeypatch):
|
||||
monkeypatch.setattr(module, "get_user_object", AsyncMock(side_effect=UserNotFoundError(user_id="default_user_id")))
|
||||
|
||||
principal = await evaluation_principal("default_user_id")
|
||||
|
||||
assert (principal.user_id, principal.user_role) == ("default_user_id", LitellmUserRoles.PROXY_ADMIN)
|
||||
assert await principal_can_pay_for(principal, ("eval-judge",)) is True
|
||||
assert principal_call_metadata(principal)["user_api_key_user_id"] == "default_user_id"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_an_unreadable_creator_row_propagates_so_the_caller_withholds(proxy, monkeypatch):
|
||||
monkeypatch.setattr(module, "get_user_object", AsyncMock(side_effect=RuntimeError("db down")))
|
||||
|
||||
with pytest.raises(RuntimeError, match="db down"):
|
||||
await evaluation_principal("eval-admin")
|
||||
|
|
@ -38,9 +38,9 @@ const DIRECTION_OPTIONS: readonly { value: ShadowEvalDirection; label: string }[
|
|||
|
||||
const START_FORM_DESCRIPTION: Record<ShadowEvalDirection, string> = {
|
||||
forward:
|
||||
"Duplicates a sampled slice of the selected targets' traffic (keys, teams, or users) through the auto-router and has an LLM judge compare both answers blind. Each target gets its own spend budget. The router's answers are never served to users; judge calls bill to the sampled traffic's own identity.",
|
||||
"Duplicates a sampled slice of the selected targets' traffic (keys, teams, or users) through the auto-router and has an LLM judge compare both answers blind. Each target gets its own spend budget. The router's answers are never served to users; shadow and judge calls bill to you and count against your budgets.",
|
||||
reverse:
|
||||
"Duplicates a sampled slice of the traffic the auto-router already serves against a fixed baseline model and has an LLM judge compare both answers blind. Each target gets its own spend budget. The baseline's answers are never served to users; judge calls bill to the sampled traffic's own identity.",
|
||||
"Duplicates a sampled slice of the traffic the auto-router already serves against a fixed baseline model and has an LLM judge compare both answers blind. Each target gets its own spend budget. The baseline's answers are never served to users; shadow and judge calls bill to you and count against your budgets.",
|
||||
};
|
||||
|
||||
const DURATION_OPTIONS = [
|
||||
|
|
|
|||
5
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
5
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -1475,8 +1475,9 @@ export interface paths {
|
|||
* eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the
|
||||
* job's window ends, or the job is stopped, so one target running out of budget does
|
||||
* not end sampling for the others; sampling changes propagate to pods within about 10
|
||||
* seconds. Shadow and judge calls bill to the sampled request's own identity but are
|
||||
* excluded from request counts and auto-router adoption metrics.
|
||||
* seconds. Shadow and judge calls run as the admin who started the job: their spend is
|
||||
* that admin's, the admin's total and per-model budgets withhold samples once exhausted,
|
||||
* and they are excluded from request counts and auto-router adoption metrics.
|
||||
*/
|
||||
post: operations["start_shadow_eval_auto_router_shadow_eval_start_post"];
|
||||
delete?: never;
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue