This commit is contained in:
tin-berri 2026-10-04 11:52:41 -04:00 • committed by GitHub
commit a36693f63d
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
9 changed files with 363 additions and 115 deletions

View file

@ -30,7 +30,6 @@ from litellm.constants import INTERNAL_CALL_ORIGIN_METADATA_KEY
from litellm.integrations.custom_logger import CustomLogger
from litellm.integrations.websearch_interception.tools import is_web_search_tool_responses
from litellm.litellm_core_utils.core_helpers import get_litellm_metadata_from_kwargs, independent_snapshot
from litellm.litellm_core_utils.internal_call_metadata import sanitized_forwardable_call_metadata
from litellm.litellm_core_utils.llm_judge import (
default_router_provider,
extract_text_from_content,
@ -620,50 +619,30 @@ def _record_funnel_event(job_id: str, stage: "ShadowEvalFunnelStage") -> None:
verbose_logger.debug("shadow_eval: funnel increment failed for %s: %s", job_id, e)
async def _key_or_team_is_over_budget(metadata: Mapping[str, object]) -> bool:
"""Whether the shadowed key or its team is over budget, decided by the same owners
the request path uses, so counter keys and thresholds can never drift from auth's.
Advisory and fail-open: real traffic on an over-budget key is already rejected at
auth (so nothing reaches the success hook), and this gate only closes the race
where the key crosses its budget while a request is in flight.
"""
async def _admit_creator(creator_user_id: str | None, models: Sequence[str]) -> Mapping[str, object] | None:
"""The call metadata the job's creator pays under, or None when the creator cannot pay
for ``models`` or that cannot be verified, so the sample is withheld before any spend."""
if creator_user_id is None:
return None
try:
from litellm.exceptions import BudgetExceededError
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.auth.auth_checks import (
_team_max_budget_check,
_virtual_key_max_budget_check,
get_team_object,
from litellm.proxy.auth.evaluation_principal import (
evaluation_principal,
principal_call_metadata,
principal_can_pay_for,
)
from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache
except ImportError:
return False
auth: Final = metadata.get("user_api_key_auth")
if not isinstance(auth, UserAPIKeyAuth):
return False
try:
await _virtual_key_max_budget_check(valid_token=auth, proxy_logging_obj=proxy_logging_obj)
if auth.team_id:
team: Final = await get_team_object(
team_id=auth.team_id,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
check_cache_only=True,
)
await _team_max_budget_check(team_object=team, valid_token=auth, proxy_logging_obj=proxy_logging_obj)
except BudgetExceededError:
return True
except Exception as e: # noqa: BLE001 # advisory gate: a failed read must not block sampling
verbose_logger.debug("shadow_eval: budget read failed: %s", e)
return False
principal: Final = await evaluation_principal(creator_user_id)
if not await principal_can_pay_for(principal, models):
return None
return principal_call_metadata(principal)
except Exception as e: # noqa: BLE001 # an unverifiable creator budget withholds the sample rather than spending
verbose_logger.warning("shadow_eval: creator %s budget unverifiable, sample withheld: %s", creator_user_id, e)
return None
def _forwarded_team_id(metadata: Mapping[str, object]) -> str | None:
"""The shadowed key's team, the identity the judge call already carries in its metadata
and the router already selects deployments with. Read here too so the arm choice, which
happens before the router sees the call, is made under the same team."""
"""The shadowed key's team, the scope start-time validation resolved the judge under,
so the judge's router-or-SDK arm is chosen the way the job was validated."""
team_id: Final = metadata.get("user_api_key_team_id")
return team_id if isinstance(team_id, str) and team_id else None
@ -749,6 +728,7 @@ class ActiveShadowEvalJob(BaseModel):
judge_model: str
max_turns: int
max_budget: float | None = None
created_by: str | None = None
ends_at: datetime
attempts: int = 0
spend: float = 0.0
@ -833,6 +813,7 @@ class ShadowEvalLogger(CustomLogger):
job_spend_reader: Callable[[str, float, float], Awaitable[float]] | None = None,
job_spend_writer: Callable[[str, float], Awaitable[None]] | None = None,
funnel_recorder: Callable[[str, "ShadowEvalFunnelStage"], None] | None = None,
creator_admission: Callable[[str | None, Sequence[str]], Awaitable[Mapping[str, object] | None]] | None = None,
) -> None:
"""Providers are callables so the proxy's lazily-initialized globals are resolved
at call time, not at logger construction. The spend reader and writer wrap the
@ -843,6 +824,7 @@ class ShadowEvalLogger(CustomLogger):
self._read_job_spend = job_spend_reader or _job_spend_from_counter
self._write_job_spend = job_spend_writer or _add_job_spend_to_counter
self._record_funnel = funnel_recorder or _record_funnel_event
self._admit_creator = creator_admission or _admit_creator
self._inflight_shadow_tasks: int = 0
# Starts per job since the last cache fill, never decremented within a
# generation; the refill absorbs written rows and resets.
@ -1055,8 +1037,9 @@ class ShadowEvalLogger(CustomLogger):
) -> None:
"""Budget gates once per sampled request, then every router arm in turn: shadow
call -> blind judge -> one attempt row stamped with the arm. The gates that
decline to spend on an admitted sample (no DB to record into, an over-budget key,
an unverifiable or exhausted eval budget) count the REQUEST withheld before any
decline to spend on an admitted sample (no DB to record into, an unverifiable or
exhausted eval budget, a creator who cannot pay for every model the sample would
call) count the REQUEST withheld before any
arm runs, so funnel counters stay per-request and a leg's eligible traffic still
reconciles as not_sampled + unjudgeable + shed + withheld + sampled requests,
where each sampled request writes one attempt row per arm. A budget crossed
@ -1069,9 +1052,6 @@ class ShadowEvalLogger(CustomLogger):
if prisma is None:
self._record_funnel(job.id, "withheld")
return
if await _key_or_team_is_over_budget(parent_metadata):
self._record_funnel(job.id, "withheld")
return
if job.max_budget is not None:
try:
spend: Final = await self._read_job_spend(_job_spend_counter_key(job.id), job.spend, job.max_budget)
@ -1082,6 +1062,12 @@ class ShadowEvalLogger(CustomLogger):
if spend >= job.max_budget:
self._record_funnel(job.id, "withheld")
return
creator_metadata: Final = await self._admit_creator(
job.created_by, (*(job.arm_target(arm) for arm in job.arm_router_names), job.judge_model)
)
if creator_metadata is None:
self._record_funnel(job.id, "withheld")
return
for arm_router in job.arm_router_names:
await self._run_shadow_arm(
prisma=prisma,
@ -1097,6 +1083,7 @@ class ShadowEvalLogger(CustomLogger):
control_tier=control_tier,
shadow_params=shadow_params,
parent_metadata=parent_metadata,
creator_metadata=creator_metadata,
)
async def _run_shadow_arm(
@ -1114,12 +1101,13 @@ class ShadowEvalLogger(CustomLogger):
control_tier: str | None,
shadow_params: Mapping[str, object],
parent_metadata: Mapping[str, object],
creator_metadata: Mapping[str, object],
) -> None:
"""One arm's pipeline: shadow call -> blind judge -> one attempt row, every exit
recording this arm's outcome, so one arm's fault never silences a sibling arm."""
try:
shadow: Final = await self._call_router_shadow(
job.arm_target(arm_router), messages, shadow_params, parent_metadata
job.arm_target(arm_router), messages, shadow_params, creator_metadata
)
except Exception as e: # noqa: BLE001 # detached task: nothing billed yet, record and never raise
verbose_logger.debug("shadow_eval: pipeline failed for %s: %s", request_id, e)
@ -1161,6 +1149,7 @@ class ShadowEvalLogger(CustomLogger):
shadow_text=shadow.text,
tools=shadow_params.get("tools"),
parent_metadata=parent_metadata,
creator_metadata=creator_metadata,
)
if isinstance(verdict, _CallFailure):
await self._record_attempt(
@ -1268,18 +1257,19 @@ class ShadowEvalLogger(CustomLogger):
target_model: str,
messages: Sequence[Mapping[str, object]],
shadow_params: Mapping[str, object],
parent_metadata: Mapping[str, object],
creator_metadata: Mapping[str, object],
) -> "_ShadowResponse | _CallFailure":
"""Send the prompt through the arm nobody was served: the auto-router under
evaluation, or a reverse job's fixed baseline model. The metadata carries the
shadowed key's identity (spend attribution) and receives a routing decision
write-back, which a plain baseline model simply never makes."""
evaluation, or a reverse job's fixed baseline model. The call runs as the job's
creator and receives a routing decision write-back, which a plain baseline model
simply never makes."""
router: Final = self._router_provider()
if router is None:
return _CallFailure("no router configured on this pod")
shadow_metadata: Final[dict[str, object]] = ( # mutable-ok: router writes its routing decision back
sanitized_forwardable_call_metadata(parent_metadata, SHADOW_EVAL_ROUTER_CALL_ORIGIN)
)
shadow_metadata: Final[dict[str, object]] = { # mutable-ok: router writes its routing decision back
**creator_metadata,
INTERNAL_CALL_ORIGIN_METADATA_KEY: SHADOW_EVAL_ROUTER_CALL_ORIGIN,
}
try:
response: Final = await router.acompletion(
model=target_model,
@ -1321,6 +1311,7 @@ class ShadowEvalLogger(CustomLogger):
shadow_text: str,
tools: object,
parent_metadata: Mapping[str, object],
creator_metadata: Mapping[str, object],
) -> "_JudgeVerdict | _CallFailure":
"""Blind pairwise judge with A/B labels randomized to cancel position bias. Both
arms were offered the same tools, so the judge is shown their definitions too: a
@ -1334,7 +1325,7 @@ class ShadowEvalLogger(CustomLogger):
for m in messages
if m.get("content") is not None
)
judge_metadata: Final = sanitized_forwardable_call_metadata(parent_metadata, SHADOW_EVAL_JUDGE_CALL_ORIGIN)
judge_metadata: Final = {**creator_metadata, INTERNAL_CALL_ORIGIN_METADATA_KEY: SHADOW_EVAL_JUDGE_CALL_ORIGIN}
judge_messages: Final = [
{"role": "system", "content": PAIRWISE_JUDGE_SYSTEM_PROMPT},
{

View file

@ -1,9 +1,9 @@
"""Metadata a request forwards to the internal LLM sub-calls it triggers.
Internal features (the auto-router's classifier and embeddings, shadow eval's shadow and
judge calls) bill real provider spend that nobody typed a prompt for. That spend must land
on the same key/team/org/user as the request that caused it, so the sub-call carries the
caller's identity metadata, minus two things that must never be forwarded as-is:
Internal features (the auto-router's classifier, embeddings and context compaction) bill
real provider spend that nobody typed a prompt for. That spend must land on the same
key/team/org/user as the request that caused it, so the sub-call carries the caller's
identity metadata, minus two things that must never be forwarded as-is:
* ``user_api_key_budget_reservation`` (and the reservation nested inside
``user_api_key_auth``) belongs to the parent completion. If a sub-call's cost callback
@ -161,8 +161,8 @@ def sanitized_forwardable_call_metadata(
) -> dict[str, object]: # mutable-ok: SDK metadata kwarg
"""Just the caller's identity, stamped with the sub-call's origin.
For sub-calls detached from the parent request (shadow eval), which outlive it and
must not inherit per-request state such as its routing decision or logging payload.
For sub-calls that must not inherit the parent's per-request state, such as its
routing decision or logging payload.
"""
identity: Final = {k: v for k, v in parent_metadata.items() if k in FORWARDABLE_IDENTITY_METADATA_KEYS}
return _sanitized(identity) | {INTERNAL_CALL_ORIGIN_METADATA_KEY: call_origin}

View file

@ -0,0 +1,88 @@
"""The principal a shadow evaluation's own LLM calls run as: the admin who created the job.
Evaluation calls are spend the creator chose to incur, so they are attributed, routed and
budget-checked as that admin's own requests, through the same owners an authenticated
request uses. Nothing the sampled request carried decides who pays for them.
"""
from collections.abc import Mapping, Sequence
from types import MappingProxyType
from typing import Final
from pydantic import TypeAdapter
from litellm._logging import verbose_proxy_logger
from litellm.exceptions import BudgetExceededError
from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy.auth.auth_checks import effective_user_role, get_user_object
from litellm.proxy.auth.fallback_budget import is_token_within_budget_for_model
from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup
from litellm.types.proxy.auth.auth_checks import UserNotFoundError
_MODEL_BUDGETS: Final[TypeAdapter[Mapping[str, object] | None]] = TypeAdapter(Mapping[str, object] | None)
async def evaluation_principal(creator_user_id: str) -> UserAPIKeyAuth:
"""The creator as a principal carrying their current budgets.
A creator with no user row is the master key's admin, which has no personal budget to
enforce: the spend is still attributed to that id. Any other read failure propagates, so
the caller withholds work it cannot verify the creator can pay for.
"""
from litellm.proxy.proxy_server import prisma_client, proxy_logging_obj, user_api_key_cache
try:
user: Final = await get_user_object(
user_id=creator_user_id,
prisma_client=prisma_client,
user_api_key_cache=user_api_key_cache,
user_id_upsert=False,
proxy_logging_obj=proxy_logging_obj,
)
except UserNotFoundError:
return UserAPIKeyAuth(user_id=creator_user_id, user_role=LitellmUserRoles.PROXY_ADMIN)
if user is None:
return UserAPIKeyAuth(user_id=creator_user_id, user_role=LitellmUserRoles.PROXY_ADMIN)
return UserAPIKeyAuth(
user_id=user.user_id,
user_role=effective_user_role(user.user_role),
user_email=user.user_email,
user_spend=user.spend,
user_max_budget=user.max_budget,
user_model_max_budget=_MODEL_BUDGETS.validate_python(user.model_max_budget),
)
async def principal_can_pay_for(principal: UserAPIKeyAuth, models: Sequence[str]) -> bool:
"""Whether the principal is within its total budget and every per-model budget for
``models``, read through the owners the request path enforces them with."""
from litellm.proxy.proxy_server import llm_router, model_max_budget_limiter
if llm_router is None:
return False
try:
for model in models:
if not await is_token_within_budget_for_model(model=model, valid_token=principal, llm_router=llm_router):
return False
if principal.user_id is not None and principal.user_model_max_budget:
await model_max_budget_limiter.is_user_within_model_budget(
user_id=principal.user_id, user_model_max_budget=principal.user_model_max_budget, model=model
)
except BudgetExceededError as e:
verbose_proxy_logger.debug("shadow_eval: creator %s is over budget: %s", principal.user_id, e)
return False
return True
def principal_call_metadata(principal: UserAPIKeyAuth) -> Mapping[str, object]:
"""Request metadata naming ``principal`` exactly as auth stamps it on a request the
principal sent, so every spend writer, limiter and router filter reads one identity."""
stamped: Final[dict[str, object]] = {} # mutable-ok: the stampers write into a request dict
request: Final = {"metadata": stamped}
LiteLLMProxyRequestSetup.add_user_api_key_auth_to_request_metadata(
data=request, user_api_key_dict=principal, _metadata_variable_name="metadata"
)
LiteLLMProxyRequestSetup.add_budget_metadata_to_request_metadata(
data=request, user_api_key_dict=principal, _metadata_variable_name="metadata"
)
return MappingProxyType(stamped)

View file

@ -1650,6 +1650,27 @@ class LiteLLMProxyRequestSetup:
)
return user_api_key_logged_metadata
@staticmethod
def add_budget_metadata_to_request_metadata(
data: dict, user_api_key_dict: UserAPIKeyAuth, _metadata_variable_name: str
) -> None:
"""The principal's budgets and spend: prometheus reports them and the per-model budget
limiter reads the ``*_model_max_budget`` entries to charge spend."""
data[_metadata_variable_name]["user_api_key_team_max_budget"] = user_api_key_dict.team_max_budget
data[_metadata_variable_name]["user_api_key_team_spend"] = user_api_key_dict.team_spend
data[_metadata_variable_name]["user_api_key_team_model_max_budget"] = user_api_key_dict.team_model_max_budget
data[_metadata_variable_name]["user_api_key_request_route"] = user_api_key_dict.request_route
data[_metadata_variable_name]["user_api_key_spend"] = user_api_key_dict.spend
data[_metadata_variable_name]["user_api_key_max_budget"] = user_api_key_dict.max_budget
data[_metadata_variable_name]["user_api_key_model_max_budget"] = user_api_key_dict.model_max_budget
data[_metadata_variable_name]["user_api_key_end_user_model_max_budget"] = (
user_api_key_dict.end_user_model_max_budget
)
data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend
data[_metadata_variable_name]["user_api_key_user_max_budget"] = user_api_key_dict.user_max_budget
data[_metadata_variable_name]["user_api_key_user_model_max_budget"] = user_api_key_dict.user_model_max_budget
data[_metadata_variable_name].update(carried_budget_metadata(user_api_key_dict))
@staticmethod
def add_user_api_key_auth_to_request_metadata(
data: dict,
@ -2405,28 +2426,10 @@ async def add_litellm_data_to_request(
_metadata_variable_name=_metadata_variable_name,
)
# Team spend, budget - used by prometheus.py
data[_metadata_variable_name]["user_api_key_team_max_budget"] = user_api_key_dict.team_max_budget
data[_metadata_variable_name]["user_api_key_team_spend"] = user_api_key_dict.team_spend
data[_metadata_variable_name]["user_api_key_team_model_max_budget"] = user_api_key_dict.team_model_max_budget
data[_metadata_variable_name]["user_api_key_request_route"] = user_api_key_dict.request_route
# API Key spend, budget - used by prometheus.py
data[_metadata_variable_name]["user_api_key_spend"] = user_api_key_dict.spend
data[_metadata_variable_name]["user_api_key_max_budget"] = user_api_key_dict.max_budget
data[_metadata_variable_name]["user_api_key_model_max_budget"] = user_api_key_dict.model_max_budget
data[_metadata_variable_name]["user_api_key_end_user_model_max_budget"] = (
user_api_key_dict.end_user_model_max_budget
LiteLLMProxyRequestSetup.add_budget_metadata_to_request_metadata(
data=data, user_api_key_dict=user_api_key_dict, _metadata_variable_name=_metadata_variable_name
)
# User spend, budget - used by prometheus.py
# Follow same pattern as team and API key budgets
data[_metadata_variable_name]["user_api_key_user_spend"] = user_api_key_dict.user_spend
data[_metadata_variable_name]["user_api_key_user_max_budget"] = user_api_key_dict.user_max_budget
user_model_budget: Final = user_api_key_dict.user_model_max_budget
data[_metadata_variable_name]["user_api_key_user_model_max_budget"] = user_model_budget # rebind-ok: out-param
data[_metadata_variable_name].update(carried_budget_metadata(user_api_key_dict))
data[_metadata_variable_name]["user_api_key_metadata"] = strip_callback_config(user_api_key_dict.metadata)
data[_metadata_variable_name]["user_api_key_team_metadata"] = strip_callback_config(user_api_key_dict.team_metadata)
data[_metadata_variable_name]["user_api_key_object_permission_id"] = getattr(

View file

@ -1684,8 +1684,9 @@ async def start_shadow_eval(
eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the
job's window ends, or the job is stopped, so one target running out of budget does
not end sampling for the others; sampling changes propagate to pods within about 10
seconds. Shadow and judge calls bill to the sampled request's own identity but are
excluded from request counts and auto-router adoption metrics.
seconds. Shadow and judge calls run as the admin who started the job: their spend is
that admin's, the admin's total and per-model budgets withhold samples once exhausted,
and they are excluded from request counts and auto-router adoption metrics.
"""
from litellm.proxy.proxy_server import llm_router, prisma_client

View file

@ -2,8 +2,9 @@
the detached pipeline's single attempt-row write, and the cache-first job lookup."""
import asyncio
from collections.abc import Mapping
from collections.abc import Mapping, Sequence
from datetime import datetime, timedelta, timezone
from types import MappingProxyType
from typing import Final, Literal
from unittest.mock import AsyncMock, MagicMock
@ -55,6 +56,7 @@ def _job(**overrides) -> ActiveShadowEvalJob:
max_turns=200,
ends_at=datetime.now(timezone.utc) + timedelta(days=1),
attempts=0,
created_by="eval-admin",
)
return ActiveShadowEvalJob(**{**defaults, **overrides})
@ -92,6 +94,7 @@ def _job_record(job: ActiveShadowEvalJob, target_type="key", target_id="key-hash
judge_model=job.judge_model,
max_turns=job.max_turns,
max_budget=job.max_budget,
created_by=job.created_by,
ends_at=job.ends_at,
).items():
setattr(record, field, value)
@ -230,10 +233,28 @@ def _spend_counter(store=None):
return counter, read, write
def _logger(router=None, prisma=None, jobs=(), counter_store=None, jobs_by_target=None) -> ShadowEvalLogger:
CREATOR_METADATA: Final = MappingProxyType({"user_api_key_user_id": "eval-admin", "user_api_key_hash": None})
def _creator_admission(admitted: bool = True):
"""Stands in for the proxy's creator budget owners: answers with the creator's call
metadata, or None for a creator who cannot pay, and records what it was asked."""
asked: Final[list[tuple[str | None, tuple[str, ...]]]] = []
async def admit(creator_user_id: str | None, models: Sequence[str]) -> Mapping[str, object] | None:
asked.append((creator_user_id, tuple(models)))
return CREATOR_METADATA if admitted else None
return admit, asked
def _logger(
router=None, prisma=None, jobs=(), counter_store=None, jobs_by_target=None, creator_admitted=True
) -> ShadowEvalLogger:
cache = InMemoryCache(max_size_in_memory=4, default_ttl=60)
counter, read, write = _spend_counter(counter_store)
funnel_events = []
admit, asked = _creator_admission(creator_admitted)
logger = ShadowEvalLogger(
router_provider=lambda: router,
prisma_provider=lambda: prisma,
@ -241,9 +262,11 @@ def _logger(router=None, prisma=None, jobs=(), counter_store=None, jobs_by_targe
job_spend_reader=read,
job_spend_writer=write,
funnel_recorder=lambda job_id, stage: funnel_events.append((job_id, stage)),
creator_admission=admit,
)
logger._test_counter = counter
logger._test_funnel = funnel_events
logger._test_admission_asks = asked
seeded = jobs_by_target if jobs_by_target is not None else ({("key", "key-hash"): tuple(jobs)} if jobs else None)
if seeded is not None:
cache.set_cache("shadow_eval:active_jobs", seeded)
@ -1225,24 +1248,35 @@ class TestSuccessHookSkipChain:
assert prisma.db.litellm_shadowevalattempt.create.await_count == 1
async def test_v1_messages_surface_forwards_identity_from_litellm_metadata(self):
"""/v1/messages stores identity in litellm_params.litellm_metadata, so the hook
resolves the bucket through the shared helper; every surface forwards the same
identity to the shadow and judge calls."""
@pytest.mark.parametrize(
"call_type,bucket", [("acompletion", "metadata"), ("anthropic_messages", "litellm_metadata")]
)
async def test_eval_calls_run_as_the_creator_whatever_the_sampled_caller_was(self, call_type: str, bucket: str):
"""Every surface's sampled request carries its caller's identity in a different bucket;
none of it reaches the shadow or judge call, which bill to the job's creator."""
prisma = _prisma()
router = _router()
logger = _logger(router=router, prisma=prisma, jobs=(_job(),))
hook_kwargs = _success_kwargs()
hook_kwargs = _success_kwargs(call_type=call_type)
hook_kwargs["litellm_params"] = {
"litellm_metadata": {"user_api_key_hash": "key-hash", "user_api_key_team_id": "team-1"}
bucket: {
"user_api_key_hash": "key-hash",
"user_api_key_team_id": "team-1",
"user_api_key_user_id": "sampled-user",
"user_api_key_end_user_id": "sampled-end-user",
},
"proxy_server_request": {"body": {"messages": [{"role": "user", "content": "hi"}], "max_tokens": 9}},
}
await logger.async_log_success_event(hook_kwargs, RESPONSE, None, None)
await _drain(logger)
shadow_call = router.acompletion.call_args_list[0].kwargs
assert shadow_call["metadata"]["user_api_key_hash"] == "key-hash"
assert shadow_call["metadata"]["user_api_key_team_id"] == "team-1"
assert logger._test_admission_asks == [("eval-admin", ("my-router", "judge-model"))]
assert router.acompletion.call_count == 2
for call in router.acompletion.call_args_list:
metadata = call.kwargs["metadata"]
assert metadata["user_api_key_user_id"] == "eval-admin"
assert {k: v for k, v in metadata.items() if v in ("key-hash", "team-1", "sampled-user", "sampled-end-user")} == {}
async def test_redacted_requests_are_never_shadowed(self):
"""Redaction rewrites the logged messages before callbacks run, so this hook only
@ -1513,24 +1547,23 @@ class TestShadowPipeline:
router.acompletion.assert_not_called()
assert logger._test_funnel == [("job-1", "withheld")]
async def test_over_budget_key_skips_before_any_call(self, monkeypatch: pytest.MonkeyPatch):
"""The gate delegates to the auth path's own budget owner, so an over-budget
verdict there (BudgetExceededError) skips the shadow before any provider call."""
from litellm.exceptions import BudgetExceededError
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.auth import auth_checks
monkeypatch.setattr(
auth_checks,
"_virtual_key_max_budget_check",
AsyncMock(side_effect=BudgetExceededError(current_cost=11.0, max_budget=10.0)),
)
@pytest.mark.parametrize(
"job",
[
_job(),
_job(router_names=("my-router", "router-b")),
_job(direction="reverse", baseline_model="baseline-model"),
],
)
async def test_a_creator_who_cannot_pay_withholds_the_sample_before_any_call(self, job: ActiveShadowEvalJob):
"""Admission is asked once per sample about every model the sample would call, each
arm's target plus the judge, and a creator over any of those budgets spends nothing."""
router = _router()
prisma = _prisma()
logger = _logger(router=router, prisma=prisma)
logger = _logger(router=router, prisma=prisma, creator_admitted=False)
await logger._run_shadow_eval(
job=_job(),
job=job,
request_id="req-1",
messages=({"role": "user", "content": "hi"},),
real_text="real answer",
@ -1540,12 +1573,15 @@ class TestShadowPipeline:
real_cache_hit=False,
control_tier=None,
shadow_params={},
parent_metadata={"user_api_key_auth": UserAPIKeyAuth(api_key="sk-abc", max_budget=10.0)},
parent_metadata={"user_api_key_hash": "key-hash"},
)
assert logger._test_admission_asks == [
("eval-admin", (*(job.arm_target(arm) for arm in job.arm_router_names), "judge-model"))
]
router.acompletion.assert_not_called()
prisma.db.litellm_shadowevalattempt.create.assert_not_called()
assert logger._test_funnel == [("job-1", "withheld")]
assert logger._test_funnel == [(job.id, "withheld")]
@pytest.mark.parametrize(
"router_factory,expected_error,expected_cost,expected_shadow_cost",
@ -1965,7 +2001,7 @@ class TestShadowPipeline:
assert row["judge_cost"] == 0.0
assert logger._test_counter["spend:shadow_eval:job-1"] == 0.007
async def test_sub_calls_carry_identity_and_origin_but_never_parent_request_state(self):
async def test_sub_calls_carry_the_creator_and_origin_but_never_parent_request_state(self):
prisma = _prisma()
router = _router()
logger = _logger(router=router, prisma=prisma)
@ -1995,8 +2031,9 @@ class TestShadowPipeline:
for call in (shadow_call, judge_call):
assert call["num_retries"] == 0
assert call["fallbacks"] == []
assert call["metadata"]["user_api_key_hash"] == "key-hash"
assert call["metadata"]["user_api_key_team_id"] == "team-1"
assert call["metadata"]["user_api_key_user_id"] == "eval-admin"
assert call["metadata"]["user_api_key_hash"] is None
assert "user_api_key_team_id" not in call["metadata"]
assert "user_api_key_budget_reservation" not in call["metadata"]
assert shadow_call["metadata"][INTERNAL_CALL_ORIGIN_METADATA_KEY] == SHADOW_EVAL_ROUTER_CALL_ORIGIN
assert judge_call["metadata"][INTERNAL_CALL_ORIGIN_METADATA_KEY] == SHADOW_EVAL_JUDGE_CALL_ORIGIN

View file

@ -0,0 +1,127 @@
"""The shadow-eval creator principal: who it is, what its stamped calls charge, and
whether the request path's own budget owners let it pay."""
from typing import Final
from unittest.mock import AsyncMock
import pytest
from litellm import Router
from litellm.caching.dual_cache import DualCache
from litellm.proxy import proxy_server
from litellm.proxy._types import LitellmUserRoles
from litellm.proxy.auth import evaluation_principal as module
from litellm.proxy.auth.evaluation_principal import (
evaluation_principal,
principal_call_metadata,
principal_can_pay_for,
)
from litellm.proxy.hooks.model_max_budget_limiter import _PROXY_VirtualKeyModelMaxBudgetLimiter
from litellm.models.user import LiteLLM_UserTable
from litellm.types.proxy.auth.auth_checks import UserNotFoundError
PAID_MODEL: Final = {
"model_name": "eval-judge",
"litellm_params": {
"model": "openai/gpt-4o",
"api_key": "k",
"input_cost_per_token": 0.000001,
"output_cost_per_token": 0.000002,
},
"model_info": {"id": "eval-judge-id"},
}
def _creator(**overrides) -> LiteLLM_UserTable:
fields = {"user_id": "eval-admin", "user_role": "proxy_admin", "spend": 0.0, "max_budget": 10.0}
return LiteLLM_UserTable(**{**fields, **overrides})
@pytest.fixture
def proxy(monkeypatch: pytest.MonkeyPatch) -> _PROXY_VirtualKeyModelMaxBudgetLimiter:
limiter: Final = _PROXY_VirtualKeyModelMaxBudgetLimiter(dual_cache=DualCache())
async def counter_spend(counter_key: str, fallback_spend: float, max_budget: float | None = None) -> float:
return fallback_spend
monkeypatch.setattr(proxy_server, "llm_router", Router(model_list=[PAID_MODEL]), raising=False)
monkeypatch.setattr(proxy_server, "model_max_budget_limiter", limiter, raising=False)
monkeypatch.setattr(proxy_server, "get_current_spend", counter_spend, raising=False)
monkeypatch.setattr(proxy_server, "general_settings", {}, raising=False)
monkeypatch.setattr(proxy_server, "prisma_client", object(), raising=False)
return limiter
@pytest.mark.asyncio
async def test_spend_charged_under_the_stamp_exhausts_the_budget_the_gate_reads(proxy, monkeypatch):
"""The workflow end to end: an evaluation call stamped as the creator is charged by the
real per-model limiter, and the next admission reads that same counter and refuses."""
monkeypatch.setattr(
module,
"get_user_object",
AsyncMock(
return_value=_creator(model_max_budget={"eval-judge": {"budget_limit": 0.001, "time_period": "1d"}})
),
)
principal = await evaluation_principal("eval-admin")
assert await principal_can_pay_for(principal, ("eval-judge",)) is True
stamped = principal_call_metadata(principal)
await proxy.async_log_success_event(
{
"standard_logging_object": {
"call_type": "acompletion",
"response_cost": 0.002,
"model": "openai/gpt-4o",
"model_group": "eval-judge",
"metadata": {k: stamped.get(k) for k in ("user_api_key_hash", "user_api_key_user_id")},
},
"litellm_params": {"metadata": dict(stamped)},
},
response_obj=None,
start_time=None,
end_time=None,
)
assert stamped["user_api_key_user_id"] == "eval-admin"
assert await principal_can_pay_for(principal, ("eval-judge",)) is False
@pytest.mark.asyncio
@pytest.mark.parametrize(
"creator,models,expected",
[
(_creator(spend=10.0), ("eval-judge",), False),
(_creator(spend=10.0), (), True),
(_creator(spend=9.0), ("eval-judge",), True),
(_creator(spend=999.0, max_budget=None), ("eval-judge",), True),
],
)
async def test_total_budget_gates_every_paid_model(
proxy: _PROXY_VirtualKeyModelMaxBudgetLimiter,
monkeypatch: pytest.MonkeyPatch,
creator: LiteLLM_UserTable,
models: tuple[str, ...],
expected: bool,
):
monkeypatch.setattr(module, "get_user_object", AsyncMock(return_value=creator))
assert await principal_can_pay_for(await evaluation_principal("eval-admin"), models) is expected
@pytest.mark.asyncio
async def test_a_creator_without_a_user_row_is_the_unbudgeted_master_key_admin(proxy, monkeypatch):
monkeypatch.setattr(module, "get_user_object", AsyncMock(side_effect=UserNotFoundError(user_id="default_user_id")))
principal = await evaluation_principal("default_user_id")
assert (principal.user_id, principal.user_role) == ("default_user_id", LitellmUserRoles.PROXY_ADMIN)
assert await principal_can_pay_for(principal, ("eval-judge",)) is True
assert principal_call_metadata(principal)["user_api_key_user_id"] == "default_user_id"
@pytest.mark.asyncio
async def test_an_unreadable_creator_row_propagates_so_the_caller_withholds(proxy, monkeypatch):
monkeypatch.setattr(module, "get_user_object", AsyncMock(side_effect=RuntimeError("db down")))
with pytest.raises(RuntimeError, match="db down"):
await evaluation_principal("eval-admin")

View file

@ -38,9 +38,9 @@ const DIRECTION_OPTIONS: readonly { value: ShadowEvalDirection; label: string }[
const START_FORM_DESCRIPTION: Record<ShadowEvalDirection, string> = {
forward:
"Duplicates a sampled slice of the selected targets' traffic (keys, teams, or users) through the auto-router and has an LLM judge compare both answers blind. Each target gets its own spend budget. The router's answers are never served to users; judge calls bill to the sampled traffic's own identity.",
"Duplicates a sampled slice of the selected targets' traffic (keys, teams, or users) through the auto-router and has an LLM judge compare both answers blind. Each target gets its own spend budget. The router's answers are never served to users; shadow and judge calls bill to you and count against your budgets.",
reverse:
"Duplicates a sampled slice of the traffic the auto-router already serves against a fixed baseline model and has an LLM judge compare both answers blind. Each target gets its own spend budget. The baseline's answers are never served to users; judge calls bill to the sampled traffic's own identity.",
"Duplicates a sampled slice of the traffic the auto-router already serves against a fixed baseline model and has an LLM judge compare both answers blind. Each target gets its own spend budget. The baseline's answers are never served to users; shadow and judge calls bill to you and count against your budgets.",
};
const DURATION_OPTIONS = [

View file

@ -1475,8 +1475,9 @@ export interface paths {
* eval spend, the shadow and judge calls' own cost, reaches max_budget dollars, the
* job's window ends, or the job is stopped, so one target running out of budget does
* not end sampling for the others; sampling changes propagate to pods within about 10
* seconds. Shadow and judge calls bill to the sampled request's own identity but are
* excluded from request counts and auto-router adoption metrics.
* seconds. Shadow and judge calls run as the admin who started the job: their spend is
* that admin's, the admin's total and per-model budgets withhold samples once exhausted,
* and they are excluded from request counts and auto-router adoption metrics.
*/
post: operations["start_shadow_eval_auto_router_shadow_eval_start_post"];
delete?: never;