mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
feat(shadow_eval): scope a job to model groups, ANDed with its key, team, and user targets (#39828)
A shadow eval job could only be scoped by identity, so "this user's traffic on model X across every key they own" was not expressible and a models field on the start body was silently dropped. The job now carries a models list that every target is narrowed to, matched on the requested model group with model_group_alias resolved on both sides. An unresolvable name is a 400 at start. Empty means every model, which is what every existing row reads as. The dashboard start form gains an "Only on models" picker and the job headline shows the scope.
This commit is contained in:
parent
e7dd524a3c
commit
8b6ea72845
13 changed files with 343 additions and 9 deletions
|
|
@ -0,0 +1 @@
|
|||
ALTER TABLE "LiteLLM_ShadowEvalJob" ADD COLUMN IF NOT EXISTS "models" TEXT[] NOT NULL DEFAULT ARRAY[]::TEXT[];
|
||||
|
|
@ -1536,6 +1536,7 @@ model LiteLLM_ShadowEvalJob {
|
|||
target_id String // hashed virtual key, team_id, or user_id whose traffic this leg shadows
|
||||
router_name String // first (often only) auto-router under evaluation; router_names is the full set
|
||||
router_names String[] @default([]) // all routers this job runs as shadow arms; empty on legacy rows, whose set is (router_name)
|
||||
models String[] @default([]) // model groups the sampled traffic is narrowed to; empty samples every model
|
||||
direction String @default("forward") // forward | reverse
|
||||
baseline_model String? // reverse only: the fixed model the router is judged against
|
||||
judge_model String
|
||||
|
|
|
|||
|
|
@ -37,6 +37,7 @@ from litellm.litellm_core_utils.llm_judge import (
|
|||
)
|
||||
from litellm.litellm_core_utils.redact_messages import should_redact_message_logging
|
||||
from litellm.llms.base_llm.base_utils import type_to_response_format_param
|
||||
from litellm.router_utils.common_utils import resolve_model_group_alias
|
||||
from litellm.types.management_endpoints.auto_router_endpoints import ShadowEvalDirection
|
||||
from litellm.types.utils import SHADOW_EVAL_JUDGE_CALL_ORIGIN, SHADOW_EVAL_ROUTER_CALL_ORIGIN
|
||||
|
||||
|
|
@ -650,6 +651,7 @@ class ActiveShadowEvalJob(BaseModel):
|
|||
id: str
|
||||
router_name: str
|
||||
router_names: tuple[str, ...] = ()
|
||||
models: frozenset[str] = frozenset()
|
||||
direction: ShadowEvalDirection = "forward"
|
||||
baseline_model: str | None = None
|
||||
shadow_percentage: float
|
||||
|
|
@ -692,6 +694,21 @@ class ActiveShadowEvalJob(BaseModel):
|
|||
return self.baseline_model or arm_router
|
||||
|
||||
|
||||
def _canonical_group(router: "Router | None", model_group: str) -> str:
|
||||
"""A model group in the one spelling both a job's scope and a request's model compare
|
||||
under: an alias resolves to its target so the two never fail to match on spelling."""
|
||||
return (
|
||||
resolve_model_group_alias(router.model_group_alias, model_group) if router is not None else None
|
||||
) or model_group
|
||||
|
||||
|
||||
def _scope_admits(router: "Router | None", job: "ActiveShadowEvalJob", model_group: str) -> bool:
|
||||
"""Whether the request's group is in the job's model scope. Both sides resolve through
|
||||
the router's alias map at match time, so a re-pointed alias applies to the next request
|
||||
rather than after the jobs cache rolls."""
|
||||
return not job.models or any(_canonical_group(router, name) == model_group for name in job.models)
|
||||
|
||||
|
||||
def _as_active_job(record: object, attempts: int, spend: float) -> ActiveShadowEvalJob | None:
|
||||
"""The sampling path's view of one job row, or None for a row it cannot sample: an
|
||||
unknown direction, or a reverse job with no baseline model to duplicate against.
|
||||
|
|
@ -714,7 +731,8 @@ class ShadowEvalLogger(CustomLogger):
|
|||
A job targets a virtual key, a team, or a user; a request qualifies for a job when
|
||||
any of its resolved identities (key hash, team id, user id) matches the job's
|
||||
target, so team and user jobs cover JWT-authenticated traffic, which carries no
|
||||
key hash at all."""
|
||||
key hash at all. A job scoped to model groups further requires the request's
|
||||
requested group to be one of them."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
|
|
@ -801,19 +819,24 @@ class ShadowEvalLogger(CustomLogger):
|
|||
active_jobs: Sequence[ActiveShadowEvalJob],
|
||||
request_metadata: Mapping[str, object],
|
||||
request_id: str,
|
||||
model_group: str,
|
||||
) -> tuple[ActiveShadowEvalJob, ...]:
|
||||
"""The jobs that sample this request. A key can hold one job per direction, and a
|
||||
request routed by one job's router while bypassing the other's qualifies for both;
|
||||
each is separately budgeted, so both fire. An admitting job that loses the sampling
|
||||
dice is counted, so results can weigh judged rows against the traffic they stand for."""
|
||||
dice is counted, so results can weigh judged rows against the traffic they stand for.
|
||||
A request outside a job's direction or model scope is not that job's traffic and
|
||||
goes uncounted, so the funnel stays a fraction of the traffic the job admits."""
|
||||
eligible: list[ActiveShadowEvalJob] = [] # mutable-ok: bucketed per-job admission
|
||||
now: Final = datetime.now(timezone.utc)
|
||||
router: Final = self._router_provider()
|
||||
for job in active_jobs:
|
||||
if (
|
||||
now >= job.ends_at
|
||||
or job.attempts + self._job_starts.get(job.id, 0) >= job.max_turns
|
||||
or (job.max_budget is not None and job.spend >= job.max_budget)
|
||||
or not _direction_admits(request_metadata, job)
|
||||
or not _scope_admits(router, job, model_group)
|
||||
):
|
||||
continue
|
||||
if not _sample_hits(request_id, job.id, job.shadow_percentage):
|
||||
|
|
@ -868,6 +891,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
tuple(job for target in targets for job in active_jobs.get(target, ())),
|
||||
request_metadata,
|
||||
request_id,
|
||||
_canonical_group(self._router_provider(), str(payload.get("model_group") or "")),
|
||||
)
|
||||
if not eligible:
|
||||
return
|
||||
|
|
|
|||
|
|
@ -789,6 +789,26 @@ def _for_teams(team_ids: Sequence[str | None]) -> str:
|
|||
return f" for team {', '.join(named)}" if named else ""
|
||||
|
||||
|
||||
def _validate_model_scope(llm_router: "Router | None", models: Sequence[str]) -> None:
|
||||
"""Reject a scope naming a model no request on this proxy could carry, at start rather
|
||||
than as a job that silently samples nothing. The question is "could any caller ask for
|
||||
this name", not "does it resolve for the job's teams": a user target's traffic can arrive
|
||||
on any team's key, so a team-public name is a legitimate scope for it, and an auto-router
|
||||
is one too (a forward job on router A scoped to router B samples what B serves today).
|
||||
Nothing here is ever dispatched to."""
|
||||
unreachable: Final = tuple(
|
||||
model
|
||||
for model in models
|
||||
if judge_target(llm_router, model).via == "nothing"
|
||||
and (llm_router is None or model not in llm_router.team_public_model_names)
|
||||
)
|
||||
if unreachable:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="models not served by this proxy: " + ", ".join(f"'{model}'" for model in unreachable),
|
||||
)
|
||||
|
||||
|
||||
_JUDGED_ROLES: Final[frozenset[StrategyRouterDependencyRole]] = frozenset({"tier", "default"})
|
||||
|
||||
|
||||
|
|
@ -1080,6 +1100,7 @@ class _LegRow(BaseModel):
|
|||
target_id: str
|
||||
router_name: str
|
||||
router_names: tuple[str, ...] = ()
|
||||
models: tuple[str, ...] = ()
|
||||
direction: ShadowEvalDirection
|
||||
baseline_model: str | None = None
|
||||
judge_model: str
|
||||
|
|
@ -1150,6 +1171,7 @@ def _group_response(
|
|||
for leg in sorted(legs, key=lambda leg: (leg.target_type, leg.target_id))
|
||||
),
|
||||
router_names=first.arm_router_names,
|
||||
models=first.models,
|
||||
direction=first.direction,
|
||||
baseline_model=first.baseline_model,
|
||||
judge_model=first.judge_model,
|
||||
|
|
@ -1322,7 +1344,10 @@ async def start_shadow_eval(
|
|||
A target is a virtual key, a team, or a user. Team and user targets match on the
|
||||
identity every request resolves to at auth time, so they cover JWT-authenticated
|
||||
traffic, which presents no virtual key; a user target samples that user's traffic
|
||||
across all their teams, whether it arrives on a JWT or a key they own.
|
||||
across all their teams, whether it arrives on a JWT or a key they own. models narrows
|
||||
every target to requests for those model groups, so a user plus one model samples that
|
||||
user's traffic on that model across every key they own; it is forward-only, since a
|
||||
reverse job already samples exactly the traffic its own router served.
|
||||
|
||||
A forward job answers whether the targets should adopt router_name: it samples the
|
||||
requests the router did not serve and duplicates them through it. A reverse job
|
||||
|
|
@ -1411,6 +1436,7 @@ async def start_shadow_eval(
|
|||
if data.baseline_model is not None:
|
||||
_validate_plain_model(llm_router, data.baseline_model, "baseline_model", team_ids)
|
||||
_validate_judge_is_not_a_candidate(llm_router, data, team_ids)
|
||||
_validate_model_scope(llm_router, data.models)
|
||||
|
||||
requested_targets: Final[tuple[tuple[ShadowEvalTargetType, str], ...]] = (
|
||||
*(("key", key) for key in data.api_key_ids),
|
||||
|
|
@ -1456,6 +1482,7 @@ async def start_shadow_eval(
|
|||
# a pre-router_names pod samples router_name alone, so it must be a real arm
|
||||
"router_name": data.router_names[0],
|
||||
"router_names": list(data.router_names), # mutable-ok: Prisma payload
|
||||
"models": list(data.models), # mutable-ok: Prisma payload
|
||||
"direction": data.direction,
|
||||
"baseline_model": data.baseline_model,
|
||||
"judge_model": data.judge_model,
|
||||
|
|
@ -1517,6 +1544,7 @@ async def start_shadow_eval(
|
|||
for target_type, target_id in sorted(requested_targets)
|
||||
),
|
||||
router_names=data.router_names,
|
||||
models=data.models,
|
||||
direction=data.direction,
|
||||
baseline_model=data.baseline_model,
|
||||
judge_model=data.judge_model,
|
||||
|
|
|
|||
|
|
@ -1536,6 +1536,7 @@ model LiteLLM_ShadowEvalJob {
|
|||
target_id String // hashed virtual key, team_id, or user_id whose traffic this leg shadows
|
||||
router_name String // first (often only) auto-router under evaluation; router_names is the full set
|
||||
router_names String[] @default([]) // all routers this job runs as shadow arms; empty on legacy rows, whose set is (router_name)
|
||||
models String[] @default([]) // model groups the sampled traffic is narrowed to; empty samples every model
|
||||
direction String @default("forward") // forward | reverse
|
||||
baseline_model String? // reverse only: the fixed model the router is judged against
|
||||
judge_model String
|
||||
|
|
|
|||
|
|
@ -292,6 +292,18 @@ class StartShadowEvalRequest(BaseModel):
|
|||
"to across all their teams: JWT requests carrying their subject claim and virtual keys they own"
|
||||
),
|
||||
)
|
||||
models: tuple[str, ...] = Field(
|
||||
default=(),
|
||||
max_length=100,
|
||||
description=(
|
||||
"Model groups to narrow the sampled traffic to, matched on the group the caller "
|
||||
"requested and resolved through model_group_alias, so an alias and its target are one "
|
||||
"name. Empty samples every model the targets use. This ANDs with the targets: a job "
|
||||
"over a user and one model samples that user's requests on that model across every key "
|
||||
"they own, and none of their other traffic. Forward jobs only: a reverse job samples "
|
||||
"exactly the traffic its own router served, which no other model group can name"
|
||||
),
|
||||
)
|
||||
router_name: str | None = Field(
|
||||
default=None,
|
||||
description=(
|
||||
|
|
@ -372,12 +384,20 @@ class StartShadowEvalRequest(BaseModel):
|
|||
def _round_percentage(cls, value: float) -> float:
|
||||
return round(value, 2)
|
||||
|
||||
@field_validator("api_key_ids", "team_ids", "user_ids")
|
||||
@field_validator("api_key_ids", "team_ids", "user_ids", "models")
|
||||
@classmethod
|
||||
def _dedupe_targets(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||
"""A target named twice would collide with itself on the one-active-per-(target, direction) index."""
|
||||
"""A target named twice would collide with itself on the one-active-per-(target, direction)
|
||||
index; a model named twice is one scope entry."""
|
||||
return tuple(dict.fromkeys(value))
|
||||
|
||||
@field_validator("models")
|
||||
@classmethod
|
||||
def _models_are_names(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||
if not all(name.strip() for name in value):
|
||||
raise ValueError("models must be non-empty model group names")
|
||||
return value
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _at_least_one_target_at_most_hundred(self) -> "StartShadowEvalRequest":
|
||||
total: Final = len(self.api_key_ids) + len(self.team_ids) + len(self.user_ids)
|
||||
|
|
@ -387,6 +407,18 @@ class StartShadowEvalRequest(BaseModel):
|
|||
raise ValueError("at most 100 targets per job across api_key_ids, team_ids, and user_ids")
|
||||
return self
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _model_scope_is_forward_only(self) -> "StartShadowEvalRequest":
|
||||
"""A reverse job admits exactly the requests its own router served, so every one of
|
||||
them names that router and nothing else; any other scope would sample nothing and
|
||||
the router itself is a no-op. Both readings are rejected rather than shipped as a
|
||||
job that silently never samples."""
|
||||
if self.models and self.direction == "reverse":
|
||||
raise ValueError(
|
||||
"models is only meaningful for a forward job; a reverse job samples its own router's traffic"
|
||||
)
|
||||
return self
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _baseline_model_matches_direction(self) -> "StartShadowEvalRequest":
|
||||
if self.direction == "reverse" and self.baseline_model is None:
|
||||
|
|
@ -599,6 +631,10 @@ class ShadowEvalJobResponse(BaseModel):
|
|||
"traffic and judge every arm against the same real responses"
|
||||
),
|
||||
)
|
||||
models: tuple[str, ...] = Field(
|
||||
default=(),
|
||||
description="Model groups the sampled traffic is narrowed to; empty means every model the targets use",
|
||||
)
|
||||
direction: ShadowEvalDirection = "forward"
|
||||
baseline_model: str | None = None
|
||||
judge_model: str
|
||||
|
|
|
|||
|
|
@ -1536,6 +1536,7 @@ model LiteLLM_ShadowEvalJob {
|
|||
target_id String // hashed virtual key, team_id, or user_id whose traffic this leg shadows
|
||||
router_name String // first (often only) auto-router under evaluation; router_names is the full set
|
||||
router_names String[] @default([]) // all routers this job runs as shadow arms; empty on legacy rows, whose set is (router_name)
|
||||
models String[] @default([]) // model groups the sampled traffic is narrowed to; empty samples every model
|
||||
direction String @default("forward") // forward | reverse
|
||||
baseline_model String? // reverse only: the fixed model the router is judged against
|
||||
judge_model String
|
||||
|
|
|
|||
|
|
@ -72,6 +72,7 @@ def _job_record(job: ActiveShadowEvalJob, target_type="key", target_id="key-hash
|
|||
target_id=target_id,
|
||||
router_name=job.router_name,
|
||||
router_names=job.router_names,
|
||||
models=sorted(job.models),
|
||||
direction=job.direction,
|
||||
baseline_model=job.baseline_model,
|
||||
shadow_percentage=job.shadow_percentage,
|
||||
|
|
@ -205,6 +206,7 @@ def _success_kwargs(
|
|||
request_metadata=None,
|
||||
call_type="acompletion",
|
||||
model="claude-opus",
|
||||
model_group="opus-group",
|
||||
response_cost=None,
|
||||
cache_hit=None,
|
||||
):
|
||||
|
|
@ -213,6 +215,7 @@ def _success_kwargs(
|
|||
"id": request_id,
|
||||
"call_type": call_type,
|
||||
"model": model,
|
||||
"model_group": model_group,
|
||||
"metadata": {"user_api_key_hash": api_key_hash},
|
||||
"model_parameters": {"temperature": 0.5, "stream": True},
|
||||
"response_cost": response_cost,
|
||||
|
|
@ -1007,6 +1010,79 @@ class TestTargetMatching:
|
|||
assert logger._job_starts == {"key-job": 1, "team-job": 1}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestModelScope:
|
||||
"""A job scoped to model groups samples a target's request only when the group the
|
||||
caller asked for is one of them; an out-of-scope request is not the job's traffic at
|
||||
all, so it records no funnel event, exactly like a direction mismatch."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"requested,sampled",
|
||||
[("sonnet-group", True), ("opus-group", False), ("", False)],
|
||||
ids=["in-scope-group-samples", "other-group-skips", "unknown-group-fails-closed"],
|
||||
)
|
||||
async def test_scope_admits_only_the_named_groups_and_counts_nothing_else(self, requested, sampled):
|
||||
prisma = _prisma()
|
||||
logger = _logger(router=_router(), prisma=prisma, jobs=(_job(models=frozenset({"sonnet-group", "haiku-group"})),))
|
||||
|
||||
await logger.async_log_success_event(_success_kwargs(model_group=requested), RESPONSE, None, None)
|
||||
await _drain(logger)
|
||||
|
||||
assert prisma.db.litellm_shadowevalattempt.create.await_count == (1 if sampled else 0)
|
||||
assert logger._test_funnel == []
|
||||
|
||||
async def test_an_unscoped_job_samples_every_group(self):
|
||||
prisma = _prisma()
|
||||
logger = _logger(router=_router(), prisma=prisma, jobs=(_job(),))
|
||||
|
||||
await logger.async_log_success_event(_success_kwargs(model_group="anything"), RESPONSE, None, None)
|
||||
await _drain(logger)
|
||||
|
||||
prisma.db.litellm_shadowevalattempt.create.assert_awaited_once()
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"scoped_to,requested",
|
||||
[("sonnet-group", "fast"), ("fast", "sonnet-group")],
|
||||
ids=["job-names-the-target-request-uses-the-alias", "job-names-the-alias-request-uses-the-target"],
|
||||
)
|
||||
async def test_an_alias_and_its_target_are_one_group_on_both_sides(self, scoped_to, requested):
|
||||
"""Both the job's scope and the request's group resolve through the router's alias
|
||||
map at match time, so re-pointing an alias follows config rather than freezing at
|
||||
job start."""
|
||||
router = _router()
|
||||
router.model_group_alias = {"fast": "sonnet-group"}
|
||||
prisma = _prisma(jobs=[_job_record(_job(models=frozenset({scoped_to})))])
|
||||
logger = _logger(router=router, prisma=prisma)
|
||||
|
||||
await logger.async_log_success_event(_success_kwargs(model_group=requested), RESPONSE, None, None)
|
||||
await _drain(logger)
|
||||
|
||||
prisma.db.litellm_shadowevalattempt.create.assert_awaited_once()
|
||||
assert prisma.db.litellm_shadowevaljob.find_many.await_count == 1
|
||||
|
||||
async def test_a_repointed_alias_applies_to_the_next_request_without_a_cache_refill(self):
|
||||
router = _router()
|
||||
router.model_group_alias = {"fast": "sonnet-group"}
|
||||
prisma = _prisma(jobs=[_job_record(_job(models=frozenset({"fast"})))])
|
||||
logger = _logger(router=router, prisma=prisma)
|
||||
await logger.async_log_success_event(_success_kwargs(model_group="sonnet-group"), RESPONSE, None, None)
|
||||
await _drain(logger)
|
||||
assert prisma.db.litellm_shadowevalattempt.create.await_count == 1
|
||||
|
||||
router.model_group_alias = {"fast": "haiku-group"}
|
||||
await logger.async_log_success_event(
|
||||
_success_kwargs(request_id="req-2", model_group="sonnet-group"), RESPONSE, None, None
|
||||
)
|
||||
await logger.async_log_success_event(
|
||||
_success_kwargs(request_id="req-3", model_group="haiku-group"), RESPONSE, None, None
|
||||
)
|
||||
await _drain(logger)
|
||||
|
||||
rows = [call.kwargs["data"]["request_id"] for call in prisma.db.litellm_shadowevalattempt.create.call_args_list]
|
||||
assert rows == ["req-1", "req-3"]
|
||||
assert prisma.db.litellm_shadowevaljob.find_many.await_count == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestActiveJobsCache:
|
||||
async def test_cache_miss_reads_db_once_then_serves_from_cache(self):
|
||||
|
|
|
|||
|
|
@ -882,6 +882,7 @@ def _leg_record(**overrides: object) -> MagicMock:
|
|||
"target_id": "key-hash",
|
||||
"router_name": "my-router",
|
||||
"router_names": (),
|
||||
"models": (),
|
||||
"direction": "forward",
|
||||
"baseline_model": None,
|
||||
"judge_model": "anthropic/claude-sonnet-5",
|
||||
|
|
@ -1033,6 +1034,7 @@ def _shadow_prisma(
|
|||
"target_id",
|
||||
"router_name",
|
||||
"router_names",
|
||||
"models",
|
||||
"direction",
|
||||
"baseline_model",
|
||||
"judge_model",
|
||||
|
|
@ -1533,6 +1535,87 @@ async def test_start_shadow_eval_forward_leaves_the_baseline_column_empty(monkey
|
|||
assert rows[0]["baseline_model"] is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_start_shadow_eval_writes_the_model_scope_on_every_leg_and_echoes_it(monkeypatch: pytest.MonkeyPatch):
|
||||
"""A model scope is job config, so every leg carries the same copy and both the start
|
||||
response and a later list read report it; an auto-router is a legitimate scope (a
|
||||
forward job on one router may sample what another router serves today)."""
|
||||
import litellm.proxy.proxy_server as proxy_server
|
||||
|
||||
_configure_anthropic_sdk_judge(monkeypatch)
|
||||
prisma = _shadow_prisma()
|
||||
monkeypatch.setattr(proxy_server, "prisma_client", prisma)
|
||||
monkeypatch.setattr(proxy_server, "llm_router", _shadow_router())
|
||||
|
||||
response = await start_shadow_eval(
|
||||
_start_request(api_key_ids=("key-hash", "key-hash-2"), models=("cheap", "sonnet-router")), ADMIN
|
||||
)
|
||||
|
||||
rows = prisma.db.litellm_shadowevaljob.create_many.call_args.kwargs["data"]
|
||||
assert [row["models"] for row in rows] == [["cheap", "sonnet-router"], ["cheap", "sonnet-router"]]
|
||||
assert response.models == ("cheap", "sonnet-router")
|
||||
|
||||
listed = _shadow_prisma(legs=[_leg_record(models=("cheap",)), _leg_record(id="leg-0", group_id="job-0")])
|
||||
monkeypatch.setattr(proxy_server, "prisma_client", listed)
|
||||
jobs = await list_shadow_eval_jobs(VIEWER, target_type=None, target_id=None, limit=50)
|
||||
assert {job.job_id: job.models for job in jobs} == {"job-1": ("cheap",), "job-0": ()}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_start_shadow_eval_accepts_a_team_public_scope_for_a_user_target(monkeypatch: pytest.MonkeyPatch):
|
||||
"""A user's traffic can arrive on any team's key, so a name only one team can ask for
|
||||
is a legitimate scope for a user target even though it resolves for nobody unscoped."""
|
||||
import litellm.proxy.proxy_server as proxy_server
|
||||
|
||||
_configure_anthropic_sdk_judge(monkeypatch)
|
||||
prisma = _shadow_prisma(known_users={"dev-alice": "alice@example.com"})
|
||||
monkeypatch.setattr(proxy_server, "prisma_client", prisma)
|
||||
monkeypatch.setattr(proxy_server, "llm_router", _shadow_router())
|
||||
|
||||
response = await start_shadow_eval(
|
||||
_start_request(api_key_ids=(), user_ids=("dev-alice",), models=("house-judge",)), ADMIN
|
||||
)
|
||||
|
||||
assert response.models == ("house-judge",)
|
||||
assert prisma.db.litellm_shadowevaljob.create_many.call_args.kwargs["data"][0]["models"] == ["house-judge"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_start_shadow_eval_rejects_a_model_scope_this_proxy_does_not_serve(monkeypatch: pytest.MonkeyPatch):
|
||||
"""A typo'd model name would otherwise start a job that samples nothing. Only the
|
||||
unresolvable names are reported, so the caller fixes them in one round."""
|
||||
import litellm.proxy.proxy_server as proxy_server
|
||||
|
||||
_configure_anthropic_sdk_judge(monkeypatch)
|
||||
prisma = _shadow_prisma()
|
||||
monkeypatch.setattr(proxy_server, "prisma_client", prisma)
|
||||
monkeypatch.setattr(proxy_server, "llm_router", _shadow_router())
|
||||
|
||||
with pytest.raises(HTTPException) as exc:
|
||||
await start_shadow_eval(_start_request(models=("cheap", "no-such-model-zzz")), ADMIN)
|
||||
assert exc.value.status_code == 400
|
||||
assert "'no-such-model-zzz'" in exc.value.detail
|
||||
assert "'cheap'" not in exc.value.detail
|
||||
prisma.db.litellm_shadowevaljob.create_many.assert_not_called()
|
||||
|
||||
|
||||
def test_start_request_dedupes_the_model_scope_and_rejects_blank_names():
|
||||
assert _start_request(models=("cheap", "mid", "cheap")).models == ("cheap", "mid")
|
||||
assert _start_request().models == ()
|
||||
with pytest.raises(ValidationError, match="non-empty model group names"):
|
||||
_start_request(models=("cheap", " "))
|
||||
|
||||
|
||||
def test_start_request_rejects_a_model_scope_on_a_reverse_job():
|
||||
"""Reverse admission is the router's own traffic, whose requested group is always the
|
||||
router, so a plain-model scope would sample nothing and the router itself is a no-op."""
|
||||
with pytest.raises(ValidationError, match="only meaningful for a forward job"):
|
||||
_start_request(direction="reverse", baseline_model="cheap", models=("mid",))
|
||||
with pytest.raises(ValidationError, match="only meaningful for a forward job"):
|
||||
_start_request(direction="reverse", baseline_model="cheap", models=("my-router",))
|
||||
assert _start_request(direction="reverse", baseline_model="cheap").models == ()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_start_shadow_eval_rejects_keys_this_proxy_does_not_know(monkeypatch: pytest.MonkeyPatch):
|
||||
"""A typo'd api_key_id would otherwise create a leg no traffic can ever match. Every
|
||||
|
|
|
|||
|
|
@ -104,6 +104,7 @@ const job = (overrides: Partial<ShadowEvalJob> = {}): ShadowEvalJob => ({
|
|||
status: "running",
|
||||
router_name: "claude-auto",
|
||||
router_names: ["claude-auto"],
|
||||
models: [],
|
||||
direction: "forward",
|
||||
baseline_model: null,
|
||||
judge_model: "anthropic/claude-sonnet-5",
|
||||
|
|
@ -450,6 +451,7 @@ describe("ShadowEvalSection", () => {
|
|||
api_key_ids: ["hash-alpha", "hash-beta"],
|
||||
team_ids: [],
|
||||
user_ids: [],
|
||||
models: [],
|
||||
router_names: ["gpt-auto"],
|
||||
direction: "forward",
|
||||
shadow_percentage: 10,
|
||||
|
|
@ -479,6 +481,7 @@ describe("ShadowEvalSection", () => {
|
|||
api_key_ids: [],
|
||||
team_ids: ["team-eng"],
|
||||
user_ids: [],
|
||||
models: [],
|
||||
router_names: ["gpt-auto"],
|
||||
direction: "forward",
|
||||
shadow_percentage: 10,
|
||||
|
|
@ -489,15 +492,41 @@ describe("ShadowEvalSection", () => {
|
|||
expect(start.mutate).toHaveBeenCalledWith(expectedBody);
|
||||
});
|
||||
|
||||
it("narrows a job to the picked model groups and shows the scope on the job headline", async () => {
|
||||
const user = userEvent.setup();
|
||||
const { start } = mockHooks({});
|
||||
render(<ShadowEvalSection />);
|
||||
|
||||
await user.click(screen.getByPlaceholderText("Search teams by alias"));
|
||||
const teamList = await screen.findByTestId("paginated-multi-select-list");
|
||||
await user.click(within(teamList).getByText("engineering"));
|
||||
await chooseSelectOption(user, screen.getByPlaceholderText("Every model the targets use"), "prod-claude");
|
||||
await chooseSelectOption(user, screen.getByPlaceholderText("Select up to 4 auto-routers"), "gpt-auto");
|
||||
await user.click(screen.getByPlaceholderText("Select a judge model"));
|
||||
await user.click(await screen.findByRole("option", { name: /anthropic\/claude-sonnet-5/ }));
|
||||
await user.click(screen.getByText("Start shadow eval"));
|
||||
|
||||
expect(start.mutate).toHaveBeenCalledWith(
|
||||
expect.objectContaining({ team_ids: ["team-eng"], models: ["prod-claude"] }),
|
||||
);
|
||||
|
||||
const scoped = job({ models: ["prod-claude", "prod-haiku"] });
|
||||
mockHooks({ jobs: [scoped], detailsById: { "job-1": scoped } });
|
||||
render(<ShadowEvalSection />);
|
||||
expect(screen.getByText("prod-claude, prod-haiku")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("requires a baseline model in reverse mode and submits it, while forward mode never shows the picker", async () => {
|
||||
const user = userEvent.setup();
|
||||
const { start } = mockHooks({});
|
||||
render(<ShadowEvalSection />);
|
||||
|
||||
expect(screen.queryByPlaceholderText("Select a baseline model")).not.toBeInTheDocument();
|
||||
expect(screen.getByPlaceholderText("Every model the targets use")).toBeInTheDocument();
|
||||
|
||||
await user.click(screen.getByText("Adoption check: key's traffic vs the router"));
|
||||
await user.click(await screen.findByText("Regression check: router's picks vs a baseline"));
|
||||
expect(screen.queryByPlaceholderText("Every model the targets use")).not.toBeInTheDocument();
|
||||
await user.click(screen.getByPlaceholderText("Search keys by alias"));
|
||||
const keyList = await screen.findByTestId("paginated-multi-select-list");
|
||||
await user.click(within(keyList).getByText("prod-alpha"));
|
||||
|
|
@ -516,6 +545,7 @@ describe("ShadowEvalSection", () => {
|
|||
api_key_ids: ["hash-alpha"],
|
||||
team_ids: [],
|
||||
user_ids: [],
|
||||
models: [],
|
||||
router_names: ["gpt-auto"],
|
||||
direction: "reverse",
|
||||
baseline_model: "prod-claude",
|
||||
|
|
@ -551,6 +581,7 @@ describe("ShadowEvalSection", () => {
|
|||
api_key_ids: ["hash-alpha"],
|
||||
team_ids: [],
|
||||
user_ids: [],
|
||||
models: [],
|
||||
router_names: ["gpt-auto", "claude-auto"],
|
||||
direction: "forward",
|
||||
shadow_percentage: 10,
|
||||
|
|
|
|||
|
|
@ -87,17 +87,25 @@ const targetStatus = (job: ShadowEvalJob, target: ShadowEvalJobTarget): string =
|
|||
|
||||
const jobRouters = (job: ShadowEvalJob): string => (job.router_names ?? [job.router_name]).join(", ");
|
||||
|
||||
const jobModelScope = (job: ShadowEvalJob): React.ReactNode =>
|
||||
job.models && job.models.length > 0 ? (
|
||||
<>
|
||||
{" "}
|
||||
on <span className="font-mono text-xs">{job.models.join(", ")}</span>
|
||||
</>
|
||||
) : null;
|
||||
|
||||
const jobHeadline = (job: ShadowEvalJob): React.ReactNode =>
|
||||
job.direction === "reverse" ? (
|
||||
<>
|
||||
Comparing <span className="font-mono text-xs">{jobRouters(job)}</span> to{" "}
|
||||
<span className="font-mono text-xs">{job.baseline_model}</span> on {job.shadow_percentage}% of{" "}
|
||||
<span className="font-mono text-xs">{shadowedTargetsLabel(job)}</span> traffic
|
||||
<span className="font-mono text-xs">{shadowedTargetsLabel(job)}</span> traffic{jobModelScope(job)}
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
Shadowing {job.shadow_percentage}% of <span className="font-mono text-xs">{shadowedTargetsLabel(job)}</span>{" "}
|
||||
traffic via <span className="font-mono text-xs">{jobRouters(job)}</span>
|
||||
traffic{jobModelScope(job)} via <span className="font-mono text-xs">{jobRouters(job)}</span>
|
||||
</>
|
||||
);
|
||||
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@ import { useStartShadowEval, type ShadowEvalJob } from "./useShadowEval";
|
|||
type ShadowEvalDirection = ShadowEvalJob["direction"];
|
||||
|
||||
const MAX_ROUTERS = 4;
|
||||
const MAX_MODELS = 100;
|
||||
|
||||
const RECOMMENDED_JUDGE_MODELS = ["anthropic/claude-sonnet-5", "openai/gpt-4o", "gemini/gemini-2.5-pro"] as const;
|
||||
|
||||
|
|
@ -206,6 +207,7 @@ interface StartFormValidityInputs {
|
|||
apiKeyIds: string[];
|
||||
teamIds: string[];
|
||||
userIds: string[];
|
||||
models: string[];
|
||||
routerNames: string[];
|
||||
direction: ShadowEvalDirection;
|
||||
baselineModel: string;
|
||||
|
|
@ -224,7 +226,8 @@ const startFormValidity = (inputs: StartFormValidityInputs) => {
|
|||
const routerCountValid = inputs.routerNames.length >= 1 && inputs.routerNames.length <= MAX_ROUTERS;
|
||||
const routersMatchDirection = inputs.direction === "forward" || inputs.routerNames.length === 1;
|
||||
const routersValid = routerCountValid && routersMatchDirection;
|
||||
const modelsPicked = routersValid && inputs.judgeModel !== "" && baselinePicked;
|
||||
const scopeValid = routersValid && (inputs.direction === "reverse" || inputs.models.length <= MAX_MODELS);
|
||||
const modelsPicked = scopeValid && inputs.judgeModel !== "" && baselinePicked;
|
||||
const filled = targetsPicked && modelsPicked;
|
||||
const boundsValid = percentageValid && maxBudgetValid;
|
||||
const valid = Boolean(inputs.accessToken) && filled && boundsValid;
|
||||
|
|
@ -235,6 +238,7 @@ interface StartBodyInputs {
|
|||
apiKeyIds: string[];
|
||||
teamIds: string[];
|
||||
userIds: string[];
|
||||
models: string[];
|
||||
routerNames: string[];
|
||||
direction: ShadowEvalDirection;
|
||||
baselineModel: string;
|
||||
|
|
@ -248,6 +252,7 @@ const buildStartBody = (inputs: StartBodyInputs) => ({
|
|||
api_key_ids: inputs.apiKeyIds,
|
||||
team_ids: inputs.teamIds,
|
||||
user_ids: inputs.userIds,
|
||||
models: inputs.direction === "forward" ? inputs.models : [],
|
||||
router_names: inputs.routerNames,
|
||||
direction: inputs.direction,
|
||||
...(inputs.direction === "reverse" ? { baseline_model: inputs.baselineModel } : {}),
|
||||
|
|
@ -262,6 +267,7 @@ export const StartForm: React.FC = () => {
|
|||
const [apiKeyIds, setApiKeyIds] = useState<string[]>([]);
|
||||
const [teamIds, setTeamIds] = useState<string[]>([]);
|
||||
const [userIds, setUserIds] = useState<string[]>([]);
|
||||
const [models, setModels] = useState<string[]>([]);
|
||||
const [routerNames, setRouterNames] = useState<string[]>([]);
|
||||
const [direction, setDirection] = useState<ShadowEvalDirection>("forward");
|
||||
const [baselineModel, setBaselineModel] = useState("");
|
||||
|
|
@ -272,6 +278,11 @@ export const StartForm: React.FC = () => {
|
|||
const { data: autoRouters } = useAutoRouters();
|
||||
const judgeModelOptions = useJudgeModelOptions();
|
||||
const baselineModelOptions = useBaselineModelOptions();
|
||||
const configuredGroups = usePlainModelGroups();
|
||||
const modelOptions = useMemo<SearchSelectOption[]>(
|
||||
() => [...configuredGroups].toSorted((a, b) => a.localeCompare(b)).map((name) => ({ label: name, value: name })),
|
||||
[configuredGroups],
|
||||
);
|
||||
const start = useStartShadowEval();
|
||||
|
||||
const routerOptions = useMemo<SearchSelectOption[]>(() => {
|
||||
|
|
@ -286,6 +297,7 @@ export const StartForm: React.FC = () => {
|
|||
apiKeyIds,
|
||||
teamIds,
|
||||
userIds,
|
||||
models,
|
||||
routerNames,
|
||||
direction,
|
||||
baselineModel,
|
||||
|
|
@ -299,6 +311,7 @@ export const StartForm: React.FC = () => {
|
|||
apiKeyIds,
|
||||
teamIds,
|
||||
userIds,
|
||||
models,
|
||||
routerNames,
|
||||
direction,
|
||||
baselineModel,
|
||||
|
|
@ -344,6 +357,22 @@ export const StartForm: React.FC = () => {
|
|||
<Field label="Users to shadow" htmlFor="shadow-eval-user">
|
||||
<UserSelect value={userIds} onChange={setUserIds} />
|
||||
</Field>
|
||||
{direction === "forward" && (
|
||||
<Field label="Only on models">
|
||||
<MultiSelect
|
||||
options={modelOptions}
|
||||
value={models}
|
||||
onValueChange={setModels}
|
||||
placeholder="Every model the targets use"
|
||||
emptyText="No models configured"
|
||||
/>
|
||||
{models.length > MAX_MODELS ? (
|
||||
<p className="text-xs text-destructive">Pick at most {MAX_MODELS} models</p>
|
||||
) : (
|
||||
<p className="text-xs text-muted-foreground">Narrows every target above to requests for these models</p>
|
||||
)}
|
||||
</Field>
|
||||
)}
|
||||
<RouterField
|
||||
options={routerOptions}
|
||||
routerNames={routerNames}
|
||||
|
|
|
|||
17
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
17
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -1264,7 +1264,10 @@ export interface paths {
|
|||
* A target is a virtual key, a team, or a user. Team and user targets match on the
|
||||
* identity every request resolves to at auth time, so they cover JWT-authenticated
|
||||
* traffic, which presents no virtual key; a user target samples that user's traffic
|
||||
* across all their teams, whether it arrives on a JWT or a key they own.
|
||||
* across all their teams, whether it arrives on a JWT or a key they own. models narrows
|
||||
* every target to requests for those model groups, so a user plus one model samples that
|
||||
* user's traffic on that model across every key they own; it is forward-only, since a
|
||||
* reverse job already samples exactly the traffic its own router served.
|
||||
*
|
||||
* A forward job answers whether the targets should adopt router_name: it samples the
|
||||
* requests the router did not serve and duplicates them through it. A reverse job
|
||||
|
|
@ -35805,6 +35808,12 @@ export interface components {
|
|||
* @description Most recent attempt error; detail endpoint only
|
||||
*/
|
||||
last_error?: string | null;
|
||||
/**
|
||||
* Models
|
||||
* @description Model groups the sampled traffic is narrowed to; empty means every model the targets use
|
||||
* @default []
|
||||
*/
|
||||
models: string[];
|
||||
/** @description Stratified verdicts; detail endpoint only */
|
||||
results?: components["schemas"]["ShadowEvalResult"] | null;
|
||||
/**
|
||||
|
|
@ -36228,6 +36237,12 @@ export interface components {
|
|||
* @default 10
|
||||
*/
|
||||
max_budget: number;
|
||||
/**
|
||||
* Models
|
||||
* @description Model groups to narrow the sampled traffic to, matched on the group the caller requested and resolved through model_group_alias, so an alias and its target are one name. Empty samples every model the targets use. This ANDs with the targets: a job over a user and one model samples that user's requests on that model across every key they own, and none of their other traffic. Forward jobs only: a reverse job samples exactly the traffic its own router served, which no other model group can name
|
||||
* @default []
|
||||
*/
|
||||
models: string[];
|
||||
/**
|
||||
* Router Name
|
||||
* @description The auto-router under evaluation, in either direction: the single-router spelling of router_names. Provide exactly one of the two fields
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue