mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-14 23:21:35 +00:00
fix(auto-router): exclude disconnects before first token from abandonment, cap quality-signal scans, clarify metrics
- Abandon signal now excludes disconnects that delivered zero completion tokens. These are latency/connectivity events, not quality evidence -- the caller never saw a response to judge. Measured as: client_disconnected AND completion_tokens > 0. Regression coverage in test_disconnect_before_first_token_is_not_counted_as_abandonment. - Added MAX_QUALITY_SIGNAL_ROWS = 1,000,000 cap and explicit 400 rejection when a date range contains more matching spend-log rows. Window scanning is unbounded and materializes in-memory; without a cap, a PROXY_ADMIN_VIEW_ONLY key could exhaust database and worker resources via repeated multi-year requests. Returns descriptive error directing users to narrow the range, rather than silently truncating and dropping the turns these signals compute from. - UI metric label: 'Stream abandonment' -> 'Potential Turn Abandons' (turn-granular, not session-granular). Tooltip updated to mention the before-first-token exclusion and the conflation of user abandons with dropped connections. - Removed red-text flagging of worse-than-baseline rates (worseThanBaseline logic). These rates are directional evidence only, and coloring one worse/better can mislead teams into over-interpreting self-selected baseline comparisons. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
parent
2580aaa65f
commit
941fd47804
6 changed files with 110 additions and 15 deletions
|
|
@ -9,9 +9,12 @@ both are already recorded on every request:
|
|||
do the job. High precision, low recall: it only sees callers who *can* switch
|
||||
and bother to, so an API-only integration can report zero while suffering.
|
||||
|
||||
``abandonment`` the caller hung up before the stream finished. Low precision, high recall:
|
||||
it catches the giving-up that escalation structurally misses, but a dropped
|
||||
connection and "I read enough" look the same from here.
|
||||
``abandonment`` the caller hung up after content started streaming but before it finished.
|
||||
Low precision, high recall: it catches the giving-up that escalation
|
||||
structurally misses, but a dropped connection and "I read enough" look the
|
||||
same from here. Disconnects before any content arrived are excluded --
|
||||
those are a latency or connectivity event, not a judgment on the response,
|
||||
since the caller never saw one.
|
||||
|
||||
They are reported side by side rather than blended, because the two fail in opposite
|
||||
directions and one number would hide which of them fired.
|
||||
|
|
@ -46,13 +49,16 @@ MIN_SESSION_ID_COVERAGE: Final = 0.8
|
|||
class Turn:
|
||||
"""One request, reduced to what the two signals need.
|
||||
|
||||
``escalation`` needs the model and the order; ``abandonment`` needs the disconnect flag.
|
||||
``escalation`` needs the model and the order; ``abandonment`` needs the disconnect flag
|
||||
plus ``completion_tokens``, since a disconnect that delivered nothing is not the same
|
||||
event as one that cut off a response in progress -- see ``was_abandoned``.
|
||||
Kept as a plain object rather than a row dict so the signal functions state their inputs.
|
||||
"""
|
||||
|
||||
__slots__ = (
|
||||
"api_key",
|
||||
"client_disconnected",
|
||||
"completion_tokens",
|
||||
"has_client_session_id",
|
||||
"model",
|
||||
"router_name",
|
||||
|
|
@ -68,6 +74,7 @@ class Turn:
|
|||
model: str,
|
||||
started_at: float,
|
||||
client_disconnected: bool,
|
||||
completion_tokens: int,
|
||||
router_name: str | None,
|
||||
has_client_session_id: bool,
|
||||
) -> None:
|
||||
|
|
@ -76,9 +83,20 @@ class Turn:
|
|||
self.model = model
|
||||
self.started_at = started_at
|
||||
self.client_disconnected = client_disconnected
|
||||
self.completion_tokens = completion_tokens
|
||||
self.router_name = router_name
|
||||
self.has_client_session_id = has_client_session_id
|
||||
|
||||
@property
|
||||
def was_abandoned(self) -> bool:
|
||||
"""True for a disconnect that cut off a response already in progress.
|
||||
|
||||
A disconnect with zero completion tokens delivered nothing for the caller to judge --
|
||||
that is a latency or connectivity failure, not evidence about response quality, so it
|
||||
is excluded from the abandonment signal even though ``client_disconnected`` is True.
|
||||
"""
|
||||
return self.client_disconnected and self.completion_tokens > 0
|
||||
|
||||
|
||||
class CohortSignals:
|
||||
"""What one population evidences, and how much of it there was to look at."""
|
||||
|
|
@ -191,7 +209,7 @@ def signals_for_cohort(
|
|||
|
||||
escalated: Final = sum(1 for session_turns in eligible if session_escalated(session_turns, ranks))
|
||||
eligible_turns: Final = tuple(turn for session_turns in eligible for turn in session_turns)
|
||||
abandoned: Final = sum(1 for turn in eligible_turns if turn.client_disconnected)
|
||||
abandoned: Final = sum(1 for turn in eligible_turns if turn.was_abandoned)
|
||||
return CohortSignals(
|
||||
sessions=len(eligible),
|
||||
escalation_rate_pct=round(100.0 * escalated / len(eligible), 1),
|
||||
|
|
|
|||
|
|
@ -489,12 +489,15 @@ class _QualityTurnRow(BaseModel):
|
|||
started_at: float
|
||||
router_name: str | None
|
||||
client_disconnected: bool
|
||||
completion_tokens: int
|
||||
session_turn_count: int
|
||||
api_key: str
|
||||
|
||||
|
||||
_QUALITY_TURN_ROWS: Final = TypeAdapter(list[_QualityTurnRow])
|
||||
|
||||
MAX_QUALITY_SIGNAL_ROWS: Final = 100_000
|
||||
|
||||
_QUALITY_SIGNALS_SQL: Final = """
|
||||
SELECT
|
||||
session_id,
|
||||
|
|
@ -502,12 +505,14 @@ SELECT
|
|||
EXTRACT(EPOCH FROM "startTime")::float8 AS started_at,
|
||||
(metadata #>> '{routing_decision,router_model_name}') AS router_name,
|
||||
COALESCE(metadata #>> '{error_information,error_code}' = '499', false) AS client_disconnected,
|
||||
completion_tokens,
|
||||
COUNT(*) OVER (PARTITION BY api_key, session_id)::int AS session_turn_count,
|
||||
api_key
|
||||
FROM "LiteLLM_SpendLogs"
|
||||
WHERE "startTime" >= $1::timestamp AND "startTime" < $2::timestamp
|
||||
AND session_id IS NOT NULL
|
||||
AND model <> ''
|
||||
LIMIT $3
|
||||
"""
|
||||
|
||||
|
||||
|
|
@ -564,6 +569,7 @@ def _quality_signals_for(
|
|||
model=row.model,
|
||||
started_at=row.started_at,
|
||||
client_disconnected=row.client_disconnected,
|
||||
completion_tokens=row.completion_tokens,
|
||||
router_name=row.router_name,
|
||||
has_client_session_id=row.session_turn_count > 1,
|
||||
)
|
||||
|
|
@ -650,7 +656,17 @@ async def get_auto_router_quality_signals(
|
|||
_QUALITY_SIGNALS_SQL,
|
||||
start_day.isoformat(),
|
||||
(end_day + timedelta(days=1)).isoformat(),
|
||||
MAX_QUALITY_SIGNAL_ROWS + 1,
|
||||
)
|
||||
if raw_rows is not None and len(raw_rows) > MAX_QUALITY_SIGNAL_ROWS:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=(
|
||||
f"Window contains more than {MAX_QUALITY_SIGNAL_ROWS:,} session-bearing requests; "
|
||||
"narrow the date range. Truncating would silently drop the turns these signals "
|
||||
"are computed from."
|
||||
),
|
||||
)
|
||||
turns: Final = _QUALITY_TURN_ROWS.validate_python(raw_rows or ())
|
||||
router_names: Final = tuple(sorted(frozenset(row.router_name for row in turns if row.router_name is not None)))
|
||||
return AutoRouterQualitySignalsResponse(
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ def _turn(
|
|||
model="sonnet",
|
||||
started_at=0.0,
|
||||
client_disconnected=False,
|
||||
completion_tokens=10,
|
||||
has_client_session_id=True,
|
||||
api_key="key1",
|
||||
):
|
||||
|
|
@ -38,11 +39,23 @@ def _turn(
|
|||
model=model,
|
||||
started_at=started_at,
|
||||
client_disconnected=client_disconnected,
|
||||
completion_tokens=completion_tokens,
|
||||
router_name="auto-router",
|
||||
has_client_session_id=has_client_session_id,
|
||||
)
|
||||
|
||||
|
||||
class TestTurnWasAbandoned:
|
||||
def test_disconnect_with_delivered_tokens_is_abandoned(self):
|
||||
assert _turn(client_disconnected=True, completion_tokens=1).was_abandoned is True
|
||||
|
||||
def test_disconnect_with_zero_tokens_is_not_abandoned(self):
|
||||
assert _turn(client_disconnected=True, completion_tokens=0).was_abandoned is False
|
||||
|
||||
def test_completed_turn_with_zero_tokens_is_not_abandoned(self):
|
||||
assert _turn(client_disconnected=False, completion_tokens=0).was_abandoned is False
|
||||
|
||||
|
||||
class TestRankModelsByCost:
|
||||
def test_equal_cost_models_receive_equal_rank(self, monkeypatch: pytest.MonkeyPatch):
|
||||
import litellm.proxy.db.autorouter_quality_signals as module
|
||||
|
|
@ -151,6 +164,26 @@ class TestSignalsForCohort:
|
|||
assert result.sessions == 1
|
||||
assert result.abandonment_rate_pct == 50.0
|
||||
|
||||
def test_disconnect_before_first_token_is_not_counted_as_abandonment(self):
|
||||
# A disconnect that delivered zero completion tokens never showed the caller a
|
||||
# response to judge -- it's a latency/connectivity event, not quality evidence.
|
||||
turns = [
|
||||
_turn(session_id="s1", model="haiku", started_at=1, client_disconnected=True, completion_tokens=0),
|
||||
_turn(session_id="s1", model="haiku", started_at=2, client_disconnected=False),
|
||||
_turn(
|
||||
session_id="s2",
|
||||
api_key="key1",
|
||||
model="haiku",
|
||||
started_at=1,
|
||||
client_disconnected=True,
|
||||
completion_tokens=5,
|
||||
),
|
||||
_turn(session_id="s2", api_key="key1", model="haiku", started_at=2, client_disconnected=False),
|
||||
]
|
||||
result = signals_for_cohort(turns, RANKS, REACHABLE_BY_KEY)
|
||||
assert result.sessions == 2
|
||||
assert result.abandonment_rate_pct == 25.0
|
||||
|
||||
def test_no_eligible_sessions_returns_none_rates_not_zero(self):
|
||||
# Every session already on the ceiling model: zero would misleadingly claim
|
||||
# "no miss detected" when in fact nothing could be measured.
|
||||
|
|
|
|||
|
|
@ -485,6 +485,7 @@ class TestAutoRouterQualitySignals:
|
|||
*,
|
||||
router_name: str | None = "live-auto",
|
||||
client_disconnected: bool = False,
|
||||
completion_tokens: int = 10,
|
||||
session_turn_count: int = 2,
|
||||
api_key: str = "key-1",
|
||||
) -> dict:
|
||||
|
|
@ -494,6 +495,7 @@ class TestAutoRouterQualitySignals:
|
|||
"started_at": started_at,
|
||||
"router_name": router_name,
|
||||
"client_disconnected": client_disconnected,
|
||||
"completion_tokens": completion_tokens,
|
||||
"session_turn_count": session_turn_count,
|
||||
"api_key": api_key,
|
||||
}
|
||||
|
|
@ -554,6 +556,20 @@ class TestAutoRouterQualitySignals:
|
|||
)
|
||||
assert err.value.status_code == 400
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_window_over_the_row_cap_is_rejected_not_silently_truncated(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import (
|
||||
MAX_QUALITY_SIGNAL_ROWS,
|
||||
)
|
||||
|
||||
rows = [self._row(f"s{i}", self.CHEAP, float(i)) for i in range(MAX_QUALITY_SIGNAL_ROWS + 1)]
|
||||
with pytest.raises(HTTPException) as err:
|
||||
await self._call(rows, monkeypatch)
|
||||
assert err.value.status_code == 400
|
||||
assert "narrow the date range" in err.value.detail
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_routed_escalation_is_measured_from_routed_rows(self, monkeypatch: pytest.MonkeyPatch):
|
||||
rows = [
|
||||
|
|
@ -677,3 +693,19 @@ class TestAutoRouterQualitySignals:
|
|||
response = await self._call(rows, monkeypatch)
|
||||
assert response.totals.routed.sessions == 1
|
||||
assert response.totals.routed.abandonment_rate_pct == 50.0
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_disconnect_before_first_token_is_excluded_from_abandonment(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
# Same shape as the test above, but the disconnect delivered nothing: the caller
|
||||
# never saw a response to judge, so it must not read as abandonment.
|
||||
rows = [
|
||||
self._row("s1", self.CHEAP, 1.0, client_disconnected=True, completion_tokens=0),
|
||||
self._row("s1", self.CHEAP, 2.0, client_disconnected=False),
|
||||
self._row("s2", self.PRICEY, 1.0),
|
||||
self._row("s2", self.PRICEY, 2.0),
|
||||
]
|
||||
response = await self._call(rows, monkeypatch)
|
||||
assert response.totals.routed.sessions == 1
|
||||
assert response.totals.routed.abandonment_rate_pct == 0.0
|
||||
|
|
|
|||
|
|
@ -358,7 +358,7 @@ describe("AutoRouterBenchmarksTab", () => {
|
|||
expect(screen.getAllByText("Not enough comparable non-router traffic in this window to compare")).toHaveLength(2);
|
||||
});
|
||||
|
||||
it("flags escalation as worse than baseline visually distinctly from a healthy rate", () => {
|
||||
it("renders escalation and abandonment rates without flagging them as regressions", () => {
|
||||
mockHook({ data: response([group()]) });
|
||||
mockQualityHook(
|
||||
qualityResponse({
|
||||
|
|
@ -368,7 +368,8 @@ describe("AutoRouterBenchmarksTab", () => {
|
|||
);
|
||||
renderTab();
|
||||
|
||||
expect(screen.getByText("9.0%")).toHaveClass("text-destructive");
|
||||
expect(screen.getByText("9.0%")).toHaveClass("text-foreground");
|
||||
expect(screen.getByText("9.0%")).not.toHaveClass("text-destructive");
|
||||
});
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -115,7 +115,6 @@ const QualityMetric: React.FC<{
|
|||
baseline: number | null | undefined;
|
||||
baselineUnavailableReason: string | null | undefined;
|
||||
}> = ({ label, hint, routed, baseline, baselineUnavailableReason }) => {
|
||||
const worseThanBaseline = routed != null && baseline != null && routed > baseline;
|
||||
return (
|
||||
<div className="flex flex-col gap-1.5">
|
||||
<TooltipProvider>
|
||||
|
|
@ -133,11 +132,7 @@ const QualityMetric: React.FC<{
|
|||
<TooltipContent className="max-w-xs">{hint}</TooltipContent>
|
||||
</Tooltip>
|
||||
</TooltipProvider>
|
||||
<p
|
||||
className={`text-2xl font-semibold tabular-nums ${worseThanBaseline ? "text-destructive" : "text-foreground"}`}
|
||||
>
|
||||
{ratePctLabel(routed)}
|
||||
</p>
|
||||
<p className="text-2xl font-semibold tabular-nums text-foreground">{ratePctLabel(routed)}</p>
|
||||
<p className="text-[11px] text-muted-foreground">
|
||||
{baseline == null
|
||||
? (baselineUnavailableReason && BASELINE_UNAVAILABLE_COPY[baselineUnavailableReason]) ||
|
||||
|
|
@ -159,8 +154,8 @@ const QualityCard: React.FC<{ signals: AutoRouterQualitySignals }> = ({ signals
|
|||
baselineUnavailableReason={signals.baseline_unavailable_reason}
|
||||
/>
|
||||
<QualityMetric
|
||||
label="Stream abandonment"
|
||||
hint="Share of turns where the caller hung up before the response finished. Catches the giving-up that escalation misses, but a dropped connection and 'I read enough' look the same from here."
|
||||
label="Potential Turn Abandons"
|
||||
hint="Share of turns where the caller hung up after content started streaming but before it finished. Disconnects before any content arrived are excluded — those are a latency or connectivity event, not a judgment on the response. Note: User abandons and dropped connections are conflated in this metric."
|
||||
routed={signals.routed.abandonment_rate_pct}
|
||||
baseline={signals.baseline?.abandonment_rate_pct}
|
||||
baselineUnavailableReason={signals.baseline_unavailable_reason}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue