mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cli): show full-session auto-router cost comparison (#44959)
This commit is contained in:
parent
3c79db6122
commit
9a3f000c9b
2 changed files with 55 additions and 39 deletions
|
|
@ -69,7 +69,6 @@ class Session(NamedTuple):
|
|||
baseline_model: str | None
|
||||
turns: int | None = None
|
||||
savings_estimated_turns: int | None = None
|
||||
savings_estimated_actual_spend: float | None = None
|
||||
|
||||
|
||||
class Credentials(NamedTuple):
|
||||
|
|
@ -210,10 +209,9 @@ def _session_from_payload(payload: Mapping[str, object]) -> Session | None:
|
|||
router_name: Final = printable(payload.get("router_name"))
|
||||
last_model: Final = printable(payload.get("last_model"))
|
||||
spend: Final = payload.get("spend")
|
||||
baseline_spend: Final = payload.get("savings_estimated_baseline_spend", payload.get("baseline_spend"))
|
||||
baseline_spend: Final = payload.get("baseline_spend")
|
||||
turns: Final = payload.get("turns")
|
||||
estimated_turns: Final = payload.get("savings_estimated_turns")
|
||||
estimated_actual: Final = payload.get("savings_estimated_actual_spend")
|
||||
if not router_name or not last_model:
|
||||
return None
|
||||
if not isinstance(spend, (int, float)) or isinstance(spend, bool) or not isfinite(spend):
|
||||
|
|
@ -234,19 +232,11 @@ def _session_from_payload(payload: Mapping[str, object]) -> Session | None:
|
|||
if isinstance(estimated_turns, int) and not isinstance(estimated_turns, bool) and estimated_turns >= 0
|
||||
else (0 if estimated_turns is not None else None)
|
||||
),
|
||||
savings_estimated_actual_spend=(
|
||||
float(estimated_actual)
|
||||
if isinstance(estimated_actual, (int, float))
|
||||
and not isinstance(estimated_actual, bool)
|
||||
and isfinite(estimated_actual)
|
||||
and estimated_actual >= 0
|
||||
else None
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def cache_path(cache_dir: Path, credentials: Credentials, session_id: str) -> Path:
|
||||
identity: Final = "\n".join((credentials.base_url, credentials.api_key, session_id))
|
||||
identity: Final = "\n".join(("whole-session-v1", credentials.base_url, credentials.api_key, session_id))
|
||||
return cache_dir / hashlib.sha256(identity.encode()).hexdigest()
|
||||
|
||||
|
||||
|
|
@ -342,34 +332,27 @@ def render(model: str, session: Session | None, config_dir: Path, use_color: boo
|
|||
routed: Final = paint(BOLD, f"Routed to: {model}")
|
||||
if session is None:
|
||||
return routed
|
||||
if session.savings_estimated_turns == 0 or session.baseline_spend is None:
|
||||
if session.baseline_spend is None:
|
||||
return f"{routed}{SEPARATOR}Savings unavailable"
|
||||
if session.baseline_model is None or session.baseline_spend <= 0:
|
||||
return routed
|
||||
if session.savings_estimated_turns is not None and (
|
||||
session.savings_estimated_actual_spend is None
|
||||
or session.turns is None
|
||||
or session.savings_estimated_turns > session.turns
|
||||
session.turns is None or session.savings_estimated_turns > session.turns
|
||||
):
|
||||
return f"{routed}{SEPARATOR}Savings unavailable"
|
||||
compared_spend: Final = (
|
||||
session.savings_estimated_actual_spend
|
||||
if session.savings_estimated_turns is not None and session.savings_estimated_actual_spend is not None
|
||||
else session.spend
|
||||
)
|
||||
coverage: Final = (
|
||||
f"{SEPARATOR}{session.savings_estimated_turns} of {session.turns} turns estimated"
|
||||
f"{SEPARATOR}baseline estimated for {session.savings_estimated_turns} of {session.turns} turns"
|
||||
if session.savings_estimated_turns is not None
|
||||
else ""
|
||||
)
|
||||
reference: Final = baseline_label(session.baseline_model, config_dir)
|
||||
pct: Final = round((session.baseline_spend - compared_spend) / session.baseline_spend * 100)
|
||||
pct: Final = round((session.baseline_spend - session.spend) / session.baseline_spend * 100)
|
||||
sign: Final = "-" if pct > 0 else "+" if pct < 0 else ""
|
||||
delta: Final = paint(LITELLM_COLOR, f"{sign}{abs(pct)}% vs {reference}")
|
||||
peak: Final = max(compared_spend, session.baseline_spend)
|
||||
peak: Final = max(session.spend, session.baseline_spend)
|
||||
label_width: Final = max(_display_width(session.router_name), _display_width(reference))
|
||||
rows: Final = (
|
||||
(session.router_name, compared_spend, LITELLM_COLOR),
|
||||
(session.router_name, session.spend, LITELLM_COLOR),
|
||||
(reference, session.baseline_spend, BASELINE_COLOR),
|
||||
)
|
||||
lines: Final = (
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
"""The status line script is copied verbatim to the user's machine, so these drive it the way Claude Code
|
||||
and Codex do: the documented stdin payload, a transcript on disk, and the proxy behind an injected fetch."""
|
||||
|
||||
import hashlib
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
|
|
@ -155,6 +156,19 @@ class TestCredentials:
|
|||
|
||||
|
||||
class TestSessionCache:
|
||||
def test_upgrade_does_not_reuse_a_cached_partial_baseline(self, tmp_path: Path) -> None:
|
||||
credentials: Final = Credentials("http://p", "sk-virtual")
|
||||
identity: Final = "\n".join((credentials.base_url, credentials.api_key, SESSION_ID))
|
||||
previous_cache: Final = tmp_path / hashlib.sha256(identity.encode()).hexdigest()
|
||||
tmp_path.chmod(0o700)
|
||||
previous_cache.write_text(
|
||||
json.dumps({"fetched_at": 1.0, "session": RECORDED._replace(spend=10.0)._asdict()})
|
||||
)
|
||||
whole_session: Final = RECORDED._replace(spend=10.0, baseline_spend=10.24)
|
||||
assert load_session(
|
||||
credentials, SESSION_ID, tmp_path, lambda c, s: Fetched(whole_session, True), now=lambda: 1.0
|
||||
) == whole_session
|
||||
|
||||
def test_a_definite_answer_is_served_from_the_cache_within_the_ttl(self, tmp_path):
|
||||
calls = []
|
||||
|
||||
|
|
@ -342,19 +356,39 @@ class TestRender:
|
|||
|
||||
|
||||
class TestClaudeCodeMode:
|
||||
@pytest.mark.parametrize("estimated_turns", (0, 1))
|
||||
def test_current_estimates_keep_the_routed_model_and_compare_only_covered_turns(
|
||||
self, tmp_path: Path, transcript: Path, config_dir: Path, estimated_turns: int
|
||||
@pytest.mark.parametrize(
|
||||
("estimated_turns", "baseline", "covered_actual", "covered_baseline", "delta", "amounts"),
|
||||
(
|
||||
(1, 9.5, 2.0, 1.5, "+5%", ("$10.00", "$9.50")),
|
||||
(1, 11.0, 1.0, 2.0, "-9%", ("$10.00", "$11.00")),
|
||||
(3, 9.5, 10.0, 9.5, "+5%", ("$10.00", "$9.50")),
|
||||
(0, 11.0, 0.0, None, "-9%", ("$10.00", "$11.00")),
|
||||
(0, None, 0.0, None, None, ()),
|
||||
(1, None, 2.0, 1.5, None, ()),
|
||||
),
|
||||
ids=("partial-loss", "partial-savings", "full", "legacy-savings", "unknown", "missing-total"),
|
||||
)
|
||||
def test_session_comparison_includes_all_spend_and_keeps_baseline_coverage(
|
||||
self,
|
||||
tmp_path: Path,
|
||||
transcript: Path,
|
||||
config_dir: Path,
|
||||
estimated_turns: int,
|
||||
baseline: float | None,
|
||||
covered_actual: float,
|
||||
covered_baseline: float | None,
|
||||
delta: str | None,
|
||||
amounts: tuple[str, ...],
|
||||
) -> None:
|
||||
session: Final = statusline_script._session_from_payload(
|
||||
{
|
||||
**RECORDED._asdict(),
|
||||
"spend": 10.0,
|
||||
"baseline_spend": None,
|
||||
"savings_estimated_baseline_spend": 1.5 if estimated_turns else None,
|
||||
"baseline_spend": baseline,
|
||||
"savings_estimated_baseline_spend": covered_baseline,
|
||||
"turns": 3,
|
||||
"savings_estimated_turns": estimated_turns,
|
||||
"savings_estimated_actual_spend": 2.0 if estimated_turns else 0.0,
|
||||
"savings_estimated_actual_spend": covered_actual,
|
||||
}
|
||||
)
|
||||
assert session is not None
|
||||
|
|
@ -364,14 +398,13 @@ class TestClaudeCodeMode:
|
|||
|
||||
first: Final = _run(_payload(transcript), _env(tmp_path, config_dir), fetch)
|
||||
assert first == _run(_payload(transcript), _env(tmp_path, config_dir), fetch)
|
||||
assert first.startswith("Routed to: claude-sonnet-5")
|
||||
if estimated_turns:
|
||||
assert "+33% vs Claude Opus 5 · 1 of 3 turns estimated" in first
|
||||
assert "$2.00" in first and "$1.50" in first
|
||||
assert "$10.00" not in first and "+567%" not in first
|
||||
else:
|
||||
assert "Savings unavailable" in first
|
||||
assert "%" not in first and "$" not in first
|
||||
comparison: Final = (
|
||||
f" {delta} vs Claude Opus 5 · baseline estimated for {estimated_turns} of 3 turns"
|
||||
if delta is not None
|
||||
else " · Savings unavailable"
|
||||
)
|
||||
assert first.splitlines()[0] == f"Routed to: claude-sonnet-5{comparison}"
|
||||
assert tuple(line.rsplit(" ", 1)[-1] for line in first.splitlines()[1:]) == amounts
|
||||
|
||||
@pytest.mark.parametrize("transcript_model", ("claude-auto", "anthropic/claude-opus-5"))
|
||||
def test_the_session_names_the_routed_model_even_when_the_transcript_differs(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue