fix(cli): show full-session auto-router cost comparison (#44959)

This commit is contained in:
tin-berri 2026-10-06 16:24:14 -07:00 • committed by GitHub
parent 3c79db6122
commit 9a3f000c9b
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 55 additions and 39 deletions

View file

@ -69,7 +69,6 @@ class Session(NamedTuple):
baseline_model: str | None
turns: int | None = None
savings_estimated_turns: int | None = None
savings_estimated_actual_spend: float | None = None
class Credentials(NamedTuple):
@ -210,10 +209,9 @@ def _session_from_payload(payload: Mapping[str, object]) -> Session | None:
router_name: Final = printable(payload.get("router_name"))
last_model: Final = printable(payload.get("last_model"))
spend: Final = payload.get("spend")
baseline_spend: Final = payload.get("savings_estimated_baseline_spend", payload.get("baseline_spend"))
baseline_spend: Final = payload.get("baseline_spend")
turns: Final = payload.get("turns")
estimated_turns: Final = payload.get("savings_estimated_turns")
estimated_actual: Final = payload.get("savings_estimated_actual_spend")
if not router_name or not last_model:
return None
if not isinstance(spend, (int, float)) or isinstance(spend, bool) or not isfinite(spend):
@ -234,19 +232,11 @@ def _session_from_payload(payload: Mapping[str, object]) -> Session | None:
if isinstance(estimated_turns, int) and not isinstance(estimated_turns, bool) and estimated_turns >= 0
else (0 if estimated_turns is not None else None)
),
savings_estimated_actual_spend=(
float(estimated_actual)
if isinstance(estimated_actual, (int, float))
and not isinstance(estimated_actual, bool)
and isfinite(estimated_actual)
and estimated_actual >= 0
else None
),
)
def cache_path(cache_dir: Path, credentials: Credentials, session_id: str) -> Path:
identity: Final = "\n".join((credentials.base_url, credentials.api_key, session_id))
identity: Final = "\n".join(("whole-session-v1", credentials.base_url, credentials.api_key, session_id))
return cache_dir / hashlib.sha256(identity.encode()).hexdigest()
@ -342,34 +332,27 @@ def render(model: str, session: Session | None, config_dir: Path, use_color: boo
routed: Final = paint(BOLD, f"Routed to: {model}")
if session is None:
return routed
if session.savings_estimated_turns == 0 or session.baseline_spend is None:
if session.baseline_spend is None:
return f"{routed}{SEPARATOR}Savings unavailable"
if session.baseline_model is None or session.baseline_spend <= 0:
return routed
if session.savings_estimated_turns is not None and (
session.savings_estimated_actual_spend is None
or session.turns is None
or session.savings_estimated_turns > session.turns
session.turns is None or session.savings_estimated_turns > session.turns
):
return f"{routed}{SEPARATOR}Savings unavailable"
compared_spend: Final = (
session.savings_estimated_actual_spend
if session.savings_estimated_turns is not None and session.savings_estimated_actual_spend is not None
else session.spend
)
coverage: Final = (
f"{SEPARATOR}{session.savings_estimated_turns} of {session.turns} turns estimated"
f"{SEPARATOR}baseline estimated for {session.savings_estimated_turns} of {session.turns} turns"
if session.savings_estimated_turns is not None
else ""
)
reference: Final = baseline_label(session.baseline_model, config_dir)
pct: Final = round((session.baseline_spend - compared_spend) / session.baseline_spend * 100)
pct: Final = round((session.baseline_spend - session.spend) / session.baseline_spend * 100)
sign: Final = "-" if pct > 0 else "+" if pct < 0 else ""
delta: Final = paint(LITELLM_COLOR, f"{sign}{abs(pct)}% vs {reference}")
peak: Final = max(compared_spend, session.baseline_spend)
peak: Final = max(session.spend, session.baseline_spend)
label_width: Final = max(_display_width(session.router_name), _display_width(reference))
rows: Final = (
(session.router_name, compared_spend, LITELLM_COLOR),
(session.router_name, session.spend, LITELLM_COLOR),
(reference, session.baseline_spend, BASELINE_COLOR),
)
lines: Final = (

View file

@ -1,6 +1,7 @@
"""The status line script is copied verbatim to the user's machine, so these drive it the way Claude Code
and Codex do: the documented stdin payload, a transcript on disk, and the proxy behind an injected fetch."""
import hashlib
import io
import json
import os
@ -155,6 +156,19 @@ class TestCredentials:
class TestSessionCache:
def test_upgrade_does_not_reuse_a_cached_partial_baseline(self, tmp_path: Path) -> None:
credentials: Final = Credentials("http://p", "sk-virtual")
identity: Final = "\n".join((credentials.base_url, credentials.api_key, SESSION_ID))
previous_cache: Final = tmp_path / hashlib.sha256(identity.encode()).hexdigest()
tmp_path.chmod(0o700)
previous_cache.write_text(
json.dumps({"fetched_at": 1.0, "session": RECORDED._replace(spend=10.0)._asdict()})
)
whole_session: Final = RECORDED._replace(spend=10.0, baseline_spend=10.24)
assert load_session(
credentials, SESSION_ID, tmp_path, lambda c, s: Fetched(whole_session, True), now=lambda: 1.0
) == whole_session
def test_a_definite_answer_is_served_from_the_cache_within_the_ttl(self, tmp_path):
calls = []
@ -342,19 +356,39 @@ class TestRender:
class TestClaudeCodeMode:
@pytest.mark.parametrize("estimated_turns", (0, 1))
def test_current_estimates_keep_the_routed_model_and_compare_only_covered_turns(
self, tmp_path: Path, transcript: Path, config_dir: Path, estimated_turns: int
@pytest.mark.parametrize(
("estimated_turns", "baseline", "covered_actual", "covered_baseline", "delta", "amounts"),
(
(1, 9.5, 2.0, 1.5, "+5%", ("$10.00", "$9.50")),
(1, 11.0, 1.0, 2.0, "-9%", ("$10.00", "$11.00")),
(3, 9.5, 10.0, 9.5, "+5%", ("$10.00", "$9.50")),
(0, 11.0, 0.0, None, "-9%", ("$10.00", "$11.00")),
(0, None, 0.0, None, None, ()),
(1, None, 2.0, 1.5, None, ()),
),
ids=("partial-loss", "partial-savings", "full", "legacy-savings", "unknown", "missing-total"),
)
def test_session_comparison_includes_all_spend_and_keeps_baseline_coverage(
self,
tmp_path: Path,
transcript: Path,
config_dir: Path,
estimated_turns: int,
baseline: float | None,
covered_actual: float,
covered_baseline: float | None,
delta: str | None,
amounts: tuple[str, ...],
) -> None:
session: Final = statusline_script._session_from_payload(
{
**RECORDED._asdict(),
"spend": 10.0,
"baseline_spend": None,
"savings_estimated_baseline_spend": 1.5 if estimated_turns else None,
"baseline_spend": baseline,
"savings_estimated_baseline_spend": covered_baseline,
"turns": 3,
"savings_estimated_turns": estimated_turns,
"savings_estimated_actual_spend": 2.0 if estimated_turns else 0.0,
"savings_estimated_actual_spend": covered_actual,
}
)
assert session is not None
@ -364,14 +398,13 @@ class TestClaudeCodeMode:
first: Final = _run(_payload(transcript), _env(tmp_path, config_dir), fetch)
assert first == _run(_payload(transcript), _env(tmp_path, config_dir), fetch)
assert first.startswith("Routed to: claude-sonnet-5")
if estimated_turns:
assert "+33% vs Claude Opus 5 · 1 of 3 turns estimated" in first
assert "$2.00" in first and "$1.50" in first
assert "$10.00" not in first and "+567%" not in first
else:
assert "Savings unavailable" in first
assert "%" not in first and "$" not in first
comparison: Final = (
f" {delta} vs Claude Opus 5 · baseline estimated for {estimated_turns} of 3 turns"
if delta is not None
else " · Savings unavailable"
)
assert first.splitlines()[0] == f"Routed to: claude-sonnet-5{comparison}"
assert tuple(line.rsplit(" ", 1)[-1] for line in first.splitlines()[1:]) == amounts
@pytest.mark.parametrize("transcript_model", ("claude-auto", "anthropic/claude-opus-5"))
def test_the_session_names_the_routed_model_even_when_the_transcript_differs(