diff --git a/litellm/proxy/client/cli/commands/statusline_script.py b/litellm/proxy/client/cli/commands/statusline_script.py index f137165a6e5..5aeefa0c739 100644 --- a/litellm/proxy/client/cli/commands/statusline_script.py +++ b/litellm/proxy/client/cli/commands/statusline_script.py @@ -69,7 +69,6 @@ class Session(NamedTuple): baseline_model: str | None turns: int | None = None savings_estimated_turns: int | None = None - savings_estimated_actual_spend: float | None = None class Credentials(NamedTuple): @@ -210,10 +209,9 @@ def _session_from_payload(payload: Mapping[str, object]) -> Session | None: router_name: Final = printable(payload.get("router_name")) last_model: Final = printable(payload.get("last_model")) spend: Final = payload.get("spend") - baseline_spend: Final = payload.get("savings_estimated_baseline_spend", payload.get("baseline_spend")) + baseline_spend: Final = payload.get("baseline_spend") turns: Final = payload.get("turns") estimated_turns: Final = payload.get("savings_estimated_turns") - estimated_actual: Final = payload.get("savings_estimated_actual_spend") if not router_name or not last_model: return None if not isinstance(spend, (int, float)) or isinstance(spend, bool) or not isfinite(spend): @@ -234,19 +232,11 @@ def _session_from_payload(payload: Mapping[str, object]) -> Session | None: if isinstance(estimated_turns, int) and not isinstance(estimated_turns, bool) and estimated_turns >= 0 else (0 if estimated_turns is not None else None) ), - savings_estimated_actual_spend=( - float(estimated_actual) - if isinstance(estimated_actual, (int, float)) - and not isinstance(estimated_actual, bool) - and isfinite(estimated_actual) - and estimated_actual >= 0 - else None - ), ) def cache_path(cache_dir: Path, credentials: Credentials, session_id: str) -> Path: - identity: Final = "\n".join((credentials.base_url, credentials.api_key, session_id)) + identity: Final = "\n".join(("whole-session-v1", credentials.base_url, credentials.api_key, session_id)) return cache_dir / hashlib.sha256(identity.encode()).hexdigest() @@ -342,34 +332,27 @@ def render(model: str, session: Session | None, config_dir: Path, use_color: boo routed: Final = paint(BOLD, f"Routed to: {model}") if session is None: return routed - if session.savings_estimated_turns == 0 or session.baseline_spend is None: + if session.baseline_spend is None: return f"{routed}{SEPARATOR}Savings unavailable" if session.baseline_model is None or session.baseline_spend <= 0: return routed if session.savings_estimated_turns is not None and ( - session.savings_estimated_actual_spend is None - or session.turns is None - or session.savings_estimated_turns > session.turns + session.turns is None or session.savings_estimated_turns > session.turns ): return f"{routed}{SEPARATOR}Savings unavailable" - compared_spend: Final = ( - session.savings_estimated_actual_spend - if session.savings_estimated_turns is not None and session.savings_estimated_actual_spend is not None - else session.spend - ) coverage: Final = ( - f"{SEPARATOR}{session.savings_estimated_turns} of {session.turns} turns estimated" + f"{SEPARATOR}baseline estimated for {session.savings_estimated_turns} of {session.turns} turns" if session.savings_estimated_turns is not None else "" ) reference: Final = baseline_label(session.baseline_model, config_dir) - pct: Final = round((session.baseline_spend - compared_spend) / session.baseline_spend * 100) + pct: Final = round((session.baseline_spend - session.spend) / session.baseline_spend * 100) sign: Final = "-" if pct > 0 else "+" if pct < 0 else "" delta: Final = paint(LITELLM_COLOR, f"{sign}{abs(pct)}% vs {reference}") - peak: Final = max(compared_spend, session.baseline_spend) + peak: Final = max(session.spend, session.baseline_spend) label_width: Final = max(_display_width(session.router_name), _display_width(reference)) rows: Final = ( - (session.router_name, compared_spend, LITELLM_COLOR), + (session.router_name, session.spend, LITELLM_COLOR), (reference, session.baseline_spend, BASELINE_COLOR), ) lines: Final = ( diff --git a/tests/unit/proxy/client/cli/test_statusline_script.py b/tests/unit/proxy/client/cli/test_statusline_script.py index 0cbeec86ee8..3486bf1c132 100644 --- a/tests/unit/proxy/client/cli/test_statusline_script.py +++ b/tests/unit/proxy/client/cli/test_statusline_script.py @@ -1,6 +1,7 @@ """The status line script is copied verbatim to the user's machine, so these drive it the way Claude Code and Codex do: the documented stdin payload, a transcript on disk, and the proxy behind an injected fetch.""" +import hashlib import io import json import os @@ -155,6 +156,19 @@ class TestCredentials: class TestSessionCache: + def test_upgrade_does_not_reuse_a_cached_partial_baseline(self, tmp_path: Path) -> None: + credentials: Final = Credentials("http://p", "sk-virtual") + identity: Final = "\n".join((credentials.base_url, credentials.api_key, SESSION_ID)) + previous_cache: Final = tmp_path / hashlib.sha256(identity.encode()).hexdigest() + tmp_path.chmod(0o700) + previous_cache.write_text( + json.dumps({"fetched_at": 1.0, "session": RECORDED._replace(spend=10.0)._asdict()}) + ) + whole_session: Final = RECORDED._replace(spend=10.0, baseline_spend=10.24) + assert load_session( + credentials, SESSION_ID, tmp_path, lambda c, s: Fetched(whole_session, True), now=lambda: 1.0 + ) == whole_session + def test_a_definite_answer_is_served_from_the_cache_within_the_ttl(self, tmp_path): calls = [] @@ -342,19 +356,39 @@ class TestRender: class TestClaudeCodeMode: - @pytest.mark.parametrize("estimated_turns", (0, 1)) - def test_current_estimates_keep_the_routed_model_and_compare_only_covered_turns( - self, tmp_path: Path, transcript: Path, config_dir: Path, estimated_turns: int + @pytest.mark.parametrize( + ("estimated_turns", "baseline", "covered_actual", "covered_baseline", "delta", "amounts"), + ( + (1, 9.5, 2.0, 1.5, "+5%", ("$10.00", "$9.50")), + (1, 11.0, 1.0, 2.0, "-9%", ("$10.00", "$11.00")), + (3, 9.5, 10.0, 9.5, "+5%", ("$10.00", "$9.50")), + (0, 11.0, 0.0, None, "-9%", ("$10.00", "$11.00")), + (0, None, 0.0, None, None, ()), + (1, None, 2.0, 1.5, None, ()), + ), + ids=("partial-loss", "partial-savings", "full", "legacy-savings", "unknown", "missing-total"), + ) + def test_session_comparison_includes_all_spend_and_keeps_baseline_coverage( + self, + tmp_path: Path, + transcript: Path, + config_dir: Path, + estimated_turns: int, + baseline: float | None, + covered_actual: float, + covered_baseline: float | None, + delta: str | None, + amounts: tuple[str, ...], ) -> None: session: Final = statusline_script._session_from_payload( { **RECORDED._asdict(), "spend": 10.0, - "baseline_spend": None, - "savings_estimated_baseline_spend": 1.5 if estimated_turns else None, + "baseline_spend": baseline, + "savings_estimated_baseline_spend": covered_baseline, "turns": 3, "savings_estimated_turns": estimated_turns, - "savings_estimated_actual_spend": 2.0 if estimated_turns else 0.0, + "savings_estimated_actual_spend": covered_actual, } ) assert session is not None @@ -364,14 +398,13 @@ class TestClaudeCodeMode: first: Final = _run(_payload(transcript), _env(tmp_path, config_dir), fetch) assert first == _run(_payload(transcript), _env(tmp_path, config_dir), fetch) - assert first.startswith("Routed to: claude-sonnet-5") - if estimated_turns: - assert "+33% vs Claude Opus 5 · 1 of 3 turns estimated" in first - assert "$2.00" in first and "$1.50" in first - assert "$10.00" not in first and "+567%" not in first - else: - assert "Savings unavailable" in first - assert "%" not in first and "$" not in first + comparison: Final = ( + f" {delta} vs Claude Opus 5 · baseline estimated for {estimated_turns} of 3 turns" + if delta is not None + else " · Savings unavailable" + ) + assert first.splitlines()[0] == f"Routed to: claude-sonnet-5{comparison}" + assert tuple(line.rsplit(" ", 1)[-1] for line in first.splitlines()[1:]) == amounts @pytest.mark.parametrize("transcript_model", ("claude-auto", "anthropic/claude-opus-5")) def test_the_session_names_the_routed_model_even_when_the_transcript_differs(