Merge pull request #41116 from BerriAI/litellm_auto_router_session_api_access

fix(proxy): allow LLM API keys to read auto-router sessions
This commit is contained in:
tin-berri 2026-09-14 13:49:03 -07:00 • committed by GitHub
commit 2cad7a49af
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 83 additions and 24 deletions

View file

@ -136,6 +136,9 @@ class RouteChecks:
# For llm_api_routes, also check registered pass-through endpoints
################################################
if allowed_route == "llm_api_routes":
if route == "/auto_router/session" and RouteChecks._get_request_method(request) == "GET":
return True
from litellm.proxy.pass_through_endpoints.pass_through_endpoints import (
InitPassThroughEndpointHelpers,
)

View file

@ -585,7 +585,9 @@ LiteLLM ████████░░░░░░░░░░░░░░
Claude Opus 5 ████████████████████████ $0.38
```
The routed model comes from Claude Code's own transcript, so it only names the tier model when the auto-router deployment sets `return_raw_model_name: true` (the `lite autoroute` wizard does); otherwise it shows the alias you requested. The cost lines come from `GET /auto_router/session?session_id=...`, which any virtual key may call for its own sessions, and are cached for five seconds under a per-user `$TMPDIR/litellm-statusline-<uid>` directory. The baseline is the priciest model in the router's hardest tier, the same counterfactual the auto-router's savings reports use. `lite unconfigure claude` removes the `statusLine` entry only while it still points at that script.
After the first response, the status line uses the latest routed model recorded by `GET /auto_router/session?session_id=...`, so it can show the tier model even when the transcript contains the router alias. If no session record is available, it falls back to Claude Code's transcript. Session records and costs are cached for five seconds under a per-user `$TMPDIR/litellm-statusline-<uid>` directory. The gateway records turns asynchronously, so the display can briefly lag a completed turn. Any virtual key may read its own sessions. The baseline is the priciest model in the router's hardest tier, the same counterfactual the auto-router's savings reports use. `lite unconfigure claude` removes the `statusLine` entry only while it still points at that script
After upgrading the CLI, rerun your original `lite configure claude` command with the same gateway, key and model choice to refresh `~/.litellm/statusline.py`. Keep any explicit `--model` value: omitting it removes the earlier model pin. Package upgrades alone do not refresh this installed copy
`lite codex` registers the same script as a Codex `Stop` hook for the launch, so after each turn Codex prints the same block as a system message. Codex asks once to trust the hook; the answer is remembered for later launches.

View file

@ -7,19 +7,17 @@ status refresh (about every 300ms while typing), so the proxy is asked at most o
TTL per session and every other refresh is served from a small on-disk cache that holds
only the proxy's answer, never the key.
Claude Code pipes a JSON payload on stdin (session_id, transcript_path, model); the routed
model is the `message.model` of the latest foreground assistant line in the transcript,
which is the proxy's response `model` field. That only names the tier model when the
auto-router deployment sets `return_raw_model_name: true`; otherwise it is the alias the
client requested. Codex pipes its Stop event instead (hook_event_name, session_id) and has
no transcript to read, so the routed model comes from the proxy's session record and the
result is printed as a `systemMessage` for the transcript. The proxy key is read from the
agent's own environment (the static token `lite configure claude` writes); nothing here
spawns a credential helper.
Claude Code pipes a JSON payload on stdin (session_id, transcript_path, model). After the
first foreground assistant response, the routed model comes from the proxy's session
record, falling back to the latest foreground assistant `message.model` in the transcript
when no record is available. Codex pipes its Stop event instead (hook_event_name, session_id)
and prints the session record as a `systemMessage` for the transcript. The proxy key is read
from the agent's own environment (the static token `lite configure claude` writes); nothing
here spawns a credential helper.
Cost figures come from GET /auto_router/session on the proxy, which reads the per-session
rollup written by the spend flush. That flush is asynchronous, so a turn's cost lands a
second or two after the turn; the cache TTL absorbs it.
The routed model and cost figures come from GET /auto_router/session on the proxy, which
reads the per-session rollup written by the asynchronous spend flush. The record and cache
can briefly lag a completed turn.
"""
from __future__ import annotations
@ -348,7 +346,8 @@ def status_line(
if not session_id or not credentials.usable:
return render(label, None, config_dir, color_enabled(env))
session: Final = load_session(credentials, session_id, cache_dir, fetch)
return render(label, session, config_dir, color_enabled(env))
routed_label: Final = model_label(session.last_model, config_dir) if session is not None else label
return render(routed_label, session, config_dir, color_enabled(env))
def codex_stop_message(

View file

@ -1,5 +1,6 @@
import os
from datetime import datetime
from typing import Final
from unittest.mock import MagicMock, patch
@ -3919,10 +3920,15 @@ def test_claude_code_marketplace_routes_open_to_internal_users(route):
@pytest.mark.parametrize("user_role", [None, LitellmUserRoles.INTERNAL_USER.value, LitellmUserRoles.INTERNAL_USER_VIEW_ONLY.value])
def test_auto_router_session_is_reachable_by_any_key_but_benchmarks_stays_admin_only(user_role):
valid_token = UserAPIKeyAuth(api_key="hash-of-caller", user_role=user_role)
request = MagicMock(spec=Request)
request.query_params = {"session_id": "sess-1"}
@pytest.mark.parametrize("allowed_routes", [None, ["llm_api_routes"]])
def test_auto_router_session_is_reachable_by_any_key_but_benchmarks_stays_admin_only(
user_role: str | None, allowed_routes: list[str] | None
) -> None:
valid_token: Final = UserAPIKeyAuth(api_key="hash-of-caller", user_role=user_role, allowed_routes=allowed_routes)
request: Final = Request({"type": "http", "method": "GET", "query_string": b"session_id=sess-1"})
assert RouteChecks.should_call_route("/auto_router/session", valid_token, request) is True
assert RouteChecks.is_llm_api_route("/auto_router/session") is False
RouteChecks.non_proxy_admin_allowed_routes_check(
user_obj=None,
@ -3941,3 +3947,36 @@ def test_auto_router_session_is_reachable_by_any_key_but_benchmarks_stays_admin_
valid_token=valid_token,
request_data={},
)
@pytest.mark.parametrize(
"route,method,allowed_routes",
[
("/auto_router/session", method, ["llm_api_routes"])
for method in ("POST", "PUT", "PATCH", "DELETE", "HEAD", "OPTIONS", None)
]
+ [
(route, "GET", ["llm_api_routes"])
for route in (
"/auto_router/benchmarks",
"/auto_router/test_routing",
"/auto_router/validate_complexity_router_config",
"/auto_router/session/other",
"/auto_router/sessions",
)
]
+ [
("/auto_router/session", "GET", allowed_routes)
for allowed_routes in (["/v1/messages"], ["info_routes"], ["openai_routes"])
],
)
def test_auto_router_session_read_grant_rejects_other_methods_paths_and_scopes(
route: str, method: str | None, allowed_routes: list[str]
) -> None:
valid_token: Final = UserAPIKeyAuth(api_key="hash-of-caller", allowed_routes=allowed_routes)
request: Final = Request({"type": "http", "method": method}) if method is not None else None
with pytest.raises(HTTPException) as error:
RouteChecks.should_call_route(route, valid_token, request)
assert error.value.status_code == 403

View file

@ -8,6 +8,7 @@ import re
import subprocess
import sys
from pathlib import Path
from typing import Final
import pytest
@ -297,16 +298,31 @@ class TestRender:
class TestClaudeCodeMode:
def test_the_transcript_names_the_routed_model_and_the_proxy_adds_the_savings(self, tmp_path, transcript, config_dir):
seen = []
@pytest.mark.parametrize("transcript_model", ("claude-auto", "anthropic/claude-opus-5"))
def test_the_session_names_the_routed_model_even_when_the_transcript_differs(
self, tmp_path: Path, config_dir: Path, transcript_model: str
) -> None:
transcript: Final = tmp_path / "session.jsonl"
transcript.write_text(_assistant_line(transcript_model) + "\n")
def fetch(credentials, session_id):
seen.append((credentials, session_id))
def fetch(credentials: Credentials, session_id: str) -> Fetched:
assert credentials == Credentials("http://127.0.0.1:4000", "sk-virtual")
assert session_id == SESSION_ID
return Fetched(RECORDED, definitive=True)
text = _run(_payload(transcript), _env(tmp_path, config_dir), fetch)
text: Final = _run(_payload(transcript), _env(tmp_path, config_dir), fetch)
assert text.startswith("claude-auto · Routed to: claude-sonnet-5 -63% vs Claude Opus 5\n")
assert seen == [(Credentials("http://127.0.0.1:4000", "sk-virtual"), SESSION_ID)]
def test_a_discovered_display_name_labels_the_sessions_model(
self, tmp_path: Path, transcript: Path, config_dir: Path
) -> None:
session: Final = RECORDED._replace(last_model="anthropic/claude-opus-5")
def fetch(credentials: Credentials, session_id: str) -> Fetched:
return Fetched(session, definitive=True)
text: Final = _run(_payload(transcript), _env(tmp_path, config_dir), fetch)
assert text.startswith("claude-auto · Routed to: Claude Opus 5 -63% vs Claude Opus 5\n")
def test_an_unrecorded_session_degrades_to_the_routed_line(self, tmp_path, transcript, config_dir):
assert _run(_payload(transcript), _env(tmp_path, config_dir), lambda c, s: Fetched(None, True)) == (