fix(cli/anthropic): unblock lite autoroute proxy deps, adaptive thinking, and thinking+signature streaming (#33507)

This commit is contained in:
devin-ai-integration[bot] 2026-07-16 00:44:00 -07:00 • committed by GitHub
parent 5a0e1dd1dd
commit ebc6fdb4c2
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
8 changed files with 207 additions and 6 deletions

View file

@ -495,19 +495,21 @@ Cursor is not supported: it has no equivalent file-based config to hot-patch thi
#### Install the CLI
If you don't already have the `lite` command, install it with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
`lite autoroute up` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
```bash
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/main/scripts/install-cli.sh | sh
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/main/scripts/install.sh | sh
```
This installs only `litellm[cli]`, the thin client (`lite`), not the full proxy server. To try an unreleased branch or commit instead of the latest PyPI release, set `LITELLM_CLI_REF`:
To QA an unreleased branch or commit instead of the latest PyPI release, set `LITELLM_CLI_REF`:
```bash
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch-or-commit>/scripts/install-cli.sh | \
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch-or-commit>/scripts/install.sh | \
LITELLM_CLI_REF=<branch-or-commit> sh
```
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute up`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
Point the CLI at your real proxy and key before running any `lite model-groups` or `lite autoroute` command -- like every other command in this CLI, they read `LITELLM_PROXY_URL`/`LITELLM_PROXY_API_KEY` (or `--base-url`/`--api-key`), no `lite login` required:
```bash

View file

@ -21,6 +21,7 @@ from .process import (
clear_pid_record,
is_running,
launch_proxy,
missing_proxy_runtime_modules,
poll_liveliness,
read_pid_record,
secure_create,
@ -81,6 +82,16 @@ def up() -> None:
if not CONFIG_PATH.exists():
raise click.ClickException("No config found. Run `lite autoroute configure` first.")
missing = missing_proxy_runtime_modules()
if missing:
raise click.ClickException(
"lite autoroute up launches a local litellm proxy, which needs the proxy runtime that the "
f"thin `litellm[cli]` install does not include (missing: {', '.join(missing)}). Install the "
"proxy runtime with `uv tool install --force 'litellm[proxy]'`, or to QA a branch, "
"`curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch>/scripts/install.sh | "
"LITELLM_CLI_REF=<branch> sh`."
)
try:
existing_pid = read_pid_record()
except UpError as e:

View file

@ -1,4 +1,5 @@
import contextlib
import importlib.util
import json
import os
import signal
@ -37,6 +38,20 @@ class PidRecord:
_PID_RECORD_ADAPTER = TypeAdapter(PidRecord)
_PROXY_RUNTIME_MODULES: tuple[str, ...] = ("fastapi", "uvicorn", "backoff", "orjson", "websockets", "apscheduler")
def missing_proxy_runtime_modules() -> tuple[str, ...]:
"""Proxy-server modules that ``lite autoroute up`` needs but the thin CLI install lacks.
``launch_proxy`` runs the full ``litellm.proxy.proxy_cli`` server, whose dependencies live in
the ``proxy`` extra, not the ``cli`` extra that installs the ``lite`` command. On a thin
``litellm[cli]`` install the subprocess dies with a bare ``ModuleNotFoundError``; detecting the
gap here lets ``up`` fail with an actionable message instead.
"""
return tuple(name for name in _PROXY_RUNTIME_MODULES if importlib.util.find_spec(name) is None)
def allocate_free_port() -> int:
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
sock.bind(("127.0.0.1", 0))
@ -165,6 +180,7 @@ __all__ = [
"clear_pid_record",
"is_running",
"launch_proxy",
"missing_proxy_runtime_modules",
"poll_liveliness",
"read_pid_record",
"secure_create",

View file

@ -5,12 +5,24 @@
# Needs only curl: uv is bootstrapped if missing, and uv provisions a compatible
# Python itself (reusing a suitable system one, else downloading a managed build).
#
# To install from an unreleased branch, tag, or commit instead of the latest PyPI
# release, set LITELLM_CLI_REF:
# curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch>/scripts/install.sh | \
# LITELLM_CLI_REF=<branch> sh
#
# NOTE: set -e without pipefail for POSIX sh compatibility (dash on Ubuntu/Debian
# ignores the shebang when invoked as `sh` and does not support `pipefail`).
set -eu
# NOTE: before merging, this must stay as "litellm[proxy]" to install from PyPI.
LITELLM_PACKAGE="litellm[proxy]"
# LITELLM_CLI_REF opts into installing from a branch, tag, or commit instead (for
# example, to QA lite autoroute against an unreleased branch, which needs this proxy
# runtime, not the thin litellm[cli] install).
if [ -n "${LITELLM_CLI_REF:-}" ]; then
LITELLM_PACKAGE="litellm[proxy] @ git+https://github.com/BerriAI/litellm.git@${LITELLM_CLI_REF}"
else
LITELLM_PACKAGE="litellm[proxy]"
fi
UV_VERSION="0.10.9"
# ── colours ────────────────────────────────────────────────────────────────
@ -81,7 +93,11 @@ fi
# ── install ────────────────────────────────────────────────────────────────
echo ""
header "Installing litellm[proxy]…"
if [ -n "${LITELLM_CLI_REF:-}" ]; then
header "Installing litellm[proxy] from ${LITELLM_CLI_REF}…"
else
header "Installing litellm[proxy]…"
fi
echo ""
# --python-preference system: reuse a compatible system Python when present,

View file

@ -12,6 +12,7 @@ content survives.
import asyncio
import json
from types import SimpleNamespace
from typing import AsyncIterator
from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import (
AnthropicStreamWrapper,
@ -202,3 +203,89 @@ def test_split_clears_reasoning_and_thinking_on_finish_chunk():
assert content_chunk.choices[0].delta.thinking_blocks == [{"type": "thinking"}]
assert finish_chunk.choices[0].delta.reasoning_content is None
assert finish_chunk.choices[0].delta.thinking_blocks is None
def _thinking_delta_chunk(thinking: str) -> ModelResponseStream:
return ModelResponseStream(
choices=[
StreamingChoices(
index=0,
delta=Delta(
reasoning_content=thinking,
thinking_blocks=[{"type": "thinking", "thinking": thinking, "signature": None}],
provider_specific_fields={
"thinking_blocks": [{"type": "thinking", "thinking": thinking, "signature": None}]
},
),
finish_reason=None,
)
],
)
def _signature_chunk(recap: str, signature: str) -> ModelResponseStream:
return ModelResponseStream(
choices=[
StreamingChoices(
index=0,
delta=Delta(
reasoning_content=recap,
thinking_blocks=[{"type": "thinking", "thinking": recap, "signature": signature}],
provider_specific_fields={
"thinking_blocks": [{"type": "thinking", "thinking": recap, "signature": signature}]
},
),
finish_reason=None,
)
],
)
def test_thinking_then_signature_chunk_does_not_crash_stream():
"""Regression for the /v1/messages streaming crash reported on autoroute.
Anthropic streams extended thinking as incremental ``thinking_delta`` chunks, then a
closing chunk that recaps the full accumulated thinking AND carries the signature. The
adapter used to raise ``ValueError`` on that closing chunk, killing the whole stream. It
must instead emit a single ``signature_delta`` for the recap chunk and never re-emit the
recap thinking, so the incremental thinking text is not duplicated.
"""
chunks = [
_thinking_delta_chunk("First, "),
_thinking_delta_chunk("reason."),
_signature_chunk("First, reason.", "sig-abc"),
ModelResponseStream(
choices=[StreamingChoices(index=0, delta=Delta(content="Done"), finish_reason=None)],
),
ModelResponseStream(
choices=[StreamingChoices(index=0, delta=Delta(), finish_reason="stop")],
usage=Usage(prompt_tokens=5, completion_tokens=3, total_tokens=8),
),
]
async def _aiter() -> "AsyncIterator[ModelResponseStream]":
for chunk in chunks:
yield chunk
wrapper = AnthropicStreamWrapper(completion_stream=_aiter(), model="claude-haiku-4-5")
sse = _collect_async(wrapper)
signature_deltas = [
json.loads(line[len("data: ") :])
for block in sse.split("\n\n")
for line in block.splitlines()
if line.startswith("data: ") and '"signature_delta"' in line
]
assert len(signature_deltas) == 1
assert signature_deltas[0]["delta"]["signature"] == "sig-abc"
thinking_text = "".join(
json.loads(line[len("data: ") :])["delta"]["thinking"]
for block in sse.split("\n\n")
for line in block.splitlines()
if line.startswith("data: ") and '"thinking_delta"' in line
)
assert thinking_text == "First, reason."
assert "message_stop" in sse
assert "Done" in sse

View file

@ -32,6 +32,37 @@ def _transform(model, params, litellm_params=None):
)
def test_adaptive_thinking_only_translated_to_legacy_for_haiku_4_5():
"""The minimal autoroute repro: Claude Code sends bare ``thinking={type: adaptive}``
(no ``output_config``) and the complexity router picks Haiku 4.5, which does not
support adaptive thinking. Anthropic 400s with "adaptive thinking is not supported on
this model" unless the flag is dropped, so it must be translated to the legacy extended
thinking the model does support rather than forwarded raw."""
result = _transform("claude-haiku-4-5", {"max_tokens": 8192, "thinking": {"type": "adaptive"}})
assert result["thinking"] == {
"type": "enabled",
"budget_tokens": DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
}
assert "output_config" not in result
def test_adaptive_thinking_only_dropped_for_non_reasoning_model():
"""Bare adaptive thinking on a model with no reasoning support at all is silently
dropped so the request still succeeds instead of being rejected."""
result = _transform("claude-3-5-haiku-latest", {"max_tokens": 8192, "thinking": {"type": "adaptive"}})
assert "thinking" not in result
def test_adaptive_thinking_only_preserved_for_4_6():
"""A 4.6+ model natively supports adaptive thinking, so a bare adaptive flag must not
be rewritten even without output_config."""
result = _transform("claude-sonnet-4-6", {"max_tokens": 8192, "thinking": {"type": "adaptive"}})
assert result["thinking"] == {"type": "adaptive"}
def test_effort_translated_to_legacy_thinking_for_haiku_4_5():
"""Core regression: Claude Code sends adaptive thinking + effort to Haiku 4.5
(thinking-capable, pre-4.6). Effort must be translated to legacy extended

View file

@ -69,6 +69,25 @@ class TestUpCommand:
assert result.exception is None or isinstance(result.exception, SystemExit)
assert "lite autoroute configure" in result.output
def test_refuses_with_actionable_error_when_proxy_runtime_missing(self, monkeypatch, tmp_path):
"""`up` launches a real litellm proxy, which the thin `litellm[cli]` install cannot run.
It must fail fast with an actionable message pointing at the proxy install, before it ever
tries to launch the doomed subprocess (which would otherwise die with a bare ImportError)."""
config_path, _log_path, _settings_path, _backup_path, _pid_record_path = _patch_paths(monkeypatch, tmp_path)
config_path.write_text(yaml.safe_dump({"model_list": []}))
monkeypatch.setattr(commands_module, "missing_proxy_runtime_modules", lambda: ("fastapi", "websockets"))
def _fail_if_launched(*args, **kwargs):
raise AssertionError("launch_proxy must not run when the proxy runtime is missing")
monkeypatch.setattr(commands_module, "launch_proxy", _fail_if_launched)
result = self.runner.invoke(up)
assert result.exit_code != 0
assert "fastapi, websockets" in result.output
assert "litellm[proxy]" in result.output
def test_refuses_when_pid_record_exists_and_process_still_running(self, monkeypatch, tmp_path):
config_path, _log_path, _settings_path, _backup_path, pid_record_path = _patch_paths(monkeypatch, tmp_path)
config_path.write_text(yaml.safe_dump({"model_list": []}))

View file

@ -14,6 +14,7 @@ from litellm.proxy.client.cli.commands.autoroute.process import (
clear_pid_record,
is_running,
launch_proxy,
missing_proxy_runtime_modules,
poll_liveliness,
read_pid_record,
write_pid_record,
@ -137,3 +138,21 @@ class TestPollLiveliness:
assert "exited early" in str(exc_info.value)
assert "crash log line" in str(exc_info.value)
class TestMissingProxyRuntimeModules:
def test_flags_absent_modules_only(self, monkeypatch):
"""A thin litellm[cli] install lacks the proxy runtime; the missing ones must be reported
(by name, for an actionable error) while modules that are importable are not."""
monkeypatch.setattr(
process_module,
"_PROXY_RUNTIME_MODULES",
("os", "litellm_autoroute_definitely_absent_pkg", "socket"),
)
assert missing_proxy_runtime_modules() == ("litellm_autoroute_definitely_absent_pkg",)
def test_empty_when_all_present(self, monkeypatch):
monkeypatch.setattr(process_module, "_PROXY_RUNTIME_MODULES", ("os", "socket"))
assert missing_proxy_runtime_modules() == ()