mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cli/anthropic): unblock lite autoroute proxy deps, adaptive thinking, and thinking+signature streaming (#33507)
This commit is contained in:
parent
5a0e1dd1dd
commit
ebc6fdb4c2
8 changed files with 207 additions and 6 deletions
|
|
@ -495,19 +495,21 @@ Cursor is not supported: it has no equivalent file-based config to hot-patch thi
|
|||
|
||||
#### Install the CLI
|
||||
|
||||
If you don't already have the `lite` command, install it with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
|
||||
`lite autoroute up` builds and runs a throwaway litellm proxy locally, so unlike the rest of this CLI it needs the proxy server runtime, not just the thin `litellm[cli]` client. Install `litellm[proxy]` (which ships the `lite` command too) with a single curl command -- no existing Python tooling required, `uv` is bootstrapped automatically if missing:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/main/scripts/install-cli.sh | sh
|
||||
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/main/scripts/install.sh | sh
|
||||
```
|
||||
|
||||
This installs only `litellm[cli]`, the thin client (`lite`), not the full proxy server. To try an unreleased branch or commit instead of the latest PyPI release, set `LITELLM_CLI_REF`:
|
||||
To QA an unreleased branch or commit instead of the latest PyPI release, set `LITELLM_CLI_REF`:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch-or-commit>/scripts/install-cli.sh | \
|
||||
curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch-or-commit>/scripts/install.sh | \
|
||||
LITELLM_CLI_REF=<branch-or-commit> sh
|
||||
```
|
||||
|
||||
The thin `scripts/install-cli.sh` installs only `litellm[cli]`, which is enough for `lite login`, `lite claude`, and `lite up`, but not for `lite autoroute up`; running it against a `litellm[cli]` install fails fast with a message telling you to install the proxy runtime.
|
||||
|
||||
Point the CLI at your real proxy and key before running any `lite model-groups` or `lite autoroute` command -- like every other command in this CLI, they read `LITELLM_PROXY_URL`/`LITELLM_PROXY_API_KEY` (or `--base-url`/`--api-key`), no `lite login` required:
|
||||
|
||||
```bash
|
||||
|
|
|
|||
|
|
@ -21,6 +21,7 @@ from .process import (
|
|||
clear_pid_record,
|
||||
is_running,
|
||||
launch_proxy,
|
||||
missing_proxy_runtime_modules,
|
||||
poll_liveliness,
|
||||
read_pid_record,
|
||||
secure_create,
|
||||
|
|
@ -81,6 +82,16 @@ def up() -> None:
|
|||
if not CONFIG_PATH.exists():
|
||||
raise click.ClickException("No config found. Run `lite autoroute configure` first.")
|
||||
|
||||
missing = missing_proxy_runtime_modules()
|
||||
if missing:
|
||||
raise click.ClickException(
|
||||
"lite autoroute up launches a local litellm proxy, which needs the proxy runtime that the "
|
||||
f"thin `litellm[cli]` install does not include (missing: {', '.join(missing)}). Install the "
|
||||
"proxy runtime with `uv tool install --force 'litellm[proxy]'`, or to QA a branch, "
|
||||
"`curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch>/scripts/install.sh | "
|
||||
"LITELLM_CLI_REF=<branch> sh`."
|
||||
)
|
||||
|
||||
try:
|
||||
existing_pid = read_pid_record()
|
||||
except UpError as e:
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import contextlib
|
||||
import importlib.util
|
||||
import json
|
||||
import os
|
||||
import signal
|
||||
|
|
@ -37,6 +38,20 @@ class PidRecord:
|
|||
_PID_RECORD_ADAPTER = TypeAdapter(PidRecord)
|
||||
|
||||
|
||||
_PROXY_RUNTIME_MODULES: tuple[str, ...] = ("fastapi", "uvicorn", "backoff", "orjson", "websockets", "apscheduler")
|
||||
|
||||
|
||||
def missing_proxy_runtime_modules() -> tuple[str, ...]:
|
||||
"""Proxy-server modules that ``lite autoroute up`` needs but the thin CLI install lacks.
|
||||
|
||||
``launch_proxy`` runs the full ``litellm.proxy.proxy_cli`` server, whose dependencies live in
|
||||
the ``proxy`` extra, not the ``cli`` extra that installs the ``lite`` command. On a thin
|
||||
``litellm[cli]`` install the subprocess dies with a bare ``ModuleNotFoundError``; detecting the
|
||||
gap here lets ``up`` fail with an actionable message instead.
|
||||
"""
|
||||
return tuple(name for name in _PROXY_RUNTIME_MODULES if importlib.util.find_spec(name) is None)
|
||||
|
||||
|
||||
def allocate_free_port() -> int:
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
|
||||
sock.bind(("127.0.0.1", 0))
|
||||
|
|
@ -165,6 +180,7 @@ __all__ = [
|
|||
"clear_pid_record",
|
||||
"is_running",
|
||||
"launch_proxy",
|
||||
"missing_proxy_runtime_modules",
|
||||
"poll_liveliness",
|
||||
"read_pid_record",
|
||||
"secure_create",
|
||||
|
|
|
|||
|
|
@ -5,12 +5,24 @@
|
|||
# Needs only curl: uv is bootstrapped if missing, and uv provisions a compatible
|
||||
# Python itself (reusing a suitable system one, else downloading a managed build).
|
||||
#
|
||||
# To install from an unreleased branch, tag, or commit instead of the latest PyPI
|
||||
# release, set LITELLM_CLI_REF:
|
||||
# curl -fsSL https://raw.githubusercontent.com/BerriAI/litellm/<branch>/scripts/install.sh | \
|
||||
# LITELLM_CLI_REF=<branch> sh
|
||||
#
|
||||
# NOTE: set -e without pipefail for POSIX sh compatibility (dash on Ubuntu/Debian
|
||||
# ignores the shebang when invoked as `sh` and does not support `pipefail`).
|
||||
set -eu
|
||||
|
||||
# NOTE: before merging, this must stay as "litellm[proxy]" to install from PyPI.
|
||||
LITELLM_PACKAGE="litellm[proxy]"
|
||||
# LITELLM_CLI_REF opts into installing from a branch, tag, or commit instead (for
|
||||
# example, to QA lite autoroute against an unreleased branch, which needs this proxy
|
||||
# runtime, not the thin litellm[cli] install).
|
||||
if [ -n "${LITELLM_CLI_REF:-}" ]; then
|
||||
LITELLM_PACKAGE="litellm[proxy] @ git+https://github.com/BerriAI/litellm.git@${LITELLM_CLI_REF}"
|
||||
else
|
||||
LITELLM_PACKAGE="litellm[proxy]"
|
||||
fi
|
||||
UV_VERSION="0.10.9"
|
||||
|
||||
# ── colours ────────────────────────────────────────────────────────────────
|
||||
|
|
@ -81,7 +93,11 @@ fi
|
|||
|
||||
# ── install ────────────────────────────────────────────────────────────────
|
||||
echo ""
|
||||
header "Installing litellm[proxy]…"
|
||||
if [ -n "${LITELLM_CLI_REF:-}" ]; then
|
||||
header "Installing litellm[proxy] from ${LITELLM_CLI_REF}…"
|
||||
else
|
||||
header "Installing litellm[proxy]…"
|
||||
fi
|
||||
echo ""
|
||||
|
||||
# --python-preference system: reuse a compatible system Python when present,
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ content survives.
|
|||
import asyncio
|
||||
import json
|
||||
from types import SimpleNamespace
|
||||
from typing import AsyncIterator
|
||||
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.streaming_iterator import (
|
||||
AnthropicStreamWrapper,
|
||||
|
|
@ -202,3 +203,89 @@ def test_split_clears_reasoning_and_thinking_on_finish_chunk():
|
|||
assert content_chunk.choices[0].delta.thinking_blocks == [{"type": "thinking"}]
|
||||
assert finish_chunk.choices[0].delta.reasoning_content is None
|
||||
assert finish_chunk.choices[0].delta.thinking_blocks is None
|
||||
|
||||
|
||||
def _thinking_delta_chunk(thinking: str) -> ModelResponseStream:
|
||||
return ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
index=0,
|
||||
delta=Delta(
|
||||
reasoning_content=thinking,
|
||||
thinking_blocks=[{"type": "thinking", "thinking": thinking, "signature": None}],
|
||||
provider_specific_fields={
|
||||
"thinking_blocks": [{"type": "thinking", "thinking": thinking, "signature": None}]
|
||||
},
|
||||
),
|
||||
finish_reason=None,
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def _signature_chunk(recap: str, signature: str) -> ModelResponseStream:
|
||||
return ModelResponseStream(
|
||||
choices=[
|
||||
StreamingChoices(
|
||||
index=0,
|
||||
delta=Delta(
|
||||
reasoning_content=recap,
|
||||
thinking_blocks=[{"type": "thinking", "thinking": recap, "signature": signature}],
|
||||
provider_specific_fields={
|
||||
"thinking_blocks": [{"type": "thinking", "thinking": recap, "signature": signature}]
|
||||
},
|
||||
),
|
||||
finish_reason=None,
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def test_thinking_then_signature_chunk_does_not_crash_stream():
|
||||
"""Regression for the /v1/messages streaming crash reported on autoroute.
|
||||
|
||||
Anthropic streams extended thinking as incremental ``thinking_delta`` chunks, then a
|
||||
closing chunk that recaps the full accumulated thinking AND carries the signature. The
|
||||
adapter used to raise ``ValueError`` on that closing chunk, killing the whole stream. It
|
||||
must instead emit a single ``signature_delta`` for the recap chunk and never re-emit the
|
||||
recap thinking, so the incremental thinking text is not duplicated.
|
||||
"""
|
||||
chunks = [
|
||||
_thinking_delta_chunk("First, "),
|
||||
_thinking_delta_chunk("reason."),
|
||||
_signature_chunk("First, reason.", "sig-abc"),
|
||||
ModelResponseStream(
|
||||
choices=[StreamingChoices(index=0, delta=Delta(content="Done"), finish_reason=None)],
|
||||
),
|
||||
ModelResponseStream(
|
||||
choices=[StreamingChoices(index=0, delta=Delta(), finish_reason="stop")],
|
||||
usage=Usage(prompt_tokens=5, completion_tokens=3, total_tokens=8),
|
||||
),
|
||||
]
|
||||
|
||||
async def _aiter() -> "AsyncIterator[ModelResponseStream]":
|
||||
for chunk in chunks:
|
||||
yield chunk
|
||||
|
||||
wrapper = AnthropicStreamWrapper(completion_stream=_aiter(), model="claude-haiku-4-5")
|
||||
sse = _collect_async(wrapper)
|
||||
|
||||
signature_deltas = [
|
||||
json.loads(line[len("data: ") :])
|
||||
for block in sse.split("\n\n")
|
||||
for line in block.splitlines()
|
||||
if line.startswith("data: ") and '"signature_delta"' in line
|
||||
]
|
||||
assert len(signature_deltas) == 1
|
||||
assert signature_deltas[0]["delta"]["signature"] == "sig-abc"
|
||||
|
||||
thinking_text = "".join(
|
||||
json.loads(line[len("data: ") :])["delta"]["thinking"]
|
||||
for block in sse.split("\n\n")
|
||||
for line in block.splitlines()
|
||||
if line.startswith("data: ") and '"thinking_delta"' in line
|
||||
)
|
||||
assert thinking_text == "First, reason."
|
||||
|
||||
assert "message_stop" in sse
|
||||
assert "Done" in sse
|
||||
|
|
|
|||
|
|
@ -32,6 +32,37 @@ def _transform(model, params, litellm_params=None):
|
|||
)
|
||||
|
||||
|
||||
def test_adaptive_thinking_only_translated_to_legacy_for_haiku_4_5():
|
||||
"""The minimal autoroute repro: Claude Code sends bare ``thinking={type: adaptive}``
|
||||
(no ``output_config``) and the complexity router picks Haiku 4.5, which does not
|
||||
support adaptive thinking. Anthropic 400s with "adaptive thinking is not supported on
|
||||
this model" unless the flag is dropped, so it must be translated to the legacy extended
|
||||
thinking the model does support rather than forwarded raw."""
|
||||
result = _transform("claude-haiku-4-5", {"max_tokens": 8192, "thinking": {"type": "adaptive"}})
|
||||
|
||||
assert result["thinking"] == {
|
||||
"type": "enabled",
|
||||
"budget_tokens": DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
|
||||
}
|
||||
assert "output_config" not in result
|
||||
|
||||
|
||||
def test_adaptive_thinking_only_dropped_for_non_reasoning_model():
|
||||
"""Bare adaptive thinking on a model with no reasoning support at all is silently
|
||||
dropped so the request still succeeds instead of being rejected."""
|
||||
result = _transform("claude-3-5-haiku-latest", {"max_tokens": 8192, "thinking": {"type": "adaptive"}})
|
||||
|
||||
assert "thinking" not in result
|
||||
|
||||
|
||||
def test_adaptive_thinking_only_preserved_for_4_6():
|
||||
"""A 4.6+ model natively supports adaptive thinking, so a bare adaptive flag must not
|
||||
be rewritten even without output_config."""
|
||||
result = _transform("claude-sonnet-4-6", {"max_tokens": 8192, "thinking": {"type": "adaptive"}})
|
||||
|
||||
assert result["thinking"] == {"type": "adaptive"}
|
||||
|
||||
|
||||
def test_effort_translated_to_legacy_thinking_for_haiku_4_5():
|
||||
"""Core regression: Claude Code sends adaptive thinking + effort to Haiku 4.5
|
||||
(thinking-capable, pre-4.6). Effort must be translated to legacy extended
|
||||
|
|
|
|||
|
|
@ -69,6 +69,25 @@ class TestUpCommand:
|
|||
assert result.exception is None or isinstance(result.exception, SystemExit)
|
||||
assert "lite autoroute configure" in result.output
|
||||
|
||||
def test_refuses_with_actionable_error_when_proxy_runtime_missing(self, monkeypatch, tmp_path):
|
||||
"""`up` launches a real litellm proxy, which the thin `litellm[cli]` install cannot run.
|
||||
It must fail fast with an actionable message pointing at the proxy install, before it ever
|
||||
tries to launch the doomed subprocess (which would otherwise die with a bare ImportError)."""
|
||||
config_path, _log_path, _settings_path, _backup_path, _pid_record_path = _patch_paths(monkeypatch, tmp_path)
|
||||
config_path.write_text(yaml.safe_dump({"model_list": []}))
|
||||
monkeypatch.setattr(commands_module, "missing_proxy_runtime_modules", lambda: ("fastapi", "websockets"))
|
||||
|
||||
def _fail_if_launched(*args, **kwargs):
|
||||
raise AssertionError("launch_proxy must not run when the proxy runtime is missing")
|
||||
|
||||
monkeypatch.setattr(commands_module, "launch_proxy", _fail_if_launched)
|
||||
|
||||
result = self.runner.invoke(up)
|
||||
|
||||
assert result.exit_code != 0
|
||||
assert "fastapi, websockets" in result.output
|
||||
assert "litellm[proxy]" in result.output
|
||||
|
||||
def test_refuses_when_pid_record_exists_and_process_still_running(self, monkeypatch, tmp_path):
|
||||
config_path, _log_path, _settings_path, _backup_path, pid_record_path = _patch_paths(monkeypatch, tmp_path)
|
||||
config_path.write_text(yaml.safe_dump({"model_list": []}))
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ from litellm.proxy.client.cli.commands.autoroute.process import (
|
|||
clear_pid_record,
|
||||
is_running,
|
||||
launch_proxy,
|
||||
missing_proxy_runtime_modules,
|
||||
poll_liveliness,
|
||||
read_pid_record,
|
||||
write_pid_record,
|
||||
|
|
@ -137,3 +138,21 @@ class TestPollLiveliness:
|
|||
|
||||
assert "exited early" in str(exc_info.value)
|
||||
assert "crash log line" in str(exc_info.value)
|
||||
|
||||
|
||||
class TestMissingProxyRuntimeModules:
|
||||
def test_flags_absent_modules_only(self, monkeypatch):
|
||||
"""A thin litellm[cli] install lacks the proxy runtime; the missing ones must be reported
|
||||
(by name, for an actionable error) while modules that are importable are not."""
|
||||
monkeypatch.setattr(
|
||||
process_module,
|
||||
"_PROXY_RUNTIME_MODULES",
|
||||
("os", "litellm_autoroute_definitely_absent_pkg", "socket"),
|
||||
)
|
||||
|
||||
assert missing_proxy_runtime_modules() == ("litellm_autoroute_definitely_absent_pkg",)
|
||||
|
||||
def test_empty_when_all_present(self, monkeypatch):
|
||||
monkeypatch.setattr(process_module, "_PROXY_RUNTIME_MODULES", ("os", "socket"))
|
||||
|
||||
assert missing_proxy_runtime_modules() == ()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue