diff --git a/litellm/proxy/client/cli/README.md b/litellm/proxy/client/cli/README.md index 799e50bf980..7722d402a76 100644 --- a/litellm/proxy/client/cli/README.md +++ b/litellm/proxy/client/cli/README.md @@ -440,33 +440,30 @@ litellm-proxy http request GET /health/test_connection -H "X-Custom-Header:value ### Run a Coding Agent -Run a coding agent (Claude Code, Codex, OpenCode, and others) with all of its LLM traffic routed through your LiteLLM proxy. Everything after `--` is the command to launch: +Launch a coding agent with all of its LLM traffic routed through your LiteLLM proxy. Each supported agent is its own command, so there is nothing to remember beyond the agent's name: ```bash -litellm-proxy run -- claude -litellm-proxy run -- codex -litellm-proxy run -- opencode +litellm-proxy claude +litellm-proxy codex +litellm-proxy opencode ``` -`litellm-proxy claude-code` is a shortcut for `litellm-proxy run -- claude`. +Anything you type after the agent name is forwarded to it untouched, so the usual flags keep working: -The wrapper resolves your LiteLLM key (logging in via SSO when none is stored and you are at a terminal; otherwise it expects `LITELLM_PROXY_API_KEY` or `--api-key`), checks the key against the proxy so bad credentials fail immediately instead of deep inside the agent, exports the environment variables the agent reads, then replaces itself with the agent process. +```bash +litellm-proxy claude --resume +litellm-proxy codex exec "summarize the repo" +``` -The command is detected from the binary name to pick the right variables. Claude Code gets `ANTHROPIC_BASE_URL` (the proxy root, so it appends `/v1/messages`) and `ANTHROPIC_AUTH_TOKEN`, with any stray `ANTHROPIC_API_KEY` cleared so the proxy token wins. Codex and OpenCode get `OPENAI_BASE_URL` (the proxy plus `/v1`) and `OPENAI_API_KEY`. An unrecognized command gets both sets so it works either way. +Each command resolves your LiteLLM key (logging in via SSO when none is stored and you are at a terminal; otherwise it expects `LITELLM_PROXY_API_KEY` or `--api-key`), checks the key against the proxy so bad credentials fail immediately instead of deep inside the agent, exports the environment variables the agent reads, then replaces itself with the agent process. -Options: +The right variables are picked per agent. Claude Code gets `ANTHROPIC_BASE_URL` (the proxy root, so it appends `/v1/messages`) and `ANTHROPIC_AUTH_TOKEN`, with any stray `ANTHROPIC_API_KEY` cleared so the proxy token wins. Codex and OpenCode get `OPENAI_BASE_URL` (the proxy plus `/v1`) and `OPENAI_API_KEY`. + +Options (these belong to the wrapper, so put them before the agent's own flags): -- `--model`, `-m`: Model Claude Code should request (sets `ANTHROPIC_MODEL`); must resolve on your proxy. OpenAI-style agents take the model through their own flag. -- `--small-fast-model`: Background-task model for Claude Code (sets `ANTHROPIC_SMALL_FAST_MODEL`). - `--skip-verify`: Skip the pre-launch key check (useful offline or with non-standard auth). -Forward flags the agent owns after `--`: - -```bash -litellm-proxy run --model claude-sonnet-proxy -- claude --resume -``` - -Whatever model the agent requests (default Claude names, or your `--model` override) must exist on the proxy, since requests land on the proxy's `/v1/messages` (Anthropic) or `/v1/chat/completions` and `/v1/responses` (OpenAI) endpoints. +To pin the model, pass the agent's own model flag (for example `litellm-proxy claude --model my-proxy-model` or `litellm-proxy codex -m my-proxy-model`), or export the variable the agent reads (`ANTHROPIC_MODEL` / `ANTHROPIC_SMALL_FAST_MODEL` for Claude Code); the wrapper preserves anything you already have set. Whatever model the agent ends up requesting must exist on the proxy, since requests land on the proxy's `/v1/messages` (Anthropic) or `/v1/chat/completions` and `/v1/responses` (OpenAI) endpoints. ## Environment Variables diff --git a/litellm/proxy/client/cli/commands/run.py b/litellm/proxy/client/cli/commands/agents.py similarity index 66% rename from litellm/proxy/client/cli/commands/run.py rename to litellm/proxy/client/cli/commands/agents.py index 27306d433fa..43c594eff69 100644 --- a/litellm/proxy/client/cli/commands/run.py +++ b/litellm/proxy/client/cli/commands/agents.py @@ -1,7 +1,7 @@ import os import shutil import sys -from typing import Callable, Dict, FrozenSet, Mapping, Optional, Sequence, Tuple +from typing import Callable, Dict, FrozenSet, List, Mapping, Optional, Sequence, Tuple import click import requests @@ -11,8 +11,6 @@ from .auth import get_stored_api_key, login ANTHROPIC_BASE_URL_ENV = "ANTHROPIC_BASE_URL" ANTHROPIC_AUTH_TOKEN_ENV = "ANTHROPIC_AUTH_TOKEN" ANTHROPIC_API_KEY_ENV = "ANTHROPIC_API_KEY" -ANTHROPIC_MODEL_ENV = "ANTHROPIC_MODEL" -ANTHROPIC_SMALL_FAST_MODEL_ENV = "ANTHROPIC_SMALL_FAST_MODEL" OPENAI_BASE_URL_ENV = "OPENAI_BASE_URL" OPENAI_API_KEY_ENV = "OPENAI_API_KEY" @@ -53,9 +51,6 @@ def build_agent_env( base_url: str, api_key: str, profiles: FrozenSet[str], - *, - model: Optional[str] = None, - small_fast_model: Optional[str] = None, ) -> Dict[str, str]: """Return a copy of base_env wired to route the agent through the proxy. @@ -70,10 +65,6 @@ def build_agent_env( env[ANTHROPIC_BASE_URL_ENV] = root env[ANTHROPIC_AUTH_TOKEN_ENV] = api_key env.pop(ANTHROPIC_API_KEY_ENV, None) - if model: - env[ANTHROPIC_MODEL_ENV] = model - if small_fast_model: - env[ANTHROPIC_SMALL_FAST_MODEL_ENV] = small_fast_model if PROFILE_OPENAI in profiles: env[OPENAI_BASE_URL_ENV] = root + "/v1" env[OPENAI_API_KEY_ENV] = api_key @@ -115,8 +106,6 @@ def run_agent( api_key: str, command: Sequence[str], *, - model: Optional[str] = None, - small_fast_model: Optional[str] = None, skip_verify: bool = False, base_env: Optional[Mapping[str, str]] = None, which: Callable[[str], Optional[str]] = shutil.which, @@ -129,7 +118,7 @@ def run_agent( AgentRunError for missing binaries, an unreachable proxy, or a rejected key. """ if not command: - raise AgentRunError("Nothing to run. Try `litellm-proxy run -- claude`.") + raise AgentRunError("Nothing to run.") _, profiles = agent_profile(command[0]) binary = which(command[0]) @@ -146,8 +135,6 @@ def run_agent( base_url, api_key, profiles, - model=model, - small_fast_model=small_fast_model, ) launcher(binary, command, env) @@ -178,86 +165,60 @@ def _resolve_api_key(ctx: click.Context) -> str: return api_key -_MODEL_HELP = ( - "Model the agent should request, exported as ANTHROPIC_MODEL for Claude Code. " - "Must resolve on your proxy. OpenAI-style agents take the model via their own flag." -) -_SMALL_FAST_MODEL_HELP = ( - "Background-task model for Claude Code, exported as ANTHROPIC_SMALL_FAST_MODEL." -) _SKIP_VERIFY_HELP = "Skip the pre-launch key check against the proxy." -@click.command(name="run", context_settings={"ignore_unknown_options": True}) -@click.option("--model", "-m", default=None, help=_MODEL_HELP) -@click.option("--small-fast-model", default=None, help=_SMALL_FAST_MODEL_HELP) -@click.option("--skip-verify", is_flag=True, default=False, help=_SKIP_VERIFY_HELP) -@click.argument("command", nargs=-1, type=click.UNPROCESSED, required=True) -@click.pass_context -def run( - ctx: click.Context, - model: Optional[str], - small_fast_model: Optional[str], - skip_verify: bool, - command: Sequence[str], -): - """Run a coding agent wired to your LiteLLM proxy. - - Everything after `--` is the command to execute, e.g. - `litellm-proxy run -- claude`, `litellm-proxy run -- codex`, or - `litellm-proxy run -- opencode`. Logs in with LiteLLM if needed, checks the - key against the proxy, exports the env vars the agent reads, then hands off. - """ - command = list(command) +def _launch( + ctx: click.Context, binary: str, args: Sequence[str], *, skip_verify: bool +) -> None: base_url = ctx.obj["base_url"] api_key = _resolve_api_key(ctx) - display_name, _ = agent_profile(command[0]) + display_name, _ = agent_profile(binary) click.echo( f"litellm: routing {display_name} through proxy at {base_url.rstrip('/')}" ) try: - run_agent( - base_url, - api_key, - command, - model=model, - small_fast_model=small_fast_model, - skip_verify=skip_verify, - ) + run_agent(base_url, api_key, [binary, *args], skip_verify=skip_verify) except AgentRunError as e: raise click.ClickException(str(e)) -@click.command(name="claude-code", context_settings={"ignore_unknown_options": True}) -@click.option("--model", "-m", default=None, help=_MODEL_HELP) -@click.option("--small-fast-model", default=None, help=_SMALL_FAST_MODEL_HELP) -@click.option("--skip-verify", is_flag=True, default=False, help=_SKIP_VERIFY_HELP) -@click.argument("claude_args", nargs=-1, type=click.UNPROCESSED) -@click.pass_context -def claude_code( - ctx: click.Context, - model: Optional[str], - small_fast_model: Optional[str], - skip_verify: bool, - claude_args: Sequence[str], -): - """Shortcut for `litellm-proxy run -- claude`. Extra args are forwarded to claude.""" - ctx.invoke( - run, - model=model, - small_fast_model=small_fast_model, - skip_verify=skip_verify, - command=("claude", *claude_args), +def _make_agent_command(binary: str, display_name: str) -> click.Command: + @click.command( + name=binary, + context_settings={"ignore_unknown_options": True}, + short_help=f"Run {display_name} through your LiteLLM proxy", ) + @click.option("--skip-verify", is_flag=True, default=False, help=_SKIP_VERIFY_HELP) + @click.argument("args", nargs=-1, type=click.UNPROCESSED) + @click.pass_context + def _command(ctx: click.Context, skip_verify: bool, args: Sequence[str]) -> None: + _launch(ctx, binary, list(args), skip_verify=skip_verify) + + _command.help = ( + f"Run {display_name} routed through your LiteLLM proxy.\n\n" + f"Logs in with LiteLLM if needed, verifies your key against the proxy, " + f"exports the env vars {binary} reads, then hands off. Any arguments are " + f"forwarded to `{binary}`." + ) + return _command + + +def agent_commands() -> List[click.Command]: + """Build one top-level command per known agent, e.g. `litellm-proxy claude`.""" + return [ + _make_agent_command(binary, name) + for binary, (name, _profiles) in _KNOWN_AGENTS.items() + ] __all__ = [ - "run", - "claude_code", + "agent_commands", "run_agent", "build_agent_env", "verify_proxy_key", "agent_profile", + "AgentRunError", ] diff --git a/litellm/proxy/client/cli/interface.py b/litellm/proxy/client/cli/interface.py index 7da3084497d..f37c0b9a412 100644 --- a/litellm/proxy/client/cli/interface.py +++ b/litellm/proxy/client/cli/interface.py @@ -80,6 +80,8 @@ def styled_prompt(): def show_commands(): """Display available commands.""" + from .commands.agents import agent_commands + commands = [ ("login", "Authenticate with the LiteLLM proxy server"), ("logout", "Clear stored authentication"), @@ -91,8 +93,9 @@ def show_commands(): ("keys", "Manage API keys"), ("teams", "Manage teams and team assignments"), ("users", "Manage users"), - ("run", "Run a coding agent (claude, codex, ...) through the proxy"), - ("claude-code", "Shortcut for `run -- claude`"), + ] + commands += [(c.name, c.get_short_help_str()) for c in agent_commands()] + commands += [ ("version", "Show version information"), ("help", "Show this help message"), ("quit", "Exit the interactive session"), diff --git a/litellm/proxy/client/cli/main.py b/litellm/proxy/client/cli/main.py index e213460604a..b8c483f4b08 100644 --- a/litellm/proxy/client/cli/main.py +++ b/litellm/proxy/client/cli/main.py @@ -7,6 +7,7 @@ import click from litellm._version import version as litellm_version from litellm.proxy.client.health import HealthManagementClient +from .commands.agents import agent_commands from .commands.auth import get_stored_api_key, login, logout, whoami from .commands.chat import chat from .commands.credentials import credentials @@ -15,7 +16,6 @@ from .commands.keys import keys # local imports from .commands.models import models -from .commands.run import claude_code, run from .commands.teams import teams from .commands.users import users from .interface import interactive_shell @@ -113,9 +113,9 @@ cli.add_command(keys) cli.add_command(teams) # Add the users command group cli.add_command(users) -# Add the agent runner commands -cli.add_command(run) -cli.add_command(claude_code) +# Add a top-level command per coding agent (claude, codex, opencode, ...) +for agent_command in agent_commands(): + cli.add_command(agent_command) if __name__ == "__main__": diff --git a/tests/test_litellm/proxy/client/cli/test_run_commands.py b/tests/test_litellm/proxy/client/cli/test_agents.py similarity index 80% rename from tests/test_litellm/proxy/client/cli/test_run_commands.py rename to tests/test_litellm/proxy/client/cli/test_agents.py index f824d10772f..fc21bb6e4ec 100644 --- a/tests/test_litellm/proxy/client/cli/test_run_commands.py +++ b/tests/test_litellm/proxy/client/cli/test_agents.py @@ -12,17 +12,20 @@ sys.path.insert( ) # Adds the parent directory to the system path -from litellm.proxy.client.cli.commands.run import ( +from litellm.proxy.client.cli.commands.agents import ( AgentRunError, + agent_commands, agent_profile, build_agent_env, - claude_code, - run, run_agent, verify_proxy_key, ) -RUN_MODULE = "litellm.proxy.client.cli.commands.run" +AGENTS_MODULE = "litellm.proxy.client.cli.commands.agents" + + +def _agent_command(name): + return next(c for c in agent_commands() if c.name == name) class _FakeResponse: @@ -87,34 +90,6 @@ class TestBuildAgentEnv: assert env["ANTHROPIC_AUTH_TOKEN"] == "sk-key" assert env["OPENAI_API_KEY"] == "sk-key" - def test_model_overrides_only_apply_to_anthropic(self): - anthropic = build_agent_env( - {}, - "http://localhost:4000", - "sk-key", - frozenset({"anthropic"}), - model="claude-proxy", - small_fast_model="haiku-proxy", - ) - assert anthropic["ANTHROPIC_MODEL"] == "claude-proxy" - assert anthropic["ANTHROPIC_SMALL_FAST_MODEL"] == "haiku-proxy" - - openai = build_agent_env( - {}, - "http://localhost:4000", - "sk-key", - frozenset({"openai"}), - model="claude-proxy", - ) - assert "ANTHROPIC_MODEL" not in openai - - def test_model_overrides_omitted_when_not_given(self): - env = build_agent_env( - {}, "http://localhost:4000", "sk-key", frozenset({"anthropic"}) - ) - assert "ANTHROPIC_MODEL" not in env - assert "ANTHROPIC_SMALL_FAST_MODEL" not in env - def test_preserves_unrelated_env_and_does_not_mutate_input(self): base = {"PATH": "/usr/bin", "ANTHROPIC_API_KEY": "real-key"} env = build_agent_env( @@ -255,53 +230,79 @@ class TestRunAgent: run_agent("http://localhost:4000", "sk-key", []) -class TestRunCommand: +class TestAgentCommands: def setup_method(self): self.runner = CliRunner() - def test_launches_with_stored_key_and_forwards_args(self): + def test_one_command_per_known_agent(self): + assert {c.name for c in agent_commands()} == {"claude", "codex", "opencode"} + + def test_claude_launches_with_stored_key_and_forwards_args(self): captured = {} def fake_run_agent(base_url, api_key, command, **kwargs): captured["base_url"] = base_url captured["api_key"] = api_key captured["command"] = list(command) - captured["model"] = kwargs.get("model") + captured["skip_verify"] = kwargs.get("skip_verify") - with patch(f"{RUN_MODULE}.run_agent", side_effect=fake_run_agent): + with patch(f"{AGENTS_MODULE}.run_agent", side_effect=fake_run_agent): result = self.runner.invoke( - run, - ["--model", "claude-proxy", "--", "claude", "--resume"], + _agent_command("claude"), + ["--resume", "-p", "hi"], obj={"base_url": "http://localhost:4000", "api_key": "sk-key"}, ) assert result.exit_code == 0, result.output assert captured["api_key"] == "sk-key" - assert captured["command"] == ["claude", "--resume"] - assert captured["model"] == "claude-proxy" + assert captured["command"] == ["claude", "--resume", "-p", "hi"] + assert captured["skip_verify"] is False assert ( "routing Claude Code through proxy at http://localhost:4000" in result.output ) def test_codex_shows_friendly_name(self): - with patch(f"{RUN_MODULE}.run_agent"): + captured = {} + with patch( + f"{AGENTS_MODULE}.run_agent", + side_effect=lambda b, k, c, **kw: captured.update(command=list(c)), + ): result = self.runner.invoke( - run, - ["--", "codex"], + _agent_command("codex"), + ["exec", "do a thing"], obj={"base_url": "http://localhost:4000", "api_key": "sk-key"}, ) assert result.exit_code == 0, result.output + assert captured["command"] == ["codex", "exec", "do a thing"] assert "routing Codex through proxy" in result.output + def test_skip_verify_is_consumed_not_forwarded(self): + captured = {} + + def fake_run_agent(base_url, api_key, command, **kwargs): + captured["command"] = list(command) + captured["skip_verify"] = kwargs.get("skip_verify") + + with patch(f"{AGENTS_MODULE}.run_agent", side_effect=fake_run_agent): + result = self.runner.invoke( + _agent_command("claude"), + ["--skip-verify", "--resume"], + obj={"base_url": "http://localhost:4000", "api_key": "sk-key"}, + ) + + assert result.exit_code == 0, result.output + assert captured["skip_verify"] is True + assert captured["command"] == ["claude", "--resume"] + def test_non_interactive_without_key_errors_clearly(self): with ( - patch(f"{RUN_MODULE}._is_interactive", return_value=False), - patch(f"{RUN_MODULE}.run_agent") as mock_run, + patch(f"{AGENTS_MODULE}._is_interactive", return_value=False), + patch(f"{AGENTS_MODULE}.run_agent") as mock_run, ): result = self.runner.invoke( - run, - ["--", "claude"], + _agent_command("claude"), + [], obj={"base_url": "http://localhost:4000", "api_key": None}, ) assert result.exit_code != 0 @@ -316,21 +317,21 @@ class TestRunCommand: pass with ( - patch(f"{RUN_MODULE}._is_interactive", return_value=True), - patch(f"{RUN_MODULE}.login", fake_login), + patch(f"{AGENTS_MODULE}._is_interactive", return_value=True), + patch(f"{AGENTS_MODULE}.login", fake_login), patch( - f"{RUN_MODULE}.get_stored_api_key", return_value="sk-after-login" + f"{AGENTS_MODULE}.get_stored_api_key", return_value="sk-after-login" ) as mock_get, patch( - f"{RUN_MODULE}.run_agent", + f"{AGENTS_MODULE}.run_agent", side_effect=lambda base_url, api_key, command, **k: captured.update( api_key=api_key ), ), ): result = self.runner.invoke( - run, - ["--", "claude"], + _agent_command("claude"), + [], obj={"base_url": "http://localhost:4000", "api_key": None}, ) @@ -340,34 +341,13 @@ class TestRunCommand: def test_agent_run_error_becomes_click_error(self): with patch( - f"{RUN_MODULE}.run_agent", + f"{AGENTS_MODULE}.run_agent", side_effect=AgentRunError("could not reach proxy"), ): result = self.runner.invoke( - run, - ["--", "claude"], + _agent_command("claude"), + [], obj={"base_url": "http://localhost:4000", "api_key": "sk-key"}, ) assert result.exit_code != 0 assert "could not reach proxy" in result.output - - -class TestClaudeCodeShortcut: - def setup_method(self): - self.runner = CliRunner() - - def test_delegates_to_run_with_claude_prepended(self): - captured = {} - - def fake_run_agent(base_url, api_key, command, **kwargs): - captured["command"] = list(command) - - with patch(f"{RUN_MODULE}.run_agent", side_effect=fake_run_agent): - result = self.runner.invoke( - claude_code, - ["--", "--resume"], - obj={"base_url": "http://localhost:4000", "api_key": "sk-key"}, - ) - - assert result.exit_code == 0, result.output - assert captured["command"] == ["claude", "--resume"]