mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
feat(proxy): expose uvicorn concurrency limit (#33077)
Some checks are pending
OSS Daily Guardrails / Run OSS daily safe checks (push) Waiting to run
Some checks are pending
OSS Daily Guardrails / Run OSS daily safe checks (push) Waiting to run
Expose uvicorn's limit_concurrency as a --limit_concurrency CLI flag and LIMIT_CONCURRENCY environment variable. Uvicorn counts both active tasks and accepted connections and returns HTTP 503 once the configured limit is reached. Reject non-positive limits at CLI parse time and only add the setting to the uvicorn startup arguments. Because idle connections also consume capacity, deployments should use upstream connection/header timeouts and per-client connection limits.
This commit is contained in:
parent
90b7749c9b
commit
8c0910d4a4
2 changed files with 89 additions and 0 deletions
|
|
@ -800,6 +800,19 @@ class ProxyInitializationHelpers:
|
|||
),
|
||||
envvar="MAX_REQUESTS_BEFORE_RESTART_JITTER",
|
||||
)
|
||||
@click.option(
|
||||
"--limit_concurrency",
|
||||
default=None,
|
||||
type=click.IntRange(min=1),
|
||||
help=(
|
||||
"Set uvicorn's concurrency limit. Uvicorn counts both active tasks and "
|
||||
"accepted connections and returns HTTP 503 after the limit is reached. "
|
||||
"Idle connections can consume capacity, so use upstream connection/header "
|
||||
"timeouts and per-client connection limits. Only applies to uvicorn "
|
||||
"(ignored under --run_gunicorn / --run_hypercorn / --run_granian)."
|
||||
),
|
||||
envvar="LIMIT_CONCURRENCY",
|
||||
)
|
||||
@click.option(
|
||||
"--enforce_prisma_migration_check",
|
||||
is_flag=True,
|
||||
|
|
@ -868,6 +881,7 @@ def run_server(
|
|||
timeout_worker_healthcheck,
|
||||
max_requests_before_restart,
|
||||
max_requests_before_restart_jitter: Optional[int],
|
||||
limit_concurrency: Optional[int],
|
||||
enforce_prisma_migration_check: bool,
|
||||
use_v2_migration_resolver: bool,
|
||||
reload: bool,
|
||||
|
|
@ -1241,6 +1255,8 @@ def run_server(
|
|||
if max_requests_before_restart is not None:
|
||||
uvicorn_args["limit_max_requests"] = max_requests_before_restart
|
||||
if run_gunicorn is False and run_hypercorn is False and run_granian is False:
|
||||
if limit_concurrency is not None:
|
||||
uvicorn_args["limit_concurrency"] = limit_concurrency
|
||||
if max_requests_before_restart_jitter is not None:
|
||||
ProxyInitializationHelpers._apply_uvicorn_max_requests_jitter(
|
||||
uvicorn_args=uvicorn_args,
|
||||
|
|
|
|||
|
|
@ -569,6 +569,79 @@ class TestProxyInitializationHelpers:
|
|||
), f"exit_code={result.exit_code}, output={result.output}"
|
||||
mock_uvicorn_run.assert_called_once()
|
||||
|
||||
@patch("uvicorn.run")
|
||||
@patch("atexit.register")
|
||||
@patch("litellm.proxy.db.prisma_client.PrismaManager.setup_database")
|
||||
@patch(
|
||||
"litellm.proxy.db.prisma_client.should_update_prisma_schema", return_value=False
|
||||
)
|
||||
def test_limit_concurrency_passed_to_uvicorn(
|
||||
self, mock_should_update, mock_setup_db, mock_atexit_register, mock_uvicorn_run
|
||||
):
|
||||
"""--limit_concurrency must reach uvicorn.run so uvicorn sheds load with 503
|
||||
past the cap; omitted values stay absent and non-positive values are rejected."""
|
||||
from click.testing import CliRunner
|
||||
|
||||
from litellm.proxy.proxy_cli import run_server
|
||||
|
||||
runner = CliRunner()
|
||||
mock_proxy_module = MagicMock(
|
||||
app=MagicMock(),
|
||||
ProxyConfig=MagicMock(),
|
||||
KeyManagementSettings=MagicMock(),
|
||||
save_worker_config=MagicMock(),
|
||||
)
|
||||
clean_env = {
|
||||
k: v
|
||||
for k, v in os.environ.items()
|
||||
if k not in ("DATABASE_URL", "DIRECT_URL")
|
||||
}
|
||||
with (
|
||||
patch.dict(os.environ, clean_env, clear=True),
|
||||
patch.dict(
|
||||
"sys.modules",
|
||||
{
|
||||
"proxy_server": mock_proxy_module,
|
||||
"litellm.proxy.proxy_server": mock_proxy_module,
|
||||
},
|
||||
),
|
||||
patch(
|
||||
"litellm.proxy.proxy_cli.ProxyInitializationHelpers._get_default_unvicorn_init_args"
|
||||
) as mock_get_args,
|
||||
):
|
||||
mock_get_args.side_effect = lambda *a, **k: {
|
||||
"app": "litellm.proxy.proxy_server:app",
|
||||
"host": "localhost",
|
||||
"port": 8000,
|
||||
}
|
||||
|
||||
result = runner.invoke(
|
||||
run_server, ["--local", "--limit_concurrency", "250"]
|
||||
)
|
||||
assert (
|
||||
result.exit_code == 0
|
||||
), f"exit_code={result.exit_code}, output={result.output}"
|
||||
mock_uvicorn_run.assert_called_once()
|
||||
assert mock_uvicorn_run.call_args.kwargs.get("limit_concurrency") == 250
|
||||
|
||||
mock_uvicorn_run.reset_mock()
|
||||
result = runner.invoke(run_server, ["--local"])
|
||||
assert (
|
||||
result.exit_code == 0
|
||||
), f"exit_code={result.exit_code}, output={result.output}"
|
||||
mock_uvicorn_run.assert_called_once()
|
||||
assert "limit_concurrency" not in mock_uvicorn_run.call_args.kwargs
|
||||
|
||||
for invalid_value in ("0", "-1"):
|
||||
mock_uvicorn_run.reset_mock()
|
||||
result = runner.invoke(
|
||||
run_server,
|
||||
["--local", "--limit_concurrency", invalid_value],
|
||||
)
|
||||
assert result.exit_code == 2
|
||||
assert "Invalid value for '--limit_concurrency'" in result.output
|
||||
mock_uvicorn_run.assert_not_called()
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"timeout_config,expected_timeout",
|
||||
[
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue