mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
Some checks are pending
CI Coverage / assert-ci-coverage (push) Waiting to run
CodeQL / Analyze (actions) (push) Waiting to run
CodeQL / Analyze (javascript-typescript) (push) Waiting to run
CodeQL / Analyze (python) (push) Waiting to run
Unit Tests: Proxy DB Operations / proxy-utils (push) Blocked by required conditions
Unit Tests / caching-local (push) Waiting to run
Unit Tests / core-utils (push) Waiting to run
Unit Tests / enterprise-package (push) Waiting to run
Unit Tests / enterprise-routing (push) Waiting to run
CodSpeed Benchmarks / benchmarks (push) Waiting to run
Helm unit test / unit-test (push) Waiting to run
Lens Worker Image / lens-worker-image (push) Waiting to run
Publish basedpyright base counts / publish (push) Waiting to run
Scorecard supply-chain security / Scorecard analysis (push) Waiting to run
Code Quality Checks / code-quality (push) Waiting to run
Code Quality Checks / python-310-import-smoke (push) Waiting to run
UI Unit Tests / ui-unit-tests (push) Waiting to run
Postgres Tests / proxy-security (push) Waiting to run
Postgres Tests / schema-migration (push) Waiting to run
Postgres Tests / proxy-behavior (push) Waiting to run
LiteLLM Rust / rust-lint (push) Waiting to run
LiteLLM Rust / rust-test (push) Waiting to run
LiteLLM Rust / rust-wheel (push) Waiting to run
Unit Tests: Documentation Validation / documentation (push) Waiting to run
Unit Tests: Proxy DB Operations / assert-shard-coverage (push) Waiting to run
Unit Tests: Proxy DB Operations / auth-checks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / budgets (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / custom-logging (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / db-and-spend (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / endpoints-and-responses (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / guardrails-hooks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / jwt-and-keys (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / key-generation (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / logging-misc (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-runtime (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-server-core (push) Blocked by required conditions
Unit Tests / proxy-infra (push) Waiting to run
Unit Tests / integrations (push) Waiting to run
Unit Tests / All Other Providers (push) Waiting to run
Unit Tests / Vertex AI (push) Waiting to run
Unit Tests / mcp-integration (push) Waiting to run
Unit Tests / misc (push) Waiting to run
Unit Tests / proxy-auth (push) Waiting to run
Unit Tests / proxy-endpoints (push) Waiting to run
Unit Tests / proxy-extras (push) Waiting to run
Unit Tests / proxy-server (push) Waiting to run
Unit Tests / responses-caching-types (push) Waiting to run
GitHub Actions Security Analysis / zizmor (push) Waiting to run
* fix(azure_storage): keep the DataLakeServiceClient alive until its TTL elapses Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * ci: rerun integrations shard after unrelated gitlab prompt manager timeout Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): audit azure_storage client reuse against a local Data Lake sink Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(azure_storage): cover the exact TTL expiry boundary Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(azure_storage): wait for a rejected write before flipping the sink back Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(e2e): add azure_storage log delivery cells behind an opt-in lane Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(azure_storage): restart the proxy mid burst and bound the loss to the unflushed queue Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(azure_storage): drop the redundant stop after the owned proxy exits Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(e2e): read azure_storage objects at the auth-mode-dependent layout Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(e2e): install the datalake sdk in the e2e lint environment Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(azure_storage): drop the opt-in real Azure e2e cells and their e2e-dev dependency Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: yucheng <yucheng@berri.ai>
234 lines
12 KiB
Python
234 lines
12 KiB
Python
import os
|
|
import signal
|
|
import uuid
|
|
from pathlib import Path
|
|
from typing import Final
|
|
|
|
import httpx
|
|
from _azure_storage_support import (
|
|
SINK_HOSTS,
|
|
RecordingDataLakeSink,
|
|
azure_storage_config,
|
|
azure_storage_environment,
|
|
collect_files,
|
|
)
|
|
from _s3_v2_support import matched_ids, mixed_burst, surface_reply
|
|
from integration._support.client import Gateway, JsonValue, eventually
|
|
from integration._support.process import group_members, owned_proxy_process
|
|
from integration._support.tls import server_context, write_self_signed_cert
|
|
from integration._support.wire import wire_server
|
|
|
|
WORKERS: Final = 2
|
|
FLUSH_SECONDS: Final = "1"
|
|
|
|
|
|
def _readiness_ok(candidate: Gateway) -> bool:
|
|
try:
|
|
return candidate.request("GET", "/health/readiness").status_code == 200
|
|
except httpx.TransportError:
|
|
return False
|
|
|
|
|
|
def _present_count(payloads: tuple[dict[str, JsonValue], ...], answered: tuple[tuple[str, str | None], ...]) -> int:
|
|
response_ids: Final = frozenset(response_id for response_id, _ in answered)
|
|
call_ids: Final = frozenset(call_id for _, call_id in answered if call_id is not None)
|
|
return sum(1 for payload in payloads if payload["id"] in response_ids or payload["litellm_call_id"] in call_ids)
|
|
|
|
|
|
def test_sink_outage_mid_burst_loses_only_the_outage_window_and_recovers_exactly_once(
|
|
gateway: Gateway, tmp_path: Path
|
|
) -> None:
|
|
marker: Final = f"azure-{uuid.uuid4().hex[:8]}"
|
|
sink: Final = RecordingDataLakeSink()
|
|
cert, key = write_self_signed_cert(tmp_path, SINK_HOSTS)
|
|
with (
|
|
wire_server(surface_reply) as provider,
|
|
wire_server(sink.respond, tls=server_context(cert, key), keep_alive=True) as store,
|
|
):
|
|
environment: Final = {
|
|
**azure_storage_environment(store.url, cert),
|
|
"DEFAULT_FLUSH_INTERVAL_SECONDS": FLUSH_SECONDS,
|
|
}
|
|
config: Final = azure_storage_config(tmp_path)
|
|
with (
|
|
owned_proxy_process(gateway, tmp_path, environment, config=config, workers=WORKERS) as owned,
|
|
owned.gateway.scenario() as scenario,
|
|
):
|
|
candidate: Final = owned.gateway
|
|
openai_model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key")
|
|
anthropic_model: Final = scenario.model(
|
|
model="anthropic/claude-sonnet-4-5-20250929", api_base=provider.url, api_key="synthetic-provider-key"
|
|
)
|
|
key: Final = scenario.key(models=[openai_model, anthropic_model])
|
|
first: Final = mixed_burst(candidate, openai_model, anthropic_model, key, f"{marker}-first", per_surface=2)
|
|
collect_files(sink, len(first))
|
|
attempts_before_outage: Final = sink.attempts()
|
|
sink.fail_status = 503
|
|
outage: Final = mixed_burst(
|
|
candidate, openai_model, anthropic_model, key, f"{marker}-outage", per_surface=1
|
|
)
|
|
eventually(sink.attempts, lambda count: count > attempts_before_outage, seconds=30)
|
|
readiness: Final = candidate.request("GET", "/health/readiness")
|
|
assert readiness.status_code == 200, readiness.text
|
|
sink.fail_status = 0
|
|
tail: Final = mixed_burst(candidate, openai_model, anthropic_model, key, f"{marker}-tail", per_surface=1)
|
|
answered: Final = first + outage + tail
|
|
payloads: Final = eventually(
|
|
lambda: tuple(sink.payloads().values()),
|
|
lambda stored: _present_count(stored, tail) == len(tail),
|
|
seconds=60,
|
|
)
|
|
landed: Final = matched_ids(payloads, answered)
|
|
assert sink.duplicated() == (), sink.duplicated()
|
|
assert len(sink.stored()) == len(landed), f"{len(sink.stored())} files for {len(landed)} matched ids"
|
|
assert len(landed) >= len(first) + len(tail), (
|
|
f"lost {len(answered) - len(landed)} of {len(answered)} payloads, "
|
|
f"expected at most the {len(outage)} sent during the outage"
|
|
)
|
|
assert len(answered) - len(landed) <= len(outage)
|
|
|
|
|
|
def test_slow_sink_lands_every_id_once_without_deadlock(gateway: Gateway, tmp_path: Path) -> None:
|
|
marker: Final = f"azure-{uuid.uuid4().hex[:8]}"
|
|
sink: Final = RecordingDataLakeSink(delay_seconds=0.3)
|
|
cert, key = write_self_signed_cert(tmp_path, SINK_HOSTS)
|
|
with (
|
|
wire_server(surface_reply) as provider,
|
|
wire_server(sink.respond, tls=server_context(cert, key), keep_alive=True) as store,
|
|
):
|
|
environment: Final = {
|
|
**azure_storage_environment(store.url, cert),
|
|
"DEFAULT_FLUSH_INTERVAL_SECONDS": FLUSH_SECONDS,
|
|
}
|
|
config: Final = azure_storage_config(tmp_path)
|
|
with (
|
|
owned_proxy_process(gateway, tmp_path, environment, config=config, workers=WORKERS) as owned,
|
|
owned.gateway.scenario() as scenario,
|
|
):
|
|
candidate: Final = owned.gateway
|
|
openai_model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key")
|
|
anthropic_model: Final = scenario.model(
|
|
model="anthropic/claude-sonnet-4-5-20250929", api_base=provider.url, api_key="synthetic-provider-key"
|
|
)
|
|
key: Final = scenario.key(models=[openai_model, anthropic_model])
|
|
answered: Final = mixed_burst(candidate, openai_model, anthropic_model, key, marker, per_surface=6)
|
|
payloads: Final = collect_files(sink, len(answered), seconds=70)
|
|
assert len(matched_ids(payloads, answered)) == len(answered), tuple(sink.stored())
|
|
assert len(sink.stored()) == len(answered)
|
|
assert sink.duplicated() == (), sink.duplicated()
|
|
assert sink.peak >= 1
|
|
assert store.connections() <= 2 * WORKERS, (
|
|
f"{store.connections()} sink connections for {len(answered)} uploads"
|
|
)
|
|
|
|
|
|
def test_killing_one_worker_keeps_the_other_serving_and_uploading(gateway: Gateway, tmp_path: Path) -> None:
|
|
marker: Final = f"azure-{uuid.uuid4().hex[:8]}"
|
|
sink: Final = RecordingDataLakeSink()
|
|
cert, key = write_self_signed_cert(tmp_path, SINK_HOSTS)
|
|
with (
|
|
wire_server(surface_reply) as provider,
|
|
wire_server(sink.respond, tls=server_context(cert, key), keep_alive=True) as store,
|
|
):
|
|
environment: Final = {
|
|
**azure_storage_environment(store.url, cert),
|
|
"DEFAULT_FLUSH_INTERVAL_SECONDS": FLUSH_SECONDS,
|
|
}
|
|
config: Final = azure_storage_config(tmp_path)
|
|
with (
|
|
owned_proxy_process(gateway, tmp_path, environment, config=config, workers=WORKERS) as owned,
|
|
owned.gateway.scenario() as scenario,
|
|
):
|
|
candidate: Final = owned.gateway
|
|
openai_model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key")
|
|
anthropic_model: Final = scenario.model(
|
|
model="anthropic/claude-sonnet-4-5-20250929", api_base=provider.url, api_key="synthetic-provider-key"
|
|
)
|
|
key: Final = scenario.key(models=[openai_model, anthropic_model])
|
|
first: Final = mixed_burst(candidate, openai_model, anthropic_model, key, f"{marker}-first", per_surface=2)
|
|
collect_files(sink, len(first))
|
|
workers: Final = tuple(
|
|
process for process in group_members(owned.process.pid) if process.pid != owned.process.pid
|
|
)
|
|
assert workers, "no uvicorn workers in the owned proxy process group"
|
|
os.kill(workers[0].pid, signal.SIGKILL)
|
|
eventually(lambda: _readiness_ok(candidate), lambda ok: ok, seconds=30)
|
|
rest: Final = mixed_burst(candidate, openai_model, anthropic_model, key, f"{marker}-rest", per_surface=4)
|
|
payloads: Final = eventually(
|
|
lambda: tuple(sink.payloads().values()),
|
|
lambda stored: _present_count(stored, rest) == len(rest),
|
|
seconds=60,
|
|
)
|
|
members_after: Final = eventually(
|
|
lambda: len(group_members(owned.process.pid)),
|
|
lambda count: count >= 1 + WORKERS,
|
|
seconds=30,
|
|
return_last_on_timeout=True,
|
|
)
|
|
landed: Final = matched_ids(payloads, first + rest)
|
|
assert sink.duplicated() == (), sink.duplicated()
|
|
assert len(landed) >= len(rest), f"only {len(landed)} payloads landed for {len(rest)} post-kill requests"
|
|
assert _present_count(payloads, rest) == len(rest), (
|
|
f"lost {len(rest) - _present_count(payloads, rest)} post-kill payloads; "
|
|
f"process group holds {members_after - 1} workers after the kill"
|
|
)
|
|
|
|
|
|
def test_restarting_the_proxy_before_the_queue_flushes_bounds_the_loss_to_the_unflushed_queue_and_recovers(
|
|
gateway: Gateway, tmp_path: Path
|
|
) -> None:
|
|
marker: Final = f"azure-{uuid.uuid4().hex[:8]}"
|
|
sink: Final = RecordingDataLakeSink()
|
|
cert, key = write_self_signed_cert(tmp_path, SINK_HOSTS)
|
|
with (
|
|
wire_server(surface_reply) as provider,
|
|
wire_server(sink.respond, tls=server_context(cert, key), keep_alive=True) as store,
|
|
):
|
|
environment: Final = {
|
|
**azure_storage_environment(store.url, cert),
|
|
"DEFAULT_FLUSH_INTERVAL_SECONDS": FLUSH_SECONDS,
|
|
}
|
|
config: Final = azure_storage_config(tmp_path)
|
|
with owned_proxy_process(gateway, tmp_path, environment, config=config, workers=WORKERS) as first_owned:
|
|
with first_owned.gateway.scenario() as scenario:
|
|
openai_model: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key")
|
|
anthropic_model: Final = scenario.model(
|
|
model="anthropic/claude-sonnet-4-5-20250929",
|
|
api_base=provider.url,
|
|
api_key="synthetic-provider-key",
|
|
)
|
|
first_key: Final = scenario.key(models=[openai_model, anthropic_model])
|
|
first: Final = mixed_burst(
|
|
first_owned.gateway, openai_model, anthropic_model, first_key, f"{marker}-first", per_surface=2
|
|
)
|
|
collect_files(sink, len(first))
|
|
cut: Final = mixed_burst(
|
|
first_owned.gateway, openai_model, anthropic_model, first_key, f"{marker}-cut", per_surface=2
|
|
)
|
|
with owned_proxy_process(gateway, tmp_path, environment, config=config, workers=WORKERS) as second_owned:
|
|
with second_owned.gateway.scenario() as scenario:
|
|
second_openai: Final = scenario.model(api_base=provider.url + "/v1", api_key="synthetic-provider-key")
|
|
second_anthropic: Final = scenario.model(
|
|
model="anthropic/claude-sonnet-4-5-20250929",
|
|
api_base=provider.url,
|
|
api_key="synthetic-provider-key",
|
|
)
|
|
second_key: Final = scenario.key(models=[second_openai, second_anthropic])
|
|
tail: Final = mixed_burst(
|
|
second_owned.gateway, second_openai, second_anthropic, second_key, f"{marker}-tail", per_surface=2
|
|
)
|
|
payloads: Final = eventually(
|
|
lambda: tuple(sink.payloads().values()),
|
|
lambda stored: _present_count(stored, tail) == len(tail),
|
|
seconds=60,
|
|
)
|
|
answered: Final = first + cut + tail
|
|
landed: Final = matched_ids(payloads, answered)
|
|
assert sink.duplicated() == (), sink.duplicated()
|
|
assert len(sink.stored()) == len(landed), f"{len(sink.stored())} files for {len(landed)} matched ids"
|
|
assert _present_count(payloads, first) == len(first)
|
|
assert _present_count(payloads, tail) == len(tail)
|
|
assert len(answered) - len(landed) <= len(cut), (
|
|
f"lost {len(answered) - len(landed)} of {len(answered)} payloads; the in-memory queue is dropped on "
|
|
f"restart by design, so at most the {len(cut)} pre-restart unflushed requests may be lost"
|
|
)
|