From 4b975e6f490a559ef16b74574c4cd637e20d79cd Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 9 Oct 2026 04:28:41 +0000 Subject: [PATCH] ci: render lint, unit and smoke checks as / with one collector per tier (#45480) * ci: restructure lint, unit and smoke workflows * ci: preserve source formatting in tier workflows * ci: retire per-shard coverage flags and tighten the tier workflow guards * test(ci): freeze the workflow startup safety models * ci: keep the current required check names running until the ruleset moves to the tier collectors --------- Co-authored-by: yuneng --- .github/merge-smoke-tests.json | 15 - .github/scripts/assert_ci_coverage.py | 8 + .github/scripts/run_merge_smoke.py | 140 +-- .github/workflows/_test-unit-base.yml | 283 ----- .github/workflows/check-ui-api-types.yml | 134 --- .github/workflows/ci-coverage.yml | 48 - .../workflows/publish-lint-base-counts.yml | 2 +- .github/workflows/required-checks-legacy.yml | 397 +++++++ .github/workflows/test-code-quality.yml | 223 ---- .github/workflows/test-linting.yml | 440 +++++++- .github/workflows/test-litellm-ui-lint.yml | 105 -- .github/workflows/test-litellm-ui-unit.yml | 99 -- .github/workflows/test-merge-smoke.yml | 24 +- .github/workflows/test-unit-documentation.yml | 103 -- .github/workflows/test-unit.yml | 966 ++++++++++++++---- Makefile | 7 +- codecov.yaml | 72 +- litellm/proxy/_lazy_openapi_snapshot.py | 2 +- scripts/pre_commit_lint.sh | 16 +- .../check_workflow_startup_safety.py | 168 +-- .../code_coverage_tests/test_e2e_metadata.py | 2 +- tests/code_coverage_tests/test_merge_smoke.py | 157 --- .../test_unit_passed_gate.py | 50 +- tests/e2e_harness/AGENTS.md | 2 +- tests/unit/test_circleci_path_filter.py | 2 +- tests/unit/test_detect_changes.py | 2 +- tests/unit/test_select_ui_test_scope.py | 6 +- tests/unit/test_unit_shard_missing_paths.py | 6 +- .../unit/test_unit_shard_per_test_timeout.py | 6 +- ui/litellm-dashboard/AGENTS.md | 2 +- 30 files changed, 1783 insertions(+), 1704 deletions(-) delete mode 100644 .github/merge-smoke-tests.json delete mode 100644 .github/workflows/_test-unit-base.yml delete mode 100644 .github/workflows/check-ui-api-types.yml delete mode 100644 .github/workflows/ci-coverage.yml create mode 100644 .github/workflows/required-checks-legacy.yml delete mode 100644 .github/workflows/test-code-quality.yml delete mode 100644 .github/workflows/test-litellm-ui-lint.yml delete mode 100644 .github/workflows/test-litellm-ui-unit.yml delete mode 100644 .github/workflows/test-unit-documentation.yml diff --git a/.github/merge-smoke-tests.json b/.github/merge-smoke-tests.json deleted file mode 100644 index 90d3b6a6d59..00000000000 --- a/.github/merge-smoke-tests.json +++ /dev/null @@ -1,15 +0,0 @@ -{ - "cases": { - "CHAT-JSON": "tests/unit/llms/openai/test_openai.py::test_acompletion_returns_json_reply_over_injected_transport", - "CHAT-TEXT-STREAM": "tests/unit/llms/openai/test_openai.py::test_acompletion_streams_text_deltas_over_injected_transport", - "CHAT-TOOL-STREAM": "tests/unit/llms/openai/test_openai.py::test_acompletion_streams_tool_call_arguments_over_injected_transport", - "MODEL-ALLOW": "tests/unit/proxy/auth/test_auth_checks_object_access_and_lookup.py::test_can_object_call_model_allows_listed_model_for_key", - "MODEL-DENY": "tests/unit/proxy/auth/test_auth_checks_object_access_and_lookup.py::test_can_object_call_model_denials_return_forbidden[key-key_model_access_denied]", - "COST-EXPLICIT": "tests/unit/test_cost_calculator.py::test_completion_cost_charges_explicit_per_token_rates_over_registered_ones", - "COST-ZERO": "tests/unit/test_cost_calculator.py::test_completion_cost_is_zero_when_explicit_rates_are_zero", - "LOG-CONTENT-ON": "tests/unit/litellm_core_utils/test_litellm_logging.py::test_standard_logging_payload_keeps_message_content_when_message_logging_is_on", - "LOG-CONTENT-OFF": "tests/unit/litellm_core_utils/test_litellm_logging.py::test_standard_logging_payload_redacts_message_content_when_message_logging_is_off", - "CALLBACK-SUCCESS": "tests/unit/litellm_core_utils/test_litellm_logging.py::test_async_success_handler_delivers_standard_logging_payload_to_custom_logger", - "CALLBACK-FAILURE": "tests/unit/litellm_core_utils/test_litellm_logging.py::test_async_failure_handler_delivers_failure_payload_to_custom_logger" - } -} diff --git a/.github/scripts/assert_ci_coverage.py b/.github/scripts/assert_ci_coverage.py index e90589d3c95..04c801cf5cc 100644 --- a/.github/scripts/assert_ci_coverage.py +++ b/.github/scripts/assert_ci_coverage.py @@ -36,7 +36,15 @@ GLOB_CHARS = frozenset("*?") # itself decomposed one level deeper and is checked through its own entry. SHARDED_ROOTS: tuple[str, ...] = ( "tests/test_litellm", + "tests/unit", + "tests/unit/enterprise", + "tests/unit/enterprise/enterprise_callbacks", + "tests/unit/enterprise/proxy", + "tests/unit/llms", "tests/unit/proxy", + "tests/unit/proxy/_experimental", + "tests/unit/responses", + "tests/unit/skills", ) diff --git a/.github/scripts/run_merge_smoke.py b/.github/scripts/run_merge_smoke.py index 7e74de324fc..81d47d5b45c 100644 --- a/.github/scripts/run_merge_smoke.py +++ b/.github/scripts/run_merge_smoke.py @@ -16,29 +16,11 @@ import socket import subprocess import sys import time -from collections import Counter from collections.abc import Sequence -from dataclasses import dataclass, field +from dataclasses import dataclass from pathlib import Path -from types import MappingProxyType from typing import Final, NoReturn, TextIO, cast -import pytest - -EXPECTED_CASES: Final = ( - "CHAT-JSON", - "CHAT-TEXT-STREAM", - "CHAT-TOOL-STREAM", - "MODEL-ALLOW", - "MODEL-DENY", - "COST-EXPLICIT", - "COST-ZERO", - "LOG-CONTENT-ON", - "LOG-CONTENT-OFF", - "CALLBACK-SUCCESS", - "CALLBACK-FAILURE", -) - @dataclass(frozen=True, slots=True) class CheckResult: @@ -57,8 +39,6 @@ class _Args: ready_deadline: float = 120.0 shutdown_deadline: float = 20.0 poll_interval: float = 0.5 - manifest: str = "" - rootdir: str | None = None def fail(reason: str) -> NoReturn: @@ -345,120 +325,6 @@ def _terminate(proc: subprocess.Popen[bytes], log_file: TextIO) -> None: log_file.close() -def _load_manifest(path: Path) -> MappingProxyType[str, str]: - def no_duplicates(pairs: list[tuple[object, object]]) -> dict[object, object]: - seen: dict[object, object] = {} - for key, value in pairs: - if key in seen: - raise ValueError(f"duplicate key in manifest: {key}") - seen[key] = value - return seen - - raw_value: object = cast(object, json.loads(path.read_text(), object_pairs_hook=no_duplicates)) - if not isinstance(raw_value, dict): - raise ValueError("manifest must be an object") - loaded: Final = cast(dict[object, object], raw_value) - cases_value: object = loaded.get("cases") - if not isinstance(cases_value, dict): - raise ValueError("manifest must be an object with a 'cases' object") - cases_any: Final = cast(dict[object, object], cases_value) - cases: Final = {k: v for k, v in cases_any.items() if isinstance(k, str) and isinstance(v, str)} - if len(cases) != len(cases_any): - raise ValueError("manifest 'cases' must map string ids to string node ids") - return MappingProxyType(cases) - - -@dataclass(slots=True, eq=False) -class _Recorder: - collect_failed: list[str] = field(default_factory=list) - collected: tuple[str, ...] = () - reports: dict[str, list[tuple[str, str, bool]]] = field(default_factory=dict) - - def pytest_collectreport(self, report: pytest.CollectReport) -> None: - if report.failed: - self.collect_failed.append(report.nodeid) - - def pytest_collection_finish(self, session: pytest.Session) -> None: - self.collected = tuple(item.nodeid for item in session.items) - - def pytest_runtest_logreport(self, report: pytest.TestReport) -> None: - self.reports.setdefault(report.nodeid, []).append((report.when, report.outcome, hasattr(report, "wasxfail"))) - - -def cmd_pytest(args: _Args) -> int: - try: - cases: Final = _load_manifest(Path(args.manifest)) - except (OSError, ValueError, json.JSONDecodeError) as exc: - fail(f"manifest invalid: {exc}") - if tuple(cases) != EXPECTED_CASES: - fail(f"manifest case ids must be exactly {list(EXPECTED_CASES)} in order, got {list(cases)}") - node_ids: Final = tuple(cases.values()) - if len(set(node_ids)) != len(node_ids): - fail("manifest node ids are not unique") - argv: Final = [ - *node_ids, - "-p", - "no:cacheprovider", - "-p", - "no:xdist", - "-p", - "no:rerunfailures", - "-p", - "no:randomly", - "-rA", - "-q", - *(["--rootdir", args.rootdir] if args.rootdir else []), - ] - - recorder: Final = _Recorder() - code: Final = pytest.main(argv, plugins=[recorder]) - name_of: Final = MappingProxyType({node_id: case_id for case_id, node_id in cases.items()}) - problems: Final[list[str]] = [] - if code != 0: - problems.append(f"pytest exit code {code}") - for failed_id in recorder.collect_failed: - problems.append(f"collection failed: {name_of.get(failed_id, failed_id)}") - expected: Final = Counter(node_ids) - collected: Final = Counter(recorder.collected) - for node_id in expected - collected: - problems.append(f"missing case {name_of[node_id]} ({node_id})") - for node_id in collected - expected: - problems.append(f"unexpected test collected: {node_id}") - for node_id, count in collected.items(): - if count > 1: - problems.append(f"duplicated test id: {node_id}") - if len(recorder.collected) != len(EXPECTED_CASES): - problems.append(f"collected {len(recorder.collected)} tests, expected {len(EXPECTED_CASES)}") - rows: Final[list[tuple[str, bool]]] = [] - for case_id, node_id in cases.items(): - reports = recorder.reports.get(node_id, []) - case_ok = ( - bool(reports) - and all(outcome == "passed" and not wasxfail for _, outcome, wasxfail in reports) - and {when for when, _, _ in reports} >= {"setup", "call", "teardown"} - ) - rows.append((case_id, case_ok)) - if not reports: - problems.append(f"{case_id} ({node_id}) produced no runtest reports") - continue - for when, outcome, wasxfail in reports: - if outcome != "passed": - problems.append(f"{case_id} ({node_id}) {when} outcome={outcome}") - if wasxfail: - problems.append(f"{case_id} ({node_id}) {when} was xfail/xpass") - missing_phases = {"setup", "call", "teardown"} - {when for when, _, _ in reports} - for phase in sorted(missing_phases): - problems.append(f"{case_id} ({node_id}) missing {phase} report") - for case_id, passed in rows: - print(f"{case_id} {'PASS' if passed else 'FAIL'} {cases[case_id]}") - if problems: - for problem in problems: - print(f"merge-smoke: {problem}", file=sys.stderr) - fail("pytest verdict failed") - ok("pytest 11 cases") - return 0 - - def main() -> int: parser: Final = argparse.ArgumentParser(description=__doc__) subs: Final = parser.add_subparsers(dest="command", required=True) @@ -475,16 +341,12 @@ def main() -> int: p_proxy.add_argument("--ready-deadline", type=float, default=120) p_proxy.add_argument("--shutdown-deadline", type=float, default=20) p_proxy.add_argument("--poll-interval", type=float, default=0.5) - p_test: Final = subs.add_parser("pytest") - p_test.add_argument("--manifest", required=True) - p_test.add_argument("--rootdir", default=None) args: Final = parser.parse_args(namespace=_Args()) handlers: Final = { "verify-isolation": cmd_verify_isolation, "interpreter": cmd_interpreter, "cli": cmd_cli, "proxy-startup": cmd_proxy_startup, - "pytest": cmd_pytest, } return handlers[args.command](args) diff --git a/.github/workflows/_test-unit-base.yml b/.github/workflows/_test-unit-base.yml deleted file mode 100644 index e5847e5fbea..00000000000 --- a/.github/workflows/_test-unit-base.yml +++ /dev/null @@ -1,283 +0,0 @@ -name: _Unit Test Base (Reusable) - -on: - workflow_call: - inputs: - python-version: - description: "Python version used to install dependencies and run tests" - required: false - type: string - default: "3.12" - test-path: - description: >- - Space-separated pytest paths to run. A path that no longer exists is - dropped with a warning instead of being passed to pytest, because one - missing path makes pytest-xdist collect nothing and report exit 5, which - the step treats as a drained shard. Options are passed through as - written, so use the `--flag=value` form: a bare `--ignore path` would - have its path existence-checked like any other token. - required: true - type: string - workers: - description: "Number of pytest-xdist workers" - required: false - type: number - default: 2 - reruns: - description: "Number of reruns for flaky tests" - required: false - type: number - default: 2 - timeout-minutes: - description: >- - Timeout for the test step alone. Setup (checkout, dependency install, - Prisma client generation) gets its own allowance on top, so a slow - runner or a cold binary download can never cancel passing tests. - required: false - type: number - default: 20 - job-timeout-minutes: - description: >- - Backstop for the whole job. Keep it >= `timeout-minutes` plus 40: 35 for - the per-step ceilings on the setup steps below, and 5 for the runner - overhead the job clock charges but no step owns (job init, step - transitions, post-job cleanup). That headroom is what makes the test - budget a floor rather than a hope, since setup cannot overrun into it - without failing its own step first. GitHub expressions have no - arithmetic, so the sum is passed in rather than computed. - required: false - type: number - default: 60 - test-timeout-seconds: - description: >- - Per-test ceiling enforced by pytest-timeout, covering fixture setup and - teardown as well as the test body. A test that hangs fails with a - traceback of where it was stuck instead of idling the shard until - `timeout-minutes` cancels it. Timed-out tests are excluded from reruns - because pytest-timeout arms its timer once per test and - pytest-rerunfailures reruns inside that same window, so a rerun of a - timed-out test would run with no timer at all. - required: false - type: number - default: 120 - max-failures: - description: "Stop after this many failures" - required: false - type: number - default: 10 - dist: - description: "pytest-xdist distribution mode (loadscope|load|worksteal|loadfile|no)" - required: false - type: string - default: "loadscope" - artifact-name: - description: "Unique name for the coverage artifact (must be unique per run)" - required: true - type: string - rust-bridge-artifact: - description: "Prebuilt editable Rust bridge artifact" - required: false - type: string - default: "" -permissions: - contents: read - -env: - UV_PYTHON: ${{ inputs.python-version }} - LITELLM_LOCAL_MODEL_COST_MAP: "True" - -jobs: - run: - name: Run tests - runs-on: ubuntu-latest - timeout-minutes: ${{ inputs.job-timeout-minutes }} - permissions: - contents: read - pull-requests: read - outputs: - decision: ${{ steps.changes.outputs.decision }} - has-coverage: ${{ steps.tests.outputs.has-coverage }} - - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - timeout-minutes: 3 - with: - persist-credentials: false - - - name: Detect relevant changes - id: changes - timeout-minutes: 2 - uses: ./.github/actions/detect-changes - - - name: Set up Python - if: steps.changes.outputs.decision != 'skip' - timeout-minutes: 3 - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: ${{ env.UV_PYTHON }} - - - name: Set up uv - if: steps.changes.outputs.decision != 'skip' - timeout-minutes: 3 - uses: ./.github/actions/setup-uv-with-retries - with: - version: "0.10.9" - - - name: Cache uv dependencies - if: steps.changes.outputs.decision != 'skip' - timeout-minutes: 5 - uses: ./.github/actions/cache-uv-downloads - - - name: Set up the Rust build - if: steps.changes.outputs.decision != 'skip' - timeout-minutes: 5 - uses: ./.github/actions/rust-bridge - with: - artifact: ${{ inputs.rust-bridge-artifact }} - - - name: Install dependencies - if: steps.changes.outputs.decision != 'skip' - timeout-minutes: 8 - env: - RUST_BRIDGE_ARTIFACT: ${{ inputs.rust-bridge-artifact }} - run: | - diff -u model_prices_and_context_window.json litellm/model_prices_and_context_window_backup.json - if [ -z "$RUST_BRIDGE_ARTIFACT" ]; then - .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra semantic-router --extra saml --extra caching --extra extra_proxy --extra proxy-runtime --extra utils - else - .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra semantic-router --extra saml --extra caching --extra extra_proxy --extra proxy-runtime --extra utils --no-install-project - uv pip install --no-deps --python .venv/bin/python rust-bridge-dist/*.whl - cp rust-bridge-dist/litellm/rust_bridge/_native.abi3.so litellm/rust_bridge/_native.abi3.so - uv run --no-sync python -c "import importlib.metadata; import litellm.rust_bridge._native; print(importlib.metadata.version('litellm'))" - fi - uv run --no-sync python -c 'import os, sys; print(sys.version); assert f"{sys.version_info.major}.{sys.version_info.minor}" == os.environ["UV_PYTHON"]' - - - name: Cache Prisma binaries - if: steps.changes.outputs.decision != 'skip' - timeout-minutes: 3 - uses: ./.github/actions/cache-prisma-binaries - - - name: Generate Prisma client - if: steps.changes.outputs.decision != 'skip' - timeout-minutes: 3 - run: | - uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma - - - name: Run tests - id: tests - if: steps.changes.outputs.decision != 'skip' - timeout-minutes: ${{ inputs.timeout-minutes }} - env: - TEST_PATH: ${{ inputs.test-path }} - MAX_FAILURES: ${{ inputs.max-failures }} - WORKERS: ${{ inputs.workers }} - RERUNS: ${{ inputs.reruns }} - TEST_TIMEOUT_SECONDS: ${{ inputs.test-timeout-seconds }} - DIST: ${{ inputs.dist }} - COVERAGE_CORE: sysmon - run: | - echo "has-coverage=false" >> "$GITHUB_OUTPUT" - selection="${TEST_PATH}" - if [ -z "${selection// /}" ]; then - echo "shard selection is empty; nothing to run" - exit 0 - fi - pytest_args=() - existing_paths=0 - for token in ${selection}; do - case "${token}" in - -*) pytest_args+=("${token}") ;; - *) - if [ -e "${token%%::*}" ]; then - pytest_args+=("${token}") - existing_paths=$((existing_paths + 1)) - else - echo "::warning::${token} does not exist; drop it from this shard's test-path" - fi - ;; - esac - done - if [ "${existing_paths}" -eq 0 ]; then - echo "No path in the selection exists (${selection}); nothing to run" - exit 0 - fi - xdist_args=() - if [ "${WORKERS}" != "0" ]; then - xdist_args=(-n "${WORKERS}" --dist="${DIST}") - fi - set +e - uv run --no-sync pytest "${pytest_args[@]}" \ - --tb=short -vv \ - --maxfail="${MAX_FAILURES}" \ - "${xdist_args[@]}" \ - --reruns "${RERUNS}" \ - --reruns-delay 1 \ - --timeout="${TEST_TIMEOUT_SECONDS}" \ - --rerun-except "from pytest-timeout" \ - --durations=20 \ - --cov=./litellm --cov=./enterprise/litellm_enterprise \ - --cov-report=xml:coverage.xml \ - --cov-config=pyproject.toml - status=$? - set -e - if [ -f coverage.xml ]; then - echo "has-coverage=true" >> "$GITHUB_OUTPUT" - fi - if [ "$status" -eq 5 ]; then - echo "pytest collected no tests from ${selection}; passing" - exit 0 - fi - exit "$status" - - - name: Save coverage report - if: always() && steps.changes.outputs.decision != 'skip' - uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1 - with: - name: coverage-${{ inputs.artifact-name }}-${{ github.run_id }}-${{ github.run_attempt }} - path: coverage.xml - retention-days: 1 - - upload-coverage: - name: Upload coverage to Codecov - needs: run - if: always() && needs.run.outputs.decision != 'skip' && needs.run.outputs.has-coverage == 'true' - runs-on: ubuntu-latest - permissions: - contents: read - id-token: write - pull-requests: write - - steps: - - name: Checkout code - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Download coverage report - uses: actions/download-artifact@95815c38cf2ff2164869cbab79da8d1f422bc89e # v4.2.1 - with: - pattern: coverage-${{ inputs.artifact-name }}-${{ github.run_id }}-${{ github.run_attempt }} - path: coverage-reports - merge-multiple: true - - - name: Upload to Codecov - id: codecov-upload - continue-on-error: true - uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5 - with: - use_oidc: true - directory: coverage-reports - root_dir: ${{ github.workspace }} - flags: ${{ inputs.artifact-name }} - fail_ci_if_error: false - - - name: Upload to Codecov (retry) - if: steps.codecov-upload.outcome == 'failure' - continue-on-error: true - uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5 - with: - use_oidc: true - directory: coverage-reports - root_dir: ${{ github.workspace }} - flags: ${{ inputs.artifact-name }} - fail_ci_if_error: false diff --git a/.github/workflows/check-ui-api-types.yml b/.github/workflows/check-ui-api-types.yml deleted file mode 100644 index 185c20d916d..00000000000 --- a/.github/workflows/check-ui-api-types.yml +++ /dev/null @@ -1,134 +0,0 @@ -name: Check UI API Types Sync - -on: - pull_request: - branches: - - main - - "litellm_**" - -permissions: - contents: read - -concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} - cancel-in-progress: true - -jobs: - check-sync: - name: Verify schema.d.ts matches the proxy OpenAPI spec - runs-on: ubuntu-latest - timeout-minutes: 15 - steps: - - name: Checkout repository - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - fetch-depth: 2 - - - name: Detect changes that can affect the generated types - id: changes - run: | - set -euo pipefail - if ! base="$(git rev-parse --verify --quiet HEAD^2 >/dev/null && git rev-parse HEAD^1)"; then - echo "Not a pull request merge commit, running the full check." - echo "relevant=true" >> "$GITHUB_OUTPUT" - exit 0 - fi - files="$(git diff --name-only "$base" HEAD)" - if grep -Eq '^(litellm/(proxy|types)/|ui/litellm-dashboard/(src/lib/http/schema\.d\.ts|scripts/gen-api-types\.mjs|package(-lock)?\.json)$|\.github/workflows/check-ui-api-types\.yml$)' <<< "$files"; then - echo "relevant=true" >> "$GITHUB_OUTPUT" - else - echo "No proxy, types or generator changes in this pull request, nothing to verify." - echo "relevant=false" >> "$GITHUB_OUTPUT" - fi - - - name: Set up Python - if: steps.changes.outputs.relevant == 'true' - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Set up uv - if: steps.changes.outputs.relevant == 'true' - uses: ./.github/actions/setup-uv-with-retries - with: - version: "0.10.9" - - - name: Cache uv dependencies - if: steps.changes.outputs.relevant == 'true' - uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 - with: - path: | - ~/.cache/uv - .venv - key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }} - restore-keys: | - ${{ runner.os }}-uv- - - - name: Cache the Rust build - if: steps.changes.outputs.relevant == 'true' - uses: ./.github/actions/cache-cargo-build - - - name: Install backend dependencies - if: steps.changes.outputs.relevant == 'true' - run: .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra semantic-router - - - name: Cache Prisma binaries - if: steps.changes.outputs.relevant == 'true' - uses: ./.github/actions/cache-prisma-binaries - - - name: Generate Prisma client - if: steps.changes.outputs.relevant == 'true' - run: uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma - - - name: Regenerate the lazy OpenAPI snapshot - if: steps.changes.outputs.relevant == 'true' - run: uv run --no-sync python -m litellm.proxy._lazy_openapi_snapshot - - - name: Fail if the lazy OpenAPI snapshot is stale - if: steps.changes.outputs.relevant == 'true' - run: | - if ! git diff --exit-code -- litellm/proxy/_lazy_openapi_snapshot.json; then - echo "::error file=litellm/proxy/_lazy_openapi_snapshot.json::The lazy OpenAPI snapshot is out of sync with the lazily loaded routes." - echo "" - echo "A lazily loaded route or model changed without regenerating the snapshot that /openapi.json serves for unloaded features." - echo "To fix, run from the repo root:" - echo " uv run python -m litellm.proxy._lazy_openapi_snapshot" - echo "then run npm run gen:api from ui/litellm-dashboard and commit both files." - exit 1 - fi - echo "_lazy_openapi_snapshot.json is in sync with the lazily loaded routes." - - - name: Set up Node.js - if: steps.changes.outputs.relevant == 'true' - uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5.0.0 - with: - node-version-file: ui/litellm-dashboard/.nvmrc - cache: "npm" - cache-dependency-path: ui/litellm-dashboard/package-lock.json - - - name: Install dashboard dependencies - if: steps.changes.outputs.relevant == 'true' - working-directory: ui/litellm-dashboard - run: npm ci - - - name: Regenerate types from the live spec - if: steps.changes.outputs.relevant == 'true' - working-directory: ui/litellm-dashboard - env: - LITELLM_PYTHON: "uv run --no-sync python" - run: npm run gen:api - - - name: Fail if types are stale - if: steps.changes.outputs.relevant == 'true' - run: | - if ! git diff --exit-code -- ui/litellm-dashboard/src/lib/http/schema.d.ts; then - echo "::error file=ui/litellm-dashboard/src/lib/http/schema.d.ts::Generated API types are out of sync with the proxy OpenAPI spec." - echo "" - echo "A backend route or model changed without regenerating the dashboard types." - echo "To fix, run from ui/litellm-dashboard:" - echo " npm run gen:api" - echo "then commit the updated src/lib/http/schema.d.ts." - exit 1 - fi - echo "schema.d.ts is in sync with the proxy OpenAPI spec." diff --git a/.github/workflows/ci-coverage.yml b/.github/workflows/ci-coverage.yml deleted file mode 100644 index cd36a9a7ed6..00000000000 --- a/.github/workflows/ci-coverage.yml +++ /dev/null @@ -1,48 +0,0 @@ -name: "CI Coverage" - -on: - pull_request: - branches: - - main - - "litellm_**" - push: - branches: - - main - -permissions: - contents: read - -concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.sha }} - cancel-in-progress: ${{ github.event_name == 'pull_request' }} - -jobs: - assert-ci-coverage: - name: assert-ci-coverage - runs-on: ubuntu-latest - timeout-minutes: 5 - permissions: - contents: read - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Assert every test file and Dockerfile is invoked by a job - run: | - python -m pip install "pyyaml==6.0.3" - python .github/scripts/assert_ci_coverage.py - - # The census asks whether a job names a file; this asks whether that job's -k - # then throws it back out. A file both globbed and deselected everywhere runs - # nowhere while counting as covered, which is how the caching suite went unrun. - - name: Assert no -k expression deselects a file from every job that globs it - run: python .github/scripts/assert_ci_coverage.py --slices - - - name: Assert .github/workflows/ holds only workflows, correctly named - run: python .github/scripts/assert_workflow_dir_hygiene.py diff --git a/.github/workflows/publish-lint-base-counts.yml b/.github/workflows/publish-lint-base-counts.yml index 7af00e9c842..5623ab82b90 100644 --- a/.github/workflows/publish-lint-base-counts.yml +++ b/.github/workflows/publish-lint-base-counts.yml @@ -64,7 +64,7 @@ jobs: uses: ./.github/actions/cache-prisma-binaries # The three source scanners only need the pinned dev tools (ruff and the - # stdlib checkers), the same versions test-linting.yml's lint job runs. + # stdlib checkers), the same versions test-linting.yml's python job runs. - name: Install the dev tools if: matrix.checker != 'basedpyright' run: | diff --git a/.github/workflows/required-checks-legacy.yml b/.github/workflows/required-checks-legacy.yml new file mode 100644 index 00000000000..6bba690bd8a --- /dev/null +++ b/.github/workflows/required-checks-legacy.yml @@ -0,0 +1,397 @@ +name: Required checks (legacy) + +on: + pull_request: + branches: + - main + - "litellm_**" + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.sha }} + cancel-in-progress: true + +jobs: + lint: + name: lint + permissions: + contents: read + pull-requests: read + actions: read + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + ref: ${{ github.event.pull_request.head.sha }} + fetch-depth: 1 + clean: true + persist-credentials: false + + - name: Detect relevant changes + id: changes + uses: ./.github/actions/detect-changes + + - name: Fetch gate base (merge-base with target branch) + if: steps.changes.outputs.decision != 'skip' + env: + GH_TOKEN: ${{ github.token }} + BASE_SHA: ${{ github.event.pull_request.base.sha }} + HEAD_SHA: ${{ github.event.pull_request.head.sha }} + run: | + retry() { "$@" || { sleep 15; "$@"; } || { sleep 30; "$@"; }; } + MERGE_BASE=$(retry gh api "repos/${{ github.repository }}/compare/${BASE_SHA}...${HEAD_SHA}?per_page=1" --jq '.merge_base_commit.sha') + test -n "$MERGE_BASE" + retry git fetch --no-tags --depth=1 origin "$MERGE_BASE" + echo "GATE_BASE_SHA=$MERGE_BASE" >> "$GITHUB_ENV" + + - name: Set up Python + if: steps.changes.outputs.decision != 'skip' + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Set up uv + if: steps.changes.outputs.decision != 'skip' + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + + - name: Cache uv dependencies + if: steps.changes.outputs.decision != 'skip' + uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: | + ~/.cache/uv + .venv + key: ${{ runner.os }}-uv-lint-${{ hashFiles('uv.lock') }} + restore-keys: | + ${{ runner.os }}-uv-lint- + + - name: Clean Python cache + if: steps.changes.outputs.decision != 'skip' + run: | + find . -type d -name "__pycache__" -exec rm -rf {} + || true + find . -name "*.pyc" -delete || true + + - name: Check uv.lock is up to date + if: steps.changes.outputs.decision != 'skip' + run: | + uv lock --check || (echo "❌ uv.lock is out of sync with pyproject.toml. Run 'uv lock' locally and commit the result." && exit 1) + + - name: Cache the Rust build + if: steps.changes.outputs.decision != 'skip' + uses: ./.github/actions/cache-cargo-build + + - name: Install dependencies + if: steps.changes.outputs.decision != 'skip' + run: | + uv sync --frozen --group proxy-dev --group e2e-dev + + - name: Cache Prisma binaries + if: steps.changes.outputs.decision != 'skip' + uses: ./.github/actions/cache-prisma-binaries + + - name: Generate Prisma client + if: steps.changes.outputs.decision != 'skip' + run: | + uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma + + - name: Check ruff format + if: steps.changes.outputs.decision != 'skip' + run: | + git diff --name-only --diff-filter=ACMR "$GATE_BASE_SHA" HEAD -- ':(glob)litellm/**/*.py' | grep -v '^litellm/enterprise/' > "$RUNNER_TEMP/ruff_format_files.txt" || true + if [ ! -s "$RUNNER_TEMP/ruff_format_files.txt" ]; then + echo "No changed litellm Python files to check with ruff format." + exit 0 + fi + xargs uv run --no-sync ruff format --check --exclude '/enterprise/' < "$RUNNER_TEMP/ruff_format_files.txt" + + - name: Debug - Check file state + if: steps.changes.outputs.decision != 'skip' + run: | + echo "Current branch:" + git branch --show-current + echo "Last 3 commits:" + git log --oneline -3 + echo "File content around line 43:" + head -50 litellm/litellm_core_utils/custom_logger_registry.py | tail -10 + + - name: Check MCP operation boundary + if: steps.changes.outputs.decision != 'skip' + run: uv run --no-sync python scripts/check_mcp_operation_boundary.py + + - name: Run Ruff linting + if: steps.changes.outputs.decision != 'skip' + run: | + cd litellm + uv run --no-sync ruff check . + cd .. + + - name: Run Ruff linting (test tree) + if: steps.changes.outputs.decision != 'skip' + run: | + uv run --no-sync ruff check --config ruff-tests.toml tests + + - name: Check strict ruff rules (delta vs merge-base counts) + if: steps.changes.outputs.decision != 'skip' + env: + GH_TOKEN: ${{ github.token }} + run: | + uv run --no-sync python scripts/ruff_strict_gate.py --base "$GATE_BASE_SHA" + + - name: Check type discipline (mutable collections / casts / type guards / kwargs / unexplained suppressions, delta vs merge-base counts) + if: steps.changes.outputs.decision != 'skip' + env: + GH_TOKEN: ${{ github.token }} + run: | + uv run --no-sync python scripts/type_discipline_gate.py --base "$GATE_BASE_SHA" + + - name: Check test quality (zero-assert / mock-echo tests, sys.path.insert, raw env writes, litellm global mutation, credential-gated skips, conftest snapshot inventory, delta vs merge-base counts) + if: steps.changes.outputs.decision != 'skip' + env: + GH_TOKEN: ${{ github.token }} + run: | + uv run --no-sync python scripts/test_quality_gate.py --base "$GATE_BASE_SHA" + + - name: Print OpenAI version + if: steps.changes.outputs.decision != 'skip' + run: | + uv run --no-sync python -c "import openai; print(f'OpenAI version: {openai.__version__}')" + + - name: Check basedpyright (delta vs merge-base counts) + if: steps.changes.outputs.decision != 'skip' + env: + GH_TOKEN: ${{ github.token }} + run: | + uv run --no-sync python scripts/type_check_gate.py --base "$GATE_BASE_SHA" + + - name: Check tests/e2e basedpyright (zero errors) + if: steps.changes.outputs.decision != 'skip' + run: | + if git diff --name-only --diff-filter=ACMRD "$GATE_BASE_SHA" HEAD -- ':(glob)tests/e2e/**/*.py' ':(glob)tests/e2e_harness/**/*.py' pyrightconfig.json | grep -q .; then + uv run --no-sync basedpyright tests/e2e tests/e2e_harness + else + echo "No changed tests/e2e Python files; skipping." + fi + + - name: Run the e2e harness tests + if: steps.changes.outputs.decision != 'skip' + env: + LITELLM_MASTER_KEY: sk-e2e-harness-tests-reach-no-proxy + run: | + if ! git diff --name-only --diff-filter=ACMRD "$GATE_BASE_SHA" HEAD -- tests/e2e tests/e2e_harness ':(exclude)tests/e2e/ui' pyproject.toml uv.lock .github/workflows/test-linting.yml | grep -q .; then + echo "No changed e2e harness files; skipping." + exit 0 + fi + retry() { "$@" || { sleep 15; "$@"; } || { sleep 30; "$@"; }; } + CLAUDE_VERSION="$(retry uv run --no-sync python tests/e2e/claude_code/pr_gate_version_resolver.py)" + tests/e2e/claude_code/cron_vm/install_claude_code.sh "$CLAUDE_VERSION" "$RUNNER_TEMP/claude-cli" + PATH="$RUNNER_TEMP/claude-cli:$PATH" uv run --no-sync pytest -q tests/e2e_harness + + - name: Check for circular imports + if: steps.changes.outputs.decision != 'skip' + run: | + cd litellm + uv run --no-sync python ../tests/documentation_tests/test_circular_imports.py + cd .. + + - name: Check import safety + if: steps.changes.outputs.decision != 'skip' + run: | + uv run --no-sync python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1) + + frontend-lint: + name: frontend-lint + permissions: + contents: read + runs-on: ubuntu-latest + timeout-minutes: 8 + defaults: + run: + working-directory: ui/litellm-dashboard + + steps: + - name: Checkout repository + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + fetch-depth: 1 + persist-credentials: false + + - name: Collect changed files + id: changed + env: + GH_TOKEN: ${{ github.token }} + BASE_SHA: ${{ github.event.pull_request.base.sha }} + HEAD_SHA: ${{ github.event.pull_request.head.sha }} + run: | + merge_base=$(gh api "repos/${{ github.repository }}/compare/${BASE_SHA}...${HEAD_SHA}?per_page=1" --jq '.merge_base_commit.sha') + test -n "$merge_base" + git fetch --no-tags --depth=1 origin "$merge_base" "$HEAD_SHA" + : > "$RUNNER_TEMP/prettier_files.txt" + : > "$RUNNER_TEMP/eslint_files.txt" + while IFS= read -r f; do + [ -f "$f" ] || continue + case "$f" in + *.js | *.jsx | *.ts | *.tsx | *.mjs | *.cjs) + printf '%s\n' "$f" >> "$RUNNER_TEMP/prettier_files.txt" + printf '%s\n' "$f" >> "$RUNNER_TEMP/eslint_files.txt" ;; + *.json | *.css | *.scss | *.md | *.mdx | *.yml | *.yaml | *.html) + printf '%s\n' "$f" >> "$RUNNER_TEMP/prettier_files.txt" ;; + esac + done < <(git diff --name-only --diff-filter=ACMR --relative "$merge_base" "$HEAD_SHA" -- .) + if [ -s "$RUNNER_TEMP/prettier_files.txt" ] || [ -s "$RUNNER_TEMP/eslint_files.txt" ]; then + echo "has_files=true" >> "$GITHUB_OUTPUT" + else + echo "has_files=false" >> "$GITHUB_OUTPUT" + echo "No lintable UI files changed in this PR; nothing to check." + fi + + - name: Setup Node.js + if: steps.changed.outputs.has_files == 'true' + uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5.0.0 + with: + node-version-file: ui/litellm-dashboard/.nvmrc + cache: "npm" + cache-dependency-path: ui/litellm-dashboard/package-lock.json + + - name: Install dependencies + if: steps.changed.outputs.has_files == 'true' + run: npm ci + + - name: Lint changed files (prettier + eslint) + if: steps.changed.outputs.has_files == 'true' + run: | + prettier_files=() + eslint_files=() + while IFS= read -r f; do prettier_files+=("$f"); done < "$RUNNER_TEMP/prettier_files.txt" + while IFS= read -r f; do eslint_files+=("$f"); done < "$RUNNER_TEMP/eslint_files.txt" + status=0 + if [ ${#prettier_files[@]} -gt 0 ]; then + echo "::group::Prettier (${#prettier_files[@]} files)" + npx prettier --check "${prettier_files[@]}" || { status=1; echo "::error::Unformatted files. Fix with: npm run format"; } + echo "::endgroup::" + fi + if [ ${#eslint_files[@]} -gt 0 ]; then + echo "::group::ESLint (${#eslint_files[@]} files)" + npx eslint --no-warn-ignored --pass-on-unpruned-suppressions "${eslint_files[@]}" || status=1 + echo "::endgroup::" + fi + exit $status + + - name: Check lint budgets + if: ${{ !cancelled() && steps.changed.outputs.has_files == 'true' }} + run: | + npx eslint . -f json -o "$RUNNER_TEMP/lint-report.json" || true + node scripts/check-lint-budgets.mjs "$RUNNER_TEMP/lint-report.json" eslint-budgets.json + + - name: Check for dead code (knip) + if: ${{ !cancelled() && steps.changed.outputs.has_files == 'true' }} + run: npm run knip:ci + + assert-ci-coverage: + name: assert-ci-coverage + permissions: + contents: read + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Assert every test file and Dockerfile is invoked by a job + run: | + python -m pip install "pyyaml==6.0.3" + python .github/scripts/assert_ci_coverage.py + + - name: Assert no -k expression deselects a file from every job that globs it + run: python .github/scripts/assert_ci_coverage.py --slices + + - name: Assert .github/workflows/ holds only workflows, correctly named + run: python .github/scripts/assert_workflow_dir_hygiene.py + + dashboard-build: + name: Dashboard build + runs-on: ubuntu-24.04 + timeout-minutes: 30 + steps: + - name: Checkout + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Build the dashboard stage + run: docker build --target ui-builder -f Dockerfile . + + core-checks: + name: Core checks (Python ${{ matrix.python-version }}) + runs-on: ubuntu-24.04 + timeout-minutes: 30 + strategy: + fail-fast: false + matrix: + python-version: ["3.10", "3.11", "3.12", "3.13", "3.14"] + env: + LITELLM_LOCAL_MODEL_COST_MAP: "True" + steps: + - name: Checkout + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: ${{ matrix.python-version }} + + - name: Set up uv + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + + - name: Install dependencies + run: .github/scripts/uv_sync_with_retries.sh --frozen --extra proxy --extra cli --group dev --group proxy-dev --python ${{ matrix.python-version }} + + - name: Create the loopback-only network namespace + run: | + sudo ip netns add smoke + sudo ip netns exec smoke ip link set lo up + cat > "${RUNNER_TEMP}/in-netns" <<'WRAP' + #!/usr/bin/env bash + set -euo pipefail + exec sudo --preserve-env=LITELLM_LOCAL_MODEL_COST_MAP ip netns exec smoke setpriv --reuid "$(id -u)" --regid "$(id -g)" --init-groups -- env HOME="${HOME}" PATH="${PATH}" "$@" + WRAP + chmod +x "${RUNNER_TEMP}/in-netns" + echo "IN_NETNS=${RUNNER_TEMP}/in-netns" >> "${GITHUB_ENV}" + + - name: Verify namespace isolation + run: $IN_NETNS .venv/bin/python .github/scripts/run_merge_smoke.py verify-isolation + + - name: Verify interpreter version + run: $IN_NETNS .venv/bin/python .github/scripts/run_merge_smoke.py interpreter --expect ${{ matrix.python-version }} + + - name: Import and CLI checks + run: $IN_NETNS .venv/bin/python .github/scripts/run_merge_smoke.py cli + + - name: Proxy startup check + run: $IN_NETNS .venv/bin/python .github/scripts/run_merge_smoke.py proxy-startup --diagnostics-dir "${RUNNER_TEMP}/smoke-diagnostics" + + - name: Upload smoke diagnostics + if: always() + uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1 + with: + name: merge-smoke-diagnostics-py${{ matrix.python-version }} + path: ${{ runner.temp }}/smoke-diagnostics + if-no-files-found: ignore + + - name: Remove the network namespace + if: always() + run: sudo ip netns delete smoke diff --git a/.github/workflows/test-code-quality.yml b/.github/workflows/test-code-quality.yml deleted file mode 100644 index 8a3a9107b51..00000000000 --- a/.github/workflows/test-code-quality.yml +++ /dev/null @@ -1,223 +0,0 @@ -name: Code Quality Checks - -on: - pull_request: - branches: - - main - - "litellm_**" - push: - branches: - - main - -permissions: - contents: read - -concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.sha }} - cancel-in-progress: ${{ github.event_name == 'pull_request' }} - -jobs: - code-quality: - runs-on: ubuntu-latest - timeout-minutes: 15 - - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Checkout litellm-docs into docs/my-website (for documentation_tests) - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - repository: BerriAI/litellm-docs - path: docs/my-website - persist-credentials: false - - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Set up uv - uses: ./.github/actions/setup-uv-with-retries - with: - version: "0.10.9" - - - name: Cache uv dependencies - if: github.ref == 'refs/heads/main' - uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 - with: - path: | - ~/.cache/uv - .venv - key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }} - restore-keys: | - ${{ runner.os }}-uv- - - - name: Cache uv dependencies - if: github.ref != 'refs/heads/main' - uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 - with: - path: | - ~/.cache/uv - .venv - key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }} - restore-keys: | - ${{ runner.os }}-uv- - - - name: Cache the Rust build - uses: ./.github/actions/cache-cargo-build - - - name: Install dependencies - run: uv sync --frozen --all-groups --all-extras - - - name: check_licenses - run: uv run --no-sync python ./tests/code_coverage_tests/check_licenses.py - - - name: check_provider_folders_documented - run: uv run --no-sync python ./tests/code_coverage_tests/check_provider_folders_documented.py - - - name: check_prisma_binary_cache - run: uv run --no-sync python ./tests/code_coverage_tests/check_prisma_binary_cache.py - - - name: check_workflow_startup_safety - run: uv run --no-sync python ./tests/code_coverage_tests/check_workflow_startup_safety.py - - - name: check_workflow_job_name_collisions - run: uv run --no-sync python ./tests/code_coverage_tests/check_workflow_job_name_collisions.py - - - name: test_workflow_job_name_collisions - run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_workflow_job_name_collisions.py - - - name: test_unit_passed_gate - run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_unit_passed_gate.py - - - name: test_e2e_changed_gate - run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_e2e_changed_gate.py tests/code_coverage_tests/test_e2e_idp_stack.py - - - name: test_e2e_metadata - env: - PYTHONPATH: tests/e2e - run: | - export LITELLM_MASTER_KEY="sk-$(openssl rand -hex 16)" - uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_e2e_metadata.py tests/code_coverage_tests/test_e2e_junit_report.py - - - name: Check merge smoke harness - run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_merge_smoke.py - - - name: router_code_coverage - run: uv run --no-sync python ./tests/code_coverage_tests/router_code_coverage.py - - - name: test_chat_completion_imports - run: uv run --no-sync python ./tests/code_coverage_tests/test_chat_completion_imports.py - - - name: info_log_check - run: uv run --no-sync python ./tests/code_coverage_tests/info_log_check.py - - - name: check_guardrail_apply_decorator - run: uv run --no-sync python ./tests/code_coverage_tests/check_guardrail_apply_decorator.py - - - name: test_ban_set_verbose - run: uv run --no-sync python ./tests/code_coverage_tests/test_ban_set_verbose.py - - - name: code_qa_check_tests - run: uv run --no-sync python ./tests/code_coverage_tests/code_qa_check_tests.py - - - name: check_get_model_cost_key_performance - run: uv run --no-sync python ./tests/code_coverage_tests/check_get_model_cost_key_performance.py - - - name: test_proxy_types_import - run: uv run --no-sync python ./tests/code_coverage_tests/test_proxy_types_import.py - - - name: callback_manager_test - run: uv run --no-sync python ./tests/code_coverage_tests/callback_manager_test.py - - - name: recursive_detector - run: uv run --no-sync python ./tests/code_coverage_tests/recursive_detector.py - - - name: test_router_strategy_async - run: uv run --no-sync python ./tests/code_coverage_tests/test_router_strategy_async.py - - - name: litellm_logging_code_coverage - run: uv run --no-sync python ./tests/code_coverage_tests/litellm_logging_code_coverage.py - - - name: ensure_async_clients_test - run: uv run --no-sync python ./tests/code_coverage_tests/ensure_async_clients_test.py - - - name: enforce_llms_folder_style - run: uv run --no-sync python ./tests/code_coverage_tests/enforce_llms_folder_style.py - - - name: prevent_key_leaks_in_exceptions - run: uv run --no-sync python ./tests/code_coverage_tests/prevent_key_leaks_in_exceptions.py - - - name: check_unsafe_enterprise_import - run: uv run --no-sync python ./tests/code_coverage_tests/check_unsafe_enterprise_import.py - - - name: ban_copy_deepcopy_kwargs - run: uv run --no-sync python ./tests/code_coverage_tests/ban_copy_deepcopy_kwargs.py - - - name: check_fastuuid_usage - run: uv run --no-sync python ./tests/code_coverage_tests/check_fastuuid_usage.py - - - name: check_py310_typing_imports - run: uv run --no-sync python ./tests/code_coverage_tests/check_py310_typing_imports.py - - - name: check_e2e_no_raw_requests - run: uv run --no-sync python ./tests/code_coverage_tests/check_e2e_no_raw_requests.py - - - name: check_migrations_no_data_rewrites - run: uv run --no-sync python ./tests/code_coverage_tests/check_migrations_no_data_rewrites.py - - - name: check_no_publicly_known_master_key - run: uv run --no-sync python ./tests/code_coverage_tests/check_no_publicly_known_master_key.py - - - name: test_check_no_publicly_known_master_key - run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_check_no_publicly_known_master_key.py - - - name: check_unbounded_in_lists (fails on findings not in the baseline) - run: uv run --no-sync python ./tests/code_coverage_tests/check_unbounded_in_lists.py - - - name: memory_test - run: uv run --no-sync python ./tests/code_coverage_tests/memory_test.py - - - name: documentation_test_env_keys - run: uv run --no-sync python ./tests/documentation_tests/test_env_keys.py - - - name: documentation_test_router_settings - run: uv run --no-sync python ./tests/documentation_tests/test_router_settings.py - - - name: documentation_test_api_docs - run: uv run --no-sync python ./tests/documentation_tests/test_api_docs.py - - python-310-import-smoke: - runs-on: ubuntu-latest - timeout-minutes: 15 - - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Set up Python - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.10" - - - name: Set up uv - uses: ./.github/actions/setup-uv-with-retries - with: - version: "0.10.9" - - - name: Install dependencies - run: uv sync --frozen --extra proxy --extra cli --python 3.10 - - - run: uv run --no-sync python --version - - - name: Import litellm - run: uv run --no-sync python -c "import litellm" - - - name: Check litellm CLI - run: uv run --no-sync litellm --version - - - name: Check lite CLI - run: uv run --no-sync lite version diff --git a/.github/workflows/test-linting.yml b/.github/workflows/test-linting.yml index db2fc969ace..95a32a41ad6 100644 --- a/.github/workflows/test-linting.yml +++ b/.github/workflows/test-linting.yml @@ -1,4 +1,4 @@ -name: LiteLLM Linting +name: lint on: pull_request: @@ -14,22 +14,16 @@ concurrency: cancel-in-progress: ${{ github.event_name == 'pull_request' }} jobs: - lint: - runs-on: ubuntu-latest - timeout-minutes: 15 - # actions: read lets the four lint gates download the base-counts artifacts - # published by publish-lint-base-counts.yml instead of re-scanning the - # merge-base tree in a throwaway worktree. + python: + name: python permissions: contents: read pull-requests: read actions: read - + runs-on: ubuntu-latest + timeout-minutes: 15 steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - # Check out the PR head, not the default refs/pull/N/merge: the merge ref - # folds in newer base commits, which the diff-based gates (ruff delta, - # Any-discipline) would otherwise blame on this branch. with: ref: ${{ github.event.pull_request.head.sha }} fetch-depth: 1 @@ -100,9 +94,6 @@ jobs: if: steps.changes.outputs.decision != 'skip' uses: ./.github/actions/cache-prisma-binaries - # basedpyright resolves Prisma's generated client (litellm/proxy/schema.prisma) - # only after `prisma generate` writes prisma/client.py et al. Without this the - # DB wrappers typed against the generated client would degrade to Unknown. - name: Generate Prisma client if: steps.changes.outputs.decision != 'skip' run: | @@ -211,13 +202,11 @@ jobs: if: steps.changes.outputs.decision != 'skip' run: | uv run --no-sync python -c "from litellm import *" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1) - secret-scan: - runs-on: ubuntu-latest - timeout-minutes: 5 permissions: contents: read - + runs-on: ubuntu-latest + timeout-minutes: 5 steps: - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 with: @@ -249,3 +238,418 @@ jobs: else echo "GITGUARDIAN_API_KEY not set, skipping ggshield scan" fi + ui: + name: ui + permissions: + contents: read + runs-on: ubuntu-latest + timeout-minutes: 8 + defaults: + run: + working-directory: ui/litellm-dashboard + + steps: + - name: Checkout repository + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + fetch-depth: 1 + persist-credentials: false + + - name: Collect changed files + id: changed + env: + GH_TOKEN: ${{ github.token }} + BASE_SHA: ${{ github.event.pull_request.base.sha }} + HEAD_SHA: ${{ github.event.pull_request.head.sha }} + run: | + # base.sha is the base branch tip from when the PR was opened, while + # actions/checkout leaves HEAD on a merge of the PR into the *current* + # base tip. "$BASE_SHA"...HEAD therefore spans every base-branch commit + # landed since, so a PR that touches no UI file still gets linted + # against hundreds of other people's files. Diff the PR head against its + # own merge base instead, which is exactly what this PR changed. + merge_base=$(gh api "repos/${{ github.repository }}/compare/${BASE_SHA}...${HEAD_SHA}?per_page=1" --jq '.merge_base_commit.sha') + test -n "$merge_base" + git fetch --no-tags --depth=1 origin "$merge_base" "$HEAD_SHA" + : > "$RUNNER_TEMP/prettier_files.txt" + : > "$RUNNER_TEMP/eslint_files.txt" + while IFS= read -r f; do + [ -f "$f" ] || continue + case "$f" in + *.js | *.jsx | *.ts | *.tsx | *.mjs | *.cjs) + printf '%s\n' "$f" >> "$RUNNER_TEMP/prettier_files.txt" + printf '%s\n' "$f" >> "$RUNNER_TEMP/eslint_files.txt" ;; + *.json | *.css | *.scss | *.md | *.mdx | *.yml | *.yaml | *.html) + printf '%s\n' "$f" >> "$RUNNER_TEMP/prettier_files.txt" ;; + esac + done < <(git diff --name-only --diff-filter=ACMR --relative "$merge_base" "$HEAD_SHA" -- .) + if [ -s "$RUNNER_TEMP/prettier_files.txt" ] || [ -s "$RUNNER_TEMP/eslint_files.txt" ]; then + echo "has_files=true" >> "$GITHUB_OUTPUT" + else + echo "has_files=false" >> "$GITHUB_OUTPUT" + echo "No lintable UI files changed in this PR; nothing to check." + fi + + - name: Setup Node.js + if: steps.changed.outputs.has_files == 'true' + uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5.0.0 + with: + node-version-file: ui/litellm-dashboard/.nvmrc + cache: "npm" + cache-dependency-path: ui/litellm-dashboard/package-lock.json + + - name: Install dependencies + if: steps.changed.outputs.has_files == 'true' + run: npm ci + + - name: Lint changed files (prettier + eslint) + if: steps.changed.outputs.has_files == 'true' + run: | + prettier_files=() + eslint_files=() + while IFS= read -r f; do prettier_files+=("$f"); done < "$RUNNER_TEMP/prettier_files.txt" + while IFS= read -r f; do eslint_files+=("$f"); done < "$RUNNER_TEMP/eslint_files.txt" + status=0 + if [ ${#prettier_files[@]} -gt 0 ]; then + echo "::group::Prettier (${#prettier_files[@]} files)" + npx prettier --check "${prettier_files[@]}" || { status=1; echo "::error::Unformatted files. Fix with: npm run format"; } + echo "::endgroup::" + fi + if [ ${#eslint_files[@]} -gt 0 ]; then + echo "::group::ESLint (${#eslint_files[@]} files)" + npx eslint --no-warn-ignored --pass-on-unpruned-suppressions "${eslint_files[@]}" || status=1 + echo "::endgroup::" + fi + exit $status + + - name: Check lint budgets + if: ${{ !cancelled() && steps.changed.outputs.has_files == 'true' }} + run: | + npx eslint . -f json -o "$RUNNER_TEMP/lint-report.json" || true + node scripts/check-lint-budgets.mjs "$RUNNER_TEMP/lint-report.json" eslint-budgets.json + + - name: Check for dead code (knip) + if: ${{ !cancelled() && steps.changed.outputs.has_files == 'true' }} + run: npm run knip:ci + ui-api-types: + name: ui-api-types + permissions: + contents: read + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - name: Checkout repository + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + fetch-depth: 2 + + - name: Detect changes that can affect the generated types + id: changes + run: | + set -euo pipefail + if ! base="$(git rev-parse --verify --quiet HEAD^2 >/dev/null && git rev-parse HEAD^1)"; then + echo "Not a pull request merge commit, running the full check." + echo "relevant=true" >> "$GITHUB_OUTPUT" + exit 0 + fi + files="$(git diff --name-only "$base" HEAD)" + if grep -Eq '^(litellm/(proxy|types)/|ui/litellm-dashboard/(src/lib/http/schema\.d\.ts|scripts/gen-api-types\.mjs|package(-lock)?\.json)$|\.github/workflows/test-linting\.yml$)' <<< "$files"; then + echo "relevant=true" >> "$GITHUB_OUTPUT" + else + echo "No proxy, types or generator changes in this pull request, nothing to verify." + echo "relevant=false" >> "$GITHUB_OUTPUT" + fi + + - name: Set up Python + if: steps.changes.outputs.relevant == 'true' + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Set up uv + if: steps.changes.outputs.relevant == 'true' + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + + - name: Cache uv dependencies + if: steps.changes.outputs.relevant == 'true' + uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: | + ~/.cache/uv + .venv + key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }} + restore-keys: | + ${{ runner.os }}-uv- + + - name: Cache the Rust build + if: steps.changes.outputs.relevant == 'true' + uses: ./.github/actions/cache-cargo-build + + - name: Install backend dependencies + if: steps.changes.outputs.relevant == 'true' + run: .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra semantic-router + + - name: Cache Prisma binaries + if: steps.changes.outputs.relevant == 'true' + uses: ./.github/actions/cache-prisma-binaries + + - name: Generate Prisma client + if: steps.changes.outputs.relevant == 'true' + run: uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma + + - name: Regenerate the lazy OpenAPI snapshot + if: steps.changes.outputs.relevant == 'true' + run: uv run --no-sync python -m litellm.proxy._lazy_openapi_snapshot + + - name: Fail if the lazy OpenAPI snapshot is stale + if: steps.changes.outputs.relevant == 'true' + run: | + if ! git diff --exit-code -- litellm/proxy/_lazy_openapi_snapshot.json; then + echo "::error file=litellm/proxy/_lazy_openapi_snapshot.json::The lazy OpenAPI snapshot is out of sync with the lazily loaded routes." + echo "" + echo "A lazily loaded route or model changed without regenerating the snapshot that /openapi.json serves for unloaded features." + echo "To fix, run from the repo root:" + echo " uv run python -m litellm.proxy._lazy_openapi_snapshot" + echo "then run npm run gen:api from ui/litellm-dashboard and commit both files." + exit 1 + fi + echo "_lazy_openapi_snapshot.json is in sync with the lazily loaded routes." + + - name: Set up Node.js + if: steps.changes.outputs.relevant == 'true' + uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5.0.0 + with: + node-version-file: ui/litellm-dashboard/.nvmrc + cache: "npm" + cache-dependency-path: ui/litellm-dashboard/package-lock.json + + - name: Install dashboard dependencies + if: steps.changes.outputs.relevant == 'true' + working-directory: ui/litellm-dashboard + run: npm ci + + - name: Regenerate types from the live spec + if: steps.changes.outputs.relevant == 'true' + working-directory: ui/litellm-dashboard + env: + LITELLM_PYTHON: "uv run --no-sync python" + run: npm run gen:api + + - name: Fail if types are stale + if: steps.changes.outputs.relevant == 'true' + run: | + if ! git diff --exit-code -- ui/litellm-dashboard/src/lib/http/schema.d.ts; then + echo "::error file=ui/litellm-dashboard/src/lib/http/schema.d.ts::Generated API types are out of sync with the proxy OpenAPI spec." + echo "" + echo "A backend route or model changed without regenerating the dashboard types." + echo "To fix, run from ui/litellm-dashboard:" + echo " npm run gen:api" + echo "then commit the updated src/lib/http/schema.d.ts." + exit 1 + fi + echo "schema.d.ts is in sync with the proxy OpenAPI spec." + code-quality: + name: code-quality + permissions: + contents: read + runs-on: ubuntu-latest + timeout-minutes: 15 + + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Checkout litellm-docs into docs/my-website (for documentation_tests) + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + repository: BerriAI/litellm-docs + path: docs/my-website + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Set up uv + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + + - name: Cache uv dependencies + uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: | + ~/.cache/uv + .venv + key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }} + restore-keys: | + ${{ runner.os }}-uv- + + - name: Cache the Rust build + uses: ./.github/actions/cache-cargo-build + + - name: Install dependencies + run: uv sync --frozen --all-groups --all-extras + + - name: check_licenses + run: uv run --no-sync python ./tests/code_coverage_tests/check_licenses.py + + - name: check_provider_folders_documented + run: uv run --no-sync python ./tests/code_coverage_tests/check_provider_folders_documented.py + + - name: check_prisma_binary_cache + run: uv run --no-sync python ./tests/code_coverage_tests/check_prisma_binary_cache.py + + - name: check_workflow_startup_safety + run: uv run --no-sync python ./tests/code_coverage_tests/check_workflow_startup_safety.py + + - name: check_workflow_job_name_collisions + run: uv run --no-sync python ./tests/code_coverage_tests/check_workflow_job_name_collisions.py + + - name: test_workflow_job_name_collisions + run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_workflow_job_name_collisions.py + + - name: test_unit_passed_gate + run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_unit_passed_gate.py + + - name: test_e2e_changed_gate + run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_e2e_changed_gate.py tests/code_coverage_tests/test_e2e_idp_stack.py + + - name: test_e2e_metadata + env: + PYTHONPATH: tests/e2e + run: | + export LITELLM_MASTER_KEY="sk-$(openssl rand -hex 16)" + uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_e2e_metadata.py tests/code_coverage_tests/test_e2e_junit_report.py + + - name: Check merge smoke harness + run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_merge_smoke.py + + - name: router_code_coverage + run: uv run --no-sync python ./tests/code_coverage_tests/router_code_coverage.py + + - name: test_chat_completion_imports + run: uv run --no-sync python ./tests/code_coverage_tests/test_chat_completion_imports.py + + - name: info_log_check + run: uv run --no-sync python ./tests/code_coverage_tests/info_log_check.py + + - name: check_guardrail_apply_decorator + run: uv run --no-sync python ./tests/code_coverage_tests/check_guardrail_apply_decorator.py + + - name: test_ban_set_verbose + run: uv run --no-sync python ./tests/code_coverage_tests/test_ban_set_verbose.py + + - name: code_qa_check_tests + run: uv run --no-sync python ./tests/code_coverage_tests/code_qa_check_tests.py + + - name: check_get_model_cost_key_performance + run: uv run --no-sync python ./tests/code_coverage_tests/check_get_model_cost_key_performance.py + + - name: test_proxy_types_import + run: uv run --no-sync python ./tests/code_coverage_tests/test_proxy_types_import.py + + - name: callback_manager_test + run: uv run --no-sync python ./tests/code_coverage_tests/callback_manager_test.py + + - name: recursive_detector + run: uv run --no-sync python ./tests/code_coverage_tests/recursive_detector.py + + - name: test_router_strategy_async + run: uv run --no-sync python ./tests/code_coverage_tests/test_router_strategy_async.py + + - name: litellm_logging_code_coverage + run: uv run --no-sync python ./tests/code_coverage_tests/litellm_logging_code_coverage.py + + - name: ensure_async_clients_test + run: uv run --no-sync python ./tests/code_coverage_tests/ensure_async_clients_test.py + + - name: enforce_llms_folder_style + run: uv run --no-sync python ./tests/code_coverage_tests/enforce_llms_folder_style.py + + - name: prevent_key_leaks_in_exceptions + run: uv run --no-sync python ./tests/code_coverage_tests/prevent_key_leaks_in_exceptions.py + + - name: check_unsafe_enterprise_import + run: uv run --no-sync python ./tests/code_coverage_tests/check_unsafe_enterprise_import.py + + - name: ban_copy_deepcopy_kwargs + run: uv run --no-sync python ./tests/code_coverage_tests/ban_copy_deepcopy_kwargs.py + + - name: check_fastuuid_usage + run: uv run --no-sync python ./tests/code_coverage_tests/check_fastuuid_usage.py + + - name: check_py310_typing_imports + run: uv run --no-sync python ./tests/code_coverage_tests/check_py310_typing_imports.py + + - name: check_e2e_no_raw_requests + run: uv run --no-sync python ./tests/code_coverage_tests/check_e2e_no_raw_requests.py + + - name: check_migrations_no_data_rewrites + run: uv run --no-sync python ./tests/code_coverage_tests/check_migrations_no_data_rewrites.py + + - name: check_no_publicly_known_master_key + run: uv run --no-sync python ./tests/code_coverage_tests/check_no_publicly_known_master_key.py + + - name: test_check_no_publicly_known_master_key + run: uv run --no-sync pytest -q --noconftest -p no:cacheprovider -c /dev/null tests/code_coverage_tests/test_check_no_publicly_known_master_key.py + + - name: check_unbounded_in_lists (fails on findings not in the baseline) + run: uv run --no-sync python ./tests/code_coverage_tests/check_unbounded_in_lists.py + + - name: memory_test + run: uv run --no-sync python ./tests/code_coverage_tests/memory_test.py + + - name: documentation_test_env_keys + run: uv run --no-sync python ./tests/documentation_tests/test_env_keys.py + + - name: documentation_test_router_settings + run: uv run --no-sync python ./tests/documentation_tests/test_router_settings.py + + - name: documentation_test_api_docs + run: uv run --no-sync python ./tests/documentation_tests/test_api_docs.py + ci-coverage: + name: ci-coverage + permissions: + contents: read + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Assert every test file and Dockerfile is invoked by a job + run: | + python -m pip install "pyyaml==6.0.3" + python .github/scripts/assert_ci_coverage.py + + - name: Assert no -k expression deselects a file from every job that globs it + run: python .github/scripts/assert_ci_coverage.py --slices + + - name: Assert .github/workflows/ holds only workflows, correctly named + run: python .github/scripts/assert_workflow_dir_hygiene.py + lint-passed: + name: lint passed + needs: [python, secret-scan, ui, ui-api-types, code-quality, ci-coverage] + if: always() + runs-on: ubuntu-latest + timeout-minutes: 2 + permissions: {} + steps: + - name: Require every lint job to succeed + env: + NEEDS: ${{ toJSON(needs) }} + run: | + jq -r 'to_entries[] | "\(.key): \(.value.result)"' <<< "$NEEDS" + jq -e 'all(.[]; .result == "success")' <<< "$NEEDS" > /dev/null diff --git a/.github/workflows/test-litellm-ui-lint.yml b/.github/workflows/test-litellm-ui-lint.yml deleted file mode 100644 index 9ea5100e21b..00000000000 --- a/.github/workflows/test-litellm-ui-lint.yml +++ /dev/null @@ -1,105 +0,0 @@ -name: UI Lint -permissions: - contents: read - -on: - pull_request: - branches: - - main - - "litellm_**" - -concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.sha }} - cancel-in-progress: ${{ github.event_name == 'pull_request' }} - -jobs: - frontend-lint: - runs-on: ubuntu-latest - timeout-minutes: 8 - defaults: - run: - working-directory: ui/litellm-dashboard - - steps: - - name: Checkout repository - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - fetch-depth: 1 - persist-credentials: false - - - name: Collect changed files - id: changed - env: - GH_TOKEN: ${{ github.token }} - BASE_SHA: ${{ github.event.pull_request.base.sha }} - HEAD_SHA: ${{ github.event.pull_request.head.sha }} - run: | - # base.sha is the base branch tip from when the PR was opened, while - # actions/checkout leaves HEAD on a merge of the PR into the *current* - # base tip. "$BASE_SHA"...HEAD therefore spans every base-branch commit - # landed since, so a PR that touches no UI file still gets linted - # against hundreds of other people's files. Diff the PR head against its - # own merge base instead, which is exactly what this PR changed. - merge_base=$(gh api "repos/${{ github.repository }}/compare/${BASE_SHA}...${HEAD_SHA}?per_page=1" --jq '.merge_base_commit.sha') - test -n "$merge_base" - git fetch --no-tags --depth=1 origin "$merge_base" "$HEAD_SHA" - : > "$RUNNER_TEMP/prettier_files.txt" - : > "$RUNNER_TEMP/eslint_files.txt" - while IFS= read -r f; do - [ -f "$f" ] || continue - case "$f" in - *.js | *.jsx | *.ts | *.tsx | *.mjs | *.cjs) - printf '%s\n' "$f" >> "$RUNNER_TEMP/prettier_files.txt" - printf '%s\n' "$f" >> "$RUNNER_TEMP/eslint_files.txt" ;; - *.json | *.css | *.scss | *.md | *.mdx | *.yml | *.yaml | *.html) - printf '%s\n' "$f" >> "$RUNNER_TEMP/prettier_files.txt" ;; - esac - done < <(git diff --name-only --diff-filter=ACMR --relative "$merge_base" "$HEAD_SHA" -- .) - if [ -s "$RUNNER_TEMP/prettier_files.txt" ] || [ -s "$RUNNER_TEMP/eslint_files.txt" ]; then - echo "has_files=true" >> "$GITHUB_OUTPUT" - else - echo "has_files=false" >> "$GITHUB_OUTPUT" - echo "No lintable UI files changed in this PR; nothing to check." - fi - - - name: Setup Node.js - if: steps.changed.outputs.has_files == 'true' - uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5.0.0 - with: - node-version-file: ui/litellm-dashboard/.nvmrc - cache: "npm" - cache-dependency-path: ui/litellm-dashboard/package-lock.json - - - name: Install dependencies - if: steps.changed.outputs.has_files == 'true' - run: npm ci - - - name: Lint changed files (prettier + eslint) - if: steps.changed.outputs.has_files == 'true' - run: | - prettier_files=() - eslint_files=() - while IFS= read -r f; do prettier_files+=("$f"); done < "$RUNNER_TEMP/prettier_files.txt" - while IFS= read -r f; do eslint_files+=("$f"); done < "$RUNNER_TEMP/eslint_files.txt" - status=0 - if [ ${#prettier_files[@]} -gt 0 ]; then - echo "::group::Prettier (${#prettier_files[@]} files)" - npx prettier --check "${prettier_files[@]}" || { status=1; echo "::error::Unformatted files. Fix with: npm run format"; } - echo "::endgroup::" - fi - if [ ${#eslint_files[@]} -gt 0 ]; then - echo "::group::ESLint (${#eslint_files[@]} files)" - npx eslint --no-warn-ignored --pass-on-unpruned-suppressions "${eslint_files[@]}" || status=1 - echo "::endgroup::" - fi - exit $status - - - name: Check lint budgets - if: ${{ !cancelled() && steps.changed.outputs.has_files == 'true' }} - run: | - npx eslint . -f json -o "$RUNNER_TEMP/lint-report.json" || true - node scripts/check-lint-budgets.mjs "$RUNNER_TEMP/lint-report.json" eslint-budgets.json - - - name: Check for dead code (knip) - if: ${{ !cancelled() && steps.changed.outputs.has_files == 'true' }} - run: npm run knip:ci diff --git a/.github/workflows/test-litellm-ui-unit.yml b/.github/workflows/test-litellm-ui-unit.yml deleted file mode 100644 index fcd61cedd50..00000000000 --- a/.github/workflows/test-litellm-ui-unit.yml +++ /dev/null @@ -1,99 +0,0 @@ -name: UI Unit Tests -permissions: - contents: read - pull-requests: read - -on: - pull_request: - branches: - - main - - "litellm_**" - push: - branches: - - main - -concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} - cancel-in-progress: true - -jobs: - ui-unit-tests: - runs-on: ubuntu-latest-16-cores - timeout-minutes: 20 - defaults: - run: - working-directory: ui/litellm-dashboard - - steps: - - name: Checkout repository - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - fetch-depth: 1 - persist-credentials: false - - - name: Detect relevant changes - id: changes - uses: ./.github/actions/detect-changes - with: - category: ui - - - name: Setup Node.js - if: steps.changes.outputs.decision != 'skip' - uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5.0.0 - with: - node-version-file: ui/litellm-dashboard/.nvmrc - cache: "npm" - cache-dependency-path: ui/litellm-dashboard/package-lock.json - - - name: Install dependencies - if: steps.changes.outputs.decision != 'skip' - run: npm ci - - - name: Check UI production source types - if: steps.changes.outputs.decision != 'skip' - run: npm run typecheck - - - name: Run UI type tests (Vitest) - if: steps.changes.outputs.decision != 'skip' - env: - CI: "true" - run: npm run test:types - - - name: Run UI unit tests (Vitest) - if: steps.changes.outputs.decision != 'skip' - env: - CI: "true" - GH_TOKEN: ${{ github.token }} - BASE_SHA: ${{ github.event.pull_request.base.sha }} - HEAD_SHA: ${{ github.event.pull_request.head.sha }} - run: | - full_suite() { npm run test -- --run --pool forks --maxWorkers=14; } - - if [ -z "$BASE_SHA" ]; then - echo "Push to $GITHUB_REF_NAME: running the full suite" - full_suite - exit 0 - fi - - merge_base=$(gh api "repos/${{ github.repository }}/compare/${BASE_SHA}...${HEAD_SHA}?per_page=1" --jq '.merge_base_commit.sha') - test -n "$merge_base" - git fetch --no-tags --depth=1 origin "$merge_base" "$HEAD_SHA" - changed_files=() - while IFS= read -r f; do - changed_files+=("$f") - done < <(git diff --name-only --relative "$merge_base" "$HEAD_SHA" -- .) - if [ ${#changed_files[@]} -eq 0 ]; then - echo "No UI files changed in this PR; skipping unit tests." - exit 0 - fi - - scope=$(printf '%s\n' "${changed_files[@]}" | bash "$GITHUB_WORKSPACE/.github/scripts/select_ui_test_scope.sh") - if [ "$scope" != related ]; then - echo "Pull request: ${#changed_files[@]} changed UI files reach outside src/, so related would miss their dependents; running the full suite" - full_suite - exit 0 - fi - - echo "Pull request: running tests related to ${#changed_files[@]} changed UI files" - npm run test -- related "${changed_files[@]}" --run --passWithNoTests \ - --pool forks --maxWorkers=14 diff --git a/.github/workflows/test-merge-smoke.yml b/.github/workflows/test-merge-smoke.yml index 910763c6af2..2472f2831b6 100644 --- a/.github/workflows/test-merge-smoke.yml +++ b/.github/workflows/test-merge-smoke.yml @@ -1,4 +1,4 @@ -name: Merge smoke checks +name: smoke on: pull_request: @@ -14,7 +14,7 @@ concurrency: jobs: dashboard-build: - name: Dashboard build + name: dashboard-build runs-on: ubuntu-24.04 timeout-minutes: 30 steps: @@ -27,7 +27,7 @@ jobs: run: docker build --target ui-builder -f Dockerfile . core-checks: - name: Core checks (Python ${{ matrix.python-version }}) + name: py${{ matrix.python-version }} runs-on: ubuntu-24.04 timeout-minutes: 30 strategy: @@ -79,9 +79,6 @@ jobs: - name: Proxy startup check run: $IN_NETNS .venv/bin/python .github/scripts/run_merge_smoke.py proxy-startup --diagnostics-dir "${RUNNER_TEMP}/smoke-diagnostics" - - name: Run curated smoke cases - run: $IN_NETNS .venv/bin/python .github/scripts/run_merge_smoke.py pytest --manifest .github/merge-smoke-tests.json - - name: Upload smoke diagnostics if: always() uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1 @@ -93,3 +90,18 @@ jobs: - name: Remove the network namespace if: always() run: sudo ip netns delete smoke + + smoke-passed: + name: smoke passed + needs: [dashboard-build, core-checks] + if: always() + runs-on: ubuntu-latest + timeout-minutes: 2 + permissions: {} + steps: + - name: Require every smoke job to succeed + env: + NEEDS: ${{ toJSON(needs) }} + run: | + jq -r 'to_entries[] | "\(.key): \(.value.result)"' <<< "$NEEDS" + jq -e 'all(.[]; .result == "success")' <<< "$NEEDS" > /dev/null diff --git a/.github/workflows/test-unit-documentation.yml b/.github/workflows/test-unit-documentation.yml deleted file mode 100644 index b042e182802..00000000000 --- a/.github/workflows/test-unit-documentation.yml +++ /dev/null @@ -1,103 +0,0 @@ -name: "Unit Tests: Documentation Validation" - -on: - pull_request: - branches: - - main - - "litellm_**" - push: - branches: - - main - -permissions: - contents: read - -concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.sha }} - cancel-in-progress: ${{ github.event_name == 'pull_request' }} - -jobs: - documentation: - runs-on: ubuntu-latest - timeout-minutes: 10 - permissions: - contents: read - pull-requests: read - - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - - name: Detect relevant changes - id: changes - uses: ./.github/actions/detect-changes - - - name: Checkout litellm-docs into docs/my-website (for documentation_tests) - if: steps.changes.outputs.decision != 'skip' - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - repository: BerriAI/litellm-docs - path: docs/my-website - persist-credentials: false - - - name: Set up Python - if: steps.changes.outputs.decision != 'skip' - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 - with: - python-version: "3.12" - - - name: Set up uv - if: steps.changes.outputs.decision != 'skip' - uses: ./.github/actions/setup-uv-with-retries - with: - version: "0.10.9" - - - name: Cache uv dependencies - if: steps.changes.outputs.decision != 'skip' && github.ref == 'refs/heads/main' - uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 - with: - path: | - ~/.cache/uv - .venv - key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }} - restore-keys: | - ${{ runner.os }}-uv- - - - name: Cache uv dependencies - if: steps.changes.outputs.decision != 'skip' && github.ref != 'refs/heads/main' - uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 - with: - path: | - ~/.cache/uv - .venv - key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }} - restore-keys: | - ${{ runner.os }}-uv- - - - name: Cache the Rust build - if: steps.changes.outputs.decision != 'skip' - uses: ./.github/actions/cache-cargo-build - - - name: Install dependencies - if: steps.changes.outputs.decision != 'skip' - run: | - .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra semantic-router - - - name: Cache Prisma binaries - if: steps.changes.outputs.decision != 'skip' - uses: ./.github/actions/cache-prisma-binaries - - - name: Generate Prisma client - if: steps.changes.outputs.decision != 'skip' - run: | - uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma - - # Run the same documentation tests that CircleCI ran (as direct Python scripts) - - name: Run documentation validation tests - if: steps.changes.outputs.decision != 'skip' - run: | - uv run --no-sync python ./tests/documentation_tests/test_env_keys.py - uv run --no-sync python ./tests/documentation_tests/test_router_settings.py - uv run --no-sync python ./tests/documentation_tests/test_api_docs.py - uv run --no-sync python ./tests/documentation_tests/test_circular_imports.py diff --git a/.github/workflows/test-unit.yml b/.github/workflows/test-unit.yml index bd672f4c160..66a8069180f 100644 --- a/.github/workflows/test-unit.yml +++ b/.github/workflows/test-unit.yml @@ -1,4 +1,4 @@ -name: "Unit Tests" +name: unit on: pull_request: @@ -17,26 +17,15 @@ concurrency: group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.sha }} cancel-in-progress: ${{ github.event_name == 'pull_request' }} -# One caller for every tests/test_litellm shard, replacing the nine thin workflow -# files that each wrapped a single call to _test-unit-base.yml. Adding a shard is -# now one matrix entry rather than a new file. -# -# `name` is the shard id, and each check reports as " / Run tests". -# Unit shard names are not required ruleset contexts, so matrix entries can be split freely. -# -# Every entry states its timeouts even when they equal the base workflow's -# defaults. An absent matrix key renders as an empty string, which is not a -# number, so a partially-specified entry would fail the call rather than fall -# back to the default. jobs: rust-bridge: - name: Build the Rust bridge - outputs: - artifact: ${{ steps.rust-bridge-artifact.outputs.name }} - runs-on: ubuntu-latest + name: rust-bridge permissions: contents: read pull-requests: read + outputs: + artifact: ${{ steps.rust-bridge-artifact.outputs.name }} + runs-on: ubuntu-latest env: UV_PYTHON: "3.12" steps: @@ -119,32 +108,46 @@ jobs: RUN_ID: ${{ github.run_id }} RUN_ATTEMPT: ${{ github.run_attempt }} run: echo "name=rust-bridge-${RUN_ID}-${RUN_ATTEMPT}" >> "$GITHUB_OUTPUT" - - unit: - name: ${{ matrix.shard }} - needs: rust-bridge - if: ${{ !cancelled() }} + assert-shard-coverage: permissions: contents: read - id-token: write - pull-requests: write + runs-on: ubuntu-latest + timeout-minutes: 2 + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + - name: Assert every test directory and file is claimed by a shard + run: python3 .github/scripts/assert_ci_coverage.py --shards + unit: + name: ${{ matrix.shard }} + needs: [rust-bridge, assert-shard-coverage] + if: ${{ !cancelled() }} + runs-on: ubuntu-latest + timeout-minutes: ${{ matrix.job-timeout-minutes }} + permissions: + contents: read + pull-requests: read + env: + UV_PYTHON: ${{ matrix.python-version }} + LITELLM_LOCAL_MODEL_COST_MAP: "True" strategy: fail-fast: false matrix: include: - shard: core-utils - artifact-name: core-utils - test-path: >- + test-path: |- tests/unit/decisions tests/unit/litellm_core_utils + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: enterprise-routing - artifact-name: enterprise-routing - test-path: >- + test-path: |- tests/unit/google_genai tests/unit/router_strategy tests/unit/router_utils @@ -160,69 +163,246 @@ jobs: tests/unit/enterprise/proxy/test_file_deletion_blocking.py tests/unit/enterprise/proxy/test_managed_files_access_check.py tests/unit/enterprise/proxy/test_managed_files_hook.py + python-version: "3.12" workers: 4 reruns: 2 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: integrations - artifact-name: integrations - test-path: >- + test-path: |- tests/test_litellm/integrations tests/test_litellm/tracing tests/unit/integrations + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - - shard: Vertex AI - artifact-name: llm-vertex-ai + - shard: llms-vertex-ai test-path: >- tests/unit/llms/vertex_ai + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - - shard: All Other Providers - artifact-name: llm-other-providers + - shard: llms-anthropic test-path: >- - tests/unit/llms - --ignore=tests/unit/llms/vertex_ai - --ignore=tests/unit/llms/openai - --ignore=tests/unit/llms/meta - --ignore=tests/unit/llms/base_llm/batches/base_batches_config_test.py + tests/unit/llms/anthropic + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - - shard: OpenAI and Meta Providers - artifact-name: llm-openai-meta + - shard: llms-bedrock + test-path: |- + tests/unit/llms/bedrock + tests/unit/llms/bedrock_mantle + python-version: "3.12" + workers: 4 + reruns: 0 + timeout-minutes: 20 + job-timeout-minutes: 60 + dist: loadscope + + - shard: llms-a-to-g + test-path: |- + tests/unit/llms/test_cache_control_and_reasoning.py + tests/unit/llms/test_custom_llm.py + tests/unit/llms/test_file_content_block.py + tests/unit/llms/test_file_search_responses.py + tests/unit/llms/test_lifecycle_fix.py + tests/unit/llms/test_oss_decision.py + tests/unit/llms/test_polling_url_origin_match.py + tests/unit/llms/test_predibase_transformation.py + tests/unit/llms/a2a + tests/unit/llms/aiml + tests/unit/llms/aiohttp_openai + tests/unit/llms/apiserpent + tests/unit/llms/aws_polly + tests/unit/llms/azure + tests/unit/llms/azure_ai + tests/unit/llms/base_llm + tests/unit/llms/baseten + tests/unit/llms/black_forest_labs + tests/unit/llms/bytez + tests/unit/llms/cerebras + tests/unit/llms/chat + tests/unit/llms/chatgpt + tests/unit/llms/clarifai + tests/unit/llms/claude_code + tests/unit/llms/cloudflare + tests/unit/llms/codex + tests/unit/llms/cohere + tests/unit/llms/cometapi + tests/unit/llms/compactifai + tests/unit/llms/crusoe + tests/unit/llms/custom_httpx + tests/unit/llms/dashscope + tests/unit/llms/databricks + tests/unit/llms/dataforseo + tests/unit/llms/datarobot + tests/unit/llms/deepagents + tests/unit/llms/deepgram + tests/unit/llms/deepinfra + tests/unit/llms/deepseek + tests/unit/llms/docker_model_runner + tests/unit/llms/duckduckgo + tests/unit/llms/e2b + tests/unit/llms/edenai + tests/unit/llms/elevenlabs + tests/unit/llms/exa_ai + tests/unit/llms/fal_ai + tests/unit/llms/fastcrw + tests/unit/llms/featherless_ai + tests/unit/llms/firecrawl + tests/unit/llms/fireworks_ai + tests/unit/llms/gdc + tests/unit/llms/gemini + tests/unit/llms/gigachat + tests/unit/llms/github_copilot + tests/unit/llms/gradient_ai + tests/unit/llms/groq + --ignore=tests/unit/llms/base_llm/batches/base_batches_config_test.py + python-version: "3.12" + workers: 4 + reruns: 0 + timeout-minutes: 20 + job-timeout-minutes: 60 + dist: loadscope + + - shard: llms-h-to-z + test-path: |- + tests/unit/llms/heroku + tests/unit/llms/hosted_vllm + tests/unit/llms/huggingface + tests/unit/llms/hyperbolic + tests/unit/llms/inception + tests/unit/llms/infinity + tests/unit/llms/jina_ai + tests/unit/llms/lambda_ai + tests/unit/llms/langflow + tests/unit/llms/langgraph + tests/unit/llms/laya + tests/unit/llms/lemonade + tests/unit/llms/linkup + tests/unit/llms/litellm_proxy + tests/unit/llms/llamafile + tests/unit/llms/lm_studio + tests/unit/llms/manus + tests/unit/llms/meta_llama + tests/unit/llms/minimax + tests/unit/llms/mistral + tests/unit/llms/modelscope + tests/unit/llms/mongodb + tests/unit/llms/moonshot + tests/unit/llms/nadir + tests/unit/llms/nebius + tests/unit/llms/neosantara + tests/unit/llms/nimble + tests/unit/llms/novita + tests/unit/llms/nscale + tests/unit/llms/nvidia_nim + tests/unit/llms/nvidia_riva + tests/unit/llms/oci + tests/unit/llms/ocr + tests/unit/llms/ollama + tests/unit/llms/oobabooga + tests/unit/llms/openai_like + tests/unit/llms/opencode + tests/unit/llms/openrouter + tests/unit/llms/ovhcloud + tests/unit/llms/parallel_ai + tests/unit/llms/parasail + tests/unit/llms/pass_through + tests/unit/llms/perplexity + tests/unit/llms/pg_vector + tests/unit/llms/publicai + tests/unit/llms/ragflow + tests/unit/llms/recraft + tests/unit/llms/reducto + tests/unit/llms/replicate + tests/unit/llms/runwayml + tests/unit/llms/s3_vectors + tests/unit/llms/sagemaker + tests/unit/llms/sail + tests/unit/llms/sambanova + tests/unit/llms/sap + tests/unit/llms/scaleway + tests/unit/llms/searchapi + tests/unit/llms/searxng + tests/unit/llms/serper + tests/unit/llms/snowflake + tests/unit/llms/soniox + tests/unit/llms/stability + tests/unit/llms/tavily + tests/unit/llms/tencent + tests/unit/llms/tinyfish + tests/unit/llms/together_ai + tests/unit/llms/tool_loop + tests/unit/llms/triton + tests/unit/llms/v0 + tests/unit/llms/valkey + tests/unit/llms/vercel_ai_gateway + tests/unit/llms/volcengine + tests/unit/llms/voyage + tests/unit/llms/wandb + tests/unit/llms/watsonx + tests/unit/llms/xai + tests/unit/llms/xinference + tests/unit/llms/you_com + tests/unit/llms/zai + python-version: "3.12" + workers: 4 + reruns: 0 + timeout-minutes: 20 + job-timeout-minutes: 60 + dist: loadscope + + - shard: llms-openai-meta test-path: >- tests/unit/llms/openai tests/unit/llms/meta + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - - shard: misc - artifact-name: misc + - shard: root test-path: >- tests/test_litellm/test_*.py tests/unit/test_*.py + python-version: "3.12" workers: 4 reruns: 2 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - - shard: misc-dirs - artifact-name: misc-dirs + - shard: router test-path: >- tests/unit/test_router + python-version: "3.12" + workers: 4 + reruns: 0 + timeout-minutes: 20 + job-timeout-minutes: 60 + dist: loadscope + + - shard: endpoints + test-path: >- tests/unit/a2a_protocol + tests/unit/anthropic_interface tests/unit/batches tests/unit/chat_completions tests/unit/completion_extras @@ -230,25 +410,58 @@ jobs: tests/unit/embeddings tests/unit/endpoints tests/unit/files - tests/unit/harness tests/unit/images tests/unit/interactions tests/unit/messages + tests/unit/ocr + tests/unit/passthrough tests/unit/rag + tests/unit/realtime_api tests/unit/rerank_api - tests/unit/rust_bridge - tests/unit/secret_managers tests/unit/vector_stores tests/unit/videos - --ignore=tests/unit/rust_bridge/native_route_wheel_test.py + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope + + - shard: rust-bridge-harness + test-path: >- + tests/unit/compression + tests/unit/harness + tests/unit/rust_bridge + tests/unit/sandbox + --ignore=tests/unit/rust_bridge/native_route_wheel_test.py + python-version: "3.12" + workers: 4 + reruns: 0 + timeout-minutes: 20 + job-timeout-minutes: 60 + dist: loadscope + + - shard: enterprise-repositories-secrets + test-path: >- + tests/proxy_behavior/lens/test_connection.py + tests/unit/enterprise/enterprise_callbacks/test_callback_controls.py + tests/unit/enterprise/enterprise_callbacks/test_llm_guard.py + tests/unit/enterprise/enterprise_callbacks/test_secret_detection.py + tests/unit/integration_support + tests/unit/models + tests/unit/repositories + tests/unit/secret_managers + tests/unit/skills/test_skills_main.py + tests/unit/tracing + python-version: "3.12" + workers: 2 + reruns: 0 + timeout-minutes: 20 + job-timeout-minutes: 60 + dist: loadscope - shard: proxy-auth - artifact-name: proxy-auth - test-path: >- + test-path: |- tests/unit/proxy/auth --ignore=tests/unit/proxy/auth/test_auth_checks.py --ignore=tests/unit/proxy/auth/test_user_api_key_auth.py @@ -257,27 +470,29 @@ jobs: --ignore=tests/unit/proxy/auth/test_models_fallback_endpoint.py --ignore=tests/unit/proxy/auth/test_multipart_bypass_repro.py --ignore=tests/unit/proxy/auth/test_proxy_routes.py + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: proxy-hooks-client - artifact-name: proxy-hooks-client - test-path: >- + test-path: |- tests/unit/proxy/hooks tests/unit/proxy/policy_engine tests/unit/proxy/client --ignore=tests/unit/proxy/hooks/test_banned_keyword_list.py --ignore=tests/unit/proxy/hooks/test_unit_test_max_model_budget_limiter.py + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: proxy-endpoints - artifact-name: proxy-endpoints - test-path: >- + test-path: |- tests/unit/proxy/management_endpoints tests/unit/proxy/management_helpers tests/unit/proxy/list_api @@ -292,14 +507,15 @@ jobs: --ignore=tests/unit/proxy/management_endpoints/test_key_generate_prisma.py --ignore=tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py --ignore=tests/unit/proxy/management_helpers/test_audit_logs_proxy.py + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: proxy-feature-endpoints - artifact-name: proxy-feature-endpoints - test-path: >- + test-path: |- tests/unit/proxy/guardrails --ignore=tests/unit/proxy/google_endpoints/test_gemini_agents_endpoints.py --ignore=tests/unit/proxy/google_endpoints/test_google_endpoint_routing.py @@ -328,22 +544,25 @@ jobs: tests/unit/proxy/ui_crud_endpoints tests/unit/proxy/config_resolvers tests/unit/proxy/utils + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: proxy-server - artifact-name: proxy-server - test-path: "tests/unit/proxy/proxy_server" + test-path: >- + tests/unit/proxy/proxy_server + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 60 job-timeout-minutes: 100 + dist: loadscope - shard: mcp-elicitation - artifact-name: mcp-elicitation - test-path: >- + test-path: |- --override-ini=pythonpath=tests tests/unit/experimental_mcp_client/test_mcp_client.py tests/unit/proxy/_experimental/mcp_server/test_capabilities.py @@ -353,14 +572,15 @@ jobs: tests/unit/proxy/_experimental/mcp_server/test_mcp_server_tool_calls_and_headers.py tests/unit/proxy/_experimental/mcp_server/test_operations.py tests/integration/mcp/test_interactions.py + python-version: "3.12" workers: 2 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: proxy-infra - artifact-name: proxy-infra - test-path: >- + test-path: |- tests/unit/proxy/db --ignore=tests/unit/proxy/db/db_transaction_queue/test_e2e_pod_lock_manager.py --ignore=tests/unit/proxy/db/test_update_daily_tag_spend.py @@ -369,8 +589,6 @@ jobs: tests/unit/proxy/spend_tracking --ignore=tests/unit/proxy/spend_tracking/test_search_api_logging.py tests/unit/proxy/pass_through_endpoints - tests/unit/proxy/_experimental - --ignore=tests/unit/proxy/_experimental/mcp_server tests/unit/proxy/experimental tests/unit/proxy/common_utils --ignore=tests/unit/proxy/common_utils/test_cache_aware_routing.py @@ -385,14 +603,15 @@ jobs: tests/unit/proxy/management tests/unit/proxy/management_endpoints/test_roi_calculator_endpoints.py tests/unit/proxy/roi_calculator + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: proxy-infra-root - artifact-name: proxy-infra-root - test-path: >- + test-path: |- tests/unit/proxy/test_*.py --ignore=tests/unit/proxy/test_aproxy_startup.py --ignore=tests/unit/proxy/test_credential_slot_registry.py @@ -419,32 +638,35 @@ jobs: --ignore=tests/unit/proxy/test_unit_test_proxy_hooks.py --ignore=tests/unit/proxy/test_update_spend.py --ignore=tests/unit/proxy/test_zero_cost_model_budget_bypass.py + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: caching-local - artifact-name: caching-local test-path: >- tests/unit/caching + python-version: "3.12" workers: 2 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: proxy-extras - artifact-name: proxy-extras test-path: >- tests/unit/litellm_proxy_extras + python-version: "3.12" workers: 2 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: enterprise-package - artifact-name: enterprise-package - test-path: >- + test-path: |- tests/unit/enterprise/integrations tests/unit/enterprise/proxy/auth tests/unit/enterprise/proxy/guardrails @@ -452,15 +674,17 @@ jobs: tests/unit/enterprise/proxy/test_audit_logging_endpoints.py tests/unit/enterprise/proxy/test_liteadmin.py tests/unit/enterprise/enterprise_callbacks/test_prometheus_logging_callbacks.py + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - shard: enterprise-managed-files - artifact-name: enterprise-managed-files test-path: >- tests/unit/enterprise/proxy/hooks + python-version: "3.12" workers: 4 reruns: 0 timeout-minutes: 20 @@ -468,181 +692,148 @@ jobs: dist: load - shard: responses-caching-types - artifact-name: responses-caching-types - test-path: >- - tests/unit/responses + test-path: |- + tests/unit/responses/litellm_completion_transformation + tests/unit/responses/test_additional_tools.py + tests/unit/responses/test_custom_tool_call.py + tests/unit/responses/test_dispatch.py + tests/unit/responses/test_metadata_codex_callback.py + tests/unit/responses/test_no_duplicate_spend_logs.py + tests/unit/responses/test_null_test_fix.py + tests/unit/responses/test_responses_api_bridge_flag.py + tests/unit/responses/test_responses_api_lifecycle.py + tests/unit/responses/test_responses_api_request_body.py + tests/unit/responses/test_responses_prompt_management.py + tests/unit/responses/test_responses_router_cooldown.py + tests/unit/responses/test_responses_streaming_iterator.py + tests/unit/responses/test_responses_supported_endpoints_passthrough.py + tests/unit/responses/test_responses_utils.py + tests/unit/responses/test_responses_websocket_all_providers.py + tests/unit/responses/test_rust_bridge_websocket.py + tests/unit/responses/test_sse_output_recovery.py + tests/unit/responses/test_streaming_iterator.py + tests/unit/responses/test_streaming_iterator_error_events.py + tests/unit/responses/test_text_format_conversion.py tests/unit/types - --ignore=tests/unit/responses/mcp + python-version: "3.12" workers: 2 reruns: 0 timeout-minutes: 20 job-timeout-minutes: 60 + dist: loadscope - - shard: unit - artifact-name: unit - test-path: >- - tests/unit/anthropic_interface - tests/unit/compression - tests/unit/enterprise/enterprise_callbacks/test_callback_controls.py - tests/unit/enterprise/enterprise_callbacks/test_llm_guard.py - tests/unit/enterprise/enterprise_callbacks/test_secret_detection.py - tests/unit/integration_support - tests/unit/models - tests/unit/ocr - tests/unit/passthrough - tests/unit/realtime_api - tests/unit/repositories - tests/unit/sandbox - tests/unit/skills/test_skills_main.py - tests/unit/tracing - tests/proxy_behavior/lens/test_connection.py - workers: 2 - reruns: 0 - timeout-minutes: 20 - job-timeout-minutes: 60 - uses: ./.github/workflows/_test-unit-base.yml - with: - rust-bridge-artifact: ${{ needs.rust-bridge.outputs.artifact }} - test-path: ${{ matrix.test-path }} - workers: ${{ matrix.workers }} - reruns: ${{ matrix.reruns }} - timeout-minutes: ${{ matrix.timeout-minutes }} - job-timeout-minutes: ${{ matrix.job-timeout-minutes }} - dist: ${{ matrix.dist || 'loadscope' }} - artifact-name: ${{ matrix.artifact-name }} - - lens-python-310: - name: Lens Python 3.10 - permissions: - contents: read - id-token: write - pull-requests: write - uses: ./.github/workflows/_test-unit-base.yml - with: - python-version: "3.10" - test-path: tests/unit/proxy/lens/test_inference.py - workers: 0 - reruns: 0 - timeout-minutes: 5 - artifact-name: lens-python-310 - - # Fast guard — fails the workflow when a test directory or file inside a sharded - # tree is claimed by no shard. The semantic-shard design has no catch-all bucket, - # so an unassigned child runs nowhere; assert_ci_coverage.py holds the tree list - # and reads the same test-path keys the coverage census does. - assert-shard-coverage: - runs-on: ubuntu-latest - timeout-minutes: 2 - permissions: - contents: read - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - name: Assert every test directory and file is claimed by a shard - run: python3 .github/scripts/assert_ci_coverage.py --shards - - proxy-db: - needs: assert-shard-coverage - # Display only the semantic shard name in the checks UI instead of GHA's - # default "proxy-db (key-generation, tests/unit/proxy/…, 0, loadscope, 20)" - # which includes every matrix field and gets truncated past the test-path. - name: ${{ matrix.test-group }} - permissions: - contents: read - id-token: write - pull-requests: write - strategy: - fail-fast: false - matrix: - include: - # Must run serially — event-loop conflict with the logging worker. - - test-group: key-generation + - shard: key-generation test-path: >- tests/unit/proxy/management_endpoints/test_key_generate_prisma.py + python-version: "3.12" workers: 0 + reruns: 2 + timeout-minutes: 20 + job-timeout-minutes: 60 dist: loadscope - timeout: 20 - # ---- auth: split into 2 shards ---- - - test-group: auth-checks - test-path: >- + - shard: auth-checks + test-path: |- tests/unit/proxy/auth/test_auth_checks.py tests/unit/proxy/auth/test_user_api_key_auth.py tests/unit/proxy/test_credential_slot_registry.py tests/unit/proxy/test_deprecated_key_grace_period.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - - test-group: jwt-and-keys - test-path: >- + + - shard: jwt-and-keys + test-path: |- tests/unit/proxy/auth/test_jwt.py tests/unit/proxy/management_endpoints/test_jwt_key_mapping.py tests/unit/proxy/test_proxy_custom_auth.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - # ---- test_proxy_utils.py, single shard, worksteal distribution ---- - - test-group: proxy-utils + - shard: proxy-utils test-path: >- tests/unit/proxy/test_proxy_utils.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: worksteal - timeout: 15 - # ---- proxy server: split into 2 shards ---- - - test-group: proxy-server-core - test-path: >- + - shard: proxy-server-core + test-path: |- tests/proxy_unit_tests/test_proxy_server_gemini_pass_through.py tests/unit/proxy/test__lazy_features.py tests/unit/proxy/test_aproxy_startup.py tests/unit/proxy/test_proxy_server.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - - test-group: proxy-runtime - test-path: >- + + - shard: proxy-runtime + test-path: |- tests/unit/proxy/auth/test_multipart_bypass_repro.py tests/unit/proxy/auth/test_proxy_routes.py tests/unit/proxy/middleware/test_request_size_limit_middleware.py tests/unit/proxy/test_proxy_config_unit_test.py tests/unit/proxy/test_proxy_token_counter.py tests/unit/proxy/test_server_root_path.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - - test-group: mcp-oauth - test-path: >- + - shard: mcp-oauth + test-path: |- tests/unit/proxy/_experimental/mcp_server/test_discoverable_endpoints.py tests/unit/proxy/_experimental/mcp_server/test_mcp_server_manager.py tests/unit/proxy/_experimental/mcp_server/test_db_credentials.py tests/unit/proxy/_experimental/mcp_server/outbound_credentials + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - # ---- logging: split into 2 shards ---- - - test-group: custom-logging - test-path: >- + - shard: custom-logging + test-path: |- tests/proxy_unit_tests/test_proxy_custom_logger.py tests/unit/proxy/test_custom_callback_input.py tests/unit/proxy/test_custom_logger_s3_gcs.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - - test-group: logging-misc - test-path: >- + + - shard: logging-misc + test-path: |- tests/unit/proxy/management_helpers/test_audit_logs_proxy.py tests/unit/proxy/spend_tracking/test_search_api_logging.py tests/unit/proxy/test_proxy_reject_logging.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - - test-group: db-and-spend - test-path: >- + - shard: db-and-spend + test-path: |- tests/unit/proxy/common_utils/test_proxy_encrypt_decrypt.py tests/unit/proxy/db/db_transaction_queue/test_e2e_pod_lock_manager.py tests/unit/proxy/db/test_update_daily_tag_spend.py @@ -650,30 +841,39 @@ jobs: tests/unit/proxy/test_prisma_client_backoff_retry.py tests/unit/proxy/test_update_spend.py tests/unit/skills/test_skills_db.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - # ---- guardrails + budget + hooks: split into 2 ---- - - test-group: guardrails-hooks - test-path: >- + - shard: guardrails-hooks + test-path: |- tests/unit/proxy/hooks/test_banned_keyword_list.py tests/unit/proxy/test_proxy_setting_guardrails.py tests/unit/proxy/test_unit_test_proxy_hooks.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - - test-group: budgets - test-path: >- + + - shard: budgets + test-path: |- tests/unit/proxy/auth/test_default_end_user_budget_simple.py tests/unit/proxy/hooks/test_unit_test_max_model_budget_limiter.py tests/unit/proxy/test_zero_cost_model_budget_bypass.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - - test-group: endpoints-and-responses - test-path: >- + - shard: endpoints-and-responses + test-path: |- tests/proxy_unit_tests/test_proxy_exception_mapping.py tests/unit/proxy/lens tests/unit/proxy/auth/test_models_fallback_endpoint.py @@ -692,25 +892,379 @@ jobs: tests/unit/proxy/test_reducto_ocr_route.py tests/unit/proxy/test_response_polling_pre_call_checks.py tests/unit/proxy/test_ui_path_detection.py + python-version: "3.12" workers: 4 + reruns: 2 + timeout-minutes: 15 + job-timeout-minutes: 60 dist: loadscope - timeout: 15 - uses: ./.github/workflows/_test-unit-base.yml - with: - test-path: ${{ matrix.test-path }} - workers: ${{ matrix.workers }} - reruns: 2 - timeout-minutes: ${{ matrix.timeout }} - dist: ${{ matrix.dist }} - artifact-name: proxy-db-${{ matrix.test-group }} + - shard: lens-python-310 + test-path: >- + tests/unit/proxy/lens/test_inference.py + python-version: "3.10" + workers: 0 + reruns: 0 + timeout-minutes: 5 + job-timeout-minutes: 60 + dist: loadscope + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + timeout-minutes: 3 + with: + persist-credentials: false + + - name: Detect relevant changes + id: changes + timeout-minutes: 2 + uses: ./.github/actions/detect-changes + + - name: Set up Python + if: steps.changes.outputs.decision != 'skip' + timeout-minutes: 3 + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: ${{ env.UV_PYTHON }} + + - name: Set up uv + if: steps.changes.outputs.decision != 'skip' + timeout-minutes: 3 + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + + - name: Cache uv dependencies + if: steps.changes.outputs.decision != 'skip' + timeout-minutes: 5 + uses: ./.github/actions/cache-uv-downloads + + - name: Set up the Rust build + if: steps.changes.outputs.decision != 'skip' + timeout-minutes: 5 + uses: ./.github/actions/rust-bridge + with: + artifact: ${{ needs.rust-bridge.outputs.artifact }} + + - name: Install dependencies + if: steps.changes.outputs.decision != 'skip' + timeout-minutes: 8 + env: + RUST_BRIDGE_ARTIFACT: ${{ needs.rust-bridge.outputs.artifact }} + run: | + diff -u model_prices_and_context_window.json litellm/model_prices_and_context_window_backup.json + if [ -z "$RUST_BRIDGE_ARTIFACT" ]; then + .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra semantic-router --extra saml --extra caching --extra extra_proxy --extra proxy-runtime --extra utils + else + .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra semantic-router --extra saml --extra caching --extra extra_proxy --extra proxy-runtime --extra utils --no-install-project + uv pip install --no-deps --python .venv/bin/python rust-bridge-dist/*.whl + cp rust-bridge-dist/litellm/rust_bridge/_native.abi3.so litellm/rust_bridge/_native.abi3.so + uv run --no-sync python -c "import importlib.metadata; import litellm.rust_bridge._native; print(importlib.metadata.version('litellm'))" + fi + uv run --no-sync python -c 'import os, sys; print(sys.version); assert f"{sys.version_info.major}.{sys.version_info.minor}" == os.environ["UV_PYTHON"]' + + - name: Cache Prisma binaries + if: steps.changes.outputs.decision != 'skip' + timeout-minutes: 3 + uses: ./.github/actions/cache-prisma-binaries + + - name: Generate Prisma client + if: steps.changes.outputs.decision != 'skip' + timeout-minutes: 3 + run: | + uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma + + - name: Run tests + id: tests + if: steps.changes.outputs.decision != 'skip' + timeout-minutes: ${{ matrix.timeout-minutes }} + env: + TEST_PATH: ${{ matrix.test-path }} + MAX_FAILURES: "10" + WORKERS: ${{ matrix.workers }} + RERUNS: ${{ matrix.reruns }} + TEST_TIMEOUT_SECONDS: "120" + DIST: ${{ matrix.dist }} + COVERAGE_CORE: sysmon + run: | + echo "has-coverage=false" >> "$GITHUB_OUTPUT" + selection="${TEST_PATH}" + if [ -z "${selection// /}" ]; then + echo "shard selection is empty; nothing to run" + exit 0 + fi + pytest_args=() + existing_paths=0 + for token in ${selection}; do + case "${token}" in + -*) pytest_args+=("${token}") ;; + *) + if [ -e "${token%%::*}" ]; then + pytest_args+=("${token}") + existing_paths=$((existing_paths + 1)) + else + echo "::warning::${token} does not exist; drop it from this shard's test-path" + fi + ;; + esac + done + if [ "${existing_paths}" -eq 0 ]; then + echo "No path in the selection exists (${selection}); nothing to run" + exit 0 + fi + xdist_args=() + if [ "${WORKERS}" != "0" ]; then + xdist_args=(-n "${WORKERS}" --dist="${DIST}") + fi + set +e + uv run --no-sync pytest "${pytest_args[@]}" \ + --tb=short -vv \ + --maxfail="${MAX_FAILURES}" \ + "${xdist_args[@]}" \ + --reruns "${RERUNS}" \ + --reruns-delay 1 \ + --timeout="${TEST_TIMEOUT_SECONDS}" \ + --rerun-except "from pytest-timeout" \ + --durations=20 \ + --cov=./litellm --cov=./enterprise/litellm_enterprise \ + --cov-report=xml:coverage.xml \ + --cov-config=pyproject.toml + status=$? + set -e + if [ -f coverage.xml ]; then + echo "has-coverage=true" >> "$GITHUB_OUTPUT" + fi + if [ "$status" -eq 5 ]; then + echo "pytest collected no tests from ${selection}; passing" + exit 0 + fi + exit "$status" + + - name: Save coverage report + if: always() && steps.changes.outputs.decision != 'skip' + uses: actions/upload-artifact@4cec3d8aa04e39d1a68397de0c4cd6fb9dce8ec1 # v4.6.1 + with: + name: coverage-${{ matrix.shard }}-${{ github.run_id }}-${{ github.run_attempt }} + path: coverage.xml + retention-days: 1 + ui-unit: + name: ui-unit + permissions: + contents: read + pull-requests: read + runs-on: ubuntu-latest-16-cores + timeout-minutes: 20 + defaults: + run: + working-directory: ui/litellm-dashboard + + steps: + - name: Checkout repository + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + fetch-depth: 1 + persist-credentials: false + + - name: Detect relevant changes + id: changes + uses: ./.github/actions/detect-changes + with: + category: ui + + - name: Setup Node.js + if: steps.changes.outputs.decision != 'skip' + uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5.0.0 + with: + node-version-file: ui/litellm-dashboard/.nvmrc + cache: "npm" + cache-dependency-path: ui/litellm-dashboard/package-lock.json + + - name: Install dependencies + if: steps.changes.outputs.decision != 'skip' + run: npm ci + + - name: Check UI production source types + if: steps.changes.outputs.decision != 'skip' + run: npm run typecheck + + - name: Run UI type tests (Vitest) + if: steps.changes.outputs.decision != 'skip' + env: + CI: "true" + run: npm run test:types + + - name: Run UI unit tests (Vitest) + if: steps.changes.outputs.decision != 'skip' + env: + CI: "true" + GH_TOKEN: ${{ github.token }} + BASE_SHA: ${{ github.event.pull_request.base.sha }} + HEAD_SHA: ${{ github.event.pull_request.head.sha }} + run: | + full_suite() { npm run test -- --run --pool forks --maxWorkers=14; } + + if [ -z "$BASE_SHA" ]; then + echo "Push to $GITHUB_REF_NAME: running the full suite" + full_suite + exit 0 + fi + + merge_base=$(gh api "repos/${{ github.repository }}/compare/${BASE_SHA}...${HEAD_SHA}?per_page=1" --jq '.merge_base_commit.sha') + test -n "$merge_base" + git fetch --no-tags --depth=1 origin "$merge_base" "$HEAD_SHA" + changed_files=() + while IFS= read -r f; do + changed_files+=("$f") + done < <(git diff --name-only --relative "$merge_base" "$HEAD_SHA" -- .) + if [ ${#changed_files[@]} -eq 0 ]; then + echo "No UI files changed in this PR; skipping unit tests." + exit 0 + fi + + scope=$(printf '%s\n' "${changed_files[@]}" | bash "$GITHUB_WORKSPACE/.github/scripts/select_ui_test_scope.sh") + if [ "$scope" != related ]; then + echo "Pull request: ${#changed_files[@]} changed UI files reach outside src/, so related would miss their dependents; running the full suite" + full_suite + exit 0 + fi + + echo "Pull request: running tests related to ${#changed_files[@]} changed UI files" + npm run test -- related "${changed_files[@]}" --run --passWithNoTests \ + --pool forks --maxWorkers=14 + docs: + name: docs + permissions: + contents: read + pull-requests: read + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Detect relevant changes + id: changes + uses: ./.github/actions/detect-changes + + - name: Checkout litellm-docs into docs/my-website (for documentation_tests) + if: steps.changes.outputs.decision != 'skip' + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + repository: BerriAI/litellm-docs + path: docs/my-website + persist-credentials: false + + - name: Set up Python + if: steps.changes.outputs.decision != 'skip' + uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0 + with: + python-version: "3.12" + + - name: Set up uv + if: steps.changes.outputs.decision != 'skip' + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + + - name: Cache uv dependencies + if: steps.changes.outputs.decision != 'skip' && github.ref == 'refs/heads/main' + uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: | + ~/.cache/uv + .venv + key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }} + restore-keys: | + ${{ runner.os }}-uv- + + - name: Cache uv dependencies + if: steps.changes.outputs.decision != 'skip' && github.ref != 'refs/heads/main' + uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: | + ~/.cache/uv + .venv + key: ${{ runner.os }}-uv-${{ hashFiles('uv.lock') }} + restore-keys: | + ${{ runner.os }}-uv- + + - name: Cache the Rust build + if: steps.changes.outputs.decision != 'skip' + uses: ./.github/actions/cache-cargo-build + + - name: Install dependencies + if: steps.changes.outputs.decision != 'skip' + run: | + .github/scripts/uv_sync_with_retries.sh --frozen --group ci --group proxy-dev --extra google --extra proxy --extra semantic-router + + - name: Cache Prisma binaries + if: steps.changes.outputs.decision != 'skip' + uses: ./.github/actions/cache-prisma-binaries + + - name: Generate Prisma client + if: steps.changes.outputs.decision != 'skip' + run: | + uv run --no-sync prisma generate --schema litellm/proxy/schema.prisma + + - name: Run documentation validation tests + if: steps.changes.outputs.decision != 'skip' + run: | + uv run --no-sync python ./tests/documentation_tests/test_env_keys.py + uv run --no-sync python ./tests/documentation_tests/test_router_settings.py + uv run --no-sync python ./tests/documentation_tests/test_api_docs.py + uv run --no-sync python ./tests/documentation_tests/test_circular_imports.py + coverage: + name: coverage + needs: unit + if: always() + runs-on: ubuntu-latest + timeout-minutes: 10 + permissions: + contents: read + id-token: write + pull-requests: write + + steps: + - name: Checkout code + uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + + - name: Download coverage report + uses: actions/download-artifact@95815c38cf2ff2164869cbab79da8d1f422bc89e # v4.2.1 + with: + pattern: coverage-*-${{ github.run_id }}-* + path: coverage-reports + merge-multiple: false + + - name: Upload to Codecov + id: codecov-upload + continue-on-error: true + uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5 + with: + use_oidc: true + directory: coverage-reports + root_dir: ${{ github.workspace }} + flags: unit + fail_ci_if_error: false + + - name: Upload to Codecov (retry) + if: steps.codecov-upload.outcome == 'failure' + continue-on-error: true + uses: codecov/codecov-action@0fb7174895f61a3b6b78fc075e0cd60383518dac # v5.5.5 + with: + use_oidc: true + directory: coverage-reports + root_dir: ${{ github.workspace }} + flags: unit + fail_ci_if_error: false unit-passed: name: unit passed - needs: [rust-bridge, unit, lens-python-310, assert-shard-coverage, proxy-db] + permissions: {} + needs: [rust-bridge, assert-shard-coverage, unit, ui-unit, docs] if: always() runs-on: ubuntu-latest timeout-minutes: 2 - permissions: {} steps: - name: Require every unit job to succeed env: diff --git a/Makefile b/Makefile index b8614ca4eec..da6cebedede 100644 --- a/Makefile +++ b/Makefile @@ -137,7 +137,7 @@ format-check: install-dev lint-fetch-base: @$(RESOLVE_BASE) -# Mirror test-linting.yml's lint job environment: the proxy-dev group plus a generated +# Mirror test-linting.yml's python job environment: the proxy-dev group plus a generated # Prisma client, so `basedpyright tests/e2e` resolves the same modules CI does. The # basedpyright gate itself no longer measures here (scripts/type_check_gate.py provisions its # own .venv-typecheck). --inexact tops up the venv instead of pruning the proxy extras @@ -236,7 +236,7 @@ check-circular-imports: $(LINT_DEP_INSTALL) check-import-safety: $(LINT_DEP_INSTALL) @$(UV_RUN) python -c "from litellm import *; print('[from litellm import *] OK! no issues!');" || (echo '🚨 import failed, this means you introduced unprotected imports! 🚨'; exit 1) -# Combined linting, isomorphic to test-linting.yml's lint job so a local pass means a +# Combined linting, isomorphic to test-linting.yml's python job so a local pass means a # green CI lint: it installs the same env (proxy-dev + generated Prisma client) and then # runs the diff-scoped ruff format check, whole-tree ruff check, the strict-rule / # type-discipline / basedpyright gates as a delta vs the base, then the circular-import @@ -260,8 +260,7 @@ lint-dev: lint-format-changed check-circular-imports check-import-safety # is staged (warning about changed files left unstaged); with nothing staged it falls # back to the working tree's diff against the merge base with the base branch, so a # fresh merge commit or an unstaged working tree still gets checked. Mirrors -# test-linting.yml (Python), test-litellm-ui-build.yml's frontend-lint (dashboard), and -# check-ui-api-types.yml (API-type drift), skipping any whose files aren't in scope. +# test-linting.yml (Python, UI, and API types), skipping any whose files aren't in scope. # Not auto-installed as a git hook so it never slows an unrelated human commit. check: @$(GATE_SLOT_LOCK) $(MAKE) check-inner diff --git a/codecov.yaml b/codecov.yaml index 4d93c18f3ac..ae72f6d5286 100644 --- a/codecov.yaml +++ b/codecov.yaml @@ -6,7 +6,7 @@ codecov: ignore: - "litellm-rust/**" -# Uploads are flagged per workflow/shard (GHA) or "circleci". carryforward makes +# Uploads are flagged per workflow tier (GHA) or "circleci". carryforward makes # a re-upload of a flag replace its prior session instead of accumulating a # conflicting one, and lets a commit reuse a flag from its parent when that flag # was not re-uploaded. Required because the same commit can receive the @@ -27,6 +27,76 @@ flag_management: carryforward: false - name: circleci carryforward: false + - name: core-utils + carryforward: false + - name: enterprise-routing + carryforward: false + - name: integrations + carryforward: false + - name: llm-vertex-ai + carryforward: false + - name: llm-other-providers + carryforward: false + - name: llm-openai-meta + carryforward: false + - name: misc + carryforward: false + - name: misc-dirs + carryforward: false + - name: proxy-auth + carryforward: false + - name: proxy-hooks-client + carryforward: false + - name: proxy-endpoints + carryforward: false + - name: proxy-feature-endpoints + carryforward: false + - name: proxy-server + carryforward: false + - name: mcp-elicitation + carryforward: false + - name: proxy-infra + carryforward: false + - name: proxy-infra-root + carryforward: false + - name: caching-local + carryforward: false + - name: proxy-extras + carryforward: false + - name: enterprise-package + carryforward: false + - name: enterprise-managed-files + carryforward: false + - name: responses-caching-types + carryforward: false + - name: lens-python-310 + carryforward: false + - name: proxy-db-key-generation + carryforward: false + - name: proxy-db-auth-checks + carryforward: false + - name: proxy-db-jwt-and-keys + carryforward: false + - name: proxy-db-proxy-utils + carryforward: false + - name: proxy-db-proxy-server-core + carryforward: false + - name: proxy-db-proxy-runtime + carryforward: false + - name: proxy-db-mcp-oauth + carryforward: false + - name: proxy-db-custom-logging + carryforward: false + - name: proxy-db-logging-misc + carryforward: false + - name: proxy-db-db-and-spend + carryforward: false + - name: proxy-db-guardrails-hooks + carryforward: false + - name: proxy-db-budgets + carryforward: false + - name: proxy-db-endpoints-and-responses + carryforward: false component_management: individual_components: diff --git a/litellm/proxy/_lazy_openapi_snapshot.py b/litellm/proxy/_lazy_openapi_snapshot.py index 92578aa43b9..00a05307a89 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.py +++ b/litellm/proxy/_lazy_openapi_snapshot.py @@ -3,7 +3,7 @@ Per-feature OpenAPI snapshot for lazy-loaded routers. The committed JSON is generated by `python -m litellm.proxy._lazy_openapi_snapshot` and consumed at runtime so /openapi.json can show full route info for unloaded -features without importing them. check-ui-api-types.yml (mirrored locally by +features without importing them. test-linting.yml (mirrored locally by `make check`) regenerates this file and fails when the committed copy differs, then rebuilds schema.d.ts from app.openapi() with the snapshot injected. After changing any lazily loaded route or this generator, rerun the module and commit diff --git a/scripts/pre_commit_lint.sh b/scripts/pre_commit_lint.sh index 1c793705dc6..edb69f24e29 100755 --- a/scripts/pre_commit_lint.sh +++ b/scripts/pre_commit_lint.sh @@ -9,16 +9,16 @@ # - nothing staged -> scope is the working tree's diff against the merge base # with origin's current default branch, untracked files included # The per-area checks: -# - litellm/ Python -> `make lint` (test-linting.yml's lint job) +# - litellm/ Python -> `make lint` (test-linting.yml's python job) # - tests/e2e and tests/e2e_harness Python # -> `make lint-e2e-basedpyright` (test-linting.yml's e2e type-check step) -# + raw HTTP client ban (test-code-quality.yml's check_e2e_no_raw_requests) +# + raw HTTP client ban (test-linting.yml's check_e2e_no_raw_requests) # - tests/ Python, ruff-tests.toml, scripts/check_test_quality.py, # scripts/test_quality_gate.py # -> ruff over ruff-tests.toml + `make lint-test-quality` (test-linting.yml's # test-tree ruff and test-quality gate steps) -# - dashboard -> prettier + eslint + lint budgets (test-litellm-ui-build.yml's frontend-lint) -# - proxy/types -> regenerate the lazy OpenAPI snapshot and dashboard API types, fail on drift (check-ui-api-types.yml) +# - dashboard -> prettier + eslint + lint budgets (test-linting.yml's ui job) +# - proxy/types -> regenerate the lazy OpenAPI snapshot and dashboard API types, fail on drift (test-linting.yml's ui-api-types job) # # Each block is skipped when no matching files are in scope, so unrelated commits # stay fast. This is intentionally not auto-installed as a git hook (see @@ -110,11 +110,11 @@ e2e_py_files=$(scope_match "$e2e_py_pattern") test_tree_files=$(scope_match "$test_tree_pattern") # ruff format (and CI's format step) skip enterprise; the rest of make lint covers it. fmt_files=$(printf '%s\n' "$litellm_py_files" | grep -v '^litellm/enterprise/' | existing_files) -# check-ui-api-types.yml triggers on any file under litellm/proxy or litellm/types +# test-linting.yml triggers on any file under litellm/proxy or litellm/types # (Prisma schema and configs included, not just Python) plus the generator and its # lockfiles, so match that whole trigger set rather than a Python subset. spec_files=$(scope_match "$spec_pattern") -# CI's frontend-lint runs prettier over a wider extension set than eslint; keep that +# CI's ui job runs prettier over a wider extension set than eslint; keep that # split so this flags exactly what the job would. ui_prettier_changed=$(scope_match "$ui_prettier_pattern") ui_eslint_changed=$(scope_match "$ui_eslint_pattern") @@ -176,7 +176,7 @@ EOF if [ ${#eslint_rel[@]} -gt 0 ]; then npx eslint --no-warn-ignored --pass-on-unpruned-suppressions "${eslint_rel[@]}" || rc=1 fi - # Whole-folder lint budgets, exactly as the frontend-lint job runs them: the + # Whole-folder lint budgets, exactly as the ui job runs them: the # counts are not diff-scoped, so a local pass here means the budget step will # pass in CI too. report=$(mktemp) @@ -259,7 +259,7 @@ genapi_checks() { local status=0 echo "check: checking the lazy OpenAPI snapshot and dashboard API types are in sync (npm run gen:api)" # gen-api-types.mjs imports litellm.proxy.proxy_server, which needs the proxy deps - # and an up-to-date Prisma client; check-ui-api-types.yml installs those and runs + # and an up-to-date Prisma client; test-linting.yml installs those and runs # prisma generate before gen:api, so mirror that here or a stale client can mask # drift that CI will still flag. if [ ! -d ui/litellm-dashboard/node_modules ]; then diff --git a/tests/code_coverage_tests/check_workflow_startup_safety.py b/tests/code_coverage_tests/check_workflow_startup_safety.py index cf150daef4c..935ab7b3bef 100644 --- a/tests/code_coverage_tests/check_workflow_startup_safety.py +++ b/tests/code_coverage_tests/check_workflow_startup_safety.py @@ -12,7 +12,7 @@ because CI cannot enforce them on itself. flagged: ``-`` appears in hyphenated input names like ``inputs.timeout-minutes`` and ``/`` inside ref strings, so neither can be told apart from arithmetic by inspection alone. -2. Callers of the reusable unit-test workflow keep the job timeout at or above +2. Unit matrix rows keep the job timeout at or above the test budget plus the setup ceilings plus the runner overhead below. Otherwise the job deadline preempts pytest inside its own advertised budget, which is the failure the split timeouts exist to prevent, and it shows up as @@ -24,7 +24,6 @@ because CI cannot enforce them on itself. import re import sys from collections.abc import Iterator, Mapping, Sequence -from dataclasses import dataclass from pathlib import Path from typing import Final @@ -33,8 +32,7 @@ from pydantic import BaseModel, Field, ValidationError REPO_ROOT: Final = Path(__file__).resolve().parent.parent.parent WORKFLOWS_DIR: Final = REPO_ROOT / ".github" / "workflows" -BASE_WORKFLOW: Final = "./.github/workflows/_test-unit-base.yml" -BASE_WORKFLOW_PATH: Final = WORKFLOWS_DIR / "_test-unit-base.yml" +UNIT_WORKFLOW_PATH: Final = WORKFLOWS_DIR / "test-unit.yml" # Runner time the job clock charges but no step owns: job init, the gaps between # steps, and post-job cleanup. Without it a job capped at exactly test + setup @@ -44,24 +42,26 @@ JOB_OVERHEAD_MINUTES: Final = 5 EXPRESSION: Final = re.compile(r"\$\{\{(?P.*?)\}\}", re.DOTALL) QUOTED: Final = re.compile(r"'[^']*'") ARITHMETIC: Final = re.compile(r"[+*]") -MATRIX_REF: Final = re.compile(r"^\$\{\{\s*matrix\.(?P[\w-]+)\s*\}\}$") +JOB_DEADLINE: Final = "${{ matrix.job-timeout-minutes }}" +TEST_DEADLINE: Final = "${{ matrix.timeout-minutes }}" class WorkflowStartupError(Exception): pass -class ReusableCall(BaseModel): - uses: str | None = None - with_: Mapping[str, object] = Field(default_factory=dict, alias="with") +class WorkflowJob(BaseModel): strategy: Mapping[str, object] = Field(default_factory=dict) steps: tuple[Mapping[str, object], ...] = () + timeout_minutes: object = Field(default=None, alias="timeout-minutes") - model_config = {"populate_by_name": True} + model_config = {"populate_by_name": True, "frozen": True} class WorkflowFile(BaseModel): - jobs: Mapping[str, ReusableCall] = Field(default_factory=dict) + jobs: Mapping[str, WorkflowJob] = Field(default_factory=dict) + + model_config = {"frozen": True} def parse_workflow(text: str) -> WorkflowFile | str: @@ -79,10 +79,10 @@ def arithmetic_expressions(text: str) -> Iterator[str]: yield body.strip() -def setup_ceiling_minutes(base_text: str) -> int: - """Sum the per-step timeouts on everything the base workflow runs before pytest.""" - base: Final = yaml.safe_load(base_text) - steps: Final = base["jobs"]["run"]["steps"] +def setup_ceiling_minutes(unit_text: str) -> int: + """Sum the per-step timeouts on everything the unit job runs before pytest.""" + workflow: Final = yaml.safe_load(unit_text) + steps: Final = workflow["jobs"]["unit"]["steps"] return sum( s["timeout-minutes"] for s in steps @@ -90,100 +90,38 @@ def setup_ceiling_minutes(base_text: str) -> int: ) -def base_default(base_text: str, name: str) -> int: - base: Final = yaml.safe_load(base_text) - return base[True]["workflow_call"]["inputs"][name]["default"] - - -@dataclass(frozen=True, slots=True) -class Column: - """A budget the caller reads from one column of its own matrix.""" - - name: str - - -def budget_source(job: ReusableCall, key: str, fallback: int) -> int | Column | str: - """A caller passes a literal, or `${{ matrix.x }}` naming a column of its matrix. - - Anything else comes back as the reason it could not be read, since a budget - nothing can resolve has to be reported rather than passed over. - """ - value: Final = job.with_.get(key) - if value is None: - return fallback - if isinstance(value, int): - return value - - matrix_ref: Final = MATRIX_REF.match(str(value)) - if not matrix_ref: - return f"passes `{key}: {value}`, which is neither a number nor a `matrix` reference." - return Column(matrix_ref.group("key")) - - -def matrix_rows(job: ReusableCall) -> Sequence[Mapping[str, object]]: - matrix: Final = job.strategy.get("matrix", {}) - entries: Final = matrix.get("include", ()) if isinstance(matrix, dict) else () - return tuple(e for e in entries if isinstance(e, dict)) - - -def budget_pairs(job: ReusableCall, test_source: int | Column, job_source: int | Column) -> Iterator[tuple[int, int]]: - """Pair each shard's test budget with the job budget of that same shard. - - Matrix-sourced budgets resolve per `include` row, so two matrix columns are - read off the same row rather than cross-producted across rows. - """ - if isinstance(test_source, int) and isinstance(job_source, int): - yield test_source, job_source +def timeout_contract_errors(rel: Path, workflow: WorkflowFile, ceiling: int) -> Iterator[str]: + job: Final = workflow.jobs.get("unit") + if job is None: return - - for row in matrix_rows(job): - test_budget = row.get(test_source.name) if isinstance(test_source, Column) else test_source - job_budget = row.get(job_source.name) if isinstance(job_source, Column) else job_source - if isinstance(test_budget, int) and isinstance(job_budget, int): - yield test_budget, job_budget - - -def unresolved_message(where: str, job: ReusableCall, sources: Sequence[int | Column]) -> str: - """Why no shard yielded a pair of budgets to compare. - - Naming only the columns that resolve nowhere keeps the message honest: a - column every row supplies is not what left the pair unchecked. - """ - rows: Final = matrix_rows(job) - missing: Final = tuple( - f"`matrix.{s.name}`" - for s in sources - if isinstance(s, Column) and not any(isinstance(row.get(s.name), int) for row in rows) - ) - if missing: - return ( - f"{where} reads a budget from {', '.join(missing)}, which no `include` row supplies " - "as a number, so the pair would go unchecked." - ) - return ( - f"{where} reads both budgets from its matrix, but no single `include` row supplies both " - "as numbers, so the pair would go unchecked." - ) - - -def job_errors(rel: Path, job_name: str, job: ReusableCall, ceiling: int, base_text: str) -> Iterator[str]: - where: Final = f"{rel}: job `{job_name}`" - test_source: Final = budget_source(job, "timeout-minutes", base_default(base_text, "timeout-minutes")) - job_source: Final = budget_source(job, "job-timeout-minutes", base_default(base_text, "job-timeout-minutes")) - sources: Final = (test_source, job_source) - - unreadable: Final = tuple(f"{where} {reason}" for reason in sources if isinstance(reason, str)) - if unreadable: - yield from unreadable + if job.timeout_minutes != JOB_DEADLINE: + yield f"{rel}: job `unit` sets timeout-minutes to `{job.timeout_minutes}`, not `{JOB_DEADLINE}`" + test_deadlines: Final = tuple(s.get("timeout-minutes") for s in job.steps if s.get("name") == "Run tests") + if test_deadlines != (TEST_DEADLINE,): + yield f"{rel}: job `unit` step `Run tests` must set timeout-minutes to `{TEST_DEADLINE}`, found {test_deadlines}" + matrix_value: Final = job.strategy.get("matrix", {}) + if not isinstance(matrix_value, Mapping): + yield f"{rel}: job `unit` has no readable matrix" return - - pairs: Final = tuple(budget_pairs(job, test_source, job_source)) - if not pairs: - yield unresolved_message(where, job, sources) + entries_value: Final = matrix_value.get("include", ()) + if not isinstance(entries_value, Sequence) or isinstance(entries_value, str) or not entries_value: + yield f"{rel}: job `unit` has no readable matrix include rows" return - - for test_budget, job_budget in pairs: - required = test_budget + ceiling + JOB_OVERHEAD_MINUTES + for entry in entries_value: + if not isinstance(entry, Mapping): + yield f"{rel}: job `unit` has an unreadable matrix row" + continue + shard: Final = entry.get("shard", "") + test_budget: Final = entry.get("timeout-minutes") + job_budget: Final = entry.get("job-timeout-minutes") + where: Final = f"{rel}: job `unit`, shard `{shard}`" + if not isinstance(test_budget, int) or isinstance(test_budget, bool): + yield f"{where} has no integer timeout-minutes" + continue + if not isinstance(job_budget, int) or isinstance(job_budget, bool): + yield f"{where} has no integer job-timeout-minutes" + continue + required: Final = test_budget + ceiling + JOB_OVERHEAD_MINUTES if job_budget < required: yield ( f"{where} gives pytest {test_budget}m but caps the job at " @@ -193,13 +131,7 @@ def job_errors(rel: Path, job_name: str, job: ReusableCall, ceiling: int, base_t ) -def timeout_contract_errors(rel: Path, workflow: WorkflowFile, ceiling: int, base_text: str) -> Iterator[str]: - for job_name, job in workflow.jobs.items(): - if job.uses == BASE_WORKFLOW: - yield from job_errors(rel, job_name, job, ceiling, base_text) - - -def workflow_errors(rel: Path, text: str, ceiling: int, base_text: str) -> Iterator[str]: +def workflow_errors(rel: Path, text: str, ceiling: int) -> Iterator[str]: for expression in arithmetic_expressions(text): yield ( f"{rel}: `${{{{ {expression} }}}}` uses arithmetic, which GitHub expressions do not " @@ -211,22 +143,20 @@ def workflow_errors(rel: Path, text: str, ceiling: int, base_text: str) -> Itera yield f"{rel}: {workflow}" return - yield from timeout_contract_errors(rel, workflow, ceiling, base_text) + yield from timeout_contract_errors(rel, workflow, ceiling) def main() -> None: - base_text: Final = BASE_WORKFLOW_PATH.read_text() - ceiling: Final = setup_ceiling_minutes(base_text) + unit_text: Final = UNIT_WORKFLOW_PATH.read_text() + ceiling: Final = setup_ceiling_minutes(unit_text) errors: Final = tuple( error for path in sorted(WORKFLOWS_DIR.glob("*.y*ml")) - for error in workflow_errors(path.relative_to(REPO_ROOT), path.read_text(), ceiling, base_text) + for error in workflow_errors(path.relative_to(REPO_ROOT), path.read_text(), ceiling) ) if errors: - raise WorkflowStartupError( - "Workflow startup invariants violated:\n - " + "\n - ".join(errors) - ) + raise WorkflowStartupError("Workflow startup invariants violated:\n - " + "\n - ".join(errors)) print(f"Workflow startup invariants hold (setup ceiling {ceiling}m)") diff --git a/tests/code_coverage_tests/test_e2e_metadata.py b/tests/code_coverage_tests/test_e2e_metadata.py index 57d04262969..22fe84f72a5 100644 --- a/tests/code_coverage_tests/test_e2e_metadata.py +++ b/tests/code_coverage_tests/test_e2e_metadata.py @@ -2,7 +2,7 @@ Harness logic, so it lives here rather than under tests/e2e, which holds only tests that drive a live proxy. The harness modules are imported off -``PYTHONPATH=tests/e2e``, the way the Code Quality workflow's +``PYTHONPATH=tests/e2e``, the way the lint workflow's code-quality job's test_e2e_metadata step runs this file. Call order, the failing test's last step, the per-test reset and the JUnit attach are pinned end to end in test_e2e_junit_report.py. diff --git a/tests/code_coverage_tests/test_merge_smoke.py b/tests/code_coverage_tests/test_merge_smoke.py index e3b96e222b9..550817910c0 100644 --- a/tests/code_coverage_tests/test_merge_smoke.py +++ b/tests/code_coverage_tests/test_merge_smoke.py @@ -11,37 +11,6 @@ import pytest HARNESS: Final = Path(__file__).parents[2] / ".github" / "scripts" / "run_merge_smoke.py" -CASE_IDS: Final = ( - "CHAT-JSON", - "CHAT-TEXT-STREAM", - "CHAT-TOOL-STREAM", - "MODEL-ALLOW", - "MODEL-DENY", - "COST-EXPLICIT", - "COST-ZERO", - "LOG-CONTENT-ON", - "LOG-CONTENT-OFF", - "CALLBACK-SUCCESS", - "CALLBACK-FAILURE", -) - - -def _write_fake_tests(root: Path, body: str) -> Path: - package: Final = root / "fake_tests" - package.mkdir() - (package / "test_cases.py").write_text(body) - return package - - -def _manifest(root: Path, **overrides: str) -> Path: - cases: Final[dict[str, str]] = { - case_id: f"fake_tests/test_cases.py::test_{case_id.lower().replace('-', '_')}" for case_id in CASE_IDS - } - cases.update(overrides) - path: Final = root / "manifest.json" - path.write_text(json.dumps({"cases": cases})) - return path - def _run(root: Path, *argv: str) -> subprocess.CompletedProcess[str]: return subprocess.run( @@ -53,132 +22,6 @@ def _run(root: Path, *argv: str) -> subprocess.CompletedProcess[str]: ) -def _passing_tests() -> str: - return "\n".join(f"def test_{case_id.lower().replace('-', '_')}():\n assert True" for case_id in CASE_IDS) - - -def test_all_eleven_cases_pass(tmp_path: Path) -> None: - _write_fake_tests(tmp_path, _passing_tests()) - manifest: Final = _manifest(tmp_path) - - proc: Final = _run(tmp_path, "pytest", "--manifest", str(manifest), "--rootdir", str(tmp_path)) - - assert proc.returncode == 0, proc.stderr - assert proc.stdout.count("PASS") >= 11 - for case_id in CASE_IDS: - assert f"{case_id} PASS" in proc.stdout - - -def test_missing_test_node_id_fails(tmp_path: Path) -> None: - _write_fake_tests(tmp_path, _passing_tests()) - manifest: Final = _manifest(tmp_path, **{"COST-ZERO": "fake_tests/test_cases.py::test_does_not_exist"}) - - proc: Final = _run(tmp_path, "pytest", "--manifest", str(manifest), "--rootdir", str(tmp_path)) - - assert proc.returncode != 0 - assert "COST-ZERO" in proc.stderr or "test_does_not_exist" in proc.stderr - - -def test_skipped_case_fails(tmp_path: Path) -> None: - _write_fake_tests( - tmp_path, - _passing_tests().replace( - "def test_cost_zero():\n assert True", - "def test_cost_zero():\n import pytest\n pytest.skip('nope')", - ), - ) - manifest: Final = _manifest(tmp_path) - - proc: Final = _run(tmp_path, "pytest", "--manifest", str(manifest), "--rootdir", str(tmp_path)) - - assert proc.returncode != 0 - assert "COST-ZERO" in proc.stderr - - -def test_xfail_case_fails(tmp_path: Path) -> None: - _write_fake_tests( - tmp_path, - "import pytest\n" - + _passing_tests().replace( - "def test_cost_zero():\n assert True", - "@pytest.mark.xfail\ndef test_cost_zero():\n assert False", - ), - ) - manifest: Final = _manifest(tmp_path) - - proc: Final = _run(tmp_path, "pytest", "--manifest", str(manifest), "--rootdir", str(tmp_path)) - - assert proc.returncode != 0 - assert "COST-ZERO" in proc.stderr - - -def test_xpass_case_fails(tmp_path: Path) -> None: - _write_fake_tests( - tmp_path, - "import pytest\n" - + _passing_tests().replace( - "def test_cost_zero():\n assert True", - "@pytest.mark.xfail\ndef test_cost_zero():\n assert True", - ), - ) - manifest: Final = _manifest(tmp_path) - - proc: Final = _run(tmp_path, "pytest", "--manifest", str(manifest), "--rootdir", str(tmp_path)) - - assert proc.returncode != 0 - assert "COST-ZERO" in proc.stderr - - -def test_duplicate_manifest_key_fails(tmp_path: Path) -> None: - manifest: Final = tmp_path / "manifest.json" - manifest.write_text('{"cases": {"CHAT-JSON": "a::b", "CHAT-JSON": "a::c"}}') - - proc: Final = _run(tmp_path, "pytest", "--manifest", str(manifest)) - - assert proc.returncode != 0 - assert "CHAT-JSON" in proc.stderr - - -def test_missing_case_id_fails(tmp_path: Path) -> None: - manifest: Final = tmp_path / "manifest.json" - cases: Final = {c: f"t::{c}" for c in CASE_IDS[:-1]} - manifest.write_text(json.dumps({"cases": cases})) - - proc: Final = _run(tmp_path, "pytest", "--manifest", str(manifest)) - - assert proc.returncode != 0 - assert "case ids" in proc.stderr - - -def test_extra_case_id_fails(tmp_path: Path) -> None: - manifest: Final = tmp_path / "manifest.json" - cases: Final = {c: f"t::{c}" for c in CASE_IDS} - cases["EXTRA"] = "t::x" - manifest.write_text(json.dumps({"cases": cases})) - - proc: Final = _run(tmp_path, "pytest", "--manifest", str(manifest)) - - assert proc.returncode != 0 - assert "case ids" in proc.stderr - - -def test_teardown_error_fails(tmp_path: Path) -> None: - body: Final = ( - "import pytest\n\n@pytest.fixture\ndef boom():\n yield\n raise RuntimeError('teardown-boom')\n\n" - + _passing_tests().replace( - "def test_cost_zero():\n assert True", - "def test_cost_zero(boom):\n assert True", - ) - ) - _write_fake_tests(tmp_path, body) - manifest: Final = _manifest(tmp_path) - - proc: Final = _run(tmp_path, "pytest", "--manifest", str(manifest), "--rootdir", str(tmp_path)) - - assert proc.returncode != 0 - assert "COST-ZERO" in proc.stderr - - def _fake_litellm(tmp_path: Path, script: str) -> Path: path: Final = tmp_path / "fake-litellm" path.write_text(f"#!{sys.executable}\n" + textwrap.dedent(script)) diff --git a/tests/code_coverage_tests/test_unit_passed_gate.py b/tests/code_coverage_tests/test_unit_passed_gate.py index 7b511538d40..5a0d04681d4 100644 --- a/tests/code_coverage_tests/test_unit_passed_gate.py +++ b/tests/code_coverage_tests/test_unit_passed_gate.py @@ -7,24 +7,30 @@ from typing import Final import pytest import yaml -_UNIT_WORKFLOW: Final = Path(__file__).resolve().parents[2] / ".github" / "workflows" / "test-unit.yml" -_GATE_JOB: Final = "unit-passed" +_WORKFLOWS_DIR: Final = Path(__file__).resolve().parents[2] / ".github" / "workflows" +_TIERS: Final = ( + ("test-linting.yml", "lint-passed", frozenset()), + ("test-unit.yml", "unit-passed", frozenset({"coverage"})), + ("test-merge-smoke.yml", "smoke-passed", frozenset()), +) +_Tier = tuple[str, str, frozenset[str]] -def _jobs() -> dict[str, dict[str, object]]: - return yaml.safe_load(_UNIT_WORKFLOW.read_text())["jobs"] +def _jobs(tier: _Tier) -> dict[str, dict[str, object]]: + workflow: Final = yaml.safe_load((_WORKFLOWS_DIR / tier[0]).read_text()) + return workflow["jobs"] -def _gate_script() -> str: - steps: Final = _jobs()[_GATE_JOB]["steps"] +def _gate_script(tier: _Tier) -> str: + steps: Final = _jobs(tier)[tier[1]]["steps"] assert isinstance(steps, list) and len(steps) == 1 return steps[0]["run"] -def _run_gate(results: dict[str, str]) -> subprocess.CompletedProcess[str]: +def _run_gate(tier: _Tier, results: dict[str, str]) -> subprocess.CompletedProcess[str]: needs: Final = {job: {"result": result, "outputs": {}} for job, result in results.items()} return subprocess.run( - ("bash", "--noprofile", "--norc", "-eo", "pipefail", "-c", _gate_script()), + ("bash", "--noprofile", "--norc", "-eo", "pipefail", "-c", _gate_script(tier)), env={**os.environ, "NEEDS": json.dumps(needs)}, capture_output=True, text=True, @@ -33,36 +39,40 @@ def _run_gate(results: dict[str, str]) -> subprocess.CompletedProcess[str]: ) -def _needed_jobs() -> tuple[str, ...]: - needs: Final = _jobs()[_GATE_JOB]["needs"] +def _needed_jobs(tier: _Tier) -> tuple[str, ...]: + needs: Final = _jobs(tier)[tier[1]]["needs"] assert isinstance(needs, list) return tuple(needs) -def test_the_gate_passes_when_every_needed_job_succeeded() -> None: - jobs: Final = _needed_jobs() +@pytest.mark.parametrize("tier", _TIERS, ids=("lint", "unit", "smoke")) +def test_the_gate_passes_when_every_needed_job_succeeded(tier: _Tier) -> None: + jobs: Final = _needed_jobs(tier) assert jobs - result: Final = _run_gate(dict.fromkeys(jobs, "success")) + result: Final = _run_gate(tier, {job: "success" for job in jobs}) assert result.returncode == 0, result.stdout + result.stderr assert all(f"{job}: success" in result.stdout for job in jobs), result.stdout @pytest.mark.parametrize("outcome", ("failure", "cancelled", "skipped")) -def test_the_gate_fails_when_any_needed_job_did_not_succeed(outcome: str) -> None: - jobs: Final = _needed_jobs() +@pytest.mark.parametrize("tier", _TIERS, ids=("lint", "unit", "smoke")) +def test_the_gate_fails_when_any_needed_job_did_not_succeed(tier: _Tier, outcome: str) -> None: + jobs: Final = _needed_jobs(tier) assert len(jobs) > 1 - result: Final = _run_gate({**dict.fromkeys(jobs, "success"), jobs[-1]: outcome}) + result: Final = _run_gate(tier, {**{job: "success" for job in jobs}, jobs[-1]: outcome}) assert result.returncode != 0 assert f"{jobs[-1]}: {outcome}" in result.stdout, result.stdout -def test_the_gate_waits_for_every_other_job_and_reports_even_when_they_fail() -> None: - jobs: Final = _jobs() - gate: Final = jobs[_GATE_JOB] +@pytest.mark.parametrize("tier", _TIERS, ids=("lint", "unit", "smoke")) +def test_the_gate_waits_for_every_other_job_and_reports_even_when_they_fail(tier: _Tier) -> None: + jobs: Final = _jobs(tier) + gate: Final = jobs[tier[1]] - assert set(_needed_jobs()) == set(jobs) - {_GATE_JOB} + assert set(_needed_jobs(tier)) == set(jobs) - {tier[1]} - set(tier[2]) assert gate["if"] == "always()" + assert gate["permissions"] == {} diff --git a/tests/e2e_harness/AGENTS.md b/tests/e2e_harness/AGENTS.md index 92882850b67..212e46d754c 100644 --- a/tests/e2e_harness/AGENTS.md +++ b/tests/e2e_harness/AGENTS.md @@ -14,4 +14,4 @@ LITELLM_MASTER_KEY=sk-harness uv run pytest tests/e2e_harness Rules: no `e2e` marker and no `@meta`, since nothing here drives the proxy; `@pytest.mark.covers` only where the test proves the collector or the JUnit properties read it; inputs via arguments or env vars (setting an env var through pytest's `monkeypatch` fixture is fine, patching a function, class or module is not); and the same typing bar as the suite, `make lint-e2e-basedpyright` covers this folder and allows zero errors. The raw HTTP client ban (`tests/code_coverage_tests/check_e2e_no_raw_requests.py`) applies here too -CI: the `lint` job in `.github/workflows/test-linting.yml` runs this folder whenever anything under `tests/e2e/` (except `ui/`) or `tests/e2e_harness/` changes, with the `claude` CLI installed. The CircleCI `provider_replay_harness` job also runs the provider-edge and fixture tests at the root of this folder next to `tests/code_coverage_tests/test_provider_replay_harness.py`, which imports helpers from `test_provider_edge.py` +CI: the `python` job in `.github/workflows/test-linting.yml` runs this folder whenever anything under `tests/e2e/` (except `ui/`) or `tests/e2e_harness/` changes, with the `claude` CLI installed. The CircleCI `provider_replay_harness` job also runs the provider-edge and fixture tests at the root of this folder next to `tests/code_coverage_tests/test_provider_replay_harness.py`, which imports helpers from `test_provider_edge.py` diff --git a/tests/unit/test_circleci_path_filter.py b/tests/unit/test_circleci_path_filter.py index 873258d26a2..2a4084bd8c2 100644 --- a/tests/unit/test_circleci_path_filter.py +++ b/tests/unit/test_circleci_path_filter.py @@ -43,7 +43,7 @@ def classify(category: str, changed: list[str]) -> str: DOCS = ["README.md", "docs/my_website/index.mdx", "litellm/anywhere.md"] CLIENT = ["ui/litellm-dashboard/src/App.tsx"] BACKEND = ["litellm/main.py"] -CI = [".github/workflows/test-litellm-ui-unit.yml"] +CI = [".github/workflows/test-unit.yml"] @pytest.mark.parametrize( diff --git a/tests/unit/test_detect_changes.py b/tests/unit/test_detect_changes.py index d8feb2371a3..44a4bfdf448 100644 --- a/tests/unit/test_detect_changes.py +++ b/tests/unit/test_detect_changes.py @@ -231,5 +231,5 @@ def test_ui_category_fails_open_when_the_api_fails(tmp_path: Path) -> None: def test_ui_category_runs_when_the_ui_workflows_themselves_change(tmp_path: Path) -> None: """Without this the dashboard jobs would skip on the pull request that edits them, shipping a workflow change nothing ever exercised.""" - decision, _ = _run(tmp_path, files=[".github/workflows/test-litellm-ui-unit.yml"], category="ui") + decision, _ = _run(tmp_path, files=[".github/workflows/test-unit.yml"], category="ui") assert decision == "decision=run" diff --git a/tests/unit/test_select_ui_test_scope.py b/tests/unit/test_select_ui_test_scope.py index ebfb7701e6a..4605aab956c 100644 --- a/tests/unit/test_select_ui_test_scope.py +++ b/tests/unit/test_select_ui_test_scope.py @@ -1,6 +1,6 @@ """Regression tests for the UI unit-test scope decision. -`.github/workflows/test-litellm-ui-unit.yml` narrows the dashboard's Vitest run to +`.github/workflows/test-unit.yml` narrows the dashboard's Vitest run to `vitest related ` so a pull request only pays for the tests it can affect. `related` resolves a file to the tests that import it, so a file no test imports resolves to nothing, and with `--passWithNoTests` the job then goes green @@ -26,7 +26,7 @@ import yaml REPO_ROOT = Path(__file__).resolve().parents[2] SCOPE_SCRIPT = REPO_ROOT / ".github" / "scripts" / "select_ui_test_scope.sh" -WORKFLOW = REPO_ROOT / ".github" / "workflows" / "test-litellm-ui-unit.yml" +WORKFLOW = REPO_ROOT / ".github" / "workflows" / "test-unit.yml" STEP_NAME = "Run UI unit tests (Vitest)" FULL_SUITE_ARGV = ["run", "test", "--", "--run", "--pool", "forks", "--maxWorkers=14"] @@ -75,7 +75,7 @@ def test_an_empty_change_set_fails_open_to_the_full_suite() -> None: def _step_script() -> str: workflow = yaml.safe_load(WORKFLOW.read_text()) - steps = workflow["jobs"]["ui-unit-tests"]["steps"] + steps = workflow["jobs"]["ui-unit"]["steps"] script = next(step["run"] for step in steps if step.get("name") == STEP_NAME) resolved = script.replace("${{ github.repository }}", "BerriAI/litellm") assert "${{" not in resolved, "the step uses an Actions expression this harness does not resolve" diff --git a/tests/unit/test_unit_shard_missing_paths.py b/tests/unit/test_unit_shard_missing_paths.py index 4fa9c5bd3c1..725551e19d3 100644 --- a/tests/unit/test_unit_shard_missing_paths.py +++ b/tests/unit/test_unit_shard_missing_paths.py @@ -9,7 +9,7 @@ import pytest import yaml _REPO_ROOT: Final = Path(__file__).resolve().parents[2] -_BASE_WORKFLOW: Final = _REPO_ROOT / ".github" / "workflows" / "_test-unit-base.yml" +_UNIT_WORKFLOW: Final = _REPO_ROOT / ".github" / "workflows" / "test-unit.yml" _SHARD_ENV: Final = MappingProxyType( {"MAX_FAILURES": "10", "RERUNS": "0", "DIST": "loadscope", "TEST_TIMEOUT_SECONDS": "60", "COVERAGE_CORE": "sysmon"} ) @@ -19,8 +19,8 @@ _FAILING_TEST: Final = "def test_fails():\n assert False\n" def _run_tests_script() -> str: - workflow: Final = yaml.safe_load(_BASE_WORKFLOW.read_text()) - return next(step["run"] for step in workflow["jobs"]["run"]["steps"] if step.get("name") == "Run tests") + workflow: Final = yaml.safe_load(_UNIT_WORKFLOW.read_text()) + return next(step["run"] for step in workflow["jobs"]["unit"]["steps"] if step.get("name") == "Run tests") def _run_shard(tmp_path: Path, test_path: str, workers: str) -> subprocess.CompletedProcess[str]: diff --git a/tests/unit/test_unit_shard_per_test_timeout.py b/tests/unit/test_unit_shard_per_test_timeout.py index 8096124ca8c..830b46b723d 100644 --- a/tests/unit/test_unit_shard_per_test_timeout.py +++ b/tests/unit/test_unit_shard_per_test_timeout.py @@ -10,7 +10,7 @@ import pytest import yaml _REPO_ROOT: Final = Path(__file__).resolve().parents[2] -_BASE_WORKFLOW: Final = _REPO_ROOT / ".github" / "workflows" / "_test-unit-base.yml" +_UNIT_WORKFLOW: Final = _REPO_ROOT / ".github" / "workflows" / "test-unit.yml" _SHARD_ENV: Final = MappingProxyType({"WORKERS": "2", "RERUNS": "2", "DIST": "loadscope", "TEST_TIMEOUT_SECONDS": "1"}) _HANG_GUARD_FLAGS: Final = frozenset(("-n", "--dist", "--reruns", "--reruns-delay", "--timeout", "--rerun-except")) _HUNG_TEST_MODULE: Final = """ @@ -39,8 +39,8 @@ def test_passes(): def _run_tests_script() -> str: - workflow: Final = yaml.safe_load(_BASE_WORKFLOW.read_text()) - return next(step["run"] for step in workflow["jobs"]["run"]["steps"] if step.get("name") == "Run tests") + workflow: Final = yaml.safe_load(_UNIT_WORKFLOW.read_text()) + return next(step["run"] for step in workflow["jobs"]["unit"]["steps"] if step.get("name") == "Run tests") def _pytest_invocations(script: str) -> tuple[tuple[str, ...], ...]: diff --git a/ui/litellm-dashboard/AGENTS.md b/ui/litellm-dashboard/AGENTS.md index c890c33ce9b..bfba65b2860 100644 --- a/ui/litellm-dashboard/AGENTS.md +++ b/ui/litellm-dashboard/AGENTS.md @@ -4,7 +4,7 @@ Never put LiteLLM tokens or API keys in `localStorage`. `localStorage` survives When you fix lint violations that are grandfathered in `eslint-suppressions.json`, run `eslint . --prune-suppressions` and commit the updated baseline so the gate ratchets down instead of leaving a stale suppression -`src/lib/http/schema.d.ts` is generated from the proxy's OpenAPI spec; never hand-edit it. After changing a backend route or response model that the dashboard consumes, run `npm run gen:api` and commit the result (CI `Check UI API Types Sync` enforces this) +`src/lib/http/schema.d.ts` is generated from the proxy's OpenAPI spec; never hand-edit it. After changing a backend route or response model that the dashboard consumes, run `npm run gen:api` and commit the result (CI `lint / ui-api-types` enforces this) Tests come in three tiers, named by the standard definitions. `Foo.test.tsx` is a unit test: one module, collaborators replaced by doubles, no multi-component tree, and it should run in milliseconds. `Foo.integration.test.tsx` renders a real component tree with real children and only stubs the network boundary; it costs seconds per case, so it earns its place by proving wiring that a unit test cannot reach. Browser-level tests live in `tests/e2e/ui/` as Playwright specs against a live proxy