mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
chore(e2e): port the compat-matrix cron publisher to tests/e2e/claude_code
Ports the daily cron VM publisher from the unmerged tests/claude_code checkout so the automation runs the e2e suite from litellm_internal_staging. Adds find_regressions to matrix_builder for the green to red auto-merge gate, pins the cron venv to Python 3.12, and ships the systemd units, env template, and runbook alongside
This commit is contained in:
parent
3726bceb53
commit
3fa633370d
10 changed files with 1377 additions and 0 deletions
0
tests/e2e/claude_code/_builder_unit_tests/__init__.py
Normal file
0
tests/e2e/claude_code/_builder_unit_tests/__init__.py
Normal file
148
tests/e2e/claude_code/_builder_unit_tests/test_matrix_builder.py
Normal file
148
tests/e2e/claude_code/_builder_unit_tests/test_matrix_builder.py
Normal file
|
|
@ -0,0 +1,148 @@
|
|||
"""Unit tests for `find_regressions`, the green→red detector that gates
|
||||
auto-merge on the daily compat-matrix docs PR (see `cron_vm/`).
|
||||
|
||||
Markerless harness tests: they exercise publisher plumbing, not a product
|
||||
feature, so they run without a proxy and carry no `e2e` marker.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Mapping, Union
|
||||
|
||||
from claude_code.matrix_builder import find_regressions
|
||||
|
||||
_CellSpec = Union[str, Mapping[str, str]]
|
||||
|
||||
|
||||
def _matrix(
|
||||
cells: Mapping[tuple[str, str], _CellSpec],
|
||||
*,
|
||||
names: Mapping[str, str] | None = None,
|
||||
) -> dict[str, object]:
|
||||
"""Build a minimal matrix dict from a {(feature_id, provider): status}
|
||||
or {(feature_id, provider): cell_dict} mapping."""
|
||||
names = names or {}
|
||||
features: dict[str, dict[str, dict[str, str]]] = {}
|
||||
for (feature_id, provider), value in cells.items():
|
||||
cell = {"status": value} if isinstance(value, str) else dict(value)
|
||||
features.setdefault(feature_id, {})[provider] = cell
|
||||
return {
|
||||
"features": [
|
||||
{
|
||||
"id": feature_id,
|
||||
"name": names.get(feature_id, feature_id.upper()),
|
||||
"providers": providers,
|
||||
}
|
||||
for feature_id, providers in features.items()
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def test_find_regressions_flags_pass_to_fail() -> None:
|
||||
old = _matrix({("vision", "anthropic"): "pass"})
|
||||
new = _matrix(
|
||||
{("vision", "anthropic"): {"status": "fail", "error": "credit balance too low"}}
|
||||
)
|
||||
regressions = find_regressions(old, new)
|
||||
assert len(regressions) == 1
|
||||
r = regressions[0]
|
||||
assert r["feature_id"] == "vision"
|
||||
assert r["provider"] == "anthropic"
|
||||
assert r["old_status"] == "pass"
|
||||
assert r["new_status"] == "fail"
|
||||
assert r["error"] == "credit balance too low"
|
||||
|
||||
|
||||
def test_find_regressions_ignores_red_to_red() -> None:
|
||||
"""An already-failing cell that stays failing is NOT a regression — a
|
||||
provider that's independently broken (e.g. out of credits) must not
|
||||
block the daily auto-merge forever."""
|
||||
old = _matrix({("vision", "anthropic"): "fail"})
|
||||
new = _matrix({("vision", "anthropic"): "fail"})
|
||||
assert find_regressions(old, new) == []
|
||||
|
||||
|
||||
def test_find_regressions_ignores_improvements_and_steady_green() -> None:
|
||||
old = _matrix(
|
||||
{
|
||||
("vision", "anthropic"): "fail", # red -> green
|
||||
("tool_use", "azure"): "pass", # green -> green
|
||||
}
|
||||
)
|
||||
new = _matrix(
|
||||
{
|
||||
("vision", "anthropic"): "pass",
|
||||
("tool_use", "azure"): "pass",
|
||||
}
|
||||
)
|
||||
assert find_regressions(old, new) == []
|
||||
|
||||
|
||||
def test_find_regressions_ignores_green_to_grey() -> None:
|
||||
"""green→not_tested / green→not_applicable are degradations but not
|
||||
*red* regressions; we deliberately don't block on them."""
|
||||
old = _matrix(
|
||||
{
|
||||
("vision", "azure"): "pass",
|
||||
("tool_use", "azure"): "pass",
|
||||
}
|
||||
)
|
||||
new = _matrix(
|
||||
{
|
||||
("vision", "azure"): "not_tested",
|
||||
("tool_use", "azure"): {"status": "not_applicable", "reason": "skip"},
|
||||
}
|
||||
)
|
||||
assert find_regressions(old, new) == []
|
||||
|
||||
|
||||
def test_find_regressions_ignores_new_cells_without_baseline() -> None:
|
||||
"""A cell only present in the new matrix (new feature/provider) has no
|
||||
baseline, so a fail there can't be a regression."""
|
||||
old = _matrix({("vision", "anthropic"): "pass"})
|
||||
new = _matrix(
|
||||
{
|
||||
("vision", "anthropic"): "pass",
|
||||
("brand_new_feature", "anthropic"): "fail",
|
||||
}
|
||||
)
|
||||
assert find_regressions(old, new) == []
|
||||
|
||||
|
||||
def test_find_regressions_matches_by_id_not_name() -> None:
|
||||
"""Renaming a feature's display name must not hide a regression: cells
|
||||
are matched on the stable id."""
|
||||
old = _matrix({("thinking", "anthropic"): "pass"}, names={"thinking": "Old Name"})
|
||||
new = _matrix(
|
||||
{("thinking", "anthropic"): "fail"}, names={"thinking": "Totally New Name"}
|
||||
)
|
||||
regressions = find_regressions(old, new)
|
||||
assert len(regressions) == 1
|
||||
assert regressions[0]["feature_id"] == "thinking"
|
||||
assert regressions[0]["feature_name"] == "Totally New Name"
|
||||
|
||||
|
||||
def test_find_regressions_reports_multiple_sorted() -> None:
|
||||
old = _matrix(
|
||||
{
|
||||
("vision", "anthropic"): "pass",
|
||||
("tool_use", "anthropic"): "pass",
|
||||
("vision", "azure"): "pass",
|
||||
}
|
||||
)
|
||||
new = _matrix(
|
||||
{
|
||||
("vision", "anthropic"): "fail",
|
||||
("tool_use", "anthropic"): "fail",
|
||||
("vision", "azure"): "pass", # stays green
|
||||
}
|
||||
)
|
||||
regressions = find_regressions(old, new)
|
||||
keys = [(r["feature_id"], r["provider"]) for r in regressions]
|
||||
assert keys == [("tool_use", "anthropic"), ("vision", "anthropic")]
|
||||
|
||||
|
||||
def test_find_regressions_empty_old_matrix_is_safe() -> None:
|
||||
"""No baseline at all (first publish) yields no regressions."""
|
||||
new = _matrix({("vision", "anthropic"): "fail"})
|
||||
assert find_regressions({}, new) == []
|
||||
187
tests/e2e/claude_code/cron_vm/README.md
Normal file
187
tests/e2e/claude_code/cron_vm/README.md
Normal file
|
|
@ -0,0 +1,187 @@
|
|||
# Cron VM setup for the Claude Code compatibility-matrix populator
|
||||
|
||||
The populator runs daily on a dedicated GCP VM
|
||||
(`litellm-compatibility-matrix-populator`) rather than as a GitHub
|
||||
Action. Trade-offs:
|
||||
|
||||
- ✅ Real VM means we can `gh auth login` against an account that's
|
||||
already a collaborator on `BerriAI/litellm-docs`, instead of
|
||||
provisioning a GitHub App with `pull-requests: write`.
|
||||
- ✅ Persistent state (a single `~/litellm-cron-worktree/` and its `.venv`)
|
||||
is reused across runs, so each daily run does a fast `git checkout` +
|
||||
incremental `uv sync` rather than a fresh clone + cold sync.
|
||||
- ✅ No Docker dependency — the proxy runs directly via `uv run litellm`.
|
||||
- ⚠️ The VM has to actually be on. systemd's `Persistent=true` recovers
|
||||
from short outages, but a multi-day outage means the matrix goes
|
||||
stale until the VM is back.
|
||||
- ⚠️ Provider credentials live on the VM filesystem
|
||||
(`/etc/litellm-compat-matrix.env`) instead of GitHub secrets. Treat
|
||||
the VM as an environment with comparable blast radius to a CI runner.
|
||||
|
||||
This directory used to live at `tests/claude_code/cron_vm/` (paired with
|
||||
the standalone `tests/claude_code/` suite); it now runs the maintained
|
||||
`tests/e2e/claude_code/` suite instead. The pytest env interface changed
|
||||
accordingly: the runner exports `LITELLM_PROXY_URL` / `LITELLM_MASTER_KEY`
|
||||
(previously `LITELLM_PROXY_BASE_URL` / `LITELLM_PROXY_API_KEY`), the azure
|
||||
column reads `AZURE_AI_API_KEY` / `AZURE_AI_API_BASE` (previously
|
||||
`AZURE_FOUNDRY_*`), and the GPT columns need `OPENAI_API_KEY` and
|
||||
`AZURE_API_BASE` / `AZURE_API_KEY` — see `litellm-compat-matrix.env.example`.
|
||||
|
||||
## Layout
|
||||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `run_daily.sh` | The actual cron job. Resolves versions, updates the worktree, boots the proxy, runs pytest, builds the JSON, opens (or updates) a docs PR, sweeps stale compat-matrix PRs. |
|
||||
| `build_matrix.py` | Tiny Python CLI that wraps `claude_code.matrix_builder.build_from_paths`. Exists only because the bash script needs *some* way to render the per-cell aggregation, and the builder is already Python. |
|
||||
| `check_regressions.py` | Tiny Python CLI that wraps `claude_code.matrix_builder.find_regressions`. Diffs the freshly built matrix against the currently-published one and exits `3` if any cell flipped green→red, which gates auto-merge. |
|
||||
| `litellm-compat-matrix.service` | systemd oneshot that invokes `run_daily.sh`. |
|
||||
| `litellm-compat-matrix.timer` | `OnCalendar=*-*-* 06:00:00 UTC`, `Persistent=true`. |
|
||||
| `litellm-compat-matrix.env.example` | Template for `/etc/litellm-compat-matrix.env`. |
|
||||
|
||||
## What `run_daily.sh` does
|
||||
|
||||
1. **Resolves the latest LiteLLM final release tag** (newest bare
|
||||
`vX.Y.Z`, skipping `-rc.N`/`-dev.N` pre-releases) by paging the
|
||||
GitHub Releases API (`curl | jq`).
|
||||
2. **Reads the local Claude Code CLI version** via `claude --version`.
|
||||
The cron does not auto-upgrade the CLI — operators do that
|
||||
out-of-band by running `npm install -g @anthropic-ai/claude-code@latest`.
|
||||
3. **Updates the persistent worktree** at `~/litellm-cron-worktree/`:
|
||||
`git fetch --tags --force`, `git reset --hard`,
|
||||
`git clean -fdx -e .venv -e .uv-bin`, `git checkout --force <tag>`.
|
||||
The `.venv` is preserved across runs so `uv sync --frozen` is
|
||||
incremental. Then **shims the test suite**: `tests/e2e/` in the
|
||||
worktree is rebuilt from the dev checkout — the `claude_code/` suite
|
||||
plus the five shared transport helpers it imports (`proxy_client.py`,
|
||||
`e2e_http.py`, `models.py`, `e2e_config.py`, `transport.py`) — so the
|
||||
cron always runs *today's* tests against the latest stable proxy. The
|
||||
tag's own `tests/e2e/` tree (including the EKS-harness `conftest.py`,
|
||||
whose imports the stable venv doesn't install) is deliberately not
|
||||
used.
|
||||
4. **Boots the proxy** as a `setsid` background process on port `4100`
|
||||
(so it can't collide with a developer's `:4000`), then polls
|
||||
`/health/liveliness` until it's up.
|
||||
5. **Runs pytest** on `tests/e2e/claude_code/` with `LITELLM_PROXY_URL`
|
||||
pointed at the proxy and `COMPAT_RESULTS_PATH` set so the conftest
|
||||
hook writes the per-test results artifact. Test failures become
|
||||
`fail` cells in the JSON, not script errors.
|
||||
6. **Builds `compatibility-matrix.json`** by handing the artifact +
|
||||
manifest to `build_matrix.py`.
|
||||
7. **Opens or updates a docs PR**: `gh repo clone` of `litellm-docs`
|
||||
into a tempdir, deterministic head branch
|
||||
(`compat-matrix/<litellm-version>-<claude-code-version>-<UTC-date>`),
|
||||
`--force` push **directly to `BerriAI/litellm-docs`** (the
|
||||
`mateo-berri` token has write access, so this is a same-repo branch,
|
||||
not a fork), `gh pr create`. A re-run on the same day fast-forwards
|
||||
the existing branch and `gh pr create` no-ops ("a pull request for
|
||||
branch ... already exists" is treated as success). These PRs are no
|
||||
longer gated on a second human review.
|
||||
8. **Gates auto-merge on a regression check**: before enabling
|
||||
auto-merge, `check_regressions.py` diffs the new matrix against the
|
||||
one currently on `main`. Auto-merge (`gh pr merge --auto --squash`)
|
||||
is only enabled when **no cell flipped green→red** — i.e. every
|
||||
transition is red→green, green→green, or red→red. A pre-existing red
|
||||
cell (e.g. a provider that's out of API credits) is `red→red` and
|
||||
does **not** block; only a `pass`→`fail` flip does. When a regression
|
||||
is detected the PR is still opened/updated (with a warning banner
|
||||
naming the offending cells) but auto-merge is left **off** — and any
|
||||
auto-merge a prior same-day run enabled is explicitly disabled — so a
|
||||
human reviews before it lands on the public table. The check fails
|
||||
*closed*: if it errors, auto-merge is withheld.
|
||||
9. **Sweeps stale compat-matrix PRs**: once today's PR exists, every
|
||||
other open `compat-matrix/*` PR on the docs repo is closed (and its
|
||||
bot-owned branch deleted), so at most one compat-matrix PR is ever
|
||||
open — the newest.
|
||||
|
||||
## One-time VM setup
|
||||
|
||||
Run as `mateo` on the cron VM:
|
||||
|
||||
```bash
|
||||
# 1. Toolchain
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y git nodejs npm jq curl
|
||||
curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
sudo apt-get install -y gh # or follow https://cli.github.com/
|
||||
|
||||
# 2. Claude Code CLI (the cron does NOT auto-upgrade this; rerun this
|
||||
# line out-of-band when you want a fresh CLI to be tested)
|
||||
sudo npm install -g @anthropic-ai/claude-code@latest
|
||||
|
||||
# 3. Litellm checkout. Used by systemd's WorkingDirectory and as the
|
||||
# source of the .service / .timer files. The cron itself runs out
|
||||
# of the separate worktree at ~/litellm-cron-worktree/.
|
||||
mkdir -p ~/litellm
|
||||
git clone https://github.com/BerriAI/litellm.git ~/litellm/litellm
|
||||
git -C ~/litellm/litellm checkout litellm_internal_staging
|
||||
|
||||
# 4. gh auth — must be a collaborator on BerriAI/litellm-docs.
|
||||
gh auth login # follow prompts; pick HTTPS + token paste flow
|
||||
|
||||
# 5. Provider credentials.
|
||||
sudo cp ~/litellm/litellm/tests/e2e/claude_code/cron_vm/litellm-compat-matrix.env.example \
|
||||
/etc/litellm-compat-matrix.env
|
||||
sudoedit /etc/litellm-compat-matrix.env # fill in real values
|
||||
sudo chmod 0600 /etc/litellm-compat-matrix.env
|
||||
|
||||
# 6. systemd units.
|
||||
sudo cp ~/litellm/litellm/tests/e2e/claude_code/cron_vm/litellm-compat-matrix.service /etc/systemd/system/
|
||||
sudo cp ~/litellm/litellm/tests/e2e/claude_code/cron_vm/litellm-compat-matrix.timer /etc/systemd/system/
|
||||
sudo systemctl daemon-reload
|
||||
sudo systemctl enable --now litellm-compat-matrix.timer
|
||||
```
|
||||
|
||||
## Operating it
|
||||
|
||||
```bash
|
||||
# When does it run next?
|
||||
systemctl list-timers litellm-compat-matrix.timer
|
||||
|
||||
# Trigger a real run right now (PRs to litellm-docs).
|
||||
sudo systemctl start litellm-compat-matrix.service
|
||||
|
||||
# Trigger a run that does NOT open a PR (good for first-time validation).
|
||||
SKIP_PUBLISH=1 ~/litellm/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh
|
||||
|
||||
# Narrow to one cell while debugging.
|
||||
SKIP_PUBLISH=1 PYTEST_K='basic_messaging_non_streaming and anthropic' \
|
||||
~/litellm/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh
|
||||
|
||||
# Watch the most recent run.
|
||||
journalctl -u litellm-compat-matrix.service -f
|
||||
|
||||
# Read older runs.
|
||||
journalctl -u litellm-compat-matrix.service --since '2 days ago'
|
||||
|
||||
# Disable until further notice (e.g. while debugging).
|
||||
sudo systemctl disable --now litellm-compat-matrix.timer
|
||||
```
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **The venv is pinned to Python 3.12 (`CRON_PYTHON_VERSION`).** The
|
||||
e2e suite uses PEP 695 `type` aliases, which the VM's system Python
|
||||
(3.11) can't parse; `run_daily.sh` has uv fetch a managed CPython
|
||||
into `~/litellm-cron-worktree/.uv-python/` and syncs the venv against
|
||||
it. The first run after a version bump is a cold venv rebuild.
|
||||
- **The proxy port is `4100`, not `4000`.** This is so a developer SSH'd
|
||||
into the same VM with their own `:4000` proxy doesn't collide with a
|
||||
cron run. Override with `PROXY_PORT=...` in `/etc/litellm-compat-matrix.env`
|
||||
if you need to.
|
||||
- **`uv sync --frozen` requires the resolved tag to be tagged on
|
||||
GitHub.** If the latest stable release was made but not pushed as a
|
||||
git tag, the `git checkout` step fails. Push the tag, then rerun.
|
||||
- **`GITHUB_TOKEN` rotation is your problem.** The cron does not
|
||||
refresh the token; if `mateo-berri`'s PAT in
|
||||
`/etc/litellm-compat-matrix.env` expires, the run fails at the
|
||||
`git push`/`gh pr create` step with a 401 ("Bad credentials" /
|
||||
"Authentication failed"). Mint a fresh PAT and update the env file.
|
||||
The token needs write access to `BerriAI/litellm-docs` (classic
|
||||
`repo` scope, or fine-grained Contents:RW + Pull requests:RW).
|
||||
- **First run after upgrading the Claude Code CLI is the riskiest one.**
|
||||
If the new CLI changes its wire format the matrix run can produce
|
||||
systematic failures. Always run with `SKIP_PUBLISH=1` after a CLI
|
||||
upgrade before letting the next scheduled fire happen.
|
||||
- **Disk:** the worktree's `.venv` is ~1.3 GB and the `.git` directory
|
||||
is ~1 GB. Plan for at least 5 GB free on the VM, otherwise
|
||||
`uv sync` will fail mid-run and leave you with a half-installed venv.
|
||||
52
tests/e2e/claude_code/cron_vm/build_matrix.py
Normal file
52
tests/e2e/claude_code/cron_vm/build_matrix.py
Normal file
|
|
@ -0,0 +1,52 @@
|
|||
"""Tiny CLI wrapper around `claude_code.matrix_builder.build_from_paths`.
|
||||
|
||||
Exists only so `run_daily.sh` can hand the version metadata + paths into
|
||||
the matrix builder without re-implementing it in bash. All real logic
|
||||
lives in `matrix_builder.py`.
|
||||
|
||||
The suite imports its own modules with `tests/e2e/` on sys.path (that is
|
||||
how pytest resolves them: `tests/e2e/` has no `__init__.py`, while
|
||||
`claude_code/` does), so this script bootstraps the same root — two
|
||||
levels up from this file — before importing.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import datetime
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[2]))
|
||||
|
||||
from claude_code.matrix_builder import (
|
||||
build_from_paths,
|
||||
) # noqa: E402 # needs the sys.path bootstrap above
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--manifest", type=Path, required=True)
|
||||
parser.add_argument("--results", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
parser.add_argument("--litellm-version", required=True)
|
||||
parser.add_argument("--claude-code-version", required=True)
|
||||
args = parser.parse_args()
|
||||
|
||||
generated_at = datetime.datetime.now(datetime.timezone.utc).strftime(
|
||||
"%Y-%m-%dT%H:%M:%SZ"
|
||||
)
|
||||
build_from_paths(
|
||||
manifest_path=args.manifest,
|
||||
results_path=args.results,
|
||||
litellm_version=args.litellm_version,
|
||||
claude_code_version=args.claude_code_version,
|
||||
generated_at=generated_at,
|
||||
output_path=args.output,
|
||||
)
|
||||
print(f"wrote {args.output}") # noqa: T201 # CLI output read by run_daily.sh
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
80
tests/e2e/claude_code/cron_vm/check_regressions.py
Normal file
80
tests/e2e/claude_code/cron_vm/check_regressions.py
Normal file
|
|
@ -0,0 +1,80 @@
|
|||
"""CLI: detect green→red regressions between the published matrix and a
|
||||
freshly built one, so `run_daily.sh` can decide whether to enable
|
||||
auto-merge on the daily docs PR.
|
||||
|
||||
All real logic lives in `claude_code.matrix_builder.find_regressions`;
|
||||
this file only does the I/O and maps the result onto an exit code the
|
||||
bash caller can branch on.
|
||||
|
||||
Exit codes (the bash gate depends on these exact values):
|
||||
|
||||
0 no green→red regressions -> safe to auto-merge
|
||||
3 one or more green→red regressions -> do NOT auto-merge (human review)
|
||||
2 argparse/usage error (argparse default)
|
||||
|
||||
The `--old` file is allowed to be missing: on the first-ever publish there
|
||||
is no baseline to regress against, so we exit 0.
|
||||
|
||||
Imports resolve with `tests/e2e/` on sys.path, mirroring build_matrix.py.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[2]))
|
||||
|
||||
from claude_code.matrix_builder import (
|
||||
find_regressions,
|
||||
) # noqa: E402 # needs the sys.path bootstrap above
|
||||
|
||||
REGRESSION_EXIT = 3
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument(
|
||||
"--old",
|
||||
type=Path,
|
||||
required=True,
|
||||
help="currently published matrix JSON (may be absent on first publish)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--new",
|
||||
type=Path,
|
||||
required=True,
|
||||
help="freshly built matrix JSON",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if not args.old.exists():
|
||||
print( # noqa: T201 # CLI output read by run_daily.sh
|
||||
"no published matrix to compare against "
|
||||
"(first publish); treating as no regressions"
|
||||
)
|
||||
return 0
|
||||
|
||||
old_matrix = json.loads(args.old.read_text())
|
||||
new_matrix = json.loads(args.new.read_text())
|
||||
|
||||
regressions = find_regressions(old_matrix, new_matrix)
|
||||
if not regressions:
|
||||
print("no green->red regressions detected") # noqa: T201 # CLI output
|
||||
return 0
|
||||
|
||||
print( # noqa: T201 # CLI output read by run_daily.sh
|
||||
f"detected {len(regressions)} green->red regression(s):"
|
||||
)
|
||||
for r in regressions:
|
||||
line = f" - {r['feature_name']} [{r['provider']}]: pass -> fail"
|
||||
if r["error"]:
|
||||
line += f" ({r['error'][:160]})"
|
||||
print(line) # noqa: T201 # CLI output read by run_daily.sh
|
||||
return REGRESSION_EXIT
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
|
|
@ -0,0 +1,59 @@
|
|||
# Environment file consumed by `litellm-compat-matrix.service`.
|
||||
#
|
||||
# Install at `/etc/litellm-compat-matrix.env` and chmod 0600.
|
||||
# `EnvironmentFile=-` in the unit means the service is allowed to start
|
||||
# even if this file is missing, but the populator will fail at the
|
||||
# first provider request without these credentials.
|
||||
|
||||
# Anthropic
|
||||
ANTHROPIC_API_KEY=
|
||||
|
||||
# Bedrock (invoke + converse columns; also bedrock_mantle when enabled).
|
||||
# Use Anthropic's Bedrock API-key passthrough (long-lived bearer token).
|
||||
# No AWS_ACCESS_KEY_ID/AWS_SECRET_ACCESS_KEY required for the matrix --
|
||||
# both the LiteLLM invoke and converse routes pick up
|
||||
# AWS_BEARER_TOKEN_BEDROCK when present.
|
||||
AWS_BEARER_TOKEN_BEDROCK=
|
||||
AWS_REGION_NAME=us-east-1
|
||||
|
||||
# Vertex AI (vertex_ai + vertex_ai_gpt columns).
|
||||
# On the GCP VM, the default service-account ADC from the metadata server
|
||||
# is used -- no JSON key file is needed. If you ever need to run outside
|
||||
# GCP, also export GOOGLE_APPLICATION_CREDENTIALS=/path/to/sa.json.
|
||||
VERTEXAI_PROJECT=
|
||||
VERTEXAI_LOCATION=global
|
||||
|
||||
# Azure AI Foundry (azure column — Claude models on Foundry)
|
||||
AZURE_AI_API_KEY=
|
||||
AZURE_AI_API_BASE=
|
||||
|
||||
# OpenAI (openai GPT column)
|
||||
OPENAI_API_KEY=
|
||||
|
||||
# Azure OpenAI (azure_openai GPT column)
|
||||
AZURE_API_BASE=
|
||||
AZURE_API_KEY=
|
||||
|
||||
# REQUIRED for publishing: PAT for the `mateo-berri` user, who has write
|
||||
# access on BerriAI/litellm-docs. Used to (a) resolve the latest stable
|
||||
# release, (b) push the daily compat-matrix branch directly to
|
||||
# BerriAI/litellm-docs, (c) open the same-repo PR, and (d) enable
|
||||
# squash auto-merge on it. Scopes: classic `repo` + `workflow`, or
|
||||
# fine-grained on BerriAI/litellm-docs with Contents:RW + Pull
|
||||
# requests:RW + Workflows:RW.
|
||||
# Skip by setting SKIP_PUBLISH=1 (publishes nothing; only writes the
|
||||
# matrix JSON locally).
|
||||
GITHUB_TOKEN=
|
||||
|
||||
# Optional: the bedrock_mantle column is opt-in because the AWS account
|
||||
# needs the Mantle (OpenAI-on-Bedrock) models enabled. Without this the
|
||||
# mantle cells are skipped and recorded as not_tested rather than fail.
|
||||
# COMPAT_MANTLE_CELLS=1
|
||||
|
||||
# Optional overrides; defaults are sensible for the cron VM.
|
||||
# PROXY_PORT=4100
|
||||
# LITELLM_WORKTREE=/home/mateo/litellm-cron-worktree
|
||||
# DOCS_REPO=BerriAI/litellm-docs
|
||||
# DOCS_BRANCH=main
|
||||
# DOCS_TARGET_PATH=src/data/compatibility-matrix.json
|
||||
# AUTO_MERGE_METHOD=squash
|
||||
101
tests/e2e/claude_code/cron_vm/litellm-compat-matrix.service
Normal file
101
tests/e2e/claude_code/cron_vm/litellm-compat-matrix.service
Normal file
|
|
@ -0,0 +1,101 @@
|
|||
# systemd service for the Claude Code compatibility-matrix populator.
|
||||
#
|
||||
# Triggered by `litellm-compat-matrix.timer`; not started directly. The
|
||||
# unit is a `Type=oneshot` so the timer's `OnCalendar=` semantics
|
||||
# describe "run once per day" cleanly — there's no long-lived daemon to
|
||||
# supervise; each invocation runs the populator end-to-end and exits.
|
||||
#
|
||||
# Install
|
||||
# -------
|
||||
#
|
||||
# sudo cp tests/e2e/claude_code/cron_vm/litellm-compat-matrix.service /etc/systemd/system/
|
||||
# sudo cp tests/e2e/claude_code/cron_vm/litellm-compat-matrix.timer /etc/systemd/system/
|
||||
# sudo systemctl daemon-reload
|
||||
# sudo systemctl enable --now litellm-compat-matrix.timer
|
||||
#
|
||||
# Paths are hard-coded to /home/mateo rather than using systemd's %h
|
||||
# specifier. Why: in *system* units (this one), %h is expanded at
|
||||
# parse time against the *manager's* home -- which is /root for PID 1
|
||||
# -- and *not* against the User= directive. That mismatch makes
|
||||
# ReadWritePaths point at /root/.cache (which doesn't exist), causing
|
||||
# the namespace setup to fail with status=226/NAMESPACE before the
|
||||
# script ever runs. The runtime user (`User=mateo`) must:
|
||||
#
|
||||
# * have a checkout of `BerriAI/litellm` at `~/litellm/litellm` so the
|
||||
# publisher module is importable;
|
||||
# * have a uv venv at `~/litellm/litellm/.venv` (created by
|
||||
# `uv sync --frozen` inside that checkout once);
|
||||
# * have `gh` already authenticated against an account with
|
||||
# `pull-requests: write` on `BerriAI/litellm-docs`;
|
||||
# * have provider credentials exported in `/etc/litellm-compat-matrix.env`
|
||||
# (see `litellm-compat-matrix.env.example` in this directory).
|
||||
|
||||
[Unit]
|
||||
Description=Claude Code compatibility-matrix populator (oneshot)
|
||||
Documentation=file:///home/mateo/litellm/litellm/tests/e2e/claude_code/cron_vm/README.md
|
||||
Wants=network-online.target
|
||||
After=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
User=mateo
|
||||
Group=mateo
|
||||
|
||||
# Provider credentials + any gh/PROXY_PORT overrides live here. Format
|
||||
# is the standard `KEY=value` one line per env var.
|
||||
EnvironmentFile=-/etc/litellm-compat-matrix.env
|
||||
|
||||
# systemd starts with a minimal PATH (~/usr/local/bin:/usr/bin:/bin).
|
||||
# `uv` and `claude` are installed under the runtime user's `~/.local/bin`
|
||||
# so we have to prepend it explicitly; otherwise run_daily.sh fails at
|
||||
# the up-front command-presence check.
|
||||
Environment=PATH=/home/mateo/.local/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
|
||||
|
||||
# `HOME` is auto-set to /home/mateo when User=mateo is honored, but be
|
||||
# explicit so anything that reads $HOME (e.g. uv's cache lookup, the
|
||||
# claude CLI's per-session dir) sees the right value even if a future
|
||||
# refactor flips DynamicUser= or PrivateUsers= on.
|
||||
Environment=HOME=/home/mateo
|
||||
|
||||
WorkingDirectory=/home/mateo/litellm/litellm
|
||||
|
||||
ExecStart=/home/mateo/litellm/litellm/tests/e2e/claude_code/cron_vm/run_daily.sh
|
||||
|
||||
# 90 minutes is generous: cold runs do `git clone` + `uv sync` of a new
|
||||
# tag's lockfile, which can take a couple of minutes on a 2-vCPU VM,
|
||||
# plus the full feature x provider grid of pytest cells hitting several
|
||||
# cloud providers.
|
||||
TimeoutStartSec=90min
|
||||
|
||||
# A failed run shouldn't restart automatically — the next timer fire is
|
||||
# the right retry. Reruns of the same day's matrix are idempotent.
|
||||
Restart=no
|
||||
|
||||
# Security hardening: the populator only reads the litellm checkout and
|
||||
# the env-file; everything else it writes lives in either the worktree
|
||||
# (managed) or `/tmp` (cleaned up by tempfile).
|
||||
#
|
||||
# ReadWritePaths whitelist:
|
||||
# * litellm-cron-worktree - the long-lived stable-tag checkout +
|
||||
# its `.venv` (`uv sync` rewrites every
|
||||
# run) + `.uv-bin` (pinned `uv` binary
|
||||
# cache).
|
||||
# * .cache - uv's wheel cache (~/.cache/uv) so we
|
||||
# don't redownload pinned deps each run.
|
||||
# * .claude - `claude` CLI's per-session state under
|
||||
# `~/.claude/projects/<sha>/`; created
|
||||
# on every `claude --print` invocation.
|
||||
# * .config/gh - `gh` CLI host config; technically not
|
||||
# needed when we pass GH_TOKEN inline,
|
||||
# but cheap to whitelist and prevents
|
||||
# future regressions if a code path
|
||||
# ever falls back to the host config.
|
||||
# * /tmp - mktemp -d workdir + proxy logs.
|
||||
NoNewPrivileges=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=read-only
|
||||
ReadWritePaths=/home/mateo/litellm-cron-worktree /home/mateo/.cache /home/mateo/.claude /home/mateo/.config/gh /tmp
|
||||
PrivateTmp=true
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
25
tests/e2e/claude_code/cron_vm/litellm-compat-matrix.timer
Normal file
25
tests/e2e/claude_code/cron_vm/litellm-compat-matrix.timer
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
# Daily timer for the compatibility-matrix populator.
|
||||
#
|
||||
# 06:00 UTC matches the original GitHub Actions cron schedule; chosen so
|
||||
# operators in US/EU timezones see fresh PRs at the start of their work
|
||||
# day.
|
||||
#
|
||||
# `Persistent=true` causes a missed run (VM was off / suspended) to
|
||||
# fire the next time the timer is started, which is the property we
|
||||
# want for a once-a-day job: the matrix should refresh as soon as the
|
||||
# VM is reachable again, not wait another 24h.
|
||||
#
|
||||
# `RandomizedDelaySec=10min` smears load if multiple matrix-style
|
||||
# pipelines are ever colocated on the same VM in the future.
|
||||
|
||||
[Unit]
|
||||
Description=Run the Claude Code compatibility-matrix populator daily
|
||||
|
||||
[Timer]
|
||||
OnCalendar=*-*-* 06:00:00 UTC
|
||||
Persistent=true
|
||||
RandomizedDelaySec=10min
|
||||
Unit=litellm-compat-matrix.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
645
tests/e2e/claude_code/cron_vm/run_daily.sh
Executable file
645
tests/e2e/claude_code/cron_vm/run_daily.sh
Executable file
|
|
@ -0,0 +1,645 @@
|
|||
#!/usr/bin/env bash
|
||||
# Daily Claude Code compatibility-matrix populator.
|
||||
#
|
||||
# Runs from the GCP VM `litellm-compatibility-matrix-populator` via the
|
||||
# systemd timer in this directory. The flow is:
|
||||
#
|
||||
# 1. Resolve the latest LiteLLM final release tag from the GitHub
|
||||
# Releases API.
|
||||
# 2. Update a long-lived worktree at $WORKTREE to that tag and `uv sync` it.
|
||||
# 3. Boot the proxy as a background subprocess on $PROXY_PORT (default
|
||||
# 4100; a separate port from the human-tended :4000 proxy).
|
||||
# 4. Run `pytest tests/e2e/claude_code/` against the proxy. Test
|
||||
# failures become `fail` cells in the JSON, not script errors.
|
||||
# 5. Hand the per-test results artifact + manifest to a small Python
|
||||
# CLI (`build_matrix.py`) that wraps the existing
|
||||
# `matrix_builder.build_from_paths` to produce the published
|
||||
# compatibility-matrix.json.
|
||||
# 6. `gh repo clone` litellm-docs, write the JSON to a deterministic
|
||||
# branch (`compat-matrix/<litellm>-<claude>-<UTC-date>`), commit,
|
||||
# push the branch straight to BerriAI/litellm-docs (mateo-berri has
|
||||
# write access), `gh pr create`, then — *only if no cell regressed
|
||||
# green→red versus the currently-published matrix* — enable squash
|
||||
# auto-merge so the PR merges itself once required checks pass. A
|
||||
# green→red regression leaves auto-merge off for human review; an
|
||||
# already-red cell (red→red) does not block.
|
||||
# 7. Sweep stale compat-matrix PRs: once today's PR exists, close any
|
||||
# other open `compat-matrix/*` PR (and delete its bot-owned branch)
|
||||
# so at most ONE compat-matrix PR is ever open — the newest. A
|
||||
# gate-withheld PR that nobody triages is superseded by the next
|
||||
# day's run rather than accumulating in the queue.
|
||||
#
|
||||
# Same-day reruns land on the same branch so they update the existing PR
|
||||
# rather than spawning a new one. If the JSON is byte-identical to the
|
||||
# docs branch, we skip the push entirely.
|
||||
#
|
||||
# Required commands on $PATH: git, uv, gh, jq, curl, claude, npm.
|
||||
# Required state: a litellm checkout at $LITELLM_REPO (this file lives in
|
||||
# it), $WORKTREE is created on first run, gh is already authenticated.
|
||||
#
|
||||
# Override any default by setting the matching env var; see the systemd
|
||||
# unit for the production wiring.
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
LITELLM_REPO="${LITELLM_REPO:-${HOME}/litellm/litellm}"
|
||||
WORKTREE="${LITELLM_WORKTREE:-${HOME}/litellm-cron-worktree}"
|
||||
PROXY_PORT="${PROXY_PORT:-4100}"
|
||||
PROXY_API_KEY="${PROXY_API_KEY:-sk-cron-matrix}"
|
||||
DOCS_REPO="${DOCS_REPO:-BerriAI/litellm-docs}"
|
||||
DOCS_BRANCH="${DOCS_BRANCH:-main}"
|
||||
DOCS_TARGET_PATH="${DOCS_TARGET_PATH:-src/data/compatibility-matrix.json}"
|
||||
SKIP_PUBLISH="${SKIP_PUBLISH:-0}"
|
||||
PYTEST_K="${PYTEST_K:-}"
|
||||
# The e2e suite uses PEP 695 `type` aliases, so the venv needs Python
|
||||
# >= 3.12 (also what repo CI runs) even when the VM's system python is
|
||||
# older. uv fetches a managed CPython of this version on first use --
|
||||
# checksum-verified against the manifest baked into the pinned uv
|
||||
# binary -- and installs it under ${WORKTREE}/.uv-python (see
|
||||
# UV_PYTHON_INSTALL_DIR below) so it lives inside the one tree the
|
||||
# systemd sandbox lets us write to.
|
||||
CRON_PYTHON_VERSION="${CRON_PYTHON_VERSION:-3.12}"
|
||||
# Merge method for auto-merge. BerriAI/litellm-docs only allows squash
|
||||
# merges (merge-commit and rebase are disabled at the repo level), so
|
||||
# `squash` is the only valid value here unless that changes upstream.
|
||||
AUTO_MERGE_METHOD="${AUTO_MERGE_METHOD:-squash}"
|
||||
|
||||
POPULATOR_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
WORKDIR="$(mktemp -d -t litellm-compat-matrix.XXXXXX)"
|
||||
PROXY_PID_FILE="${WORKDIR}/proxy.pid"
|
||||
|
||||
# Cleanup is intentionally aggressive: it can run on normal exit, on a
|
||||
# signal received by the script, or after a partial failure where the
|
||||
# proxy is up but ${PROXY_PID_FILE} is stale. We try four things in
|
||||
# order and stop as soon as the proxy port is free:
|
||||
#
|
||||
# 1. SIGTERM the pid recorded in proxy.pid.
|
||||
# 2. SIGKILL anything from `pgrep -f "litellm.*--port ${PROXY_PORT}"`
|
||||
# that survived. This catches the common case where the recorded
|
||||
# pid was the sh wrapper, not the long-lived python child.
|
||||
# 3. ss -K on the port (kernel kills sockets but not processes;
|
||||
# mostly useful for catching lingering CLOSE_WAITs).
|
||||
# 4. wipe ${WORKDIR}.
|
||||
cleanup() {
|
||||
local rc=$?
|
||||
set +e
|
||||
local proxy_pid
|
||||
if [[ -f "${PROXY_PID_FILE}" ]]; then
|
||||
proxy_pid="$(cat "${PROXY_PID_FILE}")"
|
||||
if [[ -n "${proxy_pid}" ]]; then
|
||||
kill -TERM "-${proxy_pid}" 2>/dev/null || kill -TERM "${proxy_pid}" 2>/dev/null || true
|
||||
for _ in 1 2 3 4 5; do
|
||||
kill -0 "${proxy_pid}" 2>/dev/null || break
|
||||
sleep 1
|
||||
done
|
||||
fi
|
||||
fi
|
||||
# Belt-and-braces: any python or uv talking to ${PROXY_PORT} that
|
||||
# survived the SIGTERM gets SIGKILL'd by name.
|
||||
pgrep -f "litellm.*--port[ =]?${PROXY_PORT}([^0-9]|$)" 2>/dev/null \
|
||||
| xargs -r kill -KILL 2>/dev/null || true
|
||||
pgrep -f "${WORKTREE}/.uv-bin/uv.*run litellm" 2>/dev/null \
|
||||
| xargs -r kill -KILL 2>/dev/null || true
|
||||
rm -rf "${WORKDIR}"
|
||||
exit "${rc}"
|
||||
}
|
||||
trap cleanup EXIT INT TERM
|
||||
|
||||
log() { printf '==> %s\n' "$*" >&2; }
|
||||
die() { printf 'ERROR: %s\n' "$*" >&2; exit 1; }
|
||||
|
||||
for cmd in git uv gh jq curl claude; do
|
||||
command -v "${cmd}" >/dev/null 2>&1 || die "missing required command: ${cmd}"
|
||||
done
|
||||
|
||||
# Publishing pushes the branch straight to BerriAI/litellm-docs and opens
|
||||
# the PR as mateo-berri, who has write access on the docs repo. The same
|
||||
# ${GITHUB_TOKEN} is reused for release-listing above, so require it up
|
||||
# front -- failing 30 minutes into a run because the env file is missing
|
||||
# one line is a waste of CI quota.
|
||||
if [[ "${SKIP_PUBLISH}" != "1" ]]; then
|
||||
[[ -n "${GITHUB_TOKEN:-}" ]] \
|
||||
|| die "GITHUB_TOKEN (mateo-berri, write access to ${DOCS_REPO}) required to push the branch and open the PR (or set SKIP_PUBLISH=1)"
|
||||
fi
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Resolve versions
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Newest PEP 440 *final* release on BerriAI/litellm. LiteLLM moved off
|
||||
# the legacy `vX.Y.Z-stable` tag convention to PEP 440: a final/stable
|
||||
# release is now a bare `vX.Y.Z` tag, while pre-releases carry a
|
||||
# `-rc.N` / `-dev.N` segment (and the old `…-stable` / `…-stable.patch.N`
|
||||
# tags are legacy and frozen at v1.83.x). We therefore select the newest
|
||||
# tag with no pre-release segment -- matching `^v[0-9]+\.[0-9]+\.[0-9]+$`
|
||||
# -- and skip drafts. The numeric version_key sort handles 1.10 > 1.9.
|
||||
#
|
||||
# Paginate through the releases endpoint instead of grabbing only page 1
|
||||
# (default page_size=30). LiteLLM ships multiple pre-releases per day, so
|
||||
# it's common to need to walk past 30+ entries before hitting the most
|
||||
# recent final release. We cap at 5 pages (500 releases) which is
|
||||
# conservatively beyond the worst observed gap.
|
||||
GH_AUTH_HEADER=()
|
||||
if [[ -n "${GITHUB_TOKEN:-}" ]]; then
|
||||
GH_AUTH_HEADER=(-H "Authorization: Bearer ${GITHUB_TOKEN}")
|
||||
fi
|
||||
RELEASES_JSON="${WORKDIR}/releases.json"
|
||||
echo "[]" >"${RELEASES_JSON}"
|
||||
for page in 1 2 3 4 5; do
|
||||
PAGE_JSON="${WORKDIR}/releases.page${page}.json"
|
||||
curl -fsS \
|
||||
-H 'Accept: application/vnd.github+json' \
|
||||
-H 'User-Agent: litellm-compat-matrix' \
|
||||
"${GH_AUTH_HEADER[@]}" \
|
||||
"https://api.github.com/repos/BerriAI/litellm/releases?per_page=100&page=${page}" \
|
||||
>"${PAGE_JSON}"
|
||||
jq -s '.[0] + .[1]' "${RELEASES_JSON}" "${PAGE_JSON}" >"${RELEASES_JSON}.merged"
|
||||
mv "${RELEASES_JSON}.merged" "${RELEASES_JSON}"
|
||||
# Stop early once we've seen at least one final release tag — no point
|
||||
# paging further for a daily script that only needs the newest.
|
||||
if jq -e '[.[] | select((.draft // false) == false) | .tag_name // "" | select(test("^v[0-9]+\\.[0-9]+\\.[0-9]+$"))] | length > 0' "${PAGE_JSON}" >/dev/null; then
|
||||
break
|
||||
fi
|
||||
# No more pages? GitHub returns an empty array past the last page.
|
||||
if [[ "$(jq 'length' "${PAGE_JSON}")" == "0" ]]; then
|
||||
break
|
||||
fi
|
||||
done
|
||||
LITELLM_VERSION="$(
|
||||
jq -r '
|
||||
[ .[]
|
||||
| select((.draft // false) == false)
|
||||
| .tag_name // empty
|
||||
| select(test("^v[0-9]+\\.[0-9]+\\.[0-9]+$"))
|
||||
]
|
||||
| sort_by(
|
||||
capture("^v(?<a>[0-9]+)\\.(?<b>[0-9]+)\\.(?<c>[0-9]+)$")
|
||||
| [(.a|tonumber), (.b|tonumber), (.c|tonumber)]
|
||||
)
|
||||
| last // empty
|
||||
' "${RELEASES_JSON}"
|
||||
)"
|
||||
[[ -n "${LITELLM_VERSION}" ]] || die "could not resolve latest PEP 440 final release (vX.Y.Z) in 5 pages of releases"
|
||||
log "resolved litellm: ${LITELLM_VERSION}"
|
||||
|
||||
CLAUDE_CODE_VERSION="$(claude --version 2>/dev/null | awk '{print $1}')"
|
||||
[[ -n "${CLAUDE_CODE_VERSION}" ]] || die "could not read 'claude --version'"
|
||||
log "local claude code: ${CLAUDE_CODE_VERSION}"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Update the worktree to that tag
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
if [[ ! -d "${WORKTREE}/.git" ]]; then
|
||||
log "first run: cloning litellm into ${WORKTREE}"
|
||||
mkdir -p "$(dirname "${WORKTREE}")"
|
||||
git clone https://github.com/BerriAI/litellm.git "${WORKTREE}"
|
||||
fi
|
||||
|
||||
log "updating worktree to ${LITELLM_VERSION}"
|
||||
git -C "${WORKTREE}" fetch --tags --force
|
||||
git -C "${WORKTREE}" reset --hard
|
||||
# Keep the venv, the .uv-bin cache, and the .uv-python managed
|
||||
# interpreter around — uv sync will reconcile the venv on every run,
|
||||
# and we don't want to re-download the pinned uv binary or the managed
|
||||
# CPython each time. Drop everything else (including any prior
|
||||
# tests/e2e/ shim) so each run starts clean before the shim below
|
||||
# rewrites it from the dev checkout.
|
||||
git -C "${WORKTREE}" clean -fdx -e .venv -e .uv-bin -e .uv-python
|
||||
git -C "${WORKTREE}" checkout --force "${LITELLM_VERSION}"
|
||||
|
||||
# Always rebuild tests/e2e/ in the worktree from the dev checkout,
|
||||
# regardless of what the resolved ${LITELLM_VERSION} tag ships. Two
|
||||
# reasons:
|
||||
#
|
||||
# * The matrix populator's job is to exercise *today's* tests against
|
||||
# the latest stable proxy. The dev checkout carries the most recent
|
||||
# test fixes that haven't yet rolled into a stable release, and we
|
||||
# want every cron run to pick those up the moment they land on
|
||||
# ${LITELLM_REPO}, not whenever the next stable release happens.
|
||||
# * The tag's own tests/e2e/ ships the full EKS e2e harness, whose
|
||||
# top-level conftest.py imports modules (e2e_db, lifecycle,
|
||||
# otel_client, ...) that the stable venv does not install. Copying
|
||||
# the whole tree would make pytest collection blow up on those
|
||||
# imports.
|
||||
#
|
||||
# So the shim is a fresh `rm -rf` of tests/e2e/ followed by copying ONLY
|
||||
# the claude_code suite plus the shared transport helpers it imports.
|
||||
# pytest puts tests/e2e/ itself on sys.path (it has no __init__.py, while
|
||||
# claude_code/ does), which is what resolves both the `claude_code.*`
|
||||
# and the bare `proxy_client` / `e2e_http` imports inside the suite.
|
||||
E2E_HELPER_FILES=(proxy_client.py e2e_http.py models.py e2e_config.py transport.py)
|
||||
if [[ ! -d "${LITELLM_REPO}/tests/e2e/claude_code" ]]; then
|
||||
die "no shim source at ${LITELLM_REPO}/tests/e2e/claude_code"
|
||||
fi
|
||||
for helper in "${E2E_HELPER_FILES[@]}"; do
|
||||
[[ -f "${LITELLM_REPO}/tests/e2e/${helper}" ]] \
|
||||
|| die "missing shim helper: ${LITELLM_REPO}/tests/e2e/${helper}"
|
||||
done
|
||||
log "shimming tests/e2e/claude_code/ + helpers from ${LITELLM_REPO} (always-overwrite)"
|
||||
rm -rf "${WORKTREE}/tests/e2e"
|
||||
mkdir -p "${WORKTREE}/tests/e2e"
|
||||
cp -r "${LITELLM_REPO}/tests/e2e/claude_code" "${WORKTREE}/tests/e2e/"
|
||||
for helper in "${E2E_HELPER_FILES[@]}"; do
|
||||
cp "${LITELLM_REPO}/tests/e2e/${helper}" "${WORKTREE}/tests/e2e/"
|
||||
done
|
||||
|
||||
# litellm pins an exact uv version in pyproject.toml's [tool.uv]
|
||||
# `required-version` field, so a system uv that's newer or older
|
||||
# refuses to sync. We pin our own local copy at the version the
|
||||
# checked-out tag asks for, cached under .uv-bin/ inside the worktree
|
||||
# so subsequent runs skip the download.
|
||||
PINNED_UV_VERSION="$(
|
||||
awk -F'"' '
|
||||
/^required-version[[:space:]]*=/ {
|
||||
# Field 2 is the value between the quotes, e.g. ">=0.10.9" or
|
||||
# "0.10.9". Strip any leading specifier prefix so we end up with
|
||||
# the bare version string, which is what /releases/download/<v>/
|
||||
# expects.
|
||||
v = $2
|
||||
sub(/^[[:space:]=<>!~]+/, "", v)
|
||||
if (v != "") { print v; exit }
|
||||
}
|
||||
' "${WORKTREE}/pyproject.toml"
|
||||
)"
|
||||
if [[ -z "${PINNED_UV_VERSION}" ]]; then
|
||||
log "no uv version pin in pyproject.toml; using system uv"
|
||||
WORKTREE_UV="$(command -v uv)"
|
||||
else
|
||||
WORKTREE_UV="${WORKTREE}/.uv-bin/uv-${PINNED_UV_VERSION}"
|
||||
if [[ ! -x "${WORKTREE_UV}" ]]; then
|
||||
log "downloading uv ${PINNED_UV_VERSION} for the worktree"
|
||||
mkdir -p "${WORKTREE}/.uv-bin"
|
||||
UV_TARBALL_NAME="uv-x86_64-unknown-linux-gnu.tar.gz"
|
||||
UV_DOWNLOAD_URL="https://github.com/astral-sh/uv/releases/download/${PINNED_UV_VERSION}/${UV_TARBALL_NAME}"
|
||||
UV_TMPDIR="$(mktemp -d -t uv-download.XXXXXX)"
|
||||
# Download the tarball and Astral's official .sha256 sidecar to disk
|
||||
# and verify the digest before extracting/executing anything. This
|
||||
# closes the supply-chain trust gap of piping a remote binary
|
||||
# straight into `tar -xzO ... > file ; chmod +x` (see CLAUDE.md
|
||||
# "CI Supply-Chain Safety").
|
||||
curl -fsSL --output "${UV_TMPDIR}/${UV_TARBALL_NAME}" "${UV_DOWNLOAD_URL}"
|
||||
curl -fsSL --output "${UV_TMPDIR}/${UV_TARBALL_NAME}.sha256" "${UV_DOWNLOAD_URL}.sha256"
|
||||
(cd "${UV_TMPDIR}" && sha256sum -c "${UV_TARBALL_NAME}.sha256") \
|
||||
|| { rm -rf "${UV_TMPDIR}"; die "uv ${PINNED_UV_VERSION} sha256 mismatch — refusing to install"; }
|
||||
tar -xzf "${UV_TMPDIR}/${UV_TARBALL_NAME}" -C "${UV_TMPDIR}" "uv-x86_64-unknown-linux-gnu/uv"
|
||||
mv "${UV_TMPDIR}/uv-x86_64-unknown-linux-gnu/uv" "${WORKTREE_UV}.tmp"
|
||||
chmod +x "${WORKTREE_UV}.tmp"
|
||||
mv "${WORKTREE_UV}.tmp" "${WORKTREE_UV}"
|
||||
rm -rf "${UV_TMPDIR}"
|
||||
fi
|
||||
fi
|
||||
# `--extra proxy` pulls fastapi/uvicorn/etc. so `uv run litellm` can
|
||||
# actually serve. `--group proxy-dev` brings in pytest and the rest of
|
||||
# what tests/e2e/claude_code/ needs. `--python` pins the venv to
|
||||
# ${CRON_PYTHON_VERSION}; the first run after a version bump recreates
|
||||
# the venv from scratch (a one-time cold sync).
|
||||
export UV_PYTHON_INSTALL_DIR="${WORKTREE}/.uv-python"
|
||||
log "uv sync --frozen --group proxy-dev --extra proxy --python ${CRON_PYTHON_VERSION} (uv ${PINNED_UV_VERSION:-system})"
|
||||
(cd "${WORKTREE}" && "${WORKTREE_UV}" sync --frozen --group proxy-dev --extra proxy --python "${CRON_PYTHON_VERSION}")
|
||||
|
||||
PROXY_CONFIG="${WORKTREE}/tests/e2e/claude_code/test_config.yaml"
|
||||
[[ -f "${PROXY_CONFIG}" ]] || die "proxy config not found at ${PROXY_CONFIG} (shim incomplete?)"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Boot the proxy
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
log "starting proxy on 127.0.0.1:${PROXY_PORT}"
|
||||
# Bind the proxy to loopback only. The populator proxy is talked to
|
||||
# exclusively by the pytest run on the same host (the health check and
|
||||
# the test env set `LITELLM_PROXY_URL=http://127.0.0.1:...`),
|
||||
# so there's no reason to expose it on the VM's external interfaces.
|
||||
# Without `--host`, `litellm` defaults to 0.0.0.0, which combined with
|
||||
# the predictable default `LITELLM_MASTER_KEY=sk-cron-matrix` would
|
||||
# allow anything that can reach :${PROXY_PORT} on the VM to authenticate
|
||||
# and burn upstream provider credentials.
|
||||
#
|
||||
# `setsid` puts the proxy in its own session+pgroup so cleanup() can
|
||||
# SIGTERM the whole tree by passing the pgid as a negative pid. We
|
||||
# write that pid to a file so cleanup() doesn't need to remember a
|
||||
# variable that might be stale by the time the trap fires.
|
||||
setsid env LITELLM_MASTER_KEY="${PROXY_API_KEY}" bash -c '
|
||||
echo "$$" > "$0"
|
||||
cd "$1"
|
||||
exec "$2" run litellm --config "$3" --host 127.0.0.1 --port "$4"
|
||||
' "${PROXY_PID_FILE}" "${WORKTREE}" "${WORKTREE_UV}" "${PROXY_CONFIG}" "${PROXY_PORT}" \
|
||||
>"${WORKDIR}/proxy.log" 2>&1 &
|
||||
disown
|
||||
|
||||
HEALTH_URL="http://127.0.0.1:${PROXY_PORT}/health/liveliness"
|
||||
for _ in $(seq 1 45); do
|
||||
if curl -fsS "${HEALTH_URL}" >/dev/null 2>&1; then
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
curl -fsS "${HEALTH_URL}" >/dev/null \
|
||||
|| { tail -50 "${WORKDIR}/proxy.log" >&2; die "proxy did not become healthy"; }
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. Run pytest
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
RESULTS_JSON="${WORKDIR}/compat-results.json"
|
||||
# The `_*_unit_tests` ignore is defensive: those harness-only trees are
|
||||
# markerless (they run without a proxy) and don't feed matrix cells, so
|
||||
# the cron skips them if/when they land in the suite.
|
||||
PYTEST_ARGS=(
|
||||
tests/e2e/claude_code/
|
||||
"--ignore-glob=*_unit_tests*"
|
||||
)
|
||||
if [[ -n "${PYTEST_K}" ]]; then
|
||||
log "PYTEST_K set; narrowing to: ${PYTEST_K}"
|
||||
PYTEST_ARGS+=(-k "${PYTEST_K}")
|
||||
fi
|
||||
|
||||
log "running pytest"
|
||||
set +e
|
||||
(
|
||||
cd "${WORKTREE}" \
|
||||
&& LITELLM_PROXY_URL="http://127.0.0.1:${PROXY_PORT}" \
|
||||
LITELLM_MASTER_KEY="${PROXY_API_KEY}" \
|
||||
COMPAT_RESULTS_PATH="${RESULTS_JSON}" \
|
||||
"${WORKTREE_UV}" run pytest "${PYTEST_ARGS[@]}"
|
||||
)
|
||||
PYTEST_EXIT=$?
|
||||
set -e
|
||||
log "pytest exit code: ${PYTEST_EXIT} (failures become 'fail' cells, not script errors)"
|
||||
[[ -f "${RESULTS_JSON}" ]] || die "pytest did not produce ${RESULTS_JSON}"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. Build the matrix JSON
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
MATRIX_JSON="${WORKDIR}/compatibility-matrix.json"
|
||||
log "building ${MATRIX_JSON}"
|
||||
(
|
||||
cd "${WORKTREE}" \
|
||||
&& "${WORKTREE_UV}" run python "${POPULATOR_DIR}/build_matrix.py" \
|
||||
--manifest "${WORKTREE}/tests/e2e/claude_code/manifest.yaml" \
|
||||
--results "${RESULTS_JSON}" \
|
||||
--output "${MATRIX_JSON}" \
|
||||
--litellm-version "${LITELLM_VERSION}" \
|
||||
--claude-code-version "${CLAUDE_CODE_VERSION}"
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 6. Open a docs-repo PR
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
if [[ "${SKIP_PUBLISH}" == "1" ]]; then
|
||||
cp "${MATRIX_JSON}" "${LITELLM_REPO}/compatibility-matrix.json"
|
||||
log "SKIP_PUBLISH=1; matrix written to ${LITELLM_REPO}/compatibility-matrix.json"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
DATE_UTC="$(date -u +%Y-%m-%d)"
|
||||
BRANCH_NAME="compat-matrix/${LITELLM_VERSION}-${CLAUDE_CODE_VERSION}-${DATE_UTC}"
|
||||
DOCS_CLONE="${WORKDIR}/litellm-docs"
|
||||
|
||||
log "cloning ${DOCS_REPO}@${DOCS_BRANCH}"
|
||||
gh repo clone "${DOCS_REPO}" "${DOCS_CLONE}" -- --depth 1 --branch "${DOCS_BRANCH}"
|
||||
|
||||
cd "${DOCS_CLONE}"
|
||||
git config user.email "litellm-bot@berri.ai"
|
||||
git config user.name "litellm-compat-matrix-bot"
|
||||
git checkout -b "${BRANCH_NAME}"
|
||||
|
||||
# Snapshot the currently-published matrix *before* we overwrite it, so the
|
||||
# auto-merge gate below can diff old→new cell statuses. On the first-ever
|
||||
# publish the file won't exist yet; we leave ${PUBLISHED_MATRIX} pointing
|
||||
# at a path that doesn't exist and let check_regressions.py treat that as
|
||||
# "no baseline → no regressions".
|
||||
PUBLISHED_MATRIX="${WORKDIR}/published-matrix.json"
|
||||
if [[ -f "${DOCS_TARGET_PATH}" ]]; then
|
||||
cp "${DOCS_TARGET_PATH}" "${PUBLISHED_MATRIX}"
|
||||
fi
|
||||
|
||||
mkdir -p "$(dirname "${DOCS_TARGET_PATH}")"
|
||||
cp "${MATRIX_JSON}" "${DOCS_TARGET_PATH}"
|
||||
git add "${DOCS_TARGET_PATH}"
|
||||
|
||||
if git diff --cached --quiet; then
|
||||
log "matrix JSON unchanged from ${DOCS_BRANCH}; skipping PR"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# --- Auto-merge regression gate --------------------------------------------
|
||||
# Only auto-merge when the new matrix is improvement-or-equal: every cell
|
||||
# transition is red→green, green→green, or red→red. If any cell flips
|
||||
# green→red (a `pass` that became `fail`), we still open/refresh the PR but
|
||||
# leave auto-merge OFF so a human reviews the regression before it lands on
|
||||
# the public docs table. A pre-existing red cell (e.g. Anthropic out of API
|
||||
# credits) is red→red and does NOT block, so the daily PR keeps flowing.
|
||||
log "checking for green->red regressions vs the published matrix"
|
||||
set +e
|
||||
REGRESSION_REPORT="$(
|
||||
cd "${WORKTREE}" \
|
||||
&& "${WORKTREE_UV}" run python "${POPULATOR_DIR}/check_regressions.py" \
|
||||
--old "${PUBLISHED_MATRIX}" \
|
||||
--new "${MATRIX_JSON}"
|
||||
)"
|
||||
REGRESSION_EXIT=$?
|
||||
set -e
|
||||
printf '%s\n' "${REGRESSION_REPORT}" | sed 's/^/ /' >&2
|
||||
# Exit 0 = clean. Exit 3 = green→red regression(s) found. Any other code
|
||||
# means the checker itself errored; fail *closed* (withhold auto-merge) so a
|
||||
# bug in the gate can never silently auto-merge a regression.
|
||||
if [[ ${REGRESSION_EXIT} -eq 0 ]]; then
|
||||
ALLOW_AUTOMERGE=1
|
||||
elif [[ ${REGRESSION_EXIT} -eq 3 ]]; then
|
||||
ALLOW_AUTOMERGE=0
|
||||
log "WARN: green->red regression(s) detected; auto-merge will be left OFF for review"
|
||||
else
|
||||
ALLOW_AUTOMERGE=0
|
||||
log "WARN: regression check errored (exit ${REGRESSION_EXIT}); withholding auto-merge to be safe"
|
||||
fi
|
||||
|
||||
GENERATED_AT="$(jq -r '.generated_at' "${MATRIX_JSON}")"
|
||||
COMMIT_MSG="$(cat <<EOF
|
||||
Update Claude Code compatibility matrix
|
||||
|
||||
litellm_version: ${LITELLM_VERSION}
|
||||
claude_code_version: ${CLAUDE_CODE_VERSION}
|
||||
generated_at: ${GENERATED_AT}
|
||||
EOF
|
||||
)"
|
||||
git commit -m "${COMMIT_MSG}"
|
||||
|
||||
# Push the branch straight to BerriAI/litellm-docs. mateo-berri has write
|
||||
# access on the docs repo, so there's no fork hop: the PR is a same-repo
|
||||
# branch PR. The temp remote carries the token in its URL, so we add it,
|
||||
# push, then immediately remove it so the token never lingers in
|
||||
# ${DOCS_CLONE}/.git/config. (${DOCS_CLONE} is also rm -rf'd by the
|
||||
# cleanup trap on exit.)
|
||||
#
|
||||
# Plain --force (not --force-with-lease) is acceptable here: the
|
||||
# compat-matrix/* branch is bot-owned, only this script ever writes to
|
||||
# it, and runs are serialized by the systemd timer. --force-with-lease
|
||||
# would require a fetch to populate the remote-tracking ref before each
|
||||
# push and adds no safety in this single-writer setup.
|
||||
PUBLISH_PUSH_URL="https://x-access-token:${GITHUB_TOKEN}@github.com/${DOCS_REPO}.git"
|
||||
git remote remove publish 2>/dev/null || true
|
||||
git remote add publish "${PUBLISH_PUSH_URL}"
|
||||
git push --force --set-upstream publish "${BRANCH_NAME}"
|
||||
git remote remove publish
|
||||
unset PUBLISH_PUSH_URL
|
||||
|
||||
# Per-feature status table for the PR body. Reviewers triage from this.
|
||||
PR_FEATURE_TABLE="$(jq -r '
|
||||
.features[] as $f
|
||||
| "- **\($f.name)**: " +
|
||||
([ .providers[] as $p
|
||||
| "\($p)=\($f.providers[$p].status // "not_tested")"
|
||||
] | join(", "))
|
||||
' "${MATRIX_JSON}")"
|
||||
|
||||
# When the gate withheld auto-merge, call it out at the top of the PR body
|
||||
# (with the offending cells) so a reviewer knows this PR needs a human and
|
||||
# why. On the clean path this section is empty. Note `$(...)` strips the
|
||||
# trailing newline, so the body below puts explicit blank lines *around*
|
||||
# the placeholder rather than relying on the heredoc's own spacing.
|
||||
if [[ "${ALLOW_AUTOMERGE}" != "1" ]]; then
|
||||
PR_REGRESSION_SECTION="$(cat <<EOF
|
||||
> [!WARNING]
|
||||
> **Auto-merge disabled:** one or more cells regressed green→red versus the
|
||||
> currently-published matrix. Review the diff before merging.
|
||||
|
||||
\`\`\`
|
||||
${REGRESSION_REPORT}
|
||||
\`\`\`
|
||||
EOF
|
||||
)"
|
||||
else
|
||||
PR_REGRESSION_SECTION=""
|
||||
fi
|
||||
|
||||
PR_TITLE="chore(compat-matrix): refresh for ${LITELLM_VERSION} + claude-code ${CLAUDE_CODE_VERSION}"
|
||||
PR_BODY="$(cat <<EOF
|
||||
Automated daily refresh of the Claude Code compatibility matrix.
|
||||
|
||||
${PR_REGRESSION_SECTION}
|
||||
|
||||
| Field | Value |
|
||||
| --- | --- |
|
||||
| litellm_version | \`${LITELLM_VERSION}\` |
|
||||
| claude_code_version | \`${CLAUDE_CODE_VERSION}\` |
|
||||
| generated_at | \`${GENERATED_AT}\` |
|
||||
|
||||
## Per-feature results
|
||||
|
||||
${PR_FEATURE_TABLE}
|
||||
|
||||
---
|
||||
|
||||
Generated by \`tests/e2e/claude_code/cron_vm/run_daily.sh\`. Close without merging if the diff looks wrong; the next cron run will reopen with fresh results.
|
||||
EOF
|
||||
)"
|
||||
|
||||
log "opening PR from ${BRANCH_NAME} -> ${DOCS_REPO}:${DOCS_BRANCH} (as mateo-berri)"
|
||||
# GH_TOKEN is mateo-berri's write-scoped token, the same identity used
|
||||
# for release-listing above. The branch lives on ${DOCS_REPO} itself, so
|
||||
# --head is a bare branch name (a same-repo PR), not `OWNER:BRANCH`.
|
||||
set +e
|
||||
PR_OUT="$(
|
||||
GH_TOKEN="${GITHUB_TOKEN}" gh pr create \
|
||||
--repo "${DOCS_REPO}" \
|
||||
--base "${DOCS_BRANCH}" \
|
||||
--head "${BRANCH_NAME}" \
|
||||
--title "${PR_TITLE}" \
|
||||
--body "${PR_BODY}" 2>&1
|
||||
)"
|
||||
PR_EXIT=$?
|
||||
set -e
|
||||
echo "${PR_OUT}"
|
||||
|
||||
if [[ ${PR_EXIT} -ne 0 ]]; then
|
||||
if grep -q "a pull request for branch.*already exists" <<<"${PR_OUT}"; then
|
||||
log "PR already exists for ${BRANCH_NAME}; updated branch in place"
|
||||
else
|
||||
die "gh pr create failed (exit ${PR_EXIT})"
|
||||
fi
|
||||
fi
|
||||
|
||||
# Enable auto-merge so the PR merges itself once the docs repo's required
|
||||
# checks pass -- we no longer gate these bot PRs on a second human
|
||||
# approval. mateo-berri authors and merges them directly. The repo only
|
||||
# permits squash merges and has auto-merge enabled at the repo level
|
||||
# (${AUTO_MERGE_METHOD} defaults to squash accordingly).
|
||||
#
|
||||
# This only fires when the regression gate above is satisfied
|
||||
# (${ALLOW_AUTOMERGE}==1): a green→red regression — or a gate error —
|
||||
# leaves auto-merge OFF so a human triages the PR.
|
||||
#
|
||||
# `gh pr merge --auto` is idempotent: re-enabling auto-merge on a PR that
|
||||
# already has it set is a no-op, so same-day reruns stay clean. It's
|
||||
# non-fatal: if auto-merge can't be enabled (e.g. the PR is already in a
|
||||
# clean/mergeable state with nothing left to wait on, or branch
|
||||
# protection isn't configured), the matrix JSON has still landed on the
|
||||
# PR and the worst case is a manual merge click.
|
||||
if [[ "${ALLOW_AUTOMERGE}" == "1" ]]; then
|
||||
log "enabling ${AUTO_MERGE_METHOD} auto-merge on ${BRANCH_NAME}"
|
||||
set +e
|
||||
GH_TOKEN="${GITHUB_TOKEN}" gh pr merge \
|
||||
"${BRANCH_NAME}" \
|
||||
--repo "${DOCS_REPO}" \
|
||||
--auto \
|
||||
"--${AUTO_MERGE_METHOD}" 2>&1 | sed 's/^/ /'
|
||||
AUTOMERGE_EXIT=${PIPESTATUS[0]}
|
||||
set -e
|
||||
if [[ ${AUTOMERGE_EXIT} -ne 0 ]]; then
|
||||
log "WARN: gh pr merge --auto exited ${AUTOMERGE_EXIT} (non-fatal)"
|
||||
fi
|
||||
else
|
||||
# Regression (or gate error): make sure auto-merge is OFF. A same-day
|
||||
# rerun may have enabled it on an earlier, clean pass, so explicitly
|
||||
# disable rather than just skipping. Non-fatal: if it was never enabled,
|
||||
# `--disable-auto` is a harmless no-op/error we swallow.
|
||||
log "leaving ${BRANCH_NAME} for manual review; disabling any prior auto-merge"
|
||||
set +e
|
||||
GH_TOKEN="${GITHUB_TOKEN}" gh pr merge \
|
||||
"${BRANCH_NAME}" \
|
||||
--repo "${DOCS_REPO}" \
|
||||
--disable-auto 2>&1 | sed 's/^/ /'
|
||||
set -e
|
||||
fi
|
||||
|
||||
# --- Stale-PR sweep ----------------------------------------------------------
|
||||
# Keep at most ONE compat-matrix PR open: today's. Any other open
|
||||
# `compat-matrix/*` PR is a leftover from a day whose regression gate
|
||||
# withheld auto-merge and nobody triaged it; the PR we just opened or
|
||||
# refreshed above carries strictly fresher results, so the old one is
|
||||
# pure queue noise. Closing is non-destructive — the PR record and its
|
||||
# regression report stay browsable; only the bot-owned branch is
|
||||
# deleted. This runs only after today's PR exists (a `die` above skips
|
||||
# it), so a failed publish can never close the queue down to zero.
|
||||
#
|
||||
# Non-fatal: a sweep failure (rate limit, transient API error) leaves
|
||||
# stale PRs for the next run to retry; it must not fail the pipeline.
|
||||
log "sweeping stale compat-matrix PRs (keeping ${BRANCH_NAME})"
|
||||
set +e
|
||||
STALE_PRS="$(
|
||||
GH_TOKEN="${GITHUB_TOKEN}" gh pr list \
|
||||
--repo "${DOCS_REPO}" \
|
||||
--state open \
|
||||
--limit 100 \
|
||||
--json number,headRefName \
|
||||
--jq '.[] | select(.headRefName | startswith("compat-matrix/")) | "\(.number)\t\(.headRefName)"'
|
||||
)"
|
||||
while IFS=$'\t' read -r stale_pr stale_head; do
|
||||
[[ -z "${stale_pr}" ]] && continue
|
||||
[[ "${stale_head}" == "${BRANCH_NAME}" ]] && continue
|
||||
GH_TOKEN="${GITHUB_TOKEN}" gh pr close "${stale_pr}" \
|
||||
--repo "${DOCS_REPO}" \
|
||||
--delete-branch \
|
||||
--comment "Superseded by the newer daily compat-matrix PR from \`${BRANCH_NAME}\`; the populator keeps only the most recent compat-matrix PR open." 2>&1 | sed 's/^/ /'
|
||||
if [[ ${PIPESTATUS[0]} -eq 0 ]]; then
|
||||
log "closed stale compat-matrix PR #${stale_pr} (${stale_head})"
|
||||
else
|
||||
log "WARN: could not close stale compat-matrix PR #${stale_pr} (non-fatal)"
|
||||
fi
|
||||
done <<<"${STALE_PRS}"
|
||||
set -e
|
||||
|
||||
log "done"
|
||||
|
|
@ -174,6 +174,86 @@ def _aggregate_cell(results: Sequence[Mapping[str, Any]]) -> Dict[str, Any]:
|
|||
return {"status": "not_tested"}
|
||||
|
||||
|
||||
def _index_cells(matrix: Mapping[str, Any]) -> dict[tuple[str, str], dict[str, Any]]:
|
||||
"""Map ``(feature_id, provider) -> cell dict`` for a built matrix.
|
||||
|
||||
Cells are keyed by the *stable* feature ``id`` (not the display
|
||||
``name``, which can be reworded without changing the underlying row)
|
||||
and the provider key, so two matrices built at different times line up
|
||||
even if feature names drift.
|
||||
"""
|
||||
out: dict[tuple[str, str], dict[str, Any]] = {}
|
||||
for feature in matrix.get("features", []) or []:
|
||||
if not isinstance(feature, Mapping):
|
||||
continue
|
||||
feature_id = feature.get("id")
|
||||
if not feature_id:
|
||||
continue
|
||||
providers = feature.get("providers", {}) or {}
|
||||
if not isinstance(providers, Mapping):
|
||||
continue
|
||||
for provider, cell in providers.items():
|
||||
if isinstance(cell, Mapping):
|
||||
out[(feature_id, provider)] = dict(cell)
|
||||
return out
|
||||
|
||||
|
||||
def find_regressions(
|
||||
old_matrix: Mapping[str, Any],
|
||||
new_matrix: Mapping[str, Any],
|
||||
) -> list[dict[str, str]]:
|
||||
"""Return the cells that flipped green→red (``pass`` → ``fail``).
|
||||
|
||||
A *regression* is defined strictly: a cell that was ``pass`` in
|
||||
``old_matrix`` and is ``fail`` in ``new_matrix``. Every other
|
||||
transition is intentionally *not* a regression:
|
||||
|
||||
* ``red → green`` / ``green → green`` — the happy path.
|
||||
* ``red → red`` — a cell that is *already* failing for an unrelated
|
||||
reason (e.g. Anthropic out of API credits) must not block
|
||||
publishing, otherwise the daily PR would never auto-merge until
|
||||
that independent issue is fixed.
|
||||
* ``green → not_tested`` / ``green → not_applicable`` — a cell going
|
||||
grey is a degradation but not a *red* regression; treating a
|
||||
skipped/flaky run as a hard block would create false positives.
|
||||
|
||||
Cells present only in ``new_matrix`` (a newly added feature or
|
||||
provider) have no baseline and therefore cannot be regressions.
|
||||
|
||||
Each returned item is a flat str→str mapping so callers (the cron's
|
||||
``check_regressions.py``) can render it without further lookups:
|
||||
``feature_id``, ``feature_name``, ``provider``, ``old_status``,
|
||||
``new_status``, ``error``.
|
||||
"""
|
||||
old_cells = _index_cells(old_matrix)
|
||||
feature_names = {
|
||||
f.get("id"): str(f.get("name", f.get("id")))
|
||||
for f in new_matrix.get("features", []) or []
|
||||
if isinstance(f, Mapping) and f.get("id")
|
||||
}
|
||||
|
||||
regressions: list[dict[str, str]] = []
|
||||
for (feature_id, provider), new_cell in sorted(
|
||||
_index_cells(new_matrix).items(), key=lambda kv: (kv[0][0], kv[0][1])
|
||||
):
|
||||
if new_cell.get("status") != "fail":
|
||||
continue
|
||||
old_cell = old_cells.get((feature_id, provider))
|
||||
if old_cell is None or old_cell.get("status") != "pass":
|
||||
continue
|
||||
regressions.append(
|
||||
{
|
||||
"feature_id": str(feature_id),
|
||||
"feature_name": feature_names.get(feature_id, str(feature_id)),
|
||||
"provider": str(provider),
|
||||
"old_status": "pass",
|
||||
"new_status": "fail",
|
||||
"error": str(new_cell.get("error", "")),
|
||||
}
|
||||
)
|
||||
return regressions
|
||||
|
||||
|
||||
def build_from_paths(
|
||||
*,
|
||||
manifest_path: Path,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue