fix(cost-map-sync): keep the bot merging without a ruleset bypass and stop pinning the prices it owns

The reconcile only looks at bot PRs against the branch the run is on, so a
dispatch from another branch never counts as the open sync PR. When a green
bot PR cannot be merged because the app is not a bypass actor yet, the run
arms auto-merge and warns instead of failing every tick. A base with no
required checks falls back to every check so a dispatch there can still
merge, and a tick whose catalogs and base match the last no-op sync skips
the install and the script.

guard-main-branch accepts litellm_cost_map_sync_* heads so the bot keeps
working once main is the default branch again.

The tests that pinned exact OpenRouter prices and limits now check that the
entries exist and are priced: the sync owns those values, and a pin would turn
every legitimate reprice into a red bot PR that pauses syncing.
This commit is contained in:
mateo-berri 2026-09-05 00:11:12 -07:00
parent 26f74e662d
commit 340569b09a
5 changed files with 73 additions and 67 deletions

View file

@ -45,7 +45,8 @@ jobs:
pr=""
if [ -n "$BOT_LOGIN" ]; then
pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --author "app/$BOT_LOGIN" \
--search "in:title \"$PR_TITLE\"" --json number,headRefName,mergeable,isCrossRepository \
--base "$GITHUB_REF_NAME" --search "in:title \"$PR_TITLE\"" \
--json number,headRefName,mergeable,isCrossRepository,autoMergeRequest \
--jq "[.[] | select((.headRefName | startswith(\"$BRANCH_PREFIX\")) and (.isCrossRepository | not))] | first // empty")"
fi
sync=false
@ -62,11 +63,21 @@ jobs:
guard="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json name,state \
--jq '.[] | select(.name == "cost-map-guard") | .state' || true)"
required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --required --json bucket \
--jq 'map(.bucket) | unique | join(",")' || true)"
--jq 'map(.bucket) | unique | join(",")' 2>/dev/null || true)"
if [ -z "$required" ]; then
required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json bucket \
--jq 'map(.bucket) | unique | join(",")' || true)"
fi
case "$guard,$required" in
SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping)
gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch
echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry."
if gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch; then
echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry."
else
if [ "$(jq -r .autoMergeRequest <<< "$pr")" = "null" ]; then
gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --auto --merge --delete-branch
fi
echo "::warning::Sync PR #$number is green but the app is not a bypass actor on $GITHUB_REF_NAME; auto-merge is armed, so one approval lands it."
fi
;;
*FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*)
echo "::warning::Sync PR #$number has a failing check (cost-map-guard=$guard, required buckets=$required); leaving it open for a human."
@ -83,23 +94,47 @@ jobs:
- name: Explain why no PR can be opened
if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID == '' && !inputs.dry_run
run: echo "::warning::COST_MAP_BOT_APP_ID is not configured, so no sync PR can be opened or merged; dispatch with dry_run to see the diff."
- name: Hash the catalogs and the base
id: catalogs
if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run
run: |
digest="$(curl -fsSL https://openrouter.ai/api/v1/models https://ai-gateway.vercel.sh/v1/models | sha256sum | cut -c1-64)"
echo "key=cost-map-sync-${GITHUB_SHA}-${digest}" >> "$GITHUB_OUTPUT"
- name: Look up whether this catalog state was already synced
id: seen
if: steps.catalogs.outputs.key != ''
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: ${{ runner.temp }}/synced
key: ${{ steps.catalogs.outputs.key }}
lookup-only: true
- name: Skip the unchanged catalogs
if: steps.seen.outputs.cache-hit == 'true'
run: echo "Neither catalog nor the base has changed since the last sync found nothing to do."
- name: Set up uv
if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run)
if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true'))
uses: ./.github/actions/setup-uv-with-retries
with:
version: "0.10.9"
- name: Run the sync
id: sync
if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run)
if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true'))
run: |
uv run --frozen --no-dev python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md"
uv run --frozen --no-dev python ci_cd/generate_model_prices_schema.py
if git diff --quiet; then
echo "Registry already in sync; no PR needed."
echo "changed=false" >> "$GITHUB_OUTPUT"
echo "$GITHUB_SHA" > "$RUNNER_TEMP/synced"
else
echo "changed=true" >> "$GITHUB_OUTPUT"
fi
- name: Remember that this catalog state needs no PR
if: steps.sync.outputs.changed == 'false' && steps.catalogs.outputs.key != ''
uses: actions/cache/save@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: ${{ runner.temp }}/synced
key: ${{ steps.catalogs.outputs.key }}
- name: Open the sync PR
if: steps.sync.outputs.changed == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run
run: |

View file

@ -34,9 +34,9 @@ jobs:
echo "::error::PRs to main must originate from the canonical repository ($BASE_REPO), not a fork ($HEAD_REPO). External contributors should open PRs against 'litellm_internal_staging' instead."
exit 1
fi
if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]]; then
if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]] || [[ "$HEAD_REF" == litellm_cost_map_sync_?* ]]; then
echo "Allowed source branch."
exit 0
fi
echo "::error::PRs to main must originate from 'litellm_internal_staging' or a 'litellm_hotfix_*' branch. Got: '$HEAD_REF'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead."
echo "::error::PRs to main must originate from 'litellm_internal_staging', a 'litellm_hotfix_*' branch, or a 'litellm_cost_map_sync_*' bot branch. Got: '$HEAD_REF'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead."
exit 1

View file

@ -247,22 +247,6 @@ def test_azure_ai_claude_1m_context_entries(cost_map: dict):
assert cost_map[model]["max_input_tokens"] == 200000, model
# OpenRouter headline rates from GET https://openrouter.ai/api/v1/models.
# These were the catalog values that disagreed with that API (and, for the
# two spotlight models, the public model pages that their source fields cite).
_OPENROUTER_LIVE_COSTS = {
"openrouter/qwen/qwen3.5-plus-02-15": (2.6e-07, 1.56e-06, None),
"openrouter/openai/gpt-oss-120b": (3.7e-08, 1.7e-07, None),
"openrouter/qwen/qwen3-coder-plus": (6.5e-07, 3.25e-06, None),
"openrouter/qwen/qwen3.5-flash-02-23": (6.5e-08, 2.6e-07, None),
"openrouter/qwen/qwen3.5-27b": (1.95e-07, 1.56e-06, None),
"openrouter/gryphe/mythomax-l2-13b": (6e-08, 6e-08, None),
"openrouter/mancer/weaver": (4e-07, 7.5e-07, None),
"openrouter/xiaomi/mimo-v2.5-pro": (4.35e-07, 8.7e-07, 3.6e-09),
"openrouter/moonshotai/kimi-k2.5": (4.5e-07, 2.25e-06, 7e-08),
"openrouter/z-ai/glm-5": (6e-07, 1.92e-06, None),
}
_OPENROUTER_STALE_COSTS = {
"openrouter/qwen/qwen3.5-plus-02-15": (4e-07, 2.4e-06),
"openrouter/openai/gpt-oss-120b": (1.8e-07, 8e-07),
@ -275,23 +259,12 @@ _OPENROUTER_STALE_COSTS = {
[_load_root_cost_map(), GetModelCostMap.load_local_model_cost_map()],
ids=["root", "bundled_backup"],
)
def test_openrouter_catalog_costs_match_live_headline_rates(cost_map: dict):
"""openrouter/* spend tracking reads these catalog fields. The values must
stay aligned with OpenRouter's published headline rate, not the stale
figures that over/under-counted by up to 30x. Both maps are checked so
the root file and bundled backup cannot drift apart."""
control = cost_map["openrouter/anthropic/claude-opus-5"]
assert control["input_cost_per_token"] == 5e-06
assert control["output_cost_per_token"] == 2.5e-05
assert control["cache_read_input_token_cost"] == 5e-07
for model, (inp, out, cache) in _OPENROUTER_LIVE_COSTS.items():
entry = cost_map[model]
assert entry["input_cost_per_token"] == inp, model
assert entry["output_cost_per_token"] == out, model
if cache is not None:
assert entry["cache_read_input_token_cost"] == cache, model
def test_openrouter_stale_catalog_costs_never_return(cost_map: dict):
"""scripts/sync_cost_map.py keeps openrouter/* aligned with OpenRouter's
published headline rates, so the live values are not pinned here: a pin
would turn every legitimate reprice into a red sync PR. The stale figures
that over/under-counted by up to 30x must never come back, in the root
file or the bundled backup."""
for model, (stale_in, stale_out) in _OPENROUTER_STALE_COSTS.items():
entry = cost_map[model]
assert entry["input_cost_per_token"] != stale_in, model

View file

@ -197,10 +197,10 @@ def test_openrouter_qwen36_plus_model_info(_local_model_cost_map):
assert model_info is not None
assert model_info["litellm_provider"] == "openrouter"
assert model_info["mode"] == "chat"
assert model_info["max_input_tokens"] == 1000000
assert model_info["max_output_tokens"] == 65536
assert model_info["input_cost_per_token"] == 3.25e-07
assert model_info["output_cost_per_token"] == 1.95e-06
assert model_info["max_input_tokens"] > 0
assert model_info["max_output_tokens"] > 0
assert model_info["input_cost_per_token"] > 0
assert model_info["output_cost_per_token"] > 0
assert model_info["supports_function_calling"] is True
assert model_info["supports_tool_choice"] is True
assert model_info["supports_reasoning"] is True
@ -3273,10 +3273,10 @@ def test_openrouter_gemini_3_1_flash_lite_preview_pricing(_local_model_cost_map)
assert model_info is not None, f"Missing model pricing entry: {model_name}"
assert model_info["litellm_provider"] == "openrouter"
assert model_info["input_cost_per_token"] == 2.5e-07
assert model_info["output_cost_per_token"] == 1.5e-06
assert model_info["max_input_tokens"] == 1048576
assert model_info["max_output_tokens"] == 65536
assert model_info["input_cost_per_token"] > 0
assert model_info["output_cost_per_token"] > 0
assert model_info["max_input_tokens"] > 0
assert model_info["max_output_tokens"] > 0
def test_gemini_3_1_flash_lite_pricing(_local_model_cost_map):
@ -3615,8 +3615,9 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map):
consistency issue, not a design choice. Same shape as the preview-variant gap
fixed in PR #25610.
Pricing matches the existing -preview entry one-for-one (input $0.25/M, output
$1.50/M, cache-read $0.025/M) — Google did not change costs at the GA cutover.
The exact prices are not pinned: scripts/sync_cost_map.py keeps openrouter/*
aligned with OpenRouter's catalog, and a pin would turn a reprice into a red
sync PR.
"""
model_name = "openrouter/google/gemini-3.1-flash-lite"
@ -3624,11 +3625,11 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map):
assert model_info is not None, f"Missing model pricing entry: {model_name}"
assert model_info["litellm_provider"] == "openrouter"
assert model_info["input_cost_per_token"] == 2.5e-07
assert model_info["output_cost_per_token"] == 1.5e-06
assert model_info["cache_read_input_token_cost"] == 2.5e-08
assert model_info["max_input_tokens"] == 1048576
assert model_info["max_output_tokens"] == 65536
assert model_info["input_cost_per_token"] > 0
assert model_info["output_cost_per_token"] > 0
assert model_info["cache_read_input_token_cost"] > 0
assert model_info["max_input_tokens"] > 0
assert model_info["max_output_tokens"] > 0
def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_map):

View file

@ -2796,9 +2796,8 @@ def test_model_info_for_openrouter_kimi_k2_5():
Test that openrouter/moonshotai/kimi-k2.5 model info is correctly configured
in model_prices_and_context_window.json.
Model properties from OpenRouter API:
- context_length: 262144
- pricing: prompt=$0.00000045, completion=$0.00000225, input_cache_read=$0.00000007
scripts/sync_cost_map.py keeps the limits and prices aligned with OpenRouter's
catalog, so they are checked for presence, not pinned.
- modality: text+image->text (supports vision)
- supports: tool_choice, tools (function calling)
"""
@ -2817,15 +2816,13 @@ def test_model_info_for_openrouter_kimi_k2_5():
assert model_info["litellm_provider"] == "openrouter"
assert model_info["mode"] == "chat"
# Verify context window
assert model_info["max_input_tokens"] == 262144
assert model_info["max_output_tokens"] == 262144
assert model_info["max_tokens"] == 262144
assert model_info["max_input_tokens"] > 0
assert model_info["max_output_tokens"] > 0
assert model_info["max_tokens"] == model_info["max_output_tokens"]
# Verify pricing
assert model_info["input_cost_per_token"] == 4.5e-07
assert model_info["output_cost_per_token"] == 2.25e-06
assert model_info["cache_read_input_token_cost"] == 7e-08
assert model_info["input_cost_per_token"] > 0
assert model_info["output_cost_per_token"] > 0
assert model_info["cache_read_input_token_cost"] > 0
# Verify capabilities
assert model_info["supports_vision"] is True