From 340569b09a2ba0fd76daff5e6e0fbec8175c0c97 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:11:12 -0700 Subject: [PATCH] fix(cost-map-sync): keep the bot merging without a ruleset bypass and stop pinning the prices it owns The reconcile only looks at bot PRs against the branch the run is on, so a dispatch from another branch never counts as the open sync PR. When a green bot PR cannot be merged because the app is not a bypass actor yet, the run arms auto-merge and warns instead of failing every tick. A base with no required checks falls back to every check so a dispatch there can still merge, and a tick whose catalogs and base match the last no-op sync skips the install and the script. guard-main-branch accepts litellm_cost_map_sync_* heads so the bot keeps working once main is the default branch again. The tests that pinned exact OpenRouter prices and limits now check that the entries exist and are priced: the sync owns those values, and a pin would turn every legitimate reprice into a red bot PR that pauses syncing. --- .github/workflows/cost-map-sync.yml | 47 ++++++++++++++++--- .github/workflows/guard-main-branch.yml | 4 +- .../test_get_model_cost_map.py | 39 +++------------ tests/test_litellm/test_cost_calculator.py | 31 ++++++------ tests/test_litellm/test_utils.py | 19 ++++---- 5 files changed, 73 insertions(+), 67 deletions(-) diff --git a/.github/workflows/cost-map-sync.yml b/.github/workflows/cost-map-sync.yml index 8e7bf3f9699..2758a2082ea 100644 --- a/.github/workflows/cost-map-sync.yml +++ b/.github/workflows/cost-map-sync.yml @@ -45,7 +45,8 @@ jobs: pr="" if [ -n "$BOT_LOGIN" ]; then pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --author "app/$BOT_LOGIN" \ - --search "in:title \"$PR_TITLE\"" --json number,headRefName,mergeable,isCrossRepository \ + --base "$GITHUB_REF_NAME" --search "in:title \"$PR_TITLE\"" \ + --json number,headRefName,mergeable,isCrossRepository,autoMergeRequest \ --jq "[.[] | select((.headRefName | startswith(\"$BRANCH_PREFIX\")) and (.isCrossRepository | not))] | first // empty")" fi sync=false @@ -62,11 +63,21 @@ jobs: guard="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json name,state \ --jq '.[] | select(.name == "cost-map-guard") | .state' || true)" required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --required --json bucket \ - --jq 'map(.bucket) | unique | join(",")' || true)" + --jq 'map(.bucket) | unique | join(",")' 2>/dev/null || true)" + if [ -z "$required" ]; then + required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json bucket \ + --jq 'map(.bucket) | unique | join(",")' || true)" + fi case "$guard,$required" in SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping) - gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch - echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry." + if gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch; then + echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry." + else + if [ "$(jq -r .autoMergeRequest <<< "$pr")" = "null" ]; then + gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --auto --merge --delete-branch + fi + echo "::warning::Sync PR #$number is green but the app is not a bypass actor on $GITHUB_REF_NAME; auto-merge is armed, so one approval lands it." + fi ;; *FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*) echo "::warning::Sync PR #$number has a failing check (cost-map-guard=$guard, required buckets=$required); leaving it open for a human." @@ -83,23 +94,47 @@ jobs: - name: Explain why no PR can be opened if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID == '' && !inputs.dry_run run: echo "::warning::COST_MAP_BOT_APP_ID is not configured, so no sync PR can be opened or merged; dispatch with dry_run to see the diff." + - name: Hash the catalogs and the base + id: catalogs + if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run + run: | + digest="$(curl -fsSL https://openrouter.ai/api/v1/models https://ai-gateway.vercel.sh/v1/models | sha256sum | cut -c1-64)" + echo "key=cost-map-sync-${GITHUB_SHA}-${digest}" >> "$GITHUB_OUTPUT" + - name: Look up whether this catalog state was already synced + id: seen + if: steps.catalogs.outputs.key != '' + uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ${{ runner.temp }}/synced + key: ${{ steps.catalogs.outputs.key }} + lookup-only: true + - name: Skip the unchanged catalogs + if: steps.seen.outputs.cache-hit == 'true' + run: echo "Neither catalog nor the base has changed since the last sync found nothing to do." - name: Set up uv - if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run) + if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true')) uses: ./.github/actions/setup-uv-with-retries with: version: "0.10.9" - name: Run the sync id: sync - if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run) + if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true')) run: | uv run --frozen --no-dev python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md" uv run --frozen --no-dev python ci_cd/generate_model_prices_schema.py if git diff --quiet; then echo "Registry already in sync; no PR needed." echo "changed=false" >> "$GITHUB_OUTPUT" + echo "$GITHUB_SHA" > "$RUNNER_TEMP/synced" else echo "changed=true" >> "$GITHUB_OUTPUT" fi + - name: Remember that this catalog state needs no PR + if: steps.sync.outputs.changed == 'false' && steps.catalogs.outputs.key != '' + uses: actions/cache/save@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ${{ runner.temp }}/synced + key: ${{ steps.catalogs.outputs.key }} - name: Open the sync PR if: steps.sync.outputs.changed == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run run: | diff --git a/.github/workflows/guard-main-branch.yml b/.github/workflows/guard-main-branch.yml index 5bc561c6441..9deeea9d988 100644 --- a/.github/workflows/guard-main-branch.yml +++ b/.github/workflows/guard-main-branch.yml @@ -34,9 +34,9 @@ jobs: echo "::error::PRs to main must originate from the canonical repository ($BASE_REPO), not a fork ($HEAD_REPO). External contributors should open PRs against 'litellm_internal_staging' instead." exit 1 fi - if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]]; then + if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]] || [[ "$HEAD_REF" == litellm_cost_map_sync_?* ]]; then echo "Allowed source branch." exit 0 fi - echo "::error::PRs to main must originate from 'litellm_internal_staging' or a 'litellm_hotfix_*' branch. Got: '$HEAD_REF'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead." + echo "::error::PRs to main must originate from 'litellm_internal_staging', a 'litellm_hotfix_*' branch, or a 'litellm_cost_map_sync_*' bot branch. Got: '$HEAD_REF'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead." exit 1 diff --git a/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py b/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py index 18185126775..51401d2843d 100644 --- a/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py +++ b/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py @@ -247,22 +247,6 @@ def test_azure_ai_claude_1m_context_entries(cost_map: dict): assert cost_map[model]["max_input_tokens"] == 200000, model -# OpenRouter headline rates from GET https://openrouter.ai/api/v1/models. -# These were the catalog values that disagreed with that API (and, for the -# two spotlight models, the public model pages that their source fields cite). -_OPENROUTER_LIVE_COSTS = { - "openrouter/qwen/qwen3.5-plus-02-15": (2.6e-07, 1.56e-06, None), - "openrouter/openai/gpt-oss-120b": (3.7e-08, 1.7e-07, None), - "openrouter/qwen/qwen3-coder-plus": (6.5e-07, 3.25e-06, None), - "openrouter/qwen/qwen3.5-flash-02-23": (6.5e-08, 2.6e-07, None), - "openrouter/qwen/qwen3.5-27b": (1.95e-07, 1.56e-06, None), - "openrouter/gryphe/mythomax-l2-13b": (6e-08, 6e-08, None), - "openrouter/mancer/weaver": (4e-07, 7.5e-07, None), - "openrouter/xiaomi/mimo-v2.5-pro": (4.35e-07, 8.7e-07, 3.6e-09), - "openrouter/moonshotai/kimi-k2.5": (4.5e-07, 2.25e-06, 7e-08), - "openrouter/z-ai/glm-5": (6e-07, 1.92e-06, None), -} - _OPENROUTER_STALE_COSTS = { "openrouter/qwen/qwen3.5-plus-02-15": (4e-07, 2.4e-06), "openrouter/openai/gpt-oss-120b": (1.8e-07, 8e-07), @@ -275,23 +259,12 @@ _OPENROUTER_STALE_COSTS = { [_load_root_cost_map(), GetModelCostMap.load_local_model_cost_map()], ids=["root", "bundled_backup"], ) -def test_openrouter_catalog_costs_match_live_headline_rates(cost_map: dict): - """openrouter/* spend tracking reads these catalog fields. The values must - stay aligned with OpenRouter's published headline rate, not the stale - figures that over/under-counted by up to 30x. Both maps are checked so - the root file and bundled backup cannot drift apart.""" - control = cost_map["openrouter/anthropic/claude-opus-5"] - assert control["input_cost_per_token"] == 5e-06 - assert control["output_cost_per_token"] == 2.5e-05 - assert control["cache_read_input_token_cost"] == 5e-07 - - for model, (inp, out, cache) in _OPENROUTER_LIVE_COSTS.items(): - entry = cost_map[model] - assert entry["input_cost_per_token"] == inp, model - assert entry["output_cost_per_token"] == out, model - if cache is not None: - assert entry["cache_read_input_token_cost"] == cache, model - +def test_openrouter_stale_catalog_costs_never_return(cost_map: dict): + """scripts/sync_cost_map.py keeps openrouter/* aligned with OpenRouter's + published headline rates, so the live values are not pinned here: a pin + would turn every legitimate reprice into a red sync PR. The stale figures + that over/under-counted by up to 30x must never come back, in the root + file or the bundled backup.""" for model, (stale_in, stale_out) in _OPENROUTER_STALE_COSTS.items(): entry = cost_map[model] assert entry["input_cost_per_token"] != stale_in, model diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 2046695f151..3b46ca8d415 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -197,10 +197,10 @@ def test_openrouter_qwen36_plus_model_info(_local_model_cost_map): assert model_info is not None assert model_info["litellm_provider"] == "openrouter" assert model_info["mode"] == "chat" - assert model_info["max_input_tokens"] == 1000000 - assert model_info["max_output_tokens"] == 65536 - assert model_info["input_cost_per_token"] == 3.25e-07 - assert model_info["output_cost_per_token"] == 1.95e-06 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 assert model_info["supports_function_calling"] is True assert model_info["supports_tool_choice"] is True assert model_info["supports_reasoning"] is True @@ -3273,10 +3273,10 @@ def test_openrouter_gemini_3_1_flash_lite_preview_pricing(_local_model_cost_map) assert model_info is not None, f"Missing model pricing entry: {model_name}" assert model_info["litellm_provider"] == "openrouter" - assert model_info["input_cost_per_token"] == 2.5e-07 - assert model_info["output_cost_per_token"] == 1.5e-06 - assert model_info["max_input_tokens"] == 1048576 - assert model_info["max_output_tokens"] == 65536 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 def test_gemini_3_1_flash_lite_pricing(_local_model_cost_map): @@ -3615,8 +3615,9 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map): consistency issue, not a design choice. Same shape as the preview-variant gap fixed in PR #25610. - Pricing matches the existing -preview entry one-for-one (input $0.25/M, output - $1.50/M, cache-read $0.025/M) — Google did not change costs at the GA cutover. + The exact prices are not pinned: scripts/sync_cost_map.py keeps openrouter/* + aligned with OpenRouter's catalog, and a pin would turn a reprice into a red + sync PR. """ model_name = "openrouter/google/gemini-3.1-flash-lite" @@ -3624,11 +3625,11 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map): assert model_info is not None, f"Missing model pricing entry: {model_name}" assert model_info["litellm_provider"] == "openrouter" - assert model_info["input_cost_per_token"] == 2.5e-07 - assert model_info["output_cost_per_token"] == 1.5e-06 - assert model_info["cache_read_input_token_cost"] == 2.5e-08 - assert model_info["max_input_tokens"] == 1048576 - assert model_info["max_output_tokens"] == 65536 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 + assert model_info["cache_read_input_token_cost"] > 0 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_map): diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 98378d81d08..9d90d47e022 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -2796,9 +2796,8 @@ def test_model_info_for_openrouter_kimi_k2_5(): Test that openrouter/moonshotai/kimi-k2.5 model info is correctly configured in model_prices_and_context_window.json. - Model properties from OpenRouter API: - - context_length: 262144 - - pricing: prompt=$0.00000045, completion=$0.00000225, input_cache_read=$0.00000007 + scripts/sync_cost_map.py keeps the limits and prices aligned with OpenRouter's + catalog, so they are checked for presence, not pinned. - modality: text+image->text (supports vision) - supports: tool_choice, tools (function calling) """ @@ -2817,15 +2816,13 @@ def test_model_info_for_openrouter_kimi_k2_5(): assert model_info["litellm_provider"] == "openrouter" assert model_info["mode"] == "chat" - # Verify context window - assert model_info["max_input_tokens"] == 262144 - assert model_info["max_output_tokens"] == 262144 - assert model_info["max_tokens"] == 262144 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 + assert model_info["max_tokens"] == model_info["max_output_tokens"] - # Verify pricing - assert model_info["input_cost_per_token"] == 4.5e-07 - assert model_info["output_cost_per_token"] == 2.25e-06 - assert model_info["cache_read_input_token_cost"] == 7e-08 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 + assert model_info["cache_read_input_token_cost"] > 0 # Verify capabilities assert model_info["supports_vision"] is True