diff --git a/.github/workflows/cost-map-sync.yml b/.github/workflows/cost-map-sync.yml index 8e7bf3f9699..2758a2082ea 100644 --- a/.github/workflows/cost-map-sync.yml +++ b/.github/workflows/cost-map-sync.yml @@ -45,7 +45,8 @@ jobs: pr="" if [ -n "$BOT_LOGIN" ]; then pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --author "app/$BOT_LOGIN" \ - --search "in:title \"$PR_TITLE\"" --json number,headRefName,mergeable,isCrossRepository \ + --base "$GITHUB_REF_NAME" --search "in:title \"$PR_TITLE\"" \ + --json number,headRefName,mergeable,isCrossRepository,autoMergeRequest \ --jq "[.[] | select((.headRefName | startswith(\"$BRANCH_PREFIX\")) and (.isCrossRepository | not))] | first // empty")" fi sync=false @@ -62,11 +63,21 @@ jobs: guard="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json name,state \ --jq '.[] | select(.name == "cost-map-guard") | .state' || true)" required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --required --json bucket \ - --jq 'map(.bucket) | unique | join(",")' || true)" + --jq 'map(.bucket) | unique | join(",")' 2>/dev/null || true)" + if [ -z "$required" ]; then + required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json bucket \ + --jq 'map(.bucket) | unique | join(",")' || true)" + fi case "$guard,$required" in SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping) - gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch - echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry." + if gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch; then + echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry." + else + if [ "$(jq -r .autoMergeRequest <<< "$pr")" = "null" ]; then + gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --auto --merge --delete-branch + fi + echo "::warning::Sync PR #$number is green but the app is not a bypass actor on $GITHUB_REF_NAME; auto-merge is armed, so one approval lands it." + fi ;; *FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*) echo "::warning::Sync PR #$number has a failing check (cost-map-guard=$guard, required buckets=$required); leaving it open for a human." @@ -83,23 +94,47 @@ jobs: - name: Explain why no PR can be opened if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID == '' && !inputs.dry_run run: echo "::warning::COST_MAP_BOT_APP_ID is not configured, so no sync PR can be opened or merged; dispatch with dry_run to see the diff." + - name: Hash the catalogs and the base + id: catalogs + if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run + run: | + digest="$(curl -fsSL https://openrouter.ai/api/v1/models https://ai-gateway.vercel.sh/v1/models | sha256sum | cut -c1-64)" + echo "key=cost-map-sync-${GITHUB_SHA}-${digest}" >> "$GITHUB_OUTPUT" + - name: Look up whether this catalog state was already synced + id: seen + if: steps.catalogs.outputs.key != '' + uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ${{ runner.temp }}/synced + key: ${{ steps.catalogs.outputs.key }} + lookup-only: true + - name: Skip the unchanged catalogs + if: steps.seen.outputs.cache-hit == 'true' + run: echo "Neither catalog nor the base has changed since the last sync found nothing to do." - name: Set up uv - if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run) + if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true')) uses: ./.github/actions/setup-uv-with-retries with: version: "0.10.9" - name: Run the sync id: sync - if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run) + if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true')) run: | uv run --frozen --no-dev python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md" uv run --frozen --no-dev python ci_cd/generate_model_prices_schema.py if git diff --quiet; then echo "Registry already in sync; no PR needed." echo "changed=false" >> "$GITHUB_OUTPUT" + echo "$GITHUB_SHA" > "$RUNNER_TEMP/synced" else echo "changed=true" >> "$GITHUB_OUTPUT" fi + - name: Remember that this catalog state needs no PR + if: steps.sync.outputs.changed == 'false' && steps.catalogs.outputs.key != '' + uses: actions/cache/save@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ${{ runner.temp }}/synced + key: ${{ steps.catalogs.outputs.key }} - name: Open the sync PR if: steps.sync.outputs.changed == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run run: | diff --git a/.github/workflows/guard-main-branch.yml b/.github/workflows/guard-main-branch.yml index 5bc561c6441..9deeea9d988 100644 --- a/.github/workflows/guard-main-branch.yml +++ b/.github/workflows/guard-main-branch.yml @@ -34,9 +34,9 @@ jobs: echo "::error::PRs to main must originate from the canonical repository ($BASE_REPO), not a fork ($HEAD_REPO). External contributors should open PRs against 'litellm_internal_staging' instead." exit 1 fi - if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]]; then + if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]] || [[ "$HEAD_REF" == litellm_cost_map_sync_?* ]]; then echo "Allowed source branch." exit 0 fi - echo "::error::PRs to main must originate from 'litellm_internal_staging' or a 'litellm_hotfix_*' branch. Got: '$HEAD_REF'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead." + echo "::error::PRs to main must originate from 'litellm_internal_staging', a 'litellm_hotfix_*' branch, or a 'litellm_cost_map_sync_*' bot branch. Got: '$HEAD_REF'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead." exit 1 diff --git a/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py b/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py index 18185126775..51401d2843d 100644 --- a/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py +++ b/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py @@ -247,22 +247,6 @@ def test_azure_ai_claude_1m_context_entries(cost_map: dict): assert cost_map[model]["max_input_tokens"] == 200000, model -# OpenRouter headline rates from GET https://openrouter.ai/api/v1/models. -# These were the catalog values that disagreed with that API (and, for the -# two spotlight models, the public model pages that their source fields cite). -_OPENROUTER_LIVE_COSTS = { - "openrouter/qwen/qwen3.5-plus-02-15": (2.6e-07, 1.56e-06, None), - "openrouter/openai/gpt-oss-120b": (3.7e-08, 1.7e-07, None), - "openrouter/qwen/qwen3-coder-plus": (6.5e-07, 3.25e-06, None), - "openrouter/qwen/qwen3.5-flash-02-23": (6.5e-08, 2.6e-07, None), - "openrouter/qwen/qwen3.5-27b": (1.95e-07, 1.56e-06, None), - "openrouter/gryphe/mythomax-l2-13b": (6e-08, 6e-08, None), - "openrouter/mancer/weaver": (4e-07, 7.5e-07, None), - "openrouter/xiaomi/mimo-v2.5-pro": (4.35e-07, 8.7e-07, 3.6e-09), - "openrouter/moonshotai/kimi-k2.5": (4.5e-07, 2.25e-06, 7e-08), - "openrouter/z-ai/glm-5": (6e-07, 1.92e-06, None), -} - _OPENROUTER_STALE_COSTS = { "openrouter/qwen/qwen3.5-plus-02-15": (4e-07, 2.4e-06), "openrouter/openai/gpt-oss-120b": (1.8e-07, 8e-07), @@ -275,23 +259,12 @@ _OPENROUTER_STALE_COSTS = { [_load_root_cost_map(), GetModelCostMap.load_local_model_cost_map()], ids=["root", "bundled_backup"], ) -def test_openrouter_catalog_costs_match_live_headline_rates(cost_map: dict): - """openrouter/* spend tracking reads these catalog fields. The values must - stay aligned with OpenRouter's published headline rate, not the stale - figures that over/under-counted by up to 30x. Both maps are checked so - the root file and bundled backup cannot drift apart.""" - control = cost_map["openrouter/anthropic/claude-opus-5"] - assert control["input_cost_per_token"] == 5e-06 - assert control["output_cost_per_token"] == 2.5e-05 - assert control["cache_read_input_token_cost"] == 5e-07 - - for model, (inp, out, cache) in _OPENROUTER_LIVE_COSTS.items(): - entry = cost_map[model] - assert entry["input_cost_per_token"] == inp, model - assert entry["output_cost_per_token"] == out, model - if cache is not None: - assert entry["cache_read_input_token_cost"] == cache, model - +def test_openrouter_stale_catalog_costs_never_return(cost_map: dict): + """scripts/sync_cost_map.py keeps openrouter/* aligned with OpenRouter's + published headline rates, so the live values are not pinned here: a pin + would turn every legitimate reprice into a red sync PR. The stale figures + that over/under-counted by up to 30x must never come back, in the root + file or the bundled backup.""" for model, (stale_in, stale_out) in _OPENROUTER_STALE_COSTS.items(): entry = cost_map[model] assert entry["input_cost_per_token"] != stale_in, model diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 2046695f151..3b46ca8d415 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -197,10 +197,10 @@ def test_openrouter_qwen36_plus_model_info(_local_model_cost_map): assert model_info is not None assert model_info["litellm_provider"] == "openrouter" assert model_info["mode"] == "chat" - assert model_info["max_input_tokens"] == 1000000 - assert model_info["max_output_tokens"] == 65536 - assert model_info["input_cost_per_token"] == 3.25e-07 - assert model_info["output_cost_per_token"] == 1.95e-06 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 assert model_info["supports_function_calling"] is True assert model_info["supports_tool_choice"] is True assert model_info["supports_reasoning"] is True @@ -3273,10 +3273,10 @@ def test_openrouter_gemini_3_1_flash_lite_preview_pricing(_local_model_cost_map) assert model_info is not None, f"Missing model pricing entry: {model_name}" assert model_info["litellm_provider"] == "openrouter" - assert model_info["input_cost_per_token"] == 2.5e-07 - assert model_info["output_cost_per_token"] == 1.5e-06 - assert model_info["max_input_tokens"] == 1048576 - assert model_info["max_output_tokens"] == 65536 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 def test_gemini_3_1_flash_lite_pricing(_local_model_cost_map): @@ -3615,8 +3615,9 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map): consistency issue, not a design choice. Same shape as the preview-variant gap fixed in PR #25610. - Pricing matches the existing -preview entry one-for-one (input $0.25/M, output - $1.50/M, cache-read $0.025/M) — Google did not change costs at the GA cutover. + The exact prices are not pinned: scripts/sync_cost_map.py keeps openrouter/* + aligned with OpenRouter's catalog, and a pin would turn a reprice into a red + sync PR. """ model_name = "openrouter/google/gemini-3.1-flash-lite" @@ -3624,11 +3625,11 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map): assert model_info is not None, f"Missing model pricing entry: {model_name}" assert model_info["litellm_provider"] == "openrouter" - assert model_info["input_cost_per_token"] == 2.5e-07 - assert model_info["output_cost_per_token"] == 1.5e-06 - assert model_info["cache_read_input_token_cost"] == 2.5e-08 - assert model_info["max_input_tokens"] == 1048576 - assert model_info["max_output_tokens"] == 65536 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 + assert model_info["cache_read_input_token_cost"] > 0 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_map): diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 98378d81d08..9d90d47e022 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -2796,9 +2796,8 @@ def test_model_info_for_openrouter_kimi_k2_5(): Test that openrouter/moonshotai/kimi-k2.5 model info is correctly configured in model_prices_and_context_window.json. - Model properties from OpenRouter API: - - context_length: 262144 - - pricing: prompt=$0.00000045, completion=$0.00000225, input_cache_read=$0.00000007 + scripts/sync_cost_map.py keeps the limits and prices aligned with OpenRouter's + catalog, so they are checked for presence, not pinned. - modality: text+image->text (supports vision) - supports: tool_choice, tools (function calling) """ @@ -2817,15 +2816,13 @@ def test_model_info_for_openrouter_kimi_k2_5(): assert model_info["litellm_provider"] == "openrouter" assert model_info["mode"] == "chat" - # Verify context window - assert model_info["max_input_tokens"] == 262144 - assert model_info["max_output_tokens"] == 262144 - assert model_info["max_tokens"] == 262144 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 + assert model_info["max_tokens"] == model_info["max_output_tokens"] - # Verify pricing - assert model_info["input_cost_per_token"] == 4.5e-07 - assert model_info["output_cost_per_token"] == 2.25e-06 - assert model_info["cache_read_input_token_cost"] == 7e-08 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 + assert model_info["cache_read_input_token_cost"] > 0 # Verify capabilities assert model_info["supports_vision"] is True