mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(cost-map-sync): keep the bot merging without a ruleset bypass and stop pinning the prices it owns
The reconcile only looks at bot PRs against the branch the run is on, so a dispatch from another branch never counts as the open sync PR. When a green bot PR cannot be merged because the app is not a bypass actor yet, the run arms auto-merge and warns instead of failing every tick. A base with no required checks falls back to every check so a dispatch there can still merge, and a tick whose catalogs and base match the last no-op sync skips the install and the script. guard-main-branch accepts litellm_cost_map_sync_* heads so the bot keeps working once main is the default branch again. The tests that pinned exact OpenRouter prices and limits now check that the entries exist and are priced: the sync owns those values, and a pin would turn every legitimate reprice into a red bot PR that pauses syncing.
This commit is contained in:
parent
26f74e662d
commit
340569b09a
5 changed files with 73 additions and 67 deletions
47
.github/workflows/cost-map-sync.yml
vendored
47
.github/workflows/cost-map-sync.yml
vendored
|
|
@ -45,7 +45,8 @@ jobs:
|
|||
pr=""
|
||||
if [ -n "$BOT_LOGIN" ]; then
|
||||
pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --author "app/$BOT_LOGIN" \
|
||||
--search "in:title \"$PR_TITLE\"" --json number,headRefName,mergeable,isCrossRepository \
|
||||
--base "$GITHUB_REF_NAME" --search "in:title \"$PR_TITLE\"" \
|
||||
--json number,headRefName,mergeable,isCrossRepository,autoMergeRequest \
|
||||
--jq "[.[] | select((.headRefName | startswith(\"$BRANCH_PREFIX\")) and (.isCrossRepository | not))] | first // empty")"
|
||||
fi
|
||||
sync=false
|
||||
|
|
@ -62,11 +63,21 @@ jobs:
|
|||
guard="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json name,state \
|
||||
--jq '.[] | select(.name == "cost-map-guard") | .state' || true)"
|
||||
required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --required --json bucket \
|
||||
--jq 'map(.bucket) | unique | join(",")' || true)"
|
||||
--jq 'map(.bucket) | unique | join(",")' 2>/dev/null || true)"
|
||||
if [ -z "$required" ]; then
|
||||
required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json bucket \
|
||||
--jq 'map(.bucket) | unique | join(",")' || true)"
|
||||
fi
|
||||
case "$guard,$required" in
|
||||
SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping)
|
||||
gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch
|
||||
echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry."
|
||||
if gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch; then
|
||||
echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry."
|
||||
else
|
||||
if [ "$(jq -r .autoMergeRequest <<< "$pr")" = "null" ]; then
|
||||
gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --auto --merge --delete-branch
|
||||
fi
|
||||
echo "::warning::Sync PR #$number is green but the app is not a bypass actor on $GITHUB_REF_NAME; auto-merge is armed, so one approval lands it."
|
||||
fi
|
||||
;;
|
||||
*FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*)
|
||||
echo "::warning::Sync PR #$number has a failing check (cost-map-guard=$guard, required buckets=$required); leaving it open for a human."
|
||||
|
|
@ -83,23 +94,47 @@ jobs:
|
|||
- name: Explain why no PR can be opened
|
||||
if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID == '' && !inputs.dry_run
|
||||
run: echo "::warning::COST_MAP_BOT_APP_ID is not configured, so no sync PR can be opened or merged; dispatch with dry_run to see the diff."
|
||||
- name: Hash the catalogs and the base
|
||||
id: catalogs
|
||||
if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run
|
||||
run: |
|
||||
digest="$(curl -fsSL https://openrouter.ai/api/v1/models https://ai-gateway.vercel.sh/v1/models | sha256sum | cut -c1-64)"
|
||||
echo "key=cost-map-sync-${GITHUB_SHA}-${digest}" >> "$GITHUB_OUTPUT"
|
||||
- name: Look up whether this catalog state was already synced
|
||||
id: seen
|
||||
if: steps.catalogs.outputs.key != ''
|
||||
uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: ${{ runner.temp }}/synced
|
||||
key: ${{ steps.catalogs.outputs.key }}
|
||||
lookup-only: true
|
||||
- name: Skip the unchanged catalogs
|
||||
if: steps.seen.outputs.cache-hit == 'true'
|
||||
run: echo "Neither catalog nor the base has changed since the last sync found nothing to do."
|
||||
- name: Set up uv
|
||||
if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run)
|
||||
if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true'))
|
||||
uses: ./.github/actions/setup-uv-with-retries
|
||||
with:
|
||||
version: "0.10.9"
|
||||
- name: Run the sync
|
||||
id: sync
|
||||
if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run)
|
||||
if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true'))
|
||||
run: |
|
||||
uv run --frozen --no-dev python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md"
|
||||
uv run --frozen --no-dev python ci_cd/generate_model_prices_schema.py
|
||||
if git diff --quiet; then
|
||||
echo "Registry already in sync; no PR needed."
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
echo "$GITHUB_SHA" > "$RUNNER_TEMP/synced"
|
||||
else
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
- name: Remember that this catalog state needs no PR
|
||||
if: steps.sync.outputs.changed == 'false' && steps.catalogs.outputs.key != ''
|
||||
uses: actions/cache/save@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
|
||||
with:
|
||||
path: ${{ runner.temp }}/synced
|
||||
key: ${{ steps.catalogs.outputs.key }}
|
||||
- name: Open the sync PR
|
||||
if: steps.sync.outputs.changed == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run
|
||||
run: |
|
||||
|
|
|
|||
4
.github/workflows/guard-main-branch.yml
vendored
4
.github/workflows/guard-main-branch.yml
vendored
|
|
@ -34,9 +34,9 @@ jobs:
|
|||
echo "::error::PRs to main must originate from the canonical repository ($BASE_REPO), not a fork ($HEAD_REPO). External contributors should open PRs against 'litellm_internal_staging' instead."
|
||||
exit 1
|
||||
fi
|
||||
if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]]; then
|
||||
if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]] || [[ "$HEAD_REF" == litellm_cost_map_sync_?* ]]; then
|
||||
echo "Allowed source branch."
|
||||
exit 0
|
||||
fi
|
||||
echo "::error::PRs to main must originate from 'litellm_internal_staging' or a 'litellm_hotfix_*' branch. Got: '$HEAD_REF'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead."
|
||||
echo "::error::PRs to main must originate from 'litellm_internal_staging', a 'litellm_hotfix_*' branch, or a 'litellm_cost_map_sync_*' bot branch. Got: '$HEAD_REF'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead."
|
||||
exit 1
|
||||
|
|
|
|||
|
|
@ -247,22 +247,6 @@ def test_azure_ai_claude_1m_context_entries(cost_map: dict):
|
|||
assert cost_map[model]["max_input_tokens"] == 200000, model
|
||||
|
||||
|
||||
# OpenRouter headline rates from GET https://openrouter.ai/api/v1/models.
|
||||
# These were the catalog values that disagreed with that API (and, for the
|
||||
# two spotlight models, the public model pages that their source fields cite).
|
||||
_OPENROUTER_LIVE_COSTS = {
|
||||
"openrouter/qwen/qwen3.5-plus-02-15": (2.6e-07, 1.56e-06, None),
|
||||
"openrouter/openai/gpt-oss-120b": (3.7e-08, 1.7e-07, None),
|
||||
"openrouter/qwen/qwen3-coder-plus": (6.5e-07, 3.25e-06, None),
|
||||
"openrouter/qwen/qwen3.5-flash-02-23": (6.5e-08, 2.6e-07, None),
|
||||
"openrouter/qwen/qwen3.5-27b": (1.95e-07, 1.56e-06, None),
|
||||
"openrouter/gryphe/mythomax-l2-13b": (6e-08, 6e-08, None),
|
||||
"openrouter/mancer/weaver": (4e-07, 7.5e-07, None),
|
||||
"openrouter/xiaomi/mimo-v2.5-pro": (4.35e-07, 8.7e-07, 3.6e-09),
|
||||
"openrouter/moonshotai/kimi-k2.5": (4.5e-07, 2.25e-06, 7e-08),
|
||||
"openrouter/z-ai/glm-5": (6e-07, 1.92e-06, None),
|
||||
}
|
||||
|
||||
_OPENROUTER_STALE_COSTS = {
|
||||
"openrouter/qwen/qwen3.5-plus-02-15": (4e-07, 2.4e-06),
|
||||
"openrouter/openai/gpt-oss-120b": (1.8e-07, 8e-07),
|
||||
|
|
@ -275,23 +259,12 @@ _OPENROUTER_STALE_COSTS = {
|
|||
[_load_root_cost_map(), GetModelCostMap.load_local_model_cost_map()],
|
||||
ids=["root", "bundled_backup"],
|
||||
)
|
||||
def test_openrouter_catalog_costs_match_live_headline_rates(cost_map: dict):
|
||||
"""openrouter/* spend tracking reads these catalog fields. The values must
|
||||
stay aligned with OpenRouter's published headline rate, not the stale
|
||||
figures that over/under-counted by up to 30x. Both maps are checked so
|
||||
the root file and bundled backup cannot drift apart."""
|
||||
control = cost_map["openrouter/anthropic/claude-opus-5"]
|
||||
assert control["input_cost_per_token"] == 5e-06
|
||||
assert control["output_cost_per_token"] == 2.5e-05
|
||||
assert control["cache_read_input_token_cost"] == 5e-07
|
||||
|
||||
for model, (inp, out, cache) in _OPENROUTER_LIVE_COSTS.items():
|
||||
entry = cost_map[model]
|
||||
assert entry["input_cost_per_token"] == inp, model
|
||||
assert entry["output_cost_per_token"] == out, model
|
||||
if cache is not None:
|
||||
assert entry["cache_read_input_token_cost"] == cache, model
|
||||
|
||||
def test_openrouter_stale_catalog_costs_never_return(cost_map: dict):
|
||||
"""scripts/sync_cost_map.py keeps openrouter/* aligned with OpenRouter's
|
||||
published headline rates, so the live values are not pinned here: a pin
|
||||
would turn every legitimate reprice into a red sync PR. The stale figures
|
||||
that over/under-counted by up to 30x must never come back, in the root
|
||||
file or the bundled backup."""
|
||||
for model, (stale_in, stale_out) in _OPENROUTER_STALE_COSTS.items():
|
||||
entry = cost_map[model]
|
||||
assert entry["input_cost_per_token"] != stale_in, model
|
||||
|
|
|
|||
|
|
@ -197,10 +197,10 @@ def test_openrouter_qwen36_plus_model_info(_local_model_cost_map):
|
|||
assert model_info is not None
|
||||
assert model_info["litellm_provider"] == "openrouter"
|
||||
assert model_info["mode"] == "chat"
|
||||
assert model_info["max_input_tokens"] == 1000000
|
||||
assert model_info["max_output_tokens"] == 65536
|
||||
assert model_info["input_cost_per_token"] == 3.25e-07
|
||||
assert model_info["output_cost_per_token"] == 1.95e-06
|
||||
assert model_info["max_input_tokens"] > 0
|
||||
assert model_info["max_output_tokens"] > 0
|
||||
assert model_info["input_cost_per_token"] > 0
|
||||
assert model_info["output_cost_per_token"] > 0
|
||||
assert model_info["supports_function_calling"] is True
|
||||
assert model_info["supports_tool_choice"] is True
|
||||
assert model_info["supports_reasoning"] is True
|
||||
|
|
@ -3273,10 +3273,10 @@ def test_openrouter_gemini_3_1_flash_lite_preview_pricing(_local_model_cost_map)
|
|||
|
||||
assert model_info is not None, f"Missing model pricing entry: {model_name}"
|
||||
assert model_info["litellm_provider"] == "openrouter"
|
||||
assert model_info["input_cost_per_token"] == 2.5e-07
|
||||
assert model_info["output_cost_per_token"] == 1.5e-06
|
||||
assert model_info["max_input_tokens"] == 1048576
|
||||
assert model_info["max_output_tokens"] == 65536
|
||||
assert model_info["input_cost_per_token"] > 0
|
||||
assert model_info["output_cost_per_token"] > 0
|
||||
assert model_info["max_input_tokens"] > 0
|
||||
assert model_info["max_output_tokens"] > 0
|
||||
|
||||
|
||||
def test_gemini_3_1_flash_lite_pricing(_local_model_cost_map):
|
||||
|
|
@ -3615,8 +3615,9 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map):
|
|||
consistency issue, not a design choice. Same shape as the preview-variant gap
|
||||
fixed in PR #25610.
|
||||
|
||||
Pricing matches the existing -preview entry one-for-one (input $0.25/M, output
|
||||
$1.50/M, cache-read $0.025/M) — Google did not change costs at the GA cutover.
|
||||
The exact prices are not pinned: scripts/sync_cost_map.py keeps openrouter/*
|
||||
aligned with OpenRouter's catalog, and a pin would turn a reprice into a red
|
||||
sync PR.
|
||||
"""
|
||||
|
||||
model_name = "openrouter/google/gemini-3.1-flash-lite"
|
||||
|
|
@ -3624,11 +3625,11 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map):
|
|||
|
||||
assert model_info is not None, f"Missing model pricing entry: {model_name}"
|
||||
assert model_info["litellm_provider"] == "openrouter"
|
||||
assert model_info["input_cost_per_token"] == 2.5e-07
|
||||
assert model_info["output_cost_per_token"] == 1.5e-06
|
||||
assert model_info["cache_read_input_token_cost"] == 2.5e-08
|
||||
assert model_info["max_input_tokens"] == 1048576
|
||||
assert model_info["max_output_tokens"] == 65536
|
||||
assert model_info["input_cost_per_token"] > 0
|
||||
assert model_info["output_cost_per_token"] > 0
|
||||
assert model_info["cache_read_input_token_cost"] > 0
|
||||
assert model_info["max_input_tokens"] > 0
|
||||
assert model_info["max_output_tokens"] > 0
|
||||
|
||||
|
||||
def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_map):
|
||||
|
|
|
|||
|
|
@ -2796,9 +2796,8 @@ def test_model_info_for_openrouter_kimi_k2_5():
|
|||
Test that openrouter/moonshotai/kimi-k2.5 model info is correctly configured
|
||||
in model_prices_and_context_window.json.
|
||||
|
||||
Model properties from OpenRouter API:
|
||||
- context_length: 262144
|
||||
- pricing: prompt=$0.00000045, completion=$0.00000225, input_cache_read=$0.00000007
|
||||
scripts/sync_cost_map.py keeps the limits and prices aligned with OpenRouter's
|
||||
catalog, so they are checked for presence, not pinned.
|
||||
- modality: text+image->text (supports vision)
|
||||
- supports: tool_choice, tools (function calling)
|
||||
"""
|
||||
|
|
@ -2817,15 +2816,13 @@ def test_model_info_for_openrouter_kimi_k2_5():
|
|||
assert model_info["litellm_provider"] == "openrouter"
|
||||
assert model_info["mode"] == "chat"
|
||||
|
||||
# Verify context window
|
||||
assert model_info["max_input_tokens"] == 262144
|
||||
assert model_info["max_output_tokens"] == 262144
|
||||
assert model_info["max_tokens"] == 262144
|
||||
assert model_info["max_input_tokens"] > 0
|
||||
assert model_info["max_output_tokens"] > 0
|
||||
assert model_info["max_tokens"] == model_info["max_output_tokens"]
|
||||
|
||||
# Verify pricing
|
||||
assert model_info["input_cost_per_token"] == 4.5e-07
|
||||
assert model_info["output_cost_per_token"] == 2.25e-06
|
||||
assert model_info["cache_read_input_token_cost"] == 7e-08
|
||||
assert model_info["input_cost_per_token"] > 0
|
||||
assert model_info["output_cost_per_token"] > 0
|
||||
assert model_info["cache_read_input_token_cost"] > 0
|
||||
|
||||
# Verify capabilities
|
||||
assert model_info["supports_vision"] is True
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue