diff --git a/.github/scripts/auto_update_price_and_context_window_file.py b/.github/scripts/auto_update_price_and_context_window_file.py deleted file mode 100644 index 461d8d347d9..00000000000 --- a/.github/scripts/auto_update_price_and_context_window_file.py +++ /dev/null @@ -1,159 +0,0 @@ -import asyncio -import aiohttp -import json - -# Asynchronously fetch data from a given URL -async def fetch_data(url): - try: - # Create an asynchronous session - async with aiohttp.ClientSession() as session: - # Send a GET request to the URL - async with session.get(url) as resp: - # Raise an error if the response status is not OK - resp.raise_for_status() - # Parse the response JSON - resp_json = await resp.json() - print("Fetch the data from URL.") - # Return the 'data' field from the JSON response - return resp_json['data'] - except Exception as e: - # Print an error message if fetching data fails - print("Error fetching data from URL:", e) - return None - -# Synchronize local data with remote data -def sync_local_data_with_remote(local_data, remote_data): - # Update existing keys in local_data with values from remote_data - for key in (set(local_data) & set(remote_data)): - local_data[key].update(remote_data[key]) - - # Add new keys from remote_data to local_data - for key in (set(remote_data) - set(local_data)): - local_data[key] = remote_data[key] - -# Write data to the json file -def write_to_file(file_path, data): - try: - # Open the file in write mode - with open(file_path, "w") as file: - # Dump the data as JSON into the file - json.dump(data, file, indent=4) - print("Values updated successfully.") - except Exception as e: - # Print an error message if writing to file fails - print("Error updating JSON file:", e) - -# Update the existing models and add the missing models for OpenRouter -def transform_openrouter_data(data): - transformed = {} - for row in data: - # Add the fields 'max_tokens' and 'input_cost_per_token' - obj = { - "max_tokens": row["context_length"], - "input_cost_per_token": float(row["pricing"]["prompt"]), - } - - # Add 'max_output_tokens' as a field if it is not None - if "top_provider" in row and "max_completion_tokens" in row["top_provider"] and row["top_provider"]["max_completion_tokens"] is not None: - obj['max_output_tokens'] = int(row["top_provider"]["max_completion_tokens"]) - - # Add the field 'output_cost_per_token' - obj.update({ - "output_cost_per_token": float(row["pricing"]["completion"]), - }) - - # Add field 'input_cost_per_image' if it exists and is non-zero - if "pricing" in row and "image" in row["pricing"] and float(row["pricing"]["image"]) != 0.0: - obj['input_cost_per_image'] = float(row["pricing"]["image"]) - - # Add the fields 'litellm_provider' and 'mode' - obj.update({ - "litellm_provider": "openrouter", - "mode": "chat" - }) - - # Add the 'supports_vision' field if the modality is 'multimodal' - if row.get('architecture', {}).get('modality') == 'multimodal': - obj['supports_vision'] = True - - # Use a composite key to store the transformed object - transformed[f'openrouter/{row["id"]}'] = obj - - return transformed - -# Update the existing models and add the missing models for Vercel AI Gateway -def transform_vercel_ai_gateway_data(data): - transformed = {} - for row in data: - obj = { - "max_tokens": row["context_window"], - "input_cost_per_token": float(row["pricing"]["input"]), - "output_cost_per_token": float(row["pricing"]["output"]), - 'max_output_tokens': row['max_tokens'], - 'max_input_tokens': row["context_window"], - } - - # Handle cache pricing if available - if "pricing" in row: - if "input_cache_read" in row["pricing"] and row["pricing"]["input_cache_read"] is not None: - obj['cache_read_input_token_cost'] = float(f"{float(row['pricing']['input_cache_read']):e}") - - if "input_cache_write" in row["pricing"] and row["pricing"]["input_cache_write"] is not None: - obj['cache_creation_input_token_cost'] = float(f"{float(row['pricing']['input_cache_write']):e}") - - mode = "embedding" if "embedding" in row["id"].lower() else "chat" - - obj.update({"litellm_provider": "vercel_ai_gateway", "mode": mode}) - - transformed[f'vercel_ai_gateway/{row["id"]}'] = obj - - return transformed - - -# Load local data from a specified file -def load_local_data(file_path): - try: - # Open the file in read mode - with open(file_path, "r") as file: - # Load and return the JSON data - return json.load(file) - except FileNotFoundError: - # Print an error message if the file is not found - print("File not found:", file_path) - return None - except json.JSONDecodeError as e: - # Print an error message if JSON decoding fails - print("Error decoding JSON:", e) - return None - -def main(): - local_file_path = "model_prices_and_context_window.json" # Path to the local data file - openrouter_url = "https://openrouter.ai/api/v1/models" # URL to fetch OpenRouter data - vercel_ai_gateway_url = "https://ai-gateway.vercel.sh/v1/models" # URL to fetch Vercel AI Gateway data - - # Load local data from file - local_data = load_local_data(local_file_path) - - # Fetch OpenRouter data - openrouter_data = asyncio.run(fetch_data(openrouter_url)) - # Transform the fetched OpenRouter data - openrouter_data = transform_openrouter_data(openrouter_data) - - # Fetch Vercel AI Gateway data - vercel_data = asyncio.run(fetch_data(vercel_ai_gateway_url)) - # Transform the fetched Vercel AI Gateway data - vercel_data = transform_vercel_ai_gateway_data(vercel_data) - - # Combine both datasets - all_remote_data = {**openrouter_data, **vercel_data} - - # If both local and openrouter data are available, synchronize and save - if local_data and all_remote_data: - sync_local_data_with_remote(local_data, all_remote_data) - write_to_file(local_file_path, local_data) - else: - print("Failed to fetch model data from either local file or URL.") - -# Entry point of the script -if __name__ == "__main__": - main() diff --git a/.github/workflows/auto_update_price_and_context_window.yml b/.github/workflows/auto_update_price_and_context_window.yml deleted file mode 100644 index 7e40a860ee9..00000000000 --- a/.github/workflows/auto_update_price_and_context_window.yml +++ /dev/null @@ -1,39 +0,0 @@ -name: Updates model_prices_and_context_window.json and Create Pull Request - -on: - schedule: - - cron: "0 0 * * 0" # Run every Sundays at midnight - #- cron: "0 0 * * *" # Run daily at midnight - -permissions: - contents: write - pull-requests: write - -jobs: - auto_update_price_and_context_window: - if: github.repository == 'BerriAI/litellm' - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - name: Set up uv - uses: ./.github/actions/setup-uv-with-retries - with: - version: "0.10.9" - - name: Update JSON Data - run: | - uv run --frozen --with 'aiohttp==3.13.3' python ".github/scripts/auto_update_price_and_context_window_file.py" - - name: Regenerate JSON Schema - run: | - uv run --frozen python ci_cd/generate_model_prices_schema.py - - name: Create Pull Request - run: | - git add model_prices_and_context_window.json model_prices_and_context_window.schema.json - git commit -m "Update model_prices_and_context_window.json file: $(date +'%Y-%m-%d')" - gh pr create --title "Update model_prices_and_context_window.json file" \ - --body "Automated update for model_prices_and_context_window.json" \ - --head auto-update-price-and-context-window-$(date +'%Y-%m-%d') \ - --base main - env: - GH_TOKEN: ${{ secrets.GH_TOKEN }} diff --git a/.github/workflows/cost-map-sync.yml b/.github/workflows/cost-map-sync.yml new file mode 100644 index 00000000000..716c5ae9204 --- /dev/null +++ b/.github/workflows/cost-map-sync.yml @@ -0,0 +1,165 @@ +name: Cost map sync + +on: + schedule: + - cron: "*/5 * * * *" + workflow_dispatch: + inputs: + dry_run: + description: "Print the diff without opening a PR" + type: boolean + default: false + +permissions: + contents: read + pull-requests: read + +concurrency: + group: cost-map-sync + cancel-in-progress: false + +env: + BRANCH_PREFIX: litellm_cost_map_sync_ + PR_TITLE: "feat(models): sync openrouter and vercel_ai_gateway pricing" + +jobs: + cost-map-sync: + if: github.repository == 'BerriAI/litellm' + runs-on: ubuntu-latest + env: + BOT_APP_ID: ${{ secrets.COST_MAP_BOT_APP_ID }} + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + - name: Mint the bot token + id: bot + if: env.BOT_APP_ID != '' + uses: actions/create-github-app-token@a8d616148505b5069dccd32f177bb87d7f39123b # v2.1.1 + with: + app-id: ${{ secrets.COST_MAP_BOT_APP_ID }} + private-key: ${{ secrets.COST_MAP_BOT_PRIVATE_KEY }} + - name: Reconcile the open sync PR + id: open + run: | + pr="" + if [ -n "$BOT_LOGIN" ]; then + pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --author "app/$BOT_LOGIN" \ + --base "$GITHUB_REF_NAME" --search "in:title \"$PR_TITLE\"" \ + --json number,headRefName,mergeable,isCrossRepository,autoMergeRequest,createdAt \ + --jq "[.[] | select((.headRefName | startswith(\"$BRANCH_PREFIX\")) and (.isCrossRepository | not))] | first // empty")" + fi + sync=false + if [ -z "$pr" ]; then + sync=true + elif [ "$(jq -r .mergeable <<< "$pr")" = "CONFLICTING" ]; then + number="$(jq -r .number <<< "$pr")" + gh pr close "$number" --repo "$GITHUB_REPOSITORY" --delete-branch \ + --comment "This sync no longer merges cleanly against $GITHUB_REF_NAME, so the next scheduled run opens a fresh one." + echo "Closed conflicting sync PR #$number." + sync=true + else + number="$(jq -r .number <<< "$pr")" + guard="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json name,state \ + --jq '.[] | select(.name == "cost-map-guard") | .state' || true)" + required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --required --json bucket \ + --jq 'map(.bucket) | unique | join(",")' 2>/dev/null || true)" + if [ -z "$required" ]; then + required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json bucket \ + --jq 'map(.bucket) | unique | join(",")' || true)" + fi + merged=false + case "$guard,$required" in + SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping) + if gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch; then + merged=true + echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry." + elif [ "$(jq -r .autoMergeRequest <<< "$pr")" = "null" ] \ + && ! gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --auto --merge --delete-branch; then + echo "::warning::Sync PR #$number is green but the app can neither merge it nor arm auto-merge on $GITHUB_REF_NAME; a human has to merge it." + else + echo "::warning::Sync PR #$number is green but the app is not a bypass actor on $GITHUB_REF_NAME; auto-merge is armed, so one approval lands it." + fi + ;; + *FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*) + echo "::warning::Sync PR #$number has a failing check (cost-map-guard=$guard, required buckets=$required); leaving it open for a human." + ;; + *) + echo "Sync PR #$number is still being checked (cost-map-guard=$guard, required buckets=$required)." + ;; + esac + if [ "$merged" = false ] && [ "$(( $(date +%s) - $(date -d "$(jq -r .createdAt <<< "$pr")" +%s) ))" -gt 86400 ]; then + echo "::error::Sync PR #$number has been open for more than a day, so syncing is stalled until someone merges or closes it." + exit 1 + fi + fi + echo "sync=$sync" >> "$GITHUB_OUTPUT" + env: + GH_TOKEN: ${{ steps.bot.outputs.token || github.token }} + BOT_LOGIN: ${{ steps.bot.outputs.app-slug }} + - name: Explain why no PR can be opened + if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID == '' && !inputs.dry_run + run: echo "::warning::COST_MAP_BOT_APP_ID is not configured, so no sync PR can be opened or merged; dispatch with dry_run to see the diff." + - name: Hash the catalogs and the sync inputs + id: catalogs + if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run + run: | + inputs="$(git rev-parse HEAD:model_prices_and_context_window.json HEAD:model_prices_and_context_window.schema.json HEAD:scripts/sync_cost_map.py HEAD:ci_cd/generate_model_prices_schema.py)" + digest="$({ echo "$inputs"; curl -fsSL https://openrouter.ai/api/v1/models https://ai-gateway.vercel.sh/v1/models; } | sha256sum | cut -c1-64)" + echo "key=cost-map-sync-${digest}" >> "$GITHUB_OUTPUT" + - name: Look up whether this catalog state was already synced + id: seen + if: steps.catalogs.outputs.key != '' + uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ${{ runner.temp }}/synced + key: ${{ steps.catalogs.outputs.key }} + lookup-only: true + - name: Skip the unchanged catalogs + if: steps.seen.outputs.cache-hit == 'true' + run: echo "Neither catalog, the map, its schema, nor the sync script has changed since the last sync found nothing to do." + - name: Set up uv + if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true')) + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + - name: Run the sync + id: sync + if: steps.open.outputs.sync == 'true' && (inputs.dry_run || (env.BOT_APP_ID != '' && steps.seen.outputs.cache-hit != 'true')) + run: | + uv run --frozen --no-dev python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md" + uv run --frozen --no-dev python ci_cd/generate_model_prices_schema.py + if git diff --quiet; then + echo "Registry already in sync; no PR needed." + echo "changed=false" >> "$GITHUB_OUTPUT" + echo "$GITHUB_SHA" > "$RUNNER_TEMP/synced" + else + echo "changed=true" >> "$GITHUB_OUTPUT" + fi + - name: Remember that this catalog state needs no PR + if: steps.sync.outputs.changed == 'false' && steps.catalogs.outputs.key != '' + uses: actions/cache/save@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 + with: + path: ${{ runner.temp }}/synced + key: ${{ steps.catalogs.outputs.key }} + - name: Open the sync PR + if: steps.sync.outputs.changed == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run + run: | + branch="${BRANCH_PREFIX}$(date -u +'%Y-%m-%d-%H%M')" + bot_user_id="$(gh api "users/${BOT_LOGIN}[bot]" --jq .id)" + git config user.name "${BOT_LOGIN}[bot]" + git config user.email "${bot_user_id}+${BOT_LOGIN}[bot]@users.noreply.github.com" + git checkout -b "$branch" + git add model_prices_and_context_window.json \ + litellm/model_prices_and_context_window_backup.json \ + model_prices_and_context_window.schema.json + git commit -m "feat(models): sync openrouter and vercel_ai_gateway pricing $(date -u +'%Y-%m-%d %H:%M')" + gh auth setup-git + git push origin "$branch" + gh pr create --title "$PR_TITLE" \ + --body-file "$RUNNER_TEMP/pr_body.md" \ + --head "$branch" \ + --base "$GITHUB_REF_NAME" + env: + GH_TOKEN: ${{ steps.bot.outputs.token }} + BOT_LOGIN: ${{ steps.bot.outputs.app-slug }} diff --git a/.github/workflows/guard-main-branch.yml b/.github/workflows/guard-main-branch.yml index 5bc561c6441..47cd2522df4 100644 --- a/.github/workflows/guard-main-branch.yml +++ b/.github/workflows/guard-main-branch.yml @@ -27,6 +27,8 @@ jobs: HEAD_REF: ${{ github.head_ref }} HEAD_REPO: ${{ github.event.pull_request.head.repo.full_name }} BASE_REPO: ${{ github.repository }} + HEAD_AUTHOR: ${{ github.event.pull_request.user.login }} + SYNC_APP_SLUG: ${{ vars.COST_MAP_BOT_APP_SLUG }} run: | echo "PR head repo: $HEAD_REPO" echo "PR head branch: $HEAD_REF" @@ -34,9 +36,9 @@ jobs: echo "::error::PRs to main must originate from the canonical repository ($BASE_REPO), not a fork ($HEAD_REPO). External contributors should open PRs against 'litellm_internal_staging' instead." exit 1 fi - if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]]; then + if [ "$HEAD_REF" = "litellm_internal_staging" ] || [[ "$HEAD_REF" == litellm_hotfix_?* ]] || { [[ "$HEAD_REF" == litellm_cost_map_sync_?* ]] && [ "$HEAD_AUTHOR" = "${SYNC_APP_SLUG}[bot]" ]; }; then echo "Allowed source branch." exit 0 fi - echo "::error::PRs to main must originate from 'litellm_internal_staging' or a 'litellm_hotfix_*' branch. Got: '$HEAD_REF'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead." + echo "::error::PRs to main must originate from 'litellm_internal_staging', a 'litellm_hotfix_*' branch, or a 'litellm_cost_map_sync_*' branch opened by the cost map sync app ('${SYNC_APP_SLUG:-}[bot]', from the COST_MAP_BOT_APP_SLUG repository variable). Got: '$HEAD_REF' by '$HEAD_AUTHOR'. If this is a contribution, retarget the PR against 'litellm_internal_staging' instead." exit 1 diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index ab29b70bdd4..e363ae1b020 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -28,6 +28,7 @@ EXTRA_BOOLEAN_KEYS = frozenset( "use_openai_responses_path", "bedrock_converse_supports_strict_tools", "thinking_always_on", + "price_varies_by_provider", } ) diff --git a/litellm/router_utils/reasoning_effort_capability.py b/litellm/router_utils/reasoning_effort_capability.py index 7b145c15a07..96ea845e3b4 100644 --- a/litellm/router_utils/reasoning_effort_capability.py +++ b/litellm/router_utils/reasoning_effort_capability.py @@ -77,7 +77,7 @@ def declared_reasoning_efforts(model_info: Mapping[str, object]) -> tuple[str, . """The entry's own answer, read through the same bare twin as the flags so both spellings of one model agree. Present-and-a-list IS the answer, so a declared [] correctly empties the group and an unknown level is dropped rather than raised: the bundled map is enum-validated by - validate-model-prices-json, but an operator can put this key on a config.yaml model_info block + cost-map-guard, but an operator can put this key on a config.yaml model_info block where that schema never runs, and one mistyped level must not fail every sibling on the proxy.""" own: Final = model_info.get(_DECLARED_EFFORTS_KEY) raw: Final = own if own is not None else _bare_model_entry(model_info).get(_DECLARED_EFFORTS_KEY) diff --git a/scripts/sync_cost_map.py b/scripts/sync_cost_map.py new file mode 100644 index 00000000000..bda89921a0a --- /dev/null +++ b/scripts/sync_cost_map.py @@ -0,0 +1,730 @@ +"""Sync the openrouter and vercel_ai_gateway entries of model_prices_and_context_window.json with the live catalogs. + +Pulls ``GET https://openrouter.ai/api/v1/models`` and ``GET https://ai-gateway.vercel.sh/v1/models``, maps the +catalog fields onto registry fields, and diffs the result against the registry. Dry run (the default) prints the +diff summary and the generated PR body; ``--write`` applies the changes to the root cost map and its ``litellm/`` +backup copy. + +Policy: +- Both catalogs price per token as decimal strings; values are normalized to six significant digits. +- Vercel long-context tiers map to the registry's ``*_above_k_tokens`` keys, which litellm applies once the + prompt exceeds N thousand tokens. A row whose tier boundaries are not whole thousands is skipped with a warning. +- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry, marked + ``price_varies_by_provider``, and keeps a marked entry in sync, but never overwrites the price of an entry without + the mark; that difference is reported as a warning. A human who curates one provider's own price drops the mark. +- Image and audio output are priced from the catalog's per-token ``image_output`` and ``audio_output`` prices. A row + whose non-text output the catalog does not price per token is skipped. +- A new entry inherits the traits no catalog expresses (adaptive thinking, sampling params, cache minimums, system + messages) from the same model's root registry entry, found by the bare model name in any mode or else its + longest dash-prefix with the same mode, so the family-wide invariants the test suite enforces hold for the route. +- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed and a + capability flag the catalog does not claim stays as curated. ``max_output_tokens`` and ``max_tokens`` move as a + pair and only when the catalog states an output ceiling; a curated output cap equal to the entry's own context + window is a copy of that window, not a ceiling, so the catalog's ceiling replaces it unless that would shrink it + more than 10x. +- A limit that would shrink, a price that would cross zero, a price that would move more than 10x either way, and + a capability flag curated as false are held back as warnings for a human instead of applied. +- Router models and rows without a usable prompt and completion price are skipped. +- A registry entry absent from its catalog is left untouched; retiring a model stays a human call. +""" + +import argparse +import json +import math +import sys +import time +from collections import Counter +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from functools import reduce +from pathlib import Path +from types import MappingProxyType +from typing import Final, Literal + +import httpx +from pydantic import BaseModel, TypeAdapter, ValidationError + +COST_MAP_RELPATHS: Final = ( + "model_prices_and_context_window.json", + "litellm/model_prices_and_context_window_backup.json", +) +OPENROUTER_MODELS_URL: Final = "https://openrouter.ai/api/v1/models" +VERCEL_MODELS_URL: Final = "https://ai-gateway.vercel.sh/v1/models" +VERCEL_TYPE_TO_MODE: Final = MappingProxyType({"language": "chat", "embedding": "embedding"}) +LIMIT_PAIR: Final = ("max_output_tokens", "max_tokens") +SWING_LIMIT: Final = 10 +INHERITED_TRAITS: Final = frozenset( + { + "prompt_cache_min_tokens", + "supports_adaptive_thinking", + "supports_sampling_params", + "supports_system_messages", + "thinking_always_on", + } +) +PR_BODY_SECTION_LIMIT: Final = 30 +GITHUB_BODY_LIMIT: Final = 65_536 + +Provider = Literal["openrouter", "vercel_ai_gateway"] +RegistryEntry = dict[str, object] +CostMap = dict[str, object] +Prices = Mapping[str, float] + +NO_PRICES: Final[Prices] = MappingProxyType({}) +NO_TRAITS: Final[Mapping[str, object]] = MappingProxyType({}) + + +class SyncError(RuntimeError): + pass + + +class OpenRouterPricing(BaseModel): + prompt: str + completion: str + input_cache_read: str | None = None + input_cache_write: str | None = None + internal_reasoning: str | None = None + image_output: str | None = None + audio: str | None = None + audio_output: str | None = None + + +class OpenRouterArchitecture(BaseModel): + input_modalities: tuple[str, ...] | None = None + output_modalities: tuple[str, ...] | None = None + + +class OpenRouterTopProvider(BaseModel): + max_completion_tokens: int | None = None + + +class OpenRouterModel(BaseModel): + id: str + context_length: int | None = None + architecture: OpenRouterArchitecture | None = None + top_provider: OpenRouterTopProvider = OpenRouterTopProvider() + pricing: OpenRouterPricing + supported_parameters: tuple[str, ...] | None = None + + +class VercelTier(BaseModel): + cost: str + min: int | None = None + max: int | None = None + + +class VercelPricing(BaseModel): + input: str | None = None + output: str | None = None + input_cache_read: str | None = None + input_cache_write: str | None = None + input_tiers: tuple[VercelTier, ...] | None = None + output_tiers: tuple[VercelTier, ...] | None = None + input_cache_read_tiers: tuple[VercelTier, ...] | None = None + input_cache_write_tiers: tuple[VercelTier, ...] | None = None + audio_input_token_cost: str | None = None + audio_output_token_cost: str | None = None + varies_by_provider: bool = False + + +class VercelModalities(BaseModel): + input: tuple[str, ...] | None = None + output: tuple[str, ...] | None = None + + +class VercelModel(BaseModel): + id: str + type: str + context_window: int | None = None + max_tokens: int | None = None + modalities: VercelModalities | None = None + pricing: VercelPricing = VercelPricing() + supported_parameters: tuple[str, ...] | None = None + deprecated_at: int | None = None + + +OPENROUTER_ADAPTER: Final = TypeAdapter(list[OpenRouterModel]) +VERCEL_ADAPTER: Final = TypeAdapter(list[VercelModel]) + + +@dataclass(frozen=True, slots=True) +class CatalogEntry: + key: str + provider: Provider + mode: str + source: str + fields: Mapping[str, object] + indicative_prices: bool = False + + +@dataclass(frozen=True, slots=True) +class Skipped: + reason: str + warning: str | None = None + + +@dataclass(frozen=True, slots=True) +class Unmappable: + problem: str + + +@dataclass(frozen=True, slots=True) +class Catalog: + provider: Provider + entries: tuple[CatalogEntry, ...] + skipped: Mapping[str, int] + warnings: tuple[str, ...] + + +def per_token(price: float) -> float: + return float(f"{price:.6g}") + + +def _token_price(raw: str | None) -> float | None: + if raw is None: + return None + value: Final = float(raw) + return per_token(value) if math.isfinite(value) and value >= 0 else None + + +def _extra_price(raw: str | None) -> float | None: + price: Final = _token_price(raw) + return price if price else None + + +def _flags( + parameters: Sequence[str] | None, inputs: Sequence[str] | None, outputs: Sequence[str] | None +) -> Mapping[str, bool]: + params: Final = frozenset(parameters or ()) + input_modalities: Final = frozenset(inputs or ()) + output_modalities: Final = frozenset(outputs or ()) + claims: Final = { + "supports_function_calling": "tools" in params, + "supports_tool_choice": "tool_choice" in params, + "supports_reasoning": "reasoning" in params, + "supports_response_schema": "structured_outputs" in params, + "supports_vision": "image" in input_modalities, + "supports_pdf_input": bool({"file", "pdf"} & input_modalities), + "supports_audio_input": "audio" in input_modalities, + "supports_video_input": "video" in input_modalities, + "supports_audio_output": "audio" in output_modalities, + } + return MappingProxyType({name: True for name, claimed in claims.items() if claimed}) + + +def _limits(max_input: int | None, max_output: int | None) -> Mapping[str, int]: + ceiling: Final = max_output if max_output is not None else max_input + return MappingProxyType( + { + **({"max_input_tokens": max_input} if max_input is not None else {}), + **({"max_output_tokens": max_output} if max_output is not None else {}), + **({"max_tokens": ceiling} if ceiling is not None else {}), + } + ) + + +def _priced(name: str, price: float | None) -> Prices: + return MappingProxyType({name: price} if price is not None else {}) + + +def _output_prices( + outputs: Sequence[str] | None, image_price: float | None, audio_price: float | None +) -> Prices | Skipped: + modalities: Final = frozenset(outputs or ("text",)) + known: Final = { + modality: price for modality, price in (("image", image_price), ("audio", audio_price)) if price is not None + } + if "text" not in modalities or not (modalities - {"text"}) <= known.keys(): + return Skipped("output priced outside the catalog") + names: Final = {"image": "output_cost_per_image_token", "audio": "output_cost_per_audio_token"} + return MappingProxyType({names[modality]: known[modality] for modality in modalities & known.keys()}) + + +def _tier_threshold(boundary: int) -> int | None: + return next((start // 1000 for start in (boundary, boundary - 1) if start > 0 and start % 1000 == 0), None) + + +@dataclass(frozen=True, slots=True) +class TierLadder: + name: str + base: float + tiers: tuple[VercelTier, ...] + thresholds: frozenset[int] + + def price_above(self, thousands: int) -> float | None: + tokens: Final = thousands * 1000 + 1 + tier: Final = next( + (tier for tier in self.tiers if (tier.min or 0) <= tokens and (tier.max is None or tokens < tier.max)), None + ) + return _token_price(tier.cost) if tier is not None else self.base + + +def _ladder(name: str, base: float | None, tiers: Sequence[VercelTier] | None) -> TierLadder | Unmappable | None: + if base is None or not tiers: + return None + ordered: Final = tuple(sorted(tiers, key=lambda tier: tier.min or 0)) + contiguous: Final = ordered[-1].max is None and all( + lower.max == upper.min for lower, upper in zip(ordered, ordered[1:], strict=False) + ) + if not contiguous: + return Unmappable(f"{name} tiers are not contiguous") + thresholds: Final = tuple(_tier_threshold(tier.min) for tier in ordered if tier.min) + if None in thresholds or any(_token_price(tier.cost) is None for tier in ordered): + return Unmappable(f"{name} tiers have a boundary that is not a whole thousand or an unusable price") + return TierLadder(name, base, ordered, frozenset(threshold for threshold in thresholds if threshold is not None)) + + +def _vercel_tiers(pricing: VercelPricing, cache_read: float | None, cache_write: float | None) -> Prices | Unmappable: + parts: Final = ( + _ladder("input_cost_per_token", _token_price(pricing.input), pricing.input_tiers), + _ladder("output_cost_per_token", _token_price(pricing.output), pricing.output_tiers), + _ladder("cache_read_input_token_cost", cache_read, pricing.input_cache_read_tiers), + _ladder("cache_creation_input_token_cost", cache_write, pricing.input_cache_write_tiers), + ) + problem: Final = next((part for part in parts if isinstance(part, Unmappable)), None) + if problem is not None: + return problem + ladders: Final = tuple(part for part in parts if isinstance(part, TierLadder)) + thresholds: Final = sorted(frozenset().union(*(ladder.thresholds for ladder in ladders))) + return MappingProxyType( + { + f"{ladder.name}_above_{thousands}k_tokens": price + for ladder in ladders + for thousands in thresholds + if (price := ladder.price_above(thousands)) is not None + } + ) + + +def _openrouter_entry(model: OpenRouterModel) -> CatalogEntry | Skipped: + pricing: Final = model.pricing + prompt: Final = _token_price(pricing.prompt) + completion: Final = _token_price(pricing.completion) + if prompt is None or completion is None: + return Skipped("unpriced or router") + inputs: Final = model.architecture.input_modalities if model.architecture else None + outputs: Final = model.architecture.output_modalities if model.architecture else None + output_prices: Final = _output_prices( + outputs, _extra_price(pricing.image_output), _extra_price(pricing.audio_output) + ) + if isinstance(output_prices, Skipped): + return output_prices + fields: Final = { + "input_cost_per_token": prompt, + "output_cost_per_token": completion, + **_limits(model.context_length, model.top_provider.max_completion_tokens), + **_priced("cache_read_input_token_cost", _extra_price(pricing.input_cache_read)), + **_priced("cache_creation_input_token_cost", _extra_price(pricing.input_cache_write)), + **_priced("output_cost_per_reasoning_token", _extra_price(pricing.internal_reasoning)), + **_priced("input_cost_per_audio_token", _extra_price(pricing.audio)), + **output_prices, + **_flags(model.supported_parameters, inputs, outputs), + } + return CatalogEntry( + key=f"openrouter/{model.id}", + provider="openrouter", + mode="chat", + source=f"https://openrouter.ai/{model.id}", + fields=MappingProxyType(fields), + ) + + +def _vercel_entry(model: VercelModel, now_ms: int) -> CatalogEntry | Skipped: + if model.deprecated_at is not None and model.deprecated_at <= now_ms: + return Skipped("deprecated") + mode: Final = VERCEL_TYPE_TO_MODE.get(model.type) + if mode is None: + return Skipped("not token priced") + pricing: Final = model.pricing + prompt: Final = _token_price(pricing.input) + completion: Final = _token_price(pricing.output if mode != "embedding" else pricing.output or "0") + if prompt is None or completion is None: + return Skipped("no usable price") + key: Final = f"vercel_ai_gateway/{model.id}" + inputs: Final = model.modalities.input if model.modalities else None + outputs: Final = model.modalities.output if model.modalities else None + output_prices: Final = _output_prices(outputs, None, _extra_price(pricing.audio_output_token_cost)) + if isinstance(output_prices, Skipped): + return output_prices + cache_read: Final = _extra_price(pricing.input_cache_read) + cache_write: Final = _extra_price(pricing.input_cache_write) + tiers: Final = _vercel_tiers(pricing, cache_read, cache_write) + if isinstance(tiers, Unmappable): + return Skipped("tiers outside the registry's thresholds", warning=f"{key}: {tiers.problem}; row skipped") + fields: Final = { + "input_cost_per_token": prompt, + "output_cost_per_token": completion, + **_limits(model.context_window, model.max_tokens), + **_priced("cache_read_input_token_cost", cache_read), + **_priced("cache_creation_input_token_cost", cache_write), + **_priced("input_cost_per_audio_token", _extra_price(pricing.audio_input_token_cost)), + **output_prices, + **tiers, + **(_flags(model.supported_parameters, inputs, outputs) if mode == "chat" else {}), + } + return CatalogEntry( + key=key, + provider="vercel_ai_gateway", + mode=mode, + source=f"https://vercel.com/ai-gateway/models/{model.id.rsplit('/', 1)[-1]}", + fields=MappingProxyType(fields), + indicative_prices=pricing.varies_by_provider, + ) + + +def _rows(raw: bytes, url: str) -> object: + parsed: Final = json.loads(raw) + rows: Final = parsed.get("data") if isinstance(parsed, dict) else parsed + if not isinstance(rows, list) or not rows: + raise SyncError(f"GET {url} returned no model rows") + return rows + + +def _catalog(provider: Provider, rows: Sequence[CatalogEntry | Skipped]) -> Catalog: + return Catalog( + provider=provider, + entries=tuple(row for row in rows if isinstance(row, CatalogEntry)), + skipped=MappingProxyType(Counter(row.reason for row in rows if isinstance(row, Skipped))), + warnings=tuple(row.warning for row in rows if isinstance(row, Skipped) and row.warning is not None), + ) + + +def load_openrouter(raw: bytes) -> Catalog: + try: + models: Final = OPENROUTER_ADAPTER.validate_python(_rows(raw, OPENROUTER_MODELS_URL)) + except ValidationError as error: + raise SyncError(f"the OpenRouter catalog no longer matches the expected shape: {error}") from error + return _catalog("openrouter", tuple(map(_openrouter_entry, models))) + + +def load_vercel(raw: bytes, now_ms: int) -> Catalog: + try: + models: Final = VERCEL_ADAPTER.validate_python(_rows(raw, VERCEL_MODELS_URL)) + except ValidationError as error: + raise SyncError(f"the Vercel AI Gateway catalog no longer matches the expected shape: {error}") from error + return _catalog("vercel_ai_gateway", tuple(_vercel_entry(model, now_ms) for model in models)) + + +@dataclass(frozen=True, slots=True) +class ProviderOutcome: + provider: Provider + added: tuple[str, ...] + updated: tuple[str, ...] + warnings: tuple[str, ...] + skipped: Mapping[str, int] + + +@dataclass(frozen=True, slots=True) +class SyncOutcome: + cost_map: CostMap + providers: tuple[ProviderOutcome, ...] + + @property + def has_changes(self) -> bool: + return any(outcome.added or outcome.updated for outcome in self.providers) + + +def _root_candidates(bare: str) -> tuple[str, ...]: + segments: Final = bare.split("-") + stems: Final = tuple( + "-".join(segments[:count]) for count in range(len(segments), 0, -1) if count >= 2 or count == len(segments) + ) + return tuple(dict.fromkeys(name for stem in stems for name in (stem, stem.replace(".", "-")))) + + +def _inherited(cost_map: CostMap, entry: CatalogEntry) -> Mapping[str, object]: + bare: Final = entry.key.rsplit("/", 1)[-1].split(":", 1)[0] + same_name: Final = frozenset((bare, bare.replace(".", "-"))) + root: Final = next( + ( + candidate + for name, candidate in ((name, cost_map.get(name)) for name in _root_candidates(bare)) + if isinstance(candidate, dict) and (name in same_name or candidate.get("mode") == entry.mode) + ), + None, + ) + if root is None: + return NO_TRAITS + return MappingProxyType({name: value for name, value in root.items() if name in INHERITED_TRAITS}) + + +def _new_entry(entry: CatalogEntry, inherited: Mapping[str, object]) -> RegistryEntry: + return dict( + sorted( + { + **inherited, + **entry.fields, + "litellm_provider": entry.provider, + "mode": entry.mode, + "source": entry.source, + **({"price_varies_by_provider": True} if entry.indicative_prices else {}), + }.items() + ) + ) + + +@dataclass(frozen=True, slots=True) +class FieldChange: + name: str + old: object + new: object + hold: str | None + + @property + def line(self) -> str: + held: Final = f" held back: {self.hold}" if self.hold else "" + return f"{self.name}: {self.old!r} -> {self.new!r}{held}" + + +def _swing(old: float, new: float) -> str | None: + if (old == 0) != (new == 0): + return "a price crossing zero" + if old and new and max(new / old, old / new) > SWING_LIMIT: + return f"a price moving more than {SWING_LIMIT}x" + return None + + +def _copied_cap_hold(current: object, ceiling: object) -> str | None: + if isinstance(current, int) and isinstance(ceiling, int) and ceiling * SWING_LIMIT < current: + return f"a copied output cap shrinking more than {SWING_LIMIT}x" + return None + + +def _hold(name: str, old: object, new: object, curated_prices_win: bool) -> str | None: + if "cost" in name and curated_prices_win: + return "the catalog price varies by provider" + if old is None: + return None + if name.startswith("supports_") and old is False: + return "a capability flag curated as false" + if name.startswith("max_") and isinstance(old, int) and isinstance(new, int) and new < old: + return "a shrinking limit" + if "cost" in name and isinstance(old, int | float) and isinstance(new, int | float): + return _swing(old, new) + return None + + +def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange, ...]: + curated_prices_win: Final = ( + entry.indicative_prices and "input_cost_per_token" in existing and not existing.get("price_varies_by_provider") + ) + scalars: Final = tuple( + FieldChange(name, existing.get(name), value, _hold(name, existing.get(name), value, curated_prices_win)) + for name, value in entry.fields.items() + if name not in LIMIT_PAIR and existing.get(name) != value + ) + ceiling: Final = entry.fields.get("max_output_tokens") + if ceiling is None: + return scalars + current: Final = existing.get("max_output_tokens", existing.get("max_tokens")) + hold: Final = ( + _copied_cap_hold(current, ceiling) + if current == existing.get("max_input_tokens") + else _hold("max_output_tokens", current, ceiling, curated_prices_win) + ) + return ( + *scalars, + *(FieldChange(name, existing.get(name), ceiling, hold) for name in LIMIT_PAIR if existing.get(name) != ceiling), + ) + + +def _updated_entry( + existing: RegistryEntry, entry: CatalogEntry +) -> tuple[RegistryEntry, tuple[str, ...], tuple[str, ...]]: + changes: Final = _changes(existing, entry) + applied: Final = {change.name: change.new for change in changes if change.hold is None} + return ( + {**existing, **dict(sorted(applied.items()))}, + tuple(change.line for change in changes if change.hold is None), + tuple(change.line for change in changes if change.hold is not None), + ) + + +def _with_new_keys_in_block(ordered: CostMap, result: CostMap, new_keys: Sequence[str], prefix: str) -> CostMap: + provider_keys: Final = tuple(key for key in ordered if key.startswith(prefix)) + if not new_keys: + return {key: result[key] for key in ordered} + if not provider_keys: + return {**{key: result[key] for key in ordered}, **{key: result[key] for key in sorted(new_keys)}} + block_end: Final = provider_keys[-1] + return { + key: value + for existing in ordered + for key, value in ( + (existing, result[existing]), + *((new, result[new]) for new in sorted(new_keys) if existing == block_end), + ) + } + + +@dataclass(frozen=True, slots=True) +class Added: + key: str + entry: RegistryEntry + + +@dataclass(frozen=True, slots=True) +class Updated: + key: str + entry: RegistryEntry + line: str + + +@dataclass(frozen=True, slots=True) +class Warned: + line: str + + +EntrySync = Added | Updated | Warned + + +def _sync_entry(cost_map: CostMap, entry: CatalogEntry) -> tuple[EntrySync, ...]: + existing: Final = cost_map.get(entry.key) + if not isinstance(existing, dict): + return (Added(key=entry.key, entry=_new_entry(entry, _inherited(cost_map, entry))),) + if existing.get("mode") != entry.mode: + return ( + Warned( + line=f"`{entry.key}` has curated mode {existing.get('mode')!r} but the catalog maps to " + f"{entry.mode!r}; left unchanged" + ), + ) + new_entry, applied, held = _updated_entry(existing, entry) + return ( + *((Updated(key=entry.key, entry=new_entry, line=f"{entry.key}: " + "; ".join(applied)),) if applied else ()), + *((Warned(line=f"{entry.key}: " + "; ".join(held)),) if held else ()), + ) + + +SyncState = tuple[CostMap, tuple[ProviderOutcome, ...]] + + +def _sync_provider(state: SyncState, catalog: Catalog) -> SyncState: + cost_map, outcomes = state + syncs: Final = tuple( + sync for entry in sorted(catalog.entries, key=lambda item: item.key) for sync in _sync_entry(cost_map, entry) + ) + outcome: Final = ProviderOutcome( + provider=catalog.provider, + added=tuple(sync.key for sync in syncs if isinstance(sync, Added)), + updated=tuple(sync.line for sync in syncs if isinstance(sync, Updated)), + warnings=(*catalog.warnings, *(sync.line for sync in syncs if isinstance(sync, Warned))), + skipped=catalog.skipped, + ) + merged: Final = {**cost_map, **{sync.key: sync.entry for sync in syncs if isinstance(sync, Added | Updated)}} + return merged, (*outcomes, outcome) + + +def compute_sync(cost_map: CostMap, catalogs: Sequence[Catalog]) -> SyncOutcome: + synced, outcomes = reduce(_sync_provider, catalogs, (dict(cost_map), ())) + return SyncOutcome(cost_map=_ordered_result(cost_map, synced, outcomes), providers=outcomes) + + +def _ordered_result(cost_map: CostMap, result: CostMap, outcomes: Sequence[ProviderOutcome]) -> CostMap: + return reduce( + lambda ordered, outcome: _with_new_keys_in_block(ordered, result, outcome.added, f"{outcome.provider}/"), + outcomes, + {key: result[key] for key in cost_map}, + ) + + +def _section_block(title: str, lines: Sequence[str], backtick: bool, limit: int | None, rest: str) -> str: + shown: Final = lines[:limit] + bullets: Final = "\n".join(f"- `{line}`" if backtick else f"- {line}" for line in shown) or "- none" + overflow: Final = len(lines) - len(shown) + trailer: Final = f"\n- and {overflow} more, see the {rest}" if overflow else "" + return f"### {title} ({len(lines)})\n{bullets}{trailer}\n" + + +def _provider_body(outcome: ProviderOutcome, limit: int | None, warnings_limit: int | None) -> str: + skipped: Final = ", ".join(f"{reason} ({count})" for reason, count in sorted(outcome.skipped.items())) or "none" + return ( + f"## {outcome.provider}\n" + "\n" + f"{_section_block('Added', outcome.added, True, limit, 'diff')}" + "\n" + f"{_section_block('Updated', outcome.updated, True, limit, 'diff')}" + "\n" + f"{_section_block('Warnings needing a human call', outcome.warnings, False, warnings_limit, 'workflow log')}" + "\n" + f"Catalog rows skipped: {skipped}\n" + ) + + +def _pr_body(outcome: SyncOutcome, section_limit: int | None, warnings_limit: int | None) -> str: + return ( + "Automated sync of the openrouter and vercel_ai_gateway entries in model_prices_and_context_window.json " + f"against `GET {OPENROUTER_MODELS_URL}` and `GET {VERCEL_MODELS_URL}` by scripts/sync_cost_map.py. " + "The cost-map-guard check enforces that this PR only adds or reprices models. Changes the script held " + "back (shrinking limits, prices crossing zero or moving more than 10x, per-provider prices on curated " + "rows, capability flags curated as false) are listed under the warnings and need a human commit.\n" + "\n" + "\n".join(_provider_body(provider, section_limit, warnings_limit) for provider in outcome.providers) + ) + + +def render_pr_body(outcome: SyncOutcome, section_limit: int | None = PR_BODY_SECTION_LIMIT) -> str: + every_warning: Final = _pr_body(outcome, section_limit, None) + if section_limit is None or len(every_warning) <= GITHUB_BODY_LIMIT: + return every_warning + return _pr_body(outcome, section_limit, section_limit) + + +def render_summary(outcome: SyncOutcome) -> str: + return " ".join( + f"{provider.provider}: added={len(provider.added)} updated={len(provider.updated)} " + f"warnings={len(provider.warnings)}" + for provider in outcome.providers + ) + + +def _fetch(url: str) -> bytes: + response: Final = httpx.get(url, timeout=30, follow_redirects=True) + if response.status_code != 200: + raise SyncError(f"GET {url} returned {response.status_code}") + return response.content + + +def _serialize(cost_map: CostMap) -> str: + return json.dumps(cost_map, indent=4, ensure_ascii=False) + "\n" + + +def main(argv: Sequence[str]) -> int: + parser: Final = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--write", action="store_true", help="apply the sync to the cost map files (default: dry run)") + parser.add_argument("--openrouter-json", type=Path, help="recorded OpenRouter catalog instead of the live API") + parser.add_argument("--vercel-json", type=Path, help="recorded Vercel AI Gateway catalog instead of the live API") + parser.add_argument("--pr-body-file", type=Path, help="write the generated PR body to this path") + parser.add_argument("--repo-root", type=Path, default=Path(__file__).resolve().parent.parent) + args: Final = parser.parse_args(argv) + + openrouter_raw: Final = ( + args.openrouter_json.read_bytes() if args.openrouter_json is not None else _fetch(OPENROUTER_MODELS_URL) + ) + vercel_raw: Final = args.vercel_json.read_bytes() if args.vercel_json is not None else _fetch(VERCEL_MODELS_URL) + catalogs: Final = (load_openrouter(openrouter_raw), load_vercel(vercel_raw, now_ms=int(time.time() * 1000))) + + cost_map_path: Final = args.repo_root / COST_MAP_RELPATHS[0] + cost_map: Final = json.loads(cost_map_path.read_text()) + outcome: Final = compute_sync(cost_map, catalogs) + + if args.pr_body_file is not None: + args.pr_body_file.write_text(render_pr_body(outcome)) + if args.write and outcome.has_changes: + for relpath in COST_MAP_RELPATHS: + (args.repo_root / relpath).write_text(_serialize(outcome.cost_map)) + print(render_summary(outcome)) + print() + print(render_pr_body(outcome, section_limit=None)) + if not args.write: + print("dry run: no files were touched") + elif not outcome.has_changes: + print("registry already in sync: no files were touched") + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main(sys.argv[1:])) + except SyncError as error: + print(f"SYNC FAILED: {error}", file=sys.stderr) + raise SystemExit(1) from error diff --git a/tests/test_litellm/fixtures/cost_map_sync/openrouter_models.json b/tests/test_litellm/fixtures/cost_map_sync/openrouter_models.json new file mode 100644 index 00000000000..eb06b6a3058 --- /dev/null +++ b/tests/test_litellm/fixtures/cost_map_sync/openrouter_models.json @@ -0,0 +1,209 @@ +{ + "data": [ + { + "id": "cohere/north-mini-code:free", + "context_length": 256000, + "architecture": { + "input_modalities": [ + "text" + ] + }, + "top_provider": { + "max_completion_tokens": 64000 + }, + "pricing": { + "prompt": "0", + "completion": "0" + }, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "max_tokens", + "presence_penalty", + "reasoning", + "seed", + "stop", + "temperature", + "tool_choice", + "tools", + "top_k", + "top_p" + ] + }, + { + "id": "deepseek/deepseek-v4-pro-0813", + "context_length": 1048576, + "architecture": { + "input_modalities": [ + "text" + ] + }, + "top_provider": { + "max_completion_tokens": 384000 + }, + "pricing": { + "prompt": "0.00000057948", + "completion": "0.00000173844", + "input_cache_read": "0.000000019316" + }, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "logit_bias", + "logprobs", + "max_tokens", + "min_p", + "presence_penalty", + "reasoning", + "reasoning_effort", + "repetition_penalty", + "response_format", + "seed", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools", + "top_k", + "top_logprobs", + "top_p" + ] + }, + { + "id": "google/gemma-4-26b-a4b-it:free", + "context_length": 262144, + "architecture": { + "input_modalities": [ + "image", + "text", + "video" + ] + }, + "top_provider": { + "max_completion_tokens": 32768 + }, + "pricing": { + "prompt": "0", + "completion": "0" + }, + "supported_parameters": [ + "include_reasoning", + "max_tokens", + "reasoning", + "response_format", + "seed", + "temperature", + "tool_choice", + "tools", + "top_p" + ] + }, + { + "id": "inception/mercury-2.5-preview", + "context_length": 260000, + "architecture": { + "input_modalities": [ + "text" + ] + }, + "top_provider": { + "max_completion_tokens": 65536 + }, + "pricing": { + "prompt": "0.00000004", + "completion": "0.00000015", + "input_cache_read": "0.000000004" + }, + "supported_parameters": [ + "include_reasoning", + "max_tokens", + "reasoning", + "reasoning_effort", + "response_format", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools" + ] + }, + { + "id": "openai/gpt-5-mini", + "context_length": 400000, + "architecture": { + "input_modalities": [ + "text", + "image", + "file" + ] + }, + "top_provider": { + "max_completion_tokens": 128000 + }, + "pricing": { + "prompt": "0.00000025", + "completion": "0.000002", + "web_search": "0.01", + "input_cache_read": "0.000000025", + "image": "0.003613" + }, + "supported_parameters": [ + "include_reasoning", + "max_completion_tokens", + "max_tokens", + "reasoning", + "reasoning_effort", + "response_format", + "seed", + "structured_outputs", + "tool_choice", + "tools" + ] + }, + { + "id": "openrouter/auto", + "context_length": 2000000, + "architecture": { + "input_modalities": [ + "text", + "image", + "audio", + "file", + "video" + ] + }, + "top_provider": { + "max_completion_tokens": null + }, + "pricing": { + "prompt": "-1", + "completion": "-1" + }, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "logit_bias", + "logprobs", + "max_tokens", + "min_p", + "prediction", + "presence_penalty", + "reasoning", + "reasoning_effort", + "repetition_penalty", + "response_format", + "seed", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools", + "top_a", + "top_k", + "top_logprobs", + "top_p", + "web_search_options" + ] + } + ] +} diff --git a/tests/test_litellm/fixtures/cost_map_sync/vercel_models.json b/tests/test_litellm/fixtures/cost_map_sync/vercel_models.json new file mode 100644 index 00000000000..82fcb72f353 --- /dev/null +++ b/tests/test_litellm/fixtures/cost_map_sync/vercel_models.json @@ -0,0 +1,142 @@ +{ + "object": "list", + "data": [ + { + "id": "alibaba/qwen3-embedding-0.6b", + "type": "embedding", + "context_window": 32768, + "max_tokens": 32768, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "pricing": { + "input": "0.00000001" + }, + "supported_parameters": null, + "deprecated_at": null + }, + { + "id": "bfl/flux-2-flex", + "type": "image", + "context_window": 0, + "max_tokens": 0, + "modalities": { + "input": [ + "text" + ], + "output": [ + "image" + ] + }, + "pricing": {}, + "supported_parameters": null, + "deprecated_at": null + }, + { + "id": "openai/gpt-4o-mini-transcribe", + "type": "transcription", + "context_window": null, + "max_tokens": null, + "modalities": { + "input": [ + "audio" + ], + "output": [ + "text" + ] + }, + "pricing": { + "input": "0.00000125", + "output": "0.000005" + }, + "supported_parameters": null, + "deprecated_at": 1750000000000 + }, + { + "id": "openai/gpt-5-mini", + "type": "language", + "context_window": 400000, + "max_tokens": 128000, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "pricing": { + "input": "0.00000025", + "output": "0.000002", + "input_cache_read": "0.000000025" + }, + "supported_parameters": [ + "max_tokens", + "stop", + "tools", + "tool_choice", + "reasoning", + "include_reasoning" + ], + "deprecated_at": null + }, + { + "id": "perplexity/sonar", + "type": "language", + "context_window": 127000, + "max_tokens": 8000, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "pricing": {}, + "supported_parameters": [ + "max_tokens", + "temperature", + "stop" + ], + "deprecated_at": null + }, + { + "id": "zai/glm-4.6", + "type": "language", + "context_window": 200000, + "max_tokens": 96000, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "pricing": { + "input": "0.0000006", + "output": "0.0000022", + "input_cache_read": "0.00000011" + }, + "supported_parameters": [ + "max_tokens", + "temperature", + "stop", + "tools", + "tool_choice", + "reasoning", + "include_reasoning" + ], + "deprecated_at": null + } + ] +} diff --git a/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py b/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py index c509c8399c9..8d297917448 100644 --- a/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py +++ b/tests/test_litellm/litellm_core_utils/test_get_model_cost_map.py @@ -255,22 +255,6 @@ def test_azure_ai_claude_1m_context_entries(cost_map: dict): assert cost_map[model]["max_input_tokens"] == 200000, model -# OpenRouter headline rates from GET https://openrouter.ai/api/v1/models. -# These were the catalog values that disagreed with that API (and, for the -# two spotlight models, the public model pages that their source fields cite). -_OPENROUTER_LIVE_COSTS = { - "openrouter/qwen/qwen3.5-plus-02-15": (2.6e-07, 1.56e-06, None), - "openrouter/openai/gpt-oss-120b": (3.7e-08, 1.7e-07, None), - "openrouter/qwen/qwen3-coder-plus": (6.5e-07, 3.25e-06, None), - "openrouter/qwen/qwen3.5-flash-02-23": (6.5e-08, 2.6e-07, None), - "openrouter/qwen/qwen3.5-27b": (1.95e-07, 1.56e-06, None), - "openrouter/gryphe/mythomax-l2-13b": (6e-08, 6e-08, None), - "openrouter/mancer/weaver": (4e-07, 7.5e-07, None), - "openrouter/xiaomi/mimo-v2.5-pro": (4.35e-07, 8.7e-07, 3.6e-09), - "openrouter/moonshotai/kimi-k2.5": (4.5e-07, 2.25e-06, 7e-08), - "openrouter/z-ai/glm-5": (6e-07, 1.92e-06, None), -} - _OPENROUTER_STALE_COSTS = { "openrouter/qwen/qwen3.5-plus-02-15": (4e-07, 2.4e-06), "openrouter/openai/gpt-oss-120b": (1.8e-07, 8e-07), @@ -283,23 +267,12 @@ _OPENROUTER_STALE_COSTS = { [_load_root_cost_map(), GetModelCostMap.load_local_model_cost_map()], ids=["root", "bundled_backup"], ) -def test_openrouter_catalog_costs_match_live_headline_rates(cost_map: dict): - """openrouter/* spend tracking reads these catalog fields. The values must - stay aligned with OpenRouter's published headline rate, not the stale - figures that over/under-counted by up to 30x. Both maps are checked so - the root file and bundled backup cannot drift apart.""" - control = cost_map["openrouter/anthropic/claude-opus-5"] - assert control["input_cost_per_token"] == 5e-06 - assert control["output_cost_per_token"] == 2.5e-05 - assert control["cache_read_input_token_cost"] == 5e-07 - - for model, (inp, out, cache) in _OPENROUTER_LIVE_COSTS.items(): - entry = cost_map[model] - assert entry["input_cost_per_token"] == inp, model - assert entry["output_cost_per_token"] == out, model - if cache is not None: - assert entry["cache_read_input_token_cost"] == cache, model - +def test_openrouter_stale_catalog_costs_never_return(cost_map: dict): + """scripts/sync_cost_map.py keeps openrouter/* aligned with OpenRouter's + published headline rates, so the live values are not pinned here: a pin + would turn every legitimate reprice into a red sync PR. The stale figures + that over/under-counted by up to 30x must never come back, in the root + file or the bundled backup.""" for model, (stale_in, stale_out) in _OPENROUTER_STALE_COSTS.items(): entry = cost_map[model] assert entry["input_cost_per_token"] != stale_in, model diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index f610821e06a..5613eedc400 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -260,10 +260,10 @@ def test_openrouter_qwen36_plus_model_info(_local_model_cost_map): assert model_info is not None assert model_info["litellm_provider"] == "openrouter" assert model_info["mode"] == "chat" - assert model_info["max_input_tokens"] == 1000000 - assert model_info["max_output_tokens"] == 65536 - assert model_info["input_cost_per_token"] == 3.25e-07 - assert model_info["output_cost_per_token"] == 1.95e-06 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 assert model_info["supports_function_calling"] is True assert model_info["supports_tool_choice"] is True assert model_info["supports_reasoning"] is True @@ -3336,10 +3336,10 @@ def test_openrouter_gemini_3_1_flash_lite_preview_pricing(_local_model_cost_map) assert model_info is not None, f"Missing model pricing entry: {model_name}" assert model_info["litellm_provider"] == "openrouter" - assert model_info["input_cost_per_token"] == 2.5e-07 - assert model_info["output_cost_per_token"] == 1.5e-06 - assert model_info["max_input_tokens"] == 1048576 - assert model_info["max_output_tokens"] == 65536 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 def test_gemini_3_1_flash_lite_pricing(_local_model_cost_map): @@ -3678,8 +3678,9 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map): consistency issue, not a design choice. Same shape as the preview-variant gap fixed in PR #25610. - Pricing matches the existing -preview entry one-for-one (input $0.25/M, output - $1.50/M, cache-read $0.025/M) — Google did not change costs at the GA cutover. + The exact prices are not pinned: scripts/sync_cost_map.py keeps openrouter/* + aligned with OpenRouter's catalog, and a pin would turn a reprice into a red + sync PR. """ model_name = "openrouter/google/gemini-3.1-flash-lite" @@ -3687,11 +3688,11 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map): assert model_info is not None, f"Missing model pricing entry: {model_name}" assert model_info["litellm_provider"] == "openrouter" - assert model_info["input_cost_per_token"] == 2.5e-07 - assert model_info["output_cost_per_token"] == 1.5e-06 - assert model_info["cache_read_input_token_cost"] == 2.5e-08 - assert model_info["max_input_tokens"] == 1048576 - assert model_info["max_output_tokens"] == 65536 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 + assert model_info["cache_read_input_token_cost"] > 0 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_map): diff --git a/tests/test_litellm/test_sync_cost_map.py b/tests/test_litellm/test_sync_cost_map.py new file mode 100644 index 00000000000..8744b7bbe51 --- /dev/null +++ b/tests/test_litellm/test_sync_cost_map.py @@ -0,0 +1,885 @@ +import importlib.util +import json +from pathlib import Path +from types import ModuleType +from typing import Final, cast + +import pytest + +from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token +from litellm.types.utils import ModelInfo, Usage + +REPO_ROOT: Final = Path(__file__).resolve().parents[2] +SCRIPT_PATH: Final = REPO_ROOT / "scripts" / "sync_cost_map.py" +GENERATOR_PATH: Final = REPO_ROOT / "ci_cd" / "generate_model_prices_schema.py" +FIXTURES: Final = Path(__file__).parent / "fixtures" / "cost_map_sync" +OPENROUTER_RAW: Final = (FIXTURES / "openrouter_models.json").read_bytes() +VERCEL_RAW: Final = (FIXTURES / "vercel_models.json").read_bytes() +NOW_MS: Final = 1757030400000 + +EXISTING_DEEPSEEK: Final = { + "input_cost_per_token": 0.00000132, + "input_cost_per_token_cache_hit": 4.4e-8, + "litellm_provider": "openrouter", + "max_input_tokens": 1048576, + "max_output_tokens": 300000, + "max_tokens": 300000, + "mode": "chat", + "output_cost_per_token": 0.00000396, + "source": "https://openrouter.ai/deepseek/deepseek-v4-pro-0813", + "supports_function_calling": True, + "supports_prompt_caching": True, + "supports_reasoning": True, + "supports_response_schema": True, + "supports_tool_choice": True, +} +EXISTING_GLM: Final = { + "litellm_provider": "vercel_ai_gateway", + "cache_read_input_token_cost": 1.1e-7, + "input_cost_per_token": 4.5e-7, + "max_input_tokens": 200000, + "max_output_tokens": 200000, + "max_tokens": 200000, + "mode": "chat", + "output_cost_per_token": 0.0000018, + "source": "https://vercel.com/ai-gateway/models/glm-4.6", + "supports_function_calling": True, + "supports_parallel_function_calling": True, + "supports_tool_choice": True, +} +BLOCK_END_OPENROUTER: Final = { + "litellm_provider": "openrouter", + "mode": "completion", + "input_cost_per_token": 0.0000015, + "output_cost_per_token": 0.000002, +} + + +def _base_map() -> dict[str, object]: + return { + "sample_spec": {"litellm_provider": "one of https://docs.litellm.ai/docs/providers"}, + "gpt-4o": {"litellm_provider": "openai", "mode": "chat"}, + "openrouter/deepseek/deepseek-v4-pro-0813": dict(EXISTING_DEEPSEEK), + "openrouter/openai/gpt-3.5-turbo-instruct": dict(BLOCK_END_OPENROUTER), + "vercel_ai_gateway/zai/glm-4.6": dict(EXISTING_GLM), + "zzz/last": {"litellm_provider": "zzz", "mode": "chat"}, + } + + +@pytest.fixture(scope="module") +def sync() -> ModuleType: + spec: Final = importlib.util.spec_from_file_location("sync_cost_map", SCRIPT_PATH) + assert spec is not None and spec.loader is not None + module: Final = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _openrouter_rows(*rows: dict[str, object]) -> bytes: + return json.dumps( + {"data": [{"pricing": {"prompt": "0.000001", "completion": "0.000002"}, **row} for row in rows]} + ).encode() + + +def _vercel_rows(*rows: dict[str, object]) -> bytes: + defaults: Final = {"type": "language", "pricing": {"input": "0.000001", "output": "0.000002"}} + return json.dumps({"data": [{**defaults, **row} for row in rows]}).encode() + + +def _run(sync: ModuleType, cost_map: dict[str, object]): + return sync.compute_sync( + cost_map, (sync.load_openrouter(OPENROUTER_RAW), sync.load_vercel(VERCEL_RAW, now_ms=NOW_MS)) + ) + + +def test_new_openrouter_entry_carries_catalog_prices_limits_and_capabilities(sync: ModuleType) -> None: + outcome: Final = _run(sync, _base_map()) + + assert outcome.cost_map["openrouter/inception/mercury-2.5-preview"] == { + "cache_read_input_token_cost": 4e-9, + "input_cost_per_token": 4e-8, + "litellm_provider": "openrouter", + "max_input_tokens": 260000, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 1.5e-7, + "source": "https://openrouter.ai/inception/mercury-2.5-preview", + "supports_function_calling": True, + "supports_reasoning": True, + "supports_response_schema": True, + "supports_tool_choice": True, + } + + +def test_input_modalities_become_capability_flags(sync: ModuleType) -> None: + outcome: Final = _run(sync, _base_map()) + + gpt5: Final = outcome.cost_map["openrouter/openai/gpt-5-mini"] + gemma: Final = outcome.cost_map["openrouter/google/gemma-4-26b-a4b-it:free"] + vercel_gpt5: Final = outcome.cost_map["vercel_ai_gateway/openai/gpt-5-mini"] + assert (gpt5["supports_vision"], gpt5["supports_pdf_input"]) == (True, True) + assert "supports_video_input" not in gpt5 and "supports_audio_input" not in gpt5 + assert "input_cost_per_image" not in gpt5 and "supports_prompt_caching" not in gpt5 + assert (gemma["supports_vision"], gemma["supports_video_input"]) == (True, True) + assert (vercel_gpt5["supports_vision"], vercel_gpt5["supports_pdf_input"]) == (True, True) + assert vercel_gpt5["cache_read_input_token_cost"] == 2.5e-8 + assert vercel_gpt5["source"] == "https://vercel.com/ai-gateway/models/gpt-5-mini" + + +def test_free_models_are_added_with_zero_prices(sync: ModuleType) -> None: + outcome: Final = _run(sync, _base_map()) + + free: Final = outcome.cost_map["openrouter/cohere/north-mini-code:free"] + assert (free["input_cost_per_token"], free["output_cost_per_token"]) == (0.0, 0.0) + assert (free["max_input_tokens"], free["max_output_tokens"], free["max_tokens"]) == (256000, 64000, 64000) + assert "cache_read_input_token_cost" not in free + + +def test_router_rows_and_unpriced_rows_are_skipped(sync: ModuleType) -> None: + outcome: Final = _run(sync, _base_map()) + + openrouter, vercel = outcome.providers + assert "openrouter/openrouter/auto" not in outcome.cost_map + assert "vercel_ai_gateway/perplexity/sonar" not in outcome.cost_map + assert dict(openrouter.skipped) == {"unpriced or router": 1} + assert dict(vercel.skipped) == {"deprecated": 1, "not token priced": 1, "no usable price": 1} + + +def test_vercel_rows_map_type_to_mode_and_drop_non_token_types(sync: ModuleType) -> None: + outcome: Final = _run(sync, _base_map()) + + embedding: Final = outcome.cost_map["vercel_ai_gateway/alibaba/qwen3-embedding-0.6b"] + assert embedding == { + "input_cost_per_token": 1e-8, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "embedding", + "output_cost_per_token": 0.0, + "source": "https://vercel.com/ai-gateway/models/qwen3-embedding-0.6b", + } + assert "vercel_ai_gateway/bfl/flux-2-flex" not in outcome.cost_map + assert "vercel_ai_gateway/openai/gpt-4o-mini-transcribe" not in outcome.cost_map + + +def test_existing_entry_is_repriced_without_losing_curated_fields(sync: ModuleType) -> None: + outcome: Final = _run(sync, _base_map()) + + deepseek: Final = outcome.cost_map["openrouter/deepseek/deepseek-v4-pro-0813"] + assert deepseek["input_cost_per_token"] == 5.7948e-7 + assert deepseek["output_cost_per_token"] == 1.73844e-6 + assert deepseek["cache_read_input_token_cost"] == 1.9316e-8 + assert deepseek["input_cost_per_token_cache_hit"] == 4.4e-8 + assert (deepseek["max_output_tokens"], deepseek["max_tokens"]) == (384000, 384000) + glm: Final = outcome.cost_map["vercel_ai_gateway/zai/glm-4.6"] + assert (glm["input_cost_per_token"], glm["output_cost_per_token"]) == (6e-7, 2.2e-6) + assert glm["supports_parallel_function_calling"] is True + assert glm["supports_reasoning"] is True + assert (glm["max_output_tokens"], glm["max_tokens"]) == (96000, 96000) + openrouter, vercel = outcome.providers + assert [line.split(":")[0] for line in openrouter.updated] == ["openrouter/deepseek/deepseek-v4-pro-0813"] + assert "input_cost_per_token: 1.32e-06 -> 5.7948e-07" in openrouter.updated[0] + assert "max_output_tokens: 300000 -> 384000; max_tokens: 300000 -> 384000" in openrouter.updated[0] + assert [line.split(":")[0] for line in vercel.updated] == ["vercel_ai_gateway/zai/glm-4.6"] + assert "max_output_tokens: 200000 -> 96000; max_tokens: 200000 -> 96000" in vercel.updated[0] + assert vercel.warnings == () + + +def test_legacy_max_tokens_moves_in_step_with_the_catalog_output_ceiling(sync: ModuleType) -> None: + legacy: Final = { + "input_cost_per_token": 4e-8, + "litellm_provider": "openrouter", + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1.5e-7, + } + outcome: Final = _run(sync, {**_base_map(), "openrouter/inception/mercury-2.5-preview": legacy}) + + mercury: Final = outcome.cost_map["openrouter/inception/mercury-2.5-preview"] + assert (mercury["max_output_tokens"], mercury["max_tokens"]) == (65536, 65536) + assert mercury["max_input_tokens"] == 260000 + assert list(mercury) == [ + *legacy, + "cache_read_input_token_cost", + "max_input_tokens", + "max_output_tokens", + "supports_function_calling", + "supports_reasoning", + "supports_response_schema", + "supports_tool_choice", + ] + + +def test_an_output_cap_copied_from_the_context_window_yields_to_the_catalog_ceiling(sync: ModuleType) -> None: + copied: Final = { + "litellm_provider": "openrouter", + "mode": "chat", + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "max_input_tokens": 163840, + "max_output_tokens": 163840, + "max_tokens": 163840, + } + genuine: Final = {**copied, "max_output_tokens": 100000, "max_tokens": 100000} + row: Final = {"context_length": 163840, "top_provider": {"max_completion_tokens": 65536}} + glitch_row: Final = {"context_length": 163840, "top_provider": {"max_completion_tokens": 8192}} + catalog: Final = sync.load_openrouter( + _openrouter_rows( + {"id": "acme/copied", **row}, {"id": "acme/genuine", **row}, {"id": "acme/glitch", **glitch_row} + ) + ) + + outcome: Final = sync.compute_sync( + { + "openrouter/acme/copied": dict(copied), + "openrouter/acme/genuine": dict(genuine), + "openrouter/acme/glitch": dict(copied), + }, + (catalog,), + ) + + applied: Final = outcome.cost_map["openrouter/acme/copied"] + assert (applied["max_input_tokens"], applied["max_output_tokens"], applied["max_tokens"]) == (163840, 65536, 65536) + assert outcome.cost_map["openrouter/acme/genuine"] == genuine + assert outcome.cost_map["openrouter/acme/glitch"] == copied + assert outcome.providers[0].updated == ( + "openrouter/acme/copied: max_output_tokens: 163840 -> 65536; max_tokens: 163840 -> 65536", + ) + assert outcome.providers[0].warnings == ( + "openrouter/acme/genuine: max_output_tokens: 100000 -> 65536 held back: a shrinking limit; " + "max_tokens: 100000 -> 65536 held back: a shrinking limit", + "openrouter/acme/glitch: max_output_tokens: 163840 -> 8192 held back: a copied output cap shrinking more " + "than 10x; max_tokens: 163840 -> 8192 held back: a copied output cap shrinking more than 10x", + ) + + +def test_output_limits_stay_put_when_the_catalog_has_no_output_ceiling(sync: ModuleType) -> None: + existing: Final = { + "litellm_provider": "openrouter", + "mode": "chat", + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "max_input_tokens": 1000, + "max_output_tokens": 500, + "max_tokens": 500, + } + catalog: Final = sync.load_openrouter(_openrouter_rows({"id": "acme/x", "context_length": 4000})) + + outcome: Final = sync.compute_sync({"openrouter/acme/x": dict(existing)}, (catalog,)) + + entry: Final = outcome.cost_map["openrouter/acme/x"] + assert (entry["max_input_tokens"], entry["max_output_tokens"], entry["max_tokens"]) == (4000, 500, 500) + assert outcome.providers[0].updated == ("openrouter/acme/x: max_input_tokens: 1000 -> 4000",) + assert outcome.providers[0].warnings == () + + +def test_untouched_entries_survive_byte_for_byte(sync: ModuleType) -> None: + outcome: Final = _run(sync, _base_map()) + + assert outcome.cost_map["gpt-4o"] == {"litellm_provider": "openai", "mode": "chat"} + assert outcome.cost_map["openrouter/openai/gpt-3.5-turbo-instruct"] == BLOCK_END_OPENROUTER + assert outcome.cost_map["sample_spec"] == _base_map()["sample_spec"] + + +def test_second_sync_is_a_no_op(sync: ModuleType) -> None: + first: Final = _run(sync, _base_map()) + + second: Final = _run(sync, dict(first.cost_map)) + + assert second.has_changes is False + assert all(not provider.added and not provider.updated for provider in second.providers) + assert list(second.cost_map) == list(first.cost_map) + + +def test_mode_mismatch_warns_and_leaves_the_entry_alone(sync: ModuleType) -> None: + outcome: Final = _run( + sync, + { + **_base_map(), + "vercel_ai_gateway/openai/gpt-5-mini": { + "litellm_provider": "vercel_ai_gateway", + "mode": "responses", + "input_cost_per_token": 1.0, + }, + }, + ) + + assert outcome.cost_map["vercel_ai_gateway/openai/gpt-5-mini"]["input_cost_per_token"] == 1.0 + vercel: Final = outcome.providers[1] + assert "vercel_ai_gateway/openai/gpt-5-mini" not in vercel.added + assert all("gpt-5-mini" not in line for line in vercel.updated) + mismatch: Final = [line for line in vercel.warnings if "gpt-5-mini" in line] + assert len(mismatch) == 1 + assert "vercel_ai_gateway/openai/gpt-5-mini" in mismatch[0] + assert "'responses'" in mismatch[0] and "'chat'" in mismatch[0] + + +def test_new_keys_land_at_the_end_of_their_provider_block(sync: ModuleType) -> None: + outcome: Final = _run(sync, _base_map()) + + assert list(outcome.cost_map) == [ + "sample_spec", + "gpt-4o", + "openrouter/deepseek/deepseek-v4-pro-0813", + "openrouter/openai/gpt-3.5-turbo-instruct", + "openrouter/cohere/north-mini-code:free", + "openrouter/google/gemma-4-26b-a4b-it:free", + "openrouter/inception/mercury-2.5-preview", + "openrouter/openai/gpt-5-mini", + "vercel_ai_gateway/zai/glm-4.6", + "vercel_ai_gateway/alibaba/qwen3-embedding-0.6b", + "vercel_ai_gateway/openai/gpt-5-mini", + "zzz/last", + ] + + +def test_provider_without_a_block_is_appended_at_the_end(sync: ModuleType) -> None: + outcome: Final = _run(sync, {"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}}) + + keys: Final = list(outcome.cost_map) + assert keys[0] == "gpt-4o" + assert keys[1:6] == sorted(keys[1:6]) and all(key.startswith("openrouter/") for key in keys[1:6]) + assert keys[6:] == sorted(keys[6:]) and all(key.startswith("vercel_ai_gateway/") for key in keys[6:]) + assert len(keys) == 9 + + +def test_pr_body_lists_changes_per_provider(sync: ModuleType) -> None: + body: Final = sync.render_pr_body(_run(sync, _base_map())) + + assert "## openrouter" in body and "## vercel_ai_gateway" in body + assert "### Added (4)" in body and "- `openrouter/inception/mercury-2.5-preview`" in body + assert "### Added (2)" in body and "- `vercel_ai_gateway/openai/gpt-5-mini`" in body + assert "- `openrouter/deepseek/deepseek-v4-pro-0813: input_cost_per_token: 1.32e-06 -> 5.7948e-07" in body + assert "Catalog rows skipped: deprecated (1), no usable price (1), not token priced (1)" in body + + +def _large_outcome(sync: ModuleType, warning_count: int): + lines: Final = tuple(f"provider/model-{index}: {'x' * 200}" for index in range(400)) + return sync.SyncOutcome( + cost_map={}, + providers=tuple( + sync.ProviderOutcome( + provider=provider, added=lines, updated=lines, warnings=lines[:warning_count], skipped={} + ) + for provider in ("openrouter", "vercel_ai_gateway") + ), + ) + + +def test_pr_body_caps_added_and_updated_but_lists_every_warning(sync: ModuleType) -> None: + body: Final = sync.render_pr_body(_large_outcome(sync, warning_count=40)) + + assert body.count("### Added (400)") == 2 and body.count("- and 370 more, see the diff") == 4 + assert "provider/model-30: " in body and body.count("- provider/model-39: ") == 2 + assert body.count("### Warnings needing a human call (40)") == 2 and "see the workflow log" not in body + + +def test_pr_body_caps_warnings_too_only_when_they_would_push_it_past_github_limit(sync: ModuleType) -> None: + body: Final = sync.render_pr_body(_large_outcome(sync, warning_count=400)) + + assert len(body) < 65_536 + assert body.count("- and 370 more, see the workflow log") == 2 + assert body.count("- `provider/model-29: ") == 4 and "provider/model-30: " not in body + + +def test_every_held_change_reaches_the_pr_body_and_the_workflow_log(sync: ModuleType, tmp_path: Path, capsys) -> None: + keys: Final = tuple(f"openrouter/acme/model-{index:02d}" for index in range(40)) + catalog: Final = { + "data": [ + {"id": key.removeprefix("openrouter/"), "pricing": {"prompt": "0.000001", "completion": "0.000002"}} + for key in keys + ] + } + cost_map: Final = {**_base_map(), **{key: {"litellm_provider": "openrouter", "mode": "embedding"} for key in keys}} + for relpath in ("model_prices_and_context_window.json", "litellm/model_prices_and_context_window_backup.json"): + (tmp_path / relpath).parent.mkdir(parents=True, exist_ok=True) + (tmp_path / relpath).write_text(json.dumps(cost_map, indent=4) + "\n") + (tmp_path / "openrouter.json").write_text(json.dumps(catalog)) + body_file: Final = tmp_path / "body.md" + + code: Final = sync.main( + [ + "--openrouter-json", + str(tmp_path / "openrouter.json"), + "--vercel-json", + str(FIXTURES / "vercel_models.json"), + "--pr-body-file", + str(body_file), + "--repo-root", + str(tmp_path), + ] + ) + + body: Final = body_file.read_text() + log: Final = capsys.readouterr().out + assert code == 0 + assert "### Warnings needing a human call (40)" in body and "more, see the workflow log" not in body + assert all(key in body for key in keys) + assert all(key in log for key in keys) and "more, see the" not in log + + +@pytest.mark.parametrize( + ("loader", "raw"), + [ + ("load_openrouter", b'{"data": []}'), + ("load_openrouter", b'{"data": [{"id": "x", "pricing": {"prompt": 1}}]}'), + ("load_vercel", b"[]"), + ("load_vercel", b'{"data": [{"id": "x"}]}'), + ], +) +def test_malformed_catalogs_fail_the_run(sync: ModuleType, loader: str, raw: bytes) -> None: + kwargs: Final = {"now_ms": NOW_MS} if loader == "load_vercel" else {} + with pytest.raises(sync.SyncError): + getattr(sync, loader)(raw, **kwargs) + + +@pytest.mark.parametrize("price", ["Infinity", "-Infinity", "NaN"]) +def test_non_finite_catalog_prices_count_as_unpriced(sync: ModuleType, price: str) -> None: + raw: Final = json.dumps({"data": [{"id": "acme/x", "pricing": {"prompt": price, "completion": "0"}}]}).encode() + + catalog: Final = sync.load_openrouter(raw) + + assert catalog.entries == () + assert dict(catalog.skipped) == {"unpriced or router": 1} + + +def _vercel_language_row(deprecated_at: int | None) -> bytes: + row: Final = { + "id": "acme/chat-1", + "type": "language", + "context_window": 1000, + "max_tokens": 100, + "pricing": {"input": "0.000001", "output": "0.000002"}, + "deprecated_at": deprecated_at, + } + return json.dumps({"data": [row]}).encode() + + +def test_a_scheduled_deprecation_keeps_syncing_until_the_date(sync: ModuleType) -> None: + scheduled: Final = sync.load_vercel(_vercel_language_row(NOW_MS + 1), now_ms=NOW_MS) + passed: Final = sync.load_vercel(_vercel_language_row(NOW_MS), now_ms=NOW_MS) + + assert [entry.key for entry in scheduled.entries] == ["vercel_ai_gateway/acme/chat-1"] + assert dict(scheduled.skipped).get("deprecated", 0) == 0 + assert passed.entries == () + assert dict(passed.skipped)["deprecated"] == 1 + + +def _repo(tmp_path: Path) -> Path: + for relpath in ("model_prices_and_context_window.json", "litellm/model_prices_and_context_window_backup.json"): + target = tmp_path / relpath + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(json.dumps(_base_map(), indent=4) + "\n") + return tmp_path + + +def test_write_updates_both_cost_map_files_identically(sync: ModuleType, tmp_path: Path, capsys) -> None: + repo: Final = _repo(tmp_path) + body_file: Final = tmp_path / "body.md" + + code: Final = sync.main( + [ + "--write", + "--openrouter-json", + str(FIXTURES / "openrouter_models.json"), + "--vercel-json", + str(FIXTURES / "vercel_models.json"), + "--pr-body-file", + str(body_file), + "--repo-root", + str(repo), + ] + ) + + root: Final = (repo / "model_prices_and_context_window.json").read_text() + backup: Final = (repo / "litellm" / "model_prices_and_context_window_backup.json").read_text() + assert code == 0 + assert root == backup + assert root.endswith("}\n") + assert json.loads(root)["openrouter/inception/mercury-2.5-preview"]["input_cost_per_token"] == 4e-8 + assert "### Added (4)" in body_file.read_text() + assert capsys.readouterr().out.startswith("openrouter: added=4 updated=1 warnings=0") + + +def test_dry_run_touches_nothing(sync: ModuleType, tmp_path: Path, capsys) -> None: + repo: Final = _repo(tmp_path) + before: Final = (repo / "model_prices_and_context_window.json").read_bytes() + + code: Final = sync.main( + [ + "--openrouter-json", + str(FIXTURES / "openrouter_models.json"), + "--vercel-json", + str(FIXTURES / "vercel_models.json"), + "--repo-root", + str(repo), + ] + ) + + assert code == 0 + assert (repo / "model_prices_and_context_window.json").read_bytes() == before + assert "dry run: no files were touched" in capsys.readouterr().out + + +def test_new_entries_inherit_model_intrinsic_traits_from_the_root_entry(sync: ModuleType) -> None: + root: Final = { + "litellm_provider": "anthropic", + "mode": "chat", + "input_cost_per_token": 5e-6, + "supports_adaptive_thinking": True, + "thinking_always_on": True, + "supports_sampling_params": False, + "supports_function_calling": False, + "supports_vision": True, + "prompt_cache_min_tokens": 1024, + "supports_web_search": True, + } + already_synced: Final = {"litellm_provider": "openrouter", "mode": "chat", "input_cost_per_token": 5e-6} + cost_map: Final = { + "claude-fable-5": dict(root), + "claude-fable-5-1": {**root, "prompt_cache_min_tokens": 512}, + "claude-embed-5": {**root, "mode": "embedding"}, + "gpt-5": {"litellm_provider": "openai", "mode": "chat", "supports_system_messages": True}, + "gpt-5-codex": {"litellm_provider": "openai", "mode": "responses", "supports_system_messages": False}, + "openrouter/anthropic/claude-fable-5:thinking": dict(already_synced), + } + openrouter: Final = sync.load_openrouter( + _openrouter_rows( + {"id": "anthropic/claude-fable-5:batch", "supported_parameters": ["tools"]}, + {"id": "anthropic/claude-fable-5:thinking"}, + {"id": "anthropic/claude-embed-5"}, + {"id": "anthropic/claude-embed-5-fast"}, + {"id": "anthropic/claude-opus-6"}, + ) + ) + vercel: Final = sync.load_vercel( + _vercel_rows( + {"id": "anthropic/claude-fable-5.1"}, + {"id": "anthropic/claude-fable-5.1-fast"}, + {"id": "openai/gpt-5-codex"}, + ), + now_ms=NOW_MS, + ) + + outcome: Final = sync.compute_sync(cost_map, (openrouter, vercel)) + + batch: Final = outcome.cost_map["openrouter/anthropic/claude-fable-5:batch"] + assert (batch["supports_adaptive_thinking"], batch["thinking_always_on"]) == (True, True) + assert (batch["supports_sampling_params"], batch["prompt_cache_min_tokens"]) == (False, 1024) + assert batch["supports_function_calling"] is True + assert not {"supports_web_search", "supports_vision"} & batch.keys() + assert outcome.cost_map["vercel_ai_gateway/anthropic/claude-fable-5.1"]["prompt_cache_min_tokens"] == 512 + fast: Final = outcome.cost_map["vercel_ai_gateway/anthropic/claude-fable-5.1-fast"] + assert (fast["prompt_cache_min_tokens"], fast["supports_adaptive_thinking"]) == (512, True) + assert "prompt_cache_min_tokens" not in outcome.cost_map["openrouter/anthropic/claude-opus-6"] + assert outcome.cost_map["openrouter/anthropic/claude-embed-5"]["supports_adaptive_thinking"] is True + assert "supports_adaptive_thinking" not in outcome.cost_map["openrouter/anthropic/claude-embed-5-fast"] + assert "supports_adaptive_thinking" not in outcome.cost_map["openrouter/anthropic/claude-fable-5:thinking"] + assert outcome.cost_map["vercel_ai_gateway/openai/gpt-5-codex"]["supports_system_messages"] is False + + +@pytest.mark.parametrize( + ("field", "old", "new", "reason"), + [ + ("input_cost_per_token", 1e-6, 0.0, "a price crossing zero"), + ("input_cost_per_token", 0.0, 1e-6, "a price crossing zero"), + ("output_cost_per_token", 1e-7, 2e-6, "a price moving more than 10x"), + ("output_cost_per_token", 2e-6, 1e-7, "a price moving more than 10x"), + ("max_input_tokens", 200000, 128000, "a shrinking limit"), + ("supports_reasoning", False, True, "a capability flag curated as false"), + ], +) +def test_out_of_bounds_changes_are_held_back_as_warnings( + sync: ModuleType, field: str, old: float, new: float, reason: str +) -> None: + existing: Final = { + "litellm_provider": "openrouter", + "mode": "chat", + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "max_input_tokens": 200000, + field: old, + } + catalog_row: Final = { + "id": "acme/x", + "context_length": 200000, + "pricing": {"prompt": "0.000001", "completion": "0.000002"}, + **({"context_length": int(new)} if field == "max_input_tokens" else {}), + **({"supported_parameters": ["reasoning"]} if field == "supports_reasoning" else {}), + } + catalog_row["pricing"] = { + **catalog_row["pricing"], + **({"prompt": str(new)} if field == "input_cost_per_token" else {}), + **({"completion": str(new)} if field == "output_cost_per_token" else {}), + } + + outcome: Final = sync.compute_sync( + {"openrouter/acme/x": dict(existing)}, (sync.load_openrouter(_openrouter_rows(catalog_row)),) + ) + + assert outcome.cost_map["openrouter/acme/x"] == existing + assert outcome.has_changes is False + assert outcome.providers[0].warnings == (f"openrouter/acme/x: {field}: {old!r} -> {new!r} held back: {reason}",) + + +def test_a_price_move_within_ten_x_is_applied(sync: ModuleType) -> None: + existing: Final = {"litellm_provider": "openrouter", "mode": "chat", "input_cost_per_token": 1e-6} + catalog: Final = sync.load_openrouter( + _openrouter_rows({"id": "acme/x", "pricing": {"prompt": "0.000009", "completion": "0"}}) + ) + + outcome: Final = sync.compute_sync({"openrouter/acme/x": existing}, (catalog,)) + + assert outcome.cost_map["openrouter/acme/x"]["input_cost_per_token"] == 9e-6 + assert outcome.providers[0].warnings == () + + +def test_vercel_long_context_tiers_map_to_above_threshold_prices(sync: ModuleType) -> None: + pricing: Final = { + "input": "0.0000015", + "input_tiers": [ + {"cost": "0.0000015", "min": 0, "max": 32001}, + {"cost": "0.0000027", "min": 32001, "max": 128001}, + {"cost": "0.0000045", "min": 128001}, + ], + "output": "0.0000075", + "output_tiers": [{"cost": "0.0000075", "max": 200001}, {"cost": "0.00001125", "min": 200001}], + "input_cache_read": "0.0000003", + "input_cache_read_tiers": [{"cost": "0.0000006", "min": 256000}], + "input_cache_write": "0.000002", + "input_cache_write_tiers": [{"cost": "0.000002", "min": 0, "max": 200001}, {"cost": "0.000004", "min": 200001}], + } + catalog: Final = sync.load_vercel(_vercel_rows({"id": "acme/long", "pricing": pricing}), now_ms=NOW_MS) + + outcome: Final = sync.compute_sync({}, (catalog,)) + + entry: Final = outcome.cost_map["vercel_ai_gateway/acme/long"] + assert entry["input_cost_per_token"] == 1.5e-6 + assert entry["input_cost_per_token_above_32k_tokens"] == 2.7e-6 + assert entry["input_cost_per_token_above_128k_tokens"] == 4.5e-6 + assert entry["output_cost_per_token_above_200k_tokens"] == 1.125e-5 + assert entry["cache_read_input_token_cost_above_256k_tokens"] == 6e-7 + assert entry["cache_creation_input_token_cost_above_200k_tokens"] == 4e-6 + assert not any(key.endswith("_above_0k_tokens") for key in entry) + assert entry["input_cost_per_token_above_200k_tokens"] == 4.5e-6 + assert entry["output_cost_per_token_above_128k_tokens"] == 7.5e-6 + assert entry["cache_read_input_token_cost_above_200k_tokens"] == 3e-7 + billed: Final = { + prompt_tokens: generic_cost_per_token( + model="acme/long", + usage=Usage(prompt_tokens=prompt_tokens, completion_tokens=1000, total_tokens=prompt_tokens + 1000), + custom_llm_provider="vercel_ai_gateway", + model_info=cast(ModelInfo, dict(entry)), + ) + for prompt_tokens in (100_000, 250_000, 300_000) + } + assert billed == { + 100_000: (pytest.approx(0.27), pytest.approx(0.0075)), + 250_000: (pytest.approx(1.125), pytest.approx(0.01125)), + 300_000: (pytest.approx(1.35), pytest.approx(0.01125)), + } + + +@pytest.mark.parametrize( + ("tiers", "problem"), + [ + ( + [{"cost": "0.000001", "min": 0, "max": 150500}, {"cost": "0.000002", "min": 150500}], + "input_cost_per_token tiers have a boundary that is not a whole thousand or an unusable price", + ), + ( + [{"cost": "0.000001", "min": 0, "max": 128000}, {"cost": "0.000002", "min": 200000}], + "input_cost_per_token tiers are not contiguous", + ), + ( + [{"cost": "0.000001", "min": 0, "max": 128000}, {"cost": "0.000002", "min": 128000, "max": 256000}], + "input_cost_per_token tiers are not contiguous", + ), + ], +) +def test_unmappable_tiers_skip_the_row_with_a_warning(sync: ModuleType, tiers: list, problem: str) -> None: + pricing: Final = {"input": "0.000001", "input_tiers": tiers, "output": "0.000002"} + catalog: Final = sync.load_vercel(_vercel_rows({"id": "acme/odd", "pricing": pricing}), now_ms=NOW_MS) + + outcome: Final = sync.compute_sync({}, (catalog,)) + + assert "vercel_ai_gateway/acme/odd" not in outcome.cost_map + assert dict(catalog.skipped) == {"tiers outside the registry's thresholds": 1} + assert outcome.providers[0].warnings == (f"vercel_ai_gateway/acme/odd: {problem}; row skipped",) + + +def test_a_price_that_varies_by_provider_seeds_and_only_overwrites_what_the_bot_seeded(sync: ModuleType) -> None: + curated: Final = { + "litellm_provider": "vercel_ai_gateway", + "mode": "chat", + "input_cost_per_token": 9e-7, + "output_cost_per_token": 2e-6, + "max_input_tokens": 100000, + } + seeded: Final = {**curated, "price_varies_by_provider": True} + cited: Final = {**curated, "source": "https://vercel.com/ai-gateway/models/cited"} + row: Final = { + "context_window": 262144, + "pricing": { + "input": "0.0000015", + "input_tiers": [{"cost": "0.0000015", "min": 0, "max": 128001}, {"cost": "0.000003", "min": 128001}], + "output": "0.000002", + "input_cache_read": "0.0000003", + "varies_by_provider": True, + }, + } + catalog: Final = sync.load_vercel( + _vercel_rows( + {"id": "acme/curated", **row}, + {"id": "acme/seeded", **row}, + {"id": "acme/fresh", **row}, + {"id": "acme/cited", **row}, + ), + now_ms=NOW_MS, + ) + + outcome: Final = sync.compute_sync( + { + "vercel_ai_gateway/acme/curated": dict(curated), + "vercel_ai_gateway/acme/seeded": dict(seeded), + "vercel_ai_gateway/acme/cited": dict(cited), + }, + (catalog,), + ) + + for key in ("vercel_ai_gateway/acme/curated", "vercel_ai_gateway/acme/cited"): + held: Final = outcome.cost_map[key] + assert (held["input_cost_per_token"], held["max_input_tokens"]) == (9e-7, 262144) + assert not any("cache_read" in name or "_above_" in name for name in held) + assert "price_varies_by_provider" not in held + resynced: Final = outcome.cost_map["vercel_ai_gateway/acme/seeded"] + assert (resynced["input_cost_per_token"], resynced["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6) + fresh: Final = outcome.cost_map["vercel_ai_gateway/acme/fresh"] + assert (fresh["input_cost_per_token"], fresh["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6) + assert fresh["cache_read_input_token_cost"] == 3e-7 + assert fresh["price_varies_by_provider"] is True + held_line: Final = ( + ": input_cost_per_token: 9e-07 -> 1.5e-06 held back: " + "the catalog price varies by provider; cache_read_input_token_cost: None -> 3e-07 held back: " + "the catalog price varies by provider; input_cost_per_token_above_128k_tokens: None -> 3e-06 held back: " + "the catalog price varies by provider" + ) + assert outcome.providers[0].warnings == ( + f"vercel_ai_gateway/acme/cited{held_line}", + f"vercel_ai_gateway/acme/curated{held_line}", + ) + + +def test_every_key_the_sync_writes_is_classified_by_the_schema_generator(sync: ModuleType) -> None: + spec: Final = importlib.util.spec_from_file_location("generate_model_prices_schema", GENERATOR_PATH) + assert spec is not None and spec.loader is not None + generator: Final = importlib.util.module_from_spec(spec) + spec.loader.exec_module(generator) + varies: Final = sync.load_vercel( + _vercel_rows( + { + "id": "acme/varies", + "context_window": 262144, + "pricing": { + "input": "0.0000015", + "input_tiers": [ + {"cost": "0.0000015", "min": 0, "max": 128001}, + {"cost": "0.000003", "min": 128001}, + ], + "output": "0.000002", + "input_cache_read": "0.0000003", + "varies_by_provider": True, + }, + } + ), + now_ms=NOW_MS, + ) + synced: Final = sync.compute_sync( + _base_map(), (sync.load_openrouter(OPENROUTER_RAW), sync.load_vercel(VERCEL_RAW, now_ms=NOW_MS), varies) + ).cost_map + + assert synced["vercel_ai_gateway/acme/varies"]["price_varies_by_provider"] is True + assert generator.validation_errors(synced, generator.build_schema(synced)) == () + + +def test_image_and_audio_outputs_are_priced_per_token_or_skipped(sync: ModuleType) -> None: + openrouter: Final = sync.load_openrouter( + _openrouter_rows( + { + "id": "openai/gpt-5-image", + "architecture": {"input_modalities": ["text", "image"], "output_modalities": ["image", "text"]}, + "pricing": {"prompt": "0.00001", "completion": "0.00001", "image_output": "0.00004"}, + }, + { + "id": "acme/talker", + "architecture": {"input_modalities": ["text", "audio"], "output_modalities": ["text", "audio"]}, + "pricing": { + "prompt": "0.000001", + "completion": "0.000002", + "audio": "0.000005", + "audio_output": "0.00001", + }, + }, + { + "id": "acme/mute", + "architecture": {"input_modalities": ["text"], "output_modalities": ["text", "audio"]}, + }, + {"id": "acme/painter", "architecture": {"output_modalities": ["image"]}}, + ) + ) + vercel: Final = sync.load_vercel( + _vercel_rows( + { + "id": "acme/speaker", + "modalities": {"input": ["text", "audio"], "output": ["text", "audio"]}, + "pricing": { + "input": "0.000001", + "output": "0.000002", + "audio_input_token_cost": "0.000004", + "audio_output_token_cost": "0.000008", + }, + }, + {"id": "acme/drawer", "modalities": {"input": ["text"], "output": ["text", "image"]}}, + ), + now_ms=NOW_MS, + ) + + outcome: Final = sync.compute_sync({}, (openrouter, vercel)) + + image: Final = outcome.cost_map["openrouter/openai/gpt-5-image"] + assert (image["output_cost_per_image_token"], image["mode"], image["supports_vision"]) == (4e-5, "chat", True) + talker: Final = outcome.cost_map["openrouter/acme/talker"] + assert (talker["input_cost_per_audio_token"], talker["output_cost_per_audio_token"]) == (5e-6, 1e-5) + assert (talker["supports_audio_input"], talker["supports_audio_output"]) == (True, True) + speaker: Final = outcome.cost_map["vercel_ai_gateway/acme/speaker"] + assert (speaker["input_cost_per_audio_token"], speaker["output_cost_per_audio_token"]) == (4e-6, 8e-6) + assert speaker["supports_audio_output"] is True + assert {"openrouter/acme/mute", "openrouter/acme/painter", "vercel_ai_gateway/acme/drawer"}.isdisjoint( + outcome.cost_map + ) + assert dict(openrouter.skipped) == {"output priced outside the catalog": 2} + assert dict(vercel.skipped) == {"output priced outside the catalog": 1} + + +def test_updates_keep_the_curated_key_order_and_append_new_keys(sync: ModuleType) -> None: + curated: Final = { + "mode": "chat", + "output_cost_per_token": 2e-6, + "input_cost_per_token": 1e-6, + "litellm_provider": "openrouter", + } + catalog: Final = sync.load_openrouter( + _openrouter_rows( + { + "id": "acme/x", + "pricing": {"prompt": "0.000003", "completion": "0.000002", "input_cache_read": "0.0000001"}, + } + ) + ) + + outcome: Final = sync.compute_sync({"openrouter/acme/x": dict(curated)}, (catalog,)) + + assert list(outcome.cost_map["openrouter/acme/x"]) == [*curated, "cache_read_input_token_cost"] + assert outcome.cost_map["openrouter/acme/x"]["input_cost_per_token"] == 3e-6 diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index c7e46829aba..aed37efd221 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -938,6 +938,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_audio_token_cost": {"type": "number"}, "cache_creation_input_token_cost": {"type": "number"}, "cache_creation_input_token_cost_above_1hr": {"type": "number"}, + "cache_creation_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_256k_tokens": {"type": "number"}, @@ -951,6 +952,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_token_cost_flex": {"type": "number"}, "cache_creation_input_token_cost_priority": {"type": "number"}, "cache_read_input_token_cost": {"type": "number"}, + "cache_read_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_256k_tokens": {"type": "number"}, @@ -1006,6 +1008,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "input_cost_per_request": {"type": "number"}, "input_cost_per_second": {"type": "number"}, "input_cost_per_token": {"type": "number"}, + "input_cost_per_token_above_32k_tokens": {"type": "number"}, "input_cost_per_token_above_128k_tokens": {"type": "number"}, "input_cost_per_token_batches": {"type": "number"}, "input_cost_per_token_cache_hit": {"type": "number"}, @@ -1062,6 +1065,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "output_cost_per_second_1080p": {"type": "number"}, "output_cost_per_second_4k": {"type": "number"}, "output_cost_per_token": {"type": "number"}, + "output_cost_per_token_above_32k_tokens": {"type": "number"}, "output_cost_per_token_above_128k_tokens": {"type": "number"}, "output_cost_per_token_above_200k_tokens": {"type": "number"}, "output_cost_per_token_above_256k_tokens": {"type": "number"}, @@ -1075,6 +1079,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "output_vector_size": {"type": "number"}, "rpd": {"type": "number"}, "rpm": {"type": "number"}, + "price_varies_by_provider": {"type": "boolean"}, "source": {"type": "string"}, "comment": {"type": "string"}, "supports_assistant_prefill": {"type": "boolean"}, @@ -1236,6 +1241,11 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): }, }, }, + "patternProperties": { + "^(input_cost_per_token|output_cost_per_token|cache_read_input_token_cost|cache_creation_input_token_cost)_above_[0-9]+k_tokens$": { + "type": "number" + }, + }, "additionalProperties": False, }, } @@ -2871,9 +2881,8 @@ def test_model_info_for_openrouter_kimi_k2_5(): Test that openrouter/moonshotai/kimi-k2.5 model info is correctly configured in model_prices_and_context_window.json. - Model properties from OpenRouter API: - - context_length: 262144 - - pricing: prompt=$0.00000045, completion=$0.00000225, input_cache_read=$0.00000007 + scripts/sync_cost_map.py keeps the limits and prices aligned with OpenRouter's + catalog, so they are checked for presence, not pinned. - modality: text+image->text (supports vision) - supports: tool_choice, tools (function calling) """ @@ -2892,15 +2901,13 @@ def test_model_info_for_openrouter_kimi_k2_5(): assert model_info["litellm_provider"] == "openrouter" assert model_info["mode"] == "chat" - # Verify context window - assert model_info["max_input_tokens"] == 262144 - assert model_info["max_output_tokens"] == 262144 - assert model_info["max_tokens"] == 262144 + assert model_info["max_input_tokens"] > 0 + assert model_info["max_output_tokens"] > 0 + assert model_info["max_tokens"] == model_info["max_output_tokens"] - # Verify pricing - assert model_info["input_cost_per_token"] == 4.5e-07 - assert model_info["output_cost_per_token"] == 2.25e-06 - assert model_info["cache_read_input_token_cost"] == 7e-08 + assert model_info["input_cost_per_token"] > 0 + assert model_info["output_cost_per_token"] > 0 + assert model_info["cache_read_input_token_cost"] > 0 # Verify capabilities assert model_info["supports_vision"] is True