From 58b3037e4b7e3d260630d2817a93ee16834e1424 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:00:08 -0700 Subject: [PATCH] fix(sync-cost-map): map vercel tiers, inherit family traits, hold out-of-bounds changes, and reconcile the open bot PR Vercel long-context tiers become *_above_k_tokens keys when contiguous on a whole thousand, and a row whose tiers do not fit is skipped with a warning. Image and audio output are priced per token, and a row with an unpriced non-text output is skipped instead of billed as text. A new entry inherits the traits no catalog expresses (cache minimum, adaptive thinking, sampling params, system messages, thinking always on) from its same-mode root, found by the bare name or its longest dash prefix. The max_tokens / max_output_tokens pair moves as a unit. Shrinking limits, prices crossing zero or moving more than 10x, and every price on an already-priced varies_by_provider row are held back and listed as warnings for a human commit. Updated entries keep their curated key order with new keys appended sorted. The workflow's own token is read-only and every write uses the GitHub App token; without the App a scheduled run explains why it cannot open a PR. Each tick first reconciles the open bot PR: a conflicting one is closed and re-synced, a green one is merged, a red one is left for a human, and a sync only runs when none is open. The sync step runs with --no-dev and only when it will be used. The hardcoded map schema in test_utils.py gains the 32k tier keys the synced map now carries. --- .github/workflows/cost-map-sync.yml | 119 ++++---- scripts/sync_cost_map.py | 364 ++++++++++++++++++----- tests/test_litellm/test_sync_cost_map.py | 341 ++++++++++++++++++++- tests/test_litellm/test_utils.py | 4 + 4 files changed, 683 insertions(+), 145 deletions(-) diff --git a/.github/workflows/cost-map-sync.yml b/.github/workflows/cost-map-sync.yml index 8fc4f343083..005e93b786c 100644 --- a/.github/workflows/cost-map-sync.yml +++ b/.github/workflows/cost-map-sync.yml @@ -11,8 +11,8 @@ on: default: false permissions: - contents: write - pull-requests: write + contents: read + pull-requests: read concurrency: group: cost-map-sync @@ -39,44 +39,72 @@ jobs: with: app-id: ${{ secrets.COST_MAP_BOT_APP_ID }} private-key: ${{ secrets.COST_MAP_BOT_PRIVATE_KEY }} + - name: Reconcile the open sync PR + id: open + run: | + pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --json number,headRefName,mergeable \ + --search "in:title \"$PR_TITLE\"" \ + --jq "[.[] | select(.headRefName | startswith(\"$BRANCH_PREFIX\"))] | first // empty")" + sync=false + if [ -z "$pr" ]; then + sync=true + elif [ -z "$BOT_APP_ID" ]; then + echo "::warning::Sync PR #$(jq -r .number <<< "$pr") is open and COST_MAP_BOT_APP_ID is not configured; leaving it to a human." + elif [ "$(jq -r .mergeable <<< "$pr")" = "CONFLICTING" ]; then + number="$(jq -r .number <<< "$pr")" + gh pr close "$number" --repo "$GITHUB_REPOSITORY" --delete-branch \ + --comment "This sync no longer merges cleanly against $GITHUB_REF_NAME, so the next scheduled run opens a fresh one." + echo "Closed conflicting sync PR #$number." + sync=true + else + number="$(jq -r .number <<< "$pr")" + guard="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json name,state \ + --jq '.[] | select(.name == "cost-map-guard") | .state' || true)" + required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --required --json bucket \ + --jq 'map(.bucket) | unique | join(",")' || true)" + case "$guard,$required" in + SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping) + gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch + echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry." + ;; + *FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*) + echo "::warning::Sync PR #$number has a failing check (cost-map-guard=$guard, required buckets=$required); leaving it open for a human." + ;; + *) + echo "Sync PR #$number is still being checked (cost-map-guard=$guard, required buckets=$required)." + ;; + esac + fi + echo "sync=$sync" >> "$GITHUB_OUTPUT" + env: + GH_TOKEN: ${{ steps.bot.outputs.token || github.token }} + - name: Explain why no PR can be opened + if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID == '' && !inputs.dry_run + run: echo "::warning::COST_MAP_BOT_APP_ID is not configured, so no sync PR can be opened or merged; dispatch with dry_run to see the diff." - name: Set up uv + if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run) uses: ./.github/actions/setup-uv-with-retries with: version: "0.10.9" - - name: Look for an already-open sync PR - id: existing - run: | - open_pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --json headRefName \ - --search "in:title \"$PR_TITLE\"" \ - --jq "[.[].headRefName | select(startswith(\"$BRANCH_PREFIX\"))] | first // empty")" - echo "open_pr=$open_pr" >> "$GITHUB_OUTPUT" - if [ -n "$open_pr" ]; then - echo "An open sync PR already exists on branch $open_pr; skipping this run." - fi - env: - GH_TOKEN: ${{ steps.bot.outputs.token || secrets.GH_TOKEN || github.token }} - name: Run the sync - if: steps.existing.outputs.open_pr == '' - run: | - uv run --frozen python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md" - uv run --frozen python ci_cd/generate_model_prices_schema.py - - name: Open the sync PR - id: pr - if: steps.existing.outputs.open_pr == '' && !inputs.dry_run + id: sync + if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run) run: | + uv run --frozen --no-dev python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md" + uv run --frozen --no-dev python ci_cd/generate_model_prices_schema.py if git diff --quiet; then echo "Registry already in sync; no PR needed." - exit 0 - fi - branch="${BRANCH_PREFIX}$(date -u +'%Y-%m-%d-%H%M')" - if [ -n "$BOT_APP_ID" ]; then - bot_user_id="$(gh api "users/${BOT_LOGIN}[bot]" --jq .id)" - git config user.name "${BOT_LOGIN}[bot]" - git config user.email "${bot_user_id}+${BOT_LOGIN}[bot]@users.noreply.github.com" + echo "changed=false" >> "$GITHUB_OUTPUT" else - git config user.name "github-actions[bot]" - git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + echo "changed=true" >> "$GITHUB_OUTPUT" fi + - name: Open the sync PR + if: steps.sync.outputs.changed == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run + run: | + branch="${BRANCH_PREFIX}$(date -u +'%Y-%m-%d-%H%M')" + bot_user_id="$(gh api "users/${BOT_LOGIN}[bot]" --jq .id)" + git config user.name "${BOT_LOGIN}[bot]" + git config user.email "${bot_user_id}+${BOT_LOGIN}[bot]@users.noreply.github.com" git checkout -b "$branch" git add model_prices_and_context_window.json \ litellm/model_prices_and_context_window_backup.json \ @@ -84,35 +112,10 @@ jobs: git commit -m "feat(models): sync openrouter and vercel_ai_gateway pricing $(date -u +'%Y-%m-%d %H:%M')" gh auth setup-git git push origin "$branch" - url="$(gh pr create --title "$PR_TITLE" \ + gh pr create --title "$PR_TITLE" \ --body-file "$RUNNER_TEMP/pr_body.md" \ --head "$branch" \ - --base "$GITHUB_REF_NAME")" - echo "url=$url" >> "$GITHUB_OUTPUT" - env: - GH_TOKEN: ${{ steps.bot.outputs.token || secrets.GH_TOKEN || github.token }} - BOT_LOGIN: ${{ steps.bot.outputs.app-slug }} - - name: Merge once every required check passes - if: steps.pr.outputs.url != '' && env.BOT_APP_ID != '' - timeout-minutes: 120 - run: | - while true; do - guard="$(gh pr checks "$PR_URL" --json name,state \ - --jq '.[] | select(.name == "cost-map-guard") | .state' || true)" - required="$(gh pr checks "$PR_URL" --required --json bucket \ - --jq 'map(.bucket) | unique | join(",")' || true)" - case "$guard,$required" in - *FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*) - echo "A check failed (cost-map-guard=$guard, required buckets=$required); leaving $PR_URL open for a human." - exit 1 - ;; - SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping) - gh pr merge "$PR_URL" --repo "$GITHUB_REPOSITORY" --merge --delete-branch - exit 0 - ;; - esac - sleep 30 - done + --base "$GITHUB_REF_NAME" env: GH_TOKEN: ${{ steps.bot.outputs.token }} - PR_URL: ${{ steps.pr.outputs.url }} + BOT_LOGIN: ${{ steps.bot.outputs.app-slug }} diff --git a/scripts/sync_cost_map.py b/scripts/sync_cost_map.py index 956e7136a88..cb24bc36f5d 100644 --- a/scripts/sync_cost_map.py +++ b/scripts/sync_cost_map.py @@ -7,8 +7,20 @@ backup copy. Policy: - Both catalogs price per token as decimal strings; values are normalized to six significant digits. -- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed, a - capability flag the catalog does not claim stays as curated, and a curated output ceiling is kept. +- Vercel long-context tiers map to the registry's ``*_above_k_tokens`` keys, which litellm applies once the + prompt exceeds N thousand tokens. A row whose tier boundaries are not whole thousands is skipped with a warning. +- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry but never overwrites a + curated price, and a difference is reported as a warning. +- Image and audio output are priced from the catalog's per-token ``image_output`` and ``audio_output`` prices. A row + whose non-text output the catalog does not price per token is skipped. +- A new entry inherits the traits no catalog expresses (adaptive thinking, sampling params, cache minimums, system + messages) from the same model's root registry entry, found by the bare model name or its longest dash-prefix + with the same mode, so the family-wide invariants the test suite enforces hold for the route too. +- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed and a + capability flag the catalog does not claim stays as curated. ``max_output_tokens`` and ``max_tokens`` move as a + pair and only when the catalog states an output ceiling. +- A limit that would shrink, a price that would cross zero, and a price that would move more than 10x either way + are held back as warnings for a human instead of applied. - Router models and rows without a usable prompt and completion price are skipped. - A registry entry absent from its catalog is left untouched; retiring a model stays a human call. """ @@ -18,6 +30,7 @@ import json import math import sys import time +from collections import Counter from collections.abc import Mapping, Sequence from dataclasses import dataclass from functools import reduce @@ -35,12 +48,26 @@ COST_MAP_RELPATHS: Final = ( OPENROUTER_MODELS_URL: Final = "https://openrouter.ai/api/v1/models" VERCEL_MODELS_URL: Final = "https://ai-gateway.vercel.sh/v1/models" VERCEL_TYPE_TO_MODE: Final = MappingProxyType({"language": "chat", "embedding": "embedding"}) -ADD_ONLY_FIELDS: Final = frozenset({"max_output_tokens", "max_tokens"}) +LIMIT_PAIR: Final = ("max_output_tokens", "max_tokens") +PRICE_SWING_LIMIT: Final = 10 +INHERITED_TRAITS: Final = frozenset( + { + "prompt_cache_min_tokens", + "supports_adaptive_thinking", + "supports_sampling_params", + "supports_system_messages", + "thinking_always_on", + } +) PR_BODY_SECTION_LIMIT: Final = 30 Provider = Literal["openrouter", "vercel_ai_gateway"] RegistryEntry = dict[str, object] CostMap = dict[str, object] +Prices = Mapping[str, float] + +NO_PRICES: Final[Prices] = MappingProxyType({}) +NO_TRAITS: Final[Mapping[str, object]] = MappingProxyType({}) class SyncError(RuntimeError): @@ -53,10 +80,14 @@ class OpenRouterPricing(BaseModel): input_cache_read: str | None = None input_cache_write: str | None = None internal_reasoning: str | None = None + image_output: str | None = None + audio: str | None = None + audio_output: str | None = None class OpenRouterArchitecture(BaseModel): input_modalities: tuple[str, ...] | None = None + output_modalities: tuple[str, ...] | None = None class OpenRouterTopProvider(BaseModel): @@ -72,15 +103,29 @@ class OpenRouterModel(BaseModel): supported_parameters: tuple[str, ...] | None = None +class VercelTier(BaseModel): + cost: str + min: int | None = None + max: int | None = None + + class VercelPricing(BaseModel): input: str | None = None output: str | None = None input_cache_read: str | None = None input_cache_write: str | None = None + input_tiers: tuple[VercelTier, ...] | None = None + output_tiers: tuple[VercelTier, ...] | None = None + input_cache_read_tiers: tuple[VercelTier, ...] | None = None + input_cache_write_tiers: tuple[VercelTier, ...] | None = None + audio_input_token_cost: str | None = None + audio_output_token_cost: str | None = None + varies_by_provider: bool = False class VercelModalities(BaseModel): input: tuple[str, ...] | None = None + output: tuple[str, ...] | None = None class VercelModel(BaseModel): @@ -105,6 +150,18 @@ class CatalogEntry: mode: str source: str fields: Mapping[str, object] + indicative_prices: bool = False + + +@dataclass(frozen=True, slots=True) +class Skipped: + reason: str + warning: str | None = None + + +@dataclass(frozen=True, slots=True) +class Unmappable: + problem: str @dataclass(frozen=True, slots=True) @@ -112,6 +169,7 @@ class Catalog: provider: Provider entries: tuple[CatalogEntry, ...] skipped: Mapping[str, int] + warnings: tuple[str, ...] def per_token(price: float) -> float: @@ -130,18 +188,22 @@ def _extra_price(raw: str | None) -> float | None: return price if price else None -def _flags(parameters: Sequence[str] | None, modalities: Sequence[str] | None) -> Mapping[str, bool]: +def _flags( + parameters: Sequence[str] | None, inputs: Sequence[str] | None, outputs: Sequence[str] | None +) -> Mapping[str, bool]: params: Final = frozenset(parameters or ()) - mods: Final = frozenset(modalities or ()) + input_modalities: Final = frozenset(inputs or ()) + output_modalities: Final = frozenset(outputs or ()) claims: Final = { "supports_function_calling": "tools" in params, "supports_tool_choice": "tool_choice" in params, "supports_reasoning": "reasoning" in params, "supports_response_schema": "structured_outputs" in params, - "supports_vision": "image" in mods, - "supports_pdf_input": bool({"file", "pdf"} & mods), - "supports_audio_input": "audio" in mods, - "supports_video_input": "video" in mods, + "supports_vision": "image" in input_modalities, + "supports_pdf_input": bool({"file", "pdf"} & input_modalities), + "supports_audio_input": "audio" in input_modalities, + "supports_video_input": "video" in input_modalities, + "supports_audio_output": "audio" in output_modalities, } return MappingProxyType({name: True for name, claimed in claims.items() if claimed}) @@ -157,23 +219,85 @@ def _limits(max_input: int | None, max_output: int | None) -> Mapping[str, int]: ) -def _priced(name: str, price: float | None) -> Mapping[str, float]: +def _priced(name: str, price: float | None) -> Prices: return MappingProxyType({name: price} if price is not None else {}) -def _openrouter_entry(model: OpenRouterModel) -> CatalogEntry | None: - prompt: Final = _token_price(model.pricing.prompt) - completion: Final = _token_price(model.pricing.completion) +def _output_prices( + outputs: Sequence[str] | None, image_price: float | None, audio_price: float | None +) -> Prices | Skipped: + modalities: Final = frozenset(outputs or ("text",)) + known: Final = { + modality: price for modality, price in (("image", image_price), ("audio", audio_price)) if price is not None + } + if "text" not in modalities or not (modalities - {"text"}) <= known.keys(): + return Skipped("output priced outside the catalog") + names: Final = {"image": "output_cost_per_image_token", "audio": "output_cost_per_audio_token"} + return MappingProxyType({names[modality]: known[modality] for modality in modalities & known.keys()}) + + +def _tier_threshold(boundary: int) -> int | None: + return next((start // 1000 for start in (boundary, boundary - 1) if start > 0 and start % 1000 == 0), None) + + +def _tiered(name: str, base: float | None, tiers: Sequence[VercelTier] | None) -> Prices | Unmappable: + if base is None or not tiers: + return NO_PRICES + ordered: Final = sorted(tiers, key=lambda tier: tier.min or 0) + contiguous: Final = ordered[-1].max is None and all( + lower.max == upper.min for lower, upper in zip(ordered, ordered[1:], strict=False) + ) + if not contiguous: + return Unmappable(f"{name} tiers are not contiguous") + steps: Final = tuple((_tier_threshold(tier.min), _token_price(tier.cost)) for tier in ordered if tier.min) + prices: Final = { + f"{name}_above_{thousands}k_tokens": price + for thousands, price in steps + if thousands is not None and price is not None + } + if len(prices) != len(steps): + return Unmappable(f"{name} tiers have a boundary that is not a whole thousand or an unusable price") + return MappingProxyType(prices) + + +def _vercel_tiers(pricing: VercelPricing, cache_read: float | None, cache_write: float | None) -> Prices | Unmappable: + parts: Final = ( + _tiered("input_cost_per_token", _token_price(pricing.input), pricing.input_tiers), + _tiered("output_cost_per_token", _token_price(pricing.output), pricing.output_tiers), + _tiered("cache_read_input_token_cost", cache_read, pricing.input_cache_read_tiers), + _tiered("cache_creation_input_token_cost", cache_write, pricing.input_cache_write_tiers), + ) + problem: Final = next((part for part in parts if isinstance(part, Unmappable)), None) + if problem is not None: + return problem + return MappingProxyType( + {name: price for part in parts if not isinstance(part, Unmappable) for name, price in part.items()} + ) + + +def _openrouter_entry(model: OpenRouterModel) -> CatalogEntry | Skipped: + pricing: Final = model.pricing + prompt: Final = _token_price(pricing.prompt) + completion: Final = _token_price(pricing.completion) if prompt is None or completion is None: - return None + return Skipped("unpriced or router") + inputs: Final = model.architecture.input_modalities if model.architecture else None + outputs: Final = model.architecture.output_modalities if model.architecture else None + output_prices: Final = _output_prices( + outputs, _extra_price(pricing.image_output), _extra_price(pricing.audio_output) + ) + if isinstance(output_prices, Skipped): + return output_prices fields: Final = { "input_cost_per_token": prompt, "output_cost_per_token": completion, **_limits(model.context_length, model.top_provider.max_completion_tokens), - **_priced("cache_read_input_token_cost", _extra_price(model.pricing.input_cache_read)), - **_priced("cache_creation_input_token_cost", _extra_price(model.pricing.input_cache_write)), - **_priced("output_cost_per_reasoning_token", _extra_price(model.pricing.internal_reasoning)), - **_flags(model.supported_parameters, model.architecture.input_modalities if model.architecture else None), + **_priced("cache_read_input_token_cost", _extra_price(pricing.input_cache_read)), + **_priced("cache_creation_input_token_cost", _extra_price(pricing.input_cache_write)), + **_priced("output_cost_per_reasoning_token", _extra_price(pricing.internal_reasoning)), + **_priced("input_cost_per_audio_token", _extra_price(pricing.audio)), + **output_prices, + **_flags(model.supported_parameters, inputs, outputs), } return CatalogEntry( key=f"openrouter/{model.id}", @@ -184,30 +308,46 @@ def _openrouter_entry(model: OpenRouterModel) -> CatalogEntry | None: ) -def _vercel_entry(model: VercelModel) -> CatalogEntry | None: +def _vercel_entry(model: VercelModel, now_ms: int) -> CatalogEntry | Skipped: + if model.deprecated_at is not None and model.deprecated_at <= now_ms: + return Skipped("deprecated") mode: Final = VERCEL_TYPE_TO_MODE.get(model.type) - prompt: Final = _token_price(model.pricing.input) - completion: Final = _token_price(model.pricing.output if mode != "embedding" else model.pricing.output or "0") - if mode is None or prompt is None or completion is None: - return None + if mode is None: + return Skipped("not token priced") + pricing: Final = model.pricing + prompt: Final = _token_price(pricing.input) + completion: Final = _token_price(pricing.output if mode != "embedding" else pricing.output or "0") + if prompt is None or completion is None: + return Skipped("no usable price") + key: Final = f"vercel_ai_gateway/{model.id}" + inputs: Final = model.modalities.input if model.modalities else None + outputs: Final = model.modalities.output if model.modalities else None + output_prices: Final = _output_prices(outputs, None, _extra_price(pricing.audio_output_token_cost)) + if isinstance(output_prices, Skipped): + return output_prices + cache_read: Final = _extra_price(pricing.input_cache_read) + cache_write: Final = _extra_price(pricing.input_cache_write) + tiers: Final = _vercel_tiers(pricing, cache_read, cache_write) + if isinstance(tiers, Unmappable): + return Skipped("tiers outside the registry's thresholds", warning=f"{key}: {tiers.problem}; row skipped") fields: Final = { "input_cost_per_token": prompt, "output_cost_per_token": completion, **_limits(model.context_window, model.max_tokens), - **_priced("cache_read_input_token_cost", _extra_price(model.pricing.input_cache_read)), - **_priced("cache_creation_input_token_cost", _extra_price(model.pricing.input_cache_write)), - **( - _flags(model.supported_parameters, model.modalities.input if model.modalities else None) - if mode == "chat" - else {} - ), + **_priced("cache_read_input_token_cost", cache_read), + **_priced("cache_creation_input_token_cost", cache_write), + **_priced("input_cost_per_audio_token", _extra_price(pricing.audio_input_token_cost)), + **output_prices, + **tiers, + **(_flags(model.supported_parameters, inputs, outputs) if mode == "chat" else {}), } return CatalogEntry( - key=f"vercel_ai_gateway/{model.id}", + key=key, provider="vercel_ai_gateway", mode=mode, source=f"https://vercel.com/ai-gateway/models/{model.id.rsplit('/', 1)[-1]}", fields=MappingProxyType(fields), + indicative_prices=pricing.varies_by_provider, ) @@ -219,17 +359,21 @@ def _rows(raw: bytes, url: str) -> object: return rows +def _catalog(provider: Provider, rows: Sequence[CatalogEntry | Skipped]) -> Catalog: + return Catalog( + provider=provider, + entries=tuple(row for row in rows if isinstance(row, CatalogEntry)), + skipped=MappingProxyType(Counter(row.reason for row in rows if isinstance(row, Skipped))), + warnings=tuple(row.warning for row in rows if isinstance(row, Skipped) and row.warning is not None), + ) + + def load_openrouter(raw: bytes) -> Catalog: try: models: Final = OPENROUTER_ADAPTER.validate_python(_rows(raw, OPENROUTER_MODELS_URL)) except ValidationError as error: raise SyncError(f"the OpenRouter catalog no longer matches the expected shape: {error}") from error - entries: Final = tuple(entry for entry in map(_openrouter_entry, models) if entry is not None) - return Catalog( - provider="openrouter", - entries=entries, - skipped=MappingProxyType({"unpriced or router": len(models) - len(entries)}), - ) + return _catalog("openrouter", tuple(map(_openrouter_entry, models))) def load_vercel(raw: bytes, now_ms: int) -> Catalog: @@ -237,20 +381,7 @@ def load_vercel(raw: bytes, now_ms: int) -> Catalog: models: Final = VERCEL_ADAPTER.validate_python(_rows(raw, VERCEL_MODELS_URL)) except ValidationError as error: raise SyncError(f"the Vercel AI Gateway catalog no longer matches the expected shape: {error}") from error - live: Final = tuple(model for model in models if model.deprecated_at is None or model.deprecated_at > now_ms) - token_priced: Final = tuple(model for model in live if model.type in VERCEL_TYPE_TO_MODE) - entries: Final = tuple(entry for entry in map(_vercel_entry, token_priced) if entry is not None) - return Catalog( - provider="vercel_ai_gateway", - entries=entries, - skipped=MappingProxyType( - { - "deprecated": len(models) - len(live), - "not token priced": len(live) - len(token_priced), - "no usable price": len(token_priced) - len(entries), - } - ), - ) + return _catalog("vercel_ai_gateway", tuple(_vercel_entry(model, now_ms) for model in models)) @dataclass(frozen=True, slots=True) @@ -272,10 +403,34 @@ class SyncOutcome: return any(outcome.added or outcome.updated for outcome in self.providers) -def _new_entry(entry: CatalogEntry) -> RegistryEntry: +def _root_candidates(bare: str) -> tuple[str, ...]: + segments: Final = bare.split("-") + stems: Final = tuple( + "-".join(segments[:count]) for count in range(len(segments), 0, -1) if count >= 2 or count == len(segments) + ) + return tuple(dict.fromkeys(name for stem in stems for name in (stem, stem.replace(".", "-")))) + + +def _inherited(cost_map: CostMap, entry: CatalogEntry) -> Mapping[str, object]: + bare: Final = entry.key.rsplit("/", 1)[-1].split(":", 1)[0] + root: Final = next( + ( + candidate + for candidate in map(cost_map.get, _root_candidates(bare)) + if isinstance(candidate, dict) and candidate.get("mode") == entry.mode + ), + None, + ) + if root is None: + return NO_TRAITS + return MappingProxyType({name: value for name, value in root.items() if name in INHERITED_TRAITS}) + + +def _new_entry(entry: CatalogEntry, inherited: Mapping[str, object]) -> RegistryEntry: return dict( sorted( { + **inherited, **entry.fields, "litellm_provider": entry.provider, "mode": entry.mode, @@ -285,15 +440,67 @@ def _new_entry(entry: CatalogEntry) -> RegistryEntry: ) -def _updated_entry(existing: RegistryEntry, entry: CatalogEntry) -> tuple[RegistryEntry, tuple[str, ...]]: - keep_limits: Final = not ADD_ONLY_FIELDS.isdisjoint(existing) - desired: Final = { - name: value for name, value in entry.fields.items() if not (keep_limits and name in ADD_ONLY_FIELDS) - } - changes: Final = tuple( - f"{name}: {existing.get(name)!r} -> {value!r}" for name, value in desired.items() if existing.get(name) != value +@dataclass(frozen=True, slots=True) +class FieldChange: + name: str + old: object + new: object + hold: str | None + + @property + def line(self) -> str: + held: Final = f" held back: {self.hold}" if self.hold else "" + return f"{self.name}: {self.old!r} -> {self.new!r}{held}" + + +def _swing(old: float, new: float) -> str | None: + if (old == 0) != (new == 0): + return "a price crossing zero" + if old and new and max(new / old, old / new) > PRICE_SWING_LIMIT: + return f"a price moving more than {PRICE_SWING_LIMIT}x" + return None + + +def _hold(name: str, old: object, new: object, curated_prices_win: bool) -> str | None: + if "cost" in name and curated_prices_win: + return "the catalog price varies by provider" + if old is None: + return None + if name.startswith("max_") and isinstance(old, int) and isinstance(new, int) and new < old: + return "a shrinking limit" + if "cost" in name and isinstance(old, int | float) and isinstance(new, int | float): + return _swing(old, new) + return None + + +def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange, ...]: + curated_prices_win: Final = entry.indicative_prices and "input_cost_per_token" in existing + scalars: Final = tuple( + FieldChange(name, existing.get(name), value, _hold(name, existing.get(name), value, curated_prices_win)) + for name, value in entry.fields.items() + if name not in LIMIT_PAIR and existing.get(name) != value + ) + ceiling: Final = entry.fields.get("max_output_tokens") + if ceiling is None: + return scalars + current: Final = existing.get("max_output_tokens", existing.get("max_tokens")) + hold: Final = _hold("max_output_tokens", current, ceiling, curated_prices_win) + return ( + *scalars, + *(FieldChange(name, existing.get(name), ceiling, hold) for name in LIMIT_PAIR if existing.get(name) != ceiling), + ) + + +def _updated_entry( + existing: RegistryEntry, entry: CatalogEntry +) -> tuple[RegistryEntry, tuple[str, ...], tuple[str, ...]]: + changes: Final = _changes(existing, entry) + applied: Final = {change.name: change.new for change in changes if change.hold is None} + return ( + {**existing, **dict(sorted(applied.items()))}, + tuple(change.line for change in changes if change.hold is None), + tuple(change.line for change in changes if change.hold is not None), ) - return dict(sorted({**existing, **desired}.items())), changes def _with_new_keys_in_block(ordered: CostMap, result: CostMap, new_keys: Sequence[str], prefix: str) -> CostMap: @@ -331,26 +538,25 @@ class Warned: line: str -@dataclass(frozen=True, slots=True) -class Unchanged: - pass +EntrySync = Added | Updated | Warned -EntrySync = Added | Updated | Warned | Unchanged - - -def _sync_entry(existing: object, entry: CatalogEntry) -> EntrySync: +def _sync_entry(cost_map: CostMap, entry: CatalogEntry) -> tuple[EntrySync, ...]: + existing: Final = cost_map.get(entry.key) if not isinstance(existing, dict): - return Added(key=entry.key, entry=_new_entry(entry)) + return (Added(key=entry.key, entry=_new_entry(entry, _inherited(cost_map, entry))),) if existing.get("mode") != entry.mode: - return Warned( - line=f"`{entry.key}` has curated mode {existing.get('mode')!r} but the catalog maps to " - f"{entry.mode!r}; left unchanged" + return ( + Warned( + line=f"`{entry.key}` has curated mode {existing.get('mode')!r} but the catalog maps to " + f"{entry.mode!r}; left unchanged" + ), ) - new_entry, changes = _updated_entry(existing, entry) - if not changes: - return Unchanged() - return Updated(key=entry.key, entry=new_entry, line=f"{entry.key}: " + "; ".join(changes)) + new_entry, applied, held = _updated_entry(existing, entry) + return ( + *((Updated(key=entry.key, entry=new_entry, line=f"{entry.key}: " + "; ".join(applied)),) if applied else ()), + *((Warned(line=f"{entry.key}: " + "; ".join(held)),) if held else ()), + ) SyncState = tuple[CostMap, tuple[ProviderOutcome, ...]] @@ -359,13 +565,13 @@ SyncState = tuple[CostMap, tuple[ProviderOutcome, ...]] def _sync_provider(state: SyncState, catalog: Catalog) -> SyncState: cost_map, outcomes = state syncs: Final = tuple( - _sync_entry(cost_map.get(entry.key), entry) for entry in sorted(catalog.entries, key=lambda item: item.key) + sync for entry in sorted(catalog.entries, key=lambda item: item.key) for sync in _sync_entry(cost_map, entry) ) outcome: Final = ProviderOutcome( provider=catalog.provider, added=tuple(sync.key for sync in syncs if isinstance(sync, Added)), updated=tuple(sync.line for sync in syncs if isinstance(sync, Updated)), - warnings=tuple(sync.line for sync in syncs if isinstance(sync, Warned)), + warnings=(*catalog.warnings, *(sync.line for sync in syncs if isinstance(sync, Warned))), skipped=catalog.skipped, ) merged: Final = {**cost_map, **{sync.key: sync.entry for sync in syncs if isinstance(sync, Added | Updated)}} @@ -412,7 +618,9 @@ def render_pr_body(outcome: SyncOutcome, section_limit: int | None = PR_BODY_SEC return ( "Automated sync of the openrouter and vercel_ai_gateway entries in model_prices_and_context_window.json " f"against `GET {OPENROUTER_MODELS_URL}` and `GET {VERCEL_MODELS_URL}` by scripts/sync_cost_map.py. " - "The cost-map-guard check enforces that this PR only adds or reprices models.\n" + "The cost-map-guard check enforces that this PR only adds or reprices models. Changes the script held " + "back (shrinking limits, prices crossing zero or moving more than 10x, per-provider prices) are listed " + "under the warnings and need a human commit.\n" "\n" + "\n".join(_provider_body(provider, section_limit) for provider in outcome.providers) ) diff --git a/tests/test_litellm/test_sync_cost_map.py b/tests/test_litellm/test_sync_cost_map.py index b693a7211d6..a9fcb22655d 100644 --- a/tests/test_litellm/test_sync_cost_map.py +++ b/tests/test_litellm/test_sync_cost_map.py @@ -71,6 +71,17 @@ def sync() -> ModuleType: return module +def _openrouter_rows(*rows: dict[str, object]) -> bytes: + return json.dumps( + {"data": [{"pricing": {"prompt": "0.000001", "completion": "0.000002"}, **row} for row in rows]} + ).encode() + + +def _vercel_rows(*rows: dict[str, object]) -> bytes: + defaults: Final = {"type": "language", "pricing": {"input": "0.000001", "output": "0.000002"}} + return json.dumps({"data": [{**defaults, **row} for row in rows]}).encode() + + def _run(sync: ModuleType, cost_map: dict[str, object]): return sync.compute_sync( cost_map, (sync.load_openrouter(OPENROUTER_RAW), sync.load_vercel(VERCEL_RAW, now_ms=NOW_MS)) @@ -157,19 +168,24 @@ def test_existing_entry_is_repriced_without_losing_curated_fields(sync: ModuleTy assert deepseek["output_cost_per_token"] == 1.73844e-6 assert deepseek["cache_read_input_token_cost"] == 1.9316e-8 assert deepseek["input_cost_per_token_cache_hit"] == 4.4e-8 - assert (deepseek["max_output_tokens"], deepseek["max_tokens"]) == (300000, 300000) + assert (deepseek["max_output_tokens"], deepseek["max_tokens"]) == (384000, 384000) glm: Final = outcome.cost_map["vercel_ai_gateway/zai/glm-4.6"] assert (glm["input_cost_per_token"], glm["output_cost_per_token"]) == (6e-7, 2.2e-6) assert glm["supports_parallel_function_calling"] is True assert glm["supports_reasoning"] is True - assert glm["max_output_tokens"] == 200000 + assert (glm["max_output_tokens"], glm["max_tokens"]) == (200000, 200000) openrouter, vercel = outcome.providers assert [line.split(":")[0] for line in openrouter.updated] == ["openrouter/deepseek/deepseek-v4-pro-0813"] assert "input_cost_per_token: 1.32e-06 -> 5.7948e-07" in openrouter.updated[0] + assert "max_output_tokens: 300000 -> 384000; max_tokens: 300000 -> 384000" in openrouter.updated[0] assert [line.split(":")[0] for line in vercel.updated] == ["vercel_ai_gateway/zai/glm-4.6"] + assert vercel.warnings == ( + "vercel_ai_gateway/zai/glm-4.6: max_output_tokens: 200000 -> 96000 held back: a shrinking limit; " + "max_tokens: 200000 -> 96000 held back: a shrinking limit", + ) -def test_legacy_max_tokens_is_never_paired_with_a_different_max_output_tokens(sync: ModuleType) -> None: +def test_legacy_max_tokens_moves_in_step_with_the_catalog_output_ceiling(sync: ModuleType) -> None: legacy: Final = { "input_cost_per_token": 4e-8, "litellm_provider": "openrouter", @@ -180,9 +196,38 @@ def test_legacy_max_tokens_is_never_paired_with_a_different_max_output_tokens(sy outcome: Final = _run(sync, {**_base_map(), "openrouter/inception/mercury-2.5-preview": legacy}) mercury: Final = outcome.cost_map["openrouter/inception/mercury-2.5-preview"] - assert mercury["max_tokens"] == 8192 - assert "max_output_tokens" not in mercury + assert (mercury["max_output_tokens"], mercury["max_tokens"]) == (65536, 65536) assert mercury["max_input_tokens"] == 260000 + assert list(mercury) == [ + *legacy, + "cache_read_input_token_cost", + "max_input_tokens", + "max_output_tokens", + "supports_function_calling", + "supports_reasoning", + "supports_response_schema", + "supports_tool_choice", + ] + + +def test_output_limits_stay_put_when_the_catalog_has_no_output_ceiling(sync: ModuleType) -> None: + existing: Final = { + "litellm_provider": "openrouter", + "mode": "chat", + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "max_input_tokens": 1000, + "max_output_tokens": 500, + "max_tokens": 500, + } + catalog: Final = sync.load_openrouter(_openrouter_rows({"id": "acme/x", "context_length": 4000})) + + outcome: Final = sync.compute_sync({"openrouter/acme/x": dict(existing)}, (catalog,)) + + entry: Final = outcome.cost_map["openrouter/acme/x"] + assert (entry["max_input_tokens"], entry["max_output_tokens"], entry["max_tokens"]) == (4000, 500, 500) + assert outcome.providers[0].updated == ("openrouter/acme/x: max_input_tokens: 1000 -> 4000",) + assert outcome.providers[0].warnings == () def test_untouched_entries_survive_byte_for_byte(sync: ModuleType) -> None: @@ -220,9 +265,10 @@ def test_mode_mismatch_warns_and_leaves_the_entry_alone(sync: ModuleType) -> Non vercel: Final = outcome.providers[1] assert "vercel_ai_gateway/openai/gpt-5-mini" not in vercel.added assert all("gpt-5-mini" not in line for line in vercel.updated) - assert len(vercel.warnings) == 1 - assert "vercel_ai_gateway/openai/gpt-5-mini" in vercel.warnings[0] - assert "'responses'" in vercel.warnings[0] and "'chat'" in vercel.warnings[0] + mismatch: Final = [line for line in vercel.warnings if "gpt-5-mini" in line] + assert len(mismatch) == 1 + assert "vercel_ai_gateway/openai/gpt-5-mini" in mismatch[0] + assert "'responses'" in mismatch[0] and "'chat'" in mismatch[0] def test_new_keys_land_at_the_end_of_their_provider_block(sync: ModuleType) -> None: @@ -360,7 +406,7 @@ def test_a_scheduled_deprecation_keeps_syncing_until_the_date(sync: ModuleType) passed: Final = sync.load_vercel(_vercel_language_row(NOW_MS), now_ms=NOW_MS) assert [entry.key for entry in scheduled.entries] == ["vercel_ai_gateway/acme/chat-1"] - assert dict(scheduled.skipped)["deprecated"] == 0 + assert dict(scheduled.skipped).get("deprecated", 0) == 0 assert passed.entries == () assert dict(passed.skipped)["deprecated"] == 1 @@ -419,3 +465,280 @@ def test_dry_run_touches_nothing(sync: ModuleType, tmp_path: Path, capsys) -> No assert code == 0 assert (repo / "model_prices_and_context_window.json").read_bytes() == before assert "dry run: no files were touched" in capsys.readouterr().out + + +def test_new_entries_inherit_model_intrinsic_traits_from_the_root_entry(sync: ModuleType) -> None: + root: Final = { + "litellm_provider": "anthropic", + "mode": "chat", + "input_cost_per_token": 5e-6, + "supports_adaptive_thinking": True, + "thinking_always_on": True, + "supports_sampling_params": False, + "supports_function_calling": False, + "supports_vision": True, + "prompt_cache_min_tokens": 1024, + "supports_web_search": True, + } + already_synced: Final = {"litellm_provider": "openrouter", "mode": "chat", "input_cost_per_token": 5e-6} + cost_map: Final = { + "claude-fable-5": dict(root), + "claude-fable-5-1": {**root, "prompt_cache_min_tokens": 512}, + "claude-embed-5": {**root, "mode": "embedding"}, + "openrouter/anthropic/claude-fable-5:thinking": dict(already_synced), + } + openrouter: Final = sync.load_openrouter( + _openrouter_rows( + {"id": "anthropic/claude-fable-5:batch", "supported_parameters": ["tools"]}, + {"id": "anthropic/claude-fable-5:thinking"}, + {"id": "anthropic/claude-embed-5"}, + {"id": "anthropic/claude-opus-6"}, + ) + ) + vercel: Final = sync.load_vercel( + _vercel_rows({"id": "anthropic/claude-fable-5.1"}, {"id": "anthropic/claude-fable-5.1-fast"}), now_ms=NOW_MS + ) + + outcome: Final = sync.compute_sync(cost_map, (openrouter, vercel)) + + batch: Final = outcome.cost_map["openrouter/anthropic/claude-fable-5:batch"] + assert (batch["supports_adaptive_thinking"], batch["thinking_always_on"]) == (True, True) + assert (batch["supports_sampling_params"], batch["prompt_cache_min_tokens"]) == (False, 1024) + assert batch["supports_function_calling"] is True + assert not {"supports_web_search", "supports_vision"} & batch.keys() + assert outcome.cost_map["vercel_ai_gateway/anthropic/claude-fable-5.1"]["prompt_cache_min_tokens"] == 512 + fast: Final = outcome.cost_map["vercel_ai_gateway/anthropic/claude-fable-5.1-fast"] + assert (fast["prompt_cache_min_tokens"], fast["supports_adaptive_thinking"]) == (512, True) + assert "prompt_cache_min_tokens" not in outcome.cost_map["openrouter/anthropic/claude-opus-6"] + assert "supports_adaptive_thinking" not in outcome.cost_map["openrouter/anthropic/claude-embed-5"] + assert "supports_adaptive_thinking" not in outcome.cost_map["openrouter/anthropic/claude-fable-5:thinking"] + + +@pytest.mark.parametrize( + ("field", "old", "new", "reason"), + [ + ("input_cost_per_token", 1e-6, 0.0, "a price crossing zero"), + ("input_cost_per_token", 0.0, 1e-6, "a price crossing zero"), + ("output_cost_per_token", 1e-7, 2e-6, "a price moving more than 10x"), + ("output_cost_per_token", 2e-6, 1e-7, "a price moving more than 10x"), + ("max_input_tokens", 200000, 128000, "a shrinking limit"), + ], +) +def test_out_of_bounds_changes_are_held_back_as_warnings( + sync: ModuleType, field: str, old: float, new: float, reason: str +) -> None: + existing: Final = { + "litellm_provider": "openrouter", + "mode": "chat", + "input_cost_per_token": 1e-6, + "output_cost_per_token": 2e-6, + "max_input_tokens": 200000, + field: old, + } + catalog_row: Final = { + "id": "acme/x", + "context_length": 200000, + "pricing": {"prompt": "0.000001", "completion": "0.000002"}, + **({"context_length": int(new)} if field == "max_input_tokens" else {}), + } + catalog_row["pricing"] = { + **catalog_row["pricing"], + **({"prompt": str(new)} if field == "input_cost_per_token" else {}), + **({"completion": str(new)} if field == "output_cost_per_token" else {}), + } + + outcome: Final = sync.compute_sync( + {"openrouter/acme/x": dict(existing)}, (sync.load_openrouter(_openrouter_rows(catalog_row)),) + ) + + assert outcome.cost_map["openrouter/acme/x"] == existing + assert outcome.has_changes is False + assert outcome.providers[0].warnings == (f"openrouter/acme/x: {field}: {old!r} -> {new!r} held back: {reason}",) + + +def test_a_price_move_within_ten_x_is_applied(sync: ModuleType) -> None: + existing: Final = {"litellm_provider": "openrouter", "mode": "chat", "input_cost_per_token": 1e-6} + catalog: Final = sync.load_openrouter( + _openrouter_rows({"id": "acme/x", "pricing": {"prompt": "0.000009", "completion": "0"}}) + ) + + outcome: Final = sync.compute_sync({"openrouter/acme/x": existing}, (catalog,)) + + assert outcome.cost_map["openrouter/acme/x"]["input_cost_per_token"] == 9e-6 + assert outcome.providers[0].warnings == () + + +def test_vercel_long_context_tiers_map_to_above_threshold_prices(sync: ModuleType) -> None: + pricing: Final = { + "input": "0.0000015", + "input_tiers": [ + {"cost": "0.0000015", "min": 0, "max": 32001}, + {"cost": "0.0000027", "min": 32001, "max": 128001}, + {"cost": "0.0000045", "min": 128001}, + ], + "output": "0.0000075", + "output_tiers": [{"cost": "0.0000075", "max": 200001}, {"cost": "0.00001125", "min": 200001}], + "input_cache_read": "0.0000003", + "input_cache_read_tiers": [{"cost": "0.0000006", "min": 256000}], + "input_cache_write": "0.000002", + "input_cache_write_tiers": [{"cost": "0.000002", "min": 0, "max": 200001}, {"cost": "0.000004", "min": 200001}], + } + catalog: Final = sync.load_vercel(_vercel_rows({"id": "acme/long", "pricing": pricing}), now_ms=NOW_MS) + + outcome: Final = sync.compute_sync({}, (catalog,)) + + entry: Final = outcome.cost_map["vercel_ai_gateway/acme/long"] + assert entry["input_cost_per_token"] == 1.5e-6 + assert entry["input_cost_per_token_above_32k_tokens"] == 2.7e-6 + assert entry["input_cost_per_token_above_128k_tokens"] == 4.5e-6 + assert entry["output_cost_per_token_above_200k_tokens"] == 1.125e-5 + assert entry["cache_read_input_token_cost_above_256k_tokens"] == 6e-7 + assert entry["cache_creation_input_token_cost_above_200k_tokens"] == 4e-6 + assert not any(key.endswith("_above_0k_tokens") for key in entry) + + +@pytest.mark.parametrize( + ("tiers", "problem"), + [ + ( + [{"cost": "0.000001", "min": 0, "max": 150500}, {"cost": "0.000002", "min": 150500}], + "input_cost_per_token tiers have a boundary that is not a whole thousand or an unusable price", + ), + ( + [{"cost": "0.000001", "min": 0, "max": 128000}, {"cost": "0.000002", "min": 200000}], + "input_cost_per_token tiers are not contiguous", + ), + ( + [{"cost": "0.000001", "min": 0, "max": 128000}, {"cost": "0.000002", "min": 128000, "max": 256000}], + "input_cost_per_token tiers are not contiguous", + ), + ], +) +def test_unmappable_tiers_skip_the_row_with_a_warning(sync: ModuleType, tiers: list, problem: str) -> None: + pricing: Final = {"input": "0.000001", "input_tiers": tiers, "output": "0.000002"} + catalog: Final = sync.load_vercel(_vercel_rows({"id": "acme/odd", "pricing": pricing}), now_ms=NOW_MS) + + outcome: Final = sync.compute_sync({}, (catalog,)) + + assert "vercel_ai_gateway/acme/odd" not in outcome.cost_map + assert dict(catalog.skipped) == {"tiers outside the registry's thresholds": 1} + assert outcome.providers[0].warnings == (f"vercel_ai_gateway/acme/odd: {problem}; row skipped",) + + +def test_a_price_that_varies_by_provider_seeds_but_never_overwrites(sync: ModuleType) -> None: + curated: Final = { + "litellm_provider": "vercel_ai_gateway", + "mode": "chat", + "input_cost_per_token": 9e-7, + "output_cost_per_token": 2e-6, + "max_input_tokens": 100000, + } + row: Final = { + "context_window": 262144, + "pricing": { + "input": "0.0000015", + "input_tiers": [{"cost": "0.0000015", "min": 0, "max": 128001}, {"cost": "0.000003", "min": 128001}], + "output": "0.000002", + "input_cache_read": "0.0000003", + "varies_by_provider": True, + }, + } + catalog: Final = sync.load_vercel( + _vercel_rows({"id": "acme/curated", **row}, {"id": "acme/fresh", **row}), now_ms=NOW_MS + ) + + outcome: Final = sync.compute_sync({"vercel_ai_gateway/acme/curated": dict(curated)}, (catalog,)) + + existing: Final = outcome.cost_map["vercel_ai_gateway/acme/curated"] + assert (existing["input_cost_per_token"], existing["max_input_tokens"]) == (9e-7, 262144) + assert not any("cache_read" in name or "_above_" in name for name in existing) + fresh: Final = outcome.cost_map["vercel_ai_gateway/acme/fresh"] + assert (fresh["input_cost_per_token"], fresh["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6) + assert fresh["cache_read_input_token_cost"] == 3e-7 + assert outcome.providers[0].warnings == ( + "vercel_ai_gateway/acme/curated: input_cost_per_token: 9e-07 -> 1.5e-06 held back: " + "the catalog price varies by provider; cache_read_input_token_cost: None -> 3e-07 held back: " + "the catalog price varies by provider; input_cost_per_token_above_128k_tokens: None -> 3e-06 held back: " + "the catalog price varies by provider", + ) + + +def test_image_and_audio_outputs_are_priced_per_token_or_skipped(sync: ModuleType) -> None: + openrouter: Final = sync.load_openrouter( + _openrouter_rows( + { + "id": "openai/gpt-5-image", + "architecture": {"input_modalities": ["text", "image"], "output_modalities": ["image", "text"]}, + "pricing": {"prompt": "0.00001", "completion": "0.00001", "image_output": "0.00004"}, + }, + { + "id": "acme/talker", + "architecture": {"input_modalities": ["text", "audio"], "output_modalities": ["text", "audio"]}, + "pricing": { + "prompt": "0.000001", + "completion": "0.000002", + "audio": "0.000005", + "audio_output": "0.00001", + }, + }, + { + "id": "acme/mute", + "architecture": {"input_modalities": ["text"], "output_modalities": ["text", "audio"]}, + }, + {"id": "acme/painter", "architecture": {"output_modalities": ["image"]}}, + ) + ) + vercel: Final = sync.load_vercel( + _vercel_rows( + { + "id": "acme/speaker", + "modalities": {"input": ["text", "audio"], "output": ["text", "audio"]}, + "pricing": { + "input": "0.000001", + "output": "0.000002", + "audio_input_token_cost": "0.000004", + "audio_output_token_cost": "0.000008", + }, + }, + {"id": "acme/drawer", "modalities": {"input": ["text"], "output": ["text", "image"]}}, + ), + now_ms=NOW_MS, + ) + + outcome: Final = sync.compute_sync({}, (openrouter, vercel)) + + image: Final = outcome.cost_map["openrouter/openai/gpt-5-image"] + assert (image["output_cost_per_image_token"], image["mode"], image["supports_vision"]) == (4e-5, "chat", True) + talker: Final = outcome.cost_map["openrouter/acme/talker"] + assert (talker["input_cost_per_audio_token"], talker["output_cost_per_audio_token"]) == (5e-6, 1e-5) + assert (talker["supports_audio_input"], talker["supports_audio_output"]) == (True, True) + speaker: Final = outcome.cost_map["vercel_ai_gateway/acme/speaker"] + assert (speaker["input_cost_per_audio_token"], speaker["output_cost_per_audio_token"]) == (4e-6, 8e-6) + assert speaker["supports_audio_output"] is True + assert {"openrouter/acme/mute", "openrouter/acme/painter", "vercel_ai_gateway/acme/drawer"}.isdisjoint( + outcome.cost_map + ) + assert dict(openrouter.skipped) == {"output priced outside the catalog": 2} + assert dict(vercel.skipped) == {"output priced outside the catalog": 1} + + +def test_updates_keep_the_curated_key_order_and_append_new_keys(sync: ModuleType) -> None: + curated: Final = { + "mode": "chat", + "output_cost_per_token": 2e-6, + "input_cost_per_token": 1e-6, + "litellm_provider": "openrouter", + } + catalog: Final = sync.load_openrouter( + _openrouter_rows( + { + "id": "acme/x", + "pricing": {"prompt": "0.000003", "completion": "0.000002", "input_cache_read": "0.0000001"}, + } + ) + ) + + outcome: Final = sync.compute_sync({"openrouter/acme/x": dict(curated)}, (catalog,)) + + assert list(outcome.cost_map["openrouter/acme/x"]) == [*curated, "cache_read_input_token_cost"] + assert outcome.cost_map["openrouter/acme/x"]["input_cost_per_token"] == 3e-6 diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 14907e17b1b..98378d81d08 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -906,6 +906,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_audio_token_cost": {"type": "number"}, "cache_creation_input_token_cost": {"type": "number"}, "cache_creation_input_token_cost_above_1hr": {"type": "number"}, + "cache_creation_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_creation_input_token_cost_above_256k_tokens": {"type": "number"}, @@ -919,6 +920,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "cache_creation_input_token_cost_flex": {"type": "number"}, "cache_creation_input_token_cost_priority": {"type": "number"}, "cache_read_input_token_cost": {"type": "number"}, + "cache_read_input_token_cost_above_32k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_128k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_200k_tokens": {"type": "number"}, "cache_read_input_token_cost_above_256k_tokens": {"type": "number"}, @@ -974,6 +976,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "input_cost_per_request": {"type": "number"}, "input_cost_per_second": {"type": "number"}, "input_cost_per_token": {"type": "number"}, + "input_cost_per_token_above_32k_tokens": {"type": "number"}, "input_cost_per_token_above_128k_tokens": {"type": "number"}, "input_cost_per_token_batches": {"type": "number"}, "input_cost_per_token_cache_hit": {"type": "number"}, @@ -1029,6 +1032,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "output_cost_per_second_1080p": {"type": "number"}, "output_cost_per_second_4k": {"type": "number"}, "output_cost_per_token": {"type": "number"}, + "output_cost_per_token_above_32k_tokens": {"type": "number"}, "output_cost_per_token_above_128k_tokens": {"type": "number"}, "output_cost_per_token_above_200k_tokens": {"type": "number"}, "output_cost_per_token_above_256k_tokens": {"type": "number"},