diff --git a/.github/scripts/auto_update_price_and_context_window_file.py b/.github/scripts/auto_update_price_and_context_window_file.py deleted file mode 100644 index 461d8d347d9..00000000000 --- a/.github/scripts/auto_update_price_and_context_window_file.py +++ /dev/null @@ -1,159 +0,0 @@ -import asyncio -import aiohttp -import json - -# Asynchronously fetch data from a given URL -async def fetch_data(url): - try: - # Create an asynchronous session - async with aiohttp.ClientSession() as session: - # Send a GET request to the URL - async with session.get(url) as resp: - # Raise an error if the response status is not OK - resp.raise_for_status() - # Parse the response JSON - resp_json = await resp.json() - print("Fetch the data from URL.") - # Return the 'data' field from the JSON response - return resp_json['data'] - except Exception as e: - # Print an error message if fetching data fails - print("Error fetching data from URL:", e) - return None - -# Synchronize local data with remote data -def sync_local_data_with_remote(local_data, remote_data): - # Update existing keys in local_data with values from remote_data - for key in (set(local_data) & set(remote_data)): - local_data[key].update(remote_data[key]) - - # Add new keys from remote_data to local_data - for key in (set(remote_data) - set(local_data)): - local_data[key] = remote_data[key] - -# Write data to the json file -def write_to_file(file_path, data): - try: - # Open the file in write mode - with open(file_path, "w") as file: - # Dump the data as JSON into the file - json.dump(data, file, indent=4) - print("Values updated successfully.") - except Exception as e: - # Print an error message if writing to file fails - print("Error updating JSON file:", e) - -# Update the existing models and add the missing models for OpenRouter -def transform_openrouter_data(data): - transformed = {} - for row in data: - # Add the fields 'max_tokens' and 'input_cost_per_token' - obj = { - "max_tokens": row["context_length"], - "input_cost_per_token": float(row["pricing"]["prompt"]), - } - - # Add 'max_output_tokens' as a field if it is not None - if "top_provider" in row and "max_completion_tokens" in row["top_provider"] and row["top_provider"]["max_completion_tokens"] is not None: - obj['max_output_tokens'] = int(row["top_provider"]["max_completion_tokens"]) - - # Add the field 'output_cost_per_token' - obj.update({ - "output_cost_per_token": float(row["pricing"]["completion"]), - }) - - # Add field 'input_cost_per_image' if it exists and is non-zero - if "pricing" in row and "image" in row["pricing"] and float(row["pricing"]["image"]) != 0.0: - obj['input_cost_per_image'] = float(row["pricing"]["image"]) - - # Add the fields 'litellm_provider' and 'mode' - obj.update({ - "litellm_provider": "openrouter", - "mode": "chat" - }) - - # Add the 'supports_vision' field if the modality is 'multimodal' - if row.get('architecture', {}).get('modality') == 'multimodal': - obj['supports_vision'] = True - - # Use a composite key to store the transformed object - transformed[f'openrouter/{row["id"]}'] = obj - - return transformed - -# Update the existing models and add the missing models for Vercel AI Gateway -def transform_vercel_ai_gateway_data(data): - transformed = {} - for row in data: - obj = { - "max_tokens": row["context_window"], - "input_cost_per_token": float(row["pricing"]["input"]), - "output_cost_per_token": float(row["pricing"]["output"]), - 'max_output_tokens': row['max_tokens'], - 'max_input_tokens': row["context_window"], - } - - # Handle cache pricing if available - if "pricing" in row: - if "input_cache_read" in row["pricing"] and row["pricing"]["input_cache_read"] is not None: - obj['cache_read_input_token_cost'] = float(f"{float(row['pricing']['input_cache_read']):e}") - - if "input_cache_write" in row["pricing"] and row["pricing"]["input_cache_write"] is not None: - obj['cache_creation_input_token_cost'] = float(f"{float(row['pricing']['input_cache_write']):e}") - - mode = "embedding" if "embedding" in row["id"].lower() else "chat" - - obj.update({"litellm_provider": "vercel_ai_gateway", "mode": mode}) - - transformed[f'vercel_ai_gateway/{row["id"]}'] = obj - - return transformed - - -# Load local data from a specified file -def load_local_data(file_path): - try: - # Open the file in read mode - with open(file_path, "r") as file: - # Load and return the JSON data - return json.load(file) - except FileNotFoundError: - # Print an error message if the file is not found - print("File not found:", file_path) - return None - except json.JSONDecodeError as e: - # Print an error message if JSON decoding fails - print("Error decoding JSON:", e) - return None - -def main(): - local_file_path = "model_prices_and_context_window.json" # Path to the local data file - openrouter_url = "https://openrouter.ai/api/v1/models" # URL to fetch OpenRouter data - vercel_ai_gateway_url = "https://ai-gateway.vercel.sh/v1/models" # URL to fetch Vercel AI Gateway data - - # Load local data from file - local_data = load_local_data(local_file_path) - - # Fetch OpenRouter data - openrouter_data = asyncio.run(fetch_data(openrouter_url)) - # Transform the fetched OpenRouter data - openrouter_data = transform_openrouter_data(openrouter_data) - - # Fetch Vercel AI Gateway data - vercel_data = asyncio.run(fetch_data(vercel_ai_gateway_url)) - # Transform the fetched Vercel AI Gateway data - vercel_data = transform_vercel_ai_gateway_data(vercel_data) - - # Combine both datasets - all_remote_data = {**openrouter_data, **vercel_data} - - # If both local and openrouter data are available, synchronize and save - if local_data and all_remote_data: - sync_local_data_with_remote(local_data, all_remote_data) - write_to_file(local_file_path, local_data) - else: - print("Failed to fetch model data from either local file or URL.") - -# Entry point of the script -if __name__ == "__main__": - main() diff --git a/.github/workflows/auto_update_price_and_context_window.yml b/.github/workflows/auto_update_price_and_context_window.yml deleted file mode 100644 index 7e40a860ee9..00000000000 --- a/.github/workflows/auto_update_price_and_context_window.yml +++ /dev/null @@ -1,39 +0,0 @@ -name: Updates model_prices_and_context_window.json and Create Pull Request - -on: - schedule: - - cron: "0 0 * * 0" # Run every Sundays at midnight - #- cron: "0 0 * * *" # Run daily at midnight - -permissions: - contents: write - pull-requests: write - -jobs: - auto_update_price_and_context_window: - if: github.repository == 'BerriAI/litellm' - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 - with: - persist-credentials: false - - name: Set up uv - uses: ./.github/actions/setup-uv-with-retries - with: - version: "0.10.9" - - name: Update JSON Data - run: | - uv run --frozen --with 'aiohttp==3.13.3' python ".github/scripts/auto_update_price_and_context_window_file.py" - - name: Regenerate JSON Schema - run: | - uv run --frozen python ci_cd/generate_model_prices_schema.py - - name: Create Pull Request - run: | - git add model_prices_and_context_window.json model_prices_and_context_window.schema.json - git commit -m "Update model_prices_and_context_window.json file: $(date +'%Y-%m-%d')" - gh pr create --title "Update model_prices_and_context_window.json file" \ - --body "Automated update for model_prices_and_context_window.json" \ - --head auto-update-price-and-context-window-$(date +'%Y-%m-%d') \ - --base main - env: - GH_TOKEN: ${{ secrets.GH_TOKEN }} diff --git a/.github/workflows/cost-map-sync.yml b/.github/workflows/cost-map-sync.yml new file mode 100644 index 00000000000..8fc4f343083 --- /dev/null +++ b/.github/workflows/cost-map-sync.yml @@ -0,0 +1,118 @@ +name: Cost map sync + +on: + schedule: + - cron: "*/5 * * * *" + workflow_dispatch: + inputs: + dry_run: + description: "Print the diff without opening a PR" + type: boolean + default: false + +permissions: + contents: write + pull-requests: write + +concurrency: + group: cost-map-sync + cancel-in-progress: false + +env: + BRANCH_PREFIX: litellm_cost_map_sync_ + PR_TITLE: "feat(models): sync openrouter and vercel_ai_gateway pricing" + +jobs: + cost-map-sync: + if: github.repository == 'BerriAI/litellm' + runs-on: ubuntu-latest + env: + BOT_APP_ID: ${{ secrets.COST_MAP_BOT_APP_ID }} + steps: + - uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0 + with: + persist-credentials: false + - name: Mint the bot token + id: bot + if: env.BOT_APP_ID != '' + uses: actions/create-github-app-token@a8d616148505b5069dccd32f177bb87d7f39123b # v2.1.1 + with: + app-id: ${{ secrets.COST_MAP_BOT_APP_ID }} + private-key: ${{ secrets.COST_MAP_BOT_PRIVATE_KEY }} + - name: Set up uv + uses: ./.github/actions/setup-uv-with-retries + with: + version: "0.10.9" + - name: Look for an already-open sync PR + id: existing + run: | + open_pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --json headRefName \ + --search "in:title \"$PR_TITLE\"" \ + --jq "[.[].headRefName | select(startswith(\"$BRANCH_PREFIX\"))] | first // empty")" + echo "open_pr=$open_pr" >> "$GITHUB_OUTPUT" + if [ -n "$open_pr" ]; then + echo "An open sync PR already exists on branch $open_pr; skipping this run." + fi + env: + GH_TOKEN: ${{ steps.bot.outputs.token || secrets.GH_TOKEN || github.token }} + - name: Run the sync + if: steps.existing.outputs.open_pr == '' + run: | + uv run --frozen python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md" + uv run --frozen python ci_cd/generate_model_prices_schema.py + - name: Open the sync PR + id: pr + if: steps.existing.outputs.open_pr == '' && !inputs.dry_run + run: | + if git diff --quiet; then + echo "Registry already in sync; no PR needed." + exit 0 + fi + branch="${BRANCH_PREFIX}$(date -u +'%Y-%m-%d-%H%M')" + if [ -n "$BOT_APP_ID" ]; then + bot_user_id="$(gh api "users/${BOT_LOGIN}[bot]" --jq .id)" + git config user.name "${BOT_LOGIN}[bot]" + git config user.email "${bot_user_id}+${BOT_LOGIN}[bot]@users.noreply.github.com" + else + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + fi + git checkout -b "$branch" + git add model_prices_and_context_window.json \ + litellm/model_prices_and_context_window_backup.json \ + model_prices_and_context_window.schema.json + git commit -m "feat(models): sync openrouter and vercel_ai_gateway pricing $(date -u +'%Y-%m-%d %H:%M')" + gh auth setup-git + git push origin "$branch" + url="$(gh pr create --title "$PR_TITLE" \ + --body-file "$RUNNER_TEMP/pr_body.md" \ + --head "$branch" \ + --base "$GITHUB_REF_NAME")" + echo "url=$url" >> "$GITHUB_OUTPUT" + env: + GH_TOKEN: ${{ steps.bot.outputs.token || secrets.GH_TOKEN || github.token }} + BOT_LOGIN: ${{ steps.bot.outputs.app-slug }} + - name: Merge once every required check passes + if: steps.pr.outputs.url != '' && env.BOT_APP_ID != '' + timeout-minutes: 120 + run: | + while true; do + guard="$(gh pr checks "$PR_URL" --json name,state \ + --jq '.[] | select(.name == "cost-map-guard") | .state' || true)" + required="$(gh pr checks "$PR_URL" --required --json bucket \ + --jq 'map(.bucket) | unique | join(",")' || true)" + case "$guard,$required" in + *FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*) + echo "A check failed (cost-map-guard=$guard, required buckets=$required); leaving $PR_URL open for a human." + exit 1 + ;; + SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping) + gh pr merge "$PR_URL" --repo "$GITHUB_REPOSITORY" --merge --delete-branch + exit 0 + ;; + esac + sleep 30 + done + env: + GH_TOKEN: ${{ steps.bot.outputs.token }} + PR_URL: ${{ steps.pr.outputs.url }} diff --git a/scripts/sync_cost_map.py b/scripts/sync_cost_map.py new file mode 100644 index 00000000000..2d68c27fa5d --- /dev/null +++ b/scripts/sync_cost_map.py @@ -0,0 +1,474 @@ +"""Sync the openrouter and vercel_ai_gateway entries of model_prices_and_context_window.json with the live catalogs. + +Pulls ``GET https://openrouter.ai/api/v1/models`` and ``GET https://ai-gateway.vercel.sh/v1/models``, maps the +catalog fields onto registry fields, and diffs the result against the registry. Dry run (the default) prints the +diff summary and the generated PR body; ``--write`` applies the changes to the root cost map and its ``litellm/`` +backup copy. + +Policy: +- Both catalogs price per token as decimal strings; values are normalized to six significant digits. +- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed, a + capability flag the catalog does not claim stays as curated, and a curated output ceiling is kept. +- Router models and rows without a usable prompt and completion price are skipped. +- A registry entry absent from its catalog is left untouched; retiring a model stays a human call. +""" + +import argparse +import json +import sys +import time +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from functools import reduce +from pathlib import Path +from types import MappingProxyType +from typing import Final, Literal + +import httpx +from pydantic import BaseModel, TypeAdapter, ValidationError + +COST_MAP_RELPATHS: Final = ( + "model_prices_and_context_window.json", + "litellm/model_prices_and_context_window_backup.json", +) +OPENROUTER_MODELS_URL: Final = "https://openrouter.ai/api/v1/models" +VERCEL_MODELS_URL: Final = "https://ai-gateway.vercel.sh/v1/models" +VERCEL_TYPE_TO_MODE: Final = MappingProxyType({"language": "chat", "embedding": "embedding"}) +ADD_ONLY_FIELDS: Final = frozenset({"max_output_tokens", "max_tokens"}) + +Provider = Literal["openrouter", "vercel_ai_gateway"] +RegistryEntry = dict[str, object] +CostMap = dict[str, object] + + +class SyncError(RuntimeError): + pass + + +class OpenRouterPricing(BaseModel): + prompt: str + completion: str + input_cache_read: str | None = None + input_cache_write: str | None = None + internal_reasoning: str | None = None + + +class OpenRouterArchitecture(BaseModel): + input_modalities: tuple[str, ...] | None = None + + +class OpenRouterTopProvider(BaseModel): + max_completion_tokens: int | None = None + + +class OpenRouterModel(BaseModel): + id: str + context_length: int | None = None + architecture: OpenRouterArchitecture | None = None + top_provider: OpenRouterTopProvider = OpenRouterTopProvider() + pricing: OpenRouterPricing + supported_parameters: tuple[str, ...] | None = None + + +class VercelPricing(BaseModel): + input: str | None = None + output: str | None = None + input_cache_read: str | None = None + input_cache_write: str | None = None + + +class VercelModalities(BaseModel): + input: tuple[str, ...] | None = None + + +class VercelModel(BaseModel): + id: str + type: str + context_window: int | None = None + max_tokens: int | None = None + modalities: VercelModalities | None = None + pricing: VercelPricing = VercelPricing() + supported_parameters: tuple[str, ...] | None = None + deprecated_at: int | None = None + + +OPENROUTER_ADAPTER: Final = TypeAdapter(list[OpenRouterModel]) +VERCEL_ADAPTER: Final = TypeAdapter(list[VercelModel]) + + +@dataclass(frozen=True, slots=True) +class CatalogEntry: + key: str + provider: Provider + mode: str + source: str + fields: Mapping[str, object] + + +@dataclass(frozen=True, slots=True) +class Catalog: + provider: Provider + entries: tuple[CatalogEntry, ...] + skipped: Mapping[str, int] + + +def per_token(price: float) -> float: + return float(f"{price:.6g}") + + +def _token_price(raw: str | None) -> float | None: + if raw is None: + return None + value: Final = float(raw) + return per_token(value) if value >= 0 else None + + +def _extra_price(raw: str | None) -> float | None: + price: Final = _token_price(raw) + return price if price else None + + +def _flags(parameters: Sequence[str] | None, modalities: Sequence[str] | None) -> Mapping[str, bool]: + params: Final = frozenset(parameters or ()) + mods: Final = frozenset(modalities or ()) + claims: Final = { + "supports_function_calling": "tools" in params, + "supports_tool_choice": "tool_choice" in params, + "supports_reasoning": "reasoning" in params, + "supports_response_schema": "structured_outputs" in params, + "supports_vision": "image" in mods, + "supports_pdf_input": bool({"file", "pdf"} & mods), + "supports_audio_input": "audio" in mods, + "supports_video_input": "video" in mods, + } + return MappingProxyType({name: True for name, claimed in claims.items() if claimed}) + + +def _limits(max_input: int | None, max_output: int | None) -> Mapping[str, int]: + ceiling: Final = max_output if max_output is not None else max_input + return MappingProxyType( + { + **({"max_input_tokens": max_input} if max_input is not None else {}), + **({"max_output_tokens": max_output} if max_output is not None else {}), + **({"max_tokens": ceiling} if ceiling is not None else {}), + } + ) + + +def _priced(name: str, price: float | None) -> Mapping[str, float]: + return MappingProxyType({name: price} if price is not None else {}) + + +def _openrouter_entry(model: OpenRouterModel) -> CatalogEntry | None: + prompt: Final = _token_price(model.pricing.prompt) + completion: Final = _token_price(model.pricing.completion) + if prompt is None or completion is None: + return None + fields: Final = { + "input_cost_per_token": prompt, + "output_cost_per_token": completion, + **_limits(model.context_length, model.top_provider.max_completion_tokens), + **_priced("cache_read_input_token_cost", _extra_price(model.pricing.input_cache_read)), + **_priced("cache_creation_input_token_cost", _extra_price(model.pricing.input_cache_write)), + **_priced("output_cost_per_reasoning_token", _extra_price(model.pricing.internal_reasoning)), + **_flags(model.supported_parameters, model.architecture.input_modalities if model.architecture else None), + } + return CatalogEntry( + key=f"openrouter/{model.id}", + provider="openrouter", + mode="chat", + source=f"https://openrouter.ai/{model.id}", + fields=MappingProxyType(fields), + ) + + +def _vercel_entry(model: VercelModel) -> CatalogEntry | None: + mode: Final = VERCEL_TYPE_TO_MODE.get(model.type) + prompt: Final = _token_price(model.pricing.input) + completion: Final = _token_price(model.pricing.output if mode != "embedding" else model.pricing.output or "0") + if mode is None or prompt is None or completion is None: + return None + fields: Final = { + "input_cost_per_token": prompt, + "output_cost_per_token": completion, + **_limits(model.context_window, model.max_tokens), + **_priced("cache_read_input_token_cost", _extra_price(model.pricing.input_cache_read)), + **_priced("cache_creation_input_token_cost", _extra_price(model.pricing.input_cache_write)), + **( + _flags(model.supported_parameters, model.modalities.input if model.modalities else None) + if mode == "chat" + else {} + ), + } + return CatalogEntry( + key=f"vercel_ai_gateway/{model.id}", + provider="vercel_ai_gateway", + mode=mode, + source=f"https://vercel.com/ai-gateway/models/{model.id.rsplit('/', 1)[-1]}", + fields=MappingProxyType(fields), + ) + + +def _rows(raw: bytes, url: str) -> object: + parsed: Final = json.loads(raw) + rows: Final = parsed.get("data") if isinstance(parsed, dict) else parsed + if not isinstance(rows, list) or not rows: + raise SyncError(f"GET {url} returned no model rows") + return rows + + +def load_openrouter(raw: bytes) -> Catalog: + try: + models: Final = OPENROUTER_ADAPTER.validate_python(_rows(raw, OPENROUTER_MODELS_URL)) + except ValidationError as error: + raise SyncError(f"the OpenRouter catalog no longer matches the expected shape: {error}") from error + entries: Final = tuple(entry for entry in map(_openrouter_entry, models) if entry is not None) + return Catalog( + provider="openrouter", + entries=entries, + skipped=MappingProxyType({"unpriced or router": len(models) - len(entries)}), + ) + + +def load_vercel(raw: bytes, now_ms: int) -> Catalog: + try: + models: Final = VERCEL_ADAPTER.validate_python(_rows(raw, VERCEL_MODELS_URL)) + except ValidationError as error: + raise SyncError(f"the Vercel AI Gateway catalog no longer matches the expected shape: {error}") from error + live: Final = tuple(model for model in models if model.deprecated_at is None or model.deprecated_at > now_ms) + token_priced: Final = tuple(model for model in live if model.type in VERCEL_TYPE_TO_MODE) + entries: Final = tuple(entry for entry in map(_vercel_entry, token_priced) if entry is not None) + return Catalog( + provider="vercel_ai_gateway", + entries=entries, + skipped=MappingProxyType( + { + "deprecated": len(models) - len(live), + "not token priced": len(live) - len(token_priced), + "no usable price": len(token_priced) - len(entries), + } + ), + ) + + +@dataclass(frozen=True, slots=True) +class ProviderOutcome: + provider: Provider + added: tuple[str, ...] + updated: tuple[str, ...] + warnings: tuple[str, ...] + skipped: Mapping[str, int] + + +@dataclass(frozen=True, slots=True) +class SyncOutcome: + cost_map: CostMap + providers: tuple[ProviderOutcome, ...] + + @property + def has_changes(self) -> bool: + return any(outcome.added or outcome.updated for outcome in self.providers) + + +def _new_entry(entry: CatalogEntry) -> RegistryEntry: + return dict( + sorted( + { + **entry.fields, + "litellm_provider": entry.provider, + "mode": entry.mode, + "source": entry.source, + }.items() + ) + ) + + +def _updated_entry(existing: RegistryEntry, entry: CatalogEntry) -> tuple[RegistryEntry, tuple[str, ...]]: + keep_limits: Final = not ADD_ONLY_FIELDS.isdisjoint(existing) + desired: Final = { + name: value for name, value in entry.fields.items() if not (keep_limits and name in ADD_ONLY_FIELDS) + } + changes: Final = tuple( + f"{name}: {existing.get(name)!r} -> {value!r}" for name, value in desired.items() if existing.get(name) != value + ) + return dict(sorted({**existing, **desired}.items())), changes + + +def _with_new_keys_in_block(ordered: CostMap, result: CostMap, new_keys: Sequence[str], prefix: str) -> CostMap: + provider_keys: Final = tuple(key for key in ordered if key.startswith(prefix)) + if not new_keys: + return {key: result[key] for key in ordered} + if not provider_keys: + return {**{key: result[key] for key in ordered}, **{key: result[key] for key in sorted(new_keys)}} + block_end: Final = provider_keys[-1] + return { + key: value + for existing in ordered + for key, value in ( + (existing, result[existing]), + *((new, result[new]) for new in sorted(new_keys) if existing == block_end), + ) + } + + +@dataclass(frozen=True, slots=True) +class Added: + key: str + entry: RegistryEntry + + +@dataclass(frozen=True, slots=True) +class Updated: + key: str + entry: RegistryEntry + line: str + + +@dataclass(frozen=True, slots=True) +class Warned: + line: str + + +@dataclass(frozen=True, slots=True) +class Unchanged: + pass + + +EntrySync = Added | Updated | Warned | Unchanged + + +def _sync_entry(existing: object, entry: CatalogEntry) -> EntrySync: + if not isinstance(existing, dict): + return Added(key=entry.key, entry=_new_entry(entry)) + if existing.get("mode") != entry.mode: + return Warned( + line=f"`{entry.key}` has curated mode {existing.get('mode')!r} but the catalog maps to " + f"{entry.mode!r}; left unchanged" + ) + new_entry, changes = _updated_entry(existing, entry) + if not changes: + return Unchanged() + return Updated(key=entry.key, entry=new_entry, line=f"{entry.key}: " + "; ".join(changes)) + + +SyncState = tuple[CostMap, tuple[ProviderOutcome, ...]] + + +def _sync_provider(state: SyncState, catalog: Catalog) -> SyncState: + cost_map, outcomes = state + syncs: Final = tuple( + _sync_entry(cost_map.get(entry.key), entry) for entry in sorted(catalog.entries, key=lambda item: item.key) + ) + outcome: Final = ProviderOutcome( + provider=catalog.provider, + added=tuple(sync.key for sync in syncs if isinstance(sync, Added)), + updated=tuple(sync.line for sync in syncs if isinstance(sync, Updated)), + warnings=tuple(sync.line for sync in syncs if isinstance(sync, Warned)), + skipped=catalog.skipped, + ) + merged: Final = {**cost_map, **{sync.key: sync.entry for sync in syncs if isinstance(sync, Added | Updated)}} + return merged, (*outcomes, outcome) + + +def compute_sync(cost_map: CostMap, catalogs: Sequence[Catalog]) -> SyncOutcome: + synced, outcomes = reduce(_sync_provider, catalogs, (dict(cost_map), ())) + return SyncOutcome(cost_map=_ordered_result(cost_map, synced, outcomes), providers=outcomes) + + +def _ordered_result(cost_map: CostMap, result: CostMap, outcomes: Sequence[ProviderOutcome]) -> CostMap: + return reduce( + lambda ordered, outcome: _with_new_keys_in_block(ordered, result, outcome.added, f"{outcome.provider}/"), + outcomes, + {key: result[key] for key in cost_map}, + ) + + +def _section_block(title: str, lines: Sequence[str], backtick: bool) -> str: + bullets: Final = "\n".join(f"- `{line}`" if backtick else f"- {line}" for line in lines) or "- none" + return f"### {title} ({len(lines)})\n{bullets}\n" + + +def _provider_body(outcome: ProviderOutcome) -> str: + skipped: Final = ", ".join(f"{reason} ({count})" for reason, count in sorted(outcome.skipped.items())) or "none" + return ( + f"## {outcome.provider}\n" + "\n" + f"{_section_block('Added', outcome.added, backtick=True)}" + "\n" + f"{_section_block('Updated', outcome.updated, backtick=True)}" + "\n" + f"{_section_block('Warnings needing a human call', outcome.warnings, backtick=False)}" + "\n" + f"Catalog rows skipped: {skipped}\n" + ) + + +def render_pr_body(outcome: SyncOutcome) -> str: + return ( + "Automated sync of the openrouter and vercel_ai_gateway entries in model_prices_and_context_window.json " + f"against `GET {OPENROUTER_MODELS_URL}` and `GET {VERCEL_MODELS_URL}` by scripts/sync_cost_map.py. " + "The cost-map-guard check enforces that this PR only adds or reprices models.\n" + "\n" + "\n".join(_provider_body(provider) for provider in outcome.providers) + ) + + +def render_summary(outcome: SyncOutcome) -> str: + return " ".join( + f"{provider.provider}: added={len(provider.added)} updated={len(provider.updated)} " + f"warnings={len(provider.warnings)}" + for provider in outcome.providers + ) + + +def _fetch(url: str) -> bytes: + response: Final = httpx.get(url, timeout=30, follow_redirects=True) + if response.status_code != 200: + raise SyncError(f"GET {url} returned {response.status_code}") + return response.content + + +def _serialize(cost_map: CostMap) -> str: + return json.dumps(cost_map, indent=4, ensure_ascii=False) + "\n" + + +def main(argv: Sequence[str]) -> int: + parser: Final = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--write", action="store_true", help="apply the sync to the cost map files (default: dry run)") + parser.add_argument("--openrouter-json", type=Path, help="recorded OpenRouter catalog instead of the live API") + parser.add_argument("--vercel-json", type=Path, help="recorded Vercel AI Gateway catalog instead of the live API") + parser.add_argument("--pr-body-file", type=Path, help="write the generated PR body to this path") + parser.add_argument("--repo-root", type=Path, default=Path(__file__).resolve().parent.parent) + args: Final = parser.parse_args(argv) + + openrouter_raw: Final = ( + args.openrouter_json.read_bytes() if args.openrouter_json is not None else _fetch(OPENROUTER_MODELS_URL) + ) + vercel_raw: Final = args.vercel_json.read_bytes() if args.vercel_json is not None else _fetch(VERCEL_MODELS_URL) + catalogs: Final = (load_openrouter(openrouter_raw), load_vercel(vercel_raw, now_ms=int(time.time() * 1000))) + + cost_map_path: Final = args.repo_root / COST_MAP_RELPATHS[0] + cost_map: Final = json.loads(cost_map_path.read_text()) + outcome: Final = compute_sync(cost_map, catalogs) + body: Final = render_pr_body(outcome) + + if args.pr_body_file is not None: + args.pr_body_file.write_text(body) + if args.write and outcome.has_changes: + for relpath in COST_MAP_RELPATHS: + (args.repo_root / relpath).write_text(_serialize(outcome.cost_map)) + print(render_summary(outcome)) + print() + print(body) + if not args.write: + print("dry run: no files were touched") + elif not outcome.has_changes: + print("registry already in sync: no files were touched") + return 0 + + +if __name__ == "__main__": + try: + raise SystemExit(main(sys.argv[1:])) + except SyncError as error: + print(f"SYNC FAILED: {error}", file=sys.stderr) + raise SystemExit(1) from error diff --git a/tests/test_litellm/fixtures/cost_map_sync/openrouter_models.json b/tests/test_litellm/fixtures/cost_map_sync/openrouter_models.json new file mode 100644 index 00000000000..eb06b6a3058 --- /dev/null +++ b/tests/test_litellm/fixtures/cost_map_sync/openrouter_models.json @@ -0,0 +1,209 @@ +{ + "data": [ + { + "id": "cohere/north-mini-code:free", + "context_length": 256000, + "architecture": { + "input_modalities": [ + "text" + ] + }, + "top_provider": { + "max_completion_tokens": 64000 + }, + "pricing": { + "prompt": "0", + "completion": "0" + }, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "max_tokens", + "presence_penalty", + "reasoning", + "seed", + "stop", + "temperature", + "tool_choice", + "tools", + "top_k", + "top_p" + ] + }, + { + "id": "deepseek/deepseek-v4-pro-0813", + "context_length": 1048576, + "architecture": { + "input_modalities": [ + "text" + ] + }, + "top_provider": { + "max_completion_tokens": 384000 + }, + "pricing": { + "prompt": "0.00000057948", + "completion": "0.00000173844", + "input_cache_read": "0.000000019316" + }, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "logit_bias", + "logprobs", + "max_tokens", + "min_p", + "presence_penalty", + "reasoning", + "reasoning_effort", + "repetition_penalty", + "response_format", + "seed", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools", + "top_k", + "top_logprobs", + "top_p" + ] + }, + { + "id": "google/gemma-4-26b-a4b-it:free", + "context_length": 262144, + "architecture": { + "input_modalities": [ + "image", + "text", + "video" + ] + }, + "top_provider": { + "max_completion_tokens": 32768 + }, + "pricing": { + "prompt": "0", + "completion": "0" + }, + "supported_parameters": [ + "include_reasoning", + "max_tokens", + "reasoning", + "response_format", + "seed", + "temperature", + "tool_choice", + "tools", + "top_p" + ] + }, + { + "id": "inception/mercury-2.5-preview", + "context_length": 260000, + "architecture": { + "input_modalities": [ + "text" + ] + }, + "top_provider": { + "max_completion_tokens": 65536 + }, + "pricing": { + "prompt": "0.00000004", + "completion": "0.00000015", + "input_cache_read": "0.000000004" + }, + "supported_parameters": [ + "include_reasoning", + "max_tokens", + "reasoning", + "reasoning_effort", + "response_format", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools" + ] + }, + { + "id": "openai/gpt-5-mini", + "context_length": 400000, + "architecture": { + "input_modalities": [ + "text", + "image", + "file" + ] + }, + "top_provider": { + "max_completion_tokens": 128000 + }, + "pricing": { + "prompt": "0.00000025", + "completion": "0.000002", + "web_search": "0.01", + "input_cache_read": "0.000000025", + "image": "0.003613" + }, + "supported_parameters": [ + "include_reasoning", + "max_completion_tokens", + "max_tokens", + "reasoning", + "reasoning_effort", + "response_format", + "seed", + "structured_outputs", + "tool_choice", + "tools" + ] + }, + { + "id": "openrouter/auto", + "context_length": 2000000, + "architecture": { + "input_modalities": [ + "text", + "image", + "audio", + "file", + "video" + ] + }, + "top_provider": { + "max_completion_tokens": null + }, + "pricing": { + "prompt": "-1", + "completion": "-1" + }, + "supported_parameters": [ + "frequency_penalty", + "include_reasoning", + "logit_bias", + "logprobs", + "max_tokens", + "min_p", + "prediction", + "presence_penalty", + "reasoning", + "reasoning_effort", + "repetition_penalty", + "response_format", + "seed", + "stop", + "structured_outputs", + "temperature", + "tool_choice", + "tools", + "top_a", + "top_k", + "top_logprobs", + "top_p", + "web_search_options" + ] + } + ] +} diff --git a/tests/test_litellm/fixtures/cost_map_sync/vercel_models.json b/tests/test_litellm/fixtures/cost_map_sync/vercel_models.json new file mode 100644 index 00000000000..82fcb72f353 --- /dev/null +++ b/tests/test_litellm/fixtures/cost_map_sync/vercel_models.json @@ -0,0 +1,142 @@ +{ + "object": "list", + "data": [ + { + "id": "alibaba/qwen3-embedding-0.6b", + "type": "embedding", + "context_window": 32768, + "max_tokens": 32768, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "pricing": { + "input": "0.00000001" + }, + "supported_parameters": null, + "deprecated_at": null + }, + { + "id": "bfl/flux-2-flex", + "type": "image", + "context_window": 0, + "max_tokens": 0, + "modalities": { + "input": [ + "text" + ], + "output": [ + "image" + ] + }, + "pricing": {}, + "supported_parameters": null, + "deprecated_at": null + }, + { + "id": "openai/gpt-4o-mini-transcribe", + "type": "transcription", + "context_window": null, + "max_tokens": null, + "modalities": { + "input": [ + "audio" + ], + "output": [ + "text" + ] + }, + "pricing": { + "input": "0.00000125", + "output": "0.000005" + }, + "supported_parameters": null, + "deprecated_at": 1750000000000 + }, + { + "id": "openai/gpt-5-mini", + "type": "language", + "context_window": 400000, + "max_tokens": 128000, + "modalities": { + "input": [ + "text", + "image", + "pdf" + ], + "output": [ + "text" + ] + }, + "pricing": { + "input": "0.00000025", + "output": "0.000002", + "input_cache_read": "0.000000025" + }, + "supported_parameters": [ + "max_tokens", + "stop", + "tools", + "tool_choice", + "reasoning", + "include_reasoning" + ], + "deprecated_at": null + }, + { + "id": "perplexity/sonar", + "type": "language", + "context_window": 127000, + "max_tokens": 8000, + "modalities": { + "input": [ + "text", + "image" + ], + "output": [ + "text" + ] + }, + "pricing": {}, + "supported_parameters": [ + "max_tokens", + "temperature", + "stop" + ], + "deprecated_at": null + }, + { + "id": "zai/glm-4.6", + "type": "language", + "context_window": 200000, + "max_tokens": 96000, + "modalities": { + "input": [ + "text" + ], + "output": [ + "text" + ] + }, + "pricing": { + "input": "0.0000006", + "output": "0.0000022", + "input_cache_read": "0.00000011" + }, + "supported_parameters": [ + "max_tokens", + "temperature", + "stop", + "tools", + "tool_choice", + "reasoning", + "include_reasoning" + ], + "deprecated_at": null + } + ] +} diff --git a/tests/test_litellm/test_sync_cost_map.py b/tests/test_litellm/test_sync_cost_map.py new file mode 100644 index 00000000000..013336885eb --- /dev/null +++ b/tests/test_litellm/test_sync_cost_map.py @@ -0,0 +1,354 @@ +import importlib.util +import json +from pathlib import Path +from types import ModuleType +from typing import Final + +import pytest + +REPO_ROOT: Final = Path(__file__).resolve().parents[2] +SCRIPT_PATH: Final = REPO_ROOT / "scripts" / "sync_cost_map.py" +FIXTURES: Final = Path(__file__).parent / "fixtures" / "cost_map_sync" +OPENROUTER_RAW: Final = (FIXTURES / "openrouter_models.json").read_bytes() +VERCEL_RAW: Final = (FIXTURES / "vercel_models.json").read_bytes() +NOW_MS: Final = 1757030400000 + +EXISTING_DEEPSEEK: Final = { + "input_cost_per_token": 0.00000132, + "input_cost_per_token_cache_hit": 4.4e-8, + "litellm_provider": "openrouter", + "max_input_tokens": 1048576, + "max_output_tokens": 300000, + "max_tokens": 300000, + "mode": "chat", + "output_cost_per_token": 0.00000396, + "source": "https://openrouter.ai/deepseek/deepseek-v4-pro-0813", + "supports_function_calling": True, + "supports_prompt_caching": True, + "supports_reasoning": True, + "supports_response_schema": True, + "supports_tool_choice": True, +} +EXISTING_GLM: Final = { + "litellm_provider": "vercel_ai_gateway", + "cache_read_input_token_cost": 1.1e-7, + "input_cost_per_token": 4.5e-7, + "max_input_tokens": 200000, + "max_output_tokens": 200000, + "max_tokens": 200000, + "mode": "chat", + "output_cost_per_token": 0.0000018, + "source": "https://vercel.com/ai-gateway/models/glm-4.6", + "supports_function_calling": True, + "supports_parallel_function_calling": True, + "supports_tool_choice": True, +} +BLOCK_END_OPENROUTER: Final = { + "litellm_provider": "openrouter", + "mode": "completion", + "input_cost_per_token": 0.0000015, + "output_cost_per_token": 0.000002, +} + + +def _base_map() -> dict[str, object]: + return { + "sample_spec": {"litellm_provider": "one of https://docs.litellm.ai/docs/providers"}, + "gpt-4o": {"litellm_provider": "openai", "mode": "chat"}, + "openrouter/deepseek/deepseek-v4-pro-0813": dict(EXISTING_DEEPSEEK), + "openrouter/openai/gpt-3.5-turbo-instruct": dict(BLOCK_END_OPENROUTER), + "vercel_ai_gateway/zai/glm-4.6": dict(EXISTING_GLM), + "zzz/last": {"litellm_provider": "zzz", "mode": "chat"}, + } + + +@pytest.fixture(scope="module") +def sync() -> ModuleType: + spec = importlib.util.spec_from_file_location("sync_cost_map", SCRIPT_PATH) + assert spec is not None and spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def _run(sync: ModuleType, cost_map: dict[str, object]): + return sync.compute_sync( + cost_map, (sync.load_openrouter(OPENROUTER_RAW), sync.load_vercel(VERCEL_RAW, now_ms=NOW_MS)) + ) + + +def test_new_openrouter_entry_carries_catalog_prices_limits_and_capabilities(sync: ModuleType) -> None: + outcome = _run(sync, _base_map()) + + assert outcome.cost_map["openrouter/inception/mercury-2.5-preview"] == { + "cache_read_input_token_cost": 4e-9, + "input_cost_per_token": 4e-8, + "litellm_provider": "openrouter", + "max_input_tokens": 260000, + "max_output_tokens": 65536, + "max_tokens": 65536, + "mode": "chat", + "output_cost_per_token": 1.5e-7, + "source": "https://openrouter.ai/inception/mercury-2.5-preview", + "supports_function_calling": True, + "supports_reasoning": True, + "supports_response_schema": True, + "supports_tool_choice": True, + } + + +def test_input_modalities_become_capability_flags(sync: ModuleType) -> None: + outcome = _run(sync, _base_map()) + + gpt5 = outcome.cost_map["openrouter/openai/gpt-5-mini"] + gemma = outcome.cost_map["openrouter/google/gemma-4-26b-a4b-it:free"] + vercel_gpt5 = outcome.cost_map["vercel_ai_gateway/openai/gpt-5-mini"] + assert (gpt5["supports_vision"], gpt5["supports_pdf_input"]) == (True, True) + assert "supports_video_input" not in gpt5 and "supports_audio_input" not in gpt5 + assert "input_cost_per_image" not in gpt5 and "supports_prompt_caching" not in gpt5 + assert (gemma["supports_vision"], gemma["supports_video_input"]) == (True, True) + assert (vercel_gpt5["supports_vision"], vercel_gpt5["supports_pdf_input"]) == (True, True) + assert vercel_gpt5["cache_read_input_token_cost"] == 2.5e-8 + assert vercel_gpt5["source"] == "https://vercel.com/ai-gateway/models/gpt-5-mini" + + +def test_free_models_are_added_with_zero_prices(sync: ModuleType) -> None: + outcome = _run(sync, _base_map()) + + free = outcome.cost_map["openrouter/cohere/north-mini-code:free"] + assert (free["input_cost_per_token"], free["output_cost_per_token"]) == (0.0, 0.0) + assert (free["max_input_tokens"], free["max_output_tokens"], free["max_tokens"]) == (256000, 64000, 64000) + assert "cache_read_input_token_cost" not in free + + +def test_router_rows_and_unpriced_rows_are_skipped(sync: ModuleType) -> None: + outcome = _run(sync, _base_map()) + + openrouter, vercel = outcome.providers + assert "openrouter/openrouter/auto" not in outcome.cost_map + assert "vercel_ai_gateway/perplexity/sonar" not in outcome.cost_map + assert dict(openrouter.skipped) == {"unpriced or router": 1} + assert dict(vercel.skipped) == {"deprecated": 1, "not token priced": 1, "no usable price": 1} + + +def test_vercel_rows_map_type_to_mode_and_drop_non_token_types(sync: ModuleType) -> None: + outcome = _run(sync, _base_map()) + + embedding = outcome.cost_map["vercel_ai_gateway/alibaba/qwen3-embedding-0.6b"] + assert embedding == { + "input_cost_per_token": 1e-8, + "litellm_provider": "vercel_ai_gateway", + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "max_tokens": 32768, + "mode": "embedding", + "output_cost_per_token": 0.0, + "source": "https://vercel.com/ai-gateway/models/qwen3-embedding-0.6b", + } + assert "vercel_ai_gateway/bfl/flux-2-flex" not in outcome.cost_map + assert "vercel_ai_gateway/openai/gpt-4o-mini-transcribe" not in outcome.cost_map + + +def test_existing_entry_is_repriced_without_losing_curated_fields(sync: ModuleType) -> None: + outcome = _run(sync, _base_map()) + + deepseek = outcome.cost_map["openrouter/deepseek/deepseek-v4-pro-0813"] + assert deepseek["input_cost_per_token"] == 5.7948e-7 + assert deepseek["output_cost_per_token"] == 1.73844e-6 + assert deepseek["cache_read_input_token_cost"] == 1.9316e-8 + assert deepseek["input_cost_per_token_cache_hit"] == 4.4e-8 + assert (deepseek["max_output_tokens"], deepseek["max_tokens"]) == (300000, 300000) + glm = outcome.cost_map["vercel_ai_gateway/zai/glm-4.6"] + assert (glm["input_cost_per_token"], glm["output_cost_per_token"]) == (6e-7, 2.2e-6) + assert glm["supports_parallel_function_calling"] is True + assert glm["supports_reasoning"] is True + assert glm["max_output_tokens"] == 200000 + openrouter, vercel = outcome.providers + assert [line.split(":")[0] for line in openrouter.updated] == ["openrouter/deepseek/deepseek-v4-pro-0813"] + assert "input_cost_per_token: 1.32e-06 -> 5.7948e-07" in openrouter.updated[0] + assert [line.split(":")[0] for line in vercel.updated] == ["vercel_ai_gateway/zai/glm-4.6"] + + +def test_legacy_max_tokens_is_never_paired_with_a_different_max_output_tokens(sync: ModuleType) -> None: + legacy = { + "input_cost_per_token": 4e-8, + "litellm_provider": "openrouter", + "max_tokens": 8192, + "mode": "chat", + "output_cost_per_token": 1.5e-7, + } + outcome = _run(sync, {**_base_map(), "openrouter/inception/mercury-2.5-preview": legacy}) + + mercury = outcome.cost_map["openrouter/inception/mercury-2.5-preview"] + assert mercury["max_tokens"] == 8192 + assert "max_output_tokens" not in mercury + assert mercury["max_input_tokens"] == 260000 + + +def test_untouched_entries_survive_byte_for_byte(sync: ModuleType) -> None: + outcome = _run(sync, _base_map()) + + assert outcome.cost_map["gpt-4o"] == {"litellm_provider": "openai", "mode": "chat"} + assert outcome.cost_map["openrouter/openai/gpt-3.5-turbo-instruct"] == BLOCK_END_OPENROUTER + assert outcome.cost_map["sample_spec"] == _base_map()["sample_spec"] + + +def test_second_sync_is_a_no_op(sync: ModuleType) -> None: + first = _run(sync, _base_map()) + + second = _run(sync, dict(first.cost_map)) + + assert second.has_changes is False + assert all(not provider.added and not provider.updated for provider in second.providers) + assert list(second.cost_map) == list(first.cost_map) + + +def test_mode_mismatch_warns_and_leaves_the_entry_alone(sync: ModuleType) -> None: + cost_map = _base_map() + cost_map["vercel_ai_gateway/openai/gpt-5-mini"] = { + "litellm_provider": "vercel_ai_gateway", + "mode": "responses", + "input_cost_per_token": 1.0, + } + + outcome = _run(sync, cost_map) + + assert outcome.cost_map["vercel_ai_gateway/openai/gpt-5-mini"]["input_cost_per_token"] == 1.0 + vercel = outcome.providers[1] + assert "vercel_ai_gateway/openai/gpt-5-mini" not in outcome.providers[1].added + assert all("gpt-5-mini" not in line for line in vercel.updated) + assert len(vercel.warnings) == 1 + assert "vercel_ai_gateway/openai/gpt-5-mini" in vercel.warnings[0] + assert "'responses'" in vercel.warnings[0] and "'chat'" in vercel.warnings[0] + + +def test_new_keys_land_at_the_end_of_their_provider_block(sync: ModuleType) -> None: + outcome = _run(sync, _base_map()) + + assert list(outcome.cost_map) == [ + "sample_spec", + "gpt-4o", + "openrouter/deepseek/deepseek-v4-pro-0813", + "openrouter/openai/gpt-3.5-turbo-instruct", + "openrouter/cohere/north-mini-code:free", + "openrouter/google/gemma-4-26b-a4b-it:free", + "openrouter/inception/mercury-2.5-preview", + "openrouter/openai/gpt-5-mini", + "vercel_ai_gateway/zai/glm-4.6", + "vercel_ai_gateway/alibaba/qwen3-embedding-0.6b", + "vercel_ai_gateway/openai/gpt-5-mini", + "zzz/last", + ] + + +def test_provider_without_a_block_is_appended_at_the_end(sync: ModuleType) -> None: + outcome = _run(sync, {"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}}) + + keys = list(outcome.cost_map) + assert keys[0] == "gpt-4o" + assert keys[1:6] == sorted(keys[1:6]) and all(key.startswith("openrouter/") for key in keys[1:6]) + assert keys[6:] == sorted(keys[6:]) and all(key.startswith("vercel_ai_gateway/") for key in keys[6:]) + assert len(keys) == 9 + + +def test_pr_body_lists_changes_per_provider(sync: ModuleType) -> None: + body = sync.render_pr_body(_run(sync, _base_map())) + + assert "## openrouter" in body and "## vercel_ai_gateway" in body + assert "### Added (4)" in body and "- `openrouter/inception/mercury-2.5-preview`" in body + assert "### Added (2)" in body and "- `vercel_ai_gateway/openai/gpt-5-mini`" in body + assert "- `openrouter/deepseek/deepseek-v4-pro-0813: input_cost_per_token: 1.32e-06 -> 5.7948e-07" in body + assert "Catalog rows skipped: deprecated (1), no usable price (1), not token priced (1)" in body + + +@pytest.mark.parametrize( + ("loader", "raw"), + [ + ("load_openrouter", b'{"data": []}'), + ("load_openrouter", b'{"data": [{"id": "x", "pricing": {"prompt": 1}}]}'), + ("load_vercel", b"[]"), + ("load_vercel", b'{"data": [{"id": "x"}]}'), + ], +) +def test_malformed_catalogs_fail_the_run(sync: ModuleType, loader: str, raw: bytes) -> None: + kwargs = {"now_ms": NOW_MS} if loader == "load_vercel" else {} + with pytest.raises(sync.SyncError): + getattr(sync, loader)(raw, **kwargs) + + +def _vercel_language_row(deprecated_at: int | None) -> bytes: + row = { + "id": "acme/chat-1", + "type": "language", + "context_window": 1000, + "max_tokens": 100, + "pricing": {"input": "0.000001", "output": "0.000002"}, + "deprecated_at": deprecated_at, + } + return json.dumps({"data": [row]}).encode() + + +def test_a_scheduled_deprecation_keeps_syncing_until_the_date(sync: ModuleType) -> None: + scheduled = sync.load_vercel(_vercel_language_row(NOW_MS + 1), now_ms=NOW_MS) + passed = sync.load_vercel(_vercel_language_row(NOW_MS), now_ms=NOW_MS) + + assert [entry.key for entry in scheduled.entries] == ["vercel_ai_gateway/acme/chat-1"] + assert dict(scheduled.skipped)["deprecated"] == 0 + assert passed.entries == () + assert dict(passed.skipped)["deprecated"] == 1 + + +def _repo(tmp_path: Path) -> Path: + for relpath in ("model_prices_and_context_window.json", "litellm/model_prices_and_context_window_backup.json"): + target = tmp_path / relpath + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(json.dumps(_base_map(), indent=4) + "\n") + return tmp_path + + +def test_write_updates_both_cost_map_files_identically(sync: ModuleType, tmp_path: Path, capsys) -> None: + repo = _repo(tmp_path) + body_file = tmp_path / "body.md" + + code = sync.main( + [ + "--write", + "--openrouter-json", + str(FIXTURES / "openrouter_models.json"), + "--vercel-json", + str(FIXTURES / "vercel_models.json"), + "--pr-body-file", + str(body_file), + "--repo-root", + str(repo), + ] + ) + + root = (repo / "model_prices_and_context_window.json").read_text() + backup = (repo / "litellm" / "model_prices_and_context_window_backup.json").read_text() + assert code == 0 + assert root == backup + assert root.endswith("}\n") + assert json.loads(root)["openrouter/inception/mercury-2.5-preview"]["input_cost_per_token"] == 4e-8 + assert "### Added (4)" in body_file.read_text() + assert capsys.readouterr().out.startswith("openrouter: added=4 updated=1 warnings=0") + + +def test_dry_run_touches_nothing(sync: ModuleType, tmp_path: Path, capsys) -> None: + repo = _repo(tmp_path) + before = (repo / "model_prices_and_context_window.json").read_bytes() + + code = sync.main( + [ + "--openrouter-json", + str(FIXTURES / "openrouter_models.json"), + "--vercel-json", + str(FIXTURES / "vercel_models.json"), + "--repo-root", + str(repo), + ] + ) + + assert code == 0 + assert (repo / "model_prices_and_context_window.json").read_bytes() == before + assert "dry run: no files were touched" in capsys.readouterr().out