mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
feat(ci): add the cost map sync bot for openrouter and vercel_ai_gateway
This commit is contained in:
parent
61bed79566
commit
d9e7448938
7 changed files with 1297 additions and 198 deletions
|
|
@ -1,159 +0,0 @@
|
|||
import asyncio
|
||||
import aiohttp
|
||||
import json
|
||||
|
||||
# Asynchronously fetch data from a given URL
|
||||
async def fetch_data(url):
|
||||
try:
|
||||
# Create an asynchronous session
|
||||
async with aiohttp.ClientSession() as session:
|
||||
# Send a GET request to the URL
|
||||
async with session.get(url) as resp:
|
||||
# Raise an error if the response status is not OK
|
||||
resp.raise_for_status()
|
||||
# Parse the response JSON
|
||||
resp_json = await resp.json()
|
||||
print("Fetch the data from URL.")
|
||||
# Return the 'data' field from the JSON response
|
||||
return resp_json['data']
|
||||
except Exception as e:
|
||||
# Print an error message if fetching data fails
|
||||
print("Error fetching data from URL:", e)
|
||||
return None
|
||||
|
||||
# Synchronize local data with remote data
|
||||
def sync_local_data_with_remote(local_data, remote_data):
|
||||
# Update existing keys in local_data with values from remote_data
|
||||
for key in (set(local_data) & set(remote_data)):
|
||||
local_data[key].update(remote_data[key])
|
||||
|
||||
# Add new keys from remote_data to local_data
|
||||
for key in (set(remote_data) - set(local_data)):
|
||||
local_data[key] = remote_data[key]
|
||||
|
||||
# Write data to the json file
|
||||
def write_to_file(file_path, data):
|
||||
try:
|
||||
# Open the file in write mode
|
||||
with open(file_path, "w") as file:
|
||||
# Dump the data as JSON into the file
|
||||
json.dump(data, file, indent=4)
|
||||
print("Values updated successfully.")
|
||||
except Exception as e:
|
||||
# Print an error message if writing to file fails
|
||||
print("Error updating JSON file:", e)
|
||||
|
||||
# Update the existing models and add the missing models for OpenRouter
|
||||
def transform_openrouter_data(data):
|
||||
transformed = {}
|
||||
for row in data:
|
||||
# Add the fields 'max_tokens' and 'input_cost_per_token'
|
||||
obj = {
|
||||
"max_tokens": row["context_length"],
|
||||
"input_cost_per_token": float(row["pricing"]["prompt"]),
|
||||
}
|
||||
|
||||
# Add 'max_output_tokens' as a field if it is not None
|
||||
if "top_provider" in row and "max_completion_tokens" in row["top_provider"] and row["top_provider"]["max_completion_tokens"] is not None:
|
||||
obj['max_output_tokens'] = int(row["top_provider"]["max_completion_tokens"])
|
||||
|
||||
# Add the field 'output_cost_per_token'
|
||||
obj.update({
|
||||
"output_cost_per_token": float(row["pricing"]["completion"]),
|
||||
})
|
||||
|
||||
# Add field 'input_cost_per_image' if it exists and is non-zero
|
||||
if "pricing" in row and "image" in row["pricing"] and float(row["pricing"]["image"]) != 0.0:
|
||||
obj['input_cost_per_image'] = float(row["pricing"]["image"])
|
||||
|
||||
# Add the fields 'litellm_provider' and 'mode'
|
||||
obj.update({
|
||||
"litellm_provider": "openrouter",
|
||||
"mode": "chat"
|
||||
})
|
||||
|
||||
# Add the 'supports_vision' field if the modality is 'multimodal'
|
||||
if row.get('architecture', {}).get('modality') == 'multimodal':
|
||||
obj['supports_vision'] = True
|
||||
|
||||
# Use a composite key to store the transformed object
|
||||
transformed[f'openrouter/{row["id"]}'] = obj
|
||||
|
||||
return transformed
|
||||
|
||||
# Update the existing models and add the missing models for Vercel AI Gateway
|
||||
def transform_vercel_ai_gateway_data(data):
|
||||
transformed = {}
|
||||
for row in data:
|
||||
obj = {
|
||||
"max_tokens": row["context_window"],
|
||||
"input_cost_per_token": float(row["pricing"]["input"]),
|
||||
"output_cost_per_token": float(row["pricing"]["output"]),
|
||||
'max_output_tokens': row['max_tokens'],
|
||||
'max_input_tokens': row["context_window"],
|
||||
}
|
||||
|
||||
# Handle cache pricing if available
|
||||
if "pricing" in row:
|
||||
if "input_cache_read" in row["pricing"] and row["pricing"]["input_cache_read"] is not None:
|
||||
obj['cache_read_input_token_cost'] = float(f"{float(row['pricing']['input_cache_read']):e}")
|
||||
|
||||
if "input_cache_write" in row["pricing"] and row["pricing"]["input_cache_write"] is not None:
|
||||
obj['cache_creation_input_token_cost'] = float(f"{float(row['pricing']['input_cache_write']):e}")
|
||||
|
||||
mode = "embedding" if "embedding" in row["id"].lower() else "chat"
|
||||
|
||||
obj.update({"litellm_provider": "vercel_ai_gateway", "mode": mode})
|
||||
|
||||
transformed[f'vercel_ai_gateway/{row["id"]}'] = obj
|
||||
|
||||
return transformed
|
||||
|
||||
|
||||
# Load local data from a specified file
|
||||
def load_local_data(file_path):
|
||||
try:
|
||||
# Open the file in read mode
|
||||
with open(file_path, "r") as file:
|
||||
# Load and return the JSON data
|
||||
return json.load(file)
|
||||
except FileNotFoundError:
|
||||
# Print an error message if the file is not found
|
||||
print("File not found:", file_path)
|
||||
return None
|
||||
except json.JSONDecodeError as e:
|
||||
# Print an error message if JSON decoding fails
|
||||
print("Error decoding JSON:", e)
|
||||
return None
|
||||
|
||||
def main():
|
||||
local_file_path = "model_prices_and_context_window.json" # Path to the local data file
|
||||
openrouter_url = "https://openrouter.ai/api/v1/models" # URL to fetch OpenRouter data
|
||||
vercel_ai_gateway_url = "https://ai-gateway.vercel.sh/v1/models" # URL to fetch Vercel AI Gateway data
|
||||
|
||||
# Load local data from file
|
||||
local_data = load_local_data(local_file_path)
|
||||
|
||||
# Fetch OpenRouter data
|
||||
openrouter_data = asyncio.run(fetch_data(openrouter_url))
|
||||
# Transform the fetched OpenRouter data
|
||||
openrouter_data = transform_openrouter_data(openrouter_data)
|
||||
|
||||
# Fetch Vercel AI Gateway data
|
||||
vercel_data = asyncio.run(fetch_data(vercel_ai_gateway_url))
|
||||
# Transform the fetched Vercel AI Gateway data
|
||||
vercel_data = transform_vercel_ai_gateway_data(vercel_data)
|
||||
|
||||
# Combine both datasets
|
||||
all_remote_data = {**openrouter_data, **vercel_data}
|
||||
|
||||
# If both local and openrouter data are available, synchronize and save
|
||||
if local_data and all_remote_data:
|
||||
sync_local_data_with_remote(local_data, all_remote_data)
|
||||
write_to_file(local_file_path, local_data)
|
||||
else:
|
||||
print("Failed to fetch model data from either local file or URL.")
|
||||
|
||||
# Entry point of the script
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -1,39 +0,0 @@
|
|||
name: Updates model_prices_and_context_window.json and Create Pull Request
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: "0 0 * * 0" # Run every Sundays at midnight
|
||||
#- cron: "0 0 * * *" # Run daily at midnight
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
auto_update_price_and_context_window:
|
||||
if: github.repository == 'BerriAI/litellm'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Set up uv
|
||||
uses: ./.github/actions/setup-uv-with-retries
|
||||
with:
|
||||
version: "0.10.9"
|
||||
- name: Update JSON Data
|
||||
run: |
|
||||
uv run --frozen --with 'aiohttp==3.13.3' python ".github/scripts/auto_update_price_and_context_window_file.py"
|
||||
- name: Regenerate JSON Schema
|
||||
run: |
|
||||
uv run --frozen python ci_cd/generate_model_prices_schema.py
|
||||
- name: Create Pull Request
|
||||
run: |
|
||||
git add model_prices_and_context_window.json model_prices_and_context_window.schema.json
|
||||
git commit -m "Update model_prices_and_context_window.json file: $(date +'%Y-%m-%d')"
|
||||
gh pr create --title "Update model_prices_and_context_window.json file" \
|
||||
--body "Automated update for model_prices_and_context_window.json" \
|
||||
--head auto-update-price-and-context-window-$(date +'%Y-%m-%d') \
|
||||
--base main
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GH_TOKEN }}
|
||||
118
.github/workflows/cost-map-sync.yml
vendored
Normal file
118
.github/workflows/cost-map-sync.yml
vendored
Normal file
|
|
@ -0,0 +1,118 @@
|
|||
name: Cost map sync
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: "*/5 * * * *"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
dry_run:
|
||||
description: "Print the diff without opening a PR"
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
concurrency:
|
||||
group: cost-map-sync
|
||||
cancel-in-progress: false
|
||||
|
||||
env:
|
||||
BRANCH_PREFIX: litellm_cost_map_sync_
|
||||
PR_TITLE: "feat(models): sync openrouter and vercel_ai_gateway pricing"
|
||||
|
||||
jobs:
|
||||
cost-map-sync:
|
||||
if: github.repository == 'BerriAI/litellm'
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
BOT_APP_ID: ${{ secrets.COST_MAP_BOT_APP_ID }}
|
||||
steps:
|
||||
- uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Mint the bot token
|
||||
id: bot
|
||||
if: env.BOT_APP_ID != ''
|
||||
uses: actions/create-github-app-token@a8d616148505b5069dccd32f177bb87d7f39123b # v2.1.1
|
||||
with:
|
||||
app-id: ${{ secrets.COST_MAP_BOT_APP_ID }}
|
||||
private-key: ${{ secrets.COST_MAP_BOT_PRIVATE_KEY }}
|
||||
- name: Set up uv
|
||||
uses: ./.github/actions/setup-uv-with-retries
|
||||
with:
|
||||
version: "0.10.9"
|
||||
- name: Look for an already-open sync PR
|
||||
id: existing
|
||||
run: |
|
||||
open_pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --json headRefName \
|
||||
--search "in:title \"$PR_TITLE\"" \
|
||||
--jq "[.[].headRefName | select(startswith(\"$BRANCH_PREFIX\"))] | first // empty")"
|
||||
echo "open_pr=$open_pr" >> "$GITHUB_OUTPUT"
|
||||
if [ -n "$open_pr" ]; then
|
||||
echo "An open sync PR already exists on branch $open_pr; skipping this run."
|
||||
fi
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.bot.outputs.token || secrets.GH_TOKEN || github.token }}
|
||||
- name: Run the sync
|
||||
if: steps.existing.outputs.open_pr == ''
|
||||
run: |
|
||||
uv run --frozen python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md"
|
||||
uv run --frozen python ci_cd/generate_model_prices_schema.py
|
||||
- name: Open the sync PR
|
||||
id: pr
|
||||
if: steps.existing.outputs.open_pr == '' && !inputs.dry_run
|
||||
run: |
|
||||
if git diff --quiet; then
|
||||
echo "Registry already in sync; no PR needed."
|
||||
exit 0
|
||||
fi
|
||||
branch="${BRANCH_PREFIX}$(date -u +'%Y-%m-%d-%H%M')"
|
||||
if [ -n "$BOT_APP_ID" ]; then
|
||||
bot_user_id="$(gh api "users/${BOT_LOGIN}[bot]" --jq .id)"
|
||||
git config user.name "${BOT_LOGIN}[bot]"
|
||||
git config user.email "${bot_user_id}+${BOT_LOGIN}[bot]@users.noreply.github.com"
|
||||
else
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
||||
fi
|
||||
git checkout -b "$branch"
|
||||
git add model_prices_and_context_window.json \
|
||||
litellm/model_prices_and_context_window_backup.json \
|
||||
model_prices_and_context_window.schema.json
|
||||
git commit -m "feat(models): sync openrouter and vercel_ai_gateway pricing $(date -u +'%Y-%m-%d %H:%M')"
|
||||
gh auth setup-git
|
||||
git push origin "$branch"
|
||||
url="$(gh pr create --title "$PR_TITLE" \
|
||||
--body-file "$RUNNER_TEMP/pr_body.md" \
|
||||
--head "$branch" \
|
||||
--base "$GITHUB_REF_NAME")"
|
||||
echo "url=$url" >> "$GITHUB_OUTPUT"
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.bot.outputs.token || secrets.GH_TOKEN || github.token }}
|
||||
BOT_LOGIN: ${{ steps.bot.outputs.app-slug }}
|
||||
- name: Merge once every required check passes
|
||||
if: steps.pr.outputs.url != '' && env.BOT_APP_ID != ''
|
||||
timeout-minutes: 120
|
||||
run: |
|
||||
while true; do
|
||||
guard="$(gh pr checks "$PR_URL" --json name,state \
|
||||
--jq '.[] | select(.name == "cost-map-guard") | .state' || true)"
|
||||
required="$(gh pr checks "$PR_URL" --required --json bucket \
|
||||
--jq 'map(.bucket) | unique | join(",")' || true)"
|
||||
case "$guard,$required" in
|
||||
*FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*)
|
||||
echo "A check failed (cost-map-guard=$guard, required buckets=$required); leaving $PR_URL open for a human."
|
||||
exit 1
|
||||
;;
|
||||
SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping)
|
||||
gh pr merge "$PR_URL" --repo "$GITHUB_REPOSITORY" --merge --delete-branch
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
sleep 30
|
||||
done
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.bot.outputs.token }}
|
||||
PR_URL: ${{ steps.pr.outputs.url }}
|
||||
474
scripts/sync_cost_map.py
Normal file
474
scripts/sync_cost_map.py
Normal file
|
|
@ -0,0 +1,474 @@
|
|||
"""Sync the openrouter and vercel_ai_gateway entries of model_prices_and_context_window.json with the live catalogs.
|
||||
|
||||
Pulls ``GET https://openrouter.ai/api/v1/models`` and ``GET https://ai-gateway.vercel.sh/v1/models``, maps the
|
||||
catalog fields onto registry fields, and diffs the result against the registry. Dry run (the default) prints the
|
||||
diff summary and the generated PR body; ``--write`` applies the changes to the root cost map and its ``litellm/``
|
||||
backup copy.
|
||||
|
||||
Policy:
|
||||
- Both catalogs price per token as decimal strings; values are normalized to six significant digits.
|
||||
- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed, a
|
||||
capability flag the catalog does not claim stays as curated, and a curated output ceiling is kept.
|
||||
- Router models and rows without a usable prompt and completion price are skipped.
|
||||
- A registry entry absent from its catalog is left untouched; retiring a model stays a human call.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
from functools import reduce
|
||||
from pathlib import Path
|
||||
from types import MappingProxyType
|
||||
from typing import Final, Literal
|
||||
|
||||
import httpx
|
||||
from pydantic import BaseModel, TypeAdapter, ValidationError
|
||||
|
||||
COST_MAP_RELPATHS: Final = (
|
||||
"model_prices_and_context_window.json",
|
||||
"litellm/model_prices_and_context_window_backup.json",
|
||||
)
|
||||
OPENROUTER_MODELS_URL: Final = "https://openrouter.ai/api/v1/models"
|
||||
VERCEL_MODELS_URL: Final = "https://ai-gateway.vercel.sh/v1/models"
|
||||
VERCEL_TYPE_TO_MODE: Final = MappingProxyType({"language": "chat", "embedding": "embedding"})
|
||||
ADD_ONLY_FIELDS: Final = frozenset({"max_output_tokens", "max_tokens"})
|
||||
|
||||
Provider = Literal["openrouter", "vercel_ai_gateway"]
|
||||
RegistryEntry = dict[str, object]
|
||||
CostMap = dict[str, object]
|
||||
|
||||
|
||||
class SyncError(RuntimeError):
|
||||
pass
|
||||
|
||||
|
||||
class OpenRouterPricing(BaseModel):
|
||||
prompt: str
|
||||
completion: str
|
||||
input_cache_read: str | None = None
|
||||
input_cache_write: str | None = None
|
||||
internal_reasoning: str | None = None
|
||||
|
||||
|
||||
class OpenRouterArchitecture(BaseModel):
|
||||
input_modalities: tuple[str, ...] | None = None
|
||||
|
||||
|
||||
class OpenRouterTopProvider(BaseModel):
|
||||
max_completion_tokens: int | None = None
|
||||
|
||||
|
||||
class OpenRouterModel(BaseModel):
|
||||
id: str
|
||||
context_length: int | None = None
|
||||
architecture: OpenRouterArchitecture | None = None
|
||||
top_provider: OpenRouterTopProvider = OpenRouterTopProvider()
|
||||
pricing: OpenRouterPricing
|
||||
supported_parameters: tuple[str, ...] | None = None
|
||||
|
||||
|
||||
class VercelPricing(BaseModel):
|
||||
input: str | None = None
|
||||
output: str | None = None
|
||||
input_cache_read: str | None = None
|
||||
input_cache_write: str | None = None
|
||||
|
||||
|
||||
class VercelModalities(BaseModel):
|
||||
input: tuple[str, ...] | None = None
|
||||
|
||||
|
||||
class VercelModel(BaseModel):
|
||||
id: str
|
||||
type: str
|
||||
context_window: int | None = None
|
||||
max_tokens: int | None = None
|
||||
modalities: VercelModalities | None = None
|
||||
pricing: VercelPricing = VercelPricing()
|
||||
supported_parameters: tuple[str, ...] | None = None
|
||||
deprecated_at: int | None = None
|
||||
|
||||
|
||||
OPENROUTER_ADAPTER: Final = TypeAdapter(list[OpenRouterModel])
|
||||
VERCEL_ADAPTER: Final = TypeAdapter(list[VercelModel])
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class CatalogEntry:
|
||||
key: str
|
||||
provider: Provider
|
||||
mode: str
|
||||
source: str
|
||||
fields: Mapping[str, object]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Catalog:
|
||||
provider: Provider
|
||||
entries: tuple[CatalogEntry, ...]
|
||||
skipped: Mapping[str, int]
|
||||
|
||||
|
||||
def per_token(price: float) -> float:
|
||||
return float(f"{price:.6g}")
|
||||
|
||||
|
||||
def _token_price(raw: str | None) -> float | None:
|
||||
if raw is None:
|
||||
return None
|
||||
value: Final = float(raw)
|
||||
return per_token(value) if value >= 0 else None
|
||||
|
||||
|
||||
def _extra_price(raw: str | None) -> float | None:
|
||||
price: Final = _token_price(raw)
|
||||
return price if price else None
|
||||
|
||||
|
||||
def _flags(parameters: Sequence[str] | None, modalities: Sequence[str] | None) -> Mapping[str, bool]:
|
||||
params: Final = frozenset(parameters or ())
|
||||
mods: Final = frozenset(modalities or ())
|
||||
claims: Final = {
|
||||
"supports_function_calling": "tools" in params,
|
||||
"supports_tool_choice": "tool_choice" in params,
|
||||
"supports_reasoning": "reasoning" in params,
|
||||
"supports_response_schema": "structured_outputs" in params,
|
||||
"supports_vision": "image" in mods,
|
||||
"supports_pdf_input": bool({"file", "pdf"} & mods),
|
||||
"supports_audio_input": "audio" in mods,
|
||||
"supports_video_input": "video" in mods,
|
||||
}
|
||||
return MappingProxyType({name: True for name, claimed in claims.items() if claimed})
|
||||
|
||||
|
||||
def _limits(max_input: int | None, max_output: int | None) -> Mapping[str, int]:
|
||||
ceiling: Final = max_output if max_output is not None else max_input
|
||||
return MappingProxyType(
|
||||
{
|
||||
**({"max_input_tokens": max_input} if max_input is not None else {}),
|
||||
**({"max_output_tokens": max_output} if max_output is not None else {}),
|
||||
**({"max_tokens": ceiling} if ceiling is not None else {}),
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _priced(name: str, price: float | None) -> Mapping[str, float]:
|
||||
return MappingProxyType({name: price} if price is not None else {})
|
||||
|
||||
|
||||
def _openrouter_entry(model: OpenRouterModel) -> CatalogEntry | None:
|
||||
prompt: Final = _token_price(model.pricing.prompt)
|
||||
completion: Final = _token_price(model.pricing.completion)
|
||||
if prompt is None or completion is None:
|
||||
return None
|
||||
fields: Final = {
|
||||
"input_cost_per_token": prompt,
|
||||
"output_cost_per_token": completion,
|
||||
**_limits(model.context_length, model.top_provider.max_completion_tokens),
|
||||
**_priced("cache_read_input_token_cost", _extra_price(model.pricing.input_cache_read)),
|
||||
**_priced("cache_creation_input_token_cost", _extra_price(model.pricing.input_cache_write)),
|
||||
**_priced("output_cost_per_reasoning_token", _extra_price(model.pricing.internal_reasoning)),
|
||||
**_flags(model.supported_parameters, model.architecture.input_modalities if model.architecture else None),
|
||||
}
|
||||
return CatalogEntry(
|
||||
key=f"openrouter/{model.id}",
|
||||
provider="openrouter",
|
||||
mode="chat",
|
||||
source=f"https://openrouter.ai/{model.id}",
|
||||
fields=MappingProxyType(fields),
|
||||
)
|
||||
|
||||
|
||||
def _vercel_entry(model: VercelModel) -> CatalogEntry | None:
|
||||
mode: Final = VERCEL_TYPE_TO_MODE.get(model.type)
|
||||
prompt: Final = _token_price(model.pricing.input)
|
||||
completion: Final = _token_price(model.pricing.output if mode != "embedding" else model.pricing.output or "0")
|
||||
if mode is None or prompt is None or completion is None:
|
||||
return None
|
||||
fields: Final = {
|
||||
"input_cost_per_token": prompt,
|
||||
"output_cost_per_token": completion,
|
||||
**_limits(model.context_window, model.max_tokens),
|
||||
**_priced("cache_read_input_token_cost", _extra_price(model.pricing.input_cache_read)),
|
||||
**_priced("cache_creation_input_token_cost", _extra_price(model.pricing.input_cache_write)),
|
||||
**(
|
||||
_flags(model.supported_parameters, model.modalities.input if model.modalities else None)
|
||||
if mode == "chat"
|
||||
else {}
|
||||
),
|
||||
}
|
||||
return CatalogEntry(
|
||||
key=f"vercel_ai_gateway/{model.id}",
|
||||
provider="vercel_ai_gateway",
|
||||
mode=mode,
|
||||
source=f"https://vercel.com/ai-gateway/models/{model.id.rsplit('/', 1)[-1]}",
|
||||
fields=MappingProxyType(fields),
|
||||
)
|
||||
|
||||
|
||||
def _rows(raw: bytes, url: str) -> object:
|
||||
parsed: Final = json.loads(raw)
|
||||
rows: Final = parsed.get("data") if isinstance(parsed, dict) else parsed
|
||||
if not isinstance(rows, list) or not rows:
|
||||
raise SyncError(f"GET {url} returned no model rows")
|
||||
return rows
|
||||
|
||||
|
||||
def load_openrouter(raw: bytes) -> Catalog:
|
||||
try:
|
||||
models: Final = OPENROUTER_ADAPTER.validate_python(_rows(raw, OPENROUTER_MODELS_URL))
|
||||
except ValidationError as error:
|
||||
raise SyncError(f"the OpenRouter catalog no longer matches the expected shape: {error}") from error
|
||||
entries: Final = tuple(entry for entry in map(_openrouter_entry, models) if entry is not None)
|
||||
return Catalog(
|
||||
provider="openrouter",
|
||||
entries=entries,
|
||||
skipped=MappingProxyType({"unpriced or router": len(models) - len(entries)}),
|
||||
)
|
||||
|
||||
|
||||
def load_vercel(raw: bytes, now_ms: int) -> Catalog:
|
||||
try:
|
||||
models: Final = VERCEL_ADAPTER.validate_python(_rows(raw, VERCEL_MODELS_URL))
|
||||
except ValidationError as error:
|
||||
raise SyncError(f"the Vercel AI Gateway catalog no longer matches the expected shape: {error}") from error
|
||||
live: Final = tuple(model for model in models if model.deprecated_at is None or model.deprecated_at > now_ms)
|
||||
token_priced: Final = tuple(model for model in live if model.type in VERCEL_TYPE_TO_MODE)
|
||||
entries: Final = tuple(entry for entry in map(_vercel_entry, token_priced) if entry is not None)
|
||||
return Catalog(
|
||||
provider="vercel_ai_gateway",
|
||||
entries=entries,
|
||||
skipped=MappingProxyType(
|
||||
{
|
||||
"deprecated": len(models) - len(live),
|
||||
"not token priced": len(live) - len(token_priced),
|
||||
"no usable price": len(token_priced) - len(entries),
|
||||
}
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ProviderOutcome:
|
||||
provider: Provider
|
||||
added: tuple[str, ...]
|
||||
updated: tuple[str, ...]
|
||||
warnings: tuple[str, ...]
|
||||
skipped: Mapping[str, int]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SyncOutcome:
|
||||
cost_map: CostMap
|
||||
providers: tuple[ProviderOutcome, ...]
|
||||
|
||||
@property
|
||||
def has_changes(self) -> bool:
|
||||
return any(outcome.added or outcome.updated for outcome in self.providers)
|
||||
|
||||
|
||||
def _new_entry(entry: CatalogEntry) -> RegistryEntry:
|
||||
return dict(
|
||||
sorted(
|
||||
{
|
||||
**entry.fields,
|
||||
"litellm_provider": entry.provider,
|
||||
"mode": entry.mode,
|
||||
"source": entry.source,
|
||||
}.items()
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def _updated_entry(existing: RegistryEntry, entry: CatalogEntry) -> tuple[RegistryEntry, tuple[str, ...]]:
|
||||
keep_limits: Final = not ADD_ONLY_FIELDS.isdisjoint(existing)
|
||||
desired: Final = {
|
||||
name: value for name, value in entry.fields.items() if not (keep_limits and name in ADD_ONLY_FIELDS)
|
||||
}
|
||||
changes: Final = tuple(
|
||||
f"{name}: {existing.get(name)!r} -> {value!r}" for name, value in desired.items() if existing.get(name) != value
|
||||
)
|
||||
return dict(sorted({**existing, **desired}.items())), changes
|
||||
|
||||
|
||||
def _with_new_keys_in_block(ordered: CostMap, result: CostMap, new_keys: Sequence[str], prefix: str) -> CostMap:
|
||||
provider_keys: Final = tuple(key for key in ordered if key.startswith(prefix))
|
||||
if not new_keys:
|
||||
return {key: result[key] for key in ordered}
|
||||
if not provider_keys:
|
||||
return {**{key: result[key] for key in ordered}, **{key: result[key] for key in sorted(new_keys)}}
|
||||
block_end: Final = provider_keys[-1]
|
||||
return {
|
||||
key: value
|
||||
for existing in ordered
|
||||
for key, value in (
|
||||
(existing, result[existing]),
|
||||
*((new, result[new]) for new in sorted(new_keys) if existing == block_end),
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Added:
|
||||
key: str
|
||||
entry: RegistryEntry
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Updated:
|
||||
key: str
|
||||
entry: RegistryEntry
|
||||
line: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Warned:
|
||||
line: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Unchanged:
|
||||
pass
|
||||
|
||||
|
||||
EntrySync = Added | Updated | Warned | Unchanged
|
||||
|
||||
|
||||
def _sync_entry(existing: object, entry: CatalogEntry) -> EntrySync:
|
||||
if not isinstance(existing, dict):
|
||||
return Added(key=entry.key, entry=_new_entry(entry))
|
||||
if existing.get("mode") != entry.mode:
|
||||
return Warned(
|
||||
line=f"`{entry.key}` has curated mode {existing.get('mode')!r} but the catalog maps to "
|
||||
f"{entry.mode!r}; left unchanged"
|
||||
)
|
||||
new_entry, changes = _updated_entry(existing, entry)
|
||||
if not changes:
|
||||
return Unchanged()
|
||||
return Updated(key=entry.key, entry=new_entry, line=f"{entry.key}: " + "; ".join(changes))
|
||||
|
||||
|
||||
SyncState = tuple[CostMap, tuple[ProviderOutcome, ...]]
|
||||
|
||||
|
||||
def _sync_provider(state: SyncState, catalog: Catalog) -> SyncState:
|
||||
cost_map, outcomes = state
|
||||
syncs: Final = tuple(
|
||||
_sync_entry(cost_map.get(entry.key), entry) for entry in sorted(catalog.entries, key=lambda item: item.key)
|
||||
)
|
||||
outcome: Final = ProviderOutcome(
|
||||
provider=catalog.provider,
|
||||
added=tuple(sync.key for sync in syncs if isinstance(sync, Added)),
|
||||
updated=tuple(sync.line for sync in syncs if isinstance(sync, Updated)),
|
||||
warnings=tuple(sync.line for sync in syncs if isinstance(sync, Warned)),
|
||||
skipped=catalog.skipped,
|
||||
)
|
||||
merged: Final = {**cost_map, **{sync.key: sync.entry for sync in syncs if isinstance(sync, Added | Updated)}}
|
||||
return merged, (*outcomes, outcome)
|
||||
|
||||
|
||||
def compute_sync(cost_map: CostMap, catalogs: Sequence[Catalog]) -> SyncOutcome:
|
||||
synced, outcomes = reduce(_sync_provider, catalogs, (dict(cost_map), ()))
|
||||
return SyncOutcome(cost_map=_ordered_result(cost_map, synced, outcomes), providers=outcomes)
|
||||
|
||||
|
||||
def _ordered_result(cost_map: CostMap, result: CostMap, outcomes: Sequence[ProviderOutcome]) -> CostMap:
|
||||
return reduce(
|
||||
lambda ordered, outcome: _with_new_keys_in_block(ordered, result, outcome.added, f"{outcome.provider}/"),
|
||||
outcomes,
|
||||
{key: result[key] for key in cost_map},
|
||||
)
|
||||
|
||||
|
||||
def _section_block(title: str, lines: Sequence[str], backtick: bool) -> str:
|
||||
bullets: Final = "\n".join(f"- `{line}`" if backtick else f"- {line}" for line in lines) or "- none"
|
||||
return f"### {title} ({len(lines)})\n{bullets}\n"
|
||||
|
||||
|
||||
def _provider_body(outcome: ProviderOutcome) -> str:
|
||||
skipped: Final = ", ".join(f"{reason} ({count})" for reason, count in sorted(outcome.skipped.items())) or "none"
|
||||
return (
|
||||
f"## {outcome.provider}\n"
|
||||
"\n"
|
||||
f"{_section_block('Added', outcome.added, backtick=True)}"
|
||||
"\n"
|
||||
f"{_section_block('Updated', outcome.updated, backtick=True)}"
|
||||
"\n"
|
||||
f"{_section_block('Warnings needing a human call', outcome.warnings, backtick=False)}"
|
||||
"\n"
|
||||
f"Catalog rows skipped: {skipped}\n"
|
||||
)
|
||||
|
||||
|
||||
def render_pr_body(outcome: SyncOutcome) -> str:
|
||||
return (
|
||||
"Automated sync of the openrouter and vercel_ai_gateway entries in model_prices_and_context_window.json "
|
||||
f"against `GET {OPENROUTER_MODELS_URL}` and `GET {VERCEL_MODELS_URL}` by scripts/sync_cost_map.py. "
|
||||
"The cost-map-guard check enforces that this PR only adds or reprices models.\n"
|
||||
"\n" + "\n".join(_provider_body(provider) for provider in outcome.providers)
|
||||
)
|
||||
|
||||
|
||||
def render_summary(outcome: SyncOutcome) -> str:
|
||||
return " ".join(
|
||||
f"{provider.provider}: added={len(provider.added)} updated={len(provider.updated)} "
|
||||
f"warnings={len(provider.warnings)}"
|
||||
for provider in outcome.providers
|
||||
)
|
||||
|
||||
|
||||
def _fetch(url: str) -> bytes:
|
||||
response: Final = httpx.get(url, timeout=30, follow_redirects=True)
|
||||
if response.status_code != 200:
|
||||
raise SyncError(f"GET {url} returned {response.status_code}")
|
||||
return response.content
|
||||
|
||||
|
||||
def _serialize(cost_map: CostMap) -> str:
|
||||
return json.dumps(cost_map, indent=4, ensure_ascii=False) + "\n"
|
||||
|
||||
|
||||
def main(argv: Sequence[str]) -> int:
|
||||
parser: Final = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--write", action="store_true", help="apply the sync to the cost map files (default: dry run)")
|
||||
parser.add_argument("--openrouter-json", type=Path, help="recorded OpenRouter catalog instead of the live API")
|
||||
parser.add_argument("--vercel-json", type=Path, help="recorded Vercel AI Gateway catalog instead of the live API")
|
||||
parser.add_argument("--pr-body-file", type=Path, help="write the generated PR body to this path")
|
||||
parser.add_argument("--repo-root", type=Path, default=Path(__file__).resolve().parent.parent)
|
||||
args: Final = parser.parse_args(argv)
|
||||
|
||||
openrouter_raw: Final = (
|
||||
args.openrouter_json.read_bytes() if args.openrouter_json is not None else _fetch(OPENROUTER_MODELS_URL)
|
||||
)
|
||||
vercel_raw: Final = args.vercel_json.read_bytes() if args.vercel_json is not None else _fetch(VERCEL_MODELS_URL)
|
||||
catalogs: Final = (load_openrouter(openrouter_raw), load_vercel(vercel_raw, now_ms=int(time.time() * 1000)))
|
||||
|
||||
cost_map_path: Final = args.repo_root / COST_MAP_RELPATHS[0]
|
||||
cost_map: Final = json.loads(cost_map_path.read_text())
|
||||
outcome: Final = compute_sync(cost_map, catalogs)
|
||||
body: Final = render_pr_body(outcome)
|
||||
|
||||
if args.pr_body_file is not None:
|
||||
args.pr_body_file.write_text(body)
|
||||
if args.write and outcome.has_changes:
|
||||
for relpath in COST_MAP_RELPATHS:
|
||||
(args.repo_root / relpath).write_text(_serialize(outcome.cost_map))
|
||||
print(render_summary(outcome))
|
||||
print()
|
||||
print(body)
|
||||
if not args.write:
|
||||
print("dry run: no files were touched")
|
||||
elif not outcome.has_changes:
|
||||
print("registry already in sync: no files were touched")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
raise SystemExit(main(sys.argv[1:]))
|
||||
except SyncError as error:
|
||||
print(f"SYNC FAILED: {error}", file=sys.stderr)
|
||||
raise SystemExit(1) from error
|
||||
209
tests/test_litellm/fixtures/cost_map_sync/openrouter_models.json
Normal file
209
tests/test_litellm/fixtures/cost_map_sync/openrouter_models.json
Normal file
|
|
@ -0,0 +1,209 @@
|
|||
{
|
||||
"data": [
|
||||
{
|
||||
"id": "cohere/north-mini-code:free",
|
||||
"context_length": 256000,
|
||||
"architecture": {
|
||||
"input_modalities": [
|
||||
"text"
|
||||
]
|
||||
},
|
||||
"top_provider": {
|
||||
"max_completion_tokens": 64000
|
||||
},
|
||||
"pricing": {
|
||||
"prompt": "0",
|
||||
"completion": "0"
|
||||
},
|
||||
"supported_parameters": [
|
||||
"frequency_penalty",
|
||||
"include_reasoning",
|
||||
"max_tokens",
|
||||
"presence_penalty",
|
||||
"reasoning",
|
||||
"seed",
|
||||
"stop",
|
||||
"temperature",
|
||||
"tool_choice",
|
||||
"tools",
|
||||
"top_k",
|
||||
"top_p"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "deepseek/deepseek-v4-pro-0813",
|
||||
"context_length": 1048576,
|
||||
"architecture": {
|
||||
"input_modalities": [
|
||||
"text"
|
||||
]
|
||||
},
|
||||
"top_provider": {
|
||||
"max_completion_tokens": 384000
|
||||
},
|
||||
"pricing": {
|
||||
"prompt": "0.00000057948",
|
||||
"completion": "0.00000173844",
|
||||
"input_cache_read": "0.000000019316"
|
||||
},
|
||||
"supported_parameters": [
|
||||
"frequency_penalty",
|
||||
"include_reasoning",
|
||||
"logit_bias",
|
||||
"logprobs",
|
||||
"max_tokens",
|
||||
"min_p",
|
||||
"presence_penalty",
|
||||
"reasoning",
|
||||
"reasoning_effort",
|
||||
"repetition_penalty",
|
||||
"response_format",
|
||||
"seed",
|
||||
"stop",
|
||||
"structured_outputs",
|
||||
"temperature",
|
||||
"tool_choice",
|
||||
"tools",
|
||||
"top_k",
|
||||
"top_logprobs",
|
||||
"top_p"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "google/gemma-4-26b-a4b-it:free",
|
||||
"context_length": 262144,
|
||||
"architecture": {
|
||||
"input_modalities": [
|
||||
"image",
|
||||
"text",
|
||||
"video"
|
||||
]
|
||||
},
|
||||
"top_provider": {
|
||||
"max_completion_tokens": 32768
|
||||
},
|
||||
"pricing": {
|
||||
"prompt": "0",
|
||||
"completion": "0"
|
||||
},
|
||||
"supported_parameters": [
|
||||
"include_reasoning",
|
||||
"max_tokens",
|
||||
"reasoning",
|
||||
"response_format",
|
||||
"seed",
|
||||
"temperature",
|
||||
"tool_choice",
|
||||
"tools",
|
||||
"top_p"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "inception/mercury-2.5-preview",
|
||||
"context_length": 260000,
|
||||
"architecture": {
|
||||
"input_modalities": [
|
||||
"text"
|
||||
]
|
||||
},
|
||||
"top_provider": {
|
||||
"max_completion_tokens": 65536
|
||||
},
|
||||
"pricing": {
|
||||
"prompt": "0.00000004",
|
||||
"completion": "0.00000015",
|
||||
"input_cache_read": "0.000000004"
|
||||
},
|
||||
"supported_parameters": [
|
||||
"include_reasoning",
|
||||
"max_tokens",
|
||||
"reasoning",
|
||||
"reasoning_effort",
|
||||
"response_format",
|
||||
"stop",
|
||||
"structured_outputs",
|
||||
"temperature",
|
||||
"tool_choice",
|
||||
"tools"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "openai/gpt-5-mini",
|
||||
"context_length": 400000,
|
||||
"architecture": {
|
||||
"input_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"file"
|
||||
]
|
||||
},
|
||||
"top_provider": {
|
||||
"max_completion_tokens": 128000
|
||||
},
|
||||
"pricing": {
|
||||
"prompt": "0.00000025",
|
||||
"completion": "0.000002",
|
||||
"web_search": "0.01",
|
||||
"input_cache_read": "0.000000025",
|
||||
"image": "0.003613"
|
||||
},
|
||||
"supported_parameters": [
|
||||
"include_reasoning",
|
||||
"max_completion_tokens",
|
||||
"max_tokens",
|
||||
"reasoning",
|
||||
"reasoning_effort",
|
||||
"response_format",
|
||||
"seed",
|
||||
"structured_outputs",
|
||||
"tool_choice",
|
||||
"tools"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "openrouter/auto",
|
||||
"context_length": 2000000,
|
||||
"architecture": {
|
||||
"input_modalities": [
|
||||
"text",
|
||||
"image",
|
||||
"audio",
|
||||
"file",
|
||||
"video"
|
||||
]
|
||||
},
|
||||
"top_provider": {
|
||||
"max_completion_tokens": null
|
||||
},
|
||||
"pricing": {
|
||||
"prompt": "-1",
|
||||
"completion": "-1"
|
||||
},
|
||||
"supported_parameters": [
|
||||
"frequency_penalty",
|
||||
"include_reasoning",
|
||||
"logit_bias",
|
||||
"logprobs",
|
||||
"max_tokens",
|
||||
"min_p",
|
||||
"prediction",
|
||||
"presence_penalty",
|
||||
"reasoning",
|
||||
"reasoning_effort",
|
||||
"repetition_penalty",
|
||||
"response_format",
|
||||
"seed",
|
||||
"stop",
|
||||
"structured_outputs",
|
||||
"temperature",
|
||||
"tool_choice",
|
||||
"tools",
|
||||
"top_a",
|
||||
"top_k",
|
||||
"top_logprobs",
|
||||
"top_p",
|
||||
"web_search_options"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
142
tests/test_litellm/fixtures/cost_map_sync/vercel_models.json
Normal file
142
tests/test_litellm/fixtures/cost_map_sync/vercel_models.json
Normal file
|
|
@ -0,0 +1,142 @@
|
|||
{
|
||||
"object": "list",
|
||||
"data": [
|
||||
{
|
||||
"id": "alibaba/qwen3-embedding-0.6b",
|
||||
"type": "embedding",
|
||||
"context_window": 32768,
|
||||
"max_tokens": 32768,
|
||||
"modalities": {
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"output": [
|
||||
"text"
|
||||
]
|
||||
},
|
||||
"pricing": {
|
||||
"input": "0.00000001"
|
||||
},
|
||||
"supported_parameters": null,
|
||||
"deprecated_at": null
|
||||
},
|
||||
{
|
||||
"id": "bfl/flux-2-flex",
|
||||
"type": "image",
|
||||
"context_window": 0,
|
||||
"max_tokens": 0,
|
||||
"modalities": {
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"output": [
|
||||
"image"
|
||||
]
|
||||
},
|
||||
"pricing": {},
|
||||
"supported_parameters": null,
|
||||
"deprecated_at": null
|
||||
},
|
||||
{
|
||||
"id": "openai/gpt-4o-mini-transcribe",
|
||||
"type": "transcription",
|
||||
"context_window": null,
|
||||
"max_tokens": null,
|
||||
"modalities": {
|
||||
"input": [
|
||||
"audio"
|
||||
],
|
||||
"output": [
|
||||
"text"
|
||||
]
|
||||
},
|
||||
"pricing": {
|
||||
"input": "0.00000125",
|
||||
"output": "0.000005"
|
||||
},
|
||||
"supported_parameters": null,
|
||||
"deprecated_at": 1750000000000
|
||||
},
|
||||
{
|
||||
"id": "openai/gpt-5-mini",
|
||||
"type": "language",
|
||||
"context_window": 400000,
|
||||
"max_tokens": 128000,
|
||||
"modalities": {
|
||||
"input": [
|
||||
"text",
|
||||
"image",
|
||||
"pdf"
|
||||
],
|
||||
"output": [
|
||||
"text"
|
||||
]
|
||||
},
|
||||
"pricing": {
|
||||
"input": "0.00000025",
|
||||
"output": "0.000002",
|
||||
"input_cache_read": "0.000000025"
|
||||
},
|
||||
"supported_parameters": [
|
||||
"max_tokens",
|
||||
"stop",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"reasoning",
|
||||
"include_reasoning"
|
||||
],
|
||||
"deprecated_at": null
|
||||
},
|
||||
{
|
||||
"id": "perplexity/sonar",
|
||||
"type": "language",
|
||||
"context_window": 127000,
|
||||
"max_tokens": 8000,
|
||||
"modalities": {
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"output": [
|
||||
"text"
|
||||
]
|
||||
},
|
||||
"pricing": {},
|
||||
"supported_parameters": [
|
||||
"max_tokens",
|
||||
"temperature",
|
||||
"stop"
|
||||
],
|
||||
"deprecated_at": null
|
||||
},
|
||||
{
|
||||
"id": "zai/glm-4.6",
|
||||
"type": "language",
|
||||
"context_window": 200000,
|
||||
"max_tokens": 96000,
|
||||
"modalities": {
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"output": [
|
||||
"text"
|
||||
]
|
||||
},
|
||||
"pricing": {
|
||||
"input": "0.0000006",
|
||||
"output": "0.0000022",
|
||||
"input_cache_read": "0.00000011"
|
||||
},
|
||||
"supported_parameters": [
|
||||
"max_tokens",
|
||||
"temperature",
|
||||
"stop",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"reasoning",
|
||||
"include_reasoning"
|
||||
],
|
||||
"deprecated_at": null
|
||||
}
|
||||
]
|
||||
}
|
||||
354
tests/test_litellm/test_sync_cost_map.py
Normal file
354
tests/test_litellm/test_sync_cost_map.py
Normal file
|
|
@ -0,0 +1,354 @@
|
|||
import importlib.util
|
||||
import json
|
||||
from pathlib import Path
|
||||
from types import ModuleType
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
REPO_ROOT: Final = Path(__file__).resolve().parents[2]
|
||||
SCRIPT_PATH: Final = REPO_ROOT / "scripts" / "sync_cost_map.py"
|
||||
FIXTURES: Final = Path(__file__).parent / "fixtures" / "cost_map_sync"
|
||||
OPENROUTER_RAW: Final = (FIXTURES / "openrouter_models.json").read_bytes()
|
||||
VERCEL_RAW: Final = (FIXTURES / "vercel_models.json").read_bytes()
|
||||
NOW_MS: Final = 1757030400000
|
||||
|
||||
EXISTING_DEEPSEEK: Final = {
|
||||
"input_cost_per_token": 0.00000132,
|
||||
"input_cost_per_token_cache_hit": 4.4e-8,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 300000,
|
||||
"max_tokens": 300000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 0.00000396,
|
||||
"source": "https://openrouter.ai/deepseek/deepseek-v4-pro-0813",
|
||||
"supports_function_calling": True,
|
||||
"supports_prompt_caching": True,
|
||||
"supports_reasoning": True,
|
||||
"supports_response_schema": True,
|
||||
"supports_tool_choice": True,
|
||||
}
|
||||
EXISTING_GLM: Final = {
|
||||
"litellm_provider": "vercel_ai_gateway",
|
||||
"cache_read_input_token_cost": 1.1e-7,
|
||||
"input_cost_per_token": 4.5e-7,
|
||||
"max_input_tokens": 200000,
|
||||
"max_output_tokens": 200000,
|
||||
"max_tokens": 200000,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 0.0000018,
|
||||
"source": "https://vercel.com/ai-gateway/models/glm-4.6",
|
||||
"supports_function_calling": True,
|
||||
"supports_parallel_function_calling": True,
|
||||
"supports_tool_choice": True,
|
||||
}
|
||||
BLOCK_END_OPENROUTER: Final = {
|
||||
"litellm_provider": "openrouter",
|
||||
"mode": "completion",
|
||||
"input_cost_per_token": 0.0000015,
|
||||
"output_cost_per_token": 0.000002,
|
||||
}
|
||||
|
||||
|
||||
def _base_map() -> dict[str, object]:
|
||||
return {
|
||||
"sample_spec": {"litellm_provider": "one of https://docs.litellm.ai/docs/providers"},
|
||||
"gpt-4o": {"litellm_provider": "openai", "mode": "chat"},
|
||||
"openrouter/deepseek/deepseek-v4-pro-0813": dict(EXISTING_DEEPSEEK),
|
||||
"openrouter/openai/gpt-3.5-turbo-instruct": dict(BLOCK_END_OPENROUTER),
|
||||
"vercel_ai_gateway/zai/glm-4.6": dict(EXISTING_GLM),
|
||||
"zzz/last": {"litellm_provider": "zzz", "mode": "chat"},
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def sync() -> ModuleType:
|
||||
spec = importlib.util.spec_from_file_location("sync_cost_map", SCRIPT_PATH)
|
||||
assert spec is not None and spec.loader is not None
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def _run(sync: ModuleType, cost_map: dict[str, object]):
|
||||
return sync.compute_sync(
|
||||
cost_map, (sync.load_openrouter(OPENROUTER_RAW), sync.load_vercel(VERCEL_RAW, now_ms=NOW_MS))
|
||||
)
|
||||
|
||||
|
||||
def test_new_openrouter_entry_carries_catalog_prices_limits_and_capabilities(sync: ModuleType) -> None:
|
||||
outcome = _run(sync, _base_map())
|
||||
|
||||
assert outcome.cost_map["openrouter/inception/mercury-2.5-preview"] == {
|
||||
"cache_read_input_token_cost": 4e-9,
|
||||
"input_cost_per_token": 4e-8,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_input_tokens": 260000,
|
||||
"max_output_tokens": 65536,
|
||||
"max_tokens": 65536,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.5e-7,
|
||||
"source": "https://openrouter.ai/inception/mercury-2.5-preview",
|
||||
"supports_function_calling": True,
|
||||
"supports_reasoning": True,
|
||||
"supports_response_schema": True,
|
||||
"supports_tool_choice": True,
|
||||
}
|
||||
|
||||
|
||||
def test_input_modalities_become_capability_flags(sync: ModuleType) -> None:
|
||||
outcome = _run(sync, _base_map())
|
||||
|
||||
gpt5 = outcome.cost_map["openrouter/openai/gpt-5-mini"]
|
||||
gemma = outcome.cost_map["openrouter/google/gemma-4-26b-a4b-it:free"]
|
||||
vercel_gpt5 = outcome.cost_map["vercel_ai_gateway/openai/gpt-5-mini"]
|
||||
assert (gpt5["supports_vision"], gpt5["supports_pdf_input"]) == (True, True)
|
||||
assert "supports_video_input" not in gpt5 and "supports_audio_input" not in gpt5
|
||||
assert "input_cost_per_image" not in gpt5 and "supports_prompt_caching" not in gpt5
|
||||
assert (gemma["supports_vision"], gemma["supports_video_input"]) == (True, True)
|
||||
assert (vercel_gpt5["supports_vision"], vercel_gpt5["supports_pdf_input"]) == (True, True)
|
||||
assert vercel_gpt5["cache_read_input_token_cost"] == 2.5e-8
|
||||
assert vercel_gpt5["source"] == "https://vercel.com/ai-gateway/models/gpt-5-mini"
|
||||
|
||||
|
||||
def test_free_models_are_added_with_zero_prices(sync: ModuleType) -> None:
|
||||
outcome = _run(sync, _base_map())
|
||||
|
||||
free = outcome.cost_map["openrouter/cohere/north-mini-code:free"]
|
||||
assert (free["input_cost_per_token"], free["output_cost_per_token"]) == (0.0, 0.0)
|
||||
assert (free["max_input_tokens"], free["max_output_tokens"], free["max_tokens"]) == (256000, 64000, 64000)
|
||||
assert "cache_read_input_token_cost" not in free
|
||||
|
||||
|
||||
def test_router_rows_and_unpriced_rows_are_skipped(sync: ModuleType) -> None:
|
||||
outcome = _run(sync, _base_map())
|
||||
|
||||
openrouter, vercel = outcome.providers
|
||||
assert "openrouter/openrouter/auto" not in outcome.cost_map
|
||||
assert "vercel_ai_gateway/perplexity/sonar" not in outcome.cost_map
|
||||
assert dict(openrouter.skipped) == {"unpriced or router": 1}
|
||||
assert dict(vercel.skipped) == {"deprecated": 1, "not token priced": 1, "no usable price": 1}
|
||||
|
||||
|
||||
def test_vercel_rows_map_type_to_mode_and_drop_non_token_types(sync: ModuleType) -> None:
|
||||
outcome = _run(sync, _base_map())
|
||||
|
||||
embedding = outcome.cost_map["vercel_ai_gateway/alibaba/qwen3-embedding-0.6b"]
|
||||
assert embedding == {
|
||||
"input_cost_per_token": 1e-8,
|
||||
"litellm_provider": "vercel_ai_gateway",
|
||||
"max_input_tokens": 32768,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "embedding",
|
||||
"output_cost_per_token": 0.0,
|
||||
"source": "https://vercel.com/ai-gateway/models/qwen3-embedding-0.6b",
|
||||
}
|
||||
assert "vercel_ai_gateway/bfl/flux-2-flex" not in outcome.cost_map
|
||||
assert "vercel_ai_gateway/openai/gpt-4o-mini-transcribe" not in outcome.cost_map
|
||||
|
||||
|
||||
def test_existing_entry_is_repriced_without_losing_curated_fields(sync: ModuleType) -> None:
|
||||
outcome = _run(sync, _base_map())
|
||||
|
||||
deepseek = outcome.cost_map["openrouter/deepseek/deepseek-v4-pro-0813"]
|
||||
assert deepseek["input_cost_per_token"] == 5.7948e-7
|
||||
assert deepseek["output_cost_per_token"] == 1.73844e-6
|
||||
assert deepseek["cache_read_input_token_cost"] == 1.9316e-8
|
||||
assert deepseek["input_cost_per_token_cache_hit"] == 4.4e-8
|
||||
assert (deepseek["max_output_tokens"], deepseek["max_tokens"]) == (300000, 300000)
|
||||
glm = outcome.cost_map["vercel_ai_gateway/zai/glm-4.6"]
|
||||
assert (glm["input_cost_per_token"], glm["output_cost_per_token"]) == (6e-7, 2.2e-6)
|
||||
assert glm["supports_parallel_function_calling"] is True
|
||||
assert glm["supports_reasoning"] is True
|
||||
assert glm["max_output_tokens"] == 200000
|
||||
openrouter, vercel = outcome.providers
|
||||
assert [line.split(":")[0] for line in openrouter.updated] == ["openrouter/deepseek/deepseek-v4-pro-0813"]
|
||||
assert "input_cost_per_token: 1.32e-06 -> 5.7948e-07" in openrouter.updated[0]
|
||||
assert [line.split(":")[0] for line in vercel.updated] == ["vercel_ai_gateway/zai/glm-4.6"]
|
||||
|
||||
|
||||
def test_legacy_max_tokens_is_never_paired_with_a_different_max_output_tokens(sync: ModuleType) -> None:
|
||||
legacy = {
|
||||
"input_cost_per_token": 4e-8,
|
||||
"litellm_provider": "openrouter",
|
||||
"max_tokens": 8192,
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 1.5e-7,
|
||||
}
|
||||
outcome = _run(sync, {**_base_map(), "openrouter/inception/mercury-2.5-preview": legacy})
|
||||
|
||||
mercury = outcome.cost_map["openrouter/inception/mercury-2.5-preview"]
|
||||
assert mercury["max_tokens"] == 8192
|
||||
assert "max_output_tokens" not in mercury
|
||||
assert mercury["max_input_tokens"] == 260000
|
||||
|
||||
|
||||
def test_untouched_entries_survive_byte_for_byte(sync: ModuleType) -> None:
|
||||
outcome = _run(sync, _base_map())
|
||||
|
||||
assert outcome.cost_map["gpt-4o"] == {"litellm_provider": "openai", "mode": "chat"}
|
||||
assert outcome.cost_map["openrouter/openai/gpt-3.5-turbo-instruct"] == BLOCK_END_OPENROUTER
|
||||
assert outcome.cost_map["sample_spec"] == _base_map()["sample_spec"]
|
||||
|
||||
|
||||
def test_second_sync_is_a_no_op(sync: ModuleType) -> None:
|
||||
first = _run(sync, _base_map())
|
||||
|
||||
second = _run(sync, dict(first.cost_map))
|
||||
|
||||
assert second.has_changes is False
|
||||
assert all(not provider.added and not provider.updated for provider in second.providers)
|
||||
assert list(second.cost_map) == list(first.cost_map)
|
||||
|
||||
|
||||
def test_mode_mismatch_warns_and_leaves_the_entry_alone(sync: ModuleType) -> None:
|
||||
cost_map = _base_map()
|
||||
cost_map["vercel_ai_gateway/openai/gpt-5-mini"] = {
|
||||
"litellm_provider": "vercel_ai_gateway",
|
||||
"mode": "responses",
|
||||
"input_cost_per_token": 1.0,
|
||||
}
|
||||
|
||||
outcome = _run(sync, cost_map)
|
||||
|
||||
assert outcome.cost_map["vercel_ai_gateway/openai/gpt-5-mini"]["input_cost_per_token"] == 1.0
|
||||
vercel = outcome.providers[1]
|
||||
assert "vercel_ai_gateway/openai/gpt-5-mini" not in outcome.providers[1].added
|
||||
assert all("gpt-5-mini" not in line for line in vercel.updated)
|
||||
assert len(vercel.warnings) == 1
|
||||
assert "vercel_ai_gateway/openai/gpt-5-mini" in vercel.warnings[0]
|
||||
assert "'responses'" in vercel.warnings[0] and "'chat'" in vercel.warnings[0]
|
||||
|
||||
|
||||
def test_new_keys_land_at_the_end_of_their_provider_block(sync: ModuleType) -> None:
|
||||
outcome = _run(sync, _base_map())
|
||||
|
||||
assert list(outcome.cost_map) == [
|
||||
"sample_spec",
|
||||
"gpt-4o",
|
||||
"openrouter/deepseek/deepseek-v4-pro-0813",
|
||||
"openrouter/openai/gpt-3.5-turbo-instruct",
|
||||
"openrouter/cohere/north-mini-code:free",
|
||||
"openrouter/google/gemma-4-26b-a4b-it:free",
|
||||
"openrouter/inception/mercury-2.5-preview",
|
||||
"openrouter/openai/gpt-5-mini",
|
||||
"vercel_ai_gateway/zai/glm-4.6",
|
||||
"vercel_ai_gateway/alibaba/qwen3-embedding-0.6b",
|
||||
"vercel_ai_gateway/openai/gpt-5-mini",
|
||||
"zzz/last",
|
||||
]
|
||||
|
||||
|
||||
def test_provider_without_a_block_is_appended_at_the_end(sync: ModuleType) -> None:
|
||||
outcome = _run(sync, {"gpt-4o": {"litellm_provider": "openai", "mode": "chat"}})
|
||||
|
||||
keys = list(outcome.cost_map)
|
||||
assert keys[0] == "gpt-4o"
|
||||
assert keys[1:6] == sorted(keys[1:6]) and all(key.startswith("openrouter/") for key in keys[1:6])
|
||||
assert keys[6:] == sorted(keys[6:]) and all(key.startswith("vercel_ai_gateway/") for key in keys[6:])
|
||||
assert len(keys) == 9
|
||||
|
||||
|
||||
def test_pr_body_lists_changes_per_provider(sync: ModuleType) -> None:
|
||||
body = sync.render_pr_body(_run(sync, _base_map()))
|
||||
|
||||
assert "## openrouter" in body and "## vercel_ai_gateway" in body
|
||||
assert "### Added (4)" in body and "- `openrouter/inception/mercury-2.5-preview`" in body
|
||||
assert "### Added (2)" in body and "- `vercel_ai_gateway/openai/gpt-5-mini`" in body
|
||||
assert "- `openrouter/deepseek/deepseek-v4-pro-0813: input_cost_per_token: 1.32e-06 -> 5.7948e-07" in body
|
||||
assert "Catalog rows skipped: deprecated (1), no usable price (1), not token priced (1)" in body
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("loader", "raw"),
|
||||
[
|
||||
("load_openrouter", b'{"data": []}'),
|
||||
("load_openrouter", b'{"data": [{"id": "x", "pricing": {"prompt": 1}}]}'),
|
||||
("load_vercel", b"[]"),
|
||||
("load_vercel", b'{"data": [{"id": "x"}]}'),
|
||||
],
|
||||
)
|
||||
def test_malformed_catalogs_fail_the_run(sync: ModuleType, loader: str, raw: bytes) -> None:
|
||||
kwargs = {"now_ms": NOW_MS} if loader == "load_vercel" else {}
|
||||
with pytest.raises(sync.SyncError):
|
||||
getattr(sync, loader)(raw, **kwargs)
|
||||
|
||||
|
||||
def _vercel_language_row(deprecated_at: int | None) -> bytes:
|
||||
row = {
|
||||
"id": "acme/chat-1",
|
||||
"type": "language",
|
||||
"context_window": 1000,
|
||||
"max_tokens": 100,
|
||||
"pricing": {"input": "0.000001", "output": "0.000002"},
|
||||
"deprecated_at": deprecated_at,
|
||||
}
|
||||
return json.dumps({"data": [row]}).encode()
|
||||
|
||||
|
||||
def test_a_scheduled_deprecation_keeps_syncing_until_the_date(sync: ModuleType) -> None:
|
||||
scheduled = sync.load_vercel(_vercel_language_row(NOW_MS + 1), now_ms=NOW_MS)
|
||||
passed = sync.load_vercel(_vercel_language_row(NOW_MS), now_ms=NOW_MS)
|
||||
|
||||
assert [entry.key for entry in scheduled.entries] == ["vercel_ai_gateway/acme/chat-1"]
|
||||
assert dict(scheduled.skipped)["deprecated"] == 0
|
||||
assert passed.entries == ()
|
||||
assert dict(passed.skipped)["deprecated"] == 1
|
||||
|
||||
|
||||
def _repo(tmp_path: Path) -> Path:
|
||||
for relpath in ("model_prices_and_context_window.json", "litellm/model_prices_and_context_window_backup.json"):
|
||||
target = tmp_path / relpath
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
target.write_text(json.dumps(_base_map(), indent=4) + "\n")
|
||||
return tmp_path
|
||||
|
||||
|
||||
def test_write_updates_both_cost_map_files_identically(sync: ModuleType, tmp_path: Path, capsys) -> None:
|
||||
repo = _repo(tmp_path)
|
||||
body_file = tmp_path / "body.md"
|
||||
|
||||
code = sync.main(
|
||||
[
|
||||
"--write",
|
||||
"--openrouter-json",
|
||||
str(FIXTURES / "openrouter_models.json"),
|
||||
"--vercel-json",
|
||||
str(FIXTURES / "vercel_models.json"),
|
||||
"--pr-body-file",
|
||||
str(body_file),
|
||||
"--repo-root",
|
||||
str(repo),
|
||||
]
|
||||
)
|
||||
|
||||
root = (repo / "model_prices_and_context_window.json").read_text()
|
||||
backup = (repo / "litellm" / "model_prices_and_context_window_backup.json").read_text()
|
||||
assert code == 0
|
||||
assert root == backup
|
||||
assert root.endswith("}\n")
|
||||
assert json.loads(root)["openrouter/inception/mercury-2.5-preview"]["input_cost_per_token"] == 4e-8
|
||||
assert "### Added (4)" in body_file.read_text()
|
||||
assert capsys.readouterr().out.startswith("openrouter: added=4 updated=1 warnings=0")
|
||||
|
||||
|
||||
def test_dry_run_touches_nothing(sync: ModuleType, tmp_path: Path, capsys) -> None:
|
||||
repo = _repo(tmp_path)
|
||||
before = (repo / "model_prices_and_context_window.json").read_bytes()
|
||||
|
||||
code = sync.main(
|
||||
[
|
||||
"--openrouter-json",
|
||||
str(FIXTURES / "openrouter_models.json"),
|
||||
"--vercel-json",
|
||||
str(FIXTURES / "vercel_models.json"),
|
||||
"--repo-root",
|
||||
str(repo),
|
||||
]
|
||||
)
|
||||
|
||||
assert code == 0
|
||||
assert (repo / "model_prices_and_context_window.json").read_bytes() == before
|
||||
assert "dry run: no files were touched" in capsys.readouterr().out
|
||||
Loading…
Add table
Reference in a new issue