mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(sync-cost-map): map vercel tiers, inherit family traits, hold out-of-bounds changes, and reconcile the open bot PR
Vercel long-context tiers become *_above_<N>k_tokens keys when contiguous on a whole thousand, and a row whose tiers do not fit is skipped with a warning. Image and audio output are priced per token, and a row with an unpriced non-text output is skipped instead of billed as text. A new entry inherits the traits no catalog expresses (cache minimum, adaptive thinking, sampling params, system messages, thinking always on) from its same-mode root, found by the bare name or its longest dash prefix. The max_tokens / max_output_tokens pair moves as a unit. Shrinking limits, prices crossing zero or moving more than 10x, and every price on an already-priced varies_by_provider row are held back and listed as warnings for a human commit. Updated entries keep their curated key order with new keys appended sorted. The workflow's own token is read-only and every write uses the GitHub App token; without the App a scheduled run explains why it cannot open a PR. Each tick first reconciles the open bot PR: a conflicting one is closed and re-synced, a green one is merged, a red one is left for a human, and a sync only runs when none is open. The sync step runs with --no-dev and only when it will be used. The hardcoded map schema in test_utils.py gains the 32k tier keys the synced map now carries.
This commit is contained in:
parent
a13daf7c5c
commit
58b3037e4b
4 changed files with 683 additions and 145 deletions
119
.github/workflows/cost-map-sync.yml
vendored
119
.github/workflows/cost-map-sync.yml
vendored
|
|
@ -11,8 +11,8 @@ on:
|
|||
default: false
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
contents: read
|
||||
pull-requests: read
|
||||
|
||||
concurrency:
|
||||
group: cost-map-sync
|
||||
|
|
@ -39,44 +39,72 @@ jobs:
|
|||
with:
|
||||
app-id: ${{ secrets.COST_MAP_BOT_APP_ID }}
|
||||
private-key: ${{ secrets.COST_MAP_BOT_PRIVATE_KEY }}
|
||||
- name: Reconcile the open sync PR
|
||||
id: open
|
||||
run: |
|
||||
pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --json number,headRefName,mergeable \
|
||||
--search "in:title \"$PR_TITLE\"" \
|
||||
--jq "[.[] | select(.headRefName | startswith(\"$BRANCH_PREFIX\"))] | first // empty")"
|
||||
sync=false
|
||||
if [ -z "$pr" ]; then
|
||||
sync=true
|
||||
elif [ -z "$BOT_APP_ID" ]; then
|
||||
echo "::warning::Sync PR #$(jq -r .number <<< "$pr") is open and COST_MAP_BOT_APP_ID is not configured; leaving it to a human."
|
||||
elif [ "$(jq -r .mergeable <<< "$pr")" = "CONFLICTING" ]; then
|
||||
number="$(jq -r .number <<< "$pr")"
|
||||
gh pr close "$number" --repo "$GITHUB_REPOSITORY" --delete-branch \
|
||||
--comment "This sync no longer merges cleanly against $GITHUB_REF_NAME, so the next scheduled run opens a fresh one."
|
||||
echo "Closed conflicting sync PR #$number."
|
||||
sync=true
|
||||
else
|
||||
number="$(jq -r .number <<< "$pr")"
|
||||
guard="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --json name,state \
|
||||
--jq '.[] | select(.name == "cost-map-guard") | .state' || true)"
|
||||
required="$(gh pr checks "$number" --repo "$GITHUB_REPOSITORY" --required --json bucket \
|
||||
--jq 'map(.bucket) | unique | join(",")' || true)"
|
||||
case "$guard,$required" in
|
||||
SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping)
|
||||
gh pr merge "$number" --repo "$GITHUB_REPOSITORY" --merge --delete-branch
|
||||
echo "Merged sync PR #$number; the next scheduled run syncs from the merged registry."
|
||||
;;
|
||||
*FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*)
|
||||
echo "::warning::Sync PR #$number has a failing check (cost-map-guard=$guard, required buckets=$required); leaving it open for a human."
|
||||
;;
|
||||
*)
|
||||
echo "Sync PR #$number is still being checked (cost-map-guard=$guard, required buckets=$required)."
|
||||
;;
|
||||
esac
|
||||
fi
|
||||
echo "sync=$sync" >> "$GITHUB_OUTPUT"
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.bot.outputs.token || github.token }}
|
||||
- name: Explain why no PR can be opened
|
||||
if: steps.open.outputs.sync == 'true' && env.BOT_APP_ID == '' && !inputs.dry_run
|
||||
run: echo "::warning::COST_MAP_BOT_APP_ID is not configured, so no sync PR can be opened or merged; dispatch with dry_run to see the diff."
|
||||
- name: Set up uv
|
||||
if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run)
|
||||
uses: ./.github/actions/setup-uv-with-retries
|
||||
with:
|
||||
version: "0.10.9"
|
||||
- name: Look for an already-open sync PR
|
||||
id: existing
|
||||
run: |
|
||||
open_pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --limit 100 --json headRefName \
|
||||
--search "in:title \"$PR_TITLE\"" \
|
||||
--jq "[.[].headRefName | select(startswith(\"$BRANCH_PREFIX\"))] | first // empty")"
|
||||
echo "open_pr=$open_pr" >> "$GITHUB_OUTPUT"
|
||||
if [ -n "$open_pr" ]; then
|
||||
echo "An open sync PR already exists on branch $open_pr; skipping this run."
|
||||
fi
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.bot.outputs.token || secrets.GH_TOKEN || github.token }}
|
||||
- name: Run the sync
|
||||
if: steps.existing.outputs.open_pr == ''
|
||||
run: |
|
||||
uv run --frozen python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md"
|
||||
uv run --frozen python ci_cd/generate_model_prices_schema.py
|
||||
- name: Open the sync PR
|
||||
id: pr
|
||||
if: steps.existing.outputs.open_pr == '' && !inputs.dry_run
|
||||
id: sync
|
||||
if: steps.open.outputs.sync == 'true' && (env.BOT_APP_ID != '' || inputs.dry_run)
|
||||
run: |
|
||||
uv run --frozen --no-dev python scripts/sync_cost_map.py --write --pr-body-file "$RUNNER_TEMP/pr_body.md"
|
||||
uv run --frozen --no-dev python ci_cd/generate_model_prices_schema.py
|
||||
if git diff --quiet; then
|
||||
echo "Registry already in sync; no PR needed."
|
||||
exit 0
|
||||
fi
|
||||
branch="${BRANCH_PREFIX}$(date -u +'%Y-%m-%d-%H%M')"
|
||||
if [ -n "$BOT_APP_ID" ]; then
|
||||
bot_user_id="$(gh api "users/${BOT_LOGIN}[bot]" --jq .id)"
|
||||
git config user.name "${BOT_LOGIN}[bot]"
|
||||
git config user.email "${bot_user_id}+${BOT_LOGIN}[bot]@users.noreply.github.com"
|
||||
echo "changed=false" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
||||
echo "changed=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
- name: Open the sync PR
|
||||
if: steps.sync.outputs.changed == 'true' && env.BOT_APP_ID != '' && !inputs.dry_run
|
||||
run: |
|
||||
branch="${BRANCH_PREFIX}$(date -u +'%Y-%m-%d-%H%M')"
|
||||
bot_user_id="$(gh api "users/${BOT_LOGIN}[bot]" --jq .id)"
|
||||
git config user.name "${BOT_LOGIN}[bot]"
|
||||
git config user.email "${bot_user_id}+${BOT_LOGIN}[bot]@users.noreply.github.com"
|
||||
git checkout -b "$branch"
|
||||
git add model_prices_and_context_window.json \
|
||||
litellm/model_prices_and_context_window_backup.json \
|
||||
|
|
@ -84,35 +112,10 @@ jobs:
|
|||
git commit -m "feat(models): sync openrouter and vercel_ai_gateway pricing $(date -u +'%Y-%m-%d %H:%M')"
|
||||
gh auth setup-git
|
||||
git push origin "$branch"
|
||||
url="$(gh pr create --title "$PR_TITLE" \
|
||||
gh pr create --title "$PR_TITLE" \
|
||||
--body-file "$RUNNER_TEMP/pr_body.md" \
|
||||
--head "$branch" \
|
||||
--base "$GITHUB_REF_NAME")"
|
||||
echo "url=$url" >> "$GITHUB_OUTPUT"
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.bot.outputs.token || secrets.GH_TOKEN || github.token }}
|
||||
BOT_LOGIN: ${{ steps.bot.outputs.app-slug }}
|
||||
- name: Merge once every required check passes
|
||||
if: steps.pr.outputs.url != '' && env.BOT_APP_ID != ''
|
||||
timeout-minutes: 120
|
||||
run: |
|
||||
while true; do
|
||||
guard="$(gh pr checks "$PR_URL" --json name,state \
|
||||
--jq '.[] | select(.name == "cost-map-guard") | .state' || true)"
|
||||
required="$(gh pr checks "$PR_URL" --required --json bucket \
|
||||
--jq 'map(.bucket) | unique | join(",")' || true)"
|
||||
case "$guard,$required" in
|
||||
*FAILURE*|*CANCELLED*|*TIMED_OUT*|*ACTION_REQUIRED*|*fail*|*cancel*)
|
||||
echo "A check failed (cost-map-guard=$guard, required buckets=$required); leaving $PR_URL open for a human."
|
||||
exit 1
|
||||
;;
|
||||
SUCCESS,pass|SUCCESS,pass,skipping|SUCCESS,skipping)
|
||||
gh pr merge "$PR_URL" --repo "$GITHUB_REPOSITORY" --merge --delete-branch
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
sleep 30
|
||||
done
|
||||
--base "$GITHUB_REF_NAME"
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.bot.outputs.token }}
|
||||
PR_URL: ${{ steps.pr.outputs.url }}
|
||||
BOT_LOGIN: ${{ steps.bot.outputs.app-slug }}
|
||||
|
|
|
|||
|
|
@ -7,8 +7,20 @@ backup copy.
|
|||
|
||||
Policy:
|
||||
- Both catalogs price per token as decimal strings; values are normalized to six significant digits.
|
||||
- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed, a
|
||||
capability flag the catalog does not claim stays as curated, and a curated output ceiling is kept.
|
||||
- Vercel long-context tiers map to the registry's ``*_above_<N>k_tokens`` keys, which litellm applies once the
|
||||
prompt exceeds N thousand tokens. A row whose tier boundaries are not whole thousands is skipped with a warning.
|
||||
- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry but never overwrites a
|
||||
curated price, and a difference is reported as a warning.
|
||||
- Image and audio output are priced from the catalog's per-token ``image_output`` and ``audio_output`` prices. A row
|
||||
whose non-text output the catalog does not price per token is skipped.
|
||||
- A new entry inherits the traits no catalog expresses (adaptive thinking, sampling params, cache minimums, system
|
||||
messages) from the same model's root registry entry, found by the bare model name or its longest dash-prefix
|
||||
with the same mode, so the family-wide invariants the test suite enforces hold for the route too.
|
||||
- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed and a
|
||||
capability flag the catalog does not claim stays as curated. ``max_output_tokens`` and ``max_tokens`` move as a
|
||||
pair and only when the catalog states an output ceiling.
|
||||
- A limit that would shrink, a price that would cross zero, and a price that would move more than 10x either way
|
||||
are held back as warnings for a human instead of applied.
|
||||
- Router models and rows without a usable prompt and completion price are skipped.
|
||||
- A registry entry absent from its catalog is left untouched; retiring a model stays a human call.
|
||||
"""
|
||||
|
|
@ -18,6 +30,7 @@ import json
|
|||
import math
|
||||
import sys
|
||||
import time
|
||||
from collections import Counter
|
||||
from collections.abc import Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
from functools import reduce
|
||||
|
|
@ -35,12 +48,26 @@ COST_MAP_RELPATHS: Final = (
|
|||
OPENROUTER_MODELS_URL: Final = "https://openrouter.ai/api/v1/models"
|
||||
VERCEL_MODELS_URL: Final = "https://ai-gateway.vercel.sh/v1/models"
|
||||
VERCEL_TYPE_TO_MODE: Final = MappingProxyType({"language": "chat", "embedding": "embedding"})
|
||||
ADD_ONLY_FIELDS: Final = frozenset({"max_output_tokens", "max_tokens"})
|
||||
LIMIT_PAIR: Final = ("max_output_tokens", "max_tokens")
|
||||
PRICE_SWING_LIMIT: Final = 10
|
||||
INHERITED_TRAITS: Final = frozenset(
|
||||
{
|
||||
"prompt_cache_min_tokens",
|
||||
"supports_adaptive_thinking",
|
||||
"supports_sampling_params",
|
||||
"supports_system_messages",
|
||||
"thinking_always_on",
|
||||
}
|
||||
)
|
||||
PR_BODY_SECTION_LIMIT: Final = 30
|
||||
|
||||
Provider = Literal["openrouter", "vercel_ai_gateway"]
|
||||
RegistryEntry = dict[str, object]
|
||||
CostMap = dict[str, object]
|
||||
Prices = Mapping[str, float]
|
||||
|
||||
NO_PRICES: Final[Prices] = MappingProxyType({})
|
||||
NO_TRAITS: Final[Mapping[str, object]] = MappingProxyType({})
|
||||
|
||||
|
||||
class SyncError(RuntimeError):
|
||||
|
|
@ -53,10 +80,14 @@ class OpenRouterPricing(BaseModel):
|
|||
input_cache_read: str | None = None
|
||||
input_cache_write: str | None = None
|
||||
internal_reasoning: str | None = None
|
||||
image_output: str | None = None
|
||||
audio: str | None = None
|
||||
audio_output: str | None = None
|
||||
|
||||
|
||||
class OpenRouterArchitecture(BaseModel):
|
||||
input_modalities: tuple[str, ...] | None = None
|
||||
output_modalities: tuple[str, ...] | None = None
|
||||
|
||||
|
||||
class OpenRouterTopProvider(BaseModel):
|
||||
|
|
@ -72,15 +103,29 @@ class OpenRouterModel(BaseModel):
|
|||
supported_parameters: tuple[str, ...] | None = None
|
||||
|
||||
|
||||
class VercelTier(BaseModel):
|
||||
cost: str
|
||||
min: int | None = None
|
||||
max: int | None = None
|
||||
|
||||
|
||||
class VercelPricing(BaseModel):
|
||||
input: str | None = None
|
||||
output: str | None = None
|
||||
input_cache_read: str | None = None
|
||||
input_cache_write: str | None = None
|
||||
input_tiers: tuple[VercelTier, ...] | None = None
|
||||
output_tiers: tuple[VercelTier, ...] | None = None
|
||||
input_cache_read_tiers: tuple[VercelTier, ...] | None = None
|
||||
input_cache_write_tiers: tuple[VercelTier, ...] | None = None
|
||||
audio_input_token_cost: str | None = None
|
||||
audio_output_token_cost: str | None = None
|
||||
varies_by_provider: bool = False
|
||||
|
||||
|
||||
class VercelModalities(BaseModel):
|
||||
input: tuple[str, ...] | None = None
|
||||
output: tuple[str, ...] | None = None
|
||||
|
||||
|
||||
class VercelModel(BaseModel):
|
||||
|
|
@ -105,6 +150,18 @@ class CatalogEntry:
|
|||
mode: str
|
||||
source: str
|
||||
fields: Mapping[str, object]
|
||||
indicative_prices: bool = False
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Skipped:
|
||||
reason: str
|
||||
warning: str | None = None
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Unmappable:
|
||||
problem: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
|
|
@ -112,6 +169,7 @@ class Catalog:
|
|||
provider: Provider
|
||||
entries: tuple[CatalogEntry, ...]
|
||||
skipped: Mapping[str, int]
|
||||
warnings: tuple[str, ...]
|
||||
|
||||
|
||||
def per_token(price: float) -> float:
|
||||
|
|
@ -130,18 +188,22 @@ def _extra_price(raw: str | None) -> float | None:
|
|||
return price if price else None
|
||||
|
||||
|
||||
def _flags(parameters: Sequence[str] | None, modalities: Sequence[str] | None) -> Mapping[str, bool]:
|
||||
def _flags(
|
||||
parameters: Sequence[str] | None, inputs: Sequence[str] | None, outputs: Sequence[str] | None
|
||||
) -> Mapping[str, bool]:
|
||||
params: Final = frozenset(parameters or ())
|
||||
mods: Final = frozenset(modalities or ())
|
||||
input_modalities: Final = frozenset(inputs or ())
|
||||
output_modalities: Final = frozenset(outputs or ())
|
||||
claims: Final = {
|
||||
"supports_function_calling": "tools" in params,
|
||||
"supports_tool_choice": "tool_choice" in params,
|
||||
"supports_reasoning": "reasoning" in params,
|
||||
"supports_response_schema": "structured_outputs" in params,
|
||||
"supports_vision": "image" in mods,
|
||||
"supports_pdf_input": bool({"file", "pdf"} & mods),
|
||||
"supports_audio_input": "audio" in mods,
|
||||
"supports_video_input": "video" in mods,
|
||||
"supports_vision": "image" in input_modalities,
|
||||
"supports_pdf_input": bool({"file", "pdf"} & input_modalities),
|
||||
"supports_audio_input": "audio" in input_modalities,
|
||||
"supports_video_input": "video" in input_modalities,
|
||||
"supports_audio_output": "audio" in output_modalities,
|
||||
}
|
||||
return MappingProxyType({name: True for name, claimed in claims.items() if claimed})
|
||||
|
||||
|
|
@ -157,23 +219,85 @@ def _limits(max_input: int | None, max_output: int | None) -> Mapping[str, int]:
|
|||
)
|
||||
|
||||
|
||||
def _priced(name: str, price: float | None) -> Mapping[str, float]:
|
||||
def _priced(name: str, price: float | None) -> Prices:
|
||||
return MappingProxyType({name: price} if price is not None else {})
|
||||
|
||||
|
||||
def _openrouter_entry(model: OpenRouterModel) -> CatalogEntry | None:
|
||||
prompt: Final = _token_price(model.pricing.prompt)
|
||||
completion: Final = _token_price(model.pricing.completion)
|
||||
def _output_prices(
|
||||
outputs: Sequence[str] | None, image_price: float | None, audio_price: float | None
|
||||
) -> Prices | Skipped:
|
||||
modalities: Final = frozenset(outputs or ("text",))
|
||||
known: Final = {
|
||||
modality: price for modality, price in (("image", image_price), ("audio", audio_price)) if price is not None
|
||||
}
|
||||
if "text" not in modalities or not (modalities - {"text"}) <= known.keys():
|
||||
return Skipped("output priced outside the catalog")
|
||||
names: Final = {"image": "output_cost_per_image_token", "audio": "output_cost_per_audio_token"}
|
||||
return MappingProxyType({names[modality]: known[modality] for modality in modalities & known.keys()})
|
||||
|
||||
|
||||
def _tier_threshold(boundary: int) -> int | None:
|
||||
return next((start // 1000 for start in (boundary, boundary - 1) if start > 0 and start % 1000 == 0), None)
|
||||
|
||||
|
||||
def _tiered(name: str, base: float | None, tiers: Sequence[VercelTier] | None) -> Prices | Unmappable:
|
||||
if base is None or not tiers:
|
||||
return NO_PRICES
|
||||
ordered: Final = sorted(tiers, key=lambda tier: tier.min or 0)
|
||||
contiguous: Final = ordered[-1].max is None and all(
|
||||
lower.max == upper.min for lower, upper in zip(ordered, ordered[1:], strict=False)
|
||||
)
|
||||
if not contiguous:
|
||||
return Unmappable(f"{name} tiers are not contiguous")
|
||||
steps: Final = tuple((_tier_threshold(tier.min), _token_price(tier.cost)) for tier in ordered if tier.min)
|
||||
prices: Final = {
|
||||
f"{name}_above_{thousands}k_tokens": price
|
||||
for thousands, price in steps
|
||||
if thousands is not None and price is not None
|
||||
}
|
||||
if len(prices) != len(steps):
|
||||
return Unmappable(f"{name} tiers have a boundary that is not a whole thousand or an unusable price")
|
||||
return MappingProxyType(prices)
|
||||
|
||||
|
||||
def _vercel_tiers(pricing: VercelPricing, cache_read: float | None, cache_write: float | None) -> Prices | Unmappable:
|
||||
parts: Final = (
|
||||
_tiered("input_cost_per_token", _token_price(pricing.input), pricing.input_tiers),
|
||||
_tiered("output_cost_per_token", _token_price(pricing.output), pricing.output_tiers),
|
||||
_tiered("cache_read_input_token_cost", cache_read, pricing.input_cache_read_tiers),
|
||||
_tiered("cache_creation_input_token_cost", cache_write, pricing.input_cache_write_tiers),
|
||||
)
|
||||
problem: Final = next((part for part in parts if isinstance(part, Unmappable)), None)
|
||||
if problem is not None:
|
||||
return problem
|
||||
return MappingProxyType(
|
||||
{name: price for part in parts if not isinstance(part, Unmappable) for name, price in part.items()}
|
||||
)
|
||||
|
||||
|
||||
def _openrouter_entry(model: OpenRouterModel) -> CatalogEntry | Skipped:
|
||||
pricing: Final = model.pricing
|
||||
prompt: Final = _token_price(pricing.prompt)
|
||||
completion: Final = _token_price(pricing.completion)
|
||||
if prompt is None or completion is None:
|
||||
return None
|
||||
return Skipped("unpriced or router")
|
||||
inputs: Final = model.architecture.input_modalities if model.architecture else None
|
||||
outputs: Final = model.architecture.output_modalities if model.architecture else None
|
||||
output_prices: Final = _output_prices(
|
||||
outputs, _extra_price(pricing.image_output), _extra_price(pricing.audio_output)
|
||||
)
|
||||
if isinstance(output_prices, Skipped):
|
||||
return output_prices
|
||||
fields: Final = {
|
||||
"input_cost_per_token": prompt,
|
||||
"output_cost_per_token": completion,
|
||||
**_limits(model.context_length, model.top_provider.max_completion_tokens),
|
||||
**_priced("cache_read_input_token_cost", _extra_price(model.pricing.input_cache_read)),
|
||||
**_priced("cache_creation_input_token_cost", _extra_price(model.pricing.input_cache_write)),
|
||||
**_priced("output_cost_per_reasoning_token", _extra_price(model.pricing.internal_reasoning)),
|
||||
**_flags(model.supported_parameters, model.architecture.input_modalities if model.architecture else None),
|
||||
**_priced("cache_read_input_token_cost", _extra_price(pricing.input_cache_read)),
|
||||
**_priced("cache_creation_input_token_cost", _extra_price(pricing.input_cache_write)),
|
||||
**_priced("output_cost_per_reasoning_token", _extra_price(pricing.internal_reasoning)),
|
||||
**_priced("input_cost_per_audio_token", _extra_price(pricing.audio)),
|
||||
**output_prices,
|
||||
**_flags(model.supported_parameters, inputs, outputs),
|
||||
}
|
||||
return CatalogEntry(
|
||||
key=f"openrouter/{model.id}",
|
||||
|
|
@ -184,30 +308,46 @@ def _openrouter_entry(model: OpenRouterModel) -> CatalogEntry | None:
|
|||
)
|
||||
|
||||
|
||||
def _vercel_entry(model: VercelModel) -> CatalogEntry | None:
|
||||
def _vercel_entry(model: VercelModel, now_ms: int) -> CatalogEntry | Skipped:
|
||||
if model.deprecated_at is not None and model.deprecated_at <= now_ms:
|
||||
return Skipped("deprecated")
|
||||
mode: Final = VERCEL_TYPE_TO_MODE.get(model.type)
|
||||
prompt: Final = _token_price(model.pricing.input)
|
||||
completion: Final = _token_price(model.pricing.output if mode != "embedding" else model.pricing.output or "0")
|
||||
if mode is None or prompt is None or completion is None:
|
||||
return None
|
||||
if mode is None:
|
||||
return Skipped("not token priced")
|
||||
pricing: Final = model.pricing
|
||||
prompt: Final = _token_price(pricing.input)
|
||||
completion: Final = _token_price(pricing.output if mode != "embedding" else pricing.output or "0")
|
||||
if prompt is None or completion is None:
|
||||
return Skipped("no usable price")
|
||||
key: Final = f"vercel_ai_gateway/{model.id}"
|
||||
inputs: Final = model.modalities.input if model.modalities else None
|
||||
outputs: Final = model.modalities.output if model.modalities else None
|
||||
output_prices: Final = _output_prices(outputs, None, _extra_price(pricing.audio_output_token_cost))
|
||||
if isinstance(output_prices, Skipped):
|
||||
return output_prices
|
||||
cache_read: Final = _extra_price(pricing.input_cache_read)
|
||||
cache_write: Final = _extra_price(pricing.input_cache_write)
|
||||
tiers: Final = _vercel_tiers(pricing, cache_read, cache_write)
|
||||
if isinstance(tiers, Unmappable):
|
||||
return Skipped("tiers outside the registry's thresholds", warning=f"{key}: {tiers.problem}; row skipped")
|
||||
fields: Final = {
|
||||
"input_cost_per_token": prompt,
|
||||
"output_cost_per_token": completion,
|
||||
**_limits(model.context_window, model.max_tokens),
|
||||
**_priced("cache_read_input_token_cost", _extra_price(model.pricing.input_cache_read)),
|
||||
**_priced("cache_creation_input_token_cost", _extra_price(model.pricing.input_cache_write)),
|
||||
**(
|
||||
_flags(model.supported_parameters, model.modalities.input if model.modalities else None)
|
||||
if mode == "chat"
|
||||
else {}
|
||||
),
|
||||
**_priced("cache_read_input_token_cost", cache_read),
|
||||
**_priced("cache_creation_input_token_cost", cache_write),
|
||||
**_priced("input_cost_per_audio_token", _extra_price(pricing.audio_input_token_cost)),
|
||||
**output_prices,
|
||||
**tiers,
|
||||
**(_flags(model.supported_parameters, inputs, outputs) if mode == "chat" else {}),
|
||||
}
|
||||
return CatalogEntry(
|
||||
key=f"vercel_ai_gateway/{model.id}",
|
||||
key=key,
|
||||
provider="vercel_ai_gateway",
|
||||
mode=mode,
|
||||
source=f"https://vercel.com/ai-gateway/models/{model.id.rsplit('/', 1)[-1]}",
|
||||
fields=MappingProxyType(fields),
|
||||
indicative_prices=pricing.varies_by_provider,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -219,17 +359,21 @@ def _rows(raw: bytes, url: str) -> object:
|
|||
return rows
|
||||
|
||||
|
||||
def _catalog(provider: Provider, rows: Sequence[CatalogEntry | Skipped]) -> Catalog:
|
||||
return Catalog(
|
||||
provider=provider,
|
||||
entries=tuple(row for row in rows if isinstance(row, CatalogEntry)),
|
||||
skipped=MappingProxyType(Counter(row.reason for row in rows if isinstance(row, Skipped))),
|
||||
warnings=tuple(row.warning for row in rows if isinstance(row, Skipped) and row.warning is not None),
|
||||
)
|
||||
|
||||
|
||||
def load_openrouter(raw: bytes) -> Catalog:
|
||||
try:
|
||||
models: Final = OPENROUTER_ADAPTER.validate_python(_rows(raw, OPENROUTER_MODELS_URL))
|
||||
except ValidationError as error:
|
||||
raise SyncError(f"the OpenRouter catalog no longer matches the expected shape: {error}") from error
|
||||
entries: Final = tuple(entry for entry in map(_openrouter_entry, models) if entry is not None)
|
||||
return Catalog(
|
||||
provider="openrouter",
|
||||
entries=entries,
|
||||
skipped=MappingProxyType({"unpriced or router": len(models) - len(entries)}),
|
||||
)
|
||||
return _catalog("openrouter", tuple(map(_openrouter_entry, models)))
|
||||
|
||||
|
||||
def load_vercel(raw: bytes, now_ms: int) -> Catalog:
|
||||
|
|
@ -237,20 +381,7 @@ def load_vercel(raw: bytes, now_ms: int) -> Catalog:
|
|||
models: Final = VERCEL_ADAPTER.validate_python(_rows(raw, VERCEL_MODELS_URL))
|
||||
except ValidationError as error:
|
||||
raise SyncError(f"the Vercel AI Gateway catalog no longer matches the expected shape: {error}") from error
|
||||
live: Final = tuple(model for model in models if model.deprecated_at is None or model.deprecated_at > now_ms)
|
||||
token_priced: Final = tuple(model for model in live if model.type in VERCEL_TYPE_TO_MODE)
|
||||
entries: Final = tuple(entry for entry in map(_vercel_entry, token_priced) if entry is not None)
|
||||
return Catalog(
|
||||
provider="vercel_ai_gateway",
|
||||
entries=entries,
|
||||
skipped=MappingProxyType(
|
||||
{
|
||||
"deprecated": len(models) - len(live),
|
||||
"not token priced": len(live) - len(token_priced),
|
||||
"no usable price": len(token_priced) - len(entries),
|
||||
}
|
||||
),
|
||||
)
|
||||
return _catalog("vercel_ai_gateway", tuple(_vercel_entry(model, now_ms) for model in models))
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
|
|
@ -272,10 +403,34 @@ class SyncOutcome:
|
|||
return any(outcome.added or outcome.updated for outcome in self.providers)
|
||||
|
||||
|
||||
def _new_entry(entry: CatalogEntry) -> RegistryEntry:
|
||||
def _root_candidates(bare: str) -> tuple[str, ...]:
|
||||
segments: Final = bare.split("-")
|
||||
stems: Final = tuple(
|
||||
"-".join(segments[:count]) for count in range(len(segments), 0, -1) if count >= 2 or count == len(segments)
|
||||
)
|
||||
return tuple(dict.fromkeys(name for stem in stems for name in (stem, stem.replace(".", "-"))))
|
||||
|
||||
|
||||
def _inherited(cost_map: CostMap, entry: CatalogEntry) -> Mapping[str, object]:
|
||||
bare: Final = entry.key.rsplit("/", 1)[-1].split(":", 1)[0]
|
||||
root: Final = next(
|
||||
(
|
||||
candidate
|
||||
for candidate in map(cost_map.get, _root_candidates(bare))
|
||||
if isinstance(candidate, dict) and candidate.get("mode") == entry.mode
|
||||
),
|
||||
None,
|
||||
)
|
||||
if root is None:
|
||||
return NO_TRAITS
|
||||
return MappingProxyType({name: value for name, value in root.items() if name in INHERITED_TRAITS})
|
||||
|
||||
|
||||
def _new_entry(entry: CatalogEntry, inherited: Mapping[str, object]) -> RegistryEntry:
|
||||
return dict(
|
||||
sorted(
|
||||
{
|
||||
**inherited,
|
||||
**entry.fields,
|
||||
"litellm_provider": entry.provider,
|
||||
"mode": entry.mode,
|
||||
|
|
@ -285,15 +440,67 @@ def _new_entry(entry: CatalogEntry) -> RegistryEntry:
|
|||
)
|
||||
|
||||
|
||||
def _updated_entry(existing: RegistryEntry, entry: CatalogEntry) -> tuple[RegistryEntry, tuple[str, ...]]:
|
||||
keep_limits: Final = not ADD_ONLY_FIELDS.isdisjoint(existing)
|
||||
desired: Final = {
|
||||
name: value for name, value in entry.fields.items() if not (keep_limits and name in ADD_ONLY_FIELDS)
|
||||
}
|
||||
changes: Final = tuple(
|
||||
f"{name}: {existing.get(name)!r} -> {value!r}" for name, value in desired.items() if existing.get(name) != value
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class FieldChange:
|
||||
name: str
|
||||
old: object
|
||||
new: object
|
||||
hold: str | None
|
||||
|
||||
@property
|
||||
def line(self) -> str:
|
||||
held: Final = f" held back: {self.hold}" if self.hold else ""
|
||||
return f"{self.name}: {self.old!r} -> {self.new!r}{held}"
|
||||
|
||||
|
||||
def _swing(old: float, new: float) -> str | None:
|
||||
if (old == 0) != (new == 0):
|
||||
return "a price crossing zero"
|
||||
if old and new and max(new / old, old / new) > PRICE_SWING_LIMIT:
|
||||
return f"a price moving more than {PRICE_SWING_LIMIT}x"
|
||||
return None
|
||||
|
||||
|
||||
def _hold(name: str, old: object, new: object, curated_prices_win: bool) -> str | None:
|
||||
if "cost" in name and curated_prices_win:
|
||||
return "the catalog price varies by provider"
|
||||
if old is None:
|
||||
return None
|
||||
if name.startswith("max_") and isinstance(old, int) and isinstance(new, int) and new < old:
|
||||
return "a shrinking limit"
|
||||
if "cost" in name and isinstance(old, int | float) and isinstance(new, int | float):
|
||||
return _swing(old, new)
|
||||
return None
|
||||
|
||||
|
||||
def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange, ...]:
|
||||
curated_prices_win: Final = entry.indicative_prices and "input_cost_per_token" in existing
|
||||
scalars: Final = tuple(
|
||||
FieldChange(name, existing.get(name), value, _hold(name, existing.get(name), value, curated_prices_win))
|
||||
for name, value in entry.fields.items()
|
||||
if name not in LIMIT_PAIR and existing.get(name) != value
|
||||
)
|
||||
ceiling: Final = entry.fields.get("max_output_tokens")
|
||||
if ceiling is None:
|
||||
return scalars
|
||||
current: Final = existing.get("max_output_tokens", existing.get("max_tokens"))
|
||||
hold: Final = _hold("max_output_tokens", current, ceiling, curated_prices_win)
|
||||
return (
|
||||
*scalars,
|
||||
*(FieldChange(name, existing.get(name), ceiling, hold) for name in LIMIT_PAIR if existing.get(name) != ceiling),
|
||||
)
|
||||
|
||||
|
||||
def _updated_entry(
|
||||
existing: RegistryEntry, entry: CatalogEntry
|
||||
) -> tuple[RegistryEntry, tuple[str, ...], tuple[str, ...]]:
|
||||
changes: Final = _changes(existing, entry)
|
||||
applied: Final = {change.name: change.new for change in changes if change.hold is None}
|
||||
return (
|
||||
{**existing, **dict(sorted(applied.items()))},
|
||||
tuple(change.line for change in changes if change.hold is None),
|
||||
tuple(change.line for change in changes if change.hold is not None),
|
||||
)
|
||||
return dict(sorted({**existing, **desired}.items())), changes
|
||||
|
||||
|
||||
def _with_new_keys_in_block(ordered: CostMap, result: CostMap, new_keys: Sequence[str], prefix: str) -> CostMap:
|
||||
|
|
@ -331,26 +538,25 @@ class Warned:
|
|||
line: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Unchanged:
|
||||
pass
|
||||
EntrySync = Added | Updated | Warned
|
||||
|
||||
|
||||
EntrySync = Added | Updated | Warned | Unchanged
|
||||
|
||||
|
||||
def _sync_entry(existing: object, entry: CatalogEntry) -> EntrySync:
|
||||
def _sync_entry(cost_map: CostMap, entry: CatalogEntry) -> tuple[EntrySync, ...]:
|
||||
existing: Final = cost_map.get(entry.key)
|
||||
if not isinstance(existing, dict):
|
||||
return Added(key=entry.key, entry=_new_entry(entry))
|
||||
return (Added(key=entry.key, entry=_new_entry(entry, _inherited(cost_map, entry))),)
|
||||
if existing.get("mode") != entry.mode:
|
||||
return Warned(
|
||||
line=f"`{entry.key}` has curated mode {existing.get('mode')!r} but the catalog maps to "
|
||||
f"{entry.mode!r}; left unchanged"
|
||||
return (
|
||||
Warned(
|
||||
line=f"`{entry.key}` has curated mode {existing.get('mode')!r} but the catalog maps to "
|
||||
f"{entry.mode!r}; left unchanged"
|
||||
),
|
||||
)
|
||||
new_entry, changes = _updated_entry(existing, entry)
|
||||
if not changes:
|
||||
return Unchanged()
|
||||
return Updated(key=entry.key, entry=new_entry, line=f"{entry.key}: " + "; ".join(changes))
|
||||
new_entry, applied, held = _updated_entry(existing, entry)
|
||||
return (
|
||||
*((Updated(key=entry.key, entry=new_entry, line=f"{entry.key}: " + "; ".join(applied)),) if applied else ()),
|
||||
*((Warned(line=f"{entry.key}: " + "; ".join(held)),) if held else ()),
|
||||
)
|
||||
|
||||
|
||||
SyncState = tuple[CostMap, tuple[ProviderOutcome, ...]]
|
||||
|
|
@ -359,13 +565,13 @@ SyncState = tuple[CostMap, tuple[ProviderOutcome, ...]]
|
|||
def _sync_provider(state: SyncState, catalog: Catalog) -> SyncState:
|
||||
cost_map, outcomes = state
|
||||
syncs: Final = tuple(
|
||||
_sync_entry(cost_map.get(entry.key), entry) for entry in sorted(catalog.entries, key=lambda item: item.key)
|
||||
sync for entry in sorted(catalog.entries, key=lambda item: item.key) for sync in _sync_entry(cost_map, entry)
|
||||
)
|
||||
outcome: Final = ProviderOutcome(
|
||||
provider=catalog.provider,
|
||||
added=tuple(sync.key for sync in syncs if isinstance(sync, Added)),
|
||||
updated=tuple(sync.line for sync in syncs if isinstance(sync, Updated)),
|
||||
warnings=tuple(sync.line for sync in syncs if isinstance(sync, Warned)),
|
||||
warnings=(*catalog.warnings, *(sync.line for sync in syncs if isinstance(sync, Warned))),
|
||||
skipped=catalog.skipped,
|
||||
)
|
||||
merged: Final = {**cost_map, **{sync.key: sync.entry for sync in syncs if isinstance(sync, Added | Updated)}}
|
||||
|
|
@ -412,7 +618,9 @@ def render_pr_body(outcome: SyncOutcome, section_limit: int | None = PR_BODY_SEC
|
|||
return (
|
||||
"Automated sync of the openrouter and vercel_ai_gateway entries in model_prices_and_context_window.json "
|
||||
f"against `GET {OPENROUTER_MODELS_URL}` and `GET {VERCEL_MODELS_URL}` by scripts/sync_cost_map.py. "
|
||||
"The cost-map-guard check enforces that this PR only adds or reprices models.\n"
|
||||
"The cost-map-guard check enforces that this PR only adds or reprices models. Changes the script held "
|
||||
"back (shrinking limits, prices crossing zero or moving more than 10x, per-provider prices) are listed "
|
||||
"under the warnings and need a human commit.\n"
|
||||
"\n" + "\n".join(_provider_body(provider, section_limit) for provider in outcome.providers)
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -71,6 +71,17 @@ def sync() -> ModuleType:
|
|||
return module
|
||||
|
||||
|
||||
def _openrouter_rows(*rows: dict[str, object]) -> bytes:
|
||||
return json.dumps(
|
||||
{"data": [{"pricing": {"prompt": "0.000001", "completion": "0.000002"}, **row} for row in rows]}
|
||||
).encode()
|
||||
|
||||
|
||||
def _vercel_rows(*rows: dict[str, object]) -> bytes:
|
||||
defaults: Final = {"type": "language", "pricing": {"input": "0.000001", "output": "0.000002"}}
|
||||
return json.dumps({"data": [{**defaults, **row} for row in rows]}).encode()
|
||||
|
||||
|
||||
def _run(sync: ModuleType, cost_map: dict[str, object]):
|
||||
return sync.compute_sync(
|
||||
cost_map, (sync.load_openrouter(OPENROUTER_RAW), sync.load_vercel(VERCEL_RAW, now_ms=NOW_MS))
|
||||
|
|
@ -157,19 +168,24 @@ def test_existing_entry_is_repriced_without_losing_curated_fields(sync: ModuleTy
|
|||
assert deepseek["output_cost_per_token"] == 1.73844e-6
|
||||
assert deepseek["cache_read_input_token_cost"] == 1.9316e-8
|
||||
assert deepseek["input_cost_per_token_cache_hit"] == 4.4e-8
|
||||
assert (deepseek["max_output_tokens"], deepseek["max_tokens"]) == (300000, 300000)
|
||||
assert (deepseek["max_output_tokens"], deepseek["max_tokens"]) == (384000, 384000)
|
||||
glm: Final = outcome.cost_map["vercel_ai_gateway/zai/glm-4.6"]
|
||||
assert (glm["input_cost_per_token"], glm["output_cost_per_token"]) == (6e-7, 2.2e-6)
|
||||
assert glm["supports_parallel_function_calling"] is True
|
||||
assert glm["supports_reasoning"] is True
|
||||
assert glm["max_output_tokens"] == 200000
|
||||
assert (glm["max_output_tokens"], glm["max_tokens"]) == (200000, 200000)
|
||||
openrouter, vercel = outcome.providers
|
||||
assert [line.split(":")[0] for line in openrouter.updated] == ["openrouter/deepseek/deepseek-v4-pro-0813"]
|
||||
assert "input_cost_per_token: 1.32e-06 -> 5.7948e-07" in openrouter.updated[0]
|
||||
assert "max_output_tokens: 300000 -> 384000; max_tokens: 300000 -> 384000" in openrouter.updated[0]
|
||||
assert [line.split(":")[0] for line in vercel.updated] == ["vercel_ai_gateway/zai/glm-4.6"]
|
||||
assert vercel.warnings == (
|
||||
"vercel_ai_gateway/zai/glm-4.6: max_output_tokens: 200000 -> 96000 held back: a shrinking limit; "
|
||||
"max_tokens: 200000 -> 96000 held back: a shrinking limit",
|
||||
)
|
||||
|
||||
|
||||
def test_legacy_max_tokens_is_never_paired_with_a_different_max_output_tokens(sync: ModuleType) -> None:
|
||||
def test_legacy_max_tokens_moves_in_step_with_the_catalog_output_ceiling(sync: ModuleType) -> None:
|
||||
legacy: Final = {
|
||||
"input_cost_per_token": 4e-8,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -180,9 +196,38 @@ def test_legacy_max_tokens_is_never_paired_with_a_different_max_output_tokens(sy
|
|||
outcome: Final = _run(sync, {**_base_map(), "openrouter/inception/mercury-2.5-preview": legacy})
|
||||
|
||||
mercury: Final = outcome.cost_map["openrouter/inception/mercury-2.5-preview"]
|
||||
assert mercury["max_tokens"] == 8192
|
||||
assert "max_output_tokens" not in mercury
|
||||
assert (mercury["max_output_tokens"], mercury["max_tokens"]) == (65536, 65536)
|
||||
assert mercury["max_input_tokens"] == 260000
|
||||
assert list(mercury) == [
|
||||
*legacy,
|
||||
"cache_read_input_token_cost",
|
||||
"max_input_tokens",
|
||||
"max_output_tokens",
|
||||
"supports_function_calling",
|
||||
"supports_reasoning",
|
||||
"supports_response_schema",
|
||||
"supports_tool_choice",
|
||||
]
|
||||
|
||||
|
||||
def test_output_limits_stay_put_when_the_catalog_has_no_output_ceiling(sync: ModuleType) -> None:
|
||||
existing: Final = {
|
||||
"litellm_provider": "openrouter",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1e-6,
|
||||
"output_cost_per_token": 2e-6,
|
||||
"max_input_tokens": 1000,
|
||||
"max_output_tokens": 500,
|
||||
"max_tokens": 500,
|
||||
}
|
||||
catalog: Final = sync.load_openrouter(_openrouter_rows({"id": "acme/x", "context_length": 4000}))
|
||||
|
||||
outcome: Final = sync.compute_sync({"openrouter/acme/x": dict(existing)}, (catalog,))
|
||||
|
||||
entry: Final = outcome.cost_map["openrouter/acme/x"]
|
||||
assert (entry["max_input_tokens"], entry["max_output_tokens"], entry["max_tokens"]) == (4000, 500, 500)
|
||||
assert outcome.providers[0].updated == ("openrouter/acme/x: max_input_tokens: 1000 -> 4000",)
|
||||
assert outcome.providers[0].warnings == ()
|
||||
|
||||
|
||||
def test_untouched_entries_survive_byte_for_byte(sync: ModuleType) -> None:
|
||||
|
|
@ -220,9 +265,10 @@ def test_mode_mismatch_warns_and_leaves_the_entry_alone(sync: ModuleType) -> Non
|
|||
vercel: Final = outcome.providers[1]
|
||||
assert "vercel_ai_gateway/openai/gpt-5-mini" not in vercel.added
|
||||
assert all("gpt-5-mini" not in line for line in vercel.updated)
|
||||
assert len(vercel.warnings) == 1
|
||||
assert "vercel_ai_gateway/openai/gpt-5-mini" in vercel.warnings[0]
|
||||
assert "'responses'" in vercel.warnings[0] and "'chat'" in vercel.warnings[0]
|
||||
mismatch: Final = [line for line in vercel.warnings if "gpt-5-mini" in line]
|
||||
assert len(mismatch) == 1
|
||||
assert "vercel_ai_gateway/openai/gpt-5-mini" in mismatch[0]
|
||||
assert "'responses'" in mismatch[0] and "'chat'" in mismatch[0]
|
||||
|
||||
|
||||
def test_new_keys_land_at_the_end_of_their_provider_block(sync: ModuleType) -> None:
|
||||
|
|
@ -360,7 +406,7 @@ def test_a_scheduled_deprecation_keeps_syncing_until_the_date(sync: ModuleType)
|
|||
passed: Final = sync.load_vercel(_vercel_language_row(NOW_MS), now_ms=NOW_MS)
|
||||
|
||||
assert [entry.key for entry in scheduled.entries] == ["vercel_ai_gateway/acme/chat-1"]
|
||||
assert dict(scheduled.skipped)["deprecated"] == 0
|
||||
assert dict(scheduled.skipped).get("deprecated", 0) == 0
|
||||
assert passed.entries == ()
|
||||
assert dict(passed.skipped)["deprecated"] == 1
|
||||
|
||||
|
|
@ -419,3 +465,280 @@ def test_dry_run_touches_nothing(sync: ModuleType, tmp_path: Path, capsys) -> No
|
|||
assert code == 0
|
||||
assert (repo / "model_prices_and_context_window.json").read_bytes() == before
|
||||
assert "dry run: no files were touched" in capsys.readouterr().out
|
||||
|
||||
|
||||
def test_new_entries_inherit_model_intrinsic_traits_from_the_root_entry(sync: ModuleType) -> None:
|
||||
root: Final = {
|
||||
"litellm_provider": "anthropic",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 5e-6,
|
||||
"supports_adaptive_thinking": True,
|
||||
"thinking_always_on": True,
|
||||
"supports_sampling_params": False,
|
||||
"supports_function_calling": False,
|
||||
"supports_vision": True,
|
||||
"prompt_cache_min_tokens": 1024,
|
||||
"supports_web_search": True,
|
||||
}
|
||||
already_synced: Final = {"litellm_provider": "openrouter", "mode": "chat", "input_cost_per_token": 5e-6}
|
||||
cost_map: Final = {
|
||||
"claude-fable-5": dict(root),
|
||||
"claude-fable-5-1": {**root, "prompt_cache_min_tokens": 512},
|
||||
"claude-embed-5": {**root, "mode": "embedding"},
|
||||
"openrouter/anthropic/claude-fable-5:thinking": dict(already_synced),
|
||||
}
|
||||
openrouter: Final = sync.load_openrouter(
|
||||
_openrouter_rows(
|
||||
{"id": "anthropic/claude-fable-5:batch", "supported_parameters": ["tools"]},
|
||||
{"id": "anthropic/claude-fable-5:thinking"},
|
||||
{"id": "anthropic/claude-embed-5"},
|
||||
{"id": "anthropic/claude-opus-6"},
|
||||
)
|
||||
)
|
||||
vercel: Final = sync.load_vercel(
|
||||
_vercel_rows({"id": "anthropic/claude-fable-5.1"}, {"id": "anthropic/claude-fable-5.1-fast"}), now_ms=NOW_MS
|
||||
)
|
||||
|
||||
outcome: Final = sync.compute_sync(cost_map, (openrouter, vercel))
|
||||
|
||||
batch: Final = outcome.cost_map["openrouter/anthropic/claude-fable-5:batch"]
|
||||
assert (batch["supports_adaptive_thinking"], batch["thinking_always_on"]) == (True, True)
|
||||
assert (batch["supports_sampling_params"], batch["prompt_cache_min_tokens"]) == (False, 1024)
|
||||
assert batch["supports_function_calling"] is True
|
||||
assert not {"supports_web_search", "supports_vision"} & batch.keys()
|
||||
assert outcome.cost_map["vercel_ai_gateway/anthropic/claude-fable-5.1"]["prompt_cache_min_tokens"] == 512
|
||||
fast: Final = outcome.cost_map["vercel_ai_gateway/anthropic/claude-fable-5.1-fast"]
|
||||
assert (fast["prompt_cache_min_tokens"], fast["supports_adaptive_thinking"]) == (512, True)
|
||||
assert "prompt_cache_min_tokens" not in outcome.cost_map["openrouter/anthropic/claude-opus-6"]
|
||||
assert "supports_adaptive_thinking" not in outcome.cost_map["openrouter/anthropic/claude-embed-5"]
|
||||
assert "supports_adaptive_thinking" not in outcome.cost_map["openrouter/anthropic/claude-fable-5:thinking"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("field", "old", "new", "reason"),
|
||||
[
|
||||
("input_cost_per_token", 1e-6, 0.0, "a price crossing zero"),
|
||||
("input_cost_per_token", 0.0, 1e-6, "a price crossing zero"),
|
||||
("output_cost_per_token", 1e-7, 2e-6, "a price moving more than 10x"),
|
||||
("output_cost_per_token", 2e-6, 1e-7, "a price moving more than 10x"),
|
||||
("max_input_tokens", 200000, 128000, "a shrinking limit"),
|
||||
],
|
||||
)
|
||||
def test_out_of_bounds_changes_are_held_back_as_warnings(
|
||||
sync: ModuleType, field: str, old: float, new: float, reason: str
|
||||
) -> None:
|
||||
existing: Final = {
|
||||
"litellm_provider": "openrouter",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 1e-6,
|
||||
"output_cost_per_token": 2e-6,
|
||||
"max_input_tokens": 200000,
|
||||
field: old,
|
||||
}
|
||||
catalog_row: Final = {
|
||||
"id": "acme/x",
|
||||
"context_length": 200000,
|
||||
"pricing": {"prompt": "0.000001", "completion": "0.000002"},
|
||||
**({"context_length": int(new)} if field == "max_input_tokens" else {}),
|
||||
}
|
||||
catalog_row["pricing"] = {
|
||||
**catalog_row["pricing"],
|
||||
**({"prompt": str(new)} if field == "input_cost_per_token" else {}),
|
||||
**({"completion": str(new)} if field == "output_cost_per_token" else {}),
|
||||
}
|
||||
|
||||
outcome: Final = sync.compute_sync(
|
||||
{"openrouter/acme/x": dict(existing)}, (sync.load_openrouter(_openrouter_rows(catalog_row)),)
|
||||
)
|
||||
|
||||
assert outcome.cost_map["openrouter/acme/x"] == existing
|
||||
assert outcome.has_changes is False
|
||||
assert outcome.providers[0].warnings == (f"openrouter/acme/x: {field}: {old!r} -> {new!r} held back: {reason}",)
|
||||
|
||||
|
||||
def test_a_price_move_within_ten_x_is_applied(sync: ModuleType) -> None:
|
||||
existing: Final = {"litellm_provider": "openrouter", "mode": "chat", "input_cost_per_token": 1e-6}
|
||||
catalog: Final = sync.load_openrouter(
|
||||
_openrouter_rows({"id": "acme/x", "pricing": {"prompt": "0.000009", "completion": "0"}})
|
||||
)
|
||||
|
||||
outcome: Final = sync.compute_sync({"openrouter/acme/x": existing}, (catalog,))
|
||||
|
||||
assert outcome.cost_map["openrouter/acme/x"]["input_cost_per_token"] == 9e-6
|
||||
assert outcome.providers[0].warnings == ()
|
||||
|
||||
|
||||
def test_vercel_long_context_tiers_map_to_above_threshold_prices(sync: ModuleType) -> None:
|
||||
pricing: Final = {
|
||||
"input": "0.0000015",
|
||||
"input_tiers": [
|
||||
{"cost": "0.0000015", "min": 0, "max": 32001},
|
||||
{"cost": "0.0000027", "min": 32001, "max": 128001},
|
||||
{"cost": "0.0000045", "min": 128001},
|
||||
],
|
||||
"output": "0.0000075",
|
||||
"output_tiers": [{"cost": "0.0000075", "max": 200001}, {"cost": "0.00001125", "min": 200001}],
|
||||
"input_cache_read": "0.0000003",
|
||||
"input_cache_read_tiers": [{"cost": "0.0000006", "min": 256000}],
|
||||
"input_cache_write": "0.000002",
|
||||
"input_cache_write_tiers": [{"cost": "0.000002", "min": 0, "max": 200001}, {"cost": "0.000004", "min": 200001}],
|
||||
}
|
||||
catalog: Final = sync.load_vercel(_vercel_rows({"id": "acme/long", "pricing": pricing}), now_ms=NOW_MS)
|
||||
|
||||
outcome: Final = sync.compute_sync({}, (catalog,))
|
||||
|
||||
entry: Final = outcome.cost_map["vercel_ai_gateway/acme/long"]
|
||||
assert entry["input_cost_per_token"] == 1.5e-6
|
||||
assert entry["input_cost_per_token_above_32k_tokens"] == 2.7e-6
|
||||
assert entry["input_cost_per_token_above_128k_tokens"] == 4.5e-6
|
||||
assert entry["output_cost_per_token_above_200k_tokens"] == 1.125e-5
|
||||
assert entry["cache_read_input_token_cost_above_256k_tokens"] == 6e-7
|
||||
assert entry["cache_creation_input_token_cost_above_200k_tokens"] == 4e-6
|
||||
assert not any(key.endswith("_above_0k_tokens") for key in entry)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("tiers", "problem"),
|
||||
[
|
||||
(
|
||||
[{"cost": "0.000001", "min": 0, "max": 150500}, {"cost": "0.000002", "min": 150500}],
|
||||
"input_cost_per_token tiers have a boundary that is not a whole thousand or an unusable price",
|
||||
),
|
||||
(
|
||||
[{"cost": "0.000001", "min": 0, "max": 128000}, {"cost": "0.000002", "min": 200000}],
|
||||
"input_cost_per_token tiers are not contiguous",
|
||||
),
|
||||
(
|
||||
[{"cost": "0.000001", "min": 0, "max": 128000}, {"cost": "0.000002", "min": 128000, "max": 256000}],
|
||||
"input_cost_per_token tiers are not contiguous",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_unmappable_tiers_skip_the_row_with_a_warning(sync: ModuleType, tiers: list, problem: str) -> None:
|
||||
pricing: Final = {"input": "0.000001", "input_tiers": tiers, "output": "0.000002"}
|
||||
catalog: Final = sync.load_vercel(_vercel_rows({"id": "acme/odd", "pricing": pricing}), now_ms=NOW_MS)
|
||||
|
||||
outcome: Final = sync.compute_sync({}, (catalog,))
|
||||
|
||||
assert "vercel_ai_gateway/acme/odd" not in outcome.cost_map
|
||||
assert dict(catalog.skipped) == {"tiers outside the registry's thresholds": 1}
|
||||
assert outcome.providers[0].warnings == (f"vercel_ai_gateway/acme/odd: {problem}; row skipped",)
|
||||
|
||||
|
||||
def test_a_price_that_varies_by_provider_seeds_but_never_overwrites(sync: ModuleType) -> None:
|
||||
curated: Final = {
|
||||
"litellm_provider": "vercel_ai_gateway",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 9e-7,
|
||||
"output_cost_per_token": 2e-6,
|
||||
"max_input_tokens": 100000,
|
||||
}
|
||||
row: Final = {
|
||||
"context_window": 262144,
|
||||
"pricing": {
|
||||
"input": "0.0000015",
|
||||
"input_tiers": [{"cost": "0.0000015", "min": 0, "max": 128001}, {"cost": "0.000003", "min": 128001}],
|
||||
"output": "0.000002",
|
||||
"input_cache_read": "0.0000003",
|
||||
"varies_by_provider": True,
|
||||
},
|
||||
}
|
||||
catalog: Final = sync.load_vercel(
|
||||
_vercel_rows({"id": "acme/curated", **row}, {"id": "acme/fresh", **row}), now_ms=NOW_MS
|
||||
)
|
||||
|
||||
outcome: Final = sync.compute_sync({"vercel_ai_gateway/acme/curated": dict(curated)}, (catalog,))
|
||||
|
||||
existing: Final = outcome.cost_map["vercel_ai_gateway/acme/curated"]
|
||||
assert (existing["input_cost_per_token"], existing["max_input_tokens"]) == (9e-7, 262144)
|
||||
assert not any("cache_read" in name or "_above_" in name for name in existing)
|
||||
fresh: Final = outcome.cost_map["vercel_ai_gateway/acme/fresh"]
|
||||
assert (fresh["input_cost_per_token"], fresh["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6)
|
||||
assert fresh["cache_read_input_token_cost"] == 3e-7
|
||||
assert outcome.providers[0].warnings == (
|
||||
"vercel_ai_gateway/acme/curated: input_cost_per_token: 9e-07 -> 1.5e-06 held back: "
|
||||
"the catalog price varies by provider; cache_read_input_token_cost: None -> 3e-07 held back: "
|
||||
"the catalog price varies by provider; input_cost_per_token_above_128k_tokens: None -> 3e-06 held back: "
|
||||
"the catalog price varies by provider",
|
||||
)
|
||||
|
||||
|
||||
def test_image_and_audio_outputs_are_priced_per_token_or_skipped(sync: ModuleType) -> None:
|
||||
openrouter: Final = sync.load_openrouter(
|
||||
_openrouter_rows(
|
||||
{
|
||||
"id": "openai/gpt-5-image",
|
||||
"architecture": {"input_modalities": ["text", "image"], "output_modalities": ["image", "text"]},
|
||||
"pricing": {"prompt": "0.00001", "completion": "0.00001", "image_output": "0.00004"},
|
||||
},
|
||||
{
|
||||
"id": "acme/talker",
|
||||
"architecture": {"input_modalities": ["text", "audio"], "output_modalities": ["text", "audio"]},
|
||||
"pricing": {
|
||||
"prompt": "0.000001",
|
||||
"completion": "0.000002",
|
||||
"audio": "0.000005",
|
||||
"audio_output": "0.00001",
|
||||
},
|
||||
},
|
||||
{
|
||||
"id": "acme/mute",
|
||||
"architecture": {"input_modalities": ["text"], "output_modalities": ["text", "audio"]},
|
||||
},
|
||||
{"id": "acme/painter", "architecture": {"output_modalities": ["image"]}},
|
||||
)
|
||||
)
|
||||
vercel: Final = sync.load_vercel(
|
||||
_vercel_rows(
|
||||
{
|
||||
"id": "acme/speaker",
|
||||
"modalities": {"input": ["text", "audio"], "output": ["text", "audio"]},
|
||||
"pricing": {
|
||||
"input": "0.000001",
|
||||
"output": "0.000002",
|
||||
"audio_input_token_cost": "0.000004",
|
||||
"audio_output_token_cost": "0.000008",
|
||||
},
|
||||
},
|
||||
{"id": "acme/drawer", "modalities": {"input": ["text"], "output": ["text", "image"]}},
|
||||
),
|
||||
now_ms=NOW_MS,
|
||||
)
|
||||
|
||||
outcome: Final = sync.compute_sync({}, (openrouter, vercel))
|
||||
|
||||
image: Final = outcome.cost_map["openrouter/openai/gpt-5-image"]
|
||||
assert (image["output_cost_per_image_token"], image["mode"], image["supports_vision"]) == (4e-5, "chat", True)
|
||||
talker: Final = outcome.cost_map["openrouter/acme/talker"]
|
||||
assert (talker["input_cost_per_audio_token"], talker["output_cost_per_audio_token"]) == (5e-6, 1e-5)
|
||||
assert (talker["supports_audio_input"], talker["supports_audio_output"]) == (True, True)
|
||||
speaker: Final = outcome.cost_map["vercel_ai_gateway/acme/speaker"]
|
||||
assert (speaker["input_cost_per_audio_token"], speaker["output_cost_per_audio_token"]) == (4e-6, 8e-6)
|
||||
assert speaker["supports_audio_output"] is True
|
||||
assert {"openrouter/acme/mute", "openrouter/acme/painter", "vercel_ai_gateway/acme/drawer"}.isdisjoint(
|
||||
outcome.cost_map
|
||||
)
|
||||
assert dict(openrouter.skipped) == {"output priced outside the catalog": 2}
|
||||
assert dict(vercel.skipped) == {"output priced outside the catalog": 1}
|
||||
|
||||
|
||||
def test_updates_keep_the_curated_key_order_and_append_new_keys(sync: ModuleType) -> None:
|
||||
curated: Final = {
|
||||
"mode": "chat",
|
||||
"output_cost_per_token": 2e-6,
|
||||
"input_cost_per_token": 1e-6,
|
||||
"litellm_provider": "openrouter",
|
||||
}
|
||||
catalog: Final = sync.load_openrouter(
|
||||
_openrouter_rows(
|
||||
{
|
||||
"id": "acme/x",
|
||||
"pricing": {"prompt": "0.000003", "completion": "0.000002", "input_cache_read": "0.0000001"},
|
||||
}
|
||||
)
|
||||
)
|
||||
|
||||
outcome: Final = sync.compute_sync({"openrouter/acme/x": dict(curated)}, (catalog,))
|
||||
|
||||
assert list(outcome.cost_map["openrouter/acme/x"]) == [*curated, "cache_read_input_token_cost"]
|
||||
assert outcome.cost_map["openrouter/acme/x"]["input_cost_per_token"] == 3e-6
|
||||
|
|
|
|||
|
|
@ -906,6 +906,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_creation_input_audio_token_cost": {"type": "number"},
|
||||
"cache_creation_input_token_cost": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_1hr": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_32k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_128k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_200k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_256k_tokens": {"type": "number"},
|
||||
|
|
@ -919,6 +920,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_creation_input_token_cost_flex": {"type": "number"},
|
||||
"cache_creation_input_token_cost_priority": {"type": "number"},
|
||||
"cache_read_input_token_cost": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_32k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_128k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_200k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_256k_tokens": {"type": "number"},
|
||||
|
|
@ -974,6 +976,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"input_cost_per_request": {"type": "number"},
|
||||
"input_cost_per_second": {"type": "number"},
|
||||
"input_cost_per_token": {"type": "number"},
|
||||
"input_cost_per_token_above_32k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_above_128k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_batches": {"type": "number"},
|
||||
"input_cost_per_token_cache_hit": {"type": "number"},
|
||||
|
|
@ -1029,6 +1032,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"output_cost_per_second_1080p": {"type": "number"},
|
||||
"output_cost_per_second_4k": {"type": "number"},
|
||||
"output_cost_per_token": {"type": "number"},
|
||||
"output_cost_per_token_above_32k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_128k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_200k_tokens": {"type": "number"},
|
||||
"output_cost_per_token_above_256k_tokens": {"type": "number"},
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue