fix(cost-map-sync): apply copied output caps and keep bot-seeded Vercel prices in sync

A curated max_output_tokens equal to the entry's own max_input_tokens is a copy of the
context window, so the catalog's ceiling replaces it instead of being held as a shrink.
A varies_by_provider price now only holds when a human priced the entry: an entry whose
source is its catalog page keeps following the catalog. The exact-name root wins trait
inheritance in any mode, a capability flag curated as false is held for a human, and the
PR body lists every held change unless that would pass GitHub's body limit.
This commit is contained in:
mateo-berri 2026-09-05 02:35:38 -07:00
parent 81142d42e2
commit 412db8dad3
2 changed files with 108 additions and 35 deletions

View file

@ -9,18 +9,20 @@ Policy:
- Both catalogs price per token as decimal strings; values are normalized to six significant digits.
- Vercel long-context tiers map to the registry's ``*_above_<N>k_tokens`` keys, which litellm applies once the
prompt exceeds N thousand tokens. A row whose tier boundaries are not whole thousands is skipped with a warning.
- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry but never overwrites a
curated price, and a difference is reported as a warning.
- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry and keeps that entry in
sync (its ``source`` is the catalog page), but never overwrites a curated price; that difference is reported as
a warning.
- Image and audio output are priced from the catalog's per-token ``image_output`` and ``audio_output`` prices. A row
whose non-text output the catalog does not price per token is skipped.
- A new entry inherits the traits no catalog expresses (adaptive thinking, sampling params, cache minimums, system
messages) from the same model's root registry entry, found by the bare model name or its longest dash-prefix
with the same mode, so the family-wide invariants the test suite enforces hold for the route too.
messages) from the same model's root registry entry, found by the bare model name in any mode or else its
longest dash-prefix with the same mode, so the family-wide invariants the test suite enforces hold for the route.
- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed and a
capability flag the catalog does not claim stays as curated. ``max_output_tokens`` and ``max_tokens`` move as a
pair and only when the catalog states an output ceiling.
- A limit that would shrink, a price that would cross zero, and a price that would move more than 10x either way
are held back as warnings for a human instead of applied.
pair and only when the catalog states an output ceiling; a curated output cap equal to the entry's own context
window is a copy of that window, not a ceiling, so the catalog's ceiling replaces it.
- A limit that would shrink, a price that would cross zero, a price that would move more than 10x either way, and
a capability flag curated as false are held back as warnings for a human instead of applied.
- Router models and rows without a usable prompt and completion price are skipped.
- A registry entry absent from its catalog is left untouched; retiring a model stays a human call.
"""
@ -60,6 +62,7 @@ INHERITED_TRAITS: Final = frozenset(
}
)
PR_BODY_SECTION_LIMIT: Final = 30
GITHUB_BODY_LIMIT: Final = 65_536
Provider = Literal["openrouter", "vercel_ai_gateway"]
RegistryEntry = dict[str, object]
@ -430,11 +433,12 @@ def _root_candidates(bare: str) -> tuple[str, ...]:
def _inherited(cost_map: CostMap, entry: CatalogEntry) -> Mapping[str, object]:
bare: Final = entry.key.rsplit("/", 1)[-1].split(":", 1)[0]
same_name: Final = frozenset((bare, bare.replace(".", "-")))
root: Final = next(
(
candidate
for candidate in map(cost_map.get, _root_candidates(bare))
if isinstance(candidate, dict) and candidate.get("mode") == entry.mode
for name, candidate in ((name, cost_map.get(name)) for name in _root_candidates(bare))
if isinstance(candidate, dict) and (name in same_name or candidate.get("mode") == entry.mode)
),
None,
)
@ -483,6 +487,8 @@ def _hold(name: str, old: object, new: object, curated_prices_win: bool) -> str
return "the catalog price varies by provider"
if old is None:
return None
if name.startswith("supports_") and old is False:
return "a capability flag curated as false"
if name.startswith("max_") and isinstance(old, int) and isinstance(new, int) and new < old:
return "a shrinking limit"
if "cost" in name and isinstance(old, int | float) and isinstance(new, int | float):
@ -491,7 +497,9 @@ def _hold(name: str, old: object, new: object, curated_prices_win: bool) -> str
def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange, ...]:
curated_prices_win: Final = entry.indicative_prices and "input_cost_per_token" in existing
curated_prices_win: Final = (
entry.indicative_prices and "input_cost_per_token" in existing and existing.get("source") != entry.source
)
scalars: Final = tuple(
FieldChange(name, existing.get(name), value, _hold(name, existing.get(name), value, curated_prices_win))
for name, value in entry.fields.items()
@ -501,7 +509,8 @@ def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange,
if ceiling is None:
return scalars
current: Final = existing.get("max_output_tokens", existing.get("max_tokens"))
hold: Final = _hold("max_output_tokens", current, ceiling, curated_prices_win)
curated_cap: Final = None if current == existing.get("max_input_tokens") else current
hold: Final = _hold("max_output_tokens", curated_cap, ceiling, curated_prices_win)
return (
*scalars,
*(FieldChange(name, existing.get(name), ceiling, hold) for name in LIMIT_PAIR if existing.get(name) != ceiling),
@ -616,7 +625,7 @@ def _section_block(title: str, lines: Sequence[str], backtick: bool, limit: int
return f"### {title} ({len(lines)})\n{bullets}{trailer}\n"
def _provider_body(outcome: ProviderOutcome, limit: int | None) -> str:
def _provider_body(outcome: ProviderOutcome, limit: int | None, warnings_limit: int | None) -> str:
skipped: Final = ", ".join(f"{reason} ({count})" for reason, count in sorted(outcome.skipped.items())) or "none"
return (
f"## {outcome.provider}\n"
@ -625,23 +634,30 @@ def _provider_body(outcome: ProviderOutcome, limit: int | None) -> str:
"\n"
f"{_section_block('Updated', outcome.updated, True, limit, 'diff')}"
"\n"
f"{_section_block('Warnings needing a human call', outcome.warnings, False, limit, 'workflow log')}"
f"{_section_block('Warnings needing a human call', outcome.warnings, False, warnings_limit, 'workflow log')}"
"\n"
f"Catalog rows skipped: {skipped}\n"
)
def render_pr_body(outcome: SyncOutcome, section_limit: int | None = PR_BODY_SECTION_LIMIT) -> str:
def _pr_body(outcome: SyncOutcome, section_limit: int | None, warnings_limit: int | None) -> str:
return (
"Automated sync of the openrouter and vercel_ai_gateway entries in model_prices_and_context_window.json "
f"against `GET {OPENROUTER_MODELS_URL}` and `GET {VERCEL_MODELS_URL}` by scripts/sync_cost_map.py. "
"The cost-map-guard check enforces that this PR only adds or reprices models. Changes the script held "
"back (shrinking limits, prices crossing zero or moving more than 10x, per-provider prices) are listed "
"under the warnings and need a human commit.\n"
"\n" + "\n".join(_provider_body(provider, section_limit) for provider in outcome.providers)
"back (shrinking limits, prices crossing zero or moving more than 10x, per-provider prices on curated "
"rows, capability flags curated as false) are listed under the warnings and need a human commit.\n"
"\n" + "\n".join(_provider_body(provider, section_limit, warnings_limit) for provider in outcome.providers)
)
def render_pr_body(outcome: SyncOutcome, section_limit: int | None = PR_BODY_SECTION_LIMIT) -> str:
every_warning: Final = _pr_body(outcome, section_limit, None)
if section_limit is None or len(every_warning) <= GITHUB_BODY_LIMIT:
return every_warning
return _pr_body(outcome, section_limit, section_limit)
def render_summary(outcome: SyncOutcome) -> str:
return " ".join(
f"{provider.provider}: added={len(provider.added)} updated={len(provider.updated)} "

View file

@ -176,16 +176,14 @@ def test_existing_entry_is_repriced_without_losing_curated_fields(sync: ModuleTy
assert (glm["input_cost_per_token"], glm["output_cost_per_token"]) == (6e-7, 2.2e-6)
assert glm["supports_parallel_function_calling"] is True
assert glm["supports_reasoning"] is True
assert (glm["max_output_tokens"], glm["max_tokens"]) == (200000, 200000)
assert (glm["max_output_tokens"], glm["max_tokens"]) == (96000, 96000)
openrouter, vercel = outcome.providers
assert [line.split(":")[0] for line in openrouter.updated] == ["openrouter/deepseek/deepseek-v4-pro-0813"]
assert "input_cost_per_token: 1.32e-06 -> 5.7948e-07" in openrouter.updated[0]
assert "max_output_tokens: 300000 -> 384000; max_tokens: 300000 -> 384000" in openrouter.updated[0]
assert [line.split(":")[0] for line in vercel.updated] == ["vercel_ai_gateway/zai/glm-4.6"]
assert vercel.warnings == (
"vercel_ai_gateway/zai/glm-4.6: max_output_tokens: 200000 -> 96000 held back: a shrinking limit; "
"max_tokens: 200000 -> 96000 held back: a shrinking limit",
)
assert "max_output_tokens: 200000 -> 96000; max_tokens: 200000 -> 96000" in vercel.updated[0]
assert vercel.warnings == ()
def test_legacy_max_tokens_moves_in_step_with_the_catalog_output_ceiling(sync: ModuleType) -> None:
@ -213,6 +211,36 @@ def test_legacy_max_tokens_moves_in_step_with_the_catalog_output_ceiling(sync: M
]
def test_an_output_cap_copied_from_the_context_window_yields_to_the_catalog_ceiling(sync: ModuleType) -> None:
copied: Final = {
"litellm_provider": "openrouter",
"mode": "chat",
"input_cost_per_token": 1e-6,
"output_cost_per_token": 2e-6,
"max_input_tokens": 163840,
"max_output_tokens": 163840,
"max_tokens": 163840,
}
genuine: Final = {**copied, "max_output_tokens": 100000, "max_tokens": 100000}
row: Final = {"context_length": 163840, "top_provider": {"max_completion_tokens": 65536}}
catalog: Final = sync.load_openrouter(_openrouter_rows({"id": "acme/copied", **row}, {"id": "acme/genuine", **row}))
outcome: Final = sync.compute_sync(
{"openrouter/acme/copied": dict(copied), "openrouter/acme/genuine": dict(genuine)}, (catalog,)
)
applied: Final = outcome.cost_map["openrouter/acme/copied"]
assert (applied["max_input_tokens"], applied["max_output_tokens"], applied["max_tokens"]) == (163840, 65536, 65536)
assert outcome.cost_map["openrouter/acme/genuine"] == genuine
assert outcome.providers[0].updated == (
"openrouter/acme/copied: max_output_tokens: 163840 -> 65536; max_tokens: 163840 -> 65536",
)
assert outcome.providers[0].warnings == (
"openrouter/acme/genuine: max_output_tokens: 100000 -> 65536 held back: a shrinking limit; "
"max_tokens: 100000 -> 65536 held back: a shrinking limit",
)
def test_output_limits_stay_put_when_the_catalog_has_no_output_ceiling(sync: ModuleType) -> None:
existing: Final = {
"litellm_provider": "openrouter",
@ -313,25 +341,36 @@ def test_pr_body_lists_changes_per_provider(sync: ModuleType) -> None:
assert "Catalog rows skipped: deprecated (1), no usable price (1), not token priced (1)" in body
def test_pr_body_caps_every_section_so_a_large_first_sync_fits_github_limit(sync: ModuleType) -> None:
def _large_outcome(sync: ModuleType, warning_count: int):
lines: Final = tuple(f"provider/model-{index}: {'x' * 200}" for index in range(400))
outcome: Final = sync.SyncOutcome(
return sync.SyncOutcome(
cost_map={},
providers=tuple(
sync.ProviderOutcome(provider=provider, added=lines, updated=lines, warnings=lines, skipped={})
sync.ProviderOutcome(
provider=provider, added=lines, updated=lines, warnings=lines[:warning_count], skipped={}
)
for provider in ("openrouter", "vercel_ai_gateway")
),
)
body: Final = sync.render_pr_body(outcome)
def test_pr_body_caps_added_and_updated_but_lists_every_warning(sync: ModuleType) -> None:
body: Final = sync.render_pr_body(_large_outcome(sync, warning_count=40))
assert body.count("### Added (400)") == 2 and body.count("- and 370 more, see the diff") == 4
assert "provider/model-30: " in body and body.count("- provider/model-39: ") == 2
assert body.count("### Warnings needing a human call (40)") == 2 and "see the workflow log" not in body
def test_pr_body_caps_warnings_too_only_when_they_would_push_it_past_github_limit(sync: ModuleType) -> None:
body: Final = sync.render_pr_body(_large_outcome(sync, warning_count=400))
assert len(body) < 65_536
assert body.count("### Added (400)") == 2 and body.count("- and 370 more, see the diff") == 4
assert body.count("- and 370 more, see the workflow log") == 2
assert body.count("- `provider/model-29: ") == 4 and "provider/model-30: " not in body
def test_workflow_log_lists_every_warning_the_capped_pr_body_drops(sync: ModuleType, tmp_path: Path, capsys) -> None:
def test_every_held_change_reaches_the_pr_body_and_the_workflow_log(sync: ModuleType, tmp_path: Path, capsys) -> None:
keys: Final = tuple(f"openrouter/acme/model-{index:02d}" for index in range(40))
catalog: Final = {
"data": [
@ -362,8 +401,8 @@ def test_workflow_log_lists_every_warning_the_capped_pr_body_drops(sync: ModuleT
body: Final = body_file.read_text()
log: Final = capsys.readouterr().out
assert code == 0
assert "### Warnings needing a human call (40)" in body and "- and 10 more, see the workflow log" in body
assert keys[29] in body and keys[30] not in body
assert "### Warnings needing a human call (40)" in body and "more, see the workflow log" not in body
assert all(key in body for key in keys)
assert all(key in log for key in keys) and "more, see the" not in log
@ -488,6 +527,8 @@ def test_new_entries_inherit_model_intrinsic_traits_from_the_root_entry(sync: Mo
"claude-fable-5": dict(root),
"claude-fable-5-1": {**root, "prompt_cache_min_tokens": 512},
"claude-embed-5": {**root, "mode": "embedding"},
"gpt-5": {"litellm_provider": "openai", "mode": "chat", "supports_system_messages": True},
"gpt-5-codex": {"litellm_provider": "openai", "mode": "responses", "supports_system_messages": False},
"openrouter/anthropic/claude-fable-5:thinking": dict(already_synced),
}
openrouter: Final = sync.load_openrouter(
@ -495,11 +536,17 @@ def test_new_entries_inherit_model_intrinsic_traits_from_the_root_entry(sync: Mo
{"id": "anthropic/claude-fable-5:batch", "supported_parameters": ["tools"]},
{"id": "anthropic/claude-fable-5:thinking"},
{"id": "anthropic/claude-embed-5"},
{"id": "anthropic/claude-embed-5-fast"},
{"id": "anthropic/claude-opus-6"},
)
)
vercel: Final = sync.load_vercel(
_vercel_rows({"id": "anthropic/claude-fable-5.1"}, {"id": "anthropic/claude-fable-5.1-fast"}), now_ms=NOW_MS
_vercel_rows(
{"id": "anthropic/claude-fable-5.1"},
{"id": "anthropic/claude-fable-5.1-fast"},
{"id": "openai/gpt-5-codex"},
),
now_ms=NOW_MS,
)
outcome: Final = sync.compute_sync(cost_map, (openrouter, vercel))
@ -513,8 +560,10 @@ def test_new_entries_inherit_model_intrinsic_traits_from_the_root_entry(sync: Mo
fast: Final = outcome.cost_map["vercel_ai_gateway/anthropic/claude-fable-5.1-fast"]
assert (fast["prompt_cache_min_tokens"], fast["supports_adaptive_thinking"]) == (512, True)
assert "prompt_cache_min_tokens" not in outcome.cost_map["openrouter/anthropic/claude-opus-6"]
assert "supports_adaptive_thinking" not in outcome.cost_map["openrouter/anthropic/claude-embed-5"]
assert outcome.cost_map["openrouter/anthropic/claude-embed-5"]["supports_adaptive_thinking"] is True
assert "supports_adaptive_thinking" not in outcome.cost_map["openrouter/anthropic/claude-embed-5-fast"]
assert "supports_adaptive_thinking" not in outcome.cost_map["openrouter/anthropic/claude-fable-5:thinking"]
assert outcome.cost_map["vercel_ai_gateway/openai/gpt-5-codex"]["supports_system_messages"] is False
@pytest.mark.parametrize(
@ -525,6 +574,7 @@ def test_new_entries_inherit_model_intrinsic_traits_from_the_root_entry(sync: Mo
("output_cost_per_token", 1e-7, 2e-6, "a price moving more than 10x"),
("output_cost_per_token", 2e-6, 1e-7, "a price moving more than 10x"),
("max_input_tokens", 200000, 128000, "a shrinking limit"),
("supports_reasoning", False, True, "a capability flag curated as false"),
],
)
def test_out_of_bounds_changes_are_held_back_as_warnings(
@ -543,6 +593,7 @@ def test_out_of_bounds_changes_are_held_back_as_warnings(
"context_length": 200000,
"pricing": {"prompt": "0.000001", "completion": "0.000002"},
**({"context_length": int(new)} if field == "max_input_tokens" else {}),
**({"supported_parameters": ["reasoning"]} if field == "supports_reasoning" else {}),
}
catalog_row["pricing"] = {
**catalog_row["pricing"],
@ -645,7 +696,7 @@ def test_unmappable_tiers_skip_the_row_with_a_warning(sync: ModuleType, tiers: l
assert outcome.providers[0].warnings == (f"vercel_ai_gateway/acme/odd: {problem}; row skipped",)
def test_a_price_that_varies_by_provider_seeds_but_never_overwrites(sync: ModuleType) -> None:
def test_a_price_that_varies_by_provider_seeds_and_only_overwrites_what_the_bot_seeded(sync: ModuleType) -> None:
curated: Final = {
"litellm_provider": "vercel_ai_gateway",
"mode": "chat",
@ -653,6 +704,7 @@ def test_a_price_that_varies_by_provider_seeds_but_never_overwrites(sync: Module
"output_cost_per_token": 2e-6,
"max_input_tokens": 100000,
}
seeded: Final = {**curated, "source": "https://vercel.com/ai-gateway/models/seeded"}
row: Final = {
"context_window": 262144,
"pricing": {
@ -664,14 +716,19 @@ def test_a_price_that_varies_by_provider_seeds_but_never_overwrites(sync: Module
},
}
catalog: Final = sync.load_vercel(
_vercel_rows({"id": "acme/curated", **row}, {"id": "acme/fresh", **row}), now_ms=NOW_MS
_vercel_rows({"id": "acme/curated", **row}, {"id": "acme/seeded", **row}, {"id": "acme/fresh", **row}),
now_ms=NOW_MS,
)
outcome: Final = sync.compute_sync({"vercel_ai_gateway/acme/curated": dict(curated)}, (catalog,))
outcome: Final = sync.compute_sync(
{"vercel_ai_gateway/acme/curated": dict(curated), "vercel_ai_gateway/acme/seeded": dict(seeded)}, (catalog,)
)
existing: Final = outcome.cost_map["vercel_ai_gateway/acme/curated"]
assert (existing["input_cost_per_token"], existing["max_input_tokens"]) == (9e-7, 262144)
assert not any("cache_read" in name or "_above_" in name for name in existing)
resynced: Final = outcome.cost_map["vercel_ai_gateway/acme/seeded"]
assert (resynced["input_cost_per_token"], resynced["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6)
fresh: Final = outcome.cost_map["vercel_ai_gateway/acme/fresh"]
assert (fresh["input_cost_per_token"], fresh["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6)
assert fresh["cache_read_input_token_cost"] == 3e-7