fix(cost-map-sync): mark bot-seeded headline prices and hold glitchy copied caps

A Vercel row seeded at a varies_by_provider headline price now carries
price_varies_by_provider: true and only a row with that mark keeps
following the catalog; a curated row citing the catalog page as source
holds its price like any other. A copied output cap still yields to the
catalog ceiling, but a ceiling more than 10x below it is held as a
catalog glitch. The map schema test learns the new key.
This commit is contained in:
mateo-berri 2026-09-05 03:05:59 -07:00
parent 412db8dad3
commit fb8f9af5a1
3 changed files with 64 additions and 21 deletions

View file

@ -9,9 +9,9 @@ Policy:
- Both catalogs price per token as decimal strings; values are normalized to six significant digits.
- Vercel long-context tiers map to the registry's ``*_above_<N>k_tokens`` keys, which litellm applies once the
prompt exceeds N thousand tokens. A row whose tier boundaries are not whole thousands is skipped with a warning.
- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry and keeps that entry in
sync (its ``source`` is the catalog page), but never overwrites a curated price; that difference is reported as
a warning.
- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry, marked
``price_varies_by_provider``, and keeps a marked entry in sync, but never overwrites the price of an entry without
the mark; that difference is reported as a warning. A human who curates one provider's own price drops the mark.
- Image and audio output are priced from the catalog's per-token ``image_output`` and ``audio_output`` prices. A row
whose non-text output the catalog does not price per token is skipped.
- A new entry inherits the traits no catalog expresses (adaptive thinking, sampling params, cache minimums, system
@ -20,7 +20,8 @@ Policy:
- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed and a
capability flag the catalog does not claim stays as curated. ``max_output_tokens`` and ``max_tokens`` move as a
pair and only when the catalog states an output ceiling; a curated output cap equal to the entry's own context
window is a copy of that window, not a ceiling, so the catalog's ceiling replaces it.
window is a copy of that window, not a ceiling, so the catalog's ceiling replaces it unless that would shrink it
more than 10x.
- A limit that would shrink, a price that would cross zero, a price that would move more than 10x either way, and
a capability flag curated as false are held back as warnings for a human instead of applied.
- Router models and rows without a usable prompt and completion price are skipped.
@ -51,7 +52,7 @@ OPENROUTER_MODELS_URL: Final = "https://openrouter.ai/api/v1/models"
VERCEL_MODELS_URL: Final = "https://ai-gateway.vercel.sh/v1/models"
VERCEL_TYPE_TO_MODE: Final = MappingProxyType({"language": "chat", "embedding": "embedding"})
LIMIT_PAIR: Final = ("max_output_tokens", "max_tokens")
PRICE_SWING_LIMIT: Final = 10
SWING_LIMIT: Final = 10
INHERITED_TRAITS: Final = frozenset(
{
"prompt_cache_min_tokens",
@ -456,6 +457,7 @@ def _new_entry(entry: CatalogEntry, inherited: Mapping[str, object]) -> Registry
"litellm_provider": entry.provider,
"mode": entry.mode,
"source": entry.source,
**({"price_varies_by_provider": True} if entry.indicative_prices else {}),
}.items()
)
)
@ -477,8 +479,14 @@ class FieldChange:
def _swing(old: float, new: float) -> str | None:
if (old == 0) != (new == 0):
return "a price crossing zero"
if old and new and max(new / old, old / new) > PRICE_SWING_LIMIT:
return f"a price moving more than {PRICE_SWING_LIMIT}x"
if old and new and max(new / old, old / new) > SWING_LIMIT:
return f"a price moving more than {SWING_LIMIT}x"
return None
def _copied_cap_hold(current: object, ceiling: object) -> str | None:
if isinstance(current, int) and isinstance(ceiling, int) and ceiling * SWING_LIMIT < current:
return f"a copied output cap shrinking more than {SWING_LIMIT}x"
return None
@ -498,7 +506,7 @@ def _hold(name: str, old: object, new: object, curated_prices_win: bool) -> str
def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange, ...]:
curated_prices_win: Final = (
entry.indicative_prices and "input_cost_per_token" in existing and existing.get("source") != entry.source
entry.indicative_prices and "input_cost_per_token" in existing and not existing.get("price_varies_by_provider")
)
scalars: Final = tuple(
FieldChange(name, existing.get(name), value, _hold(name, existing.get(name), value, curated_prices_win))
@ -509,8 +517,11 @@ def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange,
if ceiling is None:
return scalars
current: Final = existing.get("max_output_tokens", existing.get("max_tokens"))
curated_cap: Final = None if current == existing.get("max_input_tokens") else current
hold: Final = _hold("max_output_tokens", curated_cap, ceiling, curated_prices_win)
hold: Final = (
_copied_cap_hold(current, ceiling)
if current == existing.get("max_input_tokens")
else _hold("max_output_tokens", current, ceiling, curated_prices_win)
)
return (
*scalars,
*(FieldChange(name, existing.get(name), ceiling, hold) for name in LIMIT_PAIR if existing.get(name) != ceiling),

View file

@ -223,21 +223,34 @@ def test_an_output_cap_copied_from_the_context_window_yields_to_the_catalog_ceil
}
genuine: Final = {**copied, "max_output_tokens": 100000, "max_tokens": 100000}
row: Final = {"context_length": 163840, "top_provider": {"max_completion_tokens": 65536}}
catalog: Final = sync.load_openrouter(_openrouter_rows({"id": "acme/copied", **row}, {"id": "acme/genuine", **row}))
glitch_row: Final = {"context_length": 163840, "top_provider": {"max_completion_tokens": 8192}}
catalog: Final = sync.load_openrouter(
_openrouter_rows(
{"id": "acme/copied", **row}, {"id": "acme/genuine", **row}, {"id": "acme/glitch", **glitch_row}
)
)
outcome: Final = sync.compute_sync(
{"openrouter/acme/copied": dict(copied), "openrouter/acme/genuine": dict(genuine)}, (catalog,)
{
"openrouter/acme/copied": dict(copied),
"openrouter/acme/genuine": dict(genuine),
"openrouter/acme/glitch": dict(copied),
},
(catalog,),
)
applied: Final = outcome.cost_map["openrouter/acme/copied"]
assert (applied["max_input_tokens"], applied["max_output_tokens"], applied["max_tokens"]) == (163840, 65536, 65536)
assert outcome.cost_map["openrouter/acme/genuine"] == genuine
assert outcome.cost_map["openrouter/acme/glitch"] == copied
assert outcome.providers[0].updated == (
"openrouter/acme/copied: max_output_tokens: 163840 -> 65536; max_tokens: 163840 -> 65536",
)
assert outcome.providers[0].warnings == (
"openrouter/acme/genuine: max_output_tokens: 100000 -> 65536 held back: a shrinking limit; "
"max_tokens: 100000 -> 65536 held back: a shrinking limit",
"openrouter/acme/glitch: max_output_tokens: 163840 -> 8192 held back: a copied output cap shrinking more "
"than 10x; max_tokens: 163840 -> 8192 held back: a copied output cap shrinking more than 10x",
)
@ -704,7 +717,8 @@ def test_a_price_that_varies_by_provider_seeds_and_only_overwrites_what_the_bot_
"output_cost_per_token": 2e-6,
"max_input_tokens": 100000,
}
seeded: Final = {**curated, "source": "https://vercel.com/ai-gateway/models/seeded"}
seeded: Final = {**curated, "price_varies_by_provider": True}
cited: Final = {**curated, "source": "https://vercel.com/ai-gateway/models/cited"}
row: Final = {
"context_window": 262144,
"pricing": {
@ -716,27 +730,44 @@ def test_a_price_that_varies_by_provider_seeds_and_only_overwrites_what_the_bot_
},
}
catalog: Final = sync.load_vercel(
_vercel_rows({"id": "acme/curated", **row}, {"id": "acme/seeded", **row}, {"id": "acme/fresh", **row}),
_vercel_rows(
{"id": "acme/curated", **row},
{"id": "acme/seeded", **row},
{"id": "acme/fresh", **row},
{"id": "acme/cited", **row},
),
now_ms=NOW_MS,
)
outcome: Final = sync.compute_sync(
{"vercel_ai_gateway/acme/curated": dict(curated), "vercel_ai_gateway/acme/seeded": dict(seeded)}, (catalog,)
{
"vercel_ai_gateway/acme/curated": dict(curated),
"vercel_ai_gateway/acme/seeded": dict(seeded),
"vercel_ai_gateway/acme/cited": dict(cited),
},
(catalog,),
)
existing: Final = outcome.cost_map["vercel_ai_gateway/acme/curated"]
assert (existing["input_cost_per_token"], existing["max_input_tokens"]) == (9e-7, 262144)
assert not any("cache_read" in name or "_above_" in name for name in existing)
for key in ("vercel_ai_gateway/acme/curated", "vercel_ai_gateway/acme/cited"):
held: Final = outcome.cost_map[key]
assert (held["input_cost_per_token"], held["max_input_tokens"]) == (9e-7, 262144)
assert not any("cache_read" in name or "_above_" in name for name in held)
assert "price_varies_by_provider" not in held
resynced: Final = outcome.cost_map["vercel_ai_gateway/acme/seeded"]
assert (resynced["input_cost_per_token"], resynced["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6)
fresh: Final = outcome.cost_map["vercel_ai_gateway/acme/fresh"]
assert (fresh["input_cost_per_token"], fresh["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6)
assert fresh["cache_read_input_token_cost"] == 3e-7
assert outcome.providers[0].warnings == (
"vercel_ai_gateway/acme/curated: input_cost_per_token: 9e-07 -> 1.5e-06 held back: "
assert fresh["price_varies_by_provider"] is True
held_line: Final = (
": input_cost_per_token: 9e-07 -> 1.5e-06 held back: "
"the catalog price varies by provider; cache_read_input_token_cost: None -> 3e-07 held back: "
"the catalog price varies by provider; input_cost_per_token_above_128k_tokens: None -> 3e-06 held back: "
"the catalog price varies by provider",
"the catalog price varies by provider"
)
assert outcome.providers[0].warnings == (
f"vercel_ai_gateway/acme/cited{held_line}",
f"vercel_ai_gateway/acme/curated{held_line}",
)

View file

@ -1046,6 +1046,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"output_vector_size": {"type": "number"},
"rpd": {"type": "number"},
"rpm": {"type": "number"},
"price_varies_by_provider": {"type": "boolean"},
"source": {"type": "string"},
"comment": {"type": "string"},
"supports_assistant_prefill": {"type": "boolean"},