From fb8f9af5a1124884b47271f9ada2dab1b42e2cb6 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 5 Sep 2026 03:05:59 -0700 Subject: [PATCH] fix(cost-map-sync): mark bot-seeded headline prices and hold glitchy copied caps A Vercel row seeded at a varies_by_provider headline price now carries price_varies_by_provider: true and only a row with that mark keeps following the catalog; a curated row citing the catalog page as source holds its price like any other. A copied output cap still yields to the catalog ceiling, but a ceiling more than 10x below it is held as a catalog glitch. The map schema test learns the new key. --- scripts/sync_cost_map.py | 31 +++++++++----- tests/test_litellm/test_sync_cost_map.py | 53 +++++++++++++++++++----- tests/test_litellm/test_utils.py | 1 + 3 files changed, 64 insertions(+), 21 deletions(-) diff --git a/scripts/sync_cost_map.py b/scripts/sync_cost_map.py index d4433dcfa8c..bda89921a0a 100644 --- a/scripts/sync_cost_map.py +++ b/scripts/sync_cost_map.py @@ -9,9 +9,9 @@ Policy: - Both catalogs price per token as decimal strings; values are normalized to six significant digits. - Vercel long-context tiers map to the registry's ``*_above_k_tokens`` keys, which litellm applies once the prompt exceeds N thousand tokens. A row whose tier boundaries are not whole thousands is skipped with a warning. -- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry and keeps that entry in - sync (its ``source`` is the catalog page), but never overwrites a curated price; that difference is reported as - a warning. +- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry, marked + ``price_varies_by_provider``, and keeps a marked entry in sync, but never overwrites the price of an entry without + the mark; that difference is reported as a warning. A human who curates one provider's own price drops the mark. - Image and audio output are priced from the catalog's per-token ``image_output`` and ``audio_output`` prices. A row whose non-text output the catalog does not price per token is skipped. - A new entry inherits the traits no catalog expresses (adaptive thinking, sampling params, cache minimums, system @@ -20,7 +20,8 @@ Policy: - An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed and a capability flag the catalog does not claim stays as curated. ``max_output_tokens`` and ``max_tokens`` move as a pair and only when the catalog states an output ceiling; a curated output cap equal to the entry's own context - window is a copy of that window, not a ceiling, so the catalog's ceiling replaces it. + window is a copy of that window, not a ceiling, so the catalog's ceiling replaces it unless that would shrink it + more than 10x. - A limit that would shrink, a price that would cross zero, a price that would move more than 10x either way, and a capability flag curated as false are held back as warnings for a human instead of applied. - Router models and rows without a usable prompt and completion price are skipped. @@ -51,7 +52,7 @@ OPENROUTER_MODELS_URL: Final = "https://openrouter.ai/api/v1/models" VERCEL_MODELS_URL: Final = "https://ai-gateway.vercel.sh/v1/models" VERCEL_TYPE_TO_MODE: Final = MappingProxyType({"language": "chat", "embedding": "embedding"}) LIMIT_PAIR: Final = ("max_output_tokens", "max_tokens") -PRICE_SWING_LIMIT: Final = 10 +SWING_LIMIT: Final = 10 INHERITED_TRAITS: Final = frozenset( { "prompt_cache_min_tokens", @@ -456,6 +457,7 @@ def _new_entry(entry: CatalogEntry, inherited: Mapping[str, object]) -> Registry "litellm_provider": entry.provider, "mode": entry.mode, "source": entry.source, + **({"price_varies_by_provider": True} if entry.indicative_prices else {}), }.items() ) ) @@ -477,8 +479,14 @@ class FieldChange: def _swing(old: float, new: float) -> str | None: if (old == 0) != (new == 0): return "a price crossing zero" - if old and new and max(new / old, old / new) > PRICE_SWING_LIMIT: - return f"a price moving more than {PRICE_SWING_LIMIT}x" + if old and new and max(new / old, old / new) > SWING_LIMIT: + return f"a price moving more than {SWING_LIMIT}x" + return None + + +def _copied_cap_hold(current: object, ceiling: object) -> str | None: + if isinstance(current, int) and isinstance(ceiling, int) and ceiling * SWING_LIMIT < current: + return f"a copied output cap shrinking more than {SWING_LIMIT}x" return None @@ -498,7 +506,7 @@ def _hold(name: str, old: object, new: object, curated_prices_win: bool) -> str def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange, ...]: curated_prices_win: Final = ( - entry.indicative_prices and "input_cost_per_token" in existing and existing.get("source") != entry.source + entry.indicative_prices and "input_cost_per_token" in existing and not existing.get("price_varies_by_provider") ) scalars: Final = tuple( FieldChange(name, existing.get(name), value, _hold(name, existing.get(name), value, curated_prices_win)) @@ -509,8 +517,11 @@ def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange, if ceiling is None: return scalars current: Final = existing.get("max_output_tokens", existing.get("max_tokens")) - curated_cap: Final = None if current == existing.get("max_input_tokens") else current - hold: Final = _hold("max_output_tokens", curated_cap, ceiling, curated_prices_win) + hold: Final = ( + _copied_cap_hold(current, ceiling) + if current == existing.get("max_input_tokens") + else _hold("max_output_tokens", current, ceiling, curated_prices_win) + ) return ( *scalars, *(FieldChange(name, existing.get(name), ceiling, hold) for name in LIMIT_PAIR if existing.get(name) != ceiling), diff --git a/tests/test_litellm/test_sync_cost_map.py b/tests/test_litellm/test_sync_cost_map.py index 238ee05ac7c..a7dafc4888b 100644 --- a/tests/test_litellm/test_sync_cost_map.py +++ b/tests/test_litellm/test_sync_cost_map.py @@ -223,21 +223,34 @@ def test_an_output_cap_copied_from_the_context_window_yields_to_the_catalog_ceil } genuine: Final = {**copied, "max_output_tokens": 100000, "max_tokens": 100000} row: Final = {"context_length": 163840, "top_provider": {"max_completion_tokens": 65536}} - catalog: Final = sync.load_openrouter(_openrouter_rows({"id": "acme/copied", **row}, {"id": "acme/genuine", **row})) + glitch_row: Final = {"context_length": 163840, "top_provider": {"max_completion_tokens": 8192}} + catalog: Final = sync.load_openrouter( + _openrouter_rows( + {"id": "acme/copied", **row}, {"id": "acme/genuine", **row}, {"id": "acme/glitch", **glitch_row} + ) + ) outcome: Final = sync.compute_sync( - {"openrouter/acme/copied": dict(copied), "openrouter/acme/genuine": dict(genuine)}, (catalog,) + { + "openrouter/acme/copied": dict(copied), + "openrouter/acme/genuine": dict(genuine), + "openrouter/acme/glitch": dict(copied), + }, + (catalog,), ) applied: Final = outcome.cost_map["openrouter/acme/copied"] assert (applied["max_input_tokens"], applied["max_output_tokens"], applied["max_tokens"]) == (163840, 65536, 65536) assert outcome.cost_map["openrouter/acme/genuine"] == genuine + assert outcome.cost_map["openrouter/acme/glitch"] == copied assert outcome.providers[0].updated == ( "openrouter/acme/copied: max_output_tokens: 163840 -> 65536; max_tokens: 163840 -> 65536", ) assert outcome.providers[0].warnings == ( "openrouter/acme/genuine: max_output_tokens: 100000 -> 65536 held back: a shrinking limit; " "max_tokens: 100000 -> 65536 held back: a shrinking limit", + "openrouter/acme/glitch: max_output_tokens: 163840 -> 8192 held back: a copied output cap shrinking more " + "than 10x; max_tokens: 163840 -> 8192 held back: a copied output cap shrinking more than 10x", ) @@ -704,7 +717,8 @@ def test_a_price_that_varies_by_provider_seeds_and_only_overwrites_what_the_bot_ "output_cost_per_token": 2e-6, "max_input_tokens": 100000, } - seeded: Final = {**curated, "source": "https://vercel.com/ai-gateway/models/seeded"} + seeded: Final = {**curated, "price_varies_by_provider": True} + cited: Final = {**curated, "source": "https://vercel.com/ai-gateway/models/cited"} row: Final = { "context_window": 262144, "pricing": { @@ -716,27 +730,44 @@ def test_a_price_that_varies_by_provider_seeds_and_only_overwrites_what_the_bot_ }, } catalog: Final = sync.load_vercel( - _vercel_rows({"id": "acme/curated", **row}, {"id": "acme/seeded", **row}, {"id": "acme/fresh", **row}), + _vercel_rows( + {"id": "acme/curated", **row}, + {"id": "acme/seeded", **row}, + {"id": "acme/fresh", **row}, + {"id": "acme/cited", **row}, + ), now_ms=NOW_MS, ) outcome: Final = sync.compute_sync( - {"vercel_ai_gateway/acme/curated": dict(curated), "vercel_ai_gateway/acme/seeded": dict(seeded)}, (catalog,) + { + "vercel_ai_gateway/acme/curated": dict(curated), + "vercel_ai_gateway/acme/seeded": dict(seeded), + "vercel_ai_gateway/acme/cited": dict(cited), + }, + (catalog,), ) - existing: Final = outcome.cost_map["vercel_ai_gateway/acme/curated"] - assert (existing["input_cost_per_token"], existing["max_input_tokens"]) == (9e-7, 262144) - assert not any("cache_read" in name or "_above_" in name for name in existing) + for key in ("vercel_ai_gateway/acme/curated", "vercel_ai_gateway/acme/cited"): + held: Final = outcome.cost_map[key] + assert (held["input_cost_per_token"], held["max_input_tokens"]) == (9e-7, 262144) + assert not any("cache_read" in name or "_above_" in name for name in held) + assert "price_varies_by_provider" not in held resynced: Final = outcome.cost_map["vercel_ai_gateway/acme/seeded"] assert (resynced["input_cost_per_token"], resynced["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6) fresh: Final = outcome.cost_map["vercel_ai_gateway/acme/fresh"] assert (fresh["input_cost_per_token"], fresh["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6) assert fresh["cache_read_input_token_cost"] == 3e-7 - assert outcome.providers[0].warnings == ( - "vercel_ai_gateway/acme/curated: input_cost_per_token: 9e-07 -> 1.5e-06 held back: " + assert fresh["price_varies_by_provider"] is True + held_line: Final = ( + ": input_cost_per_token: 9e-07 -> 1.5e-06 held back: " "the catalog price varies by provider; cache_read_input_token_cost: None -> 3e-07 held back: " "the catalog price varies by provider; input_cost_per_token_above_128k_tokens: None -> 3e-06 held back: " - "the catalog price varies by provider", + "the catalog price varies by provider" + ) + assert outcome.providers[0].warnings == ( + f"vercel_ai_gateway/acme/cited{held_line}", + f"vercel_ai_gateway/acme/curated{held_line}", ) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 5a6aa20657d..fdcdffece34 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -1046,6 +1046,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "output_vector_size": {"type": "number"}, "rpd": {"type": "number"}, "rpm": {"type": "number"}, + "price_varies_by_provider": {"type": "boolean"}, "source": {"type": "string"}, "comment": {"type": "string"}, "supports_assistant_prefill": {"type": "boolean"},