mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(cost-map-sync): mark bot-seeded headline prices and hold glitchy copied caps
A Vercel row seeded at a varies_by_provider headline price now carries price_varies_by_provider: true and only a row with that mark keeps following the catalog; a curated row citing the catalog page as source holds its price like any other. A copied output cap still yields to the catalog ceiling, but a ceiling more than 10x below it is held as a catalog glitch. The map schema test learns the new key.
This commit is contained in:
parent
412db8dad3
commit
fb8f9af5a1
3 changed files with 64 additions and 21 deletions
|
|
@ -9,9 +9,9 @@ Policy:
|
|||
- Both catalogs price per token as decimal strings; values are normalized to six significant digits.
|
||||
- Vercel long-context tiers map to the registry's ``*_above_<N>k_tokens`` keys, which litellm applies once the
|
||||
prompt exceeds N thousand tokens. A row whose tier boundaries are not whole thousands is skipped with a warning.
|
||||
- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry and keeps that entry in
|
||||
sync (its ``source`` is the catalog page), but never overwrites a curated price; that difference is reported as
|
||||
a warning.
|
||||
- A Vercel price flagged ``varies_by_provider`` is only a headline: it seeds a new entry, marked
|
||||
``price_varies_by_provider``, and keeps a marked entry in sync, but never overwrites the price of an entry without
|
||||
the mark; that difference is reported as a warning. A human who curates one provider's own price drops the mark.
|
||||
- Image and audio output are priced from the catalog's per-token ``image_output`` and ``audio_output`` prices. A row
|
||||
whose non-text output the catalog does not price per token is skipped.
|
||||
- A new entry inherits the traits no catalog expresses (adaptive thinking, sampling params, cache minimums, system
|
||||
|
|
@ -20,7 +20,8 @@ Policy:
|
|||
- An existing entry only gains or changes the fields the catalog expresses. Nothing is ever removed and a
|
||||
capability flag the catalog does not claim stays as curated. ``max_output_tokens`` and ``max_tokens`` move as a
|
||||
pair and only when the catalog states an output ceiling; a curated output cap equal to the entry's own context
|
||||
window is a copy of that window, not a ceiling, so the catalog's ceiling replaces it.
|
||||
window is a copy of that window, not a ceiling, so the catalog's ceiling replaces it unless that would shrink it
|
||||
more than 10x.
|
||||
- A limit that would shrink, a price that would cross zero, a price that would move more than 10x either way, and
|
||||
a capability flag curated as false are held back as warnings for a human instead of applied.
|
||||
- Router models and rows without a usable prompt and completion price are skipped.
|
||||
|
|
@ -51,7 +52,7 @@ OPENROUTER_MODELS_URL: Final = "https://openrouter.ai/api/v1/models"
|
|||
VERCEL_MODELS_URL: Final = "https://ai-gateway.vercel.sh/v1/models"
|
||||
VERCEL_TYPE_TO_MODE: Final = MappingProxyType({"language": "chat", "embedding": "embedding"})
|
||||
LIMIT_PAIR: Final = ("max_output_tokens", "max_tokens")
|
||||
PRICE_SWING_LIMIT: Final = 10
|
||||
SWING_LIMIT: Final = 10
|
||||
INHERITED_TRAITS: Final = frozenset(
|
||||
{
|
||||
"prompt_cache_min_tokens",
|
||||
|
|
@ -456,6 +457,7 @@ def _new_entry(entry: CatalogEntry, inherited: Mapping[str, object]) -> Registry
|
|||
"litellm_provider": entry.provider,
|
||||
"mode": entry.mode,
|
||||
"source": entry.source,
|
||||
**({"price_varies_by_provider": True} if entry.indicative_prices else {}),
|
||||
}.items()
|
||||
)
|
||||
)
|
||||
|
|
@ -477,8 +479,14 @@ class FieldChange:
|
|||
def _swing(old: float, new: float) -> str | None:
|
||||
if (old == 0) != (new == 0):
|
||||
return "a price crossing zero"
|
||||
if old and new and max(new / old, old / new) > PRICE_SWING_LIMIT:
|
||||
return f"a price moving more than {PRICE_SWING_LIMIT}x"
|
||||
if old and new and max(new / old, old / new) > SWING_LIMIT:
|
||||
return f"a price moving more than {SWING_LIMIT}x"
|
||||
return None
|
||||
|
||||
|
||||
def _copied_cap_hold(current: object, ceiling: object) -> str | None:
|
||||
if isinstance(current, int) and isinstance(ceiling, int) and ceiling * SWING_LIMIT < current:
|
||||
return f"a copied output cap shrinking more than {SWING_LIMIT}x"
|
||||
return None
|
||||
|
||||
|
||||
|
|
@ -498,7 +506,7 @@ def _hold(name: str, old: object, new: object, curated_prices_win: bool) -> str
|
|||
|
||||
def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange, ...]:
|
||||
curated_prices_win: Final = (
|
||||
entry.indicative_prices and "input_cost_per_token" in existing and existing.get("source") != entry.source
|
||||
entry.indicative_prices and "input_cost_per_token" in existing and not existing.get("price_varies_by_provider")
|
||||
)
|
||||
scalars: Final = tuple(
|
||||
FieldChange(name, existing.get(name), value, _hold(name, existing.get(name), value, curated_prices_win))
|
||||
|
|
@ -509,8 +517,11 @@ def _changes(existing: RegistryEntry, entry: CatalogEntry) -> tuple[FieldChange,
|
|||
if ceiling is None:
|
||||
return scalars
|
||||
current: Final = existing.get("max_output_tokens", existing.get("max_tokens"))
|
||||
curated_cap: Final = None if current == existing.get("max_input_tokens") else current
|
||||
hold: Final = _hold("max_output_tokens", curated_cap, ceiling, curated_prices_win)
|
||||
hold: Final = (
|
||||
_copied_cap_hold(current, ceiling)
|
||||
if current == existing.get("max_input_tokens")
|
||||
else _hold("max_output_tokens", current, ceiling, curated_prices_win)
|
||||
)
|
||||
return (
|
||||
*scalars,
|
||||
*(FieldChange(name, existing.get(name), ceiling, hold) for name in LIMIT_PAIR if existing.get(name) != ceiling),
|
||||
|
|
|
|||
|
|
@ -223,21 +223,34 @@ def test_an_output_cap_copied_from_the_context_window_yields_to_the_catalog_ceil
|
|||
}
|
||||
genuine: Final = {**copied, "max_output_tokens": 100000, "max_tokens": 100000}
|
||||
row: Final = {"context_length": 163840, "top_provider": {"max_completion_tokens": 65536}}
|
||||
catalog: Final = sync.load_openrouter(_openrouter_rows({"id": "acme/copied", **row}, {"id": "acme/genuine", **row}))
|
||||
glitch_row: Final = {"context_length": 163840, "top_provider": {"max_completion_tokens": 8192}}
|
||||
catalog: Final = sync.load_openrouter(
|
||||
_openrouter_rows(
|
||||
{"id": "acme/copied", **row}, {"id": "acme/genuine", **row}, {"id": "acme/glitch", **glitch_row}
|
||||
)
|
||||
)
|
||||
|
||||
outcome: Final = sync.compute_sync(
|
||||
{"openrouter/acme/copied": dict(copied), "openrouter/acme/genuine": dict(genuine)}, (catalog,)
|
||||
{
|
||||
"openrouter/acme/copied": dict(copied),
|
||||
"openrouter/acme/genuine": dict(genuine),
|
||||
"openrouter/acme/glitch": dict(copied),
|
||||
},
|
||||
(catalog,),
|
||||
)
|
||||
|
||||
applied: Final = outcome.cost_map["openrouter/acme/copied"]
|
||||
assert (applied["max_input_tokens"], applied["max_output_tokens"], applied["max_tokens"]) == (163840, 65536, 65536)
|
||||
assert outcome.cost_map["openrouter/acme/genuine"] == genuine
|
||||
assert outcome.cost_map["openrouter/acme/glitch"] == copied
|
||||
assert outcome.providers[0].updated == (
|
||||
"openrouter/acme/copied: max_output_tokens: 163840 -> 65536; max_tokens: 163840 -> 65536",
|
||||
)
|
||||
assert outcome.providers[0].warnings == (
|
||||
"openrouter/acme/genuine: max_output_tokens: 100000 -> 65536 held back: a shrinking limit; "
|
||||
"max_tokens: 100000 -> 65536 held back: a shrinking limit",
|
||||
"openrouter/acme/glitch: max_output_tokens: 163840 -> 8192 held back: a copied output cap shrinking more "
|
||||
"than 10x; max_tokens: 163840 -> 8192 held back: a copied output cap shrinking more than 10x",
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -704,7 +717,8 @@ def test_a_price_that_varies_by_provider_seeds_and_only_overwrites_what_the_bot_
|
|||
"output_cost_per_token": 2e-6,
|
||||
"max_input_tokens": 100000,
|
||||
}
|
||||
seeded: Final = {**curated, "source": "https://vercel.com/ai-gateway/models/seeded"}
|
||||
seeded: Final = {**curated, "price_varies_by_provider": True}
|
||||
cited: Final = {**curated, "source": "https://vercel.com/ai-gateway/models/cited"}
|
||||
row: Final = {
|
||||
"context_window": 262144,
|
||||
"pricing": {
|
||||
|
|
@ -716,27 +730,44 @@ def test_a_price_that_varies_by_provider_seeds_and_only_overwrites_what_the_bot_
|
|||
},
|
||||
}
|
||||
catalog: Final = sync.load_vercel(
|
||||
_vercel_rows({"id": "acme/curated", **row}, {"id": "acme/seeded", **row}, {"id": "acme/fresh", **row}),
|
||||
_vercel_rows(
|
||||
{"id": "acme/curated", **row},
|
||||
{"id": "acme/seeded", **row},
|
||||
{"id": "acme/fresh", **row},
|
||||
{"id": "acme/cited", **row},
|
||||
),
|
||||
now_ms=NOW_MS,
|
||||
)
|
||||
|
||||
outcome: Final = sync.compute_sync(
|
||||
{"vercel_ai_gateway/acme/curated": dict(curated), "vercel_ai_gateway/acme/seeded": dict(seeded)}, (catalog,)
|
||||
{
|
||||
"vercel_ai_gateway/acme/curated": dict(curated),
|
||||
"vercel_ai_gateway/acme/seeded": dict(seeded),
|
||||
"vercel_ai_gateway/acme/cited": dict(cited),
|
||||
},
|
||||
(catalog,),
|
||||
)
|
||||
|
||||
existing: Final = outcome.cost_map["vercel_ai_gateway/acme/curated"]
|
||||
assert (existing["input_cost_per_token"], existing["max_input_tokens"]) == (9e-7, 262144)
|
||||
assert not any("cache_read" in name or "_above_" in name for name in existing)
|
||||
for key in ("vercel_ai_gateway/acme/curated", "vercel_ai_gateway/acme/cited"):
|
||||
held: Final = outcome.cost_map[key]
|
||||
assert (held["input_cost_per_token"], held["max_input_tokens"]) == (9e-7, 262144)
|
||||
assert not any("cache_read" in name or "_above_" in name for name in held)
|
||||
assert "price_varies_by_provider" not in held
|
||||
resynced: Final = outcome.cost_map["vercel_ai_gateway/acme/seeded"]
|
||||
assert (resynced["input_cost_per_token"], resynced["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6)
|
||||
fresh: Final = outcome.cost_map["vercel_ai_gateway/acme/fresh"]
|
||||
assert (fresh["input_cost_per_token"], fresh["input_cost_per_token_above_128k_tokens"]) == (1.5e-6, 3e-6)
|
||||
assert fresh["cache_read_input_token_cost"] == 3e-7
|
||||
assert outcome.providers[0].warnings == (
|
||||
"vercel_ai_gateway/acme/curated: input_cost_per_token: 9e-07 -> 1.5e-06 held back: "
|
||||
assert fresh["price_varies_by_provider"] is True
|
||||
held_line: Final = (
|
||||
": input_cost_per_token: 9e-07 -> 1.5e-06 held back: "
|
||||
"the catalog price varies by provider; cache_read_input_token_cost: None -> 3e-07 held back: "
|
||||
"the catalog price varies by provider; input_cost_per_token_above_128k_tokens: None -> 3e-06 held back: "
|
||||
"the catalog price varies by provider",
|
||||
"the catalog price varies by provider"
|
||||
)
|
||||
assert outcome.providers[0].warnings == (
|
||||
f"vercel_ai_gateway/acme/cited{held_line}",
|
||||
f"vercel_ai_gateway/acme/curated{held_line}",
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1046,6 +1046,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"output_vector_size": {"type": "number"},
|
||||
"rpd": {"type": "number"},
|
||||
"rpm": {"type": "number"},
|
||||
"price_varies_by_provider": {"type": "boolean"},
|
||||
"source": {"type": "string"},
|
||||
"comment": {"type": "string"},
|
||||
"supports_assistant_prefill": {"type": "boolean"},
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue