feat(cost_map): derive source_revision from the loaded bytes instead of a _metadata stamp

The revision an operator checks is now the git blob id of the exact bytes the process
loaded, the same id git rev-parse <commit>:model_prices_and_context_window.json prints,
so it is always present, never goes stale between bot writes, and needs no stamp in the
JSON that every PR touching the file would have to regenerate. The _metadata block, the
generated_at field, the schema and guard changes, and the bot stamping are dropped
This commit is contained in:
mateo-berri 2026-09-07 17:47:51 -07:00
parent aa1c76bc3b
commit 9041768fb4
19 changed files with 131 additions and 420 deletions

View file

@ -1,9 +1,6 @@
import asyncio
import aiohttp
import json
import os
import subprocess
from datetime import datetime, timezone
# Asynchronously fetch data from a given URL
async def fetch_data(url):
@ -34,28 +31,13 @@ def sync_local_data_with_remote(local_data, remote_data):
for key in (set(remote_data) - set(local_data)):
local_data[key] = remote_data[key]
def utc_now_iso():
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
def source_revision():
from_env = os.environ.get("GITHUB_SHA")
if from_env:
return from_env
return subprocess.run(["git", "rev-parse", "HEAD"], check=True, capture_output=True, text=True).stdout.strip()
def stamp_metadata(data, generated_at, revision):
return {**data, "_metadata": {"generated_at": generated_at, "source_revision": revision}}
# Write data to the json file
def write_to_file(file_path, data):
try:
# Open the file in write mode
with open(file_path, "w") as file:
# Dump the data as JSON into the file
file.write(json.dumps(data, indent=4) + "\n")
json.dump(data, file, indent=4)
print("Values updated successfully.")
except Exception as e:
# Print an error message if writing to file fails
@ -167,13 +149,8 @@ def main():
# If both local and openrouter data are available, synchronize and save
if local_data and all_remote_data:
before = json.dumps(local_data, sort_keys=True)
sync_local_data_with_remote(local_data, all_remote_data)
changed = json.dumps(local_data, sort_keys=True) != before
write_to_file(
local_file_path,
stamp_metadata(local_data, utc_now_iso(), source_revision()) if changed else local_data,
)
write_to_file(local_file_path, local_data)
else:
print("Failed to fetch model data from either local file or URL.")

View file

@ -2,8 +2,7 @@
Every pull request gets the file checks: the three cost map files parse, the backup copy matches the root file,
and the JSON schema is in sync and validates the map. Pull requests from the cost map sync bot (branches named
litellm_cost_map_sync_*) additionally may only touch those three files and may only add or update models, plus
restamp the _metadata provenance block.
litellm_cost_map_sync_*) additionally may only touch those three files and may only add or update models.
"""
from __future__ import annotations
@ -16,7 +15,7 @@ from collections.abc import Sequence
from dataclasses import dataclass
from typing import Final
from generate_model_prices_schema import BOT_LOCKED_ROOT_KEYS, build_schema, render, validation_errors
from generate_model_prices_schema import SPECIAL_ROOT_KEYS, build_schema, render, validation_errors
COST_MAP_PATH: Final = "model_prices_and_context_window.json"
BACKUP_PATH: Final = "litellm/model_prices_and_context_window_backup.json"
@ -103,7 +102,7 @@ def _bot_failures(base: Snapshot, head_map: CostMap, changed_files: Sequence[str
*(f"bot PRs may not remove fields: {ref}" for ref in removed_fields),
*(
f"bot PRs may not change {key}"
for key in sorted(BOT_LOCKED_ROOT_KEYS)
for key in sorted(SPECIAL_ROOT_KEYS)
if base_map.get(key) != head_map.get(key)
),
)

View file

@ -11,9 +11,7 @@ REPO_ROOT = Path(__file__).parent.parent
PRICES_PATH = REPO_ROOT / "model_prices_and_context_window.json"
SCHEMA_PATH = REPO_ROOT / "model_prices_and_context_window.schema.json"
METADATA_KEY = "_metadata"
SPECIAL_ROOT_KEYS = frozenset({"sample_spec", "fallback_generalizations", METADATA_KEY})
BOT_LOCKED_ROOT_KEYS = SPECIAL_ROOT_KEYS - {METADATA_KEY}
SPECIAL_ROOT_KEYS = frozenset({"sample_spec", "fallback_generalizations"})
JsonSchema = dict
@ -273,26 +271,13 @@ def build_schema(prices: dict) -> JsonSchema:
"description": (
"Schema for LiteLLM's model price and context window registry "
"(https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json). "
"Every top-level key except '_metadata', 'sample_spec', and 'fallback_generalizations' is a model id, "
"Every top-level key except 'sample_spec' and 'fallback_generalizations' is a model id, "
"optionally prefixed with its provider (e.g. 'azure/gpt-5.4'), mapping to a model entry. "
"All costs are USD per unit. New optional fields are added regularly, so consumers should "
"ignore unknown fields rather than reject them."
),
"type": "object",
"properties": {
METADATA_KEY: {
"type": "object",
"description": (
"Provenance of this file: when an automated sync last regenerated it and the commit it "
"ran against. Human edits leave it untouched; not a model entry."
),
"properties": {
"generated_at": {"type": "string", "format": "date-time"},
"source_revision": STRING,
},
"required": ["generated_at", "source_revision"],
"additionalProperties": False,
},
"sample_spec": {
"type": "object",
"description": (

View file

@ -9,18 +9,18 @@ export LITELLM_LOCAL_MODEL_COST_MAP=True
"""
import asyncio
import hashlib
import json
import os
import random
import time
from collections.abc import Awaitable, Callable
from dataclasses import dataclass
from dataclasses import dataclass, replace
from datetime import datetime, timezone
from importlib.resources import files
from typing import Final, Protocol
import httpx
from pydantic import BaseModel, ConfigDict, ValidationError
from typing_extensions import ReadOnly, TypedDict
from litellm import verbose_logger
@ -33,11 +33,10 @@ from litellm.litellm_core_utils.fallback_generalizations import (
)
FALLBACK_GENERALIZATIONS_KEY: Final = "fallback_generalizations"
METADATA_KEY: Final = "_metadata"
# Reserved top-level keys that are not model entries. They must be excluded
# from the model-count integrity check so a real upstream shrink can't be masked.
RESERVED_TOP_LEVEL_KEYS: Final = frozenset({"sample_spec", FALLBACK_GENERALIZATIONS_KEY, METADATA_KEY})
RESERVED_TOP_LEVEL_KEYS: Final = frozenset({"sample_spec", FALLBACK_GENERALIZATIONS_KEY})
def _count_model_entries(model_cost: dict) -> int:
@ -45,6 +44,11 @@ def _count_model_entries(model_cost: dict) -> int:
return sum(1 for key in model_cost if key not in RESERVED_TOP_LEVEL_KEYS)
def git_blob_id(body: bytes) -> str:
"""The sha1 git gives these bytes as a blob, so ``git rev-parse <commit>:<path>`` reproduces it for the file"""
return hashlib.sha1(b"blob %d\0" % len(body) + body, usedforsecurity=False).hexdigest()
class GetModelCostMap:
"""
Handles fetching, validating, and loading the model cost map.
@ -56,15 +60,25 @@ class GetModelCostMap:
_backup_model_count: int = -1 # -1 = not yet loaded
@staticmethod
def read_local_model_cost_map_bytes() -> bytes:
return files("litellm").joinpath("model_prices_and_context_window_backup.json").read_bytes()
@staticmethod
def read_local_model_cost_map_text() -> str:
return files("litellm").joinpath("model_prices_and_context_window_backup.json").read_text(encoding="utf-8")
return GetModelCostMap.read_local_model_cost_map_bytes().decode("utf-8")
@staticmethod
def load_local_model_cost_map_with_revision() -> "ModelCostMapReloaded":
"""The bundled backup map together with the git blob id of the file it was parsed from"""
body: Final = GetModelCostMap.read_local_model_cost_map_bytes()
content: Final = json.loads(body)
return ModelCostMapReloaded(model_cost_map=content, revision=git_blob_id(body))
@staticmethod
def load_local_model_cost_map() -> dict:
"""Load the local backup model cost map bundled with the package."""
content: Final = json.loads(GetModelCostMap.read_local_model_cost_map_text())
return content
return GetModelCostMap.load_local_model_cost_map_with_revision().model_cost_map
@classmethod
def _get_backup_model_count(cls) -> int:
@ -169,6 +183,7 @@ MODEL_COST_MAP_FETCH_MAX_WAIT_SECONDS: Final = 30.0
@dataclass(frozen=True, slots=True)
class ModelCostMapReloaded:
model_cost_map: dict # mutable-ok: adopted as litellm.model_cost, whose consumer contract is a plain mutable dict
revision: str | None = None
etag: str | None = None
@ -258,7 +273,9 @@ def _classify_fetch_response(response: httpx.Response, url: str) -> _FetchAttemp
return ModelCostMapReloadUnavailable(reason=f"invalid JSON from {url}: {e}")
if not isinstance(parsed, dict):
return ModelCostMapReloadUnavailable(reason=f"expected a JSON object from {url}, got {type(parsed).__name__}")
return ModelCostMapReloaded(model_cost_map=parsed, etag=response.headers.get("etag"))
return ModelCostMapReloaded(
model_cost_map=parsed, revision=git_blob_id(response.content), etag=response.headers.get("etag")
)
def _next_retry_wait(
@ -337,10 +354,7 @@ async def refetch_model_cost_map(
_cost_map_source_info.url = None
_cost_map_source_info.is_env_forced = True
_cost_map_source_info.fallback_reason = None
_cost_map_source_info.etag = None
return ModelCostMapReloaded(
model_cost_map=_finalize_model_cost_map(GetModelCostMap.load_local_model_cost_map())
)
return _finalize_loaded_model_cost_map(GetModelCostMap.load_local_model_cost_map_with_revision())
result: Final = await _fetch_remote_model_cost_map_with_retry(
url=url,
@ -366,8 +380,7 @@ async def refetch_model_cost_map(
_cost_map_source_info.url = url
_cost_map_source_info.is_env_forced = False
_cost_map_source_info.fallback_reason = None
_cost_map_source_info.etag = result.etag
return ModelCostMapReloaded(model_cost_map=_finalize_model_cost_map(result.model_cost_map), etag=result.etag)
return _finalize_loaded_model_cost_map(result)
class ModelCostMapSourceInfo:
@ -378,7 +391,6 @@ class ModelCostMapSourceInfo:
is_env_forced: bool = False
fallback_reason: str | None = None
loaded_at: "datetime | None" = None
generated_at: str | None = None
source_revision: str | None = None
etag: str | None = None
@ -387,28 +399,7 @@ class ModelCostMapSourceInfo:
_cost_map_source_info: Final = ModelCostMapSourceInfo()
class CostMapMetadata(BaseModel):
model_config = ConfigDict(frozen=True, extra="ignore")
generated_at: str | None = None
source_revision: str | None = None
_EMPTY_METADATA: Final = CostMapMetadata()
def _parse_metadata(raw: object) -> CostMapMetadata:
if raw is None:
return _EMPTY_METADATA
try:
return CostMapMetadata.model_validate(raw)
except ValidationError as error:
verbose_logger.warning("LiteLLM: ignoring a malformed %s block in the model cost map: %s", METADATA_KEY, error)
return _EMPTY_METADATA
class CostMapProvenance(TypedDict):
generated_at: ReadOnly[str | None]
source_revision: ReadOnly[str | None]
etag: ReadOnly[str | None]
@ -422,10 +413,10 @@ class CostMapSourceInfo(CostMapProvenance):
def get_model_cost_map_provenance() -> CostMapProvenance:
"""Which revision of the cost map this process serves: the ``_metadata`` stamp the file
carries plus the ETag the remote fetch returned (None for the bundled backup)"""
"""Which revision of the cost map this process serves: the git blob id of the bytes it loaded, the
same id ``git rev-parse <commit>:model_prices_and_context_window.json`` prints for a checkout, plus
the ETag the remote fetch returned (None for the bundled backup)"""
return {
"generated_at": _cost_map_source_info.generated_at,
"source_revision": _cost_map_source_info.source_revision,
"etag": _cost_map_source_info.etag,
}
@ -441,7 +432,7 @@ def get_model_cost_map_source_info() -> CostMapSourceInfo:
- is_env_forced: True if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage
- fallback_reason: human-readable reason if remote failed and local was used
- loaded_at: ISO 8601 time this process last loaded the map
- generated_at, source_revision: the ``_metadata`` stamp inside the loaded file
- source_revision: git blob id of the loaded file's bytes
- etag: the ETag of the remote fetch (None for the bundled backup)
"""
loaded_at: Final = _cost_map_source_info.loaded_at
@ -451,7 +442,6 @@ def get_model_cost_map_source_info() -> CostMapSourceInfo:
"is_env_forced": _cost_map_source_info.is_env_forced,
"fallback_reason": _cost_map_source_info.fallback_reason,
"loaded_at": loaded_at.isoformat() if loaded_at is not None else None,
"generated_at": _cost_map_source_info.generated_at,
"source_revision": _cost_map_source_info.source_revision,
"etag": _cost_map_source_info.etag,
}
@ -518,21 +508,24 @@ def _expand_model_aliases(model_cost: dict) -> dict:
def _finalize_model_cost_map(model_cost: dict) -> dict:
"""Extract fallback generalizations and the provenance stamp out of the raw map, then expand aliases.
"""Extract fallback generalizations out of the raw map, then expand aliases.
The ``fallback_generalizations`` block is installed into the generalizations
module and the ``_metadata`` block into the source info; both are removed from
the map so neither is ever treated as a model entry.
module and removed from the map so it is never treated as a model entry.
"""
raw: Final = model_cost.pop(FALLBACK_GENERALIZATIONS_KEY, None)
rules: Final = raw.get("rules") if isinstance(raw, dict) else None
set_fallback_generalizations(rules)
metadata: Final = _parse_metadata(model_cost.pop(METADATA_KEY, None))
_cost_map_source_info.generated_at = metadata.generated_at
_cost_map_source_info.source_revision = metadata.source_revision
return _expand_model_aliases(model_cost)
def _finalize_loaded_model_cost_map(loaded: ModelCostMapReloaded) -> ModelCostMapReloaded:
"""Record which bytes this process now serves, then finalize the map they parsed into"""
_cost_map_source_info.source_revision = loaded.revision
_cost_map_source_info.etag = loaded.etag
return replace(loaded, model_cost_map=_finalize_model_cost_map(loaded.model_cost_map))
def get_model_cost_map(
url: str,
timeout: int = 5,
@ -561,12 +554,10 @@ def get_model_cost_map(
_cost_map_source_info.url = None
_cost_map_source_info.is_env_forced = True
_cost_map_source_info.fallback_reason = None
_cost_map_source_info.etag = None
return _finalize_model_cost_map(GetModelCostMap.load_local_model_cost_map())
return _finalize_loaded_model_cost_map(GetModelCostMap.load_local_model_cost_map_with_revision()).model_cost_map
_cost_map_source_info.url = url
_cost_map_source_info.is_env_forced = False
_cost_map_source_info.etag = None
result: Final = _fetch_remote_model_cost_map_with_retry_sync(
url=url,
@ -584,7 +575,7 @@ def get_model_cost_map(
)
_cost_map_source_info.source = "local"
_cost_map_source_info.fallback_reason = f"Remote fetch failed: {result.reason}"
return _finalize_model_cost_map(GetModelCostMap.load_local_model_cost_map())
return _finalize_loaded_model_cost_map(GetModelCostMap.load_local_model_cost_map_with_revision()).model_cost_map
content: Final = result.model_cost_map
# Validate using cached count (cheap int comparison, no file I/O)
@ -598,9 +589,8 @@ def get_model_cost_map(
)
_cost_map_source_info.source = "local"
_cost_map_source_info.fallback_reason = "Remote data failed integrity validation"
return _finalize_model_cost_map(GetModelCostMap.load_local_model_cost_map())
return _finalize_loaded_model_cost_map(GetModelCostMap.load_local_model_cost_map_with_revision()).model_cost_map
_cost_map_source_info.source = "remote"
_cost_map_source_info.fallback_reason = None
_cost_map_source_info.etag = result.etag
return _finalize_model_cost_map(content)
return _finalize_loaded_model_cost_map(result).model_cost_map

View file

@ -1,8 +1,4 @@
{
"_metadata": {
"generated_at": "2026-09-07T23:38:47Z",
"source_revision": "cd681a573fd9f5b6f15a1355f46178e4e9d374d2"
},
"sample_spec": {
"code_interpreter_cost_per_session": 0.0,
"computer_use_input_cost_per_1k_tokens": 0.0,

View file

@ -17938,7 +17938,7 @@ async def get_model_cost_map_source(
- is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage
- fallback_reason: human-readable reason why remote failed (null on success)
- loaded_at: when this pod last loaded the map
- generated_at, source_revision: the _metadata stamp inside the loaded file
- source_revision: git blob id of the loaded file, what git rev-parse <commit>:<path> prints for it
- etag: the ETag of the remote fetch (null for the bundled backup)
- model_count: number of models in the currently loaded cost map
"""

View file

@ -1,8 +1,4 @@
{
"_metadata": {
"generated_at": "2026-09-07T23:38:47Z",
"source_revision": "cd681a573fd9f5b6f15a1355f46178e4e9d374d2"
},
"sample_spec": {
"code_interpreter_cost_per_session": 0.0,
"computer_use_input_cost_per_1k_tokens": 0.0,

View file

@ -1,27 +1,9 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"title": "LiteLLM model_prices_and_context_window.json",
"description": "Schema for LiteLLM's model price and context window registry (https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json). Every top-level key except '_metadata', 'sample_spec', and 'fallback_generalizations' is a model id, optionally prefixed with its provider (e.g. 'azure/gpt-5.4'), mapping to a model entry. All costs are USD per unit. New optional fields are added regularly, so consumers should ignore unknown fields rather than reject them.",
"description": "Schema for LiteLLM's model price and context window registry (https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json). Every top-level key except 'sample_spec' and 'fallback_generalizations' is a model id, optionally prefixed with its provider (e.g. 'azure/gpt-5.4'), mapping to a model entry. All costs are USD per unit. New optional fields are added regularly, so consumers should ignore unknown fields rather than reject them.",
"type": "object",
"properties": {
"_metadata": {
"type": "object",
"description": "Provenance of this file: when an automated sync last regenerated it and the commit it ran against. Human edits leave it untouched; not a model entry.",
"properties": {
"generated_at": {
"type": "string",
"format": "date-time"
},
"source_revision": {
"type": "string"
}
},
"required": [
"generated_at",
"source_revision"
],
"additionalProperties": false
},
"sample_spec": {
"type": "object",
"description": "Documentation placeholder illustrating the entry shape; not a real model and not schema-conformant (several values are prose)."

View file

@ -19,11 +19,9 @@ import argparse
import json
import os
import re
import subprocess
import sys
from collections.abc import Mapping, Sequence
from dataclasses import dataclass, field
from datetime import datetime, timezone
from pathlib import Path
from types import MappingProxyType
from typing import Final
@ -35,7 +33,6 @@ MODELS_URL: Final = "https://api.together.ai/v1/models?serverless"
DEPRECATIONS_URL: Final = "https://docs.together.ai/docs/deprecations.md"
PROVIDER: Final = "together_ai"
PREFIX: Final = "together_ai/"
METADATA_KEY: Final = "_metadata"
SOURCE_URL: Final = "https://docs.together.ai/docs/serverless-models"
COST_MAP_RELPATHS: Final = (
"model_prices_and_context_window.json",
@ -498,23 +495,6 @@ def _serialize(cost_map: CostMap) -> str:
return json.dumps(cost_map, indent=4, ensure_ascii=False) + "\n"
def stamp_metadata(cost_map: CostMap, generated_at: str, source_revision: str) -> CostMap:
return {**cost_map, METADATA_KEY: {"generated_at": generated_at, "source_revision": source_revision}}
def _utc_now_iso() -> str:
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
def _source_revision(repo_root: Path) -> str:
from_env: Final = os.environ.get("GITHUB_SHA")
if from_env:
return from_env
return subprocess.run(
("git", "rev-parse", "HEAD"), cwd=repo_root, check=True, capture_output=True, text=True
).stdout.strip()
def main(argv: Sequence[str]) -> int:
parser: Final = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--write", action="store_true", help="apply the sync to the cost map files (default: dry run)")
@ -547,9 +527,8 @@ def main(argv: Sequence[str]) -> int:
if args.pr_body_file is not None:
args.pr_body_file.write_text(body)
if args.write and outcome.has_changes:
stamped: Final = _serialize(stamp_metadata(outcome.cost_map, _utc_now_iso(), _source_revision(args.repo_root)))
for relpath in COST_MAP_RELPATHS:
(args.repo_root / relpath).write_text(stamped)
(args.repo_root / relpath).write_text(_serialize(outcome.cost_map))
print(render_summary(outcome))
print()
print(body)

View file

@ -17,11 +17,11 @@ from litellm.litellm_core_utils.fallback_generalizations import (
)
from litellm.litellm_core_utils.get_model_cost_map import (
FALLBACK_GENERALIZATIONS_KEY,
METADATA_KEY,
GetModelCostMap,
_count_model_entries,
_finalize_model_cost_map,
get_model_cost_map_provenance,
git_blob_id,
)
@ -33,18 +33,16 @@ def _load_root_cost_map() -> dict:
return json.load(f)
def _load_bundled_stamp() -> dict:
path = os.path.join(
os.path.dirname(__file__), "../../../litellm/model_prices_and_context_window_backup.json"
)
with open(path) as f:
return json.load(f)[METADATA_KEY]
def _bundled_blob_id() -> str:
path = os.path.join(os.path.dirname(__file__), "../../../litellm/model_prices_and_context_window_backup.json")
with open(path, "rb") as f:
return git_blob_id(f.read())
_STAMP = {
"generated_at": "2026-09-07T00:00:00Z",
"source_revision": "0123456789abcdef0123456789abcdef01234567",
}
def test_git_blob_id_is_what_git_hash_object_prints():
"""An operator checks a reported revision with ``git hash-object`` or ``git rev-parse <commit>:<path>``,
so the id must be git's blob sha1 of the exact bytes, not a plain sha1 or a hash of the parsed JSON."""
assert git_blob_id(b'{"gpt-5.4-mini": {"mode": "chat"}}\n') == "18b9a8381e13a3b38a2128f184f631f95829e987"
def _make_models(n: int) -> dict:
@ -57,7 +55,6 @@ def test_count_model_entries_excludes_reserved_keys():
m = _make_models(3)
m["sample_spec"] = {"foo": "bar"}
m[FALLBACK_GENERALIZATIONS_KEY] = {"rules": []}
m[METADATA_KEY] = dict(_STAMP)
assert _count_model_entries(m) == 3
@ -143,39 +140,6 @@ def test_finalize_with_no_block_clears_rules():
set_fallback_generalizations(previous)
def test_finalize_pops_metadata_and_records_provenance():
finalized = _finalize_model_cost_map({**_make_models(2), METADATA_KEY: dict(_STAMP)})
assert METADATA_KEY not in finalized
assert len(finalized) == 2
provenance = get_model_cost_map_provenance()
assert provenance["generated_at"] == _STAMP["generated_at"]
assert provenance["source_revision"] == _STAMP["source_revision"]
def test_finalize_without_metadata_clears_the_previous_stamp():
_finalize_model_cost_map({**_make_models(2), METADATA_KEY: dict(_STAMP)})
_finalize_model_cost_map(_make_models(2))
provenance = get_model_cost_map_provenance()
assert provenance["generated_at"] is None
assert provenance["source_revision"] is None
@pytest.mark.parametrize(
"raw",
["2026-09-07T00:00:00Z", {"generated_at": 42}, ["2026-09-07T00:00:00Z"]],
ids=["string", "wrong_field_type", "list"],
)
def test_finalize_tolerates_a_malformed_metadata_block(raw):
finalized = _finalize_model_cost_map({**_make_models(2), METADATA_KEY: raw})
assert METADATA_KEY not in finalized
assert len(finalized) == 2
assert get_model_cost_map_provenance()["generated_at"] is None
def test_shipped_backup_carries_the_claude_routing_rules():
"""The bundled backup must ship the Claude routing rules so a fresh install
(or an offline fallback) routes unknown Claude models without code changes.
@ -390,10 +354,6 @@ def _real_map_bytes() -> bytes:
return json.dumps(_load_root_cost_map()).encode()
def _stamped_map_bytes(stamp: dict) -> bytes:
return json.dumps({**_load_root_cost_map(), METADATA_KEY: stamp}).encode()
class _SleepRecorder:
"""Injected in place of asyncio.sleep so tests assert waits without real delay."""
@ -555,40 +515,51 @@ async def test_refetch_respects_local_env_override(monkeypatch):
@pytest.mark.asyncio
async def test_refetch_records_the_file_stamp_and_the_fetch_etag():
"""A reload reports which revision of the map it swapped in: the ``_metadata`` stamp the file
carries plus the ETag the fetch returned, with the stamp itself kept out of the model map."""
client, _ = _mock_client(
[httpx.Response(200, headers={"ETag": 'W/"abc123"'}, content=_stamped_map_bytes(_STAMP))]
)
async def test_refetch_records_the_blob_id_of_the_bytes_served_and_the_fetch_etag():
"""A reload reports which revision of the map it swapped in: the git blob id of the exact bytes the
fetch returned, so ``git rev-parse <commit>:model_prices_and_context_window.json`` can confirm it,
plus the ETag the fetch returned."""
body = _real_map_bytes()
client, _ = _mock_client([httpx.Response(200, headers={"ETag": 'W/"abc123"'}, content=body)])
result = await refetch_model_cost_map(url=_URL, sleep=_SleepRecorder(), rng=random.Random(0), client=client)
assert isinstance(result, ModelCostMapReloaded)
assert result.revision == git_blob_id(body)
assert result.etag == 'W/"abc123"'
assert METADATA_KEY not in result.model_cost_map
assert get_model_cost_map_provenance() == {
"generated_at": _STAMP["generated_at"],
"source_revision": _STAMP["source_revision"],
"etag": 'W/"abc123"',
}
assert get_model_cost_map_provenance() == {"source_revision": git_blob_id(body), "etag": 'W/"abc123"'}
@pytest.mark.asyncio
async def test_refetch_local_override_reports_the_bundled_stamp_without_an_etag(monkeypatch):
"""Forcing the bundled backup after a remote reload must drop the remote ETag, since the map
served is no longer the one that ETag identifies."""
remote, _ = _mock_client(
[httpx.Response(200, headers={"ETag": 'W/"remote"'}, content=_stamped_map_bytes(_STAMP))]
async def test_refetch_revision_follows_the_bytes_not_the_url():
"""Two fetches of the same URL that return different bytes report different revisions."""
edited = json.loads(_real_map_bytes())
edited["gpt-5.4-mini"]["input_cost_per_token"] = 0.5
client, _ = _mock_client(
[httpx.Response(200, content=_real_map_bytes()), httpx.Response(200, content=json.dumps(edited).encode())]
)
first = await refetch_model_cost_map(url=_URL, sleep=_SleepRecorder(), rng=random.Random(0), client=client)
second = await refetch_model_cost_map(url=_URL, sleep=_SleepRecorder(), rng=random.Random(0), client=client)
assert isinstance(first, ModelCostMapReloaded) and isinstance(second, ModelCostMapReloaded)
assert first.revision != second.revision
assert get_model_cost_map_provenance()["source_revision"] == second.revision
@pytest.mark.asyncio
async def test_refetch_local_override_reports_the_bundled_blob_id_without_an_etag(monkeypatch):
"""Forcing the bundled backup after a remote reload must report the backup's own blob id and drop the
remote ETag, since the map served is no longer the one that ETag identifies."""
remote, _ = _mock_client([httpx.Response(200, headers={"ETag": 'W/"remote"'}, content=_real_map_bytes())])
await refetch_model_cost_map(url=_URL, sleep=_SleepRecorder(), rng=random.Random(0), client=remote)
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
result = await refetch_model_cost_map(url=_URL, sleep=_SleepRecorder(), rng=random.Random(0))
assert isinstance(result, ModelCostMapReloaded)
assert METADATA_KEY not in result.model_cost_map
assert get_model_cost_map_provenance() == {**_load_bundled_stamp(), "etag": None}
assert result.revision == _bundled_blob_id()
assert get_model_cost_map_provenance() == {"source_revision": _bundled_blob_id(), "etag": None}
# ---------------------------------------------------------------------------
@ -633,7 +604,7 @@ def test_boot_load_retries_transient_failures_instead_of_falling_back():
source = get_model_cost_map_source_info()
assert source["source"] == "remote"
assert source["fallback_reason"] is None
assert cost_map.keys() >= _load_root_cost_map().keys() - {"sample_spec", FALLBACK_GENERALIZATIONS_KEY, METADATA_KEY}
assert cost_map.keys() >= _load_root_cost_map().keys() - {"sample_spec", FALLBACK_GENERALIZATIONS_KEY}
def test_boot_load_honors_retry_after_then_falls_back_after_max_attempts():
@ -685,37 +656,31 @@ def test_boot_load_respects_local_env_override(monkeypatch):
assert get_model_cost_map_source_info()["is_env_forced"] is True
def test_boot_load_records_the_file_stamp_and_the_fetch_etag():
client, _ = _mock_client(
[httpx.Response(200, headers={"ETag": 'W/"boot"'}, content=_stamped_map_bytes(_STAMP))],
client_cls=httpx.Client,
)
def test_boot_load_records_the_blob_id_of_the_bytes_served_and_the_fetch_etag():
body = _real_map_bytes()
client, _ = _mock_client([httpx.Response(200, headers={"ETag": 'W/"boot"'}, content=body)], client_cls=httpx.Client)
cost_map = get_model_cost_map(url=_URL, sleep=_SyncSleepRecorder(), rng=random.Random(0), client=client)
get_model_cost_map(url=_URL, sleep=_SyncSleepRecorder(), rng=random.Random(0), client=client)
assert METADATA_KEY not in cost_map
source = get_model_cost_map_source_info()
assert source["source"] == "remote"
assert source["etag"] == 'W/"boot"'
assert source["generated_at"] == _STAMP["generated_at"]
assert source["source_revision"] == _STAMP["source_revision"]
assert source["source_revision"] == git_blob_id(body)
assert source["loaded_at"] is not None
def test_boot_load_fallback_to_the_backup_drops_the_remote_etag():
"""A boot that lands on the bundled backup reports the backup's own stamp and no ETag, even
def test_boot_load_fallback_to_the_backup_reports_its_blob_id_and_drops_the_remote_etag():
"""A boot that lands on the bundled backup reports the backup's own blob id and no ETag, even
when an earlier load in the same process had fetched the remote map."""
remote, _ = _mock_client(
[httpx.Response(200, headers={"ETag": 'W/"boot"'}, content=_stamped_map_bytes(_STAMP))],
client_cls=httpx.Client,
[httpx.Response(200, headers={"ETag": 'W/"boot"'}, content=_real_map_bytes())], client_cls=httpx.Client
)
get_model_cost_map(url=_URL, sleep=_SyncSleepRecorder(), rng=random.Random(0), client=remote)
failing, _ = _mock_client([httpx.Response(404)], client_cls=httpx.Client)
cost_map = get_model_cost_map(url=_URL, sleep=_SyncSleepRecorder(), rng=random.Random(0), client=failing)
get_model_cost_map(url=_URL, sleep=_SyncSleepRecorder(), rng=random.Random(0), client=failing)
assert METADATA_KEY not in cost_map
source = get_model_cost_map_source_info()
assert source["source"] == "local"
assert source["etag"] is None
assert {"generated_at": source["generated_at"], "source_revision": source["source_revision"]} == _load_bundled_stamp()
assert source["source_revision"] == _bundled_blob_id()

View file

@ -22,7 +22,6 @@ from .conftest import VOLATILE_KEYS, normalize
_VOLATILE = VOLATILE_KEYS | frozenset({"timestamp"})
_PROVENANCE = {
"generated_at": "2026-09-07T00:00:00Z",
"source_revision": "0123456789abcdef0123456789abcdef01234567",
"etag": 'W/"cost-map-etag"',
}
@ -108,22 +107,24 @@ def test_reload_model_cost_map_happy(client, auth_as, monkeypatch, mock_prisma):
assert update_payload["reload_revision"] == {"increment": 1}
def test_reload_model_cost_map_surfaces_provenance_and_keeps_metadata_out_of_the_model_list(
def test_reload_model_cost_map_surfaces_the_blob_id_of_the_bytes_served_on_every_status_surface(
client, auth_as, monkeypatch, mock_prisma
):
"""A real refetch through the reload route reports the file's stamp and the fetch ETag on every
status surface, while the ``_metadata`` block never shows up as a model anywhere."""
"""A real refetch through the reload route reports the git blob id of the exact bytes it fetched and
the fetch ETag on the reload response, the source route, and the schedule status alike."""
import httpx
import litellm
from litellm.litellm_core_utils.get_model_cost_map import git_blob_id
from litellm.proxy import proxy_server as ps
from litellm.proxy._types import LitellmUserRoles
_attach_litellm_config(mock_prisma)
monkeypatch.setattr(ps, "prisma_client", mock_prisma)
monkeypatch.delenv("LITELLM_LOCAL_MODEL_COST_MAP", raising=False)
stamped = {**json.loads(_ROOT_COST_MAP.read_text()), "_metadata": {k: v for k, v in _PROVENANCE.items() if k != "etag"}}
served = httpx.Response(200, headers={"ETag": _PROVENANCE["etag"]}, content=json.dumps(stamped).encode())
body = _ROOT_COST_MAP.read_bytes()
expected = {"source_revision": git_blob_id(body), "etag": _PROVENANCE["etag"]}
served = httpx.Response(200, headers={"ETag": _PROVENANCE["etag"]}, content=body)
monkeypatch.setattr(
"litellm.litellm_core_utils.get_model_cost_map._default_reload_client",
lambda: httpx.AsyncClient(transport=httpx.MockTransport(lambda request: served)),
@ -144,18 +145,15 @@ def test_reload_model_cost_map_surfaces_provenance_and_keeps_metadata_out_of_the
assert reload_response.status_code == 200
reload_body = reload_response.json()
assert {key: reload_body[key] for key in _PROVENANCE} == _PROVENANCE
assert {key: reload_body[key] for key in expected} == expected
assert source_response.status_code == 200
source_body = source_response.json()
assert {key: source_body[key] for key in _PROVENANCE} == _PROVENANCE
assert {key: source_body[key] for key in expected} == expected
assert source_body["source"] == "remote"
assert status_response.status_code == 200
assert {key: status_response.json()[key] for key in _PROVENANCE} == _PROVENANCE
assert {key: status_response.json()[key] for key in expected} == expected
assert public_response.status_code == 200
public_body = public_response.json()
assert "_metadata" not in public_body
assert "_metadata" not in litellm.model_cost
assert "gpt-4o" in public_body
assert "gpt-4o" in public_response.json()
assert reload_body["models_count"] == len(litellm.model_cost)

View file

@ -1,54 +0,0 @@
"""Tests for .github/scripts/auto_update_price_and_context_window_file.py."""
import importlib.util
import json
import re
import sys
from pathlib import Path
from typing import Final
_REPO_ROOT: Final = Path(__file__).resolve().parents[2]
_MODULE_PATH: Final = _REPO_ROOT / ".github" / "scripts" / "auto_update_price_and_context_window_file.py"
_spec: Final = importlib.util.spec_from_file_location("auto_update_price_and_context_window_file", _MODULE_PATH)
script: Final = importlib.util.module_from_spec(_spec)
sys.modules[_spec.name] = script
_spec.loader.exec_module(script)
_LOCAL_FILE: Final = "model_prices_and_context_window.json"
_GENERATED_AT: Final = re.compile(r"\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z")
def _openrouter_row(model_id: str) -> dict:
return {"id": model_id, "context_length": 8192, "pricing": {"prompt": "0.000001", "completion": "0.000002"}}
def _serve(openrouter_rows: list) -> object:
async def fetch_data(url: str) -> list:
return openrouter_rows if "openrouter" in url else []
return fetch_data
def _read_local(tmp_path: Path) -> dict:
return json.loads((tmp_path / _LOCAL_FILE).read_text())
def test_main_stamps_provenance_only_when_the_sync_changed_the_file(tmp_path: Path, monkeypatch) -> None:
monkeypatch.chdir(tmp_path)
monkeypatch.setenv("GITHUB_SHA", "feedface")
monkeypatch.setattr(script, "fetch_data", _serve([_openrouter_row("acme/x")]))
(tmp_path / _LOCAL_FILE).write_text(json.dumps({"sample_spec": {"input_cost_per_token": "USD"}}, indent=4) + "\n")
script.main()
written = _read_local(tmp_path)
assert written["openrouter/acme/x"]["litellm_provider"] == "openrouter"
assert written["_metadata"]["source_revision"] == "feedface"
assert _GENERATED_AT.fullmatch(written["_metadata"]["generated_at"])
sentinel = {**written, "_metadata": {**written["_metadata"], "generated_at": "2000-01-01T00:00:00Z"}}
(tmp_path / _LOCAL_FILE).write_text(json.dumps(sentinel, indent=4) + "\n")
script.main()
assert _read_local(tmp_path) == sentinel

View file

@ -141,26 +141,6 @@ def test_bot_may_not_change_special_root_keys() -> None:
assert _failures(head) == ("bot PRs may not change fallback_generalizations",)
STAMP: Final = {"generated_at": "2026-09-07T00:00:00Z", "source_revision": "0123456789abcdef0123456789abcdef01234567"}
def test_bot_may_stamp_and_restamp_metadata() -> None:
stamped = _snapshot({**BASE_MAP, "_metadata": STAMP})
assert _failures(stamped) == ()
assert _failures(stamped, bot=False) == ()
restamped = _snapshot(
{
**BASE_MAP,
"_metadata": {**STAMP, "generated_at": "2026-09-14T00:00:00Z"},
"fallback_generalizations": {"rules": []},
}
)
assert guard.guard_failures(stamped, restamped, MAP_FILES, True) == (
"bot PRs may not change fallback_generalizations",
)
def _commit(repo: Path, cost_map: dict[str, object], message: str) -> str:
text = _serialize(cost_map)
(repo / guard.COST_MAP_PATH).write_text(text)

View file

@ -98,25 +98,6 @@ def test_schema_rejects_malformed_entries(committed_schema: dict, entry: dict):
assert not validator.is_valid({"some-model": entry})
@pytest.mark.parametrize(
"metadata",
[
"2026-09-07T00:00:00Z",
{"generated_at": "2026-09-07T00:00:00Z"},
{"source_revision": "0123456789abcdef"},
{"generated_at": "2026-09-07T00:00:00Z", "source_revision": "0123456789abcdef", "author": "bot"},
],
ids=["not_an_object", "missing_revision", "missing_generated_at", "unknown_field"],
)
def test_schema_rejects_a_malformed_metadata_block(committed_schema: dict, metadata: object):
assert not build_validator(committed_schema).is_valid({"_metadata": metadata})
def test_schema_accepts_the_provenance_stamp_as_a_non_model_root_key(committed_schema: dict):
stamp = {"generated_at": "2026-09-07T00:00:00Z", "source_revision": "0123456789abcdef0123456789abcdef01234567"}
assert build_validator(committed_schema).is_valid({"_metadata": stamp})
def test_schema_accepts_minimal_and_unknown_optional_fields(committed_schema: dict):
validator = build_validator(committed_schema)
assert validator.is_valid({"some-model": {"litellm_provider": "openai"}})

View file

@ -1,6 +1,5 @@
import importlib.util
import json
import re
from pathlib import Path
from types import MappingProxyType
@ -370,58 +369,6 @@ def test_sync_is_idempotent_over_the_repo_cost_map() -> None:
assert second.cost_map == first.cost_map
def test_stamp_metadata_adds_the_provenance_block_without_touching_models() -> None:
cost_map = {"sample_spec": {"input_cost_per_token": "USD"}, "together_ai/acme/x": {"mode": "chat"}}
stamped = sync.stamp_metadata(cost_map, "2026-09-07T00:00:00Z", "feedface")
assert stamped["_metadata"] == {"generated_at": "2026-09-07T00:00:00Z", "source_revision": "feedface"}
assert {key: value for key, value in stamped.items() if key != "_metadata"} == cost_map
assert "_metadata" not in cost_map
def _write_registry(repo_root: Path, cost_map: dict) -> None:
for relpath in sync.COST_MAP_RELPATHS:
target = repo_root / relpath
target.parent.mkdir(parents=True, exist_ok=True)
target.write_text(json.dumps(cost_map, indent=4) + "\n")
def _read_registries(repo_root: Path) -> tuple[dict, ...]:
return tuple(json.loads((repo_root / relpath).read_text()) for relpath in sync.COST_MAP_RELPATHS)
def test_write_stamps_provenance_into_both_files_only_when_the_sync_changed_them(tmp_path: Path, monkeypatch) -> None:
monkeypatch.setenv("GITHUB_SHA", "feedface")
cost_map = json.loads((ROOT / "model_prices_and_context_window.json").read_text())
dropped = next(f"together_ai/{model.id}" for model in RECORDED_CATALOG if f"together_ai/{model.id}" in cost_map)
_write_registry(tmp_path, {key: value for key, value in cost_map.items() if key not in {dropped, "_metadata"}})
argv = (
"--write",
"--models-json",
str(FIXTURES / "models_serverless.json"),
"--deprecations-md",
str(FIXTURES / "deprecations.md"),
"--repo-root",
str(tmp_path),
)
assert sync.main(argv) == 0
written = _read_registries(tmp_path)
assert written[0] == written[1]
assert dropped in written[0]
assert written[0]["_metadata"]["source_revision"] == "feedface"
assert re.fullmatch(r"\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z", written[0]["_metadata"]["generated_at"])
sentinel = {**written[0], "_metadata": {**written[0]["_metadata"], "generated_at": "2000-01-01T00:00:00Z"}}
_write_registry(tmp_path, sentinel)
assert sync.main(argv) == 0
assert _read_registries(tmp_path) == (sentinel, sentinel)
def test_pr_body_lists_every_section_and_the_skipped_types() -> None:
outcome = sync.compute_sync({}, RECORDED_CATALOG, RECORDED_DOC)
body = sync.render_pr_body(outcome)

View file

@ -21,7 +21,6 @@ from litellm._logging import (
verbose_logger,
)
from litellm.integrations.custom_logger import CustomLogger
from litellm.litellm_core_utils.get_model_cost_map import RESERVED_TOP_LEVEL_KEYS
from litellm.proxy.utils import is_valid_api_key
from litellm.types.utils import (
CallTypes,
@ -1219,12 +1218,15 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
with open(prod_json, "r") as model_prices_file:
actual_json = json.load(model_prices_file)
assert isinstance(actual_json, dict)
model_entries: Final = {
key: value for key, value in actual_json.items() if key not in RESERVED_TOP_LEVEL_KEYS
}
actual_json.pop(
"sample_spec", None
) # remove the sample, whose schema is inconsistent with the real data
actual_json.pop(
"fallback_generalizations", None
) # reserved meta key, not a model entry
# Validate schema
validate(model_entries, INTENDED_SCHEMA)
validate(actual_json, INTENDED_SCHEMA)
# Validate cost values
# Define exceptions for models that are allowed to have costs > 1
@ -1235,7 +1237,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"runwayml/seedance2", # 4K output is 150 credits/second = $1.50/second
]
is_valid, violations = validate_model_cost_values(model_entries, exceptions)
is_valid, violations = validate_model_cost_values(actual_json, exceptions)
if not is_valid:
error_message = "Cost validation failed:\n" + "\n".join(violations)
@ -1266,7 +1268,8 @@ def test_max_tokens_consistency():
inconsistencies = []
for model_name, config in models.items():
if model_name in RESERVED_TOP_LEVEL_KEYS:
# Skip the sample_spec
if model_name == "sample_spec":
continue
# Check if both max_tokens and max_output_tokens exist

View file

@ -33,15 +33,13 @@ const remoteSource = {
is_env_forced: false,
fallback_reason: null,
loaded_at: null,
generated_at: null,
source_revision: null,
etag: null,
model_count: 1234,
};
const provenance = {
loaded_at: "2026-09-07T10:00:00Z",
generated_at: "2026-09-06T23:38:47Z",
source_revision: "cd681a573fd9f5b6f15a1355f46178e4e9d374d2",
source_revision: "4273ec544726bf255ea920533e209e6022653bb4",
etag: 'W/"eb8e9a53f4cc284b"',
};
@ -66,24 +64,22 @@ describe("PriceDataReload", () => {
render(<PriceDataReload accessToken="sk-test" />);
expect(await screen.findByText("Source revision:")).toBeInTheDocument();
expect(screen.getByText("cd681a573fd9")).toBeInTheDocument();
expect(screen.getByText("4273ec544726")).toBeInTheDocument();
expect(screen.getByText("ETag:")).toBeInTheDocument();
expect(screen.getByText('W/"eb8e9a53f4cc284b"')).toBeInTheDocument();
expect(screen.getByText("Generated at:")).toBeInTheDocument();
expect(screen.getByText(new Date(provenance.generated_at).toLocaleString())).toBeInTheDocument();
expect(screen.getByText("Loaded at:")).toBeInTheDocument();
expect(screen.getByText(new Date(provenance.loaded_at).toLocaleString())).toBeInTheDocument();
});
it("shows a malformed generated_at stamp as-is instead of Invalid Date", async () => {
it("shows a malformed loaded_at as-is instead of Invalid Date", async () => {
vi.mocked(getModelCostMapSource).mockResolvedValue({
...remoteSource,
...provenance,
generated_at: "yesterday-ish",
loaded_at: "yesterday-ish",
} as never);
render(<PriceDataReload accessToken="sk-test" />);
expect(await screen.findByText("Generated at:")).toBeInTheDocument();
expect(await screen.findByText("Loaded at:")).toBeInTheDocument();
expect(screen.getByText("yesterday-ish")).toBeInTheDocument();
expect(screen.queryByText("Invalid Date")).not.toBeInTheDocument();
});
@ -92,7 +88,6 @@ describe("PriceDataReload", () => {
render(<PriceDataReload accessToken="sk-test" />);
expect(await screen.findByText("Pricing Data Source")).toBeInTheDocument();
expect(screen.queryByText("Generated at:")).not.toBeInTheDocument();
expect(screen.queryByText("Source revision:")).not.toBeInTheDocument();
expect(screen.queryByText("ETag:")).not.toBeInTheDocument();
expect(screen.queryByText("Loaded at:")).not.toBeInTheDocument();

View file

@ -50,7 +50,6 @@ interface CostMapSourceInfo {
is_env_forced: boolean;
fallback_reason: string | null;
loaded_at: string | null;
generated_at: string | null;
source_revision: string | null;
etag: string | null;
model_count: number;
@ -105,13 +104,6 @@ const formatDateTime = (dateTimeString: string | null) => {
const CostMapProvenanceRows: React.FC<{ sourceInfo: CostMapSourceInfo }> = ({ sourceInfo }) => (
<>
{sourceInfo.generated_at && (
<div className="flex items-center justify-between text-xs">
<span className="text-muted-foreground">Generated at:</span>
<span className="font-medium">{formatDateTime(sourceInfo.generated_at)}</span>
</div>
)}
{sourceInfo.source_revision && (
<div className="flex items-center justify-between gap-2 text-xs">
<span className="text-muted-foreground">Source revision:</span>

View file

@ -8685,7 +8685,7 @@ export interface paths {
* - is_env_forced: true if LITELLM_LOCAL_MODEL_COST_MAP=True forced local usage
* - fallback_reason: human-readable reason why remote failed (null on success)
* - loaded_at: when this pod last loaded the map
* - generated_at, source_revision: the _metadata stamp inside the loaded file
* - source_revision: git blob id of the loaded file, what git rev-parse <commit>:<path> prints for it
* - etag: the ETag of the remote fetch (null for the bundled backup)
* - model_count: number of models in the currently loaded cost map
*/