diff --git a/ci_cd/cost_map_guard.py b/ci_cd/cost_map_guard.py index 5842cf6f1ac..14eecf01461 100644 --- a/ci_cd/cost_map_guard.py +++ b/ci_cd/cost_map_guard.py @@ -14,8 +14,9 @@ import argparse import json import subprocess import sys -from collections.abc import Sequence +from collections.abc import Mapping, Sequence from dataclasses import dataclass +from types import MappingProxyType from typing import Final from generate_model_prices_schema import SPECIAL_ROOT_KEYS, build_schema, render, validation_errors @@ -25,6 +26,7 @@ BACKUP_PATH: Final = "litellm/model_prices_and_context_window_backup.json" SCHEMA_PATH: Final = "model_prices_and_context_window.schema.json" GUARDED_PATHS: Final = (COST_MAP_PATH, BACKUP_PATH, SCHEMA_PATH) BOT_BRANCH_PREFIX: Final = "litellm_cost_map_sync_" +EVALUATION_OUTPUT_PRICE_SOURCES: Final[Mapping[str, str]] = MappingProxyType({}) CostMap = dict[str, object] @@ -51,7 +53,31 @@ def _rendered_schema(cost_map: CostMap) -> str: return str(error) -def _file_failures(head: Snapshot, head_map: CostMap) -> tuple[str, ...]: +def _evaluation_output_price_failure( + key: str, entry: object, evaluation_output_price_sources: Mapping[str, str] +) -> str | None: + if not isinstance(entry, dict) or entry.get("mode") != "evaluation": + return None + output_cost: Final = entry.get("output_cost_per_token") + if ( + not isinstance(output_cost, (int, float)) + or isinstance(output_cost, bool) + or output_cost <= 0 + or key in evaluation_output_price_sources + ): + return None + return ( + f"{key}: mode evaluation rows bill input only; output_cost_per_token must be 0 unless the vendor " + "prices output tokens for this model (add the source to EVALUATION_OUTPUT_PRICE_SOURCES)" + ) + + +def _file_failures( + head: Snapshot, + head_map: CostMap, + *, + evaluation_output_price_sources: Mapping[str, str] = EVALUATION_OUTPUT_PRICE_SOURCES, +) -> tuple[str, ...]: schema_text: Final = _rendered_schema(head_map) if not schema_text.startswith("{"): return (schema_text,) @@ -75,6 +101,16 @@ def _file_failures(head: Snapshot, head_map: CostMap) -> tuple[str, ...]: f"{COST_MAP_PATH} does not validate against its schema: {error}" for error in validation_errors(head_map, json.loads(schema_text))[:20] ), + *( + failure + for key, entry in head_map.items() + if ( + failure := _evaluation_output_price_failure( + key, entry, evaluation_output_price_sources + ) + ) + is not None + ), ) @@ -121,13 +157,25 @@ def contract_for(bot: bool, changed_files: Sequence[str]) -> str: return "human PR, file checks only" if touches_cost_map(changed_files) else "human PR, cost map untouched" -def guard_failures(base: Snapshot, head: Snapshot, changed_files: Sequence[str], bot: bool) -> tuple[str, ...]: +def guard_failures( + base: Snapshot, + head: Snapshot, + changed_files: Sequence[str], + bot: bool, + *, + evaluation_output_price_sources: Mapping[str, str] = EVALUATION_OUTPUT_PRICE_SOURCES, +) -> tuple[str, ...]: if not bot and not touches_cost_map(changed_files): return () head_map: Final = _parse_object(head.cost_map, COST_MAP_PATH) if isinstance(head_map, str): return (head_map,) - return (*_file_failures(head, head_map), *(_bot_failures(base, head_map, changed_files) if bot else ())) + return ( + *_file_failures( + head, head_map, evaluation_output_price_sources=evaluation_output_price_sources + ), + *(_bot_failures(base, head_map, changed_files) if bot else ()), + ) def _git(*args: str) -> str | None: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6d49e7cd7fd..400f153c115 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -71768,7 +71768,7 @@ "input_cost_per_token": 4.2e-08, "litellm_provider": "azure_ai", "mode": "evaluation", - "output_cost_per_token": 4.2e-08, + "output_cost_per_token": 0.0, "source": "https://prices.azure.com/api/retail/prices?$filter=productName eq 'Microsoft Decision Models'" }, "gpt-rosalind-discovery": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6d49e7cd7fd..400f153c115 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -71768,7 +71768,7 @@ "input_cost_per_token": 4.2e-08, "litellm_provider": "azure_ai", "mode": "evaluation", - "output_cost_per_token": 4.2e-08, + "output_cost_per_token": 0.0, "source": "https://prices.azure.com/api/retail/prices?$filter=productName eq 'Microsoft Decision Models'" }, "gpt-rosalind-discovery": { diff --git a/tests/unit/test_cost_map_guard.py b/tests/unit/test_cost_map_guard.py index 1595bd9864f..e127aa08d29 100644 --- a/tests/unit/test_cost_map_guard.py +++ b/tests/unit/test_cost_map_guard.py @@ -1,3 +1,4 @@ +from collections.abc import Mapping import importlib.util import json import subprocess @@ -62,8 +63,19 @@ def _snapshot(cost_map: dict[str, object], backup: str | None = None, schema: st BASE: Final = _snapshot(BASE_MAP) -def _failures(head: object, changed_files: tuple[str, ...] = MAP_FILES, bot: bool = True) -> tuple[str, ...]: - return guard.guard_failures(BASE, head, changed_files, bot) +def _failures( + head: object, + changed_files: tuple[str, ...] = MAP_FILES, + bot: bool = True, + evaluation_output_price_sources: Mapping[str, str] = guard.EVALUATION_OUTPUT_PRICE_SOURCES, +) -> tuple[str, ...]: + return guard.guard_failures( + BASE, + head, + changed_files, + bot, + evaluation_output_price_sources=evaluation_output_price_sources, + ) def test_in_sync_files_pass_for_humans_and_bots() -> None: @@ -71,6 +83,52 @@ def test_in_sync_files_pass_for_humans_and_bots() -> None: assert _failures(BASE, bot=True) == () +EVALUATION_MODEL: Final = "azure_ai/Microsoft-Decision-1" +EVALUATION_OUTPUT_PRICE_FAILURE: Final = ( + f"{EVALUATION_MODEL}: mode evaluation rows bill input only; output_cost_per_token must be 0 unless the vendor " + "prices output tokens for this model (add the source to EVALUATION_OUTPUT_PRICE_SOURCES)" +) + + +@pytest.mark.parametrize("bot", (True, False)) +def test_evaluation_rows_with_output_price_fail_for_humans_and_bots(bot: bool) -> None: + head: Final = _snapshot( + {**BASE_MAP, EVALUATION_MODEL: _entry(mode="evaluation", output_cost_per_token=4.2e-08)} + ) + assert _failures(head, bot=bot) == (EVALUATION_OUTPUT_PRICE_FAILURE,) + + +def test_evaluation_rows_with_zero_or_missing_output_price_pass() -> None: + zero_output: Final = _snapshot( + {**BASE_MAP, EVALUATION_MODEL: _entry(mode="evaluation", output_cost_per_token=0.0)} + ) + missing_output_entry: Final = { + key: value + for key, value in _entry(mode="evaluation").items() + if key != "output_cost_per_token" + } + missing_output: Final = _snapshot({**BASE_MAP, EVALUATION_MODEL: missing_output_entry}) + + assert _failures(zero_output) == () + assert _failures(missing_output) == () + + +def test_chat_rows_with_output_price_pass() -> None: + head: Final = _snapshot( + {**BASE_MAP, "openrouter/chat-with-output": _entry(output_cost_per_token=4.2e-08)} + ) + assert _failures(head) == () + + +def test_evaluation_output_price_source_allowlist_passes() -> None: + head: Final = _snapshot( + {**BASE_MAP, EVALUATION_MODEL: _entry(mode="evaluation", output_cost_per_token=4.2e-08)} + ) + source: Final = 'https://vendor.example/pricing "Output tokens are charged per token."' + + assert _failures(head, evaluation_output_price_sources={EVALUATION_MODEL: source}) == () + + def test_bot_may_add_and_reprice_models() -> None: head = _snapshot({**BASE_MAP, "openrouter/a": _entry(9e-06, supports_vision=True), "openrouter/c": _entry()}) assert _failures(head) == ()