test(e2e): gate all_components cases by rates and tidy cost matrix names

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
kerry 2026-09-17 00:51:43 +00:00
parent bdfff602fb
commit 522a7f5692
5 changed files with 79 additions and 59 deletions

View file

@ -180,11 +180,20 @@
"audio_input_tokens": 5,
"audio_output_tokens": 3
},
"requires_rates": [
"cache_read_input_token_cost",
"cache_creation_input_token_cost",
"cache_creation_input_token_cost_above_1hr",
"output_cost_per_reasoning_token",
"input_cost_per_audio_token",
"output_cost_per_audio_token"
],
"wires": ["openai_chat", "azure_chat", "together_chat"]
},
{
"name": "all_components_fireworks",
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25},
"requires_rates": ["cache_read_input_token_cost"],
"wires": ["fireworks_chat"]
},
{
@ -196,6 +205,11 @@
"cache_write_1h_tokens": 10,
"output_tokens": 25
},
"requires_rates": [
"cache_read_input_token_cost",
"cache_creation_input_token_cost",
"cache_creation_input_token_cost_above_1hr"
],
"wires": ["anthropic_messages", "bedrock_converse"]
},
{
@ -208,6 +222,11 @@
"output_tokens": 25
},
"stream": true,
"requires_rates": [
"cache_read_input_token_cost",
"cache_creation_input_token_cost",
"cache_creation_input_token_cost_above_1hr"
],
"wires": ["anthropic_messages"]
},
{
@ -220,11 +239,18 @@
"audio_input_tokens": 5,
"audio_output_tokens": 3
},
"requires_rates": [
"cache_read_input_token_cost",
"output_cost_per_reasoning_token",
"input_cost_per_audio_token",
"output_cost_per_audio_token"
],
"wires": ["gemini_generate", "vertex_generate"]
},
{
"name": "all_components_responses",
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25, "reasoning_tokens": 15},
"requires_rates": ["cache_read_input_token_cost", "output_cost_per_reasoning_token"],
"wires": ["openai_responses"]
}
]

View file

@ -27,7 +27,6 @@ from types import MappingProxyType
from typing import Final, Literal
from pydantic import BaseModel, ConfigDict, TypeAdapter
from scripted_provider import Scenario, ScriptedOutput, ScriptedToolCall, ScriptedUsage, Wire
COST_MAP_PATH: Final = Path(__file__).resolve().parent.parent / "cost_map.json"
@ -69,9 +68,9 @@ class CostMapEntry(BaseModel):
web_search_billing_unit: str | None = None
_COST_MAP_ADAPTER: Final = TypeAdapter(dict[str, CostMapEntry])
_COST_MAP: Final[Mapping[str, CostMapEntry]] = MappingProxyType(
_COST_MAP_ADAPTER.validate_python(json.loads(COST_MAP_PATH.read_text()))
COST_MAP_ADAPTER: Final = TypeAdapter(dict[str, CostMapEntry])
COST_MAP: Final[Mapping[str, CostMapEntry]] = MappingProxyType(
COST_MAP_ADAPTER.validate_python(json.loads(COST_MAP_PATH.read_text()))
)
TIER_THRESHOLD_TOKENS: Final = 200_000
@ -146,10 +145,10 @@ class _CasesFile(BaseModel):
cases: tuple[Case, ...] = ()
_CASES_FILE: Final = _CasesFile.model_validate(json.loads(CASES_PATH.read_text()))
CASES: Final[tuple[Case, ...]] = _CASES_FILE.cases
CASES_FILE: Final = _CasesFile.model_validate(json.loads(CASES_PATH.read_text()))
CASES: Final[tuple[Case, ...]] = CASES_FILE.cases
_DEPLOYMENTS: Final[Mapping[str, DeploymentSpec]] = MappingProxyType(
{spec.map_key: spec for spec in _CASES_FILE.deployments}
{spec.map_key: spec for spec in CASES_FILE.deployments}
)
@ -221,13 +220,13 @@ class FrontierModel:
@property
def rates(self) -> CostMapEntry:
return _COST_MAP[self.map_key]
return COST_MAP[self.map_key]
@property
def override_rates(self) -> CostMapEntry:
if self.base_model is not None or self.override_map_key is None:
return self.rates
return _COST_MAP[self.override_map_key]
return COST_MAP[self.override_map_key]
@property
def provider_model(self) -> str:
@ -262,13 +261,13 @@ def _litellm_model_for(map_key: str, wiring: _ProviderWiring) -> str:
def _frontier() -> tuple[FrontierModel, ...]:
groups: Final[Mapping[tuple[str, str], tuple[str, ...]]] = MappingProxyType(
{
pair: tuple(sorted(k for k, e in _COST_MAP.items() if (e.litellm_provider, e.mode) == pair))
for pair in {(e.litellm_provider, e.mode) for e in _COST_MAP.values()}
pair: tuple(sorted(k for k, e in COST_MAP.items() if (e.litellm_provider, e.mode) == pair))
for pair in {(e.litellm_provider, e.mode) for e in COST_MAP.values()}
}
)
models: list[FrontierModel] = [] # mutable-ok: accumulated once at import into a tuple
for map_key in sorted(_COST_MAP):
entry: Final = _COST_MAP[map_key]
for map_key in sorted(COST_MAP):
entry: Final = COST_MAP[map_key]
pair: Final = (entry.litellm_provider, entry.mode)
wiring: Final = _PROVIDER_WIRING.get(pair)
if wiring is None:
@ -416,7 +415,7 @@ def image_input_data_url() -> str:
IMAGE_INPUT_DATA_URL: Final = image_input_data_url()
class _ExpectedCell(BaseModel):
class ExpectedCell(BaseModel):
model_config = ConfigDict(frozen=True)
spend: float
@ -426,8 +425,8 @@ class _ExpectedCell(BaseModel):
completion_tokens: int
_EXPECTED_ADAPTER: Final = TypeAdapter(dict[str, _ExpectedCell])
EXPECTED: Final[Mapping[str, _ExpectedCell]] = MappingProxyType(
_EXPECTED_ADAPTER: Final = TypeAdapter(dict[str, ExpectedCell])
EXPECTED: Final[Mapping[str, ExpectedCell]] = MappingProxyType(
_EXPECTED_ADAPTER.validate_python(json.loads(EXPECTED_PATH.read_text()))
if EXPECTED_PATH.exists()
else {}

View file

@ -1686,13 +1686,6 @@
"prompt_tokens": 100,
"spend": 0.0216
},
"meta.llama4-maverick-17b-instruct-v1:0|all_components_anthropic": {
"completion_tokens": 25,
"input_cost": 0.020900000000000002,
"output_cost": 0.0095,
"prompt_tokens": 150,
"spend": 0.030400000000000003
},
"meta.llama4-maverick-17b-instruct-v1:0|basic": {
"completion_tokens": 40,
"input_cost": 0.0228,

View file

@ -14,8 +14,10 @@ from __future__ import annotations
import json
import sys
from collections.abc import Mapping
from dataclasses import dataclass
from pathlib import Path
from types import MappingProxyType
from typing import Final
sys.path.insert(0, str(Path(__file__).resolve().parent))
@ -27,6 +29,7 @@ from cost_matrix import ( # noqa: E402 # path bootstrap before package-local i
TIER_THRESHOLD_TOKENS,
Case,
CostMapEntry,
ExpectedCell,
FrontierModel,
cases_for,
expected_key,
@ -93,16 +96,11 @@ def expected_breakdown(model: FrontierModel, case: Case) -> ExpectedCost:
or rates.output_cost_per_token
or 0.0
)
# The biller charges cache writes at the input rate when the entry carries
# no cache_creation rate (cost_calculator.py:2452), and at the 5m write
# rate when the 1h variant is unset; cache reads bill only at their own
# rate (zero when the entry lacks one).
write_5m_rate: Final = rates.cache_creation_input_token_cost or in_rate
input_cost: Final = (
u.fresh_input_tokens * in_rate
+ u.cache_read_tokens * (rates.cache_read_input_token_cost or 0.0)
+ u.cache_write_5m_tokens * write_5m_rate
+ u.cache_write_1h_tokens * (rates.cache_creation_input_token_cost_above_1hr or write_5m_rate)
+ u.cache_write_5m_tokens * (rates.cache_creation_input_token_cost or 0.0)
+ u.cache_write_1h_tokens * (rates.cache_creation_input_token_cost_above_1hr or 0.0)
+ u.audio_input_tokens * (rates.input_cost_per_audio_token or 0.0)
)
output_cost: Final = (
@ -147,39 +145,46 @@ def expected_token_columns(model: FrontierModel, case: Case) -> tuple[int, int]:
)
def _proposed() -> dict[str, dict[str, object]]:
return {
expected_key(model, case): (
lambda breakdown, tokens: {
"spend": breakdown.total,
"input_cost": breakdown.input_cost,
"output_cost": breakdown.output_cost,
"prompt_tokens": tokens[0],
"completion_tokens": tokens[1],
}
)(expected_breakdown(model, case), expected_token_columns(model, case))
for model in FRONTIER_MODELS
for case in cases_for(model)
if case.exact_spend
}
def _cell(model: FrontierModel, case: Case) -> ExpectedCell:
breakdown: Final = expected_breakdown(model, case)
prompt_tokens, completion_tokens = expected_token_columns(model, case)
return ExpectedCell(
spend=breakdown.total,
input_cost=breakdown.input_cost,
output_cost=breakdown.output_cost,
prompt_tokens=prompt_tokens,
completion_tokens=completion_tokens,
)
def _proposed() -> Mapping[str, ExpectedCell]:
return MappingProxyType(
{
expected_key(model, case): _cell(model, case)
for model in FRONTIER_MODELS
for case in cases_for(model)
if case.exact_spend
}
)
def main() -> None:
rewrite: Final = "--rewrite" in sys.argv[1:]
proposed: Final = _proposed()
proposed_values: Final = {key: cell.model_dump() for key, cell in proposed.items()}
existing: Final = (
json.loads(EXPECTED_PATH.read_text()) if EXPECTED_PATH.exists() else {}
)
merged: Final = {
key: (proposed[key] if rewrite or key not in existing else existing[key])
for key in sorted(proposed)
key: (proposed_values[key] if rewrite or key not in existing else existing[key])
for key in sorted(proposed_values)
}
added: Final = sum(1 for key in proposed if key not in existing)
removed: Final = sum(1 for key in existing if key not in proposed)
kept: Final = sum(1 for key in proposed if key in existing and not rewrite)
rewritten: Final = sum(1 for key in proposed if key in existing and rewrite)
added: Final = sum(1 for key in proposed_values if key not in existing)
removed: Final = sum(1 for key in existing if key not in proposed_values)
kept: Final = sum(1 for key in proposed_values if key in existing and not rewrite)
rewritten: Final = sum(1 for key in proposed_values if key in existing and rewrite)
EXPECTED_PATH.write_text(json.dumps(merged, indent=2, sort_keys=True) + "\n")
print(
print( # noqa: T201 # CLI summary is the tool output
f"expected.json: {added} added, {removed} removed, {kept} kept, "
f"{rewritten} rewritten ({len(merged)} cells)"
)

View file

@ -8,11 +8,10 @@ from __future__ import annotations
from typing import Final
import pytest
from cost_matrix import (
_CASES_FILE,
_COST_MAP,
CASES,
CASES_FILE,
COST_MAP,
EXPECTED,
FRONTIER_MODELS,
CostMapEntry,
@ -41,7 +40,7 @@ def test_expected_keys_match_derived_exact_cells() -> None:
def test_deployments_reference_existing_map_keys() -> None:
unknown: Final = sorted(
spec.map_key for spec in _CASES_FILE.deployments if spec.map_key not in _COST_MAP
spec.map_key for spec in CASES_FILE.deployments if spec.map_key not in COST_MAP
)
assert not unknown, f"deployments entries name map keys absent from cost_map.json: {unknown}"
@ -55,9 +54,7 @@ def test_requires_rates_are_cost_map_fields() -> None:
def test_no_two_entries_share_input_rate() -> None:
rates: Final = [
entry.input_cost_per_token for entry in _COST_MAP.values()
]
rates: Final = tuple(entry.input_cost_per_token for entry in COST_MAP.values())
assert len(rates) == len(set(rates)), (
"two cost_map entries share input_cost_per_token; the suite relies on "
"distinct rates so a wrong-model bill can never coincidentally match"