mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
test(e2e): drive the cost matrix from cases.json and expected.json goldens
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
813d96f26e
commit
bdfff602fb
9 changed files with 2761 additions and 957 deletions
|
|
@ -21,7 +21,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family
|
|||
- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set and excluded from the per-PR selector like the rest of `load/`, driven by `.github/workflows/test-e2e-redis-chaos.yml` and by the Buildkite `e2e-redis-chaos` step in project-releaser, which runs the proxy, Postgres and Valkey co-located with pytest in one pod and sets the opt-in), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic
|
||||
- `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate, JWT auth (access tokens issued by a real Keycloak realm, `idp.py` plus `idp_realm.json`, whose JWKS the proxy's `JWT_PUBLIC_KEY_URL` points at; see CONTRIBUTING.md for the start command and config block), and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite
|
||||
- `gateway/` - proxy configuration only (`litellm-config.yml`); no tests
|
||||
- `cost_calculation/` - cost accounting against a dedicated proxy whose whole model cost map is the test-owned `tests/e2e/cost_map.json` (loaded via `LITELLM_MODEL_COST_MAP_URL`), with provider calls answered by the scripted-provider sidecar in `scripted_provider.py`; asserts literal rate arithmetic on scripted usage across every provider wire and pricing component, deselected unless `E2E_COST_MAP_STACK` is set, driven by the Buildkite `e2e-cost-calculation` step in project-releaser, which runs a proxy booted from `gateway/cost_calculation_ci_config.yml`, Postgres and the scripted provider co-located with pytest in one pod and sets the opt-in
|
||||
- `cost_calculation/` - cost accounting against a dedicated proxy whose whole model cost map is the test-owned `tests/e2e/cost_map.json` (loaded via `LITELLM_MODEL_COST_MAP_URL`), with provider calls answered by the scripted-provider sidecar in `scripted_provider.py`; every cost-map entry is a deployment and the cases plus asserted goldens are data in `cases.json` and `expected.json` (regenerate with `generate_expected.py`), deselected unless `E2E_COST_MAP_STACK` is set, driven by the Buildkite `e2e-cost-calculation` step in project-releaser, which runs a proxy booted from `gateway/cost_calculation_ci_config.yml`, Postgres and the scripted provider co-located with pytest in one pod and sets the opt-in
|
||||
- `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke
|
||||
- `ui/` - the Admin UI browser suite: Playwright in TypeScript, driving the dashboard served by a live proxy on port 4000 (seeded postgres + mock LLM upstream; see its `run_e2e.sh`). It is a self-contained npm package with its own lockfile and does not use the Python harness, pytest markers, or the shared transport; the Python rules in this file (typed models, `Result` unions, basedpyright zero-error gate) do not apply inside it. Its only Python file, `fixtures/mock_llm_server/server.py`, is excluded from the e2e basedpyright gate via the root `pyrightconfig.json`
|
||||
|
||||
|
|
|
|||
231
tests/e2e/cost_calculation/cases.json
Normal file
231
tests/e2e/cost_calculation/cases.json
Normal file
|
|
@ -0,0 +1,231 @@
|
|||
{
|
||||
"deployments": [
|
||||
{
|
||||
"map_key": "azure/gpt-5.4-mini",
|
||||
"litellm_model": "azure/cc-pinned-deployment",
|
||||
"base_model": "azure/gpt-5.4-mini"
|
||||
}
|
||||
],
|
||||
"cases": [
|
||||
{
|
||||
"name": "basic",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40}
|
||||
},
|
||||
{
|
||||
"name": "cache_read",
|
||||
"usage": {"fresh_input_tokens": 100, "cache_read_tokens": 50, "output_tokens": 30},
|
||||
"requires_rates": ["cache_read_input_token_cost"],
|
||||
"requires_caps": ["cache_read"]
|
||||
},
|
||||
{
|
||||
"name": "cache_write_5m",
|
||||
"usage": {"fresh_input_tokens": 90, "cache_write_5m_tokens": 60, "output_tokens": 30},
|
||||
"requires_rates": ["cache_creation_input_token_cost"],
|
||||
"requires_caps": ["cache_write_5m"]
|
||||
},
|
||||
{
|
||||
"name": "cache_write_1h",
|
||||
"usage": {"fresh_input_tokens": 90, "cache_write_5m_tokens": 20, "cache_write_1h_tokens": 40, "output_tokens": 30},
|
||||
"requires_rates": ["cache_creation_input_token_cost_above_1hr", "cache_creation_input_token_cost"],
|
||||
"requires_caps": ["cache_write_1h"]
|
||||
},
|
||||
{
|
||||
"name": "reasoning",
|
||||
"usage": {"fresh_input_tokens": 100, "output_tokens": 30, "reasoning_tokens": 70},
|
||||
"requires_rates": ["output_cost_per_reasoning_token"],
|
||||
"requires_caps": ["reasoning"]
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"usage": {"fresh_input_tokens": 100, "audio_input_tokens": 25, "output_tokens": 30, "audio_output_tokens": 15},
|
||||
"requires_rates": ["input_cost_per_audio_token", "output_cost_per_audio_token"],
|
||||
"requires_caps": ["audio"]
|
||||
},
|
||||
{
|
||||
"name": "tiered",
|
||||
"usage": {"fresh_input_tokens": 200001, "output_tokens": 30},
|
||||
"requires_rates": ["input_cost_per_token_above_200k_tokens", "output_cost_per_token_above_200k_tokens"]
|
||||
},
|
||||
{
|
||||
"name": "service_tier_flex",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"service_tier": "flex",
|
||||
"requires_rates": ["input_cost_per_token_flex", "output_cost_per_token_flex"]
|
||||
},
|
||||
{
|
||||
"name": "service_tier_priority",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"service_tier": "priority",
|
||||
"requires_rates": ["input_cost_per_token_priority", "output_cost_per_token_priority"]
|
||||
},
|
||||
{
|
||||
"name": "web_search",
|
||||
"usage": {"fresh_input_tokens": 100, "output_tokens": 30, "web_search_calls": 3},
|
||||
"requires_rates": ["search_context_cost_per_query"],
|
||||
"requires_caps": ["web_search"]
|
||||
},
|
||||
{
|
||||
"name": "stream",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"stream_usage": "absent",
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["absent_usage"]
|
||||
},
|
||||
{
|
||||
"name": "response_model_override",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"response_model_override": true,
|
||||
"requires_caps": ["response_model"]
|
||||
},
|
||||
{
|
||||
"name": "stream_response_model_override",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"response_model_override": true,
|
||||
"requires_caps": ["response_model"]
|
||||
},
|
||||
{
|
||||
"name": "tool_call",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"tool_call": true,
|
||||
"requires_caps": ["tool_call"]
|
||||
},
|
||||
{
|
||||
"name": "stream_tool_call",
|
||||
"usage": {"fresh_input_tokens": 80, "output_tokens": 25},
|
||||
"stream": true,
|
||||
"tool_call": true,
|
||||
"requires_caps": ["tool_call"]
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage_tool_call",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"stream_usage": "absent",
|
||||
"tool_call": true,
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["absent_usage", "tool_call"]
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage_image_input",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"stream_usage": "absent",
|
||||
"image_input": true,
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["absent_usage", "image_input"]
|
||||
},
|
||||
{
|
||||
"name": "stream_incomplete",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"terminal": "incomplete",
|
||||
"requires_caps": ["responses_terminal"]
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage_incomplete",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"stream_usage": "absent",
|
||||
"terminal": "incomplete",
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["responses_terminal"]
|
||||
},
|
||||
{
|
||||
"name": "stream_unvalidated",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"terminal": "unvalidated",
|
||||
"requires_caps": ["responses_terminal"]
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage_unvalidated",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"stream_usage": "absent",
|
||||
"terminal": "unvalidated",
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["responses_terminal"]
|
||||
},
|
||||
{
|
||||
"name": "prompt_blocked",
|
||||
"usage": {"fresh_input_tokens": 1000, "output_tokens": 0},
|
||||
"terminal": "prompt_blocked",
|
||||
"response_model_override": true,
|
||||
"requires_caps": ["prompt_blocked"]
|
||||
},
|
||||
{
|
||||
"name": "stream_prompt_blocked",
|
||||
"usage": {"fresh_input_tokens": 1000, "output_tokens": 0},
|
||||
"stream": true,
|
||||
"terminal": "prompt_blocked",
|
||||
"response_model_override": true,
|
||||
"requires_caps": ["prompt_blocked"]
|
||||
},
|
||||
{
|
||||
"name": "all_components_chat",
|
||||
"usage": {
|
||||
"fresh_input_tokens": 80,
|
||||
"cache_read_tokens": 40,
|
||||
"cache_write_5m_tokens": 20,
|
||||
"cache_write_1h_tokens": 10,
|
||||
"output_tokens": 25,
|
||||
"reasoning_tokens": 15,
|
||||
"audio_input_tokens": 5,
|
||||
"audio_output_tokens": 3
|
||||
},
|
||||
"wires": ["openai_chat", "azure_chat", "together_chat"]
|
||||
},
|
||||
{
|
||||
"name": "all_components_fireworks",
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25},
|
||||
"wires": ["fireworks_chat"]
|
||||
},
|
||||
{
|
||||
"name": "all_components_anthropic",
|
||||
"usage": {
|
||||
"fresh_input_tokens": 80,
|
||||
"cache_read_tokens": 40,
|
||||
"cache_write_5m_tokens": 20,
|
||||
"cache_write_1h_tokens": 10,
|
||||
"output_tokens": 25
|
||||
},
|
||||
"wires": ["anthropic_messages", "bedrock_converse"]
|
||||
},
|
||||
{
|
||||
"name": "all_components_anthropic_stream",
|
||||
"usage": {
|
||||
"fresh_input_tokens": 80,
|
||||
"cache_read_tokens": 40,
|
||||
"cache_write_5m_tokens": 20,
|
||||
"cache_write_1h_tokens": 10,
|
||||
"output_tokens": 25
|
||||
},
|
||||
"stream": true,
|
||||
"wires": ["anthropic_messages"]
|
||||
},
|
||||
{
|
||||
"name": "all_components_gemini",
|
||||
"usage": {
|
||||
"fresh_input_tokens": 80,
|
||||
"cache_read_tokens": 40,
|
||||
"output_tokens": 25,
|
||||
"reasoning_tokens": 15,
|
||||
"audio_input_tokens": 5,
|
||||
"audio_output_tokens": 3
|
||||
},
|
||||
"wires": ["gemini_generate", "vertex_generate"]
|
||||
},
|
||||
{
|
||||
"name": "all_components_responses",
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25, "reasoning_tokens": 15},
|
||||
"wires": ["openai_responses"]
|
||||
}
|
||||
]
|
||||
}
|
||||
|
|
@ -1,10 +1,12 @@
|
|||
"""Cost-calculation suite fixtures.
|
||||
|
||||
Runs against a dedicated proxy whose whole model cost map is the test-owned
|
||||
``tests/e2e/cost_map.json`` (LITELLM_MODEL_COST_MAP_URL), so every deployment
|
||||
bills at rates the test asserts literal arithmetic on. Provider calls are
|
||||
answered by the scripted-provider sidecar (``scripted_provider.py``), registered
|
||||
per scenario over its control API.
|
||||
``tests/e2e/cost_map.json`` (LITELLM_MODEL_COST_MAP_URL); every map entry is a
|
||||
deployment under test, the request shapes live in ``cases.json``, and the
|
||||
asserted goldens live in ``expected.json`` (regenerate proposals with
|
||||
``generate_expected.py``). Provider calls are answered by the
|
||||
scripted-provider sidecar (``scripted_provider.py``), registered per scenario
|
||||
over its control API.
|
||||
|
||||
Deselected unless E2E_COST_MAP_STACK is set (marker `cost_map_stack`).
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -1,18 +1,16 @@
|
|||
"""The cost-calculation matrix: frontier model set, the pricing-component cases
|
||||
each model runs, and the expected-cost arithmetic.
|
||||
"""The cost-calculation matrix: the model set derived from the test cost map,
|
||||
the request/response cases from ``cases.json``, and the loaders both use.
|
||||
|
||||
Rates come from ``tests/e2e/cost_map.json``, which the proxy under test loads as
|
||||
its ENTIRE model cost map (LITELLM_MODEL_COST_MAP_URL), so an entry's rates are
|
||||
exactly what the proxy bills and nothing in the suite depends on the bundled
|
||||
map. Each model's rates are a distinct multiple of a shared base set, so a
|
||||
component billed at the wrong model's rate (or the wrong case's rate) can never
|
||||
coincidentally match.
|
||||
|
||||
Case applicability is pricing-field-gated AND wire-gated: a case runs for a
|
||||
model only when the entry carries the rate the case exercises and the wire can
|
||||
report the token kind that rate prices. When the wire cannot report a kind
|
||||
(e.g. Anthropic has no reasoning-token field, Responses reports no cache
|
||||
creation), the case is absent from the matrix rather than silently zero.
|
||||
Three data files drive the suite; nothing in Python lists models or cases:
|
||||
- ``tests/e2e/cost_map.json`` is the proxy's ENTIRE model cost map
|
||||
(LITELLM_MODEL_COST_MAP_URL); every entry becomes a deployment under test.
|
||||
- ``tests/e2e/cost_calculation/cases.json`` is the case list; each case runs
|
||||
for a model when the entry carries the rates it exercises (``requires_rates``)
|
||||
and the wire can report the token kinds involved (``requires_caps`` /
|
||||
``wires``).
|
||||
- ``tests/e2e/cost_calculation/expected.json`` holds the reviewed goldens; the
|
||||
tests assert them verbatim and never compute a price themselves. The rate
|
||||
arithmetic that proposes goldens lives in ``generate_expected.py``, not here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -26,13 +24,15 @@ from collections.abc import Mapping
|
|||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from types import MappingProxyType
|
||||
from typing import Final, Literal, TypeAlias
|
||||
from typing import Final, Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, TypeAdapter
|
||||
|
||||
from scripted_provider import Scenario, ScriptedOutput, ScriptedToolCall, ScriptedUsage, Wire
|
||||
|
||||
COST_MAP_PATH: Final = Path(__file__).resolve().parent.parent / "cost_map.json"
|
||||
CASES_PATH: Final = Path(__file__).resolve().parent / "cases.json"
|
||||
EXPECTED_PATH: Final = Path(__file__).resolve().parent / "expected.json"
|
||||
|
||||
|
||||
class SearchContextCostPerQuery(BaseModel):
|
||||
|
|
@ -77,119 +77,90 @@ _COST_MAP: Final[Mapping[str, CostMapEntry]] = MappingProxyType(
|
|||
TIER_THRESHOLD_TOKENS: Final = 200_000
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class FrontierModel:
|
||||
"""One deployment under test: the model_name the suite registers, the
|
||||
provider-prefixed litellm model string, the wire the scripted upstream
|
||||
speaks, its cost-map key, and the sibling map model the response_model
|
||||
override case reports."""
|
||||
class DeploymentSpec(BaseModel):
|
||||
"""A deployment-level fact from cases.json: when a map key needs a
|
||||
registered deployment name that is not its provider model (or a
|
||||
model_info.base_model pin), the matrix uses these instead of the defaults."""
|
||||
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
model_name: str
|
||||
litellm_model: str
|
||||
wire: Wire
|
||||
map_key: str
|
||||
override_model: str | None = None
|
||||
override_map_key: str | None = None
|
||||
# Registered as model_info.base_model; when set, the provider-reported
|
||||
# model loses to it and every case bills at this deployment's own rates.
|
||||
litellm_model: str | None = None
|
||||
base_model: str | None = None
|
||||
# Extra litellm_params merged into the /model/new registration (api_version,
|
||||
# aws_* credentials, vertex_* auth).
|
||||
litellm_params: Mapping[str, str] = MappingProxyType({})
|
||||
|
||||
@property
|
||||
def rates(self) -> CostMapEntry:
|
||||
return _COST_MAP[self.map_key]
|
||||
|
||||
@property
|
||||
def override_rates(self) -> CostMapEntry:
|
||||
if self.base_model is not None or self.override_map_key is None:
|
||||
return self.rates
|
||||
return _COST_MAP[self.override_map_key]
|
||||
|
||||
@property
|
||||
def provider_model(self) -> str:
|
||||
"""The bare provider-facing model name: litellm_model minus the provider
|
||||
prefix and any routing segment (converse/, responses/)."""
|
||||
tail: Final = self.litellm_model.split("/")[1:]
|
||||
return "/".join(tail[1:] if tail and tail[0] in ("converse", "responses") else tail)
|
||||
|
||||
@property
|
||||
def provider(self) -> str:
|
||||
return self.rates.litellm_provider
|
||||
|
||||
@property
|
||||
def api_key(self) -> str:
|
||||
# The scripted upstream ignores auth; a fixed bogus key proves the suite
|
||||
# spends zero real provider calls.
|
||||
return "sk-scripted-provider"
|
||||
|
||||
|
||||
# Response-model override targets: emit a sibling's bare provider-facing name so
|
||||
# the biller's provider-prefixed lookup lands on that sibling's map key.
|
||||
_OVERRIDE_MODELS: Final[Mapping[str, str]] = MappingProxyType({
|
||||
"gpt-5.6": "gpt-5.4-mini",
|
||||
"gpt-5.5-pro": "gpt-5.3-codex",
|
||||
"gpt-5.3-codex": "gpt-5.5-pro",
|
||||
"gpt-5.4-mini": "gpt-5.6",
|
||||
"claude-opus-5": "claude-sonnet-5",
|
||||
"claude-sonnet-5": "claude-opus-5",
|
||||
"claude-haiku-4-5": "claude-sonnet-5",
|
||||
"gemini/gemini-3.8-flash": "gemini-3.1-pro-preview",
|
||||
"gemini/gemini-3.1-pro-preview": "gemini-3.8-flash",
|
||||
"together_ai/moonshotai/Kimi-K3": "zai-org/GLM-5.3",
|
||||
"together_ai/zai-org/GLM-5.3": "moonshotai/Kimi-K3",
|
||||
"fireworks_ai/kimi-k3": "qwen3p8-max",
|
||||
"fireworks_ai/qwen3p8-max": "kimi-k3",
|
||||
"fireworks_ai/deepseek-v4p1-flash": "kimi-k3",
|
||||
})
|
||||
class Case(BaseModel):
|
||||
"""One request/response shape from cases.json; gated onto a model by
|
||||
``requires_rates`` (entry must carry each rate field), ``requires_caps``
|
||||
(the wire must report the token kind) and ``wires`` (shape is wire-specific)."""
|
||||
|
||||
_OVERRIDE_MAP_KEYS: Final[Mapping[str, str]] = MappingProxyType({
|
||||
"gpt-5.4-mini": "gpt-5.4-mini",
|
||||
"gpt-5.6": "gpt-5.6",
|
||||
"gpt-5.3-codex": "gpt-5.3-codex",
|
||||
"gpt-5.5-pro": "gpt-5.5-pro",
|
||||
"claude-sonnet-5": "claude-sonnet-5",
|
||||
"claude-opus-5": "claude-opus-5",
|
||||
"gemini-3.1-pro-preview": "gemini/gemini-3.1-pro-preview",
|
||||
"gemini-3.8-flash": "gemini/gemini-3.8-flash",
|
||||
"zai-org/GLM-5.3": "together_ai/zai-org/GLM-5.3",
|
||||
"moonshotai/Kimi-K3": "together_ai/moonshotai/Kimi-K3",
|
||||
"qwen3p8-max": "fireworks_ai/qwen3p8-max",
|
||||
"kimi-k3": "fireworks_ai/kimi-k3",
|
||||
})
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
name: str
|
||||
usage: ScriptedUsage
|
||||
stream: bool = False
|
||||
stream_usage: Literal["final_chunk", "absent"] = "final_chunk"
|
||||
service_tier: Literal["flex", "priority"] | None = None
|
||||
response_model_override: bool = False
|
||||
exact_spend: bool = True
|
||||
tool_call: bool = False
|
||||
image_input: bool = False
|
||||
terminal: Literal["completed", "incomplete", "unvalidated", "prompt_blocked"] = "completed"
|
||||
requires_rates: tuple[str, ...] = ()
|
||||
requires_caps: tuple[str, ...] = ()
|
||||
wires: tuple[Wire, ...] | None = None
|
||||
|
||||
def applies_to(self, model: FrontierModel) -> bool:
|
||||
if self.wires is not None and model.wire not in self.wires:
|
||||
return False
|
||||
caps: Final = _WIRE_CAPS[model.wire]
|
||||
if not frozenset(self.requires_caps) <= caps:
|
||||
return False
|
||||
return all(
|
||||
getattr(model.rates, field, None) is not None for field in self.requires_rates
|
||||
)
|
||||
|
||||
def scenario(self, scenario_id: str, model: FrontierModel, text: str) -> Scenario:
|
||||
return Scenario(
|
||||
scenario_id=scenario_id,
|
||||
wire=model.wire,
|
||||
usage=self.usage,
|
||||
model=model.provider_model,
|
||||
output=ScriptedOutput(
|
||||
text=text,
|
||||
response_model=model.override_model if self.response_model_override else None,
|
||||
tool_call=ScriptedToolCall(name="get_weather", arguments=TOOL_CALL_ARGUMENTS)
|
||||
if self.tool_call
|
||||
else None,
|
||||
terminal=self.terminal,
|
||||
),
|
||||
stream_usage=self.stream_usage,
|
||||
service_tier=self.service_tier,
|
||||
)
|
||||
|
||||
|
||||
_FRONTIER_SPECS: Final[tuple[tuple[str, str, Wire], ...]] = (
|
||||
("gpt-5.6", "openai/gpt-5.6", "openai_chat"),
|
||||
("gpt-5.5-pro", "openai/gpt-5.5-pro", "openai_responses"),
|
||||
("gpt-5.3-codex", "openai/gpt-5.3-codex", "openai_responses"),
|
||||
("gpt-5.4-mini", "openai/gpt-5.4-mini", "openai_chat"),
|
||||
("claude-opus-5", "anthropic/claude-opus-5", "anthropic_messages"),
|
||||
("claude-sonnet-5", "anthropic/claude-sonnet-5", "anthropic_messages"),
|
||||
("claude-haiku-4-5", "anthropic/claude-haiku-4-5", "anthropic_messages"),
|
||||
("gemini/gemini-3.8-flash", "gemini/gemini-3.8-flash", "gemini_generate"),
|
||||
("gemini/gemini-3.1-pro-preview", "gemini/gemini-3.1-pro-preview", "gemini_generate"),
|
||||
("together_ai/moonshotai/Kimi-K3", "together_ai/moonshotai/Kimi-K3", "together_chat"),
|
||||
("together_ai/zai-org/GLM-5.3", "together_ai/zai-org/GLM-5.3", "together_chat"),
|
||||
("fireworks_ai/kimi-k3", "fireworks_ai/kimi-k3", "fireworks_chat"),
|
||||
("fireworks_ai/qwen3p8-max", "fireworks_ai/qwen3p8-max", "fireworks_chat"),
|
||||
("fireworks_ai/deepseek-v4p1-flash", "fireworks_ai/deepseek-v4p1-flash", "fireworks_chat"),
|
||||
class _CasesFile(BaseModel):
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
deployments: tuple[DeploymentSpec, ...] = ()
|
||||
cases: tuple[Case, ...] = ()
|
||||
|
||||
|
||||
_CASES_FILE: Final = _CasesFile.model_validate(json.loads(CASES_PATH.read_text()))
|
||||
CASES: Final[tuple[Case, ...]] = _CASES_FILE.cases
|
||||
_DEPLOYMENTS: Final[Mapping[str, DeploymentSpec]] = MappingProxyType(
|
||||
{spec.map_key: spec for spec in _CASES_FILE.deployments}
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _ExtendedSpec:
|
||||
"""A frontier entry whose override target, model_info.base_model or extra
|
||||
litellm_params can't be derived from the map key alone."""
|
||||
class _ProviderWiring:
|
||||
"""How a (litellm_provider, mode) pair maps to a sidecar wire, the provider
|
||||
prefix on the registered litellm model string, and extra litellm_params."""
|
||||
|
||||
map_key: str
|
||||
litellm_model: str
|
||||
wire: Wire
|
||||
override_model: str | None = None
|
||||
override_map_key: str | None = None
|
||||
base_model: str | None = None
|
||||
litellm_params: Mapping[str, str] = MappingProxyType({})
|
||||
model_prefix: str | None
|
||||
litellm_params: Mapping[str, str]
|
||||
|
||||
|
||||
_AZURE_PARAMS: Final[Mapping[str, str]] = MappingProxyType({"api_version": "2025-04-01-preview"})
|
||||
|
|
@ -207,87 +178,134 @@ _VERTEX_PARAMS: Final[Mapping[str, str]] = MappingProxyType(
|
|||
}
|
||||
)
|
||||
|
||||
_EXTENDED_SPECS: Final[tuple[_ExtendedSpec, ...]] = (
|
||||
_ExtendedSpec(
|
||||
map_key="azure/gpt-5.6",
|
||||
litellm_model="azure/gpt-5.6",
|
||||
wire="azure_chat",
|
||||
override_model="gpt-5.4-mini",
|
||||
override_map_key="azure/gpt-5.4-mini",
|
||||
litellm_params=_AZURE_PARAMS,
|
||||
),
|
||||
_ExtendedSpec(
|
||||
# Deployment name is not a model; base_model pins billing so the
|
||||
# response's model field loses, proving base_model wins.
|
||||
map_key="azure/gpt-5.4-mini",
|
||||
litellm_model="azure/cc-pinned-deployment",
|
||||
wire="azure_chat",
|
||||
override_model="gpt-5.6",
|
||||
override_map_key="azure/gpt-5.6",
|
||||
base_model="azure/gpt-5.4-mini",
|
||||
litellm_params=_AZURE_PARAMS,
|
||||
),
|
||||
_ExtendedSpec(
|
||||
map_key="anthropic.claude-sonnet-5-v1:0",
|
||||
litellm_model="bedrock/converse/anthropic.claude-sonnet-5-v1:0",
|
||||
wire="bedrock_converse",
|
||||
litellm_params=_BEDROCK_PARAMS,
|
||||
),
|
||||
_ExtendedSpec(
|
||||
map_key="us.anthropic.claude-opus-5-v1:0",
|
||||
litellm_model="bedrock/converse/us.anthropic.claude-opus-5-v1:0",
|
||||
wire="bedrock_converse",
|
||||
litellm_params=_BEDROCK_PARAMS,
|
||||
),
|
||||
_ExtendedSpec(
|
||||
map_key="meta.llama4-maverick-17b-instruct-v1:0",
|
||||
litellm_model="bedrock/converse/meta.llama4-maverick-17b-instruct-v1:0",
|
||||
wire="bedrock_converse",
|
||||
litellm_params=_BEDROCK_PARAMS,
|
||||
),
|
||||
_ExtendedSpec(
|
||||
map_key="gemini-3.8-flash",
|
||||
litellm_model="vertex_ai/gemini-3.8-flash",
|
||||
wire="vertex_generate",
|
||||
override_model="gemini-3.1-pro-preview",
|
||||
override_map_key="gemini-3.1-pro-preview",
|
||||
litellm_params=_VERTEX_PARAMS,
|
||||
),
|
||||
_ExtendedSpec(
|
||||
map_key="gemini-3.1-pro-preview",
|
||||
litellm_model="vertex_ai/gemini-3.1-pro-preview",
|
||||
wire="vertex_generate",
|
||||
override_model="gemini-3.8-flash",
|
||||
override_map_key="gemini-3.8-flash",
|
||||
litellm_params=_VERTEX_PARAMS,
|
||||
),
|
||||
_PROVIDER_WIRING: Final[Mapping[tuple[str, str], _ProviderWiring]] = MappingProxyType(
|
||||
{
|
||||
("openai", "chat"): _ProviderWiring("openai_chat", "openai", MappingProxyType({})),
|
||||
("openai", "responses"): _ProviderWiring(
|
||||
"openai_responses", "openai", MappingProxyType({})
|
||||
),
|
||||
("anthropic", "chat"): _ProviderWiring(
|
||||
"anthropic_messages", "anthropic", MappingProxyType({})
|
||||
),
|
||||
("gemini", "chat"): _ProviderWiring("gemini_generate", None, MappingProxyType({})),
|
||||
("together_ai", "chat"): _ProviderWiring("together_chat", None, MappingProxyType({})),
|
||||
("fireworks_ai", "chat"): _ProviderWiring("fireworks_chat", None, MappingProxyType({})),
|
||||
("azure", "chat"): _ProviderWiring("azure_chat", None, _AZURE_PARAMS),
|
||||
("bedrock_converse", "chat"): _ProviderWiring(
|
||||
"bedrock_converse", "bedrock/converse", _BEDROCK_PARAMS
|
||||
),
|
||||
("vertex_ai-language-models", "chat"): _ProviderWiring(
|
||||
"vertex_generate", "vertex_ai", _VERTEX_PARAMS
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class FrontierModel:
|
||||
"""One deployment under test, derived from a cost-map entry: the model_name
|
||||
the suite registers, the provider-prefixed litellm model string, the wire
|
||||
the scripted upstream speaks, and the sibling map model the response_model
|
||||
override case reports."""
|
||||
|
||||
model_name: str
|
||||
litellm_model: str
|
||||
wire: Wire
|
||||
map_key: str
|
||||
override_model: str | None = None
|
||||
override_map_key: str | None = None
|
||||
# Registered as model_info.base_model; when set, the provider-reported
|
||||
# model loses to it and every case bills at this deployment's own rates.
|
||||
base_model: str | None = None
|
||||
litellm_params: Mapping[str, str] = MappingProxyType({})
|
||||
|
||||
@property
|
||||
def rates(self) -> CostMapEntry:
|
||||
return _COST_MAP[self.map_key]
|
||||
|
||||
@property
|
||||
def override_rates(self) -> CostMapEntry:
|
||||
if self.base_model is not None or self.override_map_key is None:
|
||||
return self.rates
|
||||
return _COST_MAP[self.override_map_key]
|
||||
|
||||
@property
|
||||
def provider_model(self) -> str:
|
||||
"""The bare provider-facing model name: litellm_model minus the provider
|
||||
prefix and any routing segment (converse/, responses/)."""
|
||||
return _provider_model(self.litellm_model)
|
||||
|
||||
@property
|
||||
def provider(self) -> str:
|
||||
return self.rates.litellm_provider
|
||||
|
||||
@property
|
||||
def api_key(self) -> str:
|
||||
# The scripted upstream ignores auth; a fixed bogus key proves the suite
|
||||
# spends zero real provider calls.
|
||||
return "sk-scripted-provider"
|
||||
|
||||
|
||||
def _provider_model(litellm_model: str) -> str:
|
||||
tail: Final = litellm_model.split("/")[1:]
|
||||
return "/".join(tail[1:] if tail and tail[0] in ("converse", "responses") else tail)
|
||||
|
||||
|
||||
def _litellm_model_for(map_key: str, wiring: _ProviderWiring) -> str:
|
||||
if wiring.model_prefix is None:
|
||||
return map_key
|
||||
if map_key.startswith(f"{wiring.model_prefix}/"):
|
||||
return map_key
|
||||
return f"{wiring.model_prefix}/{map_key}"
|
||||
|
||||
|
||||
def _frontier() -> tuple[FrontierModel, ...]:
|
||||
return tuple(
|
||||
FrontierModel(
|
||||
model_name=f"cc-{map_key.replace('/', '-').lower()}",
|
||||
litellm_model=litellm_model,
|
||||
wire=wire,
|
||||
map_key=map_key,
|
||||
override_model=_OVERRIDE_MODELS[map_key],
|
||||
override_map_key=_OVERRIDE_MAP_KEYS[_OVERRIDE_MODELS[map_key]],
|
||||
)
|
||||
for map_key, litellm_model, wire in _FRONTIER_SPECS
|
||||
) + tuple(
|
||||
FrontierModel(
|
||||
model_name=f"cc-{spec.map_key.replace('/', '-').replace(':', '-').replace('.', '-').lower()}",
|
||||
litellm_model=spec.litellm_model,
|
||||
wire=spec.wire,
|
||||
map_key=spec.map_key,
|
||||
override_model=spec.override_model,
|
||||
override_map_key=spec.override_map_key,
|
||||
base_model=spec.base_model,
|
||||
litellm_params=spec.litellm_params,
|
||||
)
|
||||
for spec in _EXTENDED_SPECS
|
||||
groups: Final[Mapping[tuple[str, str], tuple[str, ...]]] = MappingProxyType(
|
||||
{
|
||||
pair: tuple(sorted(k for k, e in _COST_MAP.items() if (e.litellm_provider, e.mode) == pair))
|
||||
for pair in {(e.litellm_provider, e.mode) for e in _COST_MAP.values()}
|
||||
}
|
||||
)
|
||||
models: list[FrontierModel] = [] # mutable-ok: accumulated once at import into a tuple
|
||||
for map_key in sorted(_COST_MAP):
|
||||
entry: Final = _COST_MAP[map_key]
|
||||
pair: Final = (entry.litellm_provider, entry.mode)
|
||||
wiring: Final = _PROVIDER_WIRING.get(pair)
|
||||
if wiring is None:
|
||||
raise ValueError(
|
||||
f"cost_map entry {map_key} has no wiring for "
|
||||
f"(litellm_provider={pair[0]}, mode={pair[1]}); add a "
|
||||
f"_ProviderWiring row in cost_matrix.py"
|
||||
)
|
||||
siblings: Final = groups[pair]
|
||||
override_key: Final = (
|
||||
siblings[(siblings.index(map_key) + 1) % len(siblings)] if len(siblings) > 1 else None
|
||||
)
|
||||
override_litellm: Final = (
|
||||
_litellm_model_for(override_key, wiring) if override_key is not None else None
|
||||
)
|
||||
deployment: Final = _DEPLOYMENTS.get(map_key)
|
||||
models.append(
|
||||
FrontierModel(
|
||||
model_name=f"cc-{map_key.replace('/', '-').replace(':', '-').replace('.', '-').lower()}",
|
||||
litellm_model=(
|
||||
deployment.litellm_model
|
||||
if deployment is not None and deployment.litellm_model is not None
|
||||
else _litellm_model_for(map_key, wiring)
|
||||
),
|
||||
wire=wiring.wire,
|
||||
map_key=map_key,
|
||||
override_model=(
|
||||
_provider_model(override_litellm)
|
||||
if override_litellm is not None
|
||||
else None
|
||||
),
|
||||
override_map_key=override_key,
|
||||
base_model=deployment.base_model if deployment is not None else None,
|
||||
litellm_params=wiring.litellm_params,
|
||||
)
|
||||
)
|
||||
return tuple(models)
|
||||
|
||||
|
||||
FRONTIER_MODELS: Final[tuple[FrontierModel, ...]] = _frontier()
|
||||
|
|
@ -350,71 +368,6 @@ _WIRE_CAPS: Final[Mapping[str, frozenset[str]]] = MappingProxyType({
|
|||
),
|
||||
})
|
||||
|
||||
CaseName: TypeAlias = Literal[
|
||||
"basic",
|
||||
"cache_read",
|
||||
"cache_write_5m",
|
||||
"cache_write_1h",
|
||||
"reasoning",
|
||||
"audio",
|
||||
"tiered",
|
||||
"service_tier_flex",
|
||||
"service_tier_priority",
|
||||
"web_search",
|
||||
"stream",
|
||||
"stream_no_usage",
|
||||
"response_model_override",
|
||||
"stream_response_model_override",
|
||||
"tool_call",
|
||||
"stream_no_usage_tool_call",
|
||||
"stream_no_usage_image_input",
|
||||
"stream_no_usage_incomplete",
|
||||
"stream_unvalidated",
|
||||
"stream_no_usage_unvalidated",
|
||||
"prompt_blocked",
|
||||
"stream_prompt_blocked",
|
||||
]
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Case:
|
||||
name: CaseName
|
||||
usage: ScriptedUsage
|
||||
stream: bool = False
|
||||
stream_usage: Literal["final_chunk", "absent"] = "final_chunk"
|
||||
service_tier: Literal["flex", "priority"] | None = None
|
||||
# For web_search the wire's reported call count is not always what gets
|
||||
# billed: chat-completions surfaces only expose url_citation annotations, so
|
||||
# the biller floors to one call; responses/messages/gemini report a real
|
||||
# count.
|
||||
billed_web_search_calls: int = 0
|
||||
response_model_override: bool = False
|
||||
exact_spend: bool = True
|
||||
tool_call: bool = False
|
||||
image_input: bool = False
|
||||
terminal: Literal["completed", "incomplete", "unvalidated", "prompt_blocked"] = "completed"
|
||||
|
||||
def scenario(self, scenario_id: str, model: FrontierModel, text: str) -> Scenario:
|
||||
return Scenario(
|
||||
scenario_id=scenario_id,
|
||||
wire=model.wire,
|
||||
usage=self.usage,
|
||||
model=model.provider_model,
|
||||
output=ScriptedOutput(
|
||||
text=text,
|
||||
response_model=model.override_model if self.response_model_override else None,
|
||||
tool_call=ScriptedToolCall(name="get_weather", arguments=TOOL_CALL_ARGUMENTS)
|
||||
if self.tool_call
|
||||
else None,
|
||||
terminal=self.terminal,
|
||||
),
|
||||
stream_usage=self.stream_usage,
|
||||
service_tier=self.service_tier,
|
||||
)
|
||||
|
||||
|
||||
_BASIC_USAGE: Final = ScriptedUsage(fresh_input_tokens=120, output_tokens=40)
|
||||
|
||||
TOOL_CALL_ARGUMENTS: Final = json.dumps({
|
||||
"city": "Berlin",
|
||||
"days": 7,
|
||||
|
|
@ -422,284 +375,9 @@ TOOL_CALL_ARGUMENTS: Final = json.dumps({
|
|||
"notes": "filler " * 30,
|
||||
})
|
||||
|
||||
_PROMPT_BLOCKED_USAGE: Final = ScriptedUsage(fresh_input_tokens=1000, output_tokens=0)
|
||||
|
||||
|
||||
def _web_search_case(model: FrontierModel) -> Case:
|
||||
counts_exactly: Final = model.wire in (
|
||||
"openai_responses", "anthropic_messages", "gemini_generate", "vertex_generate"
|
||||
)
|
||||
return Case(
|
||||
name="web_search",
|
||||
usage=ScriptedUsage(fresh_input_tokens=100, output_tokens=30, web_search_calls=3),
|
||||
billed_web_search_calls=3 if counts_exactly else 1,
|
||||
)
|
||||
|
||||
|
||||
def cases_for(model: FrontierModel) -> tuple[Case, ...]:
|
||||
rates: Final = model.rates
|
||||
caps: Final = _WIRE_CAPS[model.wire]
|
||||
candidates: Final[tuple[Case | None, ...]] = (
|
||||
Case(name="basic", usage=_BASIC_USAGE),
|
||||
(
|
||||
Case(name="cache_read", usage=ScriptedUsage(fresh_input_tokens=100, cache_read_tokens=50, output_tokens=30))
|
||||
if rates.cache_read_input_token_cost is not None and "cache_read" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="cache_write_5m",
|
||||
usage=ScriptedUsage(fresh_input_tokens=90, cache_write_5m_tokens=60, output_tokens=30),
|
||||
)
|
||||
if rates.cache_creation_input_token_cost is not None and "cache_write_5m" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="cache_write_1h",
|
||||
usage=ScriptedUsage(
|
||||
fresh_input_tokens=90,
|
||||
cache_write_5m_tokens=20,
|
||||
cache_write_1h_tokens=40,
|
||||
output_tokens=30,
|
||||
),
|
||||
)
|
||||
if (
|
||||
rates.cache_creation_input_token_cost_above_1hr is not None
|
||||
and rates.cache_creation_input_token_cost is not None
|
||||
and "cache_write_1h" in caps
|
||||
)
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="reasoning",
|
||||
usage=ScriptedUsage(fresh_input_tokens=100, output_tokens=30, reasoning_tokens=70),
|
||||
)
|
||||
if rates.output_cost_per_reasoning_token is not None and "reasoning" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="audio",
|
||||
usage=ScriptedUsage(
|
||||
fresh_input_tokens=100, audio_input_tokens=25, output_tokens=30, audio_output_tokens=15
|
||||
),
|
||||
)
|
||||
if (
|
||||
rates.input_cost_per_audio_token is not None
|
||||
and rates.output_cost_per_audio_token is not None
|
||||
and "audio" in caps
|
||||
)
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="tiered",
|
||||
usage=ScriptedUsage(
|
||||
fresh_input_tokens=TIER_THRESHOLD_TOKENS + 1, output_tokens=30
|
||||
),
|
||||
)
|
||||
if (
|
||||
rates.input_cost_per_token_above_200k_tokens is not None
|
||||
and rates.output_cost_per_token_above_200k_tokens is not None
|
||||
)
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(name="service_tier_flex", usage=_BASIC_USAGE, service_tier="flex")
|
||||
if rates.input_cost_per_token_flex is not None and rates.output_cost_per_token_flex is not None
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(name="service_tier_priority", usage=_BASIC_USAGE, service_tier="priority")
|
||||
if rates.input_cost_per_token_priority is not None and rates.output_cost_per_token_priority is not None
|
||||
else None
|
||||
),
|
||||
_web_search_case(model) if rates.search_context_cost_per_query is not None and "web_search" in caps else None,
|
||||
Case(name="stream", usage=_BASIC_USAGE, stream=True),
|
||||
(
|
||||
Case(
|
||||
name="stream_no_usage",
|
||||
usage=_BASIC_USAGE,
|
||||
stream=True,
|
||||
stream_usage="absent",
|
||||
exact_spend=False,
|
||||
)
|
||||
if "absent_usage" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(name="response_model_override", usage=_BASIC_USAGE, response_model_override=True)
|
||||
if "response_model" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="stream_response_model_override",
|
||||
usage=_BASIC_USAGE,
|
||||
stream=True,
|
||||
response_model_override=True,
|
||||
)
|
||||
if "response_model" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(name="tool_call", usage=_BASIC_USAGE, tool_call=True)
|
||||
if "tool_call" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="stream_no_usage_tool_call",
|
||||
usage=_BASIC_USAGE,
|
||||
stream=True,
|
||||
stream_usage="absent",
|
||||
tool_call=True,
|
||||
exact_spend=False,
|
||||
)
|
||||
if "absent_usage" in caps and "tool_call" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="stream_no_usage_image_input",
|
||||
usage=_BASIC_USAGE,
|
||||
stream=True,
|
||||
stream_usage="absent",
|
||||
image_input=True,
|
||||
exact_spend=False,
|
||||
)
|
||||
if "absent_usage" in caps and "image_input" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="stream_no_usage_incomplete",
|
||||
usage=_BASIC_USAGE,
|
||||
stream=True,
|
||||
stream_usage="absent",
|
||||
terminal="incomplete",
|
||||
exact_spend=False,
|
||||
)
|
||||
if "responses_terminal" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="stream_unvalidated",
|
||||
usage=_BASIC_USAGE,
|
||||
stream=True,
|
||||
terminal="unvalidated",
|
||||
)
|
||||
if "responses_terminal" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="stream_no_usage_unvalidated",
|
||||
usage=_BASIC_USAGE,
|
||||
stream=True,
|
||||
stream_usage="absent",
|
||||
terminal="unvalidated",
|
||||
exact_spend=False,
|
||||
)
|
||||
if "responses_terminal" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="prompt_blocked",
|
||||
usage=_PROMPT_BLOCKED_USAGE,
|
||||
terminal="prompt_blocked",
|
||||
response_model_override=True,
|
||||
)
|
||||
if "prompt_blocked" in caps
|
||||
else None
|
||||
),
|
||||
(
|
||||
Case(
|
||||
name="stream_prompt_blocked",
|
||||
usage=_PROMPT_BLOCKED_USAGE,
|
||||
stream=True,
|
||||
terminal="prompt_blocked",
|
||||
response_model_override=True,
|
||||
)
|
||||
if "prompt_blocked" in caps
|
||||
else None
|
||||
),
|
||||
)
|
||||
return tuple(case for case in candidates if case is not None)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ExpectedCost:
|
||||
"""The expected bill split the way the spend row's cost_breakdown reports
|
||||
it: the gross input component (cache reads/writes folded in), the output
|
||||
component, and the tool-usage component."""
|
||||
|
||||
input_cost: float
|
||||
output_cost: float
|
||||
tool_cost: float
|
||||
|
||||
@property
|
||||
def total(self) -> float:
|
||||
return self.input_cost + self.output_cost + self.tool_cost
|
||||
|
||||
|
||||
def expected_breakdown(model: FrontierModel, case: Case) -> ExpectedCost:
|
||||
"""Literal arithmetic on the test-map rates over the scripted token counts.
|
||||
|
||||
Input = fresh*in + read*read + 5m*create + 1h*create_1h + audio_in*audio_in;
|
||||
output = text*out + reasoning*reasoning + audio_out*audio_out; plus the
|
||||
billed web-search calls at the medium search-context rate. Above-threshold
|
||||
swaps every input/output rate to its ``_above_200k_tokens`` variant when
|
||||
total prompt tokens exceed the threshold; a service tier swaps input/output
|
||||
to the tier's variants, falling back to the base rate when a variant is
|
||||
unset -- mirroring _get_token_base_cost in litellm's cost calculator.
|
||||
"""
|
||||
rates: Final = model.override_rates if case.response_model_override else model.rates
|
||||
u: Final = case.usage
|
||||
prompt_tokens: Final = (
|
||||
u.fresh_input_tokens + u.cache_read_tokens + u.cache_write_5m_tokens
|
||||
+ u.cache_write_1h_tokens + u.audio_input_tokens
|
||||
)
|
||||
tiered: Final = prompt_tokens > TIER_THRESHOLD_TOKENS
|
||||
in_rate: Final = (
|
||||
(rates.input_cost_per_token_above_200k_tokens if tiered else None)
|
||||
or (rates.input_cost_per_token_priority if case.service_tier == "priority" else None)
|
||||
or (rates.input_cost_per_token_flex if case.service_tier == "flex" else None)
|
||||
or rates.input_cost_per_token
|
||||
or 0.0
|
||||
)
|
||||
out_rate: Final = (
|
||||
(rates.output_cost_per_token_above_200k_tokens if tiered else None)
|
||||
or (rates.output_cost_per_token_priority if case.service_tier == "priority" else None)
|
||||
or (rates.output_cost_per_token_flex if case.service_tier == "flex" else None)
|
||||
or rates.output_cost_per_token
|
||||
or 0.0
|
||||
)
|
||||
input_cost: Final = (
|
||||
u.fresh_input_tokens * in_rate
|
||||
+ u.cache_read_tokens * (rates.cache_read_input_token_cost or 0.0)
|
||||
+ u.cache_write_5m_tokens * (rates.cache_creation_input_token_cost or 0.0)
|
||||
+ u.cache_write_1h_tokens * (rates.cache_creation_input_token_cost_above_1hr or 0.0)
|
||||
+ u.audio_input_tokens * (rates.input_cost_per_audio_token or 0.0)
|
||||
)
|
||||
output_cost: Final = (
|
||||
u.output_tokens * out_rate
|
||||
+ u.reasoning_tokens * (rates.output_cost_per_reasoning_token or out_rate)
|
||||
+ u.audio_output_tokens * (rates.output_cost_per_audio_token or out_rate)
|
||||
)
|
||||
search: Final = rates.search_context_cost_per_query
|
||||
tool_cost: Final = case.billed_web_search_calls * (
|
||||
search.search_context_size_medium if search and search.search_context_size_medium else 0.0
|
||||
)
|
||||
return ExpectedCost(input_cost=input_cost, output_cost=output_cost, tool_cost=tool_cost)
|
||||
|
||||
|
||||
def expected_cost(model: FrontierModel, case: Case) -> float:
|
||||
return expected_breakdown(model, case).total
|
||||
return tuple(case for case in CASES if case.applies_to(model))
|
||||
|
||||
|
||||
def recount_cost(
|
||||
|
|
@ -738,31 +416,23 @@ def image_input_data_url() -> str:
|
|||
IMAGE_INPUT_DATA_URL: Final = image_input_data_url()
|
||||
|
||||
|
||||
def expected_token_columns(model: FrontierModel, case: Case) -> tuple[int, int]:
|
||||
"""(prompt_tokens, completion_tokens) the spend row should carry, per the
|
||||
wire's normalization: Anthropic folds cache read/write into prompt_tokens,
|
||||
everyone else reports the totals the wire emitted."""
|
||||
u: Final = case.usage
|
||||
if model.wire in ("anthropic_messages", "bedrock_converse"):
|
||||
return (
|
||||
u.fresh_input_tokens + u.cache_read_tokens + u.cache_write_5m_tokens + u.cache_write_1h_tokens,
|
||||
u.output_tokens,
|
||||
)
|
||||
if model.wire in ("gemini_generate", "vertex_generate"):
|
||||
return (
|
||||
u.fresh_input_tokens + u.cache_read_tokens + u.audio_input_tokens,
|
||||
u.output_tokens + u.reasoning_tokens + u.audio_output_tokens,
|
||||
)
|
||||
if model.wire == "openai_responses":
|
||||
return (
|
||||
u.fresh_input_tokens + u.cache_read_tokens,
|
||||
u.output_tokens + u.reasoning_tokens,
|
||||
)
|
||||
return (
|
||||
u.fresh_input_tokens
|
||||
+ u.cache_read_tokens
|
||||
+ u.cache_write_5m_tokens
|
||||
+ u.cache_write_1h_tokens
|
||||
+ u.audio_input_tokens,
|
||||
u.output_tokens + u.reasoning_tokens + u.audio_output_tokens,
|
||||
)
|
||||
class _ExpectedCell(BaseModel):
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
spend: float
|
||||
input_cost: float
|
||||
output_cost: float
|
||||
prompt_tokens: int
|
||||
completion_tokens: int
|
||||
|
||||
|
||||
_EXPECTED_ADAPTER: Final = TypeAdapter(dict[str, _ExpectedCell])
|
||||
EXPECTED: Final[Mapping[str, _ExpectedCell]] = MappingProxyType(
|
||||
_EXPECTED_ADAPTER.validate_python(json.loads(EXPECTED_PATH.read_text()))
|
||||
if EXPECTED_PATH.exists()
|
||||
else {}
|
||||
)
|
||||
|
||||
|
||||
def expected_key(model: FrontierModel, case: Case) -> str:
|
||||
return f"{model.map_key}|{case.name}"
|
||||
|
|
|
|||
2004
tests/e2e/cost_calculation/expected.json
Normal file
2004
tests/e2e/cost_calculation/expected.json
Normal file
File diff suppressed because it is too large
Load diff
189
tests/e2e/cost_calculation/generate_expected.py
Normal file
189
tests/e2e/cost_calculation/generate_expected.py
Normal file
|
|
@ -0,0 +1,189 @@
|
|||
"""Golden generator for the cost suite. Run:
|
||||
|
||||
uv run python tests/e2e/cost_calculation/generate_expected.py
|
||||
|
||||
Loads the derived matrix (models x applicable cases), computes the golden for
|
||||
each exact-spend cell from the rate arithmetic, and writes ``expected.json``
|
||||
with sorted keys. Default behaviour adds missing cells and drops stale cells
|
||||
but never overwrites an existing cell's values (a reviewed golden is
|
||||
authoritative); ``--rewrite`` recomputes everything. Prints added/removed/kept
|
||||
counts.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from cost_matrix import ( # noqa: E402 # path bootstrap before package-local imports
|
||||
EXPECTED_PATH,
|
||||
FRONTIER_MODELS,
|
||||
TIER_THRESHOLD_TOKENS,
|
||||
Case,
|
||||
CostMapEntry,
|
||||
FrontierModel,
|
||||
cases_for,
|
||||
expected_key,
|
||||
)
|
||||
|
||||
# Wires whose response surface reports a real web-search call count; the
|
||||
# chat-completions wires only expose url_citation annotations, so their billed
|
||||
# count floors to one.
|
||||
_EXACT_WEB_SEARCH_WIRES: Final = frozenset(
|
||||
{"openai_responses", "anthropic_messages", "gemini_generate", "vertex_generate"}
|
||||
)
|
||||
|
||||
|
||||
def billed_web_search_calls(model: FrontierModel, case: Case) -> int:
|
||||
if case.usage.web_search_calls == 0:
|
||||
return 0
|
||||
return case.usage.web_search_calls if model.wire in _EXACT_WEB_SEARCH_WIRES else 1
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ExpectedCost:
|
||||
"""The expected bill split the way the spend row's cost_breakdown reports
|
||||
it: the gross input component (cache reads/writes folded in), the output
|
||||
component, and the tool-usage component."""
|
||||
|
||||
input_cost: float
|
||||
output_cost: float
|
||||
tool_cost: float
|
||||
|
||||
@property
|
||||
def total(self) -> float:
|
||||
return self.input_cost + self.output_cost + self.tool_cost
|
||||
|
||||
|
||||
def expected_breakdown(model: FrontierModel, case: Case) -> ExpectedCost:
|
||||
"""Literal arithmetic on the test-map rates over the scripted token counts.
|
||||
|
||||
Input = fresh*in + read*read + 5m*create + 1h*create_1h + audio_in*audio_in;
|
||||
output = text*out + reasoning*reasoning + audio_out*audio_out; plus the
|
||||
billed web-search calls at the medium search-context rate. Above-threshold
|
||||
swaps every input/output rate to its ``_above_200k_tokens`` variant when
|
||||
total prompt tokens exceed the threshold; a service tier swaps input/output
|
||||
to the tier's variants, falling back to the base rate when a variant is
|
||||
unset -- mirroring _get_token_base_cost in litellm's cost calculator.
|
||||
"""
|
||||
rates: Final[CostMapEntry] = model.override_rates if case.response_model_override else model.rates
|
||||
u: Final = case.usage
|
||||
prompt_tokens: Final = (
|
||||
u.fresh_input_tokens + u.cache_read_tokens + u.cache_write_5m_tokens
|
||||
+ u.cache_write_1h_tokens + u.audio_input_tokens
|
||||
)
|
||||
tiered: Final = prompt_tokens > TIER_THRESHOLD_TOKENS
|
||||
in_rate: Final = (
|
||||
(rates.input_cost_per_token_above_200k_tokens if tiered else None)
|
||||
or (rates.input_cost_per_token_priority if case.service_tier == "priority" else None)
|
||||
or (rates.input_cost_per_token_flex if case.service_tier == "flex" else None)
|
||||
or rates.input_cost_per_token
|
||||
or 0.0
|
||||
)
|
||||
out_rate: Final = (
|
||||
(rates.output_cost_per_token_above_200k_tokens if tiered else None)
|
||||
or (rates.output_cost_per_token_priority if case.service_tier == "priority" else None)
|
||||
or (rates.output_cost_per_token_flex if case.service_tier == "flex" else None)
|
||||
or rates.output_cost_per_token
|
||||
or 0.0
|
||||
)
|
||||
# The biller charges cache writes at the input rate when the entry carries
|
||||
# no cache_creation rate (cost_calculator.py:2452), and at the 5m write
|
||||
# rate when the 1h variant is unset; cache reads bill only at their own
|
||||
# rate (zero when the entry lacks one).
|
||||
write_5m_rate: Final = rates.cache_creation_input_token_cost or in_rate
|
||||
input_cost: Final = (
|
||||
u.fresh_input_tokens * in_rate
|
||||
+ u.cache_read_tokens * (rates.cache_read_input_token_cost or 0.0)
|
||||
+ u.cache_write_5m_tokens * write_5m_rate
|
||||
+ u.cache_write_1h_tokens * (rates.cache_creation_input_token_cost_above_1hr or write_5m_rate)
|
||||
+ u.audio_input_tokens * (rates.input_cost_per_audio_token or 0.0)
|
||||
)
|
||||
output_cost: Final = (
|
||||
u.output_tokens * out_rate
|
||||
+ u.reasoning_tokens * (rates.output_cost_per_reasoning_token or out_rate)
|
||||
+ u.audio_output_tokens * (rates.output_cost_per_audio_token or out_rate)
|
||||
)
|
||||
search: Final = rates.search_context_cost_per_query
|
||||
tool_cost: Final = billed_web_search_calls(model, case) * (
|
||||
search.search_context_size_medium if search and search.search_context_size_medium else 0.0
|
||||
)
|
||||
return ExpectedCost(input_cost=input_cost, output_cost=output_cost, tool_cost=tool_cost)
|
||||
|
||||
|
||||
def expected_token_columns(model: FrontierModel, case: Case) -> tuple[int, int]:
|
||||
"""(prompt_tokens, completion_tokens) the spend row should carry, per the
|
||||
wire's normalization: Anthropic folds cache read/write into prompt_tokens,
|
||||
everyone else reports the totals the wire emitted."""
|
||||
u: Final = case.usage
|
||||
if model.wire in ("anthropic_messages", "bedrock_converse"):
|
||||
return (
|
||||
u.fresh_input_tokens + u.cache_read_tokens + u.cache_write_5m_tokens + u.cache_write_1h_tokens,
|
||||
u.output_tokens,
|
||||
)
|
||||
if model.wire in ("gemini_generate", "vertex_generate"):
|
||||
return (
|
||||
u.fresh_input_tokens + u.cache_read_tokens + u.audio_input_tokens,
|
||||
u.output_tokens + u.reasoning_tokens + u.audio_output_tokens,
|
||||
)
|
||||
if model.wire == "openai_responses":
|
||||
return (
|
||||
u.fresh_input_tokens + u.cache_read_tokens,
|
||||
u.output_tokens + u.reasoning_tokens,
|
||||
)
|
||||
return (
|
||||
u.fresh_input_tokens
|
||||
+ u.cache_read_tokens
|
||||
+ u.cache_write_5m_tokens
|
||||
+ u.cache_write_1h_tokens
|
||||
+ u.audio_input_tokens,
|
||||
u.output_tokens + u.reasoning_tokens + u.audio_output_tokens,
|
||||
)
|
||||
|
||||
|
||||
def _proposed() -> dict[str, dict[str, object]]:
|
||||
return {
|
||||
expected_key(model, case): (
|
||||
lambda breakdown, tokens: {
|
||||
"spend": breakdown.total,
|
||||
"input_cost": breakdown.input_cost,
|
||||
"output_cost": breakdown.output_cost,
|
||||
"prompt_tokens": tokens[0],
|
||||
"completion_tokens": tokens[1],
|
||||
}
|
||||
)(expected_breakdown(model, case), expected_token_columns(model, case))
|
||||
for model in FRONTIER_MODELS
|
||||
for case in cases_for(model)
|
||||
if case.exact_spend
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
rewrite: Final = "--rewrite" in sys.argv[1:]
|
||||
proposed: Final = _proposed()
|
||||
existing: Final = (
|
||||
json.loads(EXPECTED_PATH.read_text()) if EXPECTED_PATH.exists() else {}
|
||||
)
|
||||
merged: Final = {
|
||||
key: (proposed[key] if rewrite or key not in existing else existing[key])
|
||||
for key in sorted(proposed)
|
||||
}
|
||||
added: Final = sum(1 for key in proposed if key not in existing)
|
||||
removed: Final = sum(1 for key in existing if key not in proposed)
|
||||
kept: Final = sum(1 for key in proposed if key in existing and not rewrite)
|
||||
rewritten: Final = sum(1 for key in proposed if key in existing and rewrite)
|
||||
EXPECTED_PATH.write_text(json.dumps(merged, indent=2, sort_keys=True) + "\n")
|
||||
print(
|
||||
f"expected.json: {added} added, {removed} removed, {kept} kept, "
|
||||
f"{rewritten} rewritten ({len(merged)} cells)"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
64
tests/e2e/cost_calculation/test_matrix_data.py
Normal file
64
tests/e2e/cost_calculation/test_matrix_data.py
Normal file
|
|
@ -0,0 +1,64 @@
|
|||
"""Freshness checks for the cost suite's data files; markerless, so it runs on
|
||||
any pytest invocation of the folder without the stack. expected.json is the
|
||||
oracle: these tests check its key set against the derived matrix, never its
|
||||
values (the generator proposes, the file decides)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
from cost_matrix import (
|
||||
_CASES_FILE,
|
||||
_COST_MAP,
|
||||
CASES,
|
||||
EXPECTED,
|
||||
FRONTIER_MODELS,
|
||||
CostMapEntry,
|
||||
cases_for,
|
||||
expected_key,
|
||||
)
|
||||
|
||||
|
||||
def test_expected_keys_match_derived_exact_cells() -> None:
|
||||
derived: Final = {
|
||||
expected_key(model, case)
|
||||
for model in FRONTIER_MODELS
|
||||
for case in cases_for(model)
|
||||
if case.exact_spend
|
||||
}
|
||||
golden: Final = set(EXPECTED)
|
||||
if derived != golden:
|
||||
missing: Final = sorted(derived - golden)
|
||||
stale: Final = sorted(golden - derived)
|
||||
pytest.fail(
|
||||
"expected.json is out of sync with the derived matrix; run "
|
||||
"uv run python tests/e2e/cost_calculation/generate_expected.py "
|
||||
f"(missing: {missing}; stale: {stale})"
|
||||
)
|
||||
|
||||
|
||||
def test_deployments_reference_existing_map_keys() -> None:
|
||||
unknown: Final = sorted(
|
||||
spec.map_key for spec in _CASES_FILE.deployments if spec.map_key not in _COST_MAP
|
||||
)
|
||||
assert not unknown, f"deployments entries name map keys absent from cost_map.json: {unknown}"
|
||||
|
||||
|
||||
def test_requires_rates_are_cost_map_fields() -> None:
|
||||
fields: Final = set(CostMapEntry.model_fields)
|
||||
unknown: Final = sorted(
|
||||
{field for case in CASES for field in case.requires_rates} - fields
|
||||
)
|
||||
assert not unknown, f"requires_rates names that are not CostMapEntry fields: {unknown}"
|
||||
|
||||
|
||||
def test_no_two_entries_share_input_rate() -> None:
|
||||
rates: Final = [
|
||||
entry.input_cost_per_token for entry in _COST_MAP.values()
|
||||
]
|
||||
assert len(rates) == len(set(rates)), (
|
||||
"two cost_map entries share input_cost_per_token; the suite relies on "
|
||||
"distinct rates so a wrong-model bill can never coincidentally match"
|
||||
)
|
||||
|
|
@ -1,7 +1,7 @@
|
|||
"""Token-pricing e2e: every (frontier model, pricing-component case) cell runs a
|
||||
scripted-usage call through a deployment registered on the cost-map proxy, and
|
||||
the spend row plus response-cost header must equal literal arithmetic on the
|
||||
test map's rates.
|
||||
"""Token-pricing e2e: every (map entry, case) cell derived from cost_map.json x
|
||||
cases.json runs a scripted-usage call through a deployment registered on the
|
||||
cost-map proxy, and the spend row plus response-cost header must equal the
|
||||
reviewed golden in expected.json verbatim -- no rate arithmetic lives here.
|
||||
|
||||
Nothing here touches a real provider or the bundled cost map: the proxy's
|
||||
upstream is the scripted-provider sidecar and its entire cost map is
|
||||
|
|
@ -15,13 +15,13 @@ from typing import Final
|
|||
|
||||
from conftest import CostCalcClient, cost_rows, register_scenario_deployment
|
||||
from cost_matrix import (
|
||||
EXPECTED,
|
||||
FRONTIER_MODELS,
|
||||
IMAGE_INPUT_DATA_URL,
|
||||
Case,
|
||||
FrontierModel,
|
||||
cases_for,
|
||||
expected_cost,
|
||||
expected_token_columns,
|
||||
expected_key,
|
||||
recount_cost,
|
||||
)
|
||||
from e2e_config import unique_marker
|
||||
|
|
@ -110,17 +110,6 @@ class TestTokenPricing:
|
|||
)
|
||||
assert response.stream_error is None, f"stream carried an error event: {response.stream_error}"
|
||||
|
||||
expected: Final = expected_cost(model, case)
|
||||
if case.exact_spend and not case.stream:
|
||||
# Streamed responses commit headers before the bill is computed, so
|
||||
# the x-litellm-response-cost header is asserted only on non-stream
|
||||
# calls.
|
||||
assert response.response_cost is not None and cost_rows.approx_equal(
|
||||
response.response_cost, expected
|
||||
), (
|
||||
f"x-litellm-response-cost {response.response_cost} != expected {expected}"
|
||||
)
|
||||
|
||||
row: Final = cost_rows.poll_cost_row_where(
|
||||
client.proxy,
|
||||
scoped_key,
|
||||
|
|
@ -149,16 +138,39 @@ class TestTokenPricing:
|
|||
cost_rows.assert_total_is_sum_of_components(row)
|
||||
return
|
||||
|
||||
assert row.spend is not None and cost_rows.approx_equal(row.spend, expected), (
|
||||
f"{model.map_key}/{case.name}: spend {row.spend} != expected {expected} "
|
||||
golden: Final = EXPECTED[expected_key(model, case)]
|
||||
|
||||
if not case.stream:
|
||||
# Streamed responses commit headers before the bill is computed, so
|
||||
# the x-litellm-response-cost header is asserted only on non-stream
|
||||
# calls.
|
||||
assert response.response_cost is not None and cost_rows.approx_equal(
|
||||
response.response_cost, golden.spend
|
||||
), (
|
||||
f"x-litellm-response-cost {response.response_cost} != golden {golden.spend}"
|
||||
)
|
||||
|
||||
assert row.spend is not None and cost_rows.approx_equal(row.spend, golden.spend), (
|
||||
f"{model.map_key}/{case.name}: spend {row.spend} != golden {golden.spend} "
|
||||
f"(breakdown {row.breakdown.model_dump()})"
|
||||
)
|
||||
|
||||
prompt_tokens, completion_tokens = expected_token_columns(model, case)
|
||||
assert row.prompt_tokens == prompt_tokens, (
|
||||
f"prompt_tokens {row.prompt_tokens} != {prompt_tokens}"
|
||||
breakdown: Final = row.breakdown
|
||||
assert breakdown.input_cost is not None and cost_rows.approx_equal(
|
||||
breakdown.input_cost, golden.input_cost
|
||||
), (
|
||||
f"{model.map_key}/{case.name}: gross input_cost {breakdown.input_cost} "
|
||||
f"!= golden {golden.input_cost}; cached/written tokens billed at the input rate"
|
||||
)
|
||||
assert row.completion_tokens == completion_tokens, (
|
||||
f"completion_tokens {row.completion_tokens} != {completion_tokens}"
|
||||
assert breakdown.output_cost is not None and cost_rows.approx_equal(
|
||||
breakdown.output_cost, golden.output_cost
|
||||
), (
|
||||
f"{model.map_key}/{case.name}: output_cost {breakdown.output_cost} "
|
||||
f"!= golden {golden.output_cost}"
|
||||
)
|
||||
assert row.prompt_tokens == golden.prompt_tokens, (
|
||||
f"prompt_tokens {row.prompt_tokens} != {golden.prompt_tokens}"
|
||||
)
|
||||
assert row.completion_tokens == golden.completion_tokens, (
|
||||
f"completion_tokens {row.completion_tokens} != {golden.completion_tokens}"
|
||||
)
|
||||
cost_rows.assert_total_is_sum_of_components(row)
|
||||
|
|
|
|||
|
|
@ -1,368 +0,0 @@
|
|||
"""Wire-format e2e: one scripted upstream per provider wire, answering with a
|
||||
usage payload where every token kind the wire can report is nonzero. The spend
|
||||
row's gross input cost must equal fresh tokens at the input rate plus each cache
|
||||
and audio component at its own rate -- proving the wire's usage shape landed the
|
||||
cached tokens inside the total (OpenAI/Gemini) or as separate fields
|
||||
(Anthropic), and that the biller subtracted them before billing fresh tokens.
|
||||
|
||||
Also covers the Responses API wire (an openai/gpt-5.5-pro deployment bridged by
|
||||
the proxy to POST /responses) and a streamed Anthropic-messages case.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from collections.abc import Mapping
|
||||
from types import MappingProxyType
|
||||
from typing import Final
|
||||
|
||||
from conftest import CostCalcClient, cost_rows, register_scenario_deployment
|
||||
from cost_matrix import (
|
||||
FRONTIER_MODELS,
|
||||
Case,
|
||||
FrontierModel,
|
||||
expected_breakdown,
|
||||
expected_token_columns,
|
||||
)
|
||||
from e2e_config import unique_marker
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, ChatStreamOptions, ChatTool, ChatToolFunction
|
||||
from scripted_provider import ScriptedUsage
|
||||
|
||||
pytestmark: Final = [pytest.mark.e2e, pytest.mark.cost_map_stack] # mutable-ok: pytest only accepts a list for pytestmark
|
||||
|
||||
_MODELS: Final[Mapping[str, FrontierModel]] = MappingProxyType(
|
||||
{model.map_key: model for model in FRONTIER_MODELS}
|
||||
)
|
||||
|
||||
# One scripted usage per wire, every reportable token kind nonzero.
|
||||
_WIRE_USAGE: Final[Mapping[str, tuple[str, ScriptedUsage]]] = MappingProxyType({
|
||||
"openai_chat": (
|
||||
"gpt-5.6",
|
||||
ScriptedUsage(
|
||||
fresh_input_tokens=80,
|
||||
cache_read_tokens=40,
|
||||
cache_write_5m_tokens=20,
|
||||
cache_write_1h_tokens=10,
|
||||
output_tokens=25,
|
||||
reasoning_tokens=15,
|
||||
audio_input_tokens=5,
|
||||
audio_output_tokens=3,
|
||||
),
|
||||
),
|
||||
"openai_responses": (
|
||||
"gpt-5.5-pro",
|
||||
ScriptedUsage(
|
||||
fresh_input_tokens=80, cache_read_tokens=40, output_tokens=25, reasoning_tokens=15
|
||||
),
|
||||
),
|
||||
"anthropic_messages": (
|
||||
"claude-sonnet-5",
|
||||
ScriptedUsage(
|
||||
fresh_input_tokens=80,
|
||||
cache_read_tokens=40,
|
||||
cache_write_5m_tokens=20,
|
||||
cache_write_1h_tokens=10,
|
||||
output_tokens=25,
|
||||
),
|
||||
),
|
||||
"gemini_generate": (
|
||||
"gemini/gemini-3.8-flash",
|
||||
ScriptedUsage(
|
||||
fresh_input_tokens=80,
|
||||
cache_read_tokens=40,
|
||||
output_tokens=25,
|
||||
reasoning_tokens=15,
|
||||
audio_input_tokens=5,
|
||||
audio_output_tokens=3,
|
||||
),
|
||||
),
|
||||
"together_chat": (
|
||||
"together_ai/moonshotai/Kimi-K3",
|
||||
ScriptedUsage(
|
||||
fresh_input_tokens=80,
|
||||
cache_read_tokens=40,
|
||||
cache_write_5m_tokens=20,
|
||||
cache_write_1h_tokens=10,
|
||||
output_tokens=25,
|
||||
reasoning_tokens=15,
|
||||
audio_input_tokens=5,
|
||||
audio_output_tokens=3,
|
||||
),
|
||||
),
|
||||
"fireworks_chat": (
|
||||
"fireworks_ai/kimi-k3",
|
||||
ScriptedUsage(fresh_input_tokens=80, cache_read_tokens=40, output_tokens=25),
|
||||
),
|
||||
"azure_chat": (
|
||||
"azure/gpt-5.6",
|
||||
ScriptedUsage(
|
||||
fresh_input_tokens=80,
|
||||
cache_read_tokens=40,
|
||||
cache_write_5m_tokens=20,
|
||||
cache_write_1h_tokens=10,
|
||||
output_tokens=25,
|
||||
reasoning_tokens=15,
|
||||
audio_input_tokens=5,
|
||||
audio_output_tokens=3,
|
||||
),
|
||||
),
|
||||
"bedrock_converse": (
|
||||
"anthropic.claude-sonnet-5-v1:0",
|
||||
ScriptedUsage(
|
||||
fresh_input_tokens=80,
|
||||
cache_read_tokens=40,
|
||||
cache_write_5m_tokens=20,
|
||||
cache_write_1h_tokens=10,
|
||||
output_tokens=25,
|
||||
),
|
||||
),
|
||||
"vertex_generate": (
|
||||
"gemini-3.8-flash",
|
||||
ScriptedUsage(
|
||||
fresh_input_tokens=80,
|
||||
cache_read_tokens=40,
|
||||
output_tokens=25,
|
||||
reasoning_tokens=15,
|
||||
audio_input_tokens=5,
|
||||
audio_output_tokens=3,
|
||||
),
|
||||
),
|
||||
})
|
||||
|
||||
_SHAPE_USAGE: Final = ScriptedUsage(fresh_input_tokens=80, output_tokens=25)
|
||||
|
||||
# Renderer-level shapes the pricing matrix gates per cap, pinned here once per
|
||||
# wire so the sidecar emits prove they survive the proxy end to end.
|
||||
_SHAPES: Final[tuple[tuple[str, str, Case], ...]] = (
|
||||
*(
|
||||
(
|
||||
f"tool_call_{'stream' if stream else 'sync'}",
|
||||
wire,
|
||||
Case(name="tool_call", usage=_SHAPE_USAGE, stream=stream, tool_call=True),
|
||||
)
|
||||
for wire in _WIRE_USAGE
|
||||
for stream in (False, True)
|
||||
),
|
||||
(
|
||||
"responses_incomplete",
|
||||
"openai_responses",
|
||||
Case(name="stream_no_usage_incomplete", usage=_SHAPE_USAGE, stream=True, terminal="incomplete"),
|
||||
),
|
||||
(
|
||||
"responses_unvalidated",
|
||||
"openai_responses",
|
||||
Case(name="stream_unvalidated", usage=_SHAPE_USAGE, stream=True, terminal="unvalidated"),
|
||||
),
|
||||
(
|
||||
"gemini_prompt_blocked",
|
||||
"gemini_generate",
|
||||
Case(
|
||||
name="prompt_blocked",
|
||||
usage=ScriptedUsage(fresh_input_tokens=1000, output_tokens=0),
|
||||
terminal="prompt_blocked",
|
||||
response_model_override=True,
|
||||
),
|
||||
),
|
||||
(
|
||||
"gemini_prompt_blocked_stream",
|
||||
"gemini_generate",
|
||||
Case(
|
||||
name="stream_prompt_blocked",
|
||||
usage=ScriptedUsage(fresh_input_tokens=1000, output_tokens=0),
|
||||
stream=True,
|
||||
terminal="prompt_blocked",
|
||||
response_model_override=True,
|
||||
),
|
||||
),
|
||||
(
|
||||
"vertex_prompt_blocked",
|
||||
"vertex_generate",
|
||||
Case(
|
||||
name="prompt_blocked",
|
||||
usage=ScriptedUsage(fresh_input_tokens=1000, output_tokens=0),
|
||||
terminal="prompt_blocked",
|
||||
response_model_override=True,
|
||||
),
|
||||
),
|
||||
(
|
||||
"vertex_prompt_blocked_stream",
|
||||
"vertex_generate",
|
||||
Case(
|
||||
name="stream_prompt_blocked",
|
||||
usage=ScriptedUsage(fresh_input_tokens=1000, output_tokens=0),
|
||||
stream=True,
|
||||
terminal="prompt_blocked",
|
||||
response_model_override=True,
|
||||
),
|
||||
),
|
||||
(
|
||||
"azure_served_model_override",
|
||||
"azure_chat",
|
||||
Case(
|
||||
name="response_model_override",
|
||||
usage=_SHAPE_USAGE,
|
||||
response_model_override=True,
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _shape_id(entry: tuple[str, str, Case]) -> str:
|
||||
return entry[0]
|
||||
|
||||
|
||||
class TestWireFormats:
|
||||
@pytest.mark.parametrize("wire", tuple(_WIRE_USAGE))
|
||||
@pytest.mark.covers("quota_management.spend_tracking.scripted_wire.logs_cost")
|
||||
def test_wire_usage_shape_bills_each_component(
|
||||
self,
|
||||
client: CostCalcClient,
|
||||
resources: ResourceManager,
|
||||
scoped_key: str,
|
||||
wire: str,
|
||||
) -> None:
|
||||
map_key, usage = _WIRE_USAGE[wire]
|
||||
model: Final = _MODELS[map_key]
|
||||
case: Final = Case(name="basic", usage=usage)
|
||||
marker: Final = unique_marker()
|
||||
model_name, _handle = register_scenario_deployment(client, resources, model, case, marker)
|
||||
response: Final = client.proxy.transport.send(
|
||||
"/chat/completions",
|
||||
headers=client.proxy.transport.bearer(scoped_key),
|
||||
json=ChatBody(
|
||||
model=model_name,
|
||||
messages=(ChatMessage(role="user", content=f"{marker} scripted wire call"),),
|
||||
),
|
||||
)
|
||||
assert response.ok, f"{wire}: proxy returned {response.status_code}: {response.body[:400]}"
|
||||
|
||||
expected: Final = expected_breakdown(model, case)
|
||||
row: Final = cost_rows.poll_cost_row_where(
|
||||
client.proxy,
|
||||
scoped_key,
|
||||
lambda r: r.metadata is not None and r.metadata.cost_breakdown is not None,
|
||||
)
|
||||
assert row is not None, f"{wire}: no spend row landed"
|
||||
assert row.spend is not None and cost_rows.approx_equal(row.spend, expected.total), (
|
||||
f"{wire}: spend {row.spend} != expected {expected.total} "
|
||||
f"(breakdown {row.breakdown.model_dump()})"
|
||||
)
|
||||
breakdown: Final = row.breakdown
|
||||
assert breakdown.input_cost is not None and cost_rows.approx_equal(
|
||||
breakdown.input_cost, expected.input_cost
|
||||
), (
|
||||
f"{wire}: gross input_cost {breakdown.input_cost} != expected {expected.input_cost}; "
|
||||
"cached/written tokens billed at the input rate"
|
||||
)
|
||||
assert breakdown.output_cost is not None and cost_rows.approx_equal(
|
||||
breakdown.output_cost, expected.output_cost
|
||||
), f"{wire}: output_cost {breakdown.output_cost} != expected {expected.output_cost}"
|
||||
|
||||
prompt_tokens, completion_tokens = expected_token_columns(model, case)
|
||||
assert row.prompt_tokens == prompt_tokens, (
|
||||
f"{wire}: prompt_tokens {row.prompt_tokens} != {prompt_tokens}"
|
||||
)
|
||||
assert row.completion_tokens == completion_tokens, (
|
||||
f"{wire}: completion_tokens {row.completion_tokens} != {completion_tokens}"
|
||||
)
|
||||
cost_rows.assert_total_is_sum_of_components(row)
|
||||
|
||||
@pytest.mark.covers("quota_management.spend_tracking.scripted_wire.logs_cost")
|
||||
def test_anthropic_streamed_usage_bills_each_component(
|
||||
self, client: CostCalcClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
map_key, usage = _WIRE_USAGE["anthropic_messages"]
|
||||
model: Final = _MODELS[map_key]
|
||||
case: Final = Case(name="stream", usage=usage, stream=True)
|
||||
marker: Final = unique_marker()
|
||||
model_name, _handle = register_scenario_deployment(client, resources, model, case, marker)
|
||||
response: Final = client.proxy.transport.send(
|
||||
"/chat/completions",
|
||||
headers=client.proxy.transport.bearer(scoped_key),
|
||||
json=ChatBody(
|
||||
model=model_name,
|
||||
messages=(ChatMessage(role="user", content=f"{marker} scripted anthropic stream"),),
|
||||
stream=True,
|
||||
stream_options=ChatStreamOptions(include_usage=True),
|
||||
),
|
||||
stream=True,
|
||||
)
|
||||
assert response.ok, f"anthropic stream: proxy returned {response.status_code}: {response.body[:400]}"
|
||||
assert response.stream_done, "anthropic stream did not reach its terminal event"
|
||||
assert response.stream_error is None, f"stream carried an error event: {response.stream_error}"
|
||||
|
||||
expected: Final = expected_breakdown(model, case)
|
||||
row: Final = cost_rows.poll_cost_row_where(
|
||||
client.proxy,
|
||||
scoped_key,
|
||||
lambda r: r.metadata is not None and r.metadata.cost_breakdown is not None,
|
||||
)
|
||||
assert row is not None, "anthropic stream: no spend row landed"
|
||||
assert row.spend is not None and cost_rows.approx_equal(row.spend, expected.total), (
|
||||
f"anthropic stream: spend {row.spend} != expected {expected.total} "
|
||||
f"(breakdown {row.breakdown.model_dump()})"
|
||||
)
|
||||
cost_rows.assert_total_is_sum_of_components(row)
|
||||
|
||||
@pytest.mark.parametrize("shape_wire_case", _SHAPES, ids=_shape_id)
|
||||
@pytest.mark.covers("quota_management.spend_tracking.scripted_wire.logs_cost")
|
||||
def test_response_shape_bills_reported_usage(
|
||||
self,
|
||||
client: CostCalcClient,
|
||||
resources: ResourceManager,
|
||||
scoped_key: str,
|
||||
shape_wire_case: tuple[str, str, Case],
|
||||
) -> None:
|
||||
shape, wire, case = shape_wire_case
|
||||
map_key, _usage = _WIRE_USAGE[wire]
|
||||
model: Final = _MODELS[map_key]
|
||||
marker: Final = unique_marker()
|
||||
model_name, _handle = register_scenario_deployment(client, resources, model, case, marker)
|
||||
response: Final = client.proxy.transport.send(
|
||||
"/chat/completions",
|
||||
headers=client.proxy.transport.bearer(scoped_key),
|
||||
json=ChatBody(
|
||||
model=model_name,
|
||||
messages=(ChatMessage(role="user", content=f"{marker} scripted {shape}"),),
|
||||
stream=case.stream,
|
||||
stream_options=ChatStreamOptions(include_usage=True) if case.stream else None,
|
||||
tools=(
|
||||
(
|
||||
ChatTool(
|
||||
function=ChatToolFunction(
|
||||
name="get_weather",
|
||||
parameters={"type": "object", "properties": {"city": {"type": "string"}}},
|
||||
)
|
||||
),
|
||||
)
|
||||
if case.tool_call
|
||||
else None
|
||||
),
|
||||
),
|
||||
stream=case.stream,
|
||||
)
|
||||
assert response.ok, f"{shape}: proxy returned {response.status_code}: {response.body[:400]}"
|
||||
if case.stream:
|
||||
assert response.stream_done, f"{shape}: stream did not reach its terminal event"
|
||||
assert response.stream_error is None, f"{shape}: stream error: {response.stream_error}"
|
||||
|
||||
expected: Final = expected_breakdown(model, case)
|
||||
row: Final = cost_rows.poll_cost_row_where(
|
||||
client.proxy,
|
||||
scoped_key,
|
||||
lambda r: r.metadata is not None and r.metadata.cost_breakdown is not None,
|
||||
)
|
||||
assert row is not None, f"{shape}: no spend row landed"
|
||||
assert row.spend is not None and cost_rows.approx_equal(row.spend, expected.total), (
|
||||
f"{shape}: spend {row.spend} != expected {expected.total} "
|
||||
f"(breakdown {row.breakdown.model_dump()})"
|
||||
)
|
||||
prompt_tokens, completion_tokens = expected_token_columns(model, case)
|
||||
assert row.prompt_tokens == prompt_tokens, (
|
||||
f"{shape}: prompt_tokens {row.prompt_tokens} != {prompt_tokens}"
|
||||
)
|
||||
assert row.completion_tokens == completion_tokens, (
|
||||
f"{shape}: completion_tokens {row.completion_tokens} != {completion_tokens}"
|
||||
)
|
||||
cost_rows.assert_total_is_sum_of_components(row)
|
||||
Loading…
Add table
Reference in a new issue