diff --git a/tests/e2e/CLAUDE.md b/tests/e2e/CLAUDE.md index f89b3203622..a3e5696ef9d 100644 --- a/tests/e2e/CLAUDE.md +++ b/tests/e2e/CLAUDE.md @@ -21,7 +21,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family - `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set and excluded from the per-PR selector like the rest of `load/`, driven by `.github/workflows/test-e2e-redis-chaos.yml` and by the Buildkite `e2e-redis-chaos` step in project-releaser, which runs the proxy, Postgres and Valkey co-located with pytest in one pod and sets the opt-in), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic - `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate, JWT auth (access tokens issued by a real Keycloak realm, `idp.py` plus `idp_realm.json`, whose JWKS the proxy's `JWT_PUBLIC_KEY_URL` points at; see CONTRIBUTING.md for the start command and config block), and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite - `gateway/` - proxy configuration only (`litellm-config.yml`); no tests -- `cost_calculation/` - cost accounting against a dedicated proxy whose whole model cost map is the test-owned `tests/e2e/cost_map.json` (loaded via `LITELLM_MODEL_COST_MAP_URL`), with provider calls answered by the scripted-provider sidecar in `scripted_provider.py`; every cost-map entry is a deployment and the cases plus asserted goldens are data in `cases.json` and `expected.json` (regenerate with `generate_expected.py`; `cost_matrix.matrix_data_errors()` runs at collection time so a stale key set fails the suite's collection loudly), deselected unless `E2E_COST_MAP_STACK` is set, driven by the Buildkite `e2e-cost-calculation` step in project-releaser, which runs a proxy booted from `gateway/cost_calculation_ci_config.yml`, Postgres and the scripted provider co-located with pytest in one pod and sets the opt-in +- `cost_calculation/` - cost accounting against a dedicated proxy whose whole model cost map is the test-owned `tests/e2e/cost_map.json` (loaded via `LITELLM_MODEL_COST_MAP_URL`), with provider calls answered by the scripted-provider sidecar in `scripted_provider.py`; every cost-map entry is a deployment and the cases plus asserted goldens are data in `cases.json` (each exact-spend case carries a literal `expected` cell per map key; `cost_matrix.matrix_data_errors()` runs at collection time so a key absent from the cost map fails the suite's collection loudly), deselected unless `E2E_COST_MAP_STACK` is set, driven by the Buildkite `e2e-cost-calculation` step in project-releaser, which runs a proxy booted from `gateway/cost_calculation_ci_config.yml`, Postgres and the scripted provider co-located with pytest in one pod and sets the opt-in - `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke - `ui/` - the Admin UI browser suite: Playwright in TypeScript, driving the dashboard served by a live proxy on port 4000 (seeded postgres + mock LLM upstream; see its `run_e2e.sh`). It is a self-contained npm package with its own lockfile and does not use the Python harness, pytest markers, or the shared transport; the Python rules in this file (typed models, `Result` unions, basedpyright zero-error gate) do not apply inside it. Its only Python file, `fixtures/mock_llm_server/server.py`, is excluded from the e2e basedpyright gate via the root `pyrightconfig.json` diff --git a/tests/e2e/cost_calculation/cases.json b/tests/e2e/cost_calculation/cases.json index 49eebc85231..cda7bc6e67a 100644 --- a/tests/e2e/cost_calculation/cases.json +++ b/tests/e2e/cost_calculation/cases.json @@ -1,81 +1,254 @@ { "deployments": [ - { - "map_key": "azure/gpt-5.4-mini", - "litellm_model": "azure/cc-pinned-deployment", - "base_model": "azure/gpt-5.4-mini" - } + {"map_key": "azure/gpt-5.4-mini", "litellm_model": "azure/cc-pinned-deployment", "base_model": "azure/gpt-5.4-mini"} ], "cases": [ { "name": "basic", - "usage": {"fresh_input_tokens": 120, "output_tokens": 40} + "usage": {"fresh_input_tokens": 120, "output_tokens": 40}, + "expected": { + "anthropic.claude-sonnet-5-v1:0": {"spend": 0.034, "input_cost": 0.0204, "output_cost": 0.0136, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.6": {"spend": 0.03, "input_cost": 0.018, "output_cost": 0.012, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-haiku-4-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-opus-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-sonnet-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/kimi-k3": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/qwen3p8-max": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.1-pro-preview": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.8-flash": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.8-flash": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.4-mini": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.6": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40}, + "meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.038, "input_cost": 0.0228, "output_cost": 0.0152, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40}, + "us.anthropic.claude-opus-5-v1:0": {"spend": 0.036, "input_cost": 0.0216, "output_cost": 0.0144, "prompt_tokens": 120, "completion_tokens": 40} + } }, { "name": "cache_read", "usage": {"fresh_input_tokens": 100, "cache_read_tokens": 50, "output_tokens": 30}, - "requires_rates": ["cache_read_input_token_cost"], - "requires_caps": ["cache_read"] + "expected": { + "anthropic.claude-sonnet-5-v1:0": {"spend": 0.02805, "input_cost": 0.01785, "output_cost": 0.0102, "prompt_tokens": 150, "completion_tokens": 30}, + "azure/gpt-5.4-mini": {"spend": 0.0264, "input_cost": 0.0168, "output_cost": 0.0096, "prompt_tokens": 150, "completion_tokens": 30}, + "azure/gpt-5.6": {"spend": 0.02475, "input_cost": 0.01575, "output_cost": 0.009, "prompt_tokens": 150, "completion_tokens": 30}, + "claude-haiku-4-5": {"spend": 0.01155, "input_cost": 0.00735, "output_cost": 0.0042, "prompt_tokens": 150, "completion_tokens": 30}, + "claude-opus-5": {"spend": 0.00825, "input_cost": 0.00525, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 30}, + "claude-sonnet-5": {"spend": 0.0099, "input_cost": 0.0063, "output_cost": 0.0036, "prompt_tokens": 150, "completion_tokens": 30}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0231, "input_cost": 0.0147, "output_cost": 0.0084, "prompt_tokens": 150, "completion_tokens": 30}, + "fireworks_ai/kimi-k3": {"spend": 0.0198, "input_cost": 0.0126, "output_cost": 0.0072, "prompt_tokens": 150, "completion_tokens": 30}, + "fireworks_ai/qwen3p8-max": {"spend": 0.02145, "input_cost": 0.01365, "output_cost": 0.0078, "prompt_tokens": 150, "completion_tokens": 30}, + "gemini-3.1-pro-preview": {"spend": 0.03465, "input_cost": 0.02205, "output_cost": 0.0126, "prompt_tokens": 150, "completion_tokens": 30}, + "gemini-3.8-flash": {"spend": 0.033, "input_cost": 0.021, "output_cost": 0.012, "prompt_tokens": 150, "completion_tokens": 30}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.01485, "input_cost": 0.00945, "output_cost": 0.0054, "prompt_tokens": 150, "completion_tokens": 30}, + "gemini/gemini-3.8-flash": {"spend": 0.0132, "input_cost": 0.0084, "output_cost": 0.0048, "prompt_tokens": 150, "completion_tokens": 30}, + "gpt-5.3-codex": {"spend": 0.00495, "input_cost": 0.00315, "output_cost": 0.0018, "prompt_tokens": 150, "completion_tokens": 30}, + "gpt-5.4-mini": {"spend": 0.0066, "input_cost": 0.0042, "output_cost": 0.0024, "prompt_tokens": 150, "completion_tokens": 30}, + "gpt-5.5-pro": {"spend": 0.0033, "input_cost": 0.0021, "output_cost": 0.0012, "prompt_tokens": 150, "completion_tokens": 30}, + "gpt-5.6": {"spend": 0.00165, "input_cost": 0.00105, "output_cost": 0.0006, "prompt_tokens": 150, "completion_tokens": 30}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.0165, "input_cost": 0.0105, "output_cost": 0.006, "prompt_tokens": 150, "completion_tokens": 30}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.01815, "input_cost": 0.01155, "output_cost": 0.0066, "prompt_tokens": 150, "completion_tokens": 30}, + "us.anthropic.claude-opus-5-v1:0": {"spend": 0.0297, "input_cost": 0.0189, "output_cost": 0.0108, "prompt_tokens": 150, "completion_tokens": 30} + } }, { "name": "cache_write_5m", "usage": {"fresh_input_tokens": 90, "cache_write_5m_tokens": 60, "output_tokens": 30}, - "requires_rates": ["cache_creation_input_token_cost"], - "requires_caps": ["cache_write_5m"] + "expected": { + "anthropic.claude-sonnet-5-v1:0": {"spend": 0.0561, "input_cost": 0.0459, "output_cost": 0.0102, "prompt_tokens": 150, "completion_tokens": 30}, + "azure/gpt-5.4-mini": {"spend": 0.0528, "input_cost": 0.0432, "output_cost": 0.0096, "prompt_tokens": 150, "completion_tokens": 30}, + "azure/gpt-5.6": {"spend": 0.0495, "input_cost": 0.0405, "output_cost": 0.009, "prompt_tokens": 150, "completion_tokens": 30}, + "claude-haiku-4-5": {"spend": 0.0231, "input_cost": 0.0189, "output_cost": 0.0042, "prompt_tokens": 150, "completion_tokens": 30}, + "claude-opus-5": {"spend": 0.0165, "input_cost": 0.0135, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 30}, + "claude-sonnet-5": {"spend": 0.0198, "input_cost": 0.0162, "output_cost": 0.0036, "prompt_tokens": 150, "completion_tokens": 30}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0408, "input_cost": 0.0324, "output_cost": 0.0084, "prompt_tokens": 150, "completion_tokens": 30}, + "fireworks_ai/kimi-k3": {"spend": 0.0378, "input_cost": 0.0306, "output_cost": 0.0072, "prompt_tokens": 150, "completion_tokens": 30}, + "fireworks_ai/qwen3p8-max": {"spend": 0.0393, "input_cost": 0.0315, "output_cost": 0.0078, "prompt_tokens": 150, "completion_tokens": 30}, + "gpt-5.4-mini": {"spend": 0.0132, "input_cost": 0.0108, "output_cost": 0.0024, "prompt_tokens": 150, "completion_tokens": 30}, + "gpt-5.6": {"spend": 0.0033, "input_cost": 0.0027, "output_cost": 0.0006, "prompt_tokens": 150, "completion_tokens": 30}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.033, "input_cost": 0.027, "output_cost": 0.006, "prompt_tokens": 150, "completion_tokens": 30}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.0363, "input_cost": 0.0297, "output_cost": 0.0066, "prompt_tokens": 150, "completion_tokens": 30}, + "us.anthropic.claude-opus-5-v1:0": {"spend": 0.0594, "input_cost": 0.0486, "output_cost": 0.0108, "prompt_tokens": 150, "completion_tokens": 30} + } }, { "name": "cache_write_1h", "usage": {"fresh_input_tokens": 90, "cache_write_5m_tokens": 20, "cache_write_1h_tokens": 40, "output_tokens": 30}, - "requires_rates": ["cache_creation_input_token_cost_above_1hr", "cache_creation_input_token_cost"], - "requires_caps": ["cache_write_1h"] + "expected": { + "anthropic.claude-sonnet-5-v1:0": {"spend": 0.0629, "input_cost": 0.0527, "output_cost": 0.0102, "prompt_tokens": 150, "completion_tokens": 30}, + "azure/gpt-5.4-mini": {"spend": 0.0592, "input_cost": 0.0496, "output_cost": 0.0096, "prompt_tokens": 150, "completion_tokens": 30}, + "azure/gpt-5.6": {"spend": 0.0555, "input_cost": 0.0465, "output_cost": 0.009, "prompt_tokens": 150, "completion_tokens": 30}, + "claude-haiku-4-5": {"spend": 0.0259, "input_cost": 0.0217, "output_cost": 0.0042, "prompt_tokens": 150, "completion_tokens": 30}, + "claude-opus-5": {"spend": 0.0185, "input_cost": 0.0155, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 30}, + "claude-sonnet-5": {"spend": 0.0222, "input_cost": 0.0186, "output_cost": 0.0036, "prompt_tokens": 150, "completion_tokens": 30}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0452, "input_cost": 0.0368, "output_cost": 0.0084, "prompt_tokens": 150, "completion_tokens": 30}, + "fireworks_ai/kimi-k3": {"spend": 0.0422, "input_cost": 0.035, "output_cost": 0.0072, "prompt_tokens": 150, "completion_tokens": 30}, + "fireworks_ai/qwen3p8-max": {"spend": 0.0437, "input_cost": 0.0359, "output_cost": 0.0078, "prompt_tokens": 150, "completion_tokens": 30}, + "gpt-5.4-mini": {"spend": 0.0148, "input_cost": 0.0124, "output_cost": 0.0024, "prompt_tokens": 150, "completion_tokens": 30}, + "gpt-5.6": {"spend": 0.0037, "input_cost": 0.0031, "output_cost": 0.0006, "prompt_tokens": 150, "completion_tokens": 30}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.037, "input_cost": 0.031, "output_cost": 0.006, "prompt_tokens": 150, "completion_tokens": 30}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.0407, "input_cost": 0.0341, "output_cost": 0.0066, "prompt_tokens": 150, "completion_tokens": 30}, + "us.anthropic.claude-opus-5-v1:0": {"spend": 0.0666, "input_cost": 0.0558, "output_cost": 0.0108, "prompt_tokens": 150, "completion_tokens": 30} + } }, { "name": "reasoning", "usage": {"fresh_input_tokens": 100, "output_tokens": 30, "reasoning_tokens": 70}, - "requires_rates": ["output_cost_per_reasoning_token"], - "requires_caps": ["reasoning"] + "expected": { + "azure/gpt-5.4-mini": {"spend": 0.0816, "input_cost": 0.016, "output_cost": 0.0656, "prompt_tokens": 100, "completion_tokens": 100}, + "azure/gpt-5.6": {"spend": 0.0765, "input_cost": 0.015, "output_cost": 0.0615, "prompt_tokens": 100, "completion_tokens": 100}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0609, "input_cost": 0.014, "output_cost": 0.0469, "prompt_tokens": 100, "completion_tokens": 100}, + "fireworks_ai/kimi-k3": {"spend": 0.0577, "input_cost": 0.012, "output_cost": 0.0457, "prompt_tokens": 100, "completion_tokens": 100}, + "fireworks_ai/qwen3p8-max": {"spend": 0.0593, "input_cost": 0.013, "output_cost": 0.0463, "prompt_tokens": 100, "completion_tokens": 100}, + "gemini-3.1-pro-preview": {"spend": 0.1071, "input_cost": 0.021, "output_cost": 0.0861, "prompt_tokens": 100, "completion_tokens": 100}, + "gemini-3.8-flash": {"spend": 0.102, "input_cost": 0.02, "output_cost": 0.082, "prompt_tokens": 100, "completion_tokens": 100}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.0459, "input_cost": 0.009, "output_cost": 0.0369, "prompt_tokens": 100, "completion_tokens": 100}, + "gemini/gemini-3.8-flash": {"spend": 0.0408, "input_cost": 0.008, "output_cost": 0.0328, "prompt_tokens": 100, "completion_tokens": 100}, + "gpt-5.3-codex": {"spend": 0.0153, "input_cost": 0.003, "output_cost": 0.0123, "prompt_tokens": 100, "completion_tokens": 100}, + "gpt-5.4-mini": {"spend": 0.0204, "input_cost": 0.004, "output_cost": 0.0164, "prompt_tokens": 100, "completion_tokens": 100}, + "gpt-5.5-pro": {"spend": 0.0102, "input_cost": 0.002, "output_cost": 0.0082, "prompt_tokens": 100, "completion_tokens": 100}, + "gpt-5.6": {"spend": 0.0051, "input_cost": 0.001, "output_cost": 0.0041, "prompt_tokens": 100, "completion_tokens": 100}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.051, "input_cost": 0.01, "output_cost": 0.041, "prompt_tokens": 100, "completion_tokens": 100}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.0561, "input_cost": 0.011, "output_cost": 0.0451, "prompt_tokens": 100, "completion_tokens": 100} + } }, { "name": "audio", "usage": {"fresh_input_tokens": 100, "audio_input_tokens": 25, "output_tokens": 30, "audio_output_tokens": 15}, - "requires_rates": ["input_cost_per_audio_token", "output_cost_per_audio_token"], - "requires_caps": ["audio"] + "expected": { + "azure/gpt-5.4-mini": {"spend": 0.0664, "input_cost": 0.04, "output_cost": 0.0264, "prompt_tokens": 125, "completion_tokens": 45}, + "azure/gpt-5.6": {"spend": 0.06225, "input_cost": 0.0375, "output_cost": 0.02475, "prompt_tokens": 125, "completion_tokens": 45}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.05045, "input_cost": 0.0305, "output_cost": 0.01995, "prompt_tokens": 125, "completion_tokens": 45}, + "fireworks_ai/kimi-k3": {"spend": 0.04725, "input_cost": 0.0285, "output_cost": 0.01875, "prompt_tokens": 125, "completion_tokens": 45}, + "fireworks_ai/qwen3p8-max": {"spend": 0.04885, "input_cost": 0.0295, "output_cost": 0.01935, "prompt_tokens": 125, "completion_tokens": 45}, + "gemini-3.1-pro-preview": {"spend": 0.08715, "input_cost": 0.0525, "output_cost": 0.03465, "prompt_tokens": 125, "completion_tokens": 45}, + "gemini-3.8-flash": {"spend": 0.083, "input_cost": 0.05, "output_cost": 0.033, "prompt_tokens": 125, "completion_tokens": 45}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.03735, "input_cost": 0.0225, "output_cost": 0.01485, "prompt_tokens": 125, "completion_tokens": 45}, + "gemini/gemini-3.8-flash": {"spend": 0.0332, "input_cost": 0.02, "output_cost": 0.0132, "prompt_tokens": 125, "completion_tokens": 45}, + "gpt-5.4-mini": {"spend": 0.0166, "input_cost": 0.01, "output_cost": 0.0066, "prompt_tokens": 125, "completion_tokens": 45}, + "gpt-5.6": {"spend": 0.00415, "input_cost": 0.0025, "output_cost": 0.00165, "prompt_tokens": 125, "completion_tokens": 45}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.0415, "input_cost": 0.025, "output_cost": 0.0165, "prompt_tokens": 125, "completion_tokens": 45}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.04565, "input_cost": 0.0275, "output_cost": 0.01815, "prompt_tokens": 125, "completion_tokens": 45} + } }, { "name": "tiered", "usage": {"fresh_input_tokens": 200001, "output_tokens": 30}, - "requires_rates": ["input_cost_per_token_above_200k_tokens", "output_cost_per_token_above_200k_tokens"] + "expected": { + "azure/gpt-5.4-mini": {"spend": 256.04448, "input_cost": 256.00128, "output_cost": 0.0432, "prompt_tokens": 200001, "completion_tokens": 30}, + "azure/gpt-5.6": {"spend": 240.0417, "input_cost": 240.0012, "output_cost": 0.0405, "prompt_tokens": 200001, "completion_tokens": 30}, + "gemini-3.1-pro-preview": {"spend": 336.05838, "input_cost": 336.00168, "output_cost": 0.0567, "prompt_tokens": 200001, "completion_tokens": 30}, + "gemini-3.8-flash": {"spend": 320.0556, "input_cost": 320.0016, "output_cost": 0.054, "prompt_tokens": 200001, "completion_tokens": 30}, + "gemini/gemini-3.1-pro-preview": {"spend": 144.02502, "input_cost": 144.00072, "output_cost": 0.0243, "prompt_tokens": 200001, "completion_tokens": 30}, + "gemini/gemini-3.8-flash": {"spend": 128.02224, "input_cost": 128.00064, "output_cost": 0.0216, "prompt_tokens": 200001, "completion_tokens": 30}, + "gpt-5.3-codex": {"spend": 48.00834, "input_cost": 48.00024, "output_cost": 0.0081, "prompt_tokens": 200001, "completion_tokens": 30}, + "gpt-5.4-mini": {"spend": 64.01112, "input_cost": 64.00032, "output_cost": 0.0108, "prompt_tokens": 200001, "completion_tokens": 30}, + "gpt-5.5-pro": {"spend": 32.00556, "input_cost": 32.00016, "output_cost": 0.0054, "prompt_tokens": 200001, "completion_tokens": 30}, + "gpt-5.6": {"spend": 16.00278, "input_cost": 16.00008, "output_cost": 0.0027, "prompt_tokens": 200001, "completion_tokens": 30}, + "together_ai/moonshotai/Kimi-K3": {"spend": 160.0278, "input_cost": 160.0008, "output_cost": 0.027, "prompt_tokens": 200001, "completion_tokens": 30}, + "together_ai/zai-org/GLM-5.3": {"spend": 176.03058, "input_cost": 176.00088, "output_cost": 0.0297, "prompt_tokens": 200001, "completion_tokens": 30} + } }, { "name": "service_tier_flex", "usage": {"fresh_input_tokens": 120, "output_tokens": 40}, "service_tier": "flex", - "requires_rates": ["input_cost_per_token_flex", "output_cost_per_token_flex"] + "expected": { + "azure/gpt-5.4-mini": {"spend": 0.0448, "input_cost": 0.0288, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.6": {"spend": 0.042, "input_cost": 0.027, "output_cost": 0.015, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.1-pro-preview": {"spend": 0.0588, "input_cost": 0.0378, "output_cost": 0.021, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.8-flash": {"spend": 0.056, "input_cost": 0.036, "output_cost": 0.02, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.0252, "input_cost": 0.0162, "output_cost": 0.009, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.8-flash": {"spend": 0.0224, "input_cost": 0.0144, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.3-codex": {"spend": 0.0084, "input_cost": 0.0054, "output_cost": 0.003, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.4-mini": {"spend": 0.0112, "input_cost": 0.0072, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.0056, "input_cost": 0.0036, "output_cost": 0.002, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.6": {"spend": 0.0028, "input_cost": 0.0018, "output_cost": 0.001, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.028, "input_cost": 0.018, "output_cost": 0.01, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.0308, "input_cost": 0.0198, "output_cost": 0.011, "prompt_tokens": 120, "completion_tokens": 40} + } }, { "name": "service_tier_priority", "usage": {"fresh_input_tokens": 120, "output_tokens": 40}, "service_tier": "priority", - "requires_rates": ["input_cost_per_token_priority", "output_cost_per_token_priority"] + "expected": { + "azure/gpt-5.4-mini": {"spend": 0.04992, "input_cost": 0.03264, "output_cost": 0.01728, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.6": {"spend": 0.0468, "input_cost": 0.0306, "output_cost": 0.0162, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.1-pro-preview": {"spend": 0.06552, "input_cost": 0.04284, "output_cost": 0.02268, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.8-flash": {"spend": 0.0624, "input_cost": 0.0408, "output_cost": 0.0216, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.02808, "input_cost": 0.01836, "output_cost": 0.00972, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.8-flash": {"spend": 0.02496, "input_cost": 0.01632, "output_cost": 0.00864, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.3-codex": {"spend": 0.00936, "input_cost": 0.00612, "output_cost": 0.00324, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.4-mini": {"spend": 0.01248, "input_cost": 0.00816, "output_cost": 0.00432, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.00624, "input_cost": 0.00408, "output_cost": 0.00216, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.6": {"spend": 0.00312, "input_cost": 0.00204, "output_cost": 0.00108, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.0312, "input_cost": 0.0204, "output_cost": 0.0108, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.03432, "input_cost": 0.02244, "output_cost": 0.01188, "prompt_tokens": 120, "completion_tokens": 40} + } }, { "name": "web_search", "usage": {"fresh_input_tokens": 100, "output_tokens": 30, "web_search_calls": 3}, - "requires_rates": ["search_context_cost_per_query"], - "requires_caps": ["web_search"], - "wires": ["openai_responses", "anthropic_messages", "gemini_generate", "vertex_generate"] + "expected": { + "claude-haiku-4-5": {"spend": 0.0712, "input_cost": 0.007, "output_cost": 0.0042, "prompt_tokens": 100, "completion_tokens": 30}, + "claude-opus-5": {"spend": 0.068, "input_cost": 0.005, "output_cost": 0.003, "prompt_tokens": 100, "completion_tokens": 30}, + "claude-sonnet-5": {"spend": 0.0696, "input_cost": 0.006, "output_cost": 0.0036, "prompt_tokens": 100, "completion_tokens": 30}, + "gemini-3.1-pro-preview": {"spend": 0.0936, "input_cost": 0.021, "output_cost": 0.0126, "prompt_tokens": 100, "completion_tokens": 30}, + "gemini-3.8-flash": {"spend": 0.092, "input_cost": 0.02, "output_cost": 0.012, "prompt_tokens": 100, "completion_tokens": 30}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.0744, "input_cost": 0.009, "output_cost": 0.0054, "prompt_tokens": 100, "completion_tokens": 30}, + "gemini/gemini-3.8-flash": {"spend": 0.0728, "input_cost": 0.008, "output_cost": 0.0048, "prompt_tokens": 100, "completion_tokens": 30}, + "gpt-5.3-codex": {"spend": 0.0648, "input_cost": 0.003, "output_cost": 0.0018, "prompt_tokens": 100, "completion_tokens": 30}, + "gpt-5.5-pro": {"spend": 0.0632, "input_cost": 0.002, "output_cost": 0.0012, "prompt_tokens": 100, "completion_tokens": 30} + } }, { "name": "web_search_single", "usage": {"fresh_input_tokens": 100, "output_tokens": 30, "web_search_calls": 1}, - "requires_rates": ["search_context_cost_per_query"], - "requires_caps": ["web_search"], - "wires": ["openai_chat", "together_chat", "fireworks_chat", "azure_chat"] + "expected": { + "azure/gpt-5.4-mini": {"spend": 0.0456, "input_cost": 0.016, "output_cost": 0.0096, "prompt_tokens": 100, "completion_tokens": 30}, + "azure/gpt-5.6": {"spend": 0.044, "input_cost": 0.015, "output_cost": 0.009, "prompt_tokens": 100, "completion_tokens": 30}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0424, "input_cost": 0.014, "output_cost": 0.0084, "prompt_tokens": 100, "completion_tokens": 30}, + "fireworks_ai/kimi-k3": {"spend": 0.0392, "input_cost": 0.012, "output_cost": 0.0072, "prompt_tokens": 100, "completion_tokens": 30}, + "fireworks_ai/qwen3p8-max": {"spend": 0.0408, "input_cost": 0.013, "output_cost": 0.0078, "prompt_tokens": 100, "completion_tokens": 30}, + "gpt-5.4-mini": {"spend": 0.0264, "input_cost": 0.004, "output_cost": 0.0024, "prompt_tokens": 100, "completion_tokens": 30}, + "gpt-5.6": {"spend": 0.0216, "input_cost": 0.001, "output_cost": 0.0006, "prompt_tokens": 100, "completion_tokens": 30}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.036, "input_cost": 0.01, "output_cost": 0.006, "prompt_tokens": 100, "completion_tokens": 30}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.0376, "input_cost": 0.011, "output_cost": 0.0066, "prompt_tokens": 100, "completion_tokens": 30} + } }, { "name": "stream", "usage": {"fresh_input_tokens": 120, "output_tokens": 40}, - "stream": true + "stream": true, + "expected": { + "anthropic.claude-sonnet-5-v1:0": {"spend": 0.034, "input_cost": 0.0204, "output_cost": 0.0136, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.6": {"spend": 0.03, "input_cost": 0.018, "output_cost": 0.012, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-haiku-4-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-opus-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-sonnet-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/kimi-k3": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/qwen3p8-max": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.1-pro-preview": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.8-flash": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.8-flash": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.4-mini": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.6": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40}, + "meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.038, "input_cost": 0.0228, "output_cost": 0.0152, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40}, + "us.anthropic.claude-opus-5-v1:0": {"spend": 0.036, "input_cost": 0.0216, "output_cost": 0.0144, "prompt_tokens": 120, "completion_tokens": 40} + } }, { "name": "stream_no_usage", @@ -83,33 +256,137 @@ "stream": true, "stream_usage": "absent", "exact_spend": false, - "requires_caps": ["absent_usage"] + "models": [ + "anthropic.claude-sonnet-5-v1:0", + "azure/gpt-5.4-mini", + "azure/gpt-5.6", + "claude-haiku-4-5", + "claude-opus-5", + "claude-sonnet-5", + "fireworks_ai/deepseek-v4p1-flash", + "fireworks_ai/kimi-k3", + "fireworks_ai/qwen3p8-max", + "gemini-3.1-pro-preview", + "gemini-3.8-flash", + "gemini/gemini-3.1-pro-preview", + "gemini/gemini-3.8-flash", + "gpt-5.3-codex", + "gpt-5.4-mini", + "gpt-5.5-pro", + "gpt-5.6", + "meta.llama4-maverick-17b-instruct-v1:0", + "together_ai/moonshotai/Kimi-K3", + "together_ai/zai-org/GLM-5.3", + "us.anthropic.claude-opus-5-v1:0" + ] }, { "name": "response_model_override", "usage": {"fresh_input_tokens": 120, "output_tokens": 40}, "response_model_override": true, - "requires_caps": ["response_model"] + "expected": { + "azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.6": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-haiku-4-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-opus-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-sonnet-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/kimi-k3": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/qwen3p8-max": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.1-pro-preview": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.8-flash": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.8-flash": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.3-codex": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.4-mini": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.6": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40} + } }, { "name": "stream_response_model_override", "usage": {"fresh_input_tokens": 120, "output_tokens": 40}, "stream": true, "response_model_override": true, - "requires_caps": ["response_model"] + "expected": { + "azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.6": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-haiku-4-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-opus-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-sonnet-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/kimi-k3": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/qwen3p8-max": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.1-pro-preview": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.8-flash": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.8-flash": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.3-codex": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.4-mini": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.6": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40} + } }, { "name": "tool_call", "usage": {"fresh_input_tokens": 120, "output_tokens": 40}, "tool_call": true, - "requires_caps": ["tool_call"] + "expected": { + "anthropic.claude-sonnet-5-v1:0": {"spend": 0.034, "input_cost": 0.0204, "output_cost": 0.0136, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40}, + "azure/gpt-5.6": {"spend": 0.03, "input_cost": 0.018, "output_cost": 0.012, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-haiku-4-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-opus-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40}, + "claude-sonnet-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/kimi-k3": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40}, + "fireworks_ai/qwen3p8-max": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.1-pro-preview": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini-3.8-flash": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40}, + "gemini/gemini-3.8-flash": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.4-mini": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.6": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40}, + "meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.038, "input_cost": 0.0228, "output_cost": 0.0152, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40}, + "us.anthropic.claude-opus-5-v1:0": {"spend": 0.036, "input_cost": 0.0216, "output_cost": 0.0144, "prompt_tokens": 120, "completion_tokens": 40} + } }, { "name": "stream_tool_call", "usage": {"fresh_input_tokens": 80, "output_tokens": 25}, "stream": true, "tool_call": true, - "requires_caps": ["tool_call"] + "expected": { + "anthropic.claude-sonnet-5-v1:0": {"spend": 0.0221, "input_cost": 0.0136, "output_cost": 0.0085, "prompt_tokens": 80, "completion_tokens": 25}, + "azure/gpt-5.4-mini": {"spend": 0.0208, "input_cost": 0.0128, "output_cost": 0.008, "prompt_tokens": 80, "completion_tokens": 25}, + "azure/gpt-5.6": {"spend": 0.0195, "input_cost": 0.012, "output_cost": 0.0075, "prompt_tokens": 80, "completion_tokens": 25}, + "claude-haiku-4-5": {"spend": 0.0091, "input_cost": 0.0056, "output_cost": 0.0035, "prompt_tokens": 80, "completion_tokens": 25}, + "claude-opus-5": {"spend": 0.0065, "input_cost": 0.004, "output_cost": 0.0025, "prompt_tokens": 80, "completion_tokens": 25}, + "claude-sonnet-5": {"spend": 0.0078, "input_cost": 0.0048, "output_cost": 0.003, "prompt_tokens": 80, "completion_tokens": 25}, + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0182, "input_cost": 0.0112, "output_cost": 0.007, "prompt_tokens": 80, "completion_tokens": 25}, + "fireworks_ai/kimi-k3": {"spend": 0.0156, "input_cost": 0.0096, "output_cost": 0.006, "prompt_tokens": 80, "completion_tokens": 25}, + "fireworks_ai/qwen3p8-max": {"spend": 0.0169, "input_cost": 0.0104, "output_cost": 0.0065, "prompt_tokens": 80, "completion_tokens": 25}, + "gemini-3.1-pro-preview": {"spend": 0.0273, "input_cost": 0.0168, "output_cost": 0.0105, "prompt_tokens": 80, "completion_tokens": 25}, + "gemini-3.8-flash": {"spend": 0.026, "input_cost": 0.016, "output_cost": 0.01, "prompt_tokens": 80, "completion_tokens": 25}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.0117, "input_cost": 0.0072, "output_cost": 0.0045, "prompt_tokens": 80, "completion_tokens": 25}, + "gemini/gemini-3.8-flash": {"spend": 0.0104, "input_cost": 0.0064, "output_cost": 0.004, "prompt_tokens": 80, "completion_tokens": 25}, + "gpt-5.3-codex": {"spend": 0.0039, "input_cost": 0.0024, "output_cost": 0.0015, "prompt_tokens": 80, "completion_tokens": 25}, + "gpt-5.4-mini": {"spend": 0.0052, "input_cost": 0.0032, "output_cost": 0.002, "prompt_tokens": 80, "completion_tokens": 25}, + "gpt-5.5-pro": {"spend": 0.0026, "input_cost": 0.0016, "output_cost": 0.001, "prompt_tokens": 80, "completion_tokens": 25}, + "gpt-5.6": {"spend": 0.0013, "input_cost": 0.0008, "output_cost": 0.0005, "prompt_tokens": 80, "completion_tokens": 25}, + "meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.0247, "input_cost": 0.0152, "output_cost": 0.0095, "prompt_tokens": 80, "completion_tokens": 25}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.013, "input_cost": 0.008, "output_cost": 0.005, "prompt_tokens": 80, "completion_tokens": 25}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.0143, "input_cost": 0.0088, "output_cost": 0.0055, "prompt_tokens": 80, "completion_tokens": 25}, + "us.anthropic.claude-opus-5-v1:0": {"spend": 0.0234, "input_cost": 0.0144, "output_cost": 0.009, "prompt_tokens": 80, "completion_tokens": 25} + } }, { "name": "stream_no_usage_tool_call", @@ -118,7 +395,29 @@ "stream_usage": "absent", "tool_call": true, "exact_spend": false, - "requires_caps": ["absent_usage", "tool_call"] + "models": [ + "anthropic.claude-sonnet-5-v1:0", + "azure/gpt-5.4-mini", + "azure/gpt-5.6", + "claude-haiku-4-5", + "claude-opus-5", + "claude-sonnet-5", + "fireworks_ai/deepseek-v4p1-flash", + "fireworks_ai/kimi-k3", + "fireworks_ai/qwen3p8-max", + "gemini-3.1-pro-preview", + "gemini-3.8-flash", + "gemini/gemini-3.1-pro-preview", + "gemini/gemini-3.8-flash", + "gpt-5.3-codex", + "gpt-5.4-mini", + "gpt-5.5-pro", + "gpt-5.6", + "meta.llama4-maverick-17b-instruct-v1:0", + "together_ai/moonshotai/Kimi-K3", + "together_ai/zai-org/GLM-5.3", + "us.anthropic.claude-opus-5-v1:0" + ] }, { "name": "stream_no_usage_image_input", @@ -127,14 +426,39 @@ "stream_usage": "absent", "image_input": true, "exact_spend": false, - "requires_caps": ["absent_usage", "image_input"] + "models": [ + "anthropic.claude-sonnet-5-v1:0", + "azure/gpt-5.4-mini", + "azure/gpt-5.6", + "claude-haiku-4-5", + "claude-opus-5", + "claude-sonnet-5", + "fireworks_ai/deepseek-v4p1-flash", + "fireworks_ai/kimi-k3", + "fireworks_ai/qwen3p8-max", + "gemini-3.1-pro-preview", + "gemini-3.8-flash", + "gemini/gemini-3.1-pro-preview", + "gemini/gemini-3.8-flash", + "gpt-5.3-codex", + "gpt-5.4-mini", + "gpt-5.5-pro", + "gpt-5.6", + "meta.llama4-maverick-17b-instruct-v1:0", + "together_ai/moonshotai/Kimi-K3", + "together_ai/zai-org/GLM-5.3", + "us.anthropic.claude-opus-5-v1:0" + ] }, { "name": "stream_incomplete", "usage": {"fresh_input_tokens": 120, "output_tokens": 40}, "stream": true, "terminal": "incomplete", - "requires_caps": ["responses_terminal"] + "expected": { + "gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40} + } }, { "name": "stream_no_usage_incomplete", @@ -143,14 +467,20 @@ "stream_usage": "absent", "terminal": "incomplete", "exact_spend": false, - "requires_caps": ["responses_terminal"] + "models": [ + "gpt-5.3-codex", + "gpt-5.5-pro" + ] }, { "name": "stream_unvalidated", "usage": {"fresh_input_tokens": 120, "output_tokens": 40}, "stream": true, "terminal": "unvalidated", - "requires_caps": ["responses_terminal"] + "expected": { + "gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40} + } }, { "name": "stream_no_usage_unvalidated", @@ -159,14 +489,22 @@ "stream_usage": "absent", "terminal": "unvalidated", "exact_spend": false, - "requires_caps": ["responses_terminal"] + "models": [ + "gpt-5.3-codex", + "gpt-5.5-pro" + ] }, { "name": "prompt_blocked", "usage": {"fresh_input_tokens": 1000, "output_tokens": 0}, "terminal": "prompt_blocked", "response_model_override": true, - "requires_caps": ["prompt_blocked"] + "expected": { + "gemini-3.1-pro-preview": {"spend": 0.2, "input_cost": 0.2, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0}, + "gemini-3.8-flash": {"spend": 0.21, "input_cost": 0.21, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.08, "input_cost": 0.08, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0}, + "gemini/gemini-3.8-flash": {"spend": 0.09, "input_cost": 0.09, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0} + } }, { "name": "stream_prompt_blocked", @@ -174,77 +512,73 @@ "stream": true, "terminal": "prompt_blocked", "response_model_override": true, - "requires_caps": ["prompt_blocked"] + "expected": { + "gemini-3.1-pro-preview": {"spend": 0.2, "input_cost": 0.2, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0}, + "gemini-3.8-flash": {"spend": 0.21, "input_cost": 0.21, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.08, "input_cost": 0.08, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0}, + "gemini/gemini-3.8-flash": {"spend": 0.09, "input_cost": 0.09, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0} + } }, { "name": "all_components_chat", - "usage": { - "fresh_input_tokens": 80, - "cache_read_tokens": 40, - "cache_write_5m_tokens": 20, - "cache_write_1h_tokens": 10, - "output_tokens": 25, - "reasoning_tokens": 15, - "audio_input_tokens": 5, - "audio_output_tokens": 3 - }, - "requires_rates": [ - "output_cost_per_reasoning_token", - "input_cost_per_audio_token", - "output_cost_per_audio_token" - ], - "wires": ["openai_chat", "azure_chat", "together_chat"] + "usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "cache_write_5m_tokens": 20, "cache_write_1h_tokens": 10, "output_tokens": 25, "reasoning_tokens": 15, "audio_input_tokens": 5, "audio_output_tokens": 3}, + "expected": { + "azure/gpt-5.4-mini": {"spend": 0.0576, "input_cost": 0.03424, "output_cost": 0.02336, "prompt_tokens": 155, "completion_tokens": 43}, + "azure/gpt-5.6": {"spend": 0.054, "input_cost": 0.0321, "output_cost": 0.0219, "prompt_tokens": 155, "completion_tokens": 43}, + "gpt-5.4-mini": {"spend": 0.0144, "input_cost": 0.00856, "output_cost": 0.00584, "prompt_tokens": 155, "completion_tokens": 43}, + "gpt-5.6": {"spend": 0.0036, "input_cost": 0.00214, "output_cost": 0.00146, "prompt_tokens": 155, "completion_tokens": 43}, + "together_ai/moonshotai/Kimi-K3": {"spend": 0.036, "input_cost": 0.0214, "output_cost": 0.0146, "prompt_tokens": 155, "completion_tokens": 43}, + "together_ai/zai-org/GLM-5.3": {"spend": 0.0396, "input_cost": 0.02354, "output_cost": 0.01606, "prompt_tokens": 155, "completion_tokens": 43} + } }, { "name": "all_components_fireworks", "usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25}, - "wires": ["fireworks_chat"] + "expected": { + "fireworks_ai/deepseek-v4p1-flash": {"spend": 0.01876, "input_cost": 0.01176, "output_cost": 0.007, "prompt_tokens": 120, "completion_tokens": 25}, + "fireworks_ai/kimi-k3": {"spend": 0.01608, "input_cost": 0.01008, "output_cost": 0.006, "prompt_tokens": 120, "completion_tokens": 25}, + "fireworks_ai/qwen3p8-max": {"spend": 0.01742, "input_cost": 0.01092, "output_cost": 0.0065, "prompt_tokens": 120, "completion_tokens": 25} + } }, { "name": "all_components_anthropic", - "usage": { - "fresh_input_tokens": 80, - "cache_read_tokens": 40, - "cache_write_5m_tokens": 20, - "cache_write_1h_tokens": 10, - "output_tokens": 25 - }, - "wires": ["anthropic_messages", "bedrock_converse"] + "usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "cache_write_5m_tokens": 20, "cache_write_1h_tokens": 10, "output_tokens": 25}, + "expected": { + "anthropic.claude-sonnet-5-v1:0": {"spend": 0.03978, "input_cost": 0.03128, "output_cost": 0.0085, "prompt_tokens": 150, "completion_tokens": 25}, + "claude-haiku-4-5": {"spend": 0.01638, "input_cost": 0.01288, "output_cost": 0.0035, "prompt_tokens": 150, "completion_tokens": 25}, + "claude-opus-5": {"spend": 0.0117, "input_cost": 0.0092, "output_cost": 0.0025, "prompt_tokens": 150, "completion_tokens": 25}, + "claude-sonnet-5": {"spend": 0.01404, "input_cost": 0.01104, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 25}, + "meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.038, "input_cost": 0.0285, "output_cost": 0.0095, "prompt_tokens": 150, "completion_tokens": 25}, + "us.anthropic.claude-opus-5-v1:0": {"spend": 0.04212, "input_cost": 0.03312, "output_cost": 0.009, "prompt_tokens": 150, "completion_tokens": 25} + } }, { "name": "all_components_anthropic_stream", - "usage": { - "fresh_input_tokens": 80, - "cache_read_tokens": 40, - "cache_write_5m_tokens": 20, - "cache_write_1h_tokens": 10, - "output_tokens": 25 - }, + "usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "cache_write_5m_tokens": 20, "cache_write_1h_tokens": 10, "output_tokens": 25}, "stream": true, - "wires": ["anthropic_messages"] + "expected": { + "claude-haiku-4-5": {"spend": 0.01638, "input_cost": 0.01288, "output_cost": 0.0035, "prompt_tokens": 150, "completion_tokens": 25}, + "claude-opus-5": {"spend": 0.0117, "input_cost": 0.0092, "output_cost": 0.0025, "prompt_tokens": 150, "completion_tokens": 25}, + "claude-sonnet-5": {"spend": 0.01404, "input_cost": 0.01104, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 25} + } }, { "name": "all_components_gemini", - "usage": { - "fresh_input_tokens": 80, - "cache_read_tokens": 40, - "output_tokens": 25, - "reasoning_tokens": 15, - "audio_input_tokens": 5, - "audio_output_tokens": 3 - }, - "requires_rates": [ - "output_cost_per_reasoning_token", - "input_cost_per_audio_token", - "output_cost_per_audio_token" - ], - "wires": ["gemini_generate", "vertex_generate"] + "usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25, "reasoning_tokens": 15, "audio_input_tokens": 5, "audio_output_tokens": 3}, + "expected": { + "gemini-3.1-pro-preview": {"spend": 0.0546, "input_cost": 0.02394, "output_cost": 0.03066, "prompt_tokens": 125, "completion_tokens": 43}, + "gemini-3.8-flash": {"spend": 0.052, "input_cost": 0.0228, "output_cost": 0.0292, "prompt_tokens": 125, "completion_tokens": 43}, + "gemini/gemini-3.1-pro-preview": {"spend": 0.0234, "input_cost": 0.01026, "output_cost": 0.01314, "prompt_tokens": 125, "completion_tokens": 43}, + "gemini/gemini-3.8-flash": {"spend": 0.0208, "input_cost": 0.00912, "output_cost": 0.01168, "prompt_tokens": 125, "completion_tokens": 43} + } }, { "name": "all_components_responses", "usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25, "reasoning_tokens": 15}, - "requires_rates": ["output_cost_per_reasoning_token"], - "wires": ["openai_responses"] + "expected": { + "gpt-5.3-codex": {"spend": 0.00627, "input_cost": 0.00252, "output_cost": 0.00375, "prompt_tokens": 120, "completion_tokens": 40}, + "gpt-5.5-pro": {"spend": 0.00418, "input_cost": 0.00168, "output_cost": 0.0025, "prompt_tokens": 120, "completion_tokens": 40} + } } ] } diff --git a/tests/e2e/cost_calculation/conftest.py b/tests/e2e/cost_calculation/conftest.py index 3de9786854e..1473edb119b 100644 --- a/tests/e2e/cost_calculation/conftest.py +++ b/tests/e2e/cost_calculation/conftest.py @@ -2,9 +2,8 @@ Runs against a dedicated proxy whose whole model cost map is the test-owned ``tests/e2e/cost_map.json`` (LITELLM_MODEL_COST_MAP_URL); every map entry is a -deployment under test, the request shapes live in ``cases.json``, and the -asserted goldens live in ``expected.json`` (regenerate proposals with -``generate_expected.py``). Provider calls are answered by the +deployment under test, and the request shapes plus asserted goldens live in +``cases.json``. Provider calls are answered by the scripted-provider sidecar (``scripted_provider.py``), registered per scenario over its control API. diff --git a/tests/e2e/cost_calculation/cost_matrix.py b/tests/e2e/cost_calculation/cost_matrix.py index 68f3186809d..7999d827060 100644 --- a/tests/e2e/cost_calculation/cost_matrix.py +++ b/tests/e2e/cost_calculation/cost_matrix.py @@ -1,16 +1,13 @@ """The cost-calculation matrix: the model set derived from the test cost map, the request/response cases from ``cases.json``, and the loaders both use. -Three data files drive the suite; nothing in Python lists models or cases: +Two data files drive the suite; nothing in Python lists models or cases: - ``tests/e2e/cost_map.json`` is the proxy's ENTIRE model cost map (LITELLM_MODEL_COST_MAP_URL); every entry becomes a deployment under test. -- ``tests/e2e/cost_calculation/cases.json`` is the case list; each case runs - for a model when the entry carries the rates it exercises (``requires_rates``) - and the wire can report the token kinds involved (``requires_caps`` / - ``wires``). -- ``tests/e2e/cost_calculation/expected.json`` holds the reviewed goldens; the - tests assert them verbatim and never compute a price themselves. The rate - arithmetic that proposes goldens lives in ``generate_expected.py``, not here. +- ``tests/e2e/cost_calculation/cases.json`` is the case list plus the reviewed + goldens: each exact-spend case carries an ``expected`` cell per map key it + runs against, each recount case carries its ``models`` list, so matrix + membership and expected values are literal data read side by side. """ from __future__ import annotations @@ -26,13 +23,11 @@ from pathlib import Path from types import MappingProxyType from typing import Final, Literal -from pydantic import BaseModel, ConfigDict, TypeAdapter +from pydantic import BaseModel, ConfigDict, Field, TypeAdapter from scripted_provider import Scenario, ScriptedOutput, ScriptedToolCall, ScriptedUsage, Wire COST_MAP_PATH: Final = Path(__file__).resolve().parent.parent / "cost_map.json" CASES_PATH: Final = Path(__file__).resolve().parent / "cases.json" -EXPECTED_PATH: Final = Path(__file__).resolve().parent / "expected.json" - class SearchContextCostPerQuery(BaseModel): model_config = ConfigDict(frozen=True) @@ -88,10 +83,20 @@ class DeploymentSpec(BaseModel): base_model: str | None = None +class ExpectedCell(BaseModel): + model_config = ConfigDict(frozen=True) + + spend: float + input_cost: float + output_cost: float + prompt_tokens: int + completion_tokens: int + + class Case(BaseModel): - """One request/response shape from cases.json; gated onto a model by - ``requires_rates`` (entry must carry each rate field), ``requires_caps`` - (the wire must report the token kind) and ``wires`` (shape is wire-specific).""" + """One request/response shape from cases.json. An exact-spend case names + its models implicitly by carrying one ``expected`` golden per map key; a + recount case (``exact_spend=False``) names them in ``models`` instead.""" model_config = ConfigDict(frozen=True) @@ -105,19 +110,16 @@ class Case(BaseModel): tool_call: bool = False image_input: bool = False terminal: Literal["completed", "incomplete", "unvalidated", "prompt_blocked"] = "completed" - requires_rates: tuple[str, ...] = () - requires_caps: tuple[str, ...] = () - wires: tuple[Wire, ...] | None = None + expected: Mapping[str, ExpectedCell] = Field(default_factory=lambda: MappingProxyType({})) + models: tuple[str, ...] = () def applies_to(self, model: FrontierModel) -> bool: - if self.wires is not None and model.wire not in self.wires: - return False - caps: Final = _WIRE_CAPS[model.wire] - if not frozenset(self.requires_caps) <= caps: - return False - return all( - getattr(model.rates, field, None) is not None for field in self.requires_rates - ) + if self.exact_spend: + return model.map_key in self.expected + return model.map_key in self.models + + def expected_for(self, model: FrontierModel) -> ExpectedCell: + return self.expected[model.map_key] def scenario(self, scenario_id: str, model: FrontierModel, text: str) -> Scenario: return Scenario( @@ -309,64 +311,6 @@ def _frontier() -> tuple[FrontierModel, ...]: FRONTIER_MODELS: Final[tuple[FrontierModel, ...]] = _frontier() -# Token kinds each wire can report, gating which pricing cases apply. -_WIRE_CAPS: Final[Mapping[str, frozenset[str]]] = MappingProxyType({ - "openai_chat": frozenset( - { - "cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio", - "web_search", "response_model", "absent_usage", "tool_call", "image_input", - } - ), - "openai_responses": frozenset( - { - "cache_read", "reasoning", "web_search", "response_model", "absent_usage", - "tool_call", "image_input", "responses_terminal", - } - ), - "anthropic_messages": frozenset( - { - "cache_read", "cache_write_5m", "cache_write_1h", "web_search", - "response_model", "absent_usage", "tool_call", "image_input", - } - ), - "gemini_generate": frozenset( - { - "cache_read", "reasoning", "audio", "web_search", "response_model", - "absent_usage", "tool_call", "image_input", "prompt_blocked", - } - ), - "together_chat": frozenset( - { - "cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio", - "web_search", "response_model", "absent_usage", "tool_call", "image_input", - } - ), - "fireworks_chat": frozenset( - { - "cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio", - "web_search", "response_model", "absent_usage", "tool_call", "image_input", - } - ), - "azure_chat": frozenset( - { - "cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio", - "web_search", "response_model", "absent_usage", "tool_call", "image_input", - } - ), - "bedrock_converse": frozenset( - { - "cache_read", "cache_write_5m", "cache_write_1h", "absent_usage", - "tool_call", "image_input", - } - ), - "vertex_generate": frozenset( - { - "cache_read", "reasoning", "audio", "web_search", "response_model", - "absent_usage", "tool_call", "image_input", "prompt_blocked", - } - ), -}) - TOOL_CALL_ARGUMENTS: Final = json.dumps({ "city": "Berlin", "days": 7, @@ -415,64 +359,43 @@ def image_input_data_url() -> str: IMAGE_INPUT_DATA_URL: Final = image_input_data_url() -class ExpectedCell(BaseModel): - model_config = ConfigDict(frozen=True) - - spend: float - input_cost: float - output_cost: float - prompt_tokens: int - completion_tokens: int - - -_EXPECTED_ADAPTER: Final = TypeAdapter(dict[str, ExpectedCell]) -EXPECTED: Final[Mapping[str, ExpectedCell]] = MappingProxyType( - _EXPECTED_ADAPTER.validate_python(json.loads(EXPECTED_PATH.read_text())) - if EXPECTED_PATH.exists() - else {} -) - - -def expected_key(model: FrontierModel, case: Case) -> str: - return f"{model.map_key}|{case.name}" - - def matrix_data_errors() -> tuple[str, ...]: - """Freshness findings for the data files, as human-readable strings. + """Consistency findings for the data files, as human-readable strings. - Called at collection time by the e2e suite; also usable from - generate_expected.py's context without importing pytest. + Called at collection time by the e2e suite, so a map key named by a case + but absent from cost_map.json fails the suite's collection loudly. """ - derived: Final = { - expected_key(model, case) - for model in FRONTIER_MODELS - for case in cases_for(model) - if case.exact_spend - } - golden: Final = set(EXPECTED) unknown_deployments: Final = sorted( spec.map_key for spec in CASES_FILE.deployments if spec.map_key not in COST_MAP ) - unknown_rates: Final = sorted( - {field for case in CASES for field in case.requires_rates} - set(CostMapEntry.model_fields) + unknown_case_models: Final = sorted( + { + map_key + for case in CASES + for map_key in (*case.expected, *case.models) + if map_key not in COST_MAP + } + ) + misshapen_cases: Final = sorted( + case.name + for case in CASES + if case.exact_spend == bool(case.models) or case.exact_spend != bool(case.expected) ) input_rates: Final = tuple(entry.input_cost_per_token for entry in COST_MAP.values()) findings: Final = ( - ( - "expected.json is out of sync with the derived matrix; run " - "uv run python tests/e2e/cost_calculation/generate_expected.py " - f"(missing: {sorted(derived - golden)}; stale: {sorted(golden - derived)})" - ) - if derived != golden - else None, ( f"deployments entries name map keys absent from cost_map.json: {unknown_deployments}" if unknown_deployments else None ), ( - f"requires_rates names that are not CostMapEntry fields: {unknown_rates}" - if unknown_rates + f"case expected/models name map keys absent from cost_map.json: {unknown_case_models}" + if unknown_case_models + else None + ), + ( + f"cases must carry expected xor models (exact_spend matches the field): {misshapen_cases}" + if misshapen_cases else None ), ( diff --git a/tests/e2e/cost_calculation/expected.json b/tests/e2e/cost_calculation/expected.json deleted file mode 100644 index 984b670a82c..00000000000 --- a/tests/e2e/cost_calculation/expected.json +++ /dev/null @@ -1,2004 +0,0 @@ -{ - "anthropic.claude-sonnet-5-v1:0|all_components_anthropic": { - "completion_tokens": 25, - "input_cost": 0.03128, - "output_cost": 0.0085, - "prompt_tokens": 150, - "spend": 0.03978 - }, - "anthropic.claude-sonnet-5-v1:0|basic": { - "completion_tokens": 40, - "input_cost": 0.0204, - "output_cost": 0.013600000000000001, - "prompt_tokens": 120, - "spend": 0.034 - }, - "anthropic.claude-sonnet-5-v1:0|cache_read": { - "completion_tokens": 30, - "input_cost": 0.01785, - "output_cost": 0.0102, - "prompt_tokens": 150, - "spend": 0.028050000000000002 - }, - "anthropic.claude-sonnet-5-v1:0|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.052700000000000004, - "output_cost": 0.0102, - "prompt_tokens": 150, - "spend": 0.06290000000000001 - }, - "anthropic.claude-sonnet-5-v1:0|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.0459, - "output_cost": 0.0102, - "prompt_tokens": 150, - "spend": 0.056100000000000004 - }, - "anthropic.claude-sonnet-5-v1:0|stream": { - "completion_tokens": 40, - "input_cost": 0.0204, - "output_cost": 0.013600000000000001, - "prompt_tokens": 120, - "spend": 0.034 - }, - "anthropic.claude-sonnet-5-v1:0|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.013600000000000001, - "output_cost": 0.0085, - "prompt_tokens": 80, - "spend": 0.0221 - }, - "anthropic.claude-sonnet-5-v1:0|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0204, - "output_cost": 0.013600000000000001, - "prompt_tokens": 120, - "spend": 0.034 - }, - "azure/gpt-5.4-mini|all_components_chat": { - "completion_tokens": 43, - "input_cost": 0.03424, - "output_cost": 0.02336, - "prompt_tokens": 155, - "spend": 0.0576 - }, - "azure/gpt-5.4-mini|audio": { - "completion_tokens": 45, - "input_cost": 0.04, - "output_cost": 0.0264, - "prompt_tokens": 125, - "spend": 0.0664 - }, - "azure/gpt-5.4-mini|basic": { - "completion_tokens": 40, - "input_cost": 0.019200000000000002, - "output_cost": 0.0128, - "prompt_tokens": 120, - "spend": 0.032 - }, - "azure/gpt-5.4-mini|cache_read": { - "completion_tokens": 30, - "input_cost": 0.0168, - "output_cost": 0.009600000000000001, - "prompt_tokens": 150, - "spend": 0.0264 - }, - "azure/gpt-5.4-mini|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.049600000000000005, - "output_cost": 0.009600000000000001, - "prompt_tokens": 150, - "spend": 0.0592 - }, - "azure/gpt-5.4-mini|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.0432, - "output_cost": 0.009600000000000001, - "prompt_tokens": 150, - "spend": 0.0528 - }, - "azure/gpt-5.4-mini|reasoning": { - "completion_tokens": 100, - "input_cost": 0.016, - "output_cost": 0.0656, - "prompt_tokens": 100, - "spend": 0.0816 - }, - "azure/gpt-5.4-mini|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.019200000000000002, - "output_cost": 0.0128, - "prompt_tokens": 120, - "spend": 0.032 - }, - "azure/gpt-5.4-mini|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.0288, - "output_cost": 0.016, - "prompt_tokens": 120, - "spend": 0.0448 - }, - "azure/gpt-5.4-mini|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.03264, - "output_cost": 0.01728, - "prompt_tokens": 120, - "spend": 0.049920000000000006 - }, - "azure/gpt-5.4-mini|stream": { - "completion_tokens": 40, - "input_cost": 0.019200000000000002, - "output_cost": 0.0128, - "prompt_tokens": 120, - "spend": 0.032 - }, - "azure/gpt-5.4-mini|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.019200000000000002, - "output_cost": 0.0128, - "prompt_tokens": 120, - "spend": 0.032 - }, - "azure/gpt-5.4-mini|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.0128, - "output_cost": 0.008, - "prompt_tokens": 80, - "spend": 0.0208 - }, - "azure/gpt-5.4-mini|tiered": { - "completion_tokens": 30, - "input_cost": 256.00128, - "output_cost": 0.0432, - "prompt_tokens": 200001, - "spend": 256.04448 - }, - "azure/gpt-5.4-mini|tool_call": { - "completion_tokens": 40, - "input_cost": 0.019200000000000002, - "output_cost": 0.0128, - "prompt_tokens": 120, - "spend": 0.032 - }, - "azure/gpt-5.4-mini|web_search_single": { - "completion_tokens": 30, - "input_cost": 0.016, - "output_cost": 0.009600000000000001, - "prompt_tokens": 100, - "spend": 0.0456 - }, - "azure/gpt-5.6|all_components_chat": { - "completion_tokens": 43, - "input_cost": 0.0321, - "output_cost": 0.0219, - "prompt_tokens": 155, - "spend": 0.05399999999999999 - }, - "azure/gpt-5.6|audio": { - "completion_tokens": 45, - "input_cost": 0.0375, - "output_cost": 0.02475, - "prompt_tokens": 125, - "spend": 0.06225 - }, - "azure/gpt-5.6|basic": { - "completion_tokens": 40, - "input_cost": 0.018, - "output_cost": 0.011999999999999999, - "prompt_tokens": 120, - "spend": 0.03 - }, - "azure/gpt-5.6|cache_read": { - "completion_tokens": 30, - "input_cost": 0.01575, - "output_cost": 0.009, - "prompt_tokens": 150, - "spend": 0.02475 - }, - "azure/gpt-5.6|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.0465, - "output_cost": 0.009, - "prompt_tokens": 150, - "spend": 0.0555 - }, - "azure/gpt-5.6|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.040499999999999994, - "output_cost": 0.009, - "prompt_tokens": 150, - "spend": 0.049499999999999995 - }, - "azure/gpt-5.6|reasoning": { - "completion_tokens": 100, - "input_cost": 0.015, - "output_cost": 0.0615, - "prompt_tokens": 100, - "spend": 0.0765 - }, - "azure/gpt-5.6|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.019200000000000002, - "output_cost": 0.0128, - "prompt_tokens": 120, - "spend": 0.032 - }, - "azure/gpt-5.6|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.027, - "output_cost": 0.015, - "prompt_tokens": 120, - "spend": 0.041999999999999996 - }, - "azure/gpt-5.6|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.030600000000000002, - "output_cost": 0.0162, - "prompt_tokens": 120, - "spend": 0.0468 - }, - "azure/gpt-5.6|stream": { - "completion_tokens": 40, - "input_cost": 0.018, - "output_cost": 0.011999999999999999, - "prompt_tokens": 120, - "spend": 0.03 - }, - "azure/gpt-5.6|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.019200000000000002, - "output_cost": 0.0128, - "prompt_tokens": 120, - "spend": 0.032 - }, - "azure/gpt-5.6|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.011999999999999999, - "output_cost": 0.0075, - "prompt_tokens": 80, - "spend": 0.019499999999999997 - }, - "azure/gpt-5.6|tiered": { - "completion_tokens": 30, - "input_cost": 240.00119999999998, - "output_cost": 0.0405, - "prompt_tokens": 200001, - "spend": 240.0417 - }, - "azure/gpt-5.6|tool_call": { - "completion_tokens": 40, - "input_cost": 0.018, - "output_cost": 0.011999999999999999, - "prompt_tokens": 120, - "spend": 0.03 - }, - "azure/gpt-5.6|web_search_single": { - "completion_tokens": 30, - "input_cost": 0.015, - "output_cost": 0.009, - "prompt_tokens": 100, - "spend": 0.044 - }, - "claude-haiku-4-5|all_components_anthropic": { - "completion_tokens": 25, - "input_cost": 0.012880000000000003, - "output_cost": 0.0035000000000000005, - "prompt_tokens": 150, - "spend": 0.016380000000000002 - }, - "claude-haiku-4-5|all_components_anthropic_stream": { - "completion_tokens": 25, - "input_cost": 0.012880000000000003, - "output_cost": 0.0035000000000000005, - "prompt_tokens": 150, - "spend": 0.016380000000000002 - }, - "claude-haiku-4-5|basic": { - "completion_tokens": 40, - "input_cost": 0.008400000000000001, - "output_cost": 0.005600000000000001, - "prompt_tokens": 120, - "spend": 0.014000000000000002 - }, - "claude-haiku-4-5|cache_read": { - "completion_tokens": 30, - "input_cost": 0.007350000000000001, - "output_cost": 0.004200000000000001, - "prompt_tokens": 150, - "spend": 0.011550000000000001 - }, - "claude-haiku-4-5|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.021700000000000004, - "output_cost": 0.004200000000000001, - "prompt_tokens": 150, - "spend": 0.025900000000000006 - }, - "claude-haiku-4-5|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.0189, - "output_cost": 0.004200000000000001, - "prompt_tokens": 150, - "spend": 0.023100000000000002 - }, - "claude-haiku-4-5|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.006, - "output_cost": 0.004, - "prompt_tokens": 120, - "spend": 0.01 - }, - "claude-haiku-4-5|stream": { - "completion_tokens": 40, - "input_cost": 0.008400000000000001, - "output_cost": 0.005600000000000001, - "prompt_tokens": 120, - "spend": 0.014000000000000002 - }, - "claude-haiku-4-5|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.006, - "output_cost": 0.004, - "prompt_tokens": 120, - "spend": 0.01 - }, - "claude-haiku-4-5|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.005600000000000001, - "output_cost": 0.0035000000000000005, - "prompt_tokens": 80, - "spend": 0.0091 - }, - "claude-haiku-4-5|tool_call": { - "completion_tokens": 40, - "input_cost": 0.008400000000000001, - "output_cost": 0.005600000000000001, - "prompt_tokens": 120, - "spend": 0.014000000000000002 - }, - "claude-haiku-4-5|web_search": { - "completion_tokens": 30, - "input_cost": 0.007000000000000001, - "output_cost": 0.004200000000000001, - "prompt_tokens": 100, - "spend": 0.0712 - }, - "claude-opus-5|all_components_anthropic": { - "completion_tokens": 25, - "input_cost": 0.0092, - "output_cost": 0.0025, - "prompt_tokens": 150, - "spend": 0.0117 - }, - "claude-opus-5|all_components_anthropic_stream": { - "completion_tokens": 25, - "input_cost": 0.0092, - "output_cost": 0.0025, - "prompt_tokens": 150, - "spend": 0.0117 - }, - "claude-opus-5|basic": { - "completion_tokens": 40, - "input_cost": 0.006, - "output_cost": 0.004, - "prompt_tokens": 120, - "spend": 0.01 - }, - "claude-opus-5|cache_read": { - "completion_tokens": 30, - "input_cost": 0.00525, - "output_cost": 0.003, - "prompt_tokens": 150, - "spend": 0.00825 - }, - "claude-opus-5|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.0155, - "output_cost": 0.003, - "prompt_tokens": 150, - "spend": 0.0185 - }, - "claude-opus-5|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.013500000000000002, - "output_cost": 0.003, - "prompt_tokens": 150, - "spend": 0.0165 - }, - "claude-opus-5|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.007200000000000001, - "output_cost": 0.0048000000000000004, - "prompt_tokens": 120, - "spend": 0.012 - }, - "claude-opus-5|stream": { - "completion_tokens": 40, - "input_cost": 0.006, - "output_cost": 0.004, - "prompt_tokens": 120, - "spend": 0.01 - }, - "claude-opus-5|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.007200000000000001, - "output_cost": 0.0048000000000000004, - "prompt_tokens": 120, - "spend": 0.012 - }, - "claude-opus-5|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.004, - "output_cost": 0.0025, - "prompt_tokens": 80, - "spend": 0.006500000000000001 - }, - "claude-opus-5|tool_call": { - "completion_tokens": 40, - "input_cost": 0.006, - "output_cost": 0.004, - "prompt_tokens": 120, - "spend": 0.01 - }, - "claude-opus-5|web_search": { - "completion_tokens": 30, - "input_cost": 0.005, - "output_cost": 0.003, - "prompt_tokens": 100, - "spend": 0.068 - }, - "claude-sonnet-5|all_components_anthropic": { - "completion_tokens": 25, - "input_cost": 0.011040000000000001, - "output_cost": 0.0030000000000000005, - "prompt_tokens": 150, - "spend": 0.014040000000000002 - }, - "claude-sonnet-5|all_components_anthropic_stream": { - "completion_tokens": 25, - "input_cost": 0.011040000000000001, - "output_cost": 0.0030000000000000005, - "prompt_tokens": 150, - "spend": 0.014040000000000002 - }, - "claude-sonnet-5|basic": { - "completion_tokens": 40, - "input_cost": 0.007200000000000001, - "output_cost": 0.0048000000000000004, - "prompt_tokens": 120, - "spend": 0.012 - }, - "claude-sonnet-5|cache_read": { - "completion_tokens": 30, - "input_cost": 0.006300000000000001, - "output_cost": 0.0036000000000000003, - "prompt_tokens": 150, - "spend": 0.0099 - }, - "claude-sonnet-5|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.018600000000000002, - "output_cost": 0.0036000000000000003, - "prompt_tokens": 150, - "spend": 0.0222 - }, - "claude-sonnet-5|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.016200000000000003, - "output_cost": 0.0036000000000000003, - "prompt_tokens": 150, - "spend": 0.0198 - }, - "claude-sonnet-5|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.008400000000000001, - "output_cost": 0.005600000000000001, - "prompt_tokens": 120, - "spend": 0.014000000000000002 - }, - "claude-sonnet-5|stream": { - "completion_tokens": 40, - "input_cost": 0.007200000000000001, - "output_cost": 0.0048000000000000004, - "prompt_tokens": 120, - "spend": 0.012 - }, - "claude-sonnet-5|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.008400000000000001, - "output_cost": 0.005600000000000001, - "prompt_tokens": 120, - "spend": 0.014000000000000002 - }, - "claude-sonnet-5|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.0048000000000000004, - "output_cost": 0.0030000000000000005, - "prompt_tokens": 80, - "spend": 0.007800000000000001 - }, - "claude-sonnet-5|tool_call": { - "completion_tokens": 40, - "input_cost": 0.007200000000000001, - "output_cost": 0.0048000000000000004, - "prompt_tokens": 120, - "spend": 0.012 - }, - "claude-sonnet-5|web_search": { - "completion_tokens": 30, - "input_cost": 0.006000000000000001, - "output_cost": 0.0036000000000000003, - "prompt_tokens": 100, - "spend": 0.0696 - }, - "fireworks_ai/deepseek-v4p1-flash|all_components_fireworks": { - "completion_tokens": 25, - "input_cost": 0.011760000000000001, - "output_cost": 0.007000000000000001, - "prompt_tokens": 120, - "spend": 0.018760000000000002 - }, - "fireworks_ai/deepseek-v4p1-flash|audio": { - "completion_tokens": 45, - "input_cost": 0.030500000000000003, - "output_cost": 0.019950000000000002, - "prompt_tokens": 125, - "spend": 0.05045000000000001 - }, - "fireworks_ai/deepseek-v4p1-flash|basic": { - "completion_tokens": 40, - "input_cost": 0.016800000000000002, - "output_cost": 0.011200000000000002, - "prompt_tokens": 120, - "spend": 0.028000000000000004 - }, - "fireworks_ai/deepseek-v4p1-flash|cache_read": { - "completion_tokens": 30, - "input_cost": 0.014700000000000001, - "output_cost": 0.008400000000000001, - "prompt_tokens": 150, - "spend": 0.023100000000000002 - }, - "fireworks_ai/deepseek-v4p1-flash|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.0368, - "output_cost": 0.008400000000000001, - "prompt_tokens": 150, - "spend": 0.045200000000000004 - }, - "fireworks_ai/deepseek-v4p1-flash|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.0324, - "output_cost": 0.008400000000000001, - "prompt_tokens": 150, - "spend": 0.0408 - }, - "fireworks_ai/deepseek-v4p1-flash|reasoning": { - "completion_tokens": 100, - "input_cost": 0.014000000000000002, - "output_cost": 0.0469, - "prompt_tokens": 100, - "spend": 0.060899999999999996 - }, - "fireworks_ai/deepseek-v4p1-flash|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.014400000000000001, - "output_cost": 0.009600000000000001, - "prompt_tokens": 120, - "spend": 0.024 - }, - "fireworks_ai/deepseek-v4p1-flash|stream": { - "completion_tokens": 40, - "input_cost": 0.016800000000000002, - "output_cost": 0.011200000000000002, - "prompt_tokens": 120, - "spend": 0.028000000000000004 - }, - "fireworks_ai/deepseek-v4p1-flash|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.014400000000000001, - "output_cost": 0.009600000000000001, - "prompt_tokens": 120, - "spend": 0.024 - }, - "fireworks_ai/deepseek-v4p1-flash|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.011200000000000002, - "output_cost": 0.007000000000000001, - "prompt_tokens": 80, - "spend": 0.0182 - }, - "fireworks_ai/deepseek-v4p1-flash|tool_call": { - "completion_tokens": 40, - "input_cost": 0.016800000000000002, - "output_cost": 0.011200000000000002, - "prompt_tokens": 120, - "spend": 0.028000000000000004 - }, - "fireworks_ai/deepseek-v4p1-flash|web_search_single": { - "completion_tokens": 30, - "input_cost": 0.014000000000000002, - "output_cost": 0.008400000000000001, - "prompt_tokens": 100, - "spend": 0.04240000000000001 - }, - "fireworks_ai/kimi-k3|all_components_fireworks": { - "completion_tokens": 25, - "input_cost": 0.01008, - "output_cost": 0.006000000000000001, - "prompt_tokens": 120, - "spend": 0.01608 - }, - "fireworks_ai/kimi-k3|audio": { - "completion_tokens": 45, - "input_cost": 0.028500000000000004, - "output_cost": 0.01875, - "prompt_tokens": 125, - "spend": 0.04725 - }, - "fireworks_ai/kimi-k3|basic": { - "completion_tokens": 40, - "input_cost": 0.014400000000000001, - "output_cost": 0.009600000000000001, - "prompt_tokens": 120, - "spend": 0.024 - }, - "fireworks_ai/kimi-k3|cache_read": { - "completion_tokens": 30, - "input_cost": 0.012600000000000002, - "output_cost": 0.007200000000000001, - "prompt_tokens": 150, - "spend": 0.0198 - }, - "fireworks_ai/kimi-k3|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.035, - "output_cost": 0.007200000000000001, - "prompt_tokens": 150, - "spend": 0.0422 - }, - "fireworks_ai/kimi-k3|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.030600000000000002, - "output_cost": 0.007200000000000001, - "prompt_tokens": 150, - "spend": 0.0378 - }, - "fireworks_ai/kimi-k3|reasoning": { - "completion_tokens": 100, - "input_cost": 0.012000000000000002, - "output_cost": 0.0457, - "prompt_tokens": 100, - "spend": 0.0577 - }, - "fireworks_ai/kimi-k3|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.015600000000000003, - "output_cost": 0.010400000000000001, - "prompt_tokens": 120, - "spend": 0.026000000000000002 - }, - "fireworks_ai/kimi-k3|stream": { - "completion_tokens": 40, - "input_cost": 0.014400000000000001, - "output_cost": 0.009600000000000001, - "prompt_tokens": 120, - "spend": 0.024 - }, - "fireworks_ai/kimi-k3|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.015600000000000003, - "output_cost": 0.010400000000000001, - "prompt_tokens": 120, - "spend": 0.026000000000000002 - }, - "fireworks_ai/kimi-k3|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.009600000000000001, - "output_cost": 0.006000000000000001, - "prompt_tokens": 80, - "spend": 0.015600000000000003 - }, - "fireworks_ai/kimi-k3|tool_call": { - "completion_tokens": 40, - "input_cost": 0.014400000000000001, - "output_cost": 0.009600000000000001, - "prompt_tokens": 120, - "spend": 0.024 - }, - "fireworks_ai/kimi-k3|web_search_single": { - "completion_tokens": 30, - "input_cost": 0.012000000000000002, - "output_cost": 0.007200000000000001, - "prompt_tokens": 100, - "spend": 0.0392 - }, - "fireworks_ai/qwen3p8-max|all_components_fireworks": { - "completion_tokens": 25, - "input_cost": 0.010920000000000001, - "output_cost": 0.006500000000000001, - "prompt_tokens": 120, - "spend": 0.01742 - }, - "fireworks_ai/qwen3p8-max|audio": { - "completion_tokens": 45, - "input_cost": 0.029500000000000002, - "output_cost": 0.01935, - "prompt_tokens": 125, - "spend": 0.048850000000000005 - }, - "fireworks_ai/qwen3p8-max|basic": { - "completion_tokens": 40, - "input_cost": 0.015600000000000003, - "output_cost": 0.010400000000000001, - "prompt_tokens": 120, - "spend": 0.026000000000000002 - }, - "fireworks_ai/qwen3p8-max|cache_read": { - "completion_tokens": 30, - "input_cost": 0.01365, - "output_cost": 0.007800000000000001, - "prompt_tokens": 150, - "spend": 0.021450000000000004 - }, - "fireworks_ai/qwen3p8-max|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.0359, - "output_cost": 0.007800000000000001, - "prompt_tokens": 150, - "spend": 0.0437 - }, - "fireworks_ai/qwen3p8-max|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.0315, - "output_cost": 0.007800000000000001, - "prompt_tokens": 150, - "spend": 0.0393 - }, - "fireworks_ai/qwen3p8-max|reasoning": { - "completion_tokens": 100, - "input_cost": 0.013000000000000001, - "output_cost": 0.0463, - "prompt_tokens": 100, - "spend": 0.059300000000000005 - }, - "fireworks_ai/qwen3p8-max|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.016800000000000002, - "output_cost": 0.011200000000000002, - "prompt_tokens": 120, - "spend": 0.028000000000000004 - }, - "fireworks_ai/qwen3p8-max|stream": { - "completion_tokens": 40, - "input_cost": 0.015600000000000003, - "output_cost": 0.010400000000000001, - "prompt_tokens": 120, - "spend": 0.026000000000000002 - }, - "fireworks_ai/qwen3p8-max|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.016800000000000002, - "output_cost": 0.011200000000000002, - "prompt_tokens": 120, - "spend": 0.028000000000000004 - }, - "fireworks_ai/qwen3p8-max|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.010400000000000001, - "output_cost": 0.006500000000000001, - "prompt_tokens": 80, - "spend": 0.016900000000000002 - }, - "fireworks_ai/qwen3p8-max|tool_call": { - "completion_tokens": 40, - "input_cost": 0.015600000000000003, - "output_cost": 0.010400000000000001, - "prompt_tokens": 120, - "spend": 0.026000000000000002 - }, - "fireworks_ai/qwen3p8-max|web_search_single": { - "completion_tokens": 30, - "input_cost": 0.013000000000000001, - "output_cost": 0.007800000000000001, - "prompt_tokens": 100, - "spend": 0.0408 - }, - "gemini-3.1-pro-preview|all_components_gemini": { - "completion_tokens": 43, - "input_cost": 0.023940000000000003, - "output_cost": 0.030660000000000003, - "prompt_tokens": 125, - "spend": 0.05460000000000001 - }, - "gemini-3.1-pro-preview|audio": { - "completion_tokens": 45, - "input_cost": 0.052500000000000005, - "output_cost": 0.03465, - "prompt_tokens": 125, - "spend": 0.08715 - }, - "gemini-3.1-pro-preview|basic": { - "completion_tokens": 40, - "input_cost": 0.0252, - "output_cost": 0.016800000000000002, - "prompt_tokens": 120, - "spend": 0.042 - }, - "gemini-3.1-pro-preview|cache_read": { - "completion_tokens": 30, - "input_cost": 0.02205, - "output_cost": 0.0126, - "prompt_tokens": 150, - "spend": 0.03465 - }, - "gemini-3.1-pro-preview|prompt_blocked": { - "completion_tokens": 0, - "input_cost": 0.2, - "output_cost": 0.0, - "prompt_tokens": 1000, - "spend": 0.2 - }, - "gemini-3.1-pro-preview|reasoning": { - "completion_tokens": 100, - "input_cost": 0.021, - "output_cost": 0.0861, - "prompt_tokens": 100, - "spend": 0.1071 - }, - "gemini-3.1-pro-preview|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.024, - "output_cost": 0.016, - "prompt_tokens": 120, - "spend": 0.04 - }, - "gemini-3.1-pro-preview|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.0378, - "output_cost": 0.020999999999999998, - "prompt_tokens": 120, - "spend": 0.0588 - }, - "gemini-3.1-pro-preview|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.04284, - "output_cost": 0.02268, - "prompt_tokens": 120, - "spend": 0.06552 - }, - "gemini-3.1-pro-preview|stream": { - "completion_tokens": 40, - "input_cost": 0.0252, - "output_cost": 0.016800000000000002, - "prompt_tokens": 120, - "spend": 0.042 - }, - "gemini-3.1-pro-preview|stream_prompt_blocked": { - "completion_tokens": 0, - "input_cost": 0.2, - "output_cost": 0.0, - "prompt_tokens": 1000, - "spend": 0.2 - }, - "gemini-3.1-pro-preview|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.024, - "output_cost": 0.016, - "prompt_tokens": 120, - "spend": 0.04 - }, - "gemini-3.1-pro-preview|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.016800000000000002, - "output_cost": 0.0105, - "prompt_tokens": 80, - "spend": 0.027300000000000005 - }, - "gemini-3.1-pro-preview|tiered": { - "completion_tokens": 30, - "input_cost": 336.00168, - "output_cost": 0.0567, - "prompt_tokens": 200001, - "spend": 336.05838 - }, - "gemini-3.1-pro-preview|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0252, - "output_cost": 0.016800000000000002, - "prompt_tokens": 120, - "spend": 0.042 - }, - "gemini-3.1-pro-preview|web_search": { - "completion_tokens": 30, - "input_cost": 0.021, - "output_cost": 0.0126, - "prompt_tokens": 100, - "spend": 0.0936 - }, - "gemini-3.8-flash|all_components_gemini": { - "completion_tokens": 43, - "input_cost": 0.022799999999999997, - "output_cost": 0.0292, - "prompt_tokens": 125, - "spend": 0.052 - }, - "gemini-3.8-flash|audio": { - "completion_tokens": 45, - "input_cost": 0.05, - "output_cost": 0.033, - "prompt_tokens": 125, - "spend": 0.083 - }, - "gemini-3.8-flash|basic": { - "completion_tokens": 40, - "input_cost": 0.024, - "output_cost": 0.016, - "prompt_tokens": 120, - "spend": 0.04 - }, - "gemini-3.8-flash|cache_read": { - "completion_tokens": 30, - "input_cost": 0.021, - "output_cost": 0.012, - "prompt_tokens": 150, - "spend": 0.033 - }, - "gemini-3.8-flash|prompt_blocked": { - "completion_tokens": 0, - "input_cost": 0.21000000000000002, - "output_cost": 0.0, - "prompt_tokens": 1000, - "spend": 0.21000000000000002 - }, - "gemini-3.8-flash|reasoning": { - "completion_tokens": 100, - "input_cost": 0.02, - "output_cost": 0.082, - "prompt_tokens": 100, - "spend": 0.10200000000000001 - }, - "gemini-3.8-flash|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0252, - "output_cost": 0.016800000000000002, - "prompt_tokens": 120, - "spend": 0.042 - }, - "gemini-3.8-flash|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.036, - "output_cost": 0.02, - "prompt_tokens": 120, - "spend": 0.055999999999999994 - }, - "gemini-3.8-flash|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.0408, - "output_cost": 0.0216, - "prompt_tokens": 120, - "spend": 0.062400000000000004 - }, - "gemini-3.8-flash|stream": { - "completion_tokens": 40, - "input_cost": 0.024, - "output_cost": 0.016, - "prompt_tokens": 120, - "spend": 0.04 - }, - "gemini-3.8-flash|stream_prompt_blocked": { - "completion_tokens": 0, - "input_cost": 0.21000000000000002, - "output_cost": 0.0, - "prompt_tokens": 1000, - "spend": 0.21000000000000002 - }, - "gemini-3.8-flash|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0252, - "output_cost": 0.016800000000000002, - "prompt_tokens": 120, - "spend": 0.042 - }, - "gemini-3.8-flash|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.016, - "output_cost": 0.01, - "prompt_tokens": 80, - "spend": 0.026000000000000002 - }, - "gemini-3.8-flash|tiered": { - "completion_tokens": 30, - "input_cost": 320.0016, - "output_cost": 0.054, - "prompt_tokens": 200001, - "spend": 320.05559999999997 - }, - "gemini-3.8-flash|tool_call": { - "completion_tokens": 40, - "input_cost": 0.024, - "output_cost": 0.016, - "prompt_tokens": 120, - "spend": 0.04 - }, - "gemini-3.8-flash|web_search": { - "completion_tokens": 30, - "input_cost": 0.02, - "output_cost": 0.012, - "prompt_tokens": 100, - "spend": 0.092 - }, - "gemini/gemini-3.1-pro-preview|all_components_gemini": { - "completion_tokens": 43, - "input_cost": 0.010260000000000002, - "output_cost": 0.01314, - "prompt_tokens": 125, - "spend": 0.023400000000000004 - }, - "gemini/gemini-3.1-pro-preview|audio": { - "completion_tokens": 45, - "input_cost": 0.0225, - "output_cost": 0.014849999999999999, - "prompt_tokens": 125, - "spend": 0.037349999999999994 - }, - "gemini/gemini-3.1-pro-preview|basic": { - "completion_tokens": 40, - "input_cost": 0.0108, - "output_cost": 0.007200000000000001, - "prompt_tokens": 120, - "spend": 0.018000000000000002 - }, - "gemini/gemini-3.1-pro-preview|cache_read": { - "completion_tokens": 30, - "input_cost": 0.009450000000000002, - "output_cost": 0.0054, - "prompt_tokens": 150, - "spend": 0.014850000000000002 - }, - "gemini/gemini-3.1-pro-preview|prompt_blocked": { - "completion_tokens": 0, - "input_cost": 0.08, - "output_cost": 0.0, - "prompt_tokens": 1000, - "spend": 0.08 - }, - "gemini/gemini-3.1-pro-preview|reasoning": { - "completion_tokens": 100, - "input_cost": 0.009000000000000001, - "output_cost": 0.0369, - "prompt_tokens": 100, - "spend": 0.0459 - }, - "gemini/gemini-3.1-pro-preview|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.009600000000000001, - "output_cost": 0.0064, - "prompt_tokens": 120, - "spend": 0.016 - }, - "gemini/gemini-3.1-pro-preview|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.0162, - "output_cost": 0.009000000000000001, - "prompt_tokens": 120, - "spend": 0.0252 - }, - "gemini/gemini-3.1-pro-preview|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.01836, - "output_cost": 0.00972, - "prompt_tokens": 120, - "spend": 0.02808 - }, - "gemini/gemini-3.1-pro-preview|stream": { - "completion_tokens": 40, - "input_cost": 0.0108, - "output_cost": 0.007200000000000001, - "prompt_tokens": 120, - "spend": 0.018000000000000002 - }, - "gemini/gemini-3.1-pro-preview|stream_prompt_blocked": { - "completion_tokens": 0, - "input_cost": 0.08, - "output_cost": 0.0, - "prompt_tokens": 1000, - "spend": 0.08 - }, - "gemini/gemini-3.1-pro-preview|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.009600000000000001, - "output_cost": 0.0064, - "prompt_tokens": 120, - "spend": 0.016 - }, - "gemini/gemini-3.1-pro-preview|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.007200000000000001, - "output_cost": 0.0045000000000000005, - "prompt_tokens": 80, - "spend": 0.011700000000000002 - }, - "gemini/gemini-3.1-pro-preview|tiered": { - "completion_tokens": 30, - "input_cost": 144.00072, - "output_cost": 0.024300000000000002, - "prompt_tokens": 200001, - "spend": 144.02502 - }, - "gemini/gemini-3.1-pro-preview|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0108, - "output_cost": 0.007200000000000001, - "prompt_tokens": 120, - "spend": 0.018000000000000002 - }, - "gemini/gemini-3.1-pro-preview|web_search": { - "completion_tokens": 30, - "input_cost": 0.009000000000000001, - "output_cost": 0.0054, - "prompt_tokens": 100, - "spend": 0.0744 - }, - "gemini/gemini-3.8-flash|all_components_gemini": { - "completion_tokens": 43, - "input_cost": 0.00912, - "output_cost": 0.01168, - "prompt_tokens": 125, - "spend": 0.0208 - }, - "gemini/gemini-3.8-flash|audio": { - "completion_tokens": 45, - "input_cost": 0.02, - "output_cost": 0.0132, - "prompt_tokens": 125, - "spend": 0.0332 - }, - "gemini/gemini-3.8-flash|basic": { - "completion_tokens": 40, - "input_cost": 0.009600000000000001, - "output_cost": 0.0064, - "prompt_tokens": 120, - "spend": 0.016 - }, - "gemini/gemini-3.8-flash|cache_read": { - "completion_tokens": 30, - "input_cost": 0.0084, - "output_cost": 0.0048000000000000004, - "prompt_tokens": 150, - "spend": 0.0132 - }, - "gemini/gemini-3.8-flash|prompt_blocked": { - "completion_tokens": 0, - "input_cost": 0.09000000000000001, - "output_cost": 0.0, - "prompt_tokens": 1000, - "spend": 0.09000000000000001 - }, - "gemini/gemini-3.8-flash|reasoning": { - "completion_tokens": 100, - "input_cost": 0.008, - "output_cost": 0.0328, - "prompt_tokens": 100, - "spend": 0.0408 - }, - "gemini/gemini-3.8-flash|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0108, - "output_cost": 0.007200000000000001, - "prompt_tokens": 120, - "spend": 0.018000000000000002 - }, - "gemini/gemini-3.8-flash|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.0144, - "output_cost": 0.008, - "prompt_tokens": 120, - "spend": 0.0224 - }, - "gemini/gemini-3.8-flash|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.01632, - "output_cost": 0.00864, - "prompt_tokens": 120, - "spend": 0.024960000000000003 - }, - "gemini/gemini-3.8-flash|stream": { - "completion_tokens": 40, - "input_cost": 0.009600000000000001, - "output_cost": 0.0064, - "prompt_tokens": 120, - "spend": 0.016 - }, - "gemini/gemini-3.8-flash|stream_prompt_blocked": { - "completion_tokens": 0, - "input_cost": 0.09000000000000001, - "output_cost": 0.0, - "prompt_tokens": 1000, - "spend": 0.09000000000000001 - }, - "gemini/gemini-3.8-flash|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0108, - "output_cost": 0.007200000000000001, - "prompt_tokens": 120, - "spend": 0.018000000000000002 - }, - "gemini/gemini-3.8-flash|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.0064, - "output_cost": 0.004, - "prompt_tokens": 80, - "spend": 0.0104 - }, - "gemini/gemini-3.8-flash|tiered": { - "completion_tokens": 30, - "input_cost": 128.00064, - "output_cost": 0.0216, - "prompt_tokens": 200001, - "spend": 128.02224 - }, - "gemini/gemini-3.8-flash|tool_call": { - "completion_tokens": 40, - "input_cost": 0.009600000000000001, - "output_cost": 0.0064, - "prompt_tokens": 120, - "spend": 0.016 - }, - "gemini/gemini-3.8-flash|web_search": { - "completion_tokens": 30, - "input_cost": 0.008, - "output_cost": 0.0048000000000000004, - "prompt_tokens": 100, - "spend": 0.0728 - }, - "gpt-5.3-codex|all_components_responses": { - "completion_tokens": 40, - "input_cost": 0.00252, - "output_cost": 0.0037500000000000007, - "prompt_tokens": 120, - "spend": 0.006270000000000001 - }, - "gpt-5.3-codex|basic": { - "completion_tokens": 40, - "input_cost": 0.0036000000000000003, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 120, - "spend": 0.006 - }, - "gpt-5.3-codex|cache_read": { - "completion_tokens": 30, - "input_cost": 0.0031500000000000005, - "output_cost": 0.0018000000000000002, - "prompt_tokens": 150, - "spend": 0.00495 - }, - "gpt-5.3-codex|reasoning": { - "completion_tokens": 100, - "input_cost": 0.0030000000000000005, - "output_cost": 0.0123, - "prompt_tokens": 100, - "spend": 0.015300000000000001 - }, - "gpt-5.3-codex|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0024000000000000002, - "output_cost": 0.0016, - "prompt_tokens": 120, - "spend": 0.004 - }, - "gpt-5.3-codex|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.0054, - "output_cost": 0.003, - "prompt_tokens": 120, - "spend": 0.008400000000000001 - }, - "gpt-5.3-codex|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.00612, - "output_cost": 0.00324, - "prompt_tokens": 120, - "spend": 0.00936 - }, - "gpt-5.3-codex|stream": { - "completion_tokens": 40, - "input_cost": 0.0036000000000000003, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 120, - "spend": 0.006 - }, - "gpt-5.3-codex|stream_incomplete": { - "completion_tokens": 40, - "input_cost": 0.0036000000000000003, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 120, - "spend": 0.006 - }, - "gpt-5.3-codex|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0024000000000000002, - "output_cost": 0.0016, - "prompt_tokens": 120, - "spend": 0.004 - }, - "gpt-5.3-codex|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.0024000000000000002, - "output_cost": 0.0015000000000000002, - "prompt_tokens": 80, - "spend": 0.0039000000000000007 - }, - "gpt-5.3-codex|stream_unvalidated": { - "completion_tokens": 40, - "input_cost": 0.0036000000000000003, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 120, - "spend": 0.006 - }, - "gpt-5.3-codex|tiered": { - "completion_tokens": 30, - "input_cost": 48.000240000000005, - "output_cost": 0.0081, - "prompt_tokens": 200001, - "spend": 48.008340000000004 - }, - "gpt-5.3-codex|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0036000000000000003, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 120, - "spend": 0.006 - }, - "gpt-5.3-codex|web_search": { - "completion_tokens": 30, - "input_cost": 0.0030000000000000005, - "output_cost": 0.0018000000000000002, - "prompt_tokens": 100, - "spend": 0.0648 - }, - "gpt-5.4-mini|all_components_chat": { - "completion_tokens": 43, - "input_cost": 0.00856, - "output_cost": 0.00584, - "prompt_tokens": 155, - "spend": 0.0144 - }, - "gpt-5.4-mini|audio": { - "completion_tokens": 45, - "input_cost": 0.01, - "output_cost": 0.0066, - "prompt_tokens": 125, - "spend": 0.0166 - }, - "gpt-5.4-mini|basic": { - "completion_tokens": 40, - "input_cost": 0.0048000000000000004, - "output_cost": 0.0032, - "prompt_tokens": 120, - "spend": 0.008 - }, - "gpt-5.4-mini|cache_read": { - "completion_tokens": 30, - "input_cost": 0.0042, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 150, - "spend": 0.0066 - }, - "gpt-5.4-mini|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.012400000000000001, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 150, - "spend": 0.0148 - }, - "gpt-5.4-mini|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.0108, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 150, - "spend": 0.0132 - }, - "gpt-5.4-mini|reasoning": { - "completion_tokens": 100, - "input_cost": 0.004, - "output_cost": 0.0164, - "prompt_tokens": 100, - "spend": 0.0204 - }, - "gpt-5.4-mini|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0012000000000000001, - "output_cost": 0.0008, - "prompt_tokens": 120, - "spend": 0.002 - }, - "gpt-5.4-mini|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.0072, - "output_cost": 0.004, - "prompt_tokens": 120, - "spend": 0.0112 - }, - "gpt-5.4-mini|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.00816, - "output_cost": 0.00432, - "prompt_tokens": 120, - "spend": 0.012480000000000002 - }, - "gpt-5.4-mini|stream": { - "completion_tokens": 40, - "input_cost": 0.0048000000000000004, - "output_cost": 0.0032, - "prompt_tokens": 120, - "spend": 0.008 - }, - "gpt-5.4-mini|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0012000000000000001, - "output_cost": 0.0008, - "prompt_tokens": 120, - "spend": 0.002 - }, - "gpt-5.4-mini|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.0032, - "output_cost": 0.002, - "prompt_tokens": 80, - "spend": 0.0052 - }, - "gpt-5.4-mini|tiered": { - "completion_tokens": 30, - "input_cost": 64.00032, - "output_cost": 0.0108, - "prompt_tokens": 200001, - "spend": 64.01112 - }, - "gpt-5.4-mini|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0048000000000000004, - "output_cost": 0.0032, - "prompt_tokens": 120, - "spend": 0.008 - }, - "gpt-5.4-mini|web_search_single": { - "completion_tokens": 30, - "input_cost": 0.004, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 100, - "spend": 0.0264 - }, - "gpt-5.5-pro|all_components_responses": { - "completion_tokens": 40, - "input_cost": 0.00168, - "output_cost": 0.0025, - "prompt_tokens": 120, - "spend": 0.00418 - }, - "gpt-5.5-pro|basic": { - "completion_tokens": 40, - "input_cost": 0.0024000000000000002, - "output_cost": 0.0016, - "prompt_tokens": 120, - "spend": 0.004 - }, - "gpt-5.5-pro|cache_read": { - "completion_tokens": 30, - "input_cost": 0.0021, - "output_cost": 0.0012000000000000001, - "prompt_tokens": 150, - "spend": 0.0033 - }, - "gpt-5.5-pro|reasoning": { - "completion_tokens": 100, - "input_cost": 0.002, - "output_cost": 0.0082, - "prompt_tokens": 100, - "spend": 0.0102 - }, - "gpt-5.5-pro|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0036000000000000003, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 120, - "spend": 0.006 - }, - "gpt-5.5-pro|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.0036, - "output_cost": 0.002, - "prompt_tokens": 120, - "spend": 0.0056 - }, - "gpt-5.5-pro|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.00408, - "output_cost": 0.00216, - "prompt_tokens": 120, - "spend": 0.006240000000000001 - }, - "gpt-5.5-pro|stream": { - "completion_tokens": 40, - "input_cost": 0.0024000000000000002, - "output_cost": 0.0016, - "prompt_tokens": 120, - "spend": 0.004 - }, - "gpt-5.5-pro|stream_incomplete": { - "completion_tokens": 40, - "input_cost": 0.0024000000000000002, - "output_cost": 0.0016, - "prompt_tokens": 120, - "spend": 0.004 - }, - "gpt-5.5-pro|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0036000000000000003, - "output_cost": 0.0024000000000000002, - "prompt_tokens": 120, - "spend": 0.006 - }, - "gpt-5.5-pro|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.0016, - "output_cost": 0.001, - "prompt_tokens": 80, - "spend": 0.0026 - }, - "gpt-5.5-pro|stream_unvalidated": { - "completion_tokens": 40, - "input_cost": 0.0024000000000000002, - "output_cost": 0.0016, - "prompt_tokens": 120, - "spend": 0.004 - }, - "gpt-5.5-pro|tiered": { - "completion_tokens": 30, - "input_cost": 32.00016, - "output_cost": 0.0054, - "prompt_tokens": 200001, - "spend": 32.00556 - }, - "gpt-5.5-pro|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0024000000000000002, - "output_cost": 0.0016, - "prompt_tokens": 120, - "spend": 0.004 - }, - "gpt-5.5-pro|web_search": { - "completion_tokens": 30, - "input_cost": 0.002, - "output_cost": 0.0012000000000000001, - "prompt_tokens": 100, - "spend": 0.06319999999999999 - }, - "gpt-5.6|all_components_chat": { - "completion_tokens": 43, - "input_cost": 0.00214, - "output_cost": 0.00146, - "prompt_tokens": 155, - "spend": 0.0036 - }, - "gpt-5.6|audio": { - "completion_tokens": 45, - "input_cost": 0.0025, - "output_cost": 0.00165, - "prompt_tokens": 125, - "spend": 0.00415 - }, - "gpt-5.6|basic": { - "completion_tokens": 40, - "input_cost": 0.0012000000000000001, - "output_cost": 0.0008, - "prompt_tokens": 120, - "spend": 0.002 - }, - "gpt-5.6|cache_read": { - "completion_tokens": 30, - "input_cost": 0.00105, - "output_cost": 0.0006000000000000001, - "prompt_tokens": 150, - "spend": 0.00165 - }, - "gpt-5.6|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.0031000000000000003, - "output_cost": 0.0006000000000000001, - "prompt_tokens": 150, - "spend": 0.0037 - }, - "gpt-5.6|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.0027, - "output_cost": 0.0006000000000000001, - "prompt_tokens": 150, - "spend": 0.0033 - }, - "gpt-5.6|reasoning": { - "completion_tokens": 100, - "input_cost": 0.001, - "output_cost": 0.0041, - "prompt_tokens": 100, - "spend": 0.0051 - }, - "gpt-5.6|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0048000000000000004, - "output_cost": 0.0032, - "prompt_tokens": 120, - "spend": 0.008 - }, - "gpt-5.6|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.0018, - "output_cost": 0.001, - "prompt_tokens": 120, - "spend": 0.0028 - }, - "gpt-5.6|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.00204, - "output_cost": 0.00108, - "prompt_tokens": 120, - "spend": 0.0031200000000000004 - }, - "gpt-5.6|stream": { - "completion_tokens": 40, - "input_cost": 0.0012000000000000001, - "output_cost": 0.0008, - "prompt_tokens": 120, - "spend": 0.002 - }, - "gpt-5.6|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0048000000000000004, - "output_cost": 0.0032, - "prompt_tokens": 120, - "spend": 0.008 - }, - "gpt-5.6|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.0008, - "output_cost": 0.0005, - "prompt_tokens": 80, - "spend": 0.0013 - }, - "gpt-5.6|tiered": { - "completion_tokens": 30, - "input_cost": 16.00008, - "output_cost": 0.0027, - "prompt_tokens": 200001, - "spend": 16.00278 - }, - "gpt-5.6|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0012000000000000001, - "output_cost": 0.0008, - "prompt_tokens": 120, - "spend": 0.002 - }, - "gpt-5.6|web_search_single": { - "completion_tokens": 30, - "input_cost": 0.001, - "output_cost": 0.0006000000000000001, - "prompt_tokens": 100, - "spend": 0.0216 - }, - "meta.llama4-maverick-17b-instruct-v1:0|all_components_anthropic": { - "completion_tokens": 25, - "input_cost": 0.0285, - "output_cost": 0.0095, - "prompt_tokens": 150, - "spend": 0.038 - }, - "meta.llama4-maverick-17b-instruct-v1:0|basic": { - "completion_tokens": 40, - "input_cost": 0.0228, - "output_cost": 0.015200000000000002, - "prompt_tokens": 120, - "spend": 0.038000000000000006 - }, - "meta.llama4-maverick-17b-instruct-v1:0|stream": { - "completion_tokens": 40, - "input_cost": 0.0228, - "output_cost": 0.015200000000000002, - "prompt_tokens": 120, - "spend": 0.038000000000000006 - }, - "meta.llama4-maverick-17b-instruct-v1:0|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.015200000000000002, - "output_cost": 0.0095, - "prompt_tokens": 80, - "spend": 0.0247 - }, - "meta.llama4-maverick-17b-instruct-v1:0|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0228, - "output_cost": 0.015200000000000002, - "prompt_tokens": 120, - "spend": 0.038000000000000006 - }, - "together_ai/moonshotai/Kimi-K3|all_components_chat": { - "completion_tokens": 43, - "input_cost": 0.0214, - "output_cost": 0.0146, - "prompt_tokens": 155, - "spend": 0.036 - }, - "together_ai/moonshotai/Kimi-K3|audio": { - "completion_tokens": 45, - "input_cost": 0.025, - "output_cost": 0.0165, - "prompt_tokens": 125, - "spend": 0.0415 - }, - "together_ai/moonshotai/Kimi-K3|basic": { - "completion_tokens": 40, - "input_cost": 0.012, - "output_cost": 0.008, - "prompt_tokens": 120, - "spend": 0.02 - }, - "together_ai/moonshotai/Kimi-K3|cache_read": { - "completion_tokens": 30, - "input_cost": 0.0105, - "output_cost": 0.006, - "prompt_tokens": 150, - "spend": 0.0165 - }, - "together_ai/moonshotai/Kimi-K3|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.031, - "output_cost": 0.006, - "prompt_tokens": 150, - "spend": 0.037 - }, - "together_ai/moonshotai/Kimi-K3|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.027000000000000003, - "output_cost": 0.006, - "prompt_tokens": 150, - "spend": 0.033 - }, - "together_ai/moonshotai/Kimi-K3|reasoning": { - "completion_tokens": 100, - "input_cost": 0.01, - "output_cost": 0.041, - "prompt_tokens": 100, - "spend": 0.051000000000000004 - }, - "together_ai/moonshotai/Kimi-K3|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0132, - "output_cost": 0.0088, - "prompt_tokens": 120, - "spend": 0.022 - }, - "together_ai/moonshotai/Kimi-K3|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.018000000000000002, - "output_cost": 0.01, - "prompt_tokens": 120, - "spend": 0.028000000000000004 - }, - "together_ai/moonshotai/Kimi-K3|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.0204, - "output_cost": 0.0108, - "prompt_tokens": 120, - "spend": 0.031200000000000002 - }, - "together_ai/moonshotai/Kimi-K3|stream": { - "completion_tokens": 40, - "input_cost": 0.012, - "output_cost": 0.008, - "prompt_tokens": 120, - "spend": 0.02 - }, - "together_ai/moonshotai/Kimi-K3|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.0132, - "output_cost": 0.0088, - "prompt_tokens": 120, - "spend": 0.022 - }, - "together_ai/moonshotai/Kimi-K3|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.008, - "output_cost": 0.005, - "prompt_tokens": 80, - "spend": 0.013000000000000001 - }, - "together_ai/moonshotai/Kimi-K3|tiered": { - "completion_tokens": 30, - "input_cost": 160.0008, - "output_cost": 0.027000000000000003, - "prompt_tokens": 200001, - "spend": 160.02779999999998 - }, - "together_ai/moonshotai/Kimi-K3|tool_call": { - "completion_tokens": 40, - "input_cost": 0.012, - "output_cost": 0.008, - "prompt_tokens": 120, - "spend": 0.02 - }, - "together_ai/moonshotai/Kimi-K3|web_search_single": { - "completion_tokens": 30, - "input_cost": 0.01, - "output_cost": 0.006, - "prompt_tokens": 100, - "spend": 0.036000000000000004 - }, - "together_ai/zai-org/GLM-5.3|all_components_chat": { - "completion_tokens": 43, - "input_cost": 0.023540000000000002, - "output_cost": 0.01606, - "prompt_tokens": 155, - "spend": 0.0396 - }, - "together_ai/zai-org/GLM-5.3|audio": { - "completion_tokens": 45, - "input_cost": 0.027500000000000004, - "output_cost": 0.01815, - "prompt_tokens": 125, - "spend": 0.04565 - }, - "together_ai/zai-org/GLM-5.3|basic": { - "completion_tokens": 40, - "input_cost": 0.0132, - "output_cost": 0.0088, - "prompt_tokens": 120, - "spend": 0.022 - }, - "together_ai/zai-org/GLM-5.3|cache_read": { - "completion_tokens": 30, - "input_cost": 0.011550000000000001, - "output_cost": 0.0066, - "prompt_tokens": 150, - "spend": 0.01815 - }, - "together_ai/zai-org/GLM-5.3|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.034100000000000005, - "output_cost": 0.0066, - "prompt_tokens": 150, - "spend": 0.04070000000000001 - }, - "together_ai/zai-org/GLM-5.3|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.029699999999999997, - "output_cost": 0.0066, - "prompt_tokens": 150, - "spend": 0.0363 - }, - "together_ai/zai-org/GLM-5.3|reasoning": { - "completion_tokens": 100, - "input_cost": 0.011000000000000001, - "output_cost": 0.0451, - "prompt_tokens": 100, - "spend": 0.056100000000000004 - }, - "together_ai/zai-org/GLM-5.3|response_model_override": { - "completion_tokens": 40, - "input_cost": 0.012, - "output_cost": 0.008, - "prompt_tokens": 120, - "spend": 0.02 - }, - "together_ai/zai-org/GLM-5.3|service_tier_flex": { - "completion_tokens": 40, - "input_cost": 0.019799999999999998, - "output_cost": 0.011000000000000001, - "prompt_tokens": 120, - "spend": 0.0308 - }, - "together_ai/zai-org/GLM-5.3|service_tier_priority": { - "completion_tokens": 40, - "input_cost": 0.022439999999999998, - "output_cost": 0.01188, - "prompt_tokens": 120, - "spend": 0.034319999999999996 - }, - "together_ai/zai-org/GLM-5.3|stream": { - "completion_tokens": 40, - "input_cost": 0.0132, - "output_cost": 0.0088, - "prompt_tokens": 120, - "spend": 0.022 - }, - "together_ai/zai-org/GLM-5.3|stream_response_model_override": { - "completion_tokens": 40, - "input_cost": 0.012, - "output_cost": 0.008, - "prompt_tokens": 120, - "spend": 0.02 - }, - "together_ai/zai-org/GLM-5.3|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.0088, - "output_cost": 0.0055000000000000005, - "prompt_tokens": 80, - "spend": 0.0143 - }, - "together_ai/zai-org/GLM-5.3|tiered": { - "completion_tokens": 30, - "input_cost": 176.00088, - "output_cost": 0.0297, - "prompt_tokens": 200001, - "spend": 176.03058 - }, - "together_ai/zai-org/GLM-5.3|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0132, - "output_cost": 0.0088, - "prompt_tokens": 120, - "spend": 0.022 - }, - "together_ai/zai-org/GLM-5.3|web_search_single": { - "completion_tokens": 30, - "input_cost": 0.011000000000000001, - "output_cost": 0.0066, - "prompt_tokens": 100, - "spend": 0.0376 - }, - "us.anthropic.claude-opus-5-v1:0|all_components_anthropic": { - "completion_tokens": 25, - "input_cost": 0.033120000000000004, - "output_cost": 0.009000000000000001, - "prompt_tokens": 150, - "spend": 0.042120000000000005 - }, - "us.anthropic.claude-opus-5-v1:0|basic": { - "completion_tokens": 40, - "input_cost": 0.0216, - "output_cost": 0.014400000000000001, - "prompt_tokens": 120, - "spend": 0.036000000000000004 - }, - "us.anthropic.claude-opus-5-v1:0|cache_read": { - "completion_tokens": 30, - "input_cost": 0.018900000000000004, - "output_cost": 0.0108, - "prompt_tokens": 150, - "spend": 0.029700000000000004 - }, - "us.anthropic.claude-opus-5-v1:0|cache_write_1h": { - "completion_tokens": 30, - "input_cost": 0.0558, - "output_cost": 0.0108, - "prompt_tokens": 150, - "spend": 0.0666 - }, - "us.anthropic.claude-opus-5-v1:0|cache_write_5m": { - "completion_tokens": 30, - "input_cost": 0.048600000000000004, - "output_cost": 0.0108, - "prompt_tokens": 150, - "spend": 0.05940000000000001 - }, - "us.anthropic.claude-opus-5-v1:0|stream": { - "completion_tokens": 40, - "input_cost": 0.0216, - "output_cost": 0.014400000000000001, - "prompt_tokens": 120, - "spend": 0.036000000000000004 - }, - "us.anthropic.claude-opus-5-v1:0|stream_tool_call": { - "completion_tokens": 25, - "input_cost": 0.014400000000000001, - "output_cost": 0.009000000000000001, - "prompt_tokens": 80, - "spend": 0.023400000000000004 - }, - "us.anthropic.claude-opus-5-v1:0|tool_call": { - "completion_tokens": 40, - "input_cost": 0.0216, - "output_cost": 0.014400000000000001, - "prompt_tokens": 120, - "spend": 0.036000000000000004 - } -} diff --git a/tests/e2e/cost_calculation/generate_expected.py b/tests/e2e/cost_calculation/generate_expected.py deleted file mode 100644 index 64abdb14c99..00000000000 --- a/tests/e2e/cost_calculation/generate_expected.py +++ /dev/null @@ -1,211 +0,0 @@ -"""Golden generator for the cost suite. Run: - - uv run python tests/e2e/cost_calculation/generate_expected.py - -Loads the derived matrix (models x applicable cases), computes the golden for -each exact-spend cell from the rate arithmetic, and writes ``expected.json`` -with sorted keys. Default behaviour adds missing cells and drops stale cells -but never overwrites an existing cell's values (a reviewed golden is -authoritative); ``--rewrite`` recomputes everything. Prints added/removed/kept -counts. -""" - -from __future__ import annotations - -import json -import sys -from collections.abc import Mapping -from dataclasses import dataclass -from types import MappingProxyType -from typing import Final - -from cost_matrix import ( - EXPECTED_PATH, - FRONTIER_MODELS, - TIER_THRESHOLD_TOKENS, - Case, - CostMapEntry, - ExpectedCell, - FrontierModel, - cases_for, - expected_key, -) -from pydantic import TypeAdapter - - -def _first_present(*rates: float | None) -> float | None: - return next((rate for rate in rates if rate is not None), None) - - -@dataclass(frozen=True, slots=True) -class ExpectedCost: - """The expected bill split the way the spend row's cost_breakdown reports - it: the gross input component (cache reads/writes folded in), the output - component, and the tool-usage component.""" - - input_cost: float - output_cost: float - tool_cost: float - - @property - def total(self) -> float: - return self.input_cost + self.output_cost + self.tool_cost - - -def expected_breakdown(model: FrontierModel, case: Case) -> ExpectedCost: - """Literal arithmetic on the test-map rates over the scripted token counts. - - Input = fresh*in + read*read + 5m*create + 1h*create_1h + audio_in*audio_in; - output = text*out + reasoning*reasoning + audio_out*audio_out; plus the - billed web-search calls at the medium search-context rate. Every billed - token is a token the provider charged for: a component whose entry has no - dedicated rate bills at the ordinary input or output rate, and a present - rate (including an explicit 0.0) is authoritative. When the total prompt - tokens exceed the threshold, input/output rates come from the - ``_above_200k_tokens`` variants; a service tier takes its ``_priority`` or - ``_flex`` variant when the entry carries one, and otherwise bills at the - base rate. - """ - rates: Final[CostMapEntry] = model.override_rates if case.response_model_override else model.rates - u: Final = case.usage - prompt_tokens: Final = ( - u.fresh_input_tokens + u.cache_read_tokens + u.cache_write_5m_tokens - + u.cache_write_1h_tokens + u.audio_input_tokens - ) - tiered: Final = prompt_tokens > TIER_THRESHOLD_TOKENS - in_rate: Final = ( - _first_present( - rates.input_cost_per_token_above_200k_tokens if tiered else None, - rates.input_cost_per_token_priority if case.service_tier == "priority" else None, - rates.input_cost_per_token_flex if case.service_tier == "flex" else None, - rates.input_cost_per_token, - ) - or 0.0 - ) - out_rate: Final = ( - _first_present( - rates.output_cost_per_token_above_200k_tokens if tiered else None, - rates.output_cost_per_token_priority if case.service_tier == "priority" else None, - rates.output_cost_per_token_flex if case.service_tier == "flex" else None, - rates.output_cost_per_token, - ) - or 0.0 - ) - read_rate: Final = _first_present(rates.cache_read_input_token_cost, in_rate) or 0.0 - write_rate: Final = _first_present(rates.cache_creation_input_token_cost, in_rate) or 0.0 - write_1h_rate: Final = ( - _first_present(rates.cache_creation_input_token_cost_above_1hr, write_rate) or 0.0 - ) - audio_in_rate: Final = _first_present(rates.input_cost_per_audio_token, in_rate) or 0.0 - reasoning_rate: Final = _first_present(rates.output_cost_per_reasoning_token, out_rate) or 0.0 - audio_out_rate: Final = _first_present(rates.output_cost_per_audio_token, out_rate) or 0.0 - input_cost: Final = ( - u.fresh_input_tokens * in_rate - + u.cache_read_tokens * read_rate - + u.cache_write_5m_tokens * write_rate - + u.cache_write_1h_tokens * write_1h_rate - + u.audio_input_tokens * audio_in_rate - ) - output_cost: Final = ( - u.output_tokens * out_rate - + u.reasoning_tokens * reasoning_rate - + u.audio_output_tokens * audio_out_rate - ) - search: Final = rates.search_context_cost_per_query - medium_rate: Final = ( - search.search_context_size_medium if search is not None else None - ) - if u.web_search_calls and medium_rate is None: - raise ValueError( - f"{model.map_key}: case {case.name} bills {u.web_search_calls} web-search " - "calls but the entry has no search_context_cost_per_query medium rate" - ) - tool_cost: Final = u.web_search_calls * (medium_rate if medium_rate is not None else 0.0) - return ExpectedCost(input_cost=input_cost, output_cost=output_cost, tool_cost=tool_cost) - - -def expected_token_columns(model: FrontierModel, case: Case) -> tuple[int, int]: - """(prompt_tokens, completion_tokens) the spend row should carry, per the - wire's normalization: Anthropic folds cache read/write into prompt_tokens, - everyone else reports the totals the wire emitted.""" - u: Final = case.usage - if model.wire in ("anthropic_messages", "bedrock_converse"): - return ( - u.fresh_input_tokens + u.cache_read_tokens + u.cache_write_5m_tokens + u.cache_write_1h_tokens, - u.output_tokens, - ) - if model.wire in ("gemini_generate", "vertex_generate"): - return ( - u.fresh_input_tokens + u.cache_read_tokens + u.audio_input_tokens, - u.output_tokens + u.reasoning_tokens + u.audio_output_tokens, - ) - if model.wire == "openai_responses": - return ( - u.fresh_input_tokens + u.cache_read_tokens, - u.output_tokens + u.reasoning_tokens, - ) - return ( - u.fresh_input_tokens - + u.cache_read_tokens - + u.cache_write_5m_tokens - + u.cache_write_1h_tokens - + u.audio_input_tokens, - u.output_tokens + u.reasoning_tokens + u.audio_output_tokens, - ) - - -def _cell(model: FrontierModel, case: Case) -> ExpectedCell: - breakdown: Final = expected_breakdown(model, case) - prompt_tokens, completion_tokens = expected_token_columns(model, case) - return ExpectedCell( - spend=breakdown.total, - input_cost=breakdown.input_cost, - output_cost=breakdown.output_cost, - prompt_tokens=prompt_tokens, - completion_tokens=completion_tokens, - ) - - -def _proposed() -> Mapping[str, ExpectedCell]: - return MappingProxyType( - { - expected_key(model, case): _cell(model, case) - for model in FRONTIER_MODELS - for case in cases_for(model) - if case.exact_spend - } - ) - - -def main() -> None: - rewrite: Final = "--rewrite" in sys.argv[1:] - proposed: Final = _proposed() - proposed_values: Final = {key: cell.model_dump() for key, cell in proposed.items()} - existing: Final[Mapping[str, ExpectedCell]] = ( - TypeAdapter(dict[str, ExpectedCell]).validate_python( - json.loads(EXPECTED_PATH.read_text()) - ) - if EXPECTED_PATH.exists() - else {} - ) - merged: Final = { - key: ( - proposed_values[key] - if rewrite or key not in existing - else existing[key].model_dump() - ) - for key in sorted(proposed_values) - } - added: Final = sum(1 for key in proposed_values if key not in existing) - removed: Final = sum(1 for key in existing if key not in proposed_values) - kept: Final = sum(1 for key in proposed_values if key in existing and not rewrite) - rewritten: Final = sum(1 for key in proposed_values if key in existing and rewrite) - EXPECTED_PATH.write_text(json.dumps(merged, indent=2, sort_keys=True) + "\n") - print( # noqa: T201 # CLI summary is the tool output - f"expected.json: {added} added, {removed} removed, {kept} kept, " - f"{rewritten} rewritten ({len(merged)} cells)" - ) - - -if __name__ == "__main__": - main() diff --git a/tests/e2e/cost_calculation/test_token_pricing_e2e.py b/tests/e2e/cost_calculation/test_token_pricing_e2e.py index 346a55aa22d..03dab6be5e7 100644 --- a/tests/e2e/cost_calculation/test_token_pricing_e2e.py +++ b/tests/e2e/cost_calculation/test_token_pricing_e2e.py @@ -1,7 +1,8 @@ """Token-pricing e2e: every (map entry, case) cell derived from cost_map.json x cases.json runs a scripted-usage call through a deployment registered on the cost-map proxy, and the spend row plus response-cost header must equal the -reviewed golden in expected.json verbatim -- no rate arithmetic lives here. +reviewed golden in the case's ``expected`` cell verbatim -- no rate arithmetic +lives here. Nothing here touches a real provider or the bundled cost map: the proxy's upstream is the scripted-provider sidecar and its entire cost map is @@ -15,13 +16,11 @@ from typing import Final from conftest import CostCalcClient, cost_rows, register_scenario_deployment from cost_matrix import ( - EXPECTED, FRONTIER_MODELS, IMAGE_INPUT_DATA_URL, Case, FrontierModel, cases_for, - expected_key, matrix_data_errors, recount_cost, ) @@ -142,7 +141,7 @@ class TestTokenPricing: cost_rows.assert_total_is_sum_of_components(row) return - golden: Final = EXPECTED[expected_key(model, case)] + golden: Final = case.expected_for(model) if not case.stream: # Streamed responses commit headers before the bill is computed, so