mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
refactor(e2e): inline literal expected costs into cases.json
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
ef1f306a7d
commit
aac1456e07
7 changed files with 477 additions and 2437 deletions
|
|
@ -21,7 +21,7 @@ Each subdirectory under `tests/e2e/` is one suite, scoped to an endpoint family
|
|||
- `load/` - performance-category tests, kept OUT of the main suite: throughput/load SLO tests are a different testing category from functional e2e (variance-driven, historically flaky) and live outside this suite until re-implemented as their own pipeline (LIT-5163); do not add a live load test that runs in the default collection. What lives here: the weekly session-anomaly test (`test_weekly_session_anomaly_e2e.py`, Claude Code-shaped multi-turn sessions against real providers with ceilings on error rate, cache read/write, turn time, and spend; marked `weekly` and deselected unless `E2E_WEEKLY_ANOMALY` is set, driven by `.github/workflows/weekly_load_anomaly.yml`), the Redis chaos test (`test_redis_chaos_e2e.py`, locust load against mock deployments split round robin over `/chat/completions` and `/v1/messages`, one endpoint per simulated user, with `CLIENT PAUSE ALL` on the proxy's Redis mid-run to simulate it being down outright, asserting zero failed requests on every endpoint, budgeting RSS and CPU-per-request as ratios against the same run's healthy phase, and holding p50/p90/p99 latency and log-bytes-per-request to flat ceilings (a ratio cannot bound those two: an open breaker skips Redis instead of waiting on it, so the chaos phase can measure cheaper than baseline while still being far slower than a user should see); needs a proxy booted from `gateway/redis_chaos_ci_config.yml` on the same host with `E2E_PROXY_PID` and `E2E_PROXY_LOG` set, marked `redis_chaos`, deselected unless `E2E_REDIS_CHAOS` is set and excluded from the per-PR selector like the rest of `load/`, driven by `.github/workflows/test-e2e-redis-chaos.yml` and by the Buildkite `e2e-redis-chaos` step in project-releaser, which runs the proxy, Postgres and Valkey co-located with pytest in one pod and sets the opt-in), and markerless harness unit tests for the locust, process-usage, and session-anomaly aggregation logic
|
||||
- `other/` - the holding-pen suite for the `other.*` registry cluster with no home of its own yet: the master-key auth gate, JWT auth (access tokens issued by a real Keycloak realm, `idp.py` plus `idp_realm.json`, whose JWKS the proxy's `JWT_PUBLIC_KEY_URL` points at; see CONTRIBUTING.md for the start command and config block), and the process-lifecycle health probes (liveness, public readiness, authenticated readiness diagnostics). Promote a cluster out once it is large/stable enough for its own suite
|
||||
- `gateway/` - proxy configuration only (`litellm-config.yml`); no tests
|
||||
- `cost_calculation/` - cost accounting against a dedicated proxy whose whole model cost map is the test-owned `tests/e2e/cost_map.json` (loaded via `LITELLM_MODEL_COST_MAP_URL`), with provider calls answered by the scripted-provider sidecar in `scripted_provider.py`; every cost-map entry is a deployment and the cases plus asserted goldens are data in `cases.json` and `expected.json` (regenerate with `generate_expected.py`; `cost_matrix.matrix_data_errors()` runs at collection time so a stale key set fails the suite's collection loudly), deselected unless `E2E_COST_MAP_STACK` is set, driven by the Buildkite `e2e-cost-calculation` step in project-releaser, which runs a proxy booted from `gateway/cost_calculation_ci_config.yml`, Postgres and the scripted provider co-located with pytest in one pod and sets the opt-in
|
||||
- `cost_calculation/` - cost accounting against a dedicated proxy whose whole model cost map is the test-owned `tests/e2e/cost_map.json` (loaded via `LITELLM_MODEL_COST_MAP_URL`), with provider calls answered by the scripted-provider sidecar in `scripted_provider.py`; every cost-map entry is a deployment and the cases plus asserted goldens are data in `cases.json` (each exact-spend case carries a literal `expected` cell per map key; `cost_matrix.matrix_data_errors()` runs at collection time so a key absent from the cost map fails the suite's collection loudly), deselected unless `E2E_COST_MAP_STACK` is set, driven by the Buildkite `e2e-cost-calculation` step in project-releaser, which runs a proxy booted from `gateway/cost_calculation_ci_config.yml`, Postgres and the scripted provider co-located with pytest in one pod and sets the opt-in
|
||||
- `claude_code/` - the Claude Code compatibility matrix: drives the real `claude` CLI (and HTTP probes) against a proxy for each feature x provider cell, reporting tagged-union outcomes via the `compat_result` fixture; ships its own driver/builder/publisher plus `_*_unit_tests/` trees. The HTTP probes ride the shared transport (`ProxyClient.count_tokens` / `ProxyClient.messages`); the CLI-driving path stays bespoke
|
||||
- `ui/` - the Admin UI browser suite: Playwright in TypeScript, driving the dashboard served by a live proxy on port 4000 (seeded postgres + mock LLM upstream; see its `run_e2e.sh`). It is a self-contained npm package with its own lockfile and does not use the Python harness, pytest markers, or the shared transport; the Python rules in this file (typed models, `Result` unions, basedpyright zero-error gate) do not apply inside it. Its only Python file, `fixtures/mock_llm_server/server.py`, is excluded from the e2e basedpyright gate via the root `pyrightconfig.json`
|
||||
|
||||
|
|
|
|||
|
|
@ -1,81 +1,254 @@
|
|||
{
|
||||
"deployments": [
|
||||
{
|
||||
"map_key": "azure/gpt-5.4-mini",
|
||||
"litellm_model": "azure/cc-pinned-deployment",
|
||||
"base_model": "azure/gpt-5.4-mini"
|
||||
}
|
||||
{"map_key": "azure/gpt-5.4-mini", "litellm_model": "azure/cc-pinned-deployment", "base_model": "azure/gpt-5.4-mini"}
|
||||
],
|
||||
"cases": [
|
||||
{
|
||||
"name": "basic",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40}
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"expected": {
|
||||
"anthropic.claude-sonnet-5-v1:0": {"spend": 0.034, "input_cost": 0.0204, "output_cost": 0.0136, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.6": {"spend": 0.03, "input_cost": 0.018, "output_cost": 0.012, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-haiku-4-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-opus-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-sonnet-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.8-flash": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.4-mini": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.6": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.038, "input_cost": 0.0228, "output_cost": 0.0152, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"us.anthropic.claude-opus-5-v1:0": {"spend": 0.036, "input_cost": 0.0216, "output_cost": 0.0144, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "cache_read",
|
||||
"usage": {"fresh_input_tokens": 100, "cache_read_tokens": 50, "output_tokens": 30},
|
||||
"requires_rates": ["cache_read_input_token_cost"],
|
||||
"requires_caps": ["cache_read"]
|
||||
"expected": {
|
||||
"anthropic.claude-sonnet-5-v1:0": {"spend": 0.02805, "input_cost": 0.01785, "output_cost": 0.0102, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"azure/gpt-5.4-mini": {"spend": 0.0264, "input_cost": 0.0168, "output_cost": 0.0096, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"azure/gpt-5.6": {"spend": 0.02475, "input_cost": 0.01575, "output_cost": 0.009, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"claude-haiku-4-5": {"spend": 0.01155, "input_cost": 0.00735, "output_cost": 0.0042, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"claude-opus-5": {"spend": 0.00825, "input_cost": 0.00525, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"claude-sonnet-5": {"spend": 0.0099, "input_cost": 0.0063, "output_cost": 0.0036, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0231, "input_cost": 0.0147, "output_cost": 0.0084, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.0198, "input_cost": 0.0126, "output_cost": 0.0072, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.02145, "input_cost": 0.01365, "output_cost": 0.0078, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.03465, "input_cost": 0.02205, "output_cost": 0.0126, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gemini-3.8-flash": {"spend": 0.033, "input_cost": 0.021, "output_cost": 0.012, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.01485, "input_cost": 0.00945, "output_cost": 0.0054, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.0132, "input_cost": 0.0084, "output_cost": 0.0048, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gpt-5.3-codex": {"spend": 0.00495, "input_cost": 0.00315, "output_cost": 0.0018, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gpt-5.4-mini": {"spend": 0.0066, "input_cost": 0.0042, "output_cost": 0.0024, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gpt-5.5-pro": {"spend": 0.0033, "input_cost": 0.0021, "output_cost": 0.0012, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gpt-5.6": {"spend": 0.00165, "input_cost": 0.00105, "output_cost": 0.0006, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.0165, "input_cost": 0.0105, "output_cost": 0.006, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.01815, "input_cost": 0.01155, "output_cost": 0.0066, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"us.anthropic.claude-opus-5-v1:0": {"spend": 0.0297, "input_cost": 0.0189, "output_cost": 0.0108, "prompt_tokens": 150, "completion_tokens": 30}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "cache_write_5m",
|
||||
"usage": {"fresh_input_tokens": 90, "cache_write_5m_tokens": 60, "output_tokens": 30},
|
||||
"requires_rates": ["cache_creation_input_token_cost"],
|
||||
"requires_caps": ["cache_write_5m"]
|
||||
"expected": {
|
||||
"anthropic.claude-sonnet-5-v1:0": {"spend": 0.0561, "input_cost": 0.0459, "output_cost": 0.0102, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"azure/gpt-5.4-mini": {"spend": 0.0528, "input_cost": 0.0432, "output_cost": 0.0096, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"azure/gpt-5.6": {"spend": 0.0495, "input_cost": 0.0405, "output_cost": 0.009, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"claude-haiku-4-5": {"spend": 0.0231, "input_cost": 0.0189, "output_cost": 0.0042, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"claude-opus-5": {"spend": 0.0165, "input_cost": 0.0135, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"claude-sonnet-5": {"spend": 0.0198, "input_cost": 0.0162, "output_cost": 0.0036, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0408, "input_cost": 0.0324, "output_cost": 0.0084, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.0378, "input_cost": 0.0306, "output_cost": 0.0072, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.0393, "input_cost": 0.0315, "output_cost": 0.0078, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gpt-5.4-mini": {"spend": 0.0132, "input_cost": 0.0108, "output_cost": 0.0024, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gpt-5.6": {"spend": 0.0033, "input_cost": 0.0027, "output_cost": 0.0006, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.033, "input_cost": 0.027, "output_cost": 0.006, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.0363, "input_cost": 0.0297, "output_cost": 0.0066, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"us.anthropic.claude-opus-5-v1:0": {"spend": 0.0594, "input_cost": 0.0486, "output_cost": 0.0108, "prompt_tokens": 150, "completion_tokens": 30}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "cache_write_1h",
|
||||
"usage": {"fresh_input_tokens": 90, "cache_write_5m_tokens": 20, "cache_write_1h_tokens": 40, "output_tokens": 30},
|
||||
"requires_rates": ["cache_creation_input_token_cost_above_1hr", "cache_creation_input_token_cost"],
|
||||
"requires_caps": ["cache_write_1h"]
|
||||
"expected": {
|
||||
"anthropic.claude-sonnet-5-v1:0": {"spend": 0.0629, "input_cost": 0.0527, "output_cost": 0.0102, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"azure/gpt-5.4-mini": {"spend": 0.0592, "input_cost": 0.0496, "output_cost": 0.0096, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"azure/gpt-5.6": {"spend": 0.0555, "input_cost": 0.0465, "output_cost": 0.009, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"claude-haiku-4-5": {"spend": 0.0259, "input_cost": 0.0217, "output_cost": 0.0042, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"claude-opus-5": {"spend": 0.0185, "input_cost": 0.0155, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"claude-sonnet-5": {"spend": 0.0222, "input_cost": 0.0186, "output_cost": 0.0036, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0452, "input_cost": 0.0368, "output_cost": 0.0084, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.0422, "input_cost": 0.035, "output_cost": 0.0072, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.0437, "input_cost": 0.0359, "output_cost": 0.0078, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gpt-5.4-mini": {"spend": 0.0148, "input_cost": 0.0124, "output_cost": 0.0024, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"gpt-5.6": {"spend": 0.0037, "input_cost": 0.0031, "output_cost": 0.0006, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.037, "input_cost": 0.031, "output_cost": 0.006, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.0407, "input_cost": 0.0341, "output_cost": 0.0066, "prompt_tokens": 150, "completion_tokens": 30},
|
||||
"us.anthropic.claude-opus-5-v1:0": {"spend": 0.0666, "input_cost": 0.0558, "output_cost": 0.0108, "prompt_tokens": 150, "completion_tokens": 30}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "reasoning",
|
||||
"usage": {"fresh_input_tokens": 100, "output_tokens": 30, "reasoning_tokens": 70},
|
||||
"requires_rates": ["output_cost_per_reasoning_token"],
|
||||
"requires_caps": ["reasoning"]
|
||||
"expected": {
|
||||
"azure/gpt-5.4-mini": {"spend": 0.0816, "input_cost": 0.016, "output_cost": 0.0656, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"azure/gpt-5.6": {"spend": 0.0765, "input_cost": 0.015, "output_cost": 0.0615, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0609, "input_cost": 0.014, "output_cost": 0.0469, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.0577, "input_cost": 0.012, "output_cost": 0.0457, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.0593, "input_cost": 0.013, "output_cost": 0.0463, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.1071, "input_cost": 0.021, "output_cost": 0.0861, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"gemini-3.8-flash": {"spend": 0.102, "input_cost": 0.02, "output_cost": 0.082, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.0459, "input_cost": 0.009, "output_cost": 0.0369, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.0408, "input_cost": 0.008, "output_cost": 0.0328, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"gpt-5.3-codex": {"spend": 0.0153, "input_cost": 0.003, "output_cost": 0.0123, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"gpt-5.4-mini": {"spend": 0.0204, "input_cost": 0.004, "output_cost": 0.0164, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"gpt-5.5-pro": {"spend": 0.0102, "input_cost": 0.002, "output_cost": 0.0082, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"gpt-5.6": {"spend": 0.0051, "input_cost": 0.001, "output_cost": 0.0041, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.051, "input_cost": 0.01, "output_cost": 0.041, "prompt_tokens": 100, "completion_tokens": 100},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.0561, "input_cost": 0.011, "output_cost": 0.0451, "prompt_tokens": 100, "completion_tokens": 100}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"usage": {"fresh_input_tokens": 100, "audio_input_tokens": 25, "output_tokens": 30, "audio_output_tokens": 15},
|
||||
"requires_rates": ["input_cost_per_audio_token", "output_cost_per_audio_token"],
|
||||
"requires_caps": ["audio"]
|
||||
"expected": {
|
||||
"azure/gpt-5.4-mini": {"spend": 0.0664, "input_cost": 0.04, "output_cost": 0.0264, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"azure/gpt-5.6": {"spend": 0.06225, "input_cost": 0.0375, "output_cost": 0.02475, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.05045, "input_cost": 0.0305, "output_cost": 0.01995, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.04725, "input_cost": 0.0285, "output_cost": 0.01875, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.04885, "input_cost": 0.0295, "output_cost": 0.01935, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.08715, "input_cost": 0.0525, "output_cost": 0.03465, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"gemini-3.8-flash": {"spend": 0.083, "input_cost": 0.05, "output_cost": 0.033, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.03735, "input_cost": 0.0225, "output_cost": 0.01485, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.0332, "input_cost": 0.02, "output_cost": 0.0132, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"gpt-5.4-mini": {"spend": 0.0166, "input_cost": 0.01, "output_cost": 0.0066, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"gpt-5.6": {"spend": 0.00415, "input_cost": 0.0025, "output_cost": 0.00165, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.0415, "input_cost": 0.025, "output_cost": 0.0165, "prompt_tokens": 125, "completion_tokens": 45},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.04565, "input_cost": 0.0275, "output_cost": 0.01815, "prompt_tokens": 125, "completion_tokens": 45}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "tiered",
|
||||
"usage": {"fresh_input_tokens": 200001, "output_tokens": 30},
|
||||
"requires_rates": ["input_cost_per_token_above_200k_tokens", "output_cost_per_token_above_200k_tokens"]
|
||||
"expected": {
|
||||
"azure/gpt-5.4-mini": {"spend": 256.04448, "input_cost": 256.00128, "output_cost": 0.0432, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"azure/gpt-5.6": {"spend": 240.0417, "input_cost": 240.0012, "output_cost": 0.0405, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"gemini-3.1-pro-preview": {"spend": 336.05838, "input_cost": 336.00168, "output_cost": 0.0567, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"gemini-3.8-flash": {"spend": 320.0556, "input_cost": 320.0016, "output_cost": 0.054, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 144.02502, "input_cost": 144.00072, "output_cost": 0.0243, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"gemini/gemini-3.8-flash": {"spend": 128.02224, "input_cost": 128.00064, "output_cost": 0.0216, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"gpt-5.3-codex": {"spend": 48.00834, "input_cost": 48.00024, "output_cost": 0.0081, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"gpt-5.4-mini": {"spend": 64.01112, "input_cost": 64.00032, "output_cost": 0.0108, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"gpt-5.5-pro": {"spend": 32.00556, "input_cost": 32.00016, "output_cost": 0.0054, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"gpt-5.6": {"spend": 16.00278, "input_cost": 16.00008, "output_cost": 0.0027, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 160.0278, "input_cost": 160.0008, "output_cost": 0.027, "prompt_tokens": 200001, "completion_tokens": 30},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 176.03058, "input_cost": 176.00088, "output_cost": 0.0297, "prompt_tokens": 200001, "completion_tokens": 30}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "service_tier_flex",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"service_tier": "flex",
|
||||
"requires_rates": ["input_cost_per_token_flex", "output_cost_per_token_flex"]
|
||||
"expected": {
|
||||
"azure/gpt-5.4-mini": {"spend": 0.0448, "input_cost": 0.0288, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.6": {"spend": 0.042, "input_cost": 0.027, "output_cost": 0.015, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.0588, "input_cost": 0.0378, "output_cost": 0.021, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.8-flash": {"spend": 0.056, "input_cost": 0.036, "output_cost": 0.02, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.0252, "input_cost": 0.0162, "output_cost": 0.009, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.0224, "input_cost": 0.0144, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.3-codex": {"spend": 0.0084, "input_cost": 0.0054, "output_cost": 0.003, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.4-mini": {"spend": 0.0112, "input_cost": 0.0072, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.0056, "input_cost": 0.0036, "output_cost": 0.002, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.6": {"spend": 0.0028, "input_cost": 0.0018, "output_cost": 0.001, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.028, "input_cost": 0.018, "output_cost": 0.01, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.0308, "input_cost": 0.0198, "output_cost": 0.011, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "service_tier_priority",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"service_tier": "priority",
|
||||
"requires_rates": ["input_cost_per_token_priority", "output_cost_per_token_priority"]
|
||||
"expected": {
|
||||
"azure/gpt-5.4-mini": {"spend": 0.04992, "input_cost": 0.03264, "output_cost": 0.01728, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.6": {"spend": 0.0468, "input_cost": 0.0306, "output_cost": 0.0162, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.06552, "input_cost": 0.04284, "output_cost": 0.02268, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.8-flash": {"spend": 0.0624, "input_cost": 0.0408, "output_cost": 0.0216, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.02808, "input_cost": 0.01836, "output_cost": 0.00972, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.02496, "input_cost": 0.01632, "output_cost": 0.00864, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.3-codex": {"spend": 0.00936, "input_cost": 0.00612, "output_cost": 0.00324, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.4-mini": {"spend": 0.01248, "input_cost": 0.00816, "output_cost": 0.00432, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.00624, "input_cost": 0.00408, "output_cost": 0.00216, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.6": {"spend": 0.00312, "input_cost": 0.00204, "output_cost": 0.00108, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.0312, "input_cost": 0.0204, "output_cost": 0.0108, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.03432, "input_cost": 0.02244, "output_cost": 0.01188, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "web_search",
|
||||
"usage": {"fresh_input_tokens": 100, "output_tokens": 30, "web_search_calls": 3},
|
||||
"requires_rates": ["search_context_cost_per_query"],
|
||||
"requires_caps": ["web_search"],
|
||||
"wires": ["openai_responses", "anthropic_messages", "gemini_generate", "vertex_generate"]
|
||||
"expected": {
|
||||
"claude-haiku-4-5": {"spend": 0.0712, "input_cost": 0.007, "output_cost": 0.0042, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"claude-opus-5": {"spend": 0.068, "input_cost": 0.005, "output_cost": 0.003, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"claude-sonnet-5": {"spend": 0.0696, "input_cost": 0.006, "output_cost": 0.0036, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.0936, "input_cost": 0.021, "output_cost": 0.0126, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"gemini-3.8-flash": {"spend": 0.092, "input_cost": 0.02, "output_cost": 0.012, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.0744, "input_cost": 0.009, "output_cost": 0.0054, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.0728, "input_cost": 0.008, "output_cost": 0.0048, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"gpt-5.3-codex": {"spend": 0.0648, "input_cost": 0.003, "output_cost": 0.0018, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"gpt-5.5-pro": {"spend": 0.0632, "input_cost": 0.002, "output_cost": 0.0012, "prompt_tokens": 100, "completion_tokens": 30}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "web_search_single",
|
||||
"usage": {"fresh_input_tokens": 100, "output_tokens": 30, "web_search_calls": 1},
|
||||
"requires_rates": ["search_context_cost_per_query"],
|
||||
"requires_caps": ["web_search"],
|
||||
"wires": ["openai_chat", "together_chat", "fireworks_chat", "azure_chat"]
|
||||
"expected": {
|
||||
"azure/gpt-5.4-mini": {"spend": 0.0456, "input_cost": 0.016, "output_cost": 0.0096, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"azure/gpt-5.6": {"spend": 0.044, "input_cost": 0.015, "output_cost": 0.009, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0424, "input_cost": 0.014, "output_cost": 0.0084, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.0392, "input_cost": 0.012, "output_cost": 0.0072, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.0408, "input_cost": 0.013, "output_cost": 0.0078, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"gpt-5.4-mini": {"spend": 0.0264, "input_cost": 0.004, "output_cost": 0.0024, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"gpt-5.6": {"spend": 0.0216, "input_cost": 0.001, "output_cost": 0.0006, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.036, "input_cost": 0.01, "output_cost": 0.006, "prompt_tokens": 100, "completion_tokens": 30},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.0376, "input_cost": 0.011, "output_cost": 0.0066, "prompt_tokens": 100, "completion_tokens": 30}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "stream",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true
|
||||
"stream": true,
|
||||
"expected": {
|
||||
"anthropic.claude-sonnet-5-v1:0": {"spend": 0.034, "input_cost": 0.0204, "output_cost": 0.0136, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.6": {"spend": 0.03, "input_cost": 0.018, "output_cost": 0.012, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-haiku-4-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-opus-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-sonnet-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.8-flash": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.4-mini": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.6": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.038, "input_cost": 0.0228, "output_cost": 0.0152, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"us.anthropic.claude-opus-5-v1:0": {"spend": 0.036, "input_cost": 0.0216, "output_cost": 0.0144, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage",
|
||||
|
|
@ -83,33 +256,137 @@
|
|||
"stream": true,
|
||||
"stream_usage": "absent",
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["absent_usage"]
|
||||
"models": [
|
||||
"anthropic.claude-sonnet-5-v1:0",
|
||||
"azure/gpt-5.4-mini",
|
||||
"azure/gpt-5.6",
|
||||
"claude-haiku-4-5",
|
||||
"claude-opus-5",
|
||||
"claude-sonnet-5",
|
||||
"fireworks_ai/deepseek-v4p1-flash",
|
||||
"fireworks_ai/kimi-k3",
|
||||
"fireworks_ai/qwen3p8-max",
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3.8-flash",
|
||||
"gemini/gemini-3.1-pro-preview",
|
||||
"gemini/gemini-3.8-flash",
|
||||
"gpt-5.3-codex",
|
||||
"gpt-5.4-mini",
|
||||
"gpt-5.5-pro",
|
||||
"gpt-5.6",
|
||||
"meta.llama4-maverick-17b-instruct-v1:0",
|
||||
"together_ai/moonshotai/Kimi-K3",
|
||||
"together_ai/zai-org/GLM-5.3",
|
||||
"us.anthropic.claude-opus-5-v1:0"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "response_model_override",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"response_model_override": true,
|
||||
"requires_caps": ["response_model"]
|
||||
"expected": {
|
||||
"azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.6": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-haiku-4-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-opus-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-sonnet-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.8-flash": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.3-codex": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.4-mini": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.6": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "stream_response_model_override",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"response_model_override": true,
|
||||
"requires_caps": ["response_model"]
|
||||
"expected": {
|
||||
"azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.6": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-haiku-4-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-opus-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-sonnet-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.8-flash": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.3-codex": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.4-mini": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.6": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "tool_call",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"tool_call": true,
|
||||
"requires_caps": ["tool_call"]
|
||||
"expected": {
|
||||
"anthropic.claude-sonnet-5-v1:0": {"spend": 0.034, "input_cost": 0.0204, "output_cost": 0.0136, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.4-mini": {"spend": 0.032, "input_cost": 0.0192, "output_cost": 0.0128, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"azure/gpt-5.6": {"spend": 0.03, "input_cost": 0.018, "output_cost": 0.012, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-haiku-4-5": {"spend": 0.014, "input_cost": 0.0084, "output_cost": 0.0056, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-opus-5": {"spend": 0.01, "input_cost": 0.006, "output_cost": 0.004, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"claude-sonnet-5": {"spend": 0.012, "input_cost": 0.0072, "output_cost": 0.0048, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.028, "input_cost": 0.0168, "output_cost": 0.0112, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.024, "input_cost": 0.0144, "output_cost": 0.0096, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.026, "input_cost": 0.0156, "output_cost": 0.0104, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.042, "input_cost": 0.0252, "output_cost": 0.0168, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini-3.8-flash": {"spend": 0.04, "input_cost": 0.024, "output_cost": 0.016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.018, "input_cost": 0.0108, "output_cost": 0.0072, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.016, "input_cost": 0.0096, "output_cost": 0.0064, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.4-mini": {"spend": 0.008, "input_cost": 0.0048, "output_cost": 0.0032, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.6": {"spend": 0.002, "input_cost": 0.0012, "output_cost": 0.0008, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.038, "input_cost": 0.0228, "output_cost": 0.0152, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.02, "input_cost": 0.012, "output_cost": 0.008, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.022, "input_cost": 0.0132, "output_cost": 0.0088, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"us.anthropic.claude-opus-5-v1:0": {"spend": 0.036, "input_cost": 0.0216, "output_cost": 0.0144, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "stream_tool_call",
|
||||
"usage": {"fresh_input_tokens": 80, "output_tokens": 25},
|
||||
"stream": true,
|
||||
"tool_call": true,
|
||||
"requires_caps": ["tool_call"]
|
||||
"expected": {
|
||||
"anthropic.claude-sonnet-5-v1:0": {"spend": 0.0221, "input_cost": 0.0136, "output_cost": 0.0085, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"azure/gpt-5.4-mini": {"spend": 0.0208, "input_cost": 0.0128, "output_cost": 0.008, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"azure/gpt-5.6": {"spend": 0.0195, "input_cost": 0.012, "output_cost": 0.0075, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"claude-haiku-4-5": {"spend": 0.0091, "input_cost": 0.0056, "output_cost": 0.0035, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"claude-opus-5": {"spend": 0.0065, "input_cost": 0.004, "output_cost": 0.0025, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"claude-sonnet-5": {"spend": 0.0078, "input_cost": 0.0048, "output_cost": 0.003, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.0182, "input_cost": 0.0112, "output_cost": 0.007, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.0156, "input_cost": 0.0096, "output_cost": 0.006, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.0169, "input_cost": 0.0104, "output_cost": 0.0065, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"gemini-3.1-pro-preview": {"spend": 0.0273, "input_cost": 0.0168, "output_cost": 0.0105, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"gemini-3.8-flash": {"spend": 0.026, "input_cost": 0.016, "output_cost": 0.01, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.0117, "input_cost": 0.0072, "output_cost": 0.0045, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.0104, "input_cost": 0.0064, "output_cost": 0.004, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"gpt-5.3-codex": {"spend": 0.0039, "input_cost": 0.0024, "output_cost": 0.0015, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"gpt-5.4-mini": {"spend": 0.0052, "input_cost": 0.0032, "output_cost": 0.002, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"gpt-5.5-pro": {"spend": 0.0026, "input_cost": 0.0016, "output_cost": 0.001, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"gpt-5.6": {"spend": 0.0013, "input_cost": 0.0008, "output_cost": 0.0005, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.0247, "input_cost": 0.0152, "output_cost": 0.0095, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.013, "input_cost": 0.008, "output_cost": 0.005, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.0143, "input_cost": 0.0088, "output_cost": 0.0055, "prompt_tokens": 80, "completion_tokens": 25},
|
||||
"us.anthropic.claude-opus-5-v1:0": {"spend": 0.0234, "input_cost": 0.0144, "output_cost": 0.009, "prompt_tokens": 80, "completion_tokens": 25}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage_tool_call",
|
||||
|
|
@ -118,7 +395,29 @@
|
|||
"stream_usage": "absent",
|
||||
"tool_call": true,
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["absent_usage", "tool_call"]
|
||||
"models": [
|
||||
"anthropic.claude-sonnet-5-v1:0",
|
||||
"azure/gpt-5.4-mini",
|
||||
"azure/gpt-5.6",
|
||||
"claude-haiku-4-5",
|
||||
"claude-opus-5",
|
||||
"claude-sonnet-5",
|
||||
"fireworks_ai/deepseek-v4p1-flash",
|
||||
"fireworks_ai/kimi-k3",
|
||||
"fireworks_ai/qwen3p8-max",
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3.8-flash",
|
||||
"gemini/gemini-3.1-pro-preview",
|
||||
"gemini/gemini-3.8-flash",
|
||||
"gpt-5.3-codex",
|
||||
"gpt-5.4-mini",
|
||||
"gpt-5.5-pro",
|
||||
"gpt-5.6",
|
||||
"meta.llama4-maverick-17b-instruct-v1:0",
|
||||
"together_ai/moonshotai/Kimi-K3",
|
||||
"together_ai/zai-org/GLM-5.3",
|
||||
"us.anthropic.claude-opus-5-v1:0"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage_image_input",
|
||||
|
|
@ -127,14 +426,39 @@
|
|||
"stream_usage": "absent",
|
||||
"image_input": true,
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["absent_usage", "image_input"]
|
||||
"models": [
|
||||
"anthropic.claude-sonnet-5-v1:0",
|
||||
"azure/gpt-5.4-mini",
|
||||
"azure/gpt-5.6",
|
||||
"claude-haiku-4-5",
|
||||
"claude-opus-5",
|
||||
"claude-sonnet-5",
|
||||
"fireworks_ai/deepseek-v4p1-flash",
|
||||
"fireworks_ai/kimi-k3",
|
||||
"fireworks_ai/qwen3p8-max",
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3.8-flash",
|
||||
"gemini/gemini-3.1-pro-preview",
|
||||
"gemini/gemini-3.8-flash",
|
||||
"gpt-5.3-codex",
|
||||
"gpt-5.4-mini",
|
||||
"gpt-5.5-pro",
|
||||
"gpt-5.6",
|
||||
"meta.llama4-maverick-17b-instruct-v1:0",
|
||||
"together_ai/moonshotai/Kimi-K3",
|
||||
"together_ai/zai-org/GLM-5.3",
|
||||
"us.anthropic.claude-opus-5-v1:0"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "stream_incomplete",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"terminal": "incomplete",
|
||||
"requires_caps": ["responses_terminal"]
|
||||
"expected": {
|
||||
"gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage_incomplete",
|
||||
|
|
@ -143,14 +467,20 @@
|
|||
"stream_usage": "absent",
|
||||
"terminal": "incomplete",
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["responses_terminal"]
|
||||
"models": [
|
||||
"gpt-5.3-codex",
|
||||
"gpt-5.5-pro"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "stream_unvalidated",
|
||||
"usage": {"fresh_input_tokens": 120, "output_tokens": 40},
|
||||
"stream": true,
|
||||
"terminal": "unvalidated",
|
||||
"requires_caps": ["responses_terminal"]
|
||||
"expected": {
|
||||
"gpt-5.3-codex": {"spend": 0.006, "input_cost": 0.0036, "output_cost": 0.0024, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.004, "input_cost": 0.0024, "output_cost": 0.0016, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "stream_no_usage_unvalidated",
|
||||
|
|
@ -159,14 +489,22 @@
|
|||
"stream_usage": "absent",
|
||||
"terminal": "unvalidated",
|
||||
"exact_spend": false,
|
||||
"requires_caps": ["responses_terminal"]
|
||||
"models": [
|
||||
"gpt-5.3-codex",
|
||||
"gpt-5.5-pro"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "prompt_blocked",
|
||||
"usage": {"fresh_input_tokens": 1000, "output_tokens": 0},
|
||||
"terminal": "prompt_blocked",
|
||||
"response_model_override": true,
|
||||
"requires_caps": ["prompt_blocked"]
|
||||
"expected": {
|
||||
"gemini-3.1-pro-preview": {"spend": 0.2, "input_cost": 0.2, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0},
|
||||
"gemini-3.8-flash": {"spend": 0.21, "input_cost": 0.21, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.08, "input_cost": 0.08, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.09, "input_cost": 0.09, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "stream_prompt_blocked",
|
||||
|
|
@ -174,77 +512,73 @@
|
|||
"stream": true,
|
||||
"terminal": "prompt_blocked",
|
||||
"response_model_override": true,
|
||||
"requires_caps": ["prompt_blocked"]
|
||||
"expected": {
|
||||
"gemini-3.1-pro-preview": {"spend": 0.2, "input_cost": 0.2, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0},
|
||||
"gemini-3.8-flash": {"spend": 0.21, "input_cost": 0.21, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.08, "input_cost": 0.08, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.09, "input_cost": 0.09, "output_cost": 0.0, "prompt_tokens": 1000, "completion_tokens": 0}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "all_components_chat",
|
||||
"usage": {
|
||||
"fresh_input_tokens": 80,
|
||||
"cache_read_tokens": 40,
|
||||
"cache_write_5m_tokens": 20,
|
||||
"cache_write_1h_tokens": 10,
|
||||
"output_tokens": 25,
|
||||
"reasoning_tokens": 15,
|
||||
"audio_input_tokens": 5,
|
||||
"audio_output_tokens": 3
|
||||
},
|
||||
"requires_rates": [
|
||||
"output_cost_per_reasoning_token",
|
||||
"input_cost_per_audio_token",
|
||||
"output_cost_per_audio_token"
|
||||
],
|
||||
"wires": ["openai_chat", "azure_chat", "together_chat"]
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "cache_write_5m_tokens": 20, "cache_write_1h_tokens": 10, "output_tokens": 25, "reasoning_tokens": 15, "audio_input_tokens": 5, "audio_output_tokens": 3},
|
||||
"expected": {
|
||||
"azure/gpt-5.4-mini": {"spend": 0.0576, "input_cost": 0.03424, "output_cost": 0.02336, "prompt_tokens": 155, "completion_tokens": 43},
|
||||
"azure/gpt-5.6": {"spend": 0.054, "input_cost": 0.0321, "output_cost": 0.0219, "prompt_tokens": 155, "completion_tokens": 43},
|
||||
"gpt-5.4-mini": {"spend": 0.0144, "input_cost": 0.00856, "output_cost": 0.00584, "prompt_tokens": 155, "completion_tokens": 43},
|
||||
"gpt-5.6": {"spend": 0.0036, "input_cost": 0.00214, "output_cost": 0.00146, "prompt_tokens": 155, "completion_tokens": 43},
|
||||
"together_ai/moonshotai/Kimi-K3": {"spend": 0.036, "input_cost": 0.0214, "output_cost": 0.0146, "prompt_tokens": 155, "completion_tokens": 43},
|
||||
"together_ai/zai-org/GLM-5.3": {"spend": 0.0396, "input_cost": 0.02354, "output_cost": 0.01606, "prompt_tokens": 155, "completion_tokens": 43}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "all_components_fireworks",
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25},
|
||||
"wires": ["fireworks_chat"]
|
||||
"expected": {
|
||||
"fireworks_ai/deepseek-v4p1-flash": {"spend": 0.01876, "input_cost": 0.01176, "output_cost": 0.007, "prompt_tokens": 120, "completion_tokens": 25},
|
||||
"fireworks_ai/kimi-k3": {"spend": 0.01608, "input_cost": 0.01008, "output_cost": 0.006, "prompt_tokens": 120, "completion_tokens": 25},
|
||||
"fireworks_ai/qwen3p8-max": {"spend": 0.01742, "input_cost": 0.01092, "output_cost": 0.0065, "prompt_tokens": 120, "completion_tokens": 25}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "all_components_anthropic",
|
||||
"usage": {
|
||||
"fresh_input_tokens": 80,
|
||||
"cache_read_tokens": 40,
|
||||
"cache_write_5m_tokens": 20,
|
||||
"cache_write_1h_tokens": 10,
|
||||
"output_tokens": 25
|
||||
},
|
||||
"wires": ["anthropic_messages", "bedrock_converse"]
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "cache_write_5m_tokens": 20, "cache_write_1h_tokens": 10, "output_tokens": 25},
|
||||
"expected": {
|
||||
"anthropic.claude-sonnet-5-v1:0": {"spend": 0.03978, "input_cost": 0.03128, "output_cost": 0.0085, "prompt_tokens": 150, "completion_tokens": 25},
|
||||
"claude-haiku-4-5": {"spend": 0.01638, "input_cost": 0.01288, "output_cost": 0.0035, "prompt_tokens": 150, "completion_tokens": 25},
|
||||
"claude-opus-5": {"spend": 0.0117, "input_cost": 0.0092, "output_cost": 0.0025, "prompt_tokens": 150, "completion_tokens": 25},
|
||||
"claude-sonnet-5": {"spend": 0.01404, "input_cost": 0.01104, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 25},
|
||||
"meta.llama4-maverick-17b-instruct-v1:0": {"spend": 0.038, "input_cost": 0.0285, "output_cost": 0.0095, "prompt_tokens": 150, "completion_tokens": 25},
|
||||
"us.anthropic.claude-opus-5-v1:0": {"spend": 0.04212, "input_cost": 0.03312, "output_cost": 0.009, "prompt_tokens": 150, "completion_tokens": 25}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "all_components_anthropic_stream",
|
||||
"usage": {
|
||||
"fresh_input_tokens": 80,
|
||||
"cache_read_tokens": 40,
|
||||
"cache_write_5m_tokens": 20,
|
||||
"cache_write_1h_tokens": 10,
|
||||
"output_tokens": 25
|
||||
},
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "cache_write_5m_tokens": 20, "cache_write_1h_tokens": 10, "output_tokens": 25},
|
||||
"stream": true,
|
||||
"wires": ["anthropic_messages"]
|
||||
"expected": {
|
||||
"claude-haiku-4-5": {"spend": 0.01638, "input_cost": 0.01288, "output_cost": 0.0035, "prompt_tokens": 150, "completion_tokens": 25},
|
||||
"claude-opus-5": {"spend": 0.0117, "input_cost": 0.0092, "output_cost": 0.0025, "prompt_tokens": 150, "completion_tokens": 25},
|
||||
"claude-sonnet-5": {"spend": 0.01404, "input_cost": 0.01104, "output_cost": 0.003, "prompt_tokens": 150, "completion_tokens": 25}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "all_components_gemini",
|
||||
"usage": {
|
||||
"fresh_input_tokens": 80,
|
||||
"cache_read_tokens": 40,
|
||||
"output_tokens": 25,
|
||||
"reasoning_tokens": 15,
|
||||
"audio_input_tokens": 5,
|
||||
"audio_output_tokens": 3
|
||||
},
|
||||
"requires_rates": [
|
||||
"output_cost_per_reasoning_token",
|
||||
"input_cost_per_audio_token",
|
||||
"output_cost_per_audio_token"
|
||||
],
|
||||
"wires": ["gemini_generate", "vertex_generate"]
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25, "reasoning_tokens": 15, "audio_input_tokens": 5, "audio_output_tokens": 3},
|
||||
"expected": {
|
||||
"gemini-3.1-pro-preview": {"spend": 0.0546, "input_cost": 0.02394, "output_cost": 0.03066, "prompt_tokens": 125, "completion_tokens": 43},
|
||||
"gemini-3.8-flash": {"spend": 0.052, "input_cost": 0.0228, "output_cost": 0.0292, "prompt_tokens": 125, "completion_tokens": 43},
|
||||
"gemini/gemini-3.1-pro-preview": {"spend": 0.0234, "input_cost": 0.01026, "output_cost": 0.01314, "prompt_tokens": 125, "completion_tokens": 43},
|
||||
"gemini/gemini-3.8-flash": {"spend": 0.0208, "input_cost": 0.00912, "output_cost": 0.01168, "prompt_tokens": 125, "completion_tokens": 43}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "all_components_responses",
|
||||
"usage": {"fresh_input_tokens": 80, "cache_read_tokens": 40, "output_tokens": 25, "reasoning_tokens": 15},
|
||||
"requires_rates": ["output_cost_per_reasoning_token"],
|
||||
"wires": ["openai_responses"]
|
||||
"expected": {
|
||||
"gpt-5.3-codex": {"spend": 0.00627, "input_cost": 0.00252, "output_cost": 0.00375, "prompt_tokens": 120, "completion_tokens": 40},
|
||||
"gpt-5.5-pro": {"spend": 0.00418, "input_cost": 0.00168, "output_cost": 0.0025, "prompt_tokens": 120, "completion_tokens": 40}
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
|
|
|||
|
|
@ -2,9 +2,8 @@
|
|||
|
||||
Runs against a dedicated proxy whose whole model cost map is the test-owned
|
||||
``tests/e2e/cost_map.json`` (LITELLM_MODEL_COST_MAP_URL); every map entry is a
|
||||
deployment under test, the request shapes live in ``cases.json``, and the
|
||||
asserted goldens live in ``expected.json`` (regenerate proposals with
|
||||
``generate_expected.py``). Provider calls are answered by the
|
||||
deployment under test, and the request shapes plus asserted goldens live in
|
||||
``cases.json``. Provider calls are answered by the
|
||||
scripted-provider sidecar (``scripted_provider.py``), registered per scenario
|
||||
over its control API.
|
||||
|
||||
|
|
|
|||
|
|
@ -1,16 +1,13 @@
|
|||
"""The cost-calculation matrix: the model set derived from the test cost map,
|
||||
the request/response cases from ``cases.json``, and the loaders both use.
|
||||
|
||||
Three data files drive the suite; nothing in Python lists models or cases:
|
||||
Two data files drive the suite; nothing in Python lists models or cases:
|
||||
- ``tests/e2e/cost_map.json`` is the proxy's ENTIRE model cost map
|
||||
(LITELLM_MODEL_COST_MAP_URL); every entry becomes a deployment under test.
|
||||
- ``tests/e2e/cost_calculation/cases.json`` is the case list; each case runs
|
||||
for a model when the entry carries the rates it exercises (``requires_rates``)
|
||||
and the wire can report the token kinds involved (``requires_caps`` /
|
||||
``wires``).
|
||||
- ``tests/e2e/cost_calculation/expected.json`` holds the reviewed goldens; the
|
||||
tests assert them verbatim and never compute a price themselves. The rate
|
||||
arithmetic that proposes goldens lives in ``generate_expected.py``, not here.
|
||||
- ``tests/e2e/cost_calculation/cases.json`` is the case list plus the reviewed
|
||||
goldens: each exact-spend case carries an ``expected`` cell per map key it
|
||||
runs against, each recount case carries its ``models`` list, so matrix
|
||||
membership and expected values are literal data read side by side.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -26,13 +23,11 @@ from pathlib import Path
|
|||
from types import MappingProxyType
|
||||
from typing import Final, Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, TypeAdapter
|
||||
from pydantic import BaseModel, ConfigDict, Field, TypeAdapter
|
||||
from scripted_provider import Scenario, ScriptedOutput, ScriptedToolCall, ScriptedUsage, Wire
|
||||
|
||||
COST_MAP_PATH: Final = Path(__file__).resolve().parent.parent / "cost_map.json"
|
||||
CASES_PATH: Final = Path(__file__).resolve().parent / "cases.json"
|
||||
EXPECTED_PATH: Final = Path(__file__).resolve().parent / "expected.json"
|
||||
|
||||
|
||||
class SearchContextCostPerQuery(BaseModel):
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
|
@ -88,10 +83,20 @@ class DeploymentSpec(BaseModel):
|
|||
base_model: str | None = None
|
||||
|
||||
|
||||
class ExpectedCell(BaseModel):
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
spend: float
|
||||
input_cost: float
|
||||
output_cost: float
|
||||
prompt_tokens: int
|
||||
completion_tokens: int
|
||||
|
||||
|
||||
class Case(BaseModel):
|
||||
"""One request/response shape from cases.json; gated onto a model by
|
||||
``requires_rates`` (entry must carry each rate field), ``requires_caps``
|
||||
(the wire must report the token kind) and ``wires`` (shape is wire-specific)."""
|
||||
"""One request/response shape from cases.json. An exact-spend case names
|
||||
its models implicitly by carrying one ``expected`` golden per map key; a
|
||||
recount case (``exact_spend=False``) names them in ``models`` instead."""
|
||||
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
|
|
@ -105,19 +110,16 @@ class Case(BaseModel):
|
|||
tool_call: bool = False
|
||||
image_input: bool = False
|
||||
terminal: Literal["completed", "incomplete", "unvalidated", "prompt_blocked"] = "completed"
|
||||
requires_rates: tuple[str, ...] = ()
|
||||
requires_caps: tuple[str, ...] = ()
|
||||
wires: tuple[Wire, ...] | None = None
|
||||
expected: Mapping[str, ExpectedCell] = Field(default_factory=lambda: MappingProxyType({}))
|
||||
models: tuple[str, ...] = ()
|
||||
|
||||
def applies_to(self, model: FrontierModel) -> bool:
|
||||
if self.wires is not None and model.wire not in self.wires:
|
||||
return False
|
||||
caps: Final = _WIRE_CAPS[model.wire]
|
||||
if not frozenset(self.requires_caps) <= caps:
|
||||
return False
|
||||
return all(
|
||||
getattr(model.rates, field, None) is not None for field in self.requires_rates
|
||||
)
|
||||
if self.exact_spend:
|
||||
return model.map_key in self.expected
|
||||
return model.map_key in self.models
|
||||
|
||||
def expected_for(self, model: FrontierModel) -> ExpectedCell:
|
||||
return self.expected[model.map_key]
|
||||
|
||||
def scenario(self, scenario_id: str, model: FrontierModel, text: str) -> Scenario:
|
||||
return Scenario(
|
||||
|
|
@ -309,64 +311,6 @@ def _frontier() -> tuple[FrontierModel, ...]:
|
|||
|
||||
FRONTIER_MODELS: Final[tuple[FrontierModel, ...]] = _frontier()
|
||||
|
||||
# Token kinds each wire can report, gating which pricing cases apply.
|
||||
_WIRE_CAPS: Final[Mapping[str, frozenset[str]]] = MappingProxyType({
|
||||
"openai_chat": frozenset(
|
||||
{
|
||||
"cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio",
|
||||
"web_search", "response_model", "absent_usage", "tool_call", "image_input",
|
||||
}
|
||||
),
|
||||
"openai_responses": frozenset(
|
||||
{
|
||||
"cache_read", "reasoning", "web_search", "response_model", "absent_usage",
|
||||
"tool_call", "image_input", "responses_terminal",
|
||||
}
|
||||
),
|
||||
"anthropic_messages": frozenset(
|
||||
{
|
||||
"cache_read", "cache_write_5m", "cache_write_1h", "web_search",
|
||||
"response_model", "absent_usage", "tool_call", "image_input",
|
||||
}
|
||||
),
|
||||
"gemini_generate": frozenset(
|
||||
{
|
||||
"cache_read", "reasoning", "audio", "web_search", "response_model",
|
||||
"absent_usage", "tool_call", "image_input", "prompt_blocked",
|
||||
}
|
||||
),
|
||||
"together_chat": frozenset(
|
||||
{
|
||||
"cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio",
|
||||
"web_search", "response_model", "absent_usage", "tool_call", "image_input",
|
||||
}
|
||||
),
|
||||
"fireworks_chat": frozenset(
|
||||
{
|
||||
"cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio",
|
||||
"web_search", "response_model", "absent_usage", "tool_call", "image_input",
|
||||
}
|
||||
),
|
||||
"azure_chat": frozenset(
|
||||
{
|
||||
"cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio",
|
||||
"web_search", "response_model", "absent_usage", "tool_call", "image_input",
|
||||
}
|
||||
),
|
||||
"bedrock_converse": frozenset(
|
||||
{
|
||||
"cache_read", "cache_write_5m", "cache_write_1h", "absent_usage",
|
||||
"tool_call", "image_input",
|
||||
}
|
||||
),
|
||||
"vertex_generate": frozenset(
|
||||
{
|
||||
"cache_read", "reasoning", "audio", "web_search", "response_model",
|
||||
"absent_usage", "tool_call", "image_input", "prompt_blocked",
|
||||
}
|
||||
),
|
||||
})
|
||||
|
||||
TOOL_CALL_ARGUMENTS: Final = json.dumps({
|
||||
"city": "Berlin",
|
||||
"days": 7,
|
||||
|
|
@ -415,64 +359,43 @@ def image_input_data_url() -> str:
|
|||
IMAGE_INPUT_DATA_URL: Final = image_input_data_url()
|
||||
|
||||
|
||||
class ExpectedCell(BaseModel):
|
||||
model_config = ConfigDict(frozen=True)
|
||||
|
||||
spend: float
|
||||
input_cost: float
|
||||
output_cost: float
|
||||
prompt_tokens: int
|
||||
completion_tokens: int
|
||||
|
||||
|
||||
_EXPECTED_ADAPTER: Final = TypeAdapter(dict[str, ExpectedCell])
|
||||
EXPECTED: Final[Mapping[str, ExpectedCell]] = MappingProxyType(
|
||||
_EXPECTED_ADAPTER.validate_python(json.loads(EXPECTED_PATH.read_text()))
|
||||
if EXPECTED_PATH.exists()
|
||||
else {}
|
||||
)
|
||||
|
||||
|
||||
def expected_key(model: FrontierModel, case: Case) -> str:
|
||||
return f"{model.map_key}|{case.name}"
|
||||
|
||||
|
||||
def matrix_data_errors() -> tuple[str, ...]:
|
||||
"""Freshness findings for the data files, as human-readable strings.
|
||||
"""Consistency findings for the data files, as human-readable strings.
|
||||
|
||||
Called at collection time by the e2e suite; also usable from
|
||||
generate_expected.py's context without importing pytest.
|
||||
Called at collection time by the e2e suite, so a map key named by a case
|
||||
but absent from cost_map.json fails the suite's collection loudly.
|
||||
"""
|
||||
derived: Final = {
|
||||
expected_key(model, case)
|
||||
for model in FRONTIER_MODELS
|
||||
for case in cases_for(model)
|
||||
if case.exact_spend
|
||||
}
|
||||
golden: Final = set(EXPECTED)
|
||||
unknown_deployments: Final = sorted(
|
||||
spec.map_key for spec in CASES_FILE.deployments if spec.map_key not in COST_MAP
|
||||
)
|
||||
unknown_rates: Final = sorted(
|
||||
{field for case in CASES for field in case.requires_rates} - set(CostMapEntry.model_fields)
|
||||
unknown_case_models: Final = sorted(
|
||||
{
|
||||
map_key
|
||||
for case in CASES
|
||||
for map_key in (*case.expected, *case.models)
|
||||
if map_key not in COST_MAP
|
||||
}
|
||||
)
|
||||
misshapen_cases: Final = sorted(
|
||||
case.name
|
||||
for case in CASES
|
||||
if case.exact_spend == bool(case.models) or case.exact_spend != bool(case.expected)
|
||||
)
|
||||
input_rates: Final = tuple(entry.input_cost_per_token for entry in COST_MAP.values())
|
||||
findings: Final = (
|
||||
(
|
||||
"expected.json is out of sync with the derived matrix; run "
|
||||
"uv run python tests/e2e/cost_calculation/generate_expected.py "
|
||||
f"(missing: {sorted(derived - golden)}; stale: {sorted(golden - derived)})"
|
||||
)
|
||||
if derived != golden
|
||||
else None,
|
||||
(
|
||||
f"deployments entries name map keys absent from cost_map.json: {unknown_deployments}"
|
||||
if unknown_deployments
|
||||
else None
|
||||
),
|
||||
(
|
||||
f"requires_rates names that are not CostMapEntry fields: {unknown_rates}"
|
||||
if unknown_rates
|
||||
f"case expected/models name map keys absent from cost_map.json: {unknown_case_models}"
|
||||
if unknown_case_models
|
||||
else None
|
||||
),
|
||||
(
|
||||
f"cases must carry expected xor models (exact_spend matches the field): {misshapen_cases}"
|
||||
if misshapen_cases
|
||||
else None
|
||||
),
|
||||
(
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -1,211 +0,0 @@
|
|||
"""Golden generator for the cost suite. Run:
|
||||
|
||||
uv run python tests/e2e/cost_calculation/generate_expected.py
|
||||
|
||||
Loads the derived matrix (models x applicable cases), computes the golden for
|
||||
each exact-spend cell from the rate arithmetic, and writes ``expected.json``
|
||||
with sorted keys. Default behaviour adds missing cells and drops stale cells
|
||||
but never overwrites an existing cell's values (a reviewed golden is
|
||||
authoritative); ``--rewrite`` recomputes everything. Prints added/removed/kept
|
||||
counts.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from types import MappingProxyType
|
||||
from typing import Final
|
||||
|
||||
from cost_matrix import (
|
||||
EXPECTED_PATH,
|
||||
FRONTIER_MODELS,
|
||||
TIER_THRESHOLD_TOKENS,
|
||||
Case,
|
||||
CostMapEntry,
|
||||
ExpectedCell,
|
||||
FrontierModel,
|
||||
cases_for,
|
||||
expected_key,
|
||||
)
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
|
||||
def _first_present(*rates: float | None) -> float | None:
|
||||
return next((rate for rate in rates if rate is not None), None)
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ExpectedCost:
|
||||
"""The expected bill split the way the spend row's cost_breakdown reports
|
||||
it: the gross input component (cache reads/writes folded in), the output
|
||||
component, and the tool-usage component."""
|
||||
|
||||
input_cost: float
|
||||
output_cost: float
|
||||
tool_cost: float
|
||||
|
||||
@property
|
||||
def total(self) -> float:
|
||||
return self.input_cost + self.output_cost + self.tool_cost
|
||||
|
||||
|
||||
def expected_breakdown(model: FrontierModel, case: Case) -> ExpectedCost:
|
||||
"""Literal arithmetic on the test-map rates over the scripted token counts.
|
||||
|
||||
Input = fresh*in + read*read + 5m*create + 1h*create_1h + audio_in*audio_in;
|
||||
output = text*out + reasoning*reasoning + audio_out*audio_out; plus the
|
||||
billed web-search calls at the medium search-context rate. Every billed
|
||||
token is a token the provider charged for: a component whose entry has no
|
||||
dedicated rate bills at the ordinary input or output rate, and a present
|
||||
rate (including an explicit 0.0) is authoritative. When the total prompt
|
||||
tokens exceed the threshold, input/output rates come from the
|
||||
``_above_200k_tokens`` variants; a service tier takes its ``_priority`` or
|
||||
``_flex`` variant when the entry carries one, and otherwise bills at the
|
||||
base rate.
|
||||
"""
|
||||
rates: Final[CostMapEntry] = model.override_rates if case.response_model_override else model.rates
|
||||
u: Final = case.usage
|
||||
prompt_tokens: Final = (
|
||||
u.fresh_input_tokens + u.cache_read_tokens + u.cache_write_5m_tokens
|
||||
+ u.cache_write_1h_tokens + u.audio_input_tokens
|
||||
)
|
||||
tiered: Final = prompt_tokens > TIER_THRESHOLD_TOKENS
|
||||
in_rate: Final = (
|
||||
_first_present(
|
||||
rates.input_cost_per_token_above_200k_tokens if tiered else None,
|
||||
rates.input_cost_per_token_priority if case.service_tier == "priority" else None,
|
||||
rates.input_cost_per_token_flex if case.service_tier == "flex" else None,
|
||||
rates.input_cost_per_token,
|
||||
)
|
||||
or 0.0
|
||||
)
|
||||
out_rate: Final = (
|
||||
_first_present(
|
||||
rates.output_cost_per_token_above_200k_tokens if tiered else None,
|
||||
rates.output_cost_per_token_priority if case.service_tier == "priority" else None,
|
||||
rates.output_cost_per_token_flex if case.service_tier == "flex" else None,
|
||||
rates.output_cost_per_token,
|
||||
)
|
||||
or 0.0
|
||||
)
|
||||
read_rate: Final = _first_present(rates.cache_read_input_token_cost, in_rate) or 0.0
|
||||
write_rate: Final = _first_present(rates.cache_creation_input_token_cost, in_rate) or 0.0
|
||||
write_1h_rate: Final = (
|
||||
_first_present(rates.cache_creation_input_token_cost_above_1hr, write_rate) or 0.0
|
||||
)
|
||||
audio_in_rate: Final = _first_present(rates.input_cost_per_audio_token, in_rate) or 0.0
|
||||
reasoning_rate: Final = _first_present(rates.output_cost_per_reasoning_token, out_rate) or 0.0
|
||||
audio_out_rate: Final = _first_present(rates.output_cost_per_audio_token, out_rate) or 0.0
|
||||
input_cost: Final = (
|
||||
u.fresh_input_tokens * in_rate
|
||||
+ u.cache_read_tokens * read_rate
|
||||
+ u.cache_write_5m_tokens * write_rate
|
||||
+ u.cache_write_1h_tokens * write_1h_rate
|
||||
+ u.audio_input_tokens * audio_in_rate
|
||||
)
|
||||
output_cost: Final = (
|
||||
u.output_tokens * out_rate
|
||||
+ u.reasoning_tokens * reasoning_rate
|
||||
+ u.audio_output_tokens * audio_out_rate
|
||||
)
|
||||
search: Final = rates.search_context_cost_per_query
|
||||
medium_rate: Final = (
|
||||
search.search_context_size_medium if search is not None else None
|
||||
)
|
||||
if u.web_search_calls and medium_rate is None:
|
||||
raise ValueError(
|
||||
f"{model.map_key}: case {case.name} bills {u.web_search_calls} web-search "
|
||||
"calls but the entry has no search_context_cost_per_query medium rate"
|
||||
)
|
||||
tool_cost: Final = u.web_search_calls * (medium_rate if medium_rate is not None else 0.0)
|
||||
return ExpectedCost(input_cost=input_cost, output_cost=output_cost, tool_cost=tool_cost)
|
||||
|
||||
|
||||
def expected_token_columns(model: FrontierModel, case: Case) -> tuple[int, int]:
|
||||
"""(prompt_tokens, completion_tokens) the spend row should carry, per the
|
||||
wire's normalization: Anthropic folds cache read/write into prompt_tokens,
|
||||
everyone else reports the totals the wire emitted."""
|
||||
u: Final = case.usage
|
||||
if model.wire in ("anthropic_messages", "bedrock_converse"):
|
||||
return (
|
||||
u.fresh_input_tokens + u.cache_read_tokens + u.cache_write_5m_tokens + u.cache_write_1h_tokens,
|
||||
u.output_tokens,
|
||||
)
|
||||
if model.wire in ("gemini_generate", "vertex_generate"):
|
||||
return (
|
||||
u.fresh_input_tokens + u.cache_read_tokens + u.audio_input_tokens,
|
||||
u.output_tokens + u.reasoning_tokens + u.audio_output_tokens,
|
||||
)
|
||||
if model.wire == "openai_responses":
|
||||
return (
|
||||
u.fresh_input_tokens + u.cache_read_tokens,
|
||||
u.output_tokens + u.reasoning_tokens,
|
||||
)
|
||||
return (
|
||||
u.fresh_input_tokens
|
||||
+ u.cache_read_tokens
|
||||
+ u.cache_write_5m_tokens
|
||||
+ u.cache_write_1h_tokens
|
||||
+ u.audio_input_tokens,
|
||||
u.output_tokens + u.reasoning_tokens + u.audio_output_tokens,
|
||||
)
|
||||
|
||||
|
||||
def _cell(model: FrontierModel, case: Case) -> ExpectedCell:
|
||||
breakdown: Final = expected_breakdown(model, case)
|
||||
prompt_tokens, completion_tokens = expected_token_columns(model, case)
|
||||
return ExpectedCell(
|
||||
spend=breakdown.total,
|
||||
input_cost=breakdown.input_cost,
|
||||
output_cost=breakdown.output_cost,
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
)
|
||||
|
||||
|
||||
def _proposed() -> Mapping[str, ExpectedCell]:
|
||||
return MappingProxyType(
|
||||
{
|
||||
expected_key(model, case): _cell(model, case)
|
||||
for model in FRONTIER_MODELS
|
||||
for case in cases_for(model)
|
||||
if case.exact_spend
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
rewrite: Final = "--rewrite" in sys.argv[1:]
|
||||
proposed: Final = _proposed()
|
||||
proposed_values: Final = {key: cell.model_dump() for key, cell in proposed.items()}
|
||||
existing: Final[Mapping[str, ExpectedCell]] = (
|
||||
TypeAdapter(dict[str, ExpectedCell]).validate_python(
|
||||
json.loads(EXPECTED_PATH.read_text())
|
||||
)
|
||||
if EXPECTED_PATH.exists()
|
||||
else {}
|
||||
)
|
||||
merged: Final = {
|
||||
key: (
|
||||
proposed_values[key]
|
||||
if rewrite or key not in existing
|
||||
else existing[key].model_dump()
|
||||
)
|
||||
for key in sorted(proposed_values)
|
||||
}
|
||||
added: Final = sum(1 for key in proposed_values if key not in existing)
|
||||
removed: Final = sum(1 for key in existing if key not in proposed_values)
|
||||
kept: Final = sum(1 for key in proposed_values if key in existing and not rewrite)
|
||||
rewritten: Final = sum(1 for key in proposed_values if key in existing and rewrite)
|
||||
EXPECTED_PATH.write_text(json.dumps(merged, indent=2, sort_keys=True) + "\n")
|
||||
print( # noqa: T201 # CLI summary is the tool output
|
||||
f"expected.json: {added} added, {removed} removed, {kept} kept, "
|
||||
f"{rewritten} rewritten ({len(merged)} cells)"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -1,7 +1,8 @@
|
|||
"""Token-pricing e2e: every (map entry, case) cell derived from cost_map.json x
|
||||
cases.json runs a scripted-usage call through a deployment registered on the
|
||||
cost-map proxy, and the spend row plus response-cost header must equal the
|
||||
reviewed golden in expected.json verbatim -- no rate arithmetic lives here.
|
||||
reviewed golden in the case's ``expected`` cell verbatim -- no rate arithmetic
|
||||
lives here.
|
||||
|
||||
Nothing here touches a real provider or the bundled cost map: the proxy's
|
||||
upstream is the scripted-provider sidecar and its entire cost map is
|
||||
|
|
@ -15,13 +16,11 @@ from typing import Final
|
|||
|
||||
from conftest import CostCalcClient, cost_rows, register_scenario_deployment
|
||||
from cost_matrix import (
|
||||
EXPECTED,
|
||||
FRONTIER_MODELS,
|
||||
IMAGE_INPUT_DATA_URL,
|
||||
Case,
|
||||
FrontierModel,
|
||||
cases_for,
|
||||
expected_key,
|
||||
matrix_data_errors,
|
||||
recount_cost,
|
||||
)
|
||||
|
|
@ -142,7 +141,7 @@ class TestTokenPricing:
|
|||
cost_rows.assert_total_is_sum_of_components(row)
|
||||
return
|
||||
|
||||
golden: Final = EXPECTED[expected_key(model, case)]
|
||||
golden: Final = case.expected_for(model)
|
||||
|
||||
if not case.stream:
|
||||
# Streamed responses commit headers before the bill is computed, so
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue