test(e2e): assert the real bill for the four fixed cost gaps

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
kerry 2026-09-16 04:42:07 +00:00
parent bb6d7bf243
commit 415b06f5ff
3 changed files with 21 additions and 21 deletions

View file

@ -186,15 +186,12 @@ _WIRE_CAPS: Final[Mapping[str, frozenset[str]]] = MappingProxyType({
}
),
"openai_responses": frozenset({"cache_read", "reasoning", "web_search", "response_model", "absent_usage"}),
# Product gap: litellm hard-indexes message_delta["usage"] in
# anthropic/chat/handler.py, so a usage-absent anthropic stream raises
# KeyError; the real wire always carries it, so the case cannot be
# represented.
"anthropic_messages": frozenset({"cache_read", "cache_write_5m", "cache_write_1h", "web_search", "response_model"}),
# Product gap: the gemini transform sets ModelResponse.model from the
# request and drops the provider's modelVersion, so a response-model
# override can never be priced on this wire.
"gemini_generate": frozenset({"cache_read", "reasoning", "audio", "web_search", "absent_usage"}),
"anthropic_messages": frozenset(
{"cache_read", "cache_write_5m", "cache_write_1h", "web_search", "response_model", "absent_usage"}
),
"gemini_generate": frozenset(
{"cache_read", "reasoning", "audio", "web_search", "response_model", "absent_usage"}
),
"together_chat": frozenset(
{
"cache_read", "cache_write_5m", "cache_write_1h", "reasoning", "audio",
@ -240,9 +237,6 @@ class Case:
billed_web_search_calls: int = 0
response_model_override: bool = False
exact_spend: bool = True
# stream_usage=absent on a wire with no proxy-side token recount means the
# bill is exactly zero; asserted as such rather than skipped.
expect_zero_bill: bool = False
def scenario(self, scenario_id: str, model: FrontierModel, text: str) -> Scenario:
return Scenario(
@ -359,10 +353,6 @@ def cases_for(model: FrontierModel) -> tuple[Case, ...]:
stream=True,
stream_usage="absent",
exact_spend=False,
# The responses surface bills only provider-reported usage;
# with no usage in the stream the spend row is zero. Other
# wires recount tokens proxy-side and bill a nonzero amount.
expect_zero_bill=model.wire == "openai_responses",
)
if "absent_usage" in caps
else None

View file

@ -90,11 +90,6 @@ class TestTokenPricing:
)
assert row is not None, f"no spend row with a cost breakdown landed for {model.map_key}/{case.name}"
if not case.exact_spend and case.expect_zero_bill:
# The provider reported no usage and this wire has no proxy-side
# recount, so the bill is exactly zero.
assert row.spend is not None and row.spend == 0, f"no-usage stream billed {row.spend}: {row}"
return
if not case.exact_spend:
# stream_usage=absent: the provider reported no usage, so the row's
# token counts are the proxy's own recount; only assert a bill landed.

View file

@ -63,13 +63,18 @@
"supports_web_search": true
},
"fireworks_ai/deepseek-v4p1-flash": {
"cache_creation_input_token_cost": 0.00033,
"cache_creation_input_token_cost_above_1hr": 0.00044,
"cache_read_input_token_cost": 1.4e-05,
"input_cost_per_audio_token": 0.00066,
"input_cost_per_token": 0.00014000000000000001,
"litellm_provider": "fireworks_ai",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.00077,
"output_cost_per_reasoning_token": 0.00055,
"output_cost_per_token": 0.00028000000000000003,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
@ -82,13 +87,18 @@
"supports_web_search": true
},
"fireworks_ai/kimi-k3": {
"cache_creation_input_token_cost": 0.00033,
"cache_creation_input_token_cost_above_1hr": 0.00044,
"cache_read_input_token_cost": 1.2e-05,
"input_cost_per_audio_token": 0.00066,
"input_cost_per_token": 0.00012000000000000002,
"litellm_provider": "fireworks_ai",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.00077,
"output_cost_per_reasoning_token": 0.00055,
"output_cost_per_token": 0.00024000000000000003,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
@ -101,13 +111,18 @@
"supports_web_search": true
},
"fireworks_ai/qwen3p8-max": {
"cache_creation_input_token_cost": 0.00033,
"cache_creation_input_token_cost_above_1hr": 0.00044,
"cache_read_input_token_cost": 1.3e-05,
"input_cost_per_audio_token": 0.00066,
"input_cost_per_token": 0.00013000000000000002,
"litellm_provider": "fireworks_ai",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.00077,
"output_cost_per_reasoning_token": 0.00055,
"output_cost_per_token": 0.00026000000000000003,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,