test(e2e): clean cost map decimals and simplify scripted wire helpers

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
kerry 2026-09-16 22:57:23 +00:00
parent 9885dc8962
commit 2466975d29
3 changed files with 177 additions and 184 deletions

View file

@ -11,6 +11,7 @@ Deselected unless E2E_COST_MAP_STACK is set (marker `cost_map_stack`).
from __future__ import annotations
import functools
import importlib.util
import json
import sys
@ -21,6 +22,8 @@ from types import ModuleType
from typing import Final, Protocol, cast
import pytest
from cryptography.hazmat.primitives import serialization
from cryptography.hazmat.primitives.asymmetric import rsa
from cost_matrix import Case, FrontierModel
from e2e_config import COST_MAP_PROXY_URL, SCRIPTED_PROVIDER_PROXY_BASE
@ -112,33 +115,25 @@ def client() -> CostCalcClient:
return CostCalcClient(proxy=proxy)
_vertex_key_pem: str | None = None
@functools.cache
def _vertex_private_key_pem() -> str:
return rsa.generate_private_key(public_exponent=65537, key_size=2048).private_bytes(
serialization.Encoding.PEM,
serialization.PrivateFormat.PKCS8,
serialization.NoEncryption(),
).decode()
def _vertex_service_account_json() -> str:
"""A service-account credential JSON whose token_uri is the sidecar's
/_oauth/token route: the proxy's google-auth refresh then gets a scripted
access token without touching Google. One generated RSA key per process."""
global _vertex_key_pem # mutable-ok: session-scoped key generation cached for reuse
if _vertex_key_pem is None:
from cryptography.hazmat.primitives import serialization
from cryptography.hazmat.primitives.asymmetric import rsa
_vertex_key_pem = (
rsa.generate_private_key(public_exponent=65537, key_size=2048)
.private_bytes(
serialization.Encoding.PEM,
serialization.PrivateFormat.PKCS8,
serialization.NoEncryption(),
)
.decode()
)
access token without touching Google."""
return json.dumps(
{
"type": "service_account",
"project_id": "cc-scripted-project",
"private_key_id": "scripted",
"private_key": _vertex_key_pem,
"private_key": _vertex_private_key_pem(),
"client_email": "scripted@cc-scripted-project.iam.gserviceaccount.com",
"client_id": "0",
"auth_uri": f"{SCRIPTED_PROVIDER_PROXY_BASE}/_oauth/authorize",
@ -162,20 +157,21 @@ def register_scenario_deployment(
handle: Final = register_scenario(scenario)
resources.defer(lambda: delete_scenario(handle))
model_name: Final = f"{model.model_name}-{marker}"
extra_params: Final[dict[str, str]] = dict(model.litellm_params)
if model.wire == "vertex_generate":
extra_params["vertex_credentials"] = _vertex_service_account_json()
params: Final = {
"model": model.litellm_model,
"api_key": model.api_key,
"api_base": handle.api_base(),
**model.litellm_params,
**(
{"vertex_credentials": _vertex_service_account_json()}
if model.wire == "vertex_generate"
else {}
),
}
model_id: Final = client.proxy.register_model(
ModelNewBody(
model_name=model_name,
litellm_params=LiteLLMParamsBody.model_validate(
{
"model": model.litellm_model,
"api_key": model.api_key,
"api_base": handle.api_base(),
**extra_params,
}
),
litellm_params=LiteLLMParamsBody.model_validate(params),
model_info=ModelInfoBody(base_model=model.base_model),
)
)

View file

@ -997,36 +997,33 @@ def _bedrock_body(scenario: Scenario) -> Mapping[str, object]:
)
def _aws_str_header(name: str, value: str) -> bytes:
"""One eventstream header: 1-byte name len + name + type-7 marker + value."""
name_b: Final = name.encode()
value_b: Final = value.encode()
return (
struct.pack("!B", len(name_b))
+ name_b
+ struct.pack("!B", 7)
+ struct.pack("!H", len(value_b))
+ value_b
)
def _aws_event_frame(event_type: str, payload: Mapping[str, object]) -> bytes:
"""One application/vnd.amazon.eventstream frame: prelude + prelude CRC32 +
headers + JSON payload + message CRC32, matching botocore EventStreamBuffer."""
try:
from botocore.eventstream import crc32 as _crc32
except ImportError:
_crc32 = zlib.crc32
def _str_header(name: str, value: str) -> bytes:
name_b: Final = name.encode()
value_b: Final = value.encode()
return (
struct.pack("!B", len(name_b))
+ name_b
+ struct.pack("!B", 7)
+ struct.pack("!H", len(value_b))
+ value_b
)
payload_bytes: Final = json.dumps(payload, default=dict, separators=(",", ":")).encode()
headers_bytes: Final = (
_str_header(":event-type", event_type)
+ _str_header(":content-type", "application/json")
+ _str_header(":message-type", "event")
_aws_str_header(":event-type", event_type)
+ _aws_str_header(":content-type", "application/json")
+ _aws_str_header(":message-type", "event")
)
total_length: Final = 12 + len(headers_bytes) + len(payload_bytes) + 4
prelude: Final = struct.pack("!II", total_length, len(headers_bytes))
prelude_crc: Final = struct.pack("!I", _crc32(prelude) & 0xFFFFFFFF)
prelude_crc: Final = struct.pack("!I", zlib.crc32(prelude) & 0xFFFFFFFF)
message: Final = prelude + prelude_crc + headers_bytes + payload_bytes
return message + struct.pack("!I", _crc32(message, 0) & 0xFFFFFFFF)
return message + struct.pack("!I", zlib.crc32(message) & 0xFFFFFFFF)
def _bedrock_eventstream(scenario: Scenario) -> bytes:

View file

@ -1,4 +1,79 @@
{
"anthropic.claude-sonnet-5-v1:0": {
"cache_creation_input_token_cost": 0.00051,
"cache_creation_input_token_cost_above_1hr": 0.00068,
"cache_read_input_token_cost": 1.7e-05,
"input_cost_per_token": 0.00017,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 0.00034,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true
},
"azure/gpt-5.4-mini": {
"cache_creation_input_token_cost": 0.00048,
"cache_creation_input_token_cost_above_1hr": 0.00064,
"cache_read_input_token_cost": 1.6e-05,
"input_cost_per_audio_token": 0.00096,
"input_cost_per_token": 0.00016,
"input_cost_per_token_above_200k_tokens": 0.00128,
"input_cost_per_token_flex": 0.00024,
"input_cost_per_token_priority": 0.000272,
"litellm_provider": "azure",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.00112,
"output_cost_per_reasoning_token": 0.0008,
"output_cost_per_token": 0.00032,
"output_cost_per_token_above_200k_tokens": 0.00144,
"output_cost_per_token_flex": 0.0004,
"output_cost_per_token_priority": 0.000432,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.02
},
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_web_search": true
},
"azure/gpt-5.6": {
"cache_creation_input_token_cost": 0.00045,
"cache_creation_input_token_cost_above_1hr": 0.0006,
"cache_read_input_token_cost": 1.5e-05,
"input_cost_per_audio_token": 0.0009,
"input_cost_per_token": 0.00015,
"input_cost_per_token_above_200k_tokens": 0.0012,
"input_cost_per_token_flex": 0.000225,
"input_cost_per_token_priority": 0.000255,
"litellm_provider": "azure",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.00105,
"output_cost_per_reasoning_token": 0.00075,
"output_cost_per_token": 0.0003,
"output_cost_per_token_above_200k_tokens": 0.00135,
"output_cost_per_token_flex": 0.000375,
"output_cost_per_token_priority": 0.000405,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.02
},
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_web_search": true
},
"claude-haiku-4-5": {
"cache_creation_input_token_cost": 0.00021,
"cache_creation_input_token_cost_above_1hr": 0.00028000000000000003,
@ -134,6 +209,64 @@
"supports_reasoning": true,
"supports_web_search": true
},
"gemini-3.1-pro-preview": {
"cache_read_input_token_cost": 2.1e-05,
"input_cost_per_audio_token": 0.00126,
"input_cost_per_token": 0.00021,
"input_cost_per_token_above_200k_tokens": 0.00168,
"input_cost_per_token_flex": 0.000315,
"input_cost_per_token_priority": 0.000357,
"litellm_provider": "vertex_ai-language-models",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.00147,
"output_cost_per_reasoning_token": 0.00105,
"output_cost_per_token": 0.00042,
"output_cost_per_token_above_200k_tokens": 0.00189,
"output_cost_per_token_flex": 0.000525,
"output_cost_per_token_priority": 0.000567,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.02
},
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_web_search": true,
"web_search_billing_unit": "per_query"
},
"gemini-3.8-flash": {
"cache_read_input_token_cost": 2e-05,
"input_cost_per_audio_token": 0.0012,
"input_cost_per_token": 0.0002,
"input_cost_per_token_above_200k_tokens": 0.0016,
"input_cost_per_token_flex": 0.0003,
"input_cost_per_token_priority": 0.00034,
"litellm_provider": "vertex_ai-language-models",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.0014,
"output_cost_per_reasoning_token": 0.001,
"output_cost_per_token": 0.0004,
"output_cost_per_token_above_200k_tokens": 0.0018,
"output_cost_per_token_flex": 0.0005,
"output_cost_per_token_priority": 0.00054,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.02
},
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_web_search": true,
"web_search_billing_unit": "per_query"
},
"gemini/gemini-3.1-pro-preview": {
"cache_read_input_token_cost": 9e-06,
"input_cost_per_audio_token": 0.00054,
@ -304,139 +437,6 @@
"supports_reasoning": true,
"supports_web_search": true
},
"anthropic.claude-sonnet-5-v1:0": {
"cache_creation_input_token_cost": 0.00051,
"cache_creation_input_token_cost_above_1hr": 0.00068,
"cache_read_input_token_cost": 1.7e-05,
"input_cost_per_token": 0.00017,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 0.00034,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true
},
"azure/gpt-5.4-mini": {
"cache_creation_input_token_cost": 0.00048,
"cache_creation_input_token_cost_above_1hr": 0.00064,
"cache_read_input_token_cost": 1.6e-05,
"input_cost_per_audio_token": 0.00096,
"input_cost_per_token": 0.00016,
"input_cost_per_token_above_200k_tokens": 0.00128,
"input_cost_per_token_flex": 0.00024,
"input_cost_per_token_priority": 0.000272,
"litellm_provider": "azure",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.00112,
"output_cost_per_reasoning_token": 0.0008,
"output_cost_per_token": 0.00032,
"output_cost_per_token_above_200k_tokens": 0.00144,
"output_cost_per_token_flex": 0.0004,
"output_cost_per_token_priority": 0.000432,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.02
},
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_web_search": true
},
"azure/gpt-5.6": {
"cache_creation_input_token_cost": 0.00044999999999999996,
"cache_creation_input_token_cost_above_1hr": 0.0006000000000000001,
"cache_read_input_token_cost": 1.5e-05,
"input_cost_per_audio_token": 0.0009000000000000001,
"input_cost_per_token": 0.00015000000000000001,
"input_cost_per_token_above_200k_tokens": 0.0012000000000000001,
"input_cost_per_token_flex": 0.000225,
"input_cost_per_token_priority": 0.000255,
"litellm_provider": "azure",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.0010500000000000002,
"output_cost_per_reasoning_token": 0.00075,
"output_cost_per_token": 0.00030000000000000003,
"output_cost_per_token_above_200k_tokens": 0.00135,
"output_cost_per_token_flex": 0.000375,
"output_cost_per_token_priority": 0.00040499999999999996,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.02
},
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_web_search": true
},
"gemini-3.1-pro-preview": {
"cache_read_input_token_cost": 2.1e-05,
"input_cost_per_audio_token": 0.00126,
"input_cost_per_token": 0.00021,
"input_cost_per_token_above_200k_tokens": 0.00168,
"input_cost_per_token_flex": 0.000315,
"input_cost_per_token_priority": 0.000357,
"litellm_provider": "vertex_ai-language-models",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.00147,
"output_cost_per_reasoning_token": 0.0010500000000000002,
"output_cost_per_token": 0.00042,
"output_cost_per_token_above_200k_tokens": 0.0018900000000000001,
"output_cost_per_token_flex": 0.000525,
"output_cost_per_token_priority": 0.000567,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.02
},
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_web_search": true,
"web_search_billing_unit": "per_query"
},
"gemini-3.8-flash": {
"cache_read_input_token_cost": 2e-05,
"input_cost_per_audio_token": 0.0012,
"input_cost_per_token": 0.0002,
"input_cost_per_token_above_200k_tokens": 0.0016,
"input_cost_per_token_flex": 0.0003,
"input_cost_per_token_priority": 0.00034,
"litellm_provider": "vertex_ai-language-models",
"max_input_tokens": 2000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_audio_token": 0.0014000000000000002,
"output_cost_per_reasoning_token": 0.001,
"output_cost_per_token": 0.0004,
"output_cost_per_token_above_200k_tokens": 0.0018000000000000001,
"output_cost_per_token_flex": 0.0005,
"output_cost_per_token_priority": 0.00054,
"search_context_cost_per_query": {
"search_context_size_high": 0.03,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.02
},
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_web_search": true,
"web_search_billing_unit": "per_query"
},
"meta.llama4-maverick-17b-instruct-v1:0": {
"input_cost_per_token": 0.00019,
"litellm_provider": "bedrock_converse",
@ -508,7 +508,7 @@
"supports_web_search": true
},
"us.anthropic.claude-opus-5-v1:0": {
"cache_creation_input_token_cost": 0.0005400000000000001,
"cache_creation_input_token_cost": 0.00054,
"cache_creation_input_token_cost_above_1hr": 0.00072,
"cache_read_input_token_cost": 1.8e-05,
"input_cost_per_token": 0.00018,
@ -517,7 +517,7 @@
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 0.00036000000000000004,
"output_cost_per_token": 0.00036,
"supports_function_calling": true,
"supports_prompt_caching": true,
"supports_reasoning": true