mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
fix(cost_calculator): bill ultrafast prompts above 272k at the ultrafast long-context rates (#43764)
This commit is contained in:
parent
b781d157d7
commit
9dda4d895f
9 changed files with 443 additions and 3 deletions
|
|
@ -99,7 +99,7 @@
|
|||
"limit": 0
|
||||
},
|
||||
"reportUnknownArgumentType": {
|
||||
"limit": 44358
|
||||
"limit": 44802
|
||||
},
|
||||
"reportUnknownLambdaType": {
|
||||
"limit": 109
|
||||
|
|
|
|||
|
|
@ -203,6 +203,85 @@ fn threshold_tiers_and_boundaries() {
|
|||
assert_eq!(calculate(&specification, &flex).unwrap().input(), 600.0);
|
||||
}
|
||||
|
||||
#[rstest]
|
||||
#[case::ultrafast_above_threshold(ServiceTier::Ultrafast, 300_000, 9_301_000.0, 37_000.0)]
|
||||
#[case::ultrafast_at_threshold(ServiceTier::Ultrafast, 272_000, 544_500.0, 5_000.0)]
|
||||
#[case::standard_above_threshold(ServiceTier::Standard, 300_000, 3_300_600.0, 13_000.0)]
|
||||
#[case::priority_above_threshold(ServiceTier::Priority, 300_000, 5_701_000.0, 23_000.0)]
|
||||
fn tiered_long_context_rates_are_selected_by_service_tier(
|
||||
#[case] service_tier: ServiceTier,
|
||||
#[case] prompt_tokens: u64,
|
||||
#[case] expected_input: f64,
|
||||
#[case] expected_output: f64,
|
||||
) {
|
||||
let standard = Rates {
|
||||
cache_read: Rate::Value(3.0),
|
||||
..rates(Rate::Value(1.0), Rate::Value(2.0))
|
||||
};
|
||||
let tiers = [
|
||||
TierRates {
|
||||
tier: ServiceTier::Priority,
|
||||
rates: Rates {
|
||||
cache_read: Rate::Value(5.0),
|
||||
..rates(Rate::Value(3.0), Rate::Value(4.0))
|
||||
},
|
||||
},
|
||||
TierRates {
|
||||
tier: ServiceTier::Ultrafast,
|
||||
rates: Rates {
|
||||
cache_read: Rate::Value(7.0),
|
||||
..rates(Rate::Value(2.0), Rate::Value(5.0))
|
||||
},
|
||||
},
|
||||
];
|
||||
let threshold_tiers = [
|
||||
TierRates {
|
||||
tier: ServiceTier::Priority,
|
||||
rates: Rates {
|
||||
cache_read: Rate::Value(29.0),
|
||||
..rates(Rate::Value(19.0), Rate::Value(23.0))
|
||||
},
|
||||
},
|
||||
TierRates {
|
||||
tier: ServiceTier::Ultrafast,
|
||||
rates: Rates {
|
||||
cache_read: Rate::Value(41.0),
|
||||
..rates(Rate::Value(31.0), Rate::Value(37.0))
|
||||
},
|
||||
},
|
||||
];
|
||||
let thresholds = [ThresholdRates {
|
||||
above_prompt_tokens: 272_000,
|
||||
standard: Rates {
|
||||
cache_read: Rate::Value(17.0),
|
||||
..rates(Rate::Value(11.0), Rate::Value(13.0))
|
||||
},
|
||||
tiers: &threshold_tiers,
|
||||
}];
|
||||
let pricing = Pricing {
|
||||
standard,
|
||||
tiers: &tiers,
|
||||
thresholds: &thresholds,
|
||||
off_peak: None,
|
||||
};
|
||||
let base = request();
|
||||
let long_context_request = Request {
|
||||
usage: Usage {
|
||||
prompt_tokens,
|
||||
completion_tokens: 1_000,
|
||||
cache_read_tokens: 100,
|
||||
cache_write_tokens: 0,
|
||||
..base.usage
|
||||
},
|
||||
service_tier,
|
||||
..base
|
||||
};
|
||||
let cost = calculate(&pricing, &long_context_request).unwrap();
|
||||
|
||||
assert_eq!(cost.input(), expected_input);
|
||||
assert_eq!(cost.output(), expected_output);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compile_rejects_ambiguous_rates() {
|
||||
let duplicate = ThresholdRates {
|
||||
|
|
|
|||
|
|
@ -284,6 +284,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
cache_creation_input_token_cost_above_272k_tokens: float | None
|
||||
cache_creation_input_token_cost_above_272k_tokens_priority: float | None
|
||||
cache_creation_input_token_cost_above_272k_tokens_flex: float | None
|
||||
cache_creation_input_token_cost_above_272k_tokens_ultrafast: ReadOnly[float | None]
|
||||
cache_creation_input_token_cost_above_1hr: float | None
|
||||
cache_creation_input_token_cost_flex: float | None # OpenAI flex service tier pricing
|
||||
cache_creation_input_token_cost_priority: float | None # OpenAI priority service tier pricing
|
||||
|
|
@ -300,6 +301,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
cache_read_input_token_cost_above_272k_tokens: float | None
|
||||
cache_read_input_token_cost_above_272k_tokens_priority: float | None
|
||||
cache_read_input_token_cost_above_272k_tokens_flex: float | None
|
||||
cache_read_input_token_cost_above_272k_tokens_ultrafast: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_above_512k_tokens: float | None
|
||||
cache_read_input_token_cost_batches: ReadOnly[float | None]
|
||||
cache_read_input_token_cost_above_200k_tokens_batches: ReadOnly[float | None]
|
||||
|
|
@ -319,6 +321,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
input_cost_per_token_above_272k_tokens: float | None # GPT-5.4/5.4-pro: prompts >272K priced at 2x input
|
||||
input_cost_per_token_above_272k_tokens_priority: float | None
|
||||
input_cost_per_token_above_272k_tokens_flex: float | None
|
||||
input_cost_per_token_above_272k_tokens_ultrafast: ReadOnly[float | None]
|
||||
input_cost_per_token_above_512k_tokens: float | None # MiniMax-M3: prompts >512K priced at 2x input
|
||||
input_cost_per_character_above_128k_tokens: float | None # only for vertex ai models
|
||||
input_cost_per_query: float | None # per-request pricing: rerank, search, and Bedrock Marengo embeddings
|
||||
|
|
@ -360,6 +363,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
output_cost_per_token_above_272k_tokens: float | None # GPT-5.4/5.4-pro: prompts >272K priced at 1.5x output
|
||||
output_cost_per_token_above_272k_tokens_priority: float | None
|
||||
output_cost_per_token_above_272k_tokens_flex: float | None
|
||||
output_cost_per_token_above_272k_tokens_ultrafast: ReadOnly[float | None]
|
||||
output_cost_per_token_above_512k_tokens: float | None # MiniMax-M3: prompts >512K priced at 2x output
|
||||
output_cost_per_character_above_128k_tokens: float | None # only for vertex ai models
|
||||
output_cost_per_image: float | None
|
||||
|
|
@ -3737,6 +3741,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
cache_creation_input_token_cost_above_272k_tokens: float | None = None
|
||||
cache_creation_input_token_cost_above_272k_tokens_priority: float | None = None
|
||||
cache_creation_input_token_cost_above_272k_tokens_flex: float | None = None
|
||||
cache_creation_input_token_cost_above_272k_tokens_ultrafast: float | None = None
|
||||
cache_creation_input_token_cost_flex: float | None = None
|
||||
cache_creation_input_token_cost_priority: float | None = None
|
||||
cache_creation_input_token_cost_ultrafast: float | None = None
|
||||
|
|
@ -3749,6 +3754,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
cache_read_input_token_cost_above_200k_tokens_priority: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_priority: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_flex: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_ultrafast: float | None = None
|
||||
cache_read_input_token_cost_batches: float | None = None
|
||||
cache_read_input_token_cost_above_200k_tokens_batches: float | None = None
|
||||
cache_read_input_token_cost_above_272k_tokens_batches: float | None = None
|
||||
|
|
@ -3765,6 +3771,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
input_cost_per_token_above_200k_tokens_priority: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_priority: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_flex: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_ultrafast: float | None = None
|
||||
input_cost_per_token_above_200k_tokens_batches: float | None = None
|
||||
input_cost_per_token_above_272k_tokens_batches: float | None = None
|
||||
input_cost_per_query: float | None = None
|
||||
|
|
@ -3791,6 +3798,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
output_cost_per_token_above_200k_tokens_priority: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_priority: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_flex: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_ultrafast: float | None = None
|
||||
output_cost_per_token_above_200k_tokens_batches: float | None = None
|
||||
output_cost_per_token_above_272k_tokens_batches: float | None = None
|
||||
output_cost_per_character_above_128k_tokens: float | None = None
|
||||
|
|
|
|||
|
|
@ -6104,6 +6104,9 @@ def _get_model_info_helper(
|
|||
cache_creation_input_token_cost_above_272k_tokens_flex=_model_info.get(
|
||||
"cache_creation_input_token_cost_above_272k_tokens_flex", None
|
||||
),
|
||||
cache_creation_input_token_cost_above_272k_tokens_ultrafast=_model_info.get(
|
||||
"cache_creation_input_token_cost_above_272k_tokens_ultrafast", None
|
||||
),
|
||||
cache_creation_input_token_cost_flex=_model_info.get("cache_creation_input_token_cost_flex", None),
|
||||
cache_creation_input_token_cost_priority=_model_info.get(
|
||||
"cache_creation_input_token_cost_priority", None
|
||||
|
|
@ -6129,6 +6132,9 @@ def _get_model_info_helper(
|
|||
cache_read_input_token_cost_above_272k_tokens_flex=_model_info.get(
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex", None
|
||||
),
|
||||
cache_read_input_token_cost_above_272k_tokens_ultrafast=_model_info.get(
|
||||
"cache_read_input_token_cost_above_272k_tokens_ultrafast", None
|
||||
),
|
||||
cache_read_input_token_cost_above_512k_tokens=_model_info.get(
|
||||
"cache_read_input_token_cost_above_512k_tokens", None
|
||||
),
|
||||
|
|
@ -6167,6 +6173,9 @@ def _get_model_info_helper(
|
|||
input_cost_per_token_above_272k_tokens_flex=_model_info.get(
|
||||
"input_cost_per_token_above_272k_tokens_flex", None
|
||||
),
|
||||
input_cost_per_token_above_272k_tokens_ultrafast=_model_info.get(
|
||||
"input_cost_per_token_above_272k_tokens_ultrafast", None
|
||||
),
|
||||
input_cost_per_token_above_512k_tokens=_model_info.get("input_cost_per_token_above_512k_tokens", None),
|
||||
input_cost_per_query=_model_info.get("input_cost_per_query", None),
|
||||
cost_per_second=_model_info.get("cost_per_second", None),
|
||||
|
|
@ -6234,6 +6243,9 @@ def _get_model_info_helper(
|
|||
output_cost_per_token_above_272k_tokens_flex=_model_info.get(
|
||||
"output_cost_per_token_above_272k_tokens_flex", None
|
||||
),
|
||||
output_cost_per_token_above_272k_tokens_ultrafast=_model_info.get(
|
||||
"output_cost_per_token_above_272k_tokens_ultrafast", None
|
||||
),
|
||||
output_cost_per_token_above_512k_tokens=_model_info.get(
|
||||
"output_cost_per_token_above_512k_tokens", None
|
||||
),
|
||||
|
|
|
|||
|
|
@ -1,11 +1,15 @@
|
|||
import json
|
||||
from typing import Final
|
||||
import uuid
|
||||
from typing import Final, Literal
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
from pydantic import JsonValue
|
||||
|
||||
from tests.integration._support.client import JSON_OBJECT, Gateway, eventually, object_value, string_value
|
||||
from tests.integration._support.database import read_rows
|
||||
from tests.integration._support.upstream import delete_scenario, register_scenario
|
||||
from tests.integration.cost_calculation.cost_tracking_case import JsonResponse
|
||||
|
||||
STANDARD_INPUT_RATE: Final = 0.001
|
||||
STANDARD_OUTPUT_RATE: Final = 0.002
|
||||
|
|
@ -69,3 +73,190 @@ def test_ultrafast_service_tier_bills_ultrafast_rates_and_keeps_pricing_off_the_
|
|||
)
|
||||
assert_chat_bills_rates(gateway, model, "ultrafast", ULTRAFAST_INPUT_RATE, ULTRAFAST_OUTPUT_RATE)
|
||||
assert_chat_bills_rates(gateway, model, None, STANDARD_INPUT_RATE, STANDARD_OUTPUT_RATE)
|
||||
|
||||
|
||||
LONG_CONTEXT_PRICING: Final[dict[str, JsonValue]] = {
|
||||
"input_cost_per_token": 1e-06,
|
||||
"output_cost_per_token": 2e-06,
|
||||
"cache_read_input_token_cost": 1e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 3e-06,
|
||||
"output_cost_per_token_above_272k_tokens": 4e-06,
|
||||
"cache_read_input_token_cost_above_272k_tokens": 3e-07,
|
||||
"input_cost_per_token_ultrafast": 1e-05,
|
||||
"output_cost_per_token_ultrafast": 2e-05,
|
||||
"cache_read_input_token_cost_ultrafast": 1e-06,
|
||||
"input_cost_per_token_above_272k_tokens_ultrafast": 5e-05,
|
||||
"output_cost_per_token_above_272k_tokens_ultrafast": 6e-05,
|
||||
"cache_read_input_token_cost_above_272k_tokens_ultrafast": 5e-06,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_ultrafast": 6e-06,
|
||||
}
|
||||
LONG_PROMPT_TOKENS: Final = 300_000
|
||||
SHORT_PROMPT_TOKENS: Final = 1_000
|
||||
CACHED_TOKENS: Final = 400
|
||||
COMPLETION_TOKENS: Final = 1_000
|
||||
|
||||
|
||||
def _chat_response(service_tier: str | None, prompt_tokens: int) -> JsonResponse:
|
||||
return JsonResponse(
|
||||
content_type="application/json",
|
||||
body={
|
||||
"id": "chatcmpl-$UNIQUE_ID",
|
||||
"object": "chat.completion",
|
||||
"created": 1,
|
||||
"model": "integration-ultrafast-long-context",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": "long context answer"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": prompt_tokens,
|
||||
"completion_tokens": COMPLETION_TOKENS,
|
||||
"total_tokens": prompt_tokens + COMPLETION_TOKENS,
|
||||
"prompt_tokens_details": {"cached_tokens": CACHED_TOKENS},
|
||||
},
|
||||
**({} if service_tier is None else {"service_tier": service_tier}),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def _responses_response(service_tier: str | None, prompt_tokens: int) -> JsonResponse:
|
||||
return JsonResponse(
|
||||
content_type="application/json",
|
||||
body={
|
||||
"id": "resp_$UNIQUE_ID",
|
||||
"object": "response",
|
||||
"created_at": 1,
|
||||
"status": "completed",
|
||||
"model": "integration-ultrafast-long-context",
|
||||
"output": [
|
||||
{
|
||||
"type": "message",
|
||||
"id": "msg_$UNIQUE_ID",
|
||||
"status": "completed",
|
||||
"role": "assistant",
|
||||
"content": [{"type": "output_text", "text": "long context answer", "annotations": []}],
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"input_tokens": prompt_tokens,
|
||||
"output_tokens": COMPLETION_TOKENS,
|
||||
"total_tokens": prompt_tokens + COMPLETION_TOKENS,
|
||||
"input_tokens_details": {"cached_tokens": CACHED_TOKENS},
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
},
|
||||
**({} if service_tier is None else {"service_tier": service_tier}),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def _surface_response(
|
||||
surface: Literal["chat", "responses"], service_tier: str | None, prompt_tokens: int
|
||||
) -> JsonResponse:
|
||||
match surface:
|
||||
case "chat":
|
||||
return _chat_response(service_tier, prompt_tokens)
|
||||
case "responses":
|
||||
return _responses_response(service_tier, prompt_tokens)
|
||||
|
||||
|
||||
def _surface_request(
|
||||
surface: Literal["chat", "responses"], scenario_id: str, model: str, service_tier: str | None
|
||||
) -> tuple[str, dict[str, JsonValue], str]:
|
||||
match surface:
|
||||
case "chat":
|
||||
return (
|
||||
"/v1/chat/completions",
|
||||
{
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": "long context ultrafast control"}],
|
||||
**({} if service_tier is None else {"service_tier": service_tier}),
|
||||
},
|
||||
f"/{scenario_id}/chat/completions",
|
||||
)
|
||||
case "responses":
|
||||
return (
|
||||
"/v1/responses",
|
||||
{
|
||||
"model": model,
|
||||
"input": "long context ultrafast control",
|
||||
**({} if service_tier is None else {"service_tier": service_tier}),
|
||||
},
|
||||
f"/{scenario_id}/responses",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("service_tier", "prompt_tokens", "input_rate", "cache_read_rate", "output_rate"),
|
||||
(
|
||||
("ultrafast", LONG_PROMPT_TOKENS, 5e-05, 5e-06, 6e-05),
|
||||
("ultrafast", SHORT_PROMPT_TOKENS, 1e-05, 1e-06, 2e-05),
|
||||
(None, LONG_PROMPT_TOKENS, 3e-06, 3e-07, 4e-06),
|
||||
),
|
||||
ids=("ultrafast_above_272k", "ultrafast_below_272k", "standard_above_272k"),
|
||||
)
|
||||
@pytest.mark.parametrize("surface", ("chat", "responses"), ids=("chat", "responses"))
|
||||
def test_ultrafast_long_context_prompt_bills_ultrafast_long_context_rates(
|
||||
gateway: Gateway,
|
||||
surface: Literal["chat", "responses"],
|
||||
service_tier: str | None,
|
||||
prompt_tokens: int,
|
||||
input_rate: float,
|
||||
cache_read_rate: float,
|
||||
output_rate: float,
|
||||
) -> None:
|
||||
with gateway.scenario() as scenario:
|
||||
scenario_id: Final = f"ultrafast-long-context-{uuid.uuid4().hex}"
|
||||
handle: Final = register_scenario(
|
||||
scenario_id, _surface_response(surface, service_tier, prompt_tokens)
|
||||
)
|
||||
scenario.cleanups.callback(delete_scenario, handle)
|
||||
key: Final = scenario.key()
|
||||
model: Final = scenario.model(
|
||||
model=f"openai/integration-ultrafast-long-context-{uuid.uuid4().hex}",
|
||||
api_key=scenario_id,
|
||||
api_base=handle.api_base(),
|
||||
**LONG_CONTEXT_PRICING,
|
||||
)
|
||||
request_path, request_body, expected_upstream_path = _surface_request(surface, scenario_id, model, service_tier)
|
||||
with httpx.Client(base_url=gateway.upstream_url, trust_env=False) as upstream:
|
||||
upstream.get("/__observations").raise_for_status()
|
||||
response: Final = gateway.request(
|
||||
"POST",
|
||||
request_path,
|
||||
request_body,
|
||||
key=key,
|
||||
)
|
||||
observations: Final = JSON_OBJECT.validate_json(upstream.get("/__observations").content)["requests"]
|
||||
assert response.status_code == 200, response.text
|
||||
expected_input: Final = (prompt_tokens - CACHED_TOKENS) * input_rate + CACHED_TOKENS * cache_read_rate
|
||||
expected_output: Final = COMPLETION_TOKENS * output_rate
|
||||
expected: Final = expected_input + expected_output
|
||||
assert float(response.headers["x-litellm-response-cost"]) == pytest.approx(expected, rel=1e-6), response.text
|
||||
request_id: Final = string_value(object_value(response.json())["id"])
|
||||
rows: Final = eventually(
|
||||
lambda: read_rows(
|
||||
'SELECT spend, metadata, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id = %s',
|
||||
(request_id,),
|
||||
),
|
||||
lambda values: len(values) == 1,
|
||||
seconds=70,
|
||||
)
|
||||
assert rows[0]["prompt_tokens"] == prompt_tokens
|
||||
assert rows[0]["completion_tokens"] == COMPLETION_TOKENS
|
||||
assert float(rows[0]["spend"]) == pytest.approx(expected, rel=1e-6)
|
||||
metadata: Final = rows[0]["metadata"]
|
||||
parsed: Final = json.loads(metadata) if isinstance(metadata, str) else object_value(metadata)
|
||||
breakdown: Final = object_value(parsed["cost_breakdown"])
|
||||
assert float(breakdown["input_cost"]) == pytest.approx(expected_input, rel=1e-6)
|
||||
assert float(breakdown["output_cost"]) == pytest.approx(expected_output, rel=1e-6)
|
||||
assert isinstance(observations, list)
|
||||
assert len(observations) == 1
|
||||
observation: Final = object_value(observations[0])
|
||||
upstream_path: Final = string_value(observation["path"])
|
||||
assert upstream_path == expected_upstream_path, upstream_path
|
||||
body: Final = object_value(observation["body"])
|
||||
assert body.get("service_tier") == service_tier, body
|
||||
assert not set(LONG_CONTEXT_PRICING).intersection(body), body
|
||||
|
|
|
|||
|
|
@ -1,10 +1,13 @@
|
|||
import json
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[2]
|
||||
MAIN_PATH = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
|
|
@ -72,7 +75,23 @@ PRIORITY_LONG_CONTEXT = {
|
|||
},
|
||||
}
|
||||
|
||||
EXPECTED = {**FLEX_LONG_CONTEXT, **PRIORITY_LONG_CONTEXT}
|
||||
ULTRAFAST_LONG_CONTEXT = {
|
||||
"gpt-6-astra": {
|
||||
"input_cost_per_token_above_272k_tokens_ultrafast": 0.00012,
|
||||
"output_cost_per_token_above_272k_tokens_ultrafast": 0.00045,
|
||||
"cache_read_input_token_cost_above_272k_tokens_ultrafast": 1.2e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_ultrafast": 0.00015,
|
||||
}
|
||||
}
|
||||
|
||||
EXPECTED: Final = {
|
||||
model: {
|
||||
**FLEX_LONG_CONTEXT.get(model, {}),
|
||||
**PRIORITY_LONG_CONTEXT.get(model, {}),
|
||||
**ULTRAFAST_LONG_CONTEXT.get(model, {}),
|
||||
}
|
||||
for model in {**FLEX_LONG_CONTEXT, **PRIORITY_LONG_CONTEXT, **ULTRAFAST_LONG_CONTEXT}
|
||||
}
|
||||
|
||||
NO_PUBLISHED_PRIORITY_LONG_CONTEXT = ("gpt-5.4", "gpt-5.5")
|
||||
|
||||
|
|
@ -102,6 +121,85 @@ TIERED_COST_CASES = [
|
|||
("gpt-5.6-terra", "priority", 8e-06, 3.6e-05),
|
||||
("gpt-5.6-luna", "priority", 8e-07, 3.6e-06),
|
||||
("gpt-6-astra", "priority", 4e-05, 0.00015),
|
||||
("gpt-6-astra", "ultrafast", 0.00012, 0.00045),
|
||||
("gpt-6-sol", "priority", 8e-06, 3e-05),
|
||||
("gpt-6-luna", "priority", 4e-07, 1.5e-06),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", (MAIN_PATH, BACKUP_PATH), ids=("main", "backup"))
|
||||
def test_catalogs_contain_expected_tiered_long_context_rates(path: Path) -> None:
|
||||
catalog: Final = _load(path)
|
||||
|
||||
assert {model: {key: catalog[model][key] for key in rates} for model, rates in EXPECTED.items()} == EXPECTED, (
|
||||
"gpt-6-astra ultrafast rates per https://developers.openai.com/api/docs/pricing (2026-09-29)"
|
||||
)
|
||||
|
||||
|
||||
def test_get_model_info_preserves_expected_tiered_long_context_rates() -> None:
|
||||
assert {
|
||||
model: {key: litellm.get_model_info(model)[key] for key in rates} for model, rates in EXPECTED.items()
|
||||
} == EXPECTED
|
||||
|
||||
|
||||
@pytest.mark.parametrize(("model", "service_tier", "input_rate", "output_rate"), TIERED_COST_CASES)
|
||||
def test_tiered_long_context_cost_uses_catalog_rates(
|
||||
model: str, service_tier: str, input_rate: float, output_rate: float
|
||||
) -> None:
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=LONG_CONTEXT_PROMPT_TOKENS,
|
||||
completion_tokens=COMPLETION_TOKENS,
|
||||
total_tokens=LONG_CONTEXT_PROMPT_TOKENS + COMPLETION_TOKENS,
|
||||
)
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider="openai",
|
||||
service_tier=service_tier,
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(LONG_CONTEXT_PROMPT_TOKENS * input_rate)
|
||||
assert completion_cost == pytest.approx(COMPLETION_TOKENS * output_rate)
|
||||
|
||||
|
||||
def test_gpt_6_astra_ultrafast_long_context_costs_and_controls() -> None:
|
||||
ultrafast_usage: Final = Usage(
|
||||
prompt_tokens=300_000,
|
||||
completion_tokens=1_000,
|
||||
total_tokens=301_000,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=100, cache_creation_tokens=200),
|
||||
)
|
||||
ultrafast_prompt_cost, ultrafast_completion_cost = generic_cost_per_token(
|
||||
model="gpt-6-astra",
|
||||
usage=ultrafast_usage,
|
||||
custom_llm_provider="openai",
|
||||
service_tier="ultrafast",
|
||||
)
|
||||
standard_prompt_cost, standard_completion_cost = generic_cost_per_token(
|
||||
model="gpt-6-astra",
|
||||
usage=ultrafast_usage,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
below_threshold_usage: Final = Usage(
|
||||
prompt_tokens=271_000,
|
||||
completion_tokens=1_000,
|
||||
total_tokens=272_000,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=100, cache_creation_tokens=200),
|
||||
)
|
||||
below_threshold_prompt_cost, below_threshold_completion_cost = generic_cost_per_token(
|
||||
model="gpt-6-astra",
|
||||
usage=below_threshold_usage,
|
||||
custom_llm_provider="openai",
|
||||
service_tier="ultrafast",
|
||||
)
|
||||
|
||||
assert (ultrafast_prompt_cost, ultrafast_completion_cost) == pytest.approx(
|
||||
(299_700 * 0.00012 + 100 * 1.2e-05 + 200 * 0.00015, 1_000 * 0.00045)
|
||||
)
|
||||
assert ultrafast_prompt_cost + ultrafast_completion_cost == pytest.approx(36.4452)
|
||||
assert (standard_prompt_cost, standard_completion_cost) == pytest.approx(
|
||||
(299_700 * 0.00002 + 100 * 2e-06 + 200 * 2.5e-05, 1_000 * 7.5e-05)
|
||||
)
|
||||
assert (below_threshold_prompt_cost, below_threshold_completion_cost) == pytest.approx(
|
||||
(270_700 * 6e-05 + 100 * 6e-06 + 200 * 7.5e-05, 1_000 * 0.0003)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1829,6 +1829,34 @@ def test_register_deployment_in_model_cost_writes_both_key_families():
|
|||
_restore_model_cost_entries(model_keys)
|
||||
|
||||
|
||||
def test_router_registration_keeps_ultrafast_long_context_deployment_pricing() -> None:
|
||||
model_id: Final = "ultrafast-long-context-pricing-id"
|
||||
backend_key: Final = "openai/gpt-6-astra"
|
||||
rates: Final = {
|
||||
"input_cost_per_token_above_272k_tokens_ultrafast": 0.00012,
|
||||
"output_cost_per_token_above_272k_tokens_ultrafast": 0.00045,
|
||||
"cache_read_input_token_cost_above_272k_tokens_ultrafast": 1.2e-05,
|
||||
"cache_creation_input_token_cost_above_272k_tokens_ultrafast": 0.00015,
|
||||
}
|
||||
model_cost_entries: Final = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key)) for key in (model_id, backend_key, "gpt-6-astra")
|
||||
}
|
||||
try:
|
||||
Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "ultrafast-long-context-pricing",
|
||||
"litellm_params": {"model": backend_key, **rates},
|
||||
"model_info": {"id": model_id},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
assert {key: litellm.model_cost[model_id][key] for key in rates} == rates
|
||||
finally:
|
||||
_restore_model_cost_entries(model_cost_entries)
|
||||
|
||||
|
||||
def test_reload_keeps_custom_pricing_configured_on_litellm_params_for_a_db_model():
|
||||
"""
|
||||
A deployment added at runtime, which is what /model/new does, configures its
|
||||
|
|
|
|||
|
|
@ -768,12 +768,14 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_creation_input_token_cost_above_256k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_272k_tokens": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_272k_tokens_flex": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_272k_tokens_ultrafast": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_272k_tokens_priority": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_200k_tokens_batches": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_272k_tokens_batches": {"type": "number"},
|
||||
"cache_creation_input_token_cost_batches": {"type": "number"},
|
||||
"cache_creation_input_token_cost_flex": {"type": "number"},
|
||||
"cache_creation_input_token_cost_priority": {"type": "number"},
|
||||
"cache_creation_input_token_cost_ultrafast": {"type": "number"},
|
||||
"cache_read_input_token_cost": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_32k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_128k_tokens": {"type": "number"},
|
||||
|
|
@ -782,7 +784,9 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_read_input_token_cost_above_256k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_272k_tokens": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_272k_tokens_flex": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_272k_tokens_ultrafast": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_512k_tokens": {"type": "number"},
|
||||
"input_cost_per_token_above_272k_tokens_ultrafast": {"type": "number"},
|
||||
"cache_read_input_token_cost_batches": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_272k_tokens_batches": {"type": "number"},
|
||||
"cache_creation_input_token_cost_above_1hr_above_200k_tokens": {"type": "number"},
|
||||
|
|
@ -809,11 +813,13 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"cache_read_input_token_cost_flex": {"type": "number"},
|
||||
"cache_read_input_token_cost_priority": {"type": "number"},
|
||||
"cache_read_input_token_cost_balanced": {"type": "number"},
|
||||
"cache_read_input_token_cost_ultrafast": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_200k_tokens_priority": {"type": "number"},
|
||||
"cache_read_input_token_cost_above_272k_tokens_priority": {"type": "number"},
|
||||
"input_cost_per_token_flex": {"type": "number"},
|
||||
"input_cost_per_token_priority": {"type": "number"},
|
||||
"input_cost_per_token_balanced": {"type": "number"},
|
||||
"input_cost_per_token_ultrafast": {"type": "number"},
|
||||
"input_cost_per_token_above_200k_tokens_priority": {"type": "number"},
|
||||
"input_cost_per_token_above_272k_tokens_priority": {"type": "number"},
|
||||
"input_cost_per_token_above_272k_tokens_batches": {"type": "number"},
|
||||
|
|
@ -822,8 +828,10 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"output_cost_per_token_flex": {"type": "number"},
|
||||
"output_cost_per_token_priority": {"type": "number"},
|
||||
"output_cost_per_token_balanced": {"type": "number"},
|
||||
"output_cost_per_token_ultrafast": {"type": "number"},
|
||||
"output_cost_per_token_above_200k_tokens_priority": {"type": "number"},
|
||||
"output_cost_per_token_above_272k_tokens_priority": {"type": "number"},
|
||||
"output_cost_per_token_above_272k_tokens_ultrafast": {"type": "number"},
|
||||
"output_cost_per_token_above_272k_tokens_batches": {"type": "number"},
|
||||
"output_cost_per_token_above_272k_tokens_flex": {"type": "number"},
|
||||
"regional_endpoint_uplift_multiplier": {"type": "number"},
|
||||
|
|
|
|||
16
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
16
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -33030,6 +33030,8 @@ export interface components {
|
|||
cache_creation_input_token_cost_above_272k_tokens_flex?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Priority */
|
||||
cache_creation_input_token_cost_above_272k_tokens_priority?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Ultrafast */
|
||||
cache_creation_input_token_cost_above_272k_tokens_ultrafast?: number | null;
|
||||
/** Cache Creation Input Token Cost Batches */
|
||||
cache_creation_input_token_cost_batches?: number | null;
|
||||
/** Cache Creation Input Token Cost Flex */
|
||||
|
|
@ -33058,6 +33060,8 @@ export interface components {
|
|||
cache_read_input_token_cost_above_272k_tokens_flex?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Priority */
|
||||
cache_read_input_token_cost_above_272k_tokens_priority?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Ultrafast */
|
||||
cache_read_input_token_cost_above_272k_tokens_ultrafast?: number | null;
|
||||
/** Cache Read Input Token Cost Above 512K Tokens */
|
||||
cache_read_input_token_cost_above_512k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Balanced */
|
||||
|
|
@ -33142,6 +33146,8 @@ export interface components {
|
|||
input_cost_per_token_above_272k_tokens_flex?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Priority */
|
||||
input_cost_per_token_above_272k_tokens_priority?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Ultrafast */
|
||||
input_cost_per_token_above_272k_tokens_ultrafast?: number | null;
|
||||
/** Input Cost Per Token Above 512K Tokens */
|
||||
input_cost_per_token_above_512k_tokens?: number | null;
|
||||
/** Input Cost Per Token Balanced */
|
||||
|
|
@ -33269,6 +33275,8 @@ export interface components {
|
|||
output_cost_per_token_above_272k_tokens_flex?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Priority */
|
||||
output_cost_per_token_above_272k_tokens_priority?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Ultrafast */
|
||||
output_cost_per_token_above_272k_tokens_ultrafast?: number | null;
|
||||
/** Output Cost Per Token Above 512K Tokens */
|
||||
output_cost_per_token_above_512k_tokens?: number | null;
|
||||
/** Output Cost Per Token Balanced */
|
||||
|
|
@ -46861,6 +46869,8 @@ export interface components {
|
|||
cache_creation_input_token_cost_above_272k_tokens_flex?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Priority */
|
||||
cache_creation_input_token_cost_above_272k_tokens_priority?: number | null;
|
||||
/** Cache Creation Input Token Cost Above 272K Tokens Ultrafast */
|
||||
cache_creation_input_token_cost_above_272k_tokens_ultrafast?: number | null;
|
||||
/** Cache Creation Input Token Cost Batches */
|
||||
cache_creation_input_token_cost_batches?: number | null;
|
||||
/** Cache Creation Input Token Cost Flex */
|
||||
|
|
@ -46889,6 +46899,8 @@ export interface components {
|
|||
cache_read_input_token_cost_above_272k_tokens_flex?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Priority */
|
||||
cache_read_input_token_cost_above_272k_tokens_priority?: number | null;
|
||||
/** Cache Read Input Token Cost Above 272K Tokens Ultrafast */
|
||||
cache_read_input_token_cost_above_272k_tokens_ultrafast?: number | null;
|
||||
/** Cache Read Input Token Cost Above 512K Tokens */
|
||||
cache_read_input_token_cost_above_512k_tokens?: number | null;
|
||||
/** Cache Read Input Token Cost Balanced */
|
||||
|
|
@ -46973,6 +46985,8 @@ export interface components {
|
|||
input_cost_per_token_above_272k_tokens_flex?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Priority */
|
||||
input_cost_per_token_above_272k_tokens_priority?: number | null;
|
||||
/** Input Cost Per Token Above 272K Tokens Ultrafast */
|
||||
input_cost_per_token_above_272k_tokens_ultrafast?: number | null;
|
||||
/** Input Cost Per Token Above 512K Tokens */
|
||||
input_cost_per_token_above_512k_tokens?: number | null;
|
||||
/** Input Cost Per Token Balanced */
|
||||
|
|
@ -47100,6 +47114,8 @@ export interface components {
|
|||
output_cost_per_token_above_272k_tokens_flex?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Priority */
|
||||
output_cost_per_token_above_272k_tokens_priority?: number | null;
|
||||
/** Output Cost Per Token Above 272K Tokens Ultrafast */
|
||||
output_cost_per_token_above_272k_tokens_ultrafast?: number | null;
|
||||
/** Output Cost Per Token Above 512K Tokens */
|
||||
output_cost_per_token_above_512k_tokens?: number | null;
|
||||
/** Output Cost Per Token Balanced */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue