mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-16 23:41:43 +00:00
Merge pull request #41298 from BerriAI/litellm_drop_remaining_vendor_fact_pins
test: drop remaining tests that pin cost-map vendor facts
This commit is contained in:
commit
171888b716
44 changed files with 792 additions and 5406 deletions
File diff suppressed because it is too large
Load diff
|
|
@ -4,18 +4,13 @@ from typing import NamedTuple
|
|||
|
||||
import pytest
|
||||
|
||||
|
||||
import litellm
|
||||
from litellm.llms.bedrock.chat.converse_transformation import AmazonConverseConfig
|
||||
from litellm.llms.bedrock.common_utils import BedrockModelInfo
|
||||
from litellm.utils import _get_model_info_helper
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.types.utils import (
|
||||
Choices,
|
||||
Message,
|
||||
ModelResponse,
|
||||
PromptTokensDetailsWrapper,
|
||||
Usage,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -31,8 +26,7 @@ def local_model_cost_map(monkeypatch):
|
|||
litellm.bedrock_converse_models.update(
|
||||
key
|
||||
for key, value in litellm.model_cost.items()
|
||||
if isinstance(value, dict)
|
||||
and value.get("litellm_provider") == "bedrock_converse"
|
||||
if isinstance(value, dict) and value.get("litellm_provider") == "bedrock_converse"
|
||||
)
|
||||
yield
|
||||
finally:
|
||||
|
|
@ -56,45 +50,69 @@ class GptProfile(NamedTuple):
|
|||
GPT_5_6_PROFILES = [
|
||||
GptProfile(
|
||||
model_id="us.openai.gpt-5.6-sol",
|
||||
input_cost=4.4e-06, input_cost_above_272k=8.8e-06,
|
||||
cache_write=5.5e-06, cache_write_above_272k=1.1e-05,
|
||||
cache_read=4.4e-07, cache_read_above_272k=8.8e-07,
|
||||
output_cost=2.2e-05, output_cost_above_272k=3.3e-05,
|
||||
input_cost=4.4e-06,
|
||||
input_cost_above_272k=8.8e-06,
|
||||
cache_write=5.5e-06,
|
||||
cache_write_above_272k=1.1e-05,
|
||||
cache_read=4.4e-07,
|
||||
cache_read_above_272k=8.8e-07,
|
||||
output_cost=2.2e-05,
|
||||
output_cost_above_272k=3.3e-05,
|
||||
),
|
||||
GptProfile(
|
||||
model_id="global.openai.gpt-5.6-sol",
|
||||
input_cost=4e-06, input_cost_above_272k=8e-06,
|
||||
cache_write=5e-06, cache_write_above_272k=1e-05,
|
||||
cache_read=4e-07, cache_read_above_272k=8e-07,
|
||||
output_cost=2e-05, output_cost_above_272k=3e-05,
|
||||
input_cost=4e-06,
|
||||
input_cost_above_272k=8e-06,
|
||||
cache_write=5e-06,
|
||||
cache_write_above_272k=1e-05,
|
||||
cache_read=4e-07,
|
||||
cache_read_above_272k=8e-07,
|
||||
output_cost=2e-05,
|
||||
output_cost_above_272k=3e-05,
|
||||
),
|
||||
GptProfile(
|
||||
model_id="us.openai.gpt-5.6-terra",
|
||||
input_cost=2.2e-06, input_cost_above_272k=4.4e-06,
|
||||
cache_write=2.75e-06, cache_write_above_272k=5.5e-06,
|
||||
cache_read=2.2e-07, cache_read_above_272k=4.4e-07,
|
||||
output_cost=1.32e-05, output_cost_above_272k=1.98e-05,
|
||||
input_cost=2.2e-06,
|
||||
input_cost_above_272k=4.4e-06,
|
||||
cache_write=2.75e-06,
|
||||
cache_write_above_272k=5.5e-06,
|
||||
cache_read=2.2e-07,
|
||||
cache_read_above_272k=4.4e-07,
|
||||
output_cost=1.32e-05,
|
||||
output_cost_above_272k=1.98e-05,
|
||||
),
|
||||
GptProfile(
|
||||
model_id="global.openai.gpt-5.6-terra",
|
||||
input_cost=2e-06, input_cost_above_272k=4e-06,
|
||||
cache_write=2.5e-06, cache_write_above_272k=5e-06,
|
||||
cache_read=2e-07, cache_read_above_272k=4e-07,
|
||||
output_cost=1.2e-05, output_cost_above_272k=1.8e-05,
|
||||
input_cost=2e-06,
|
||||
input_cost_above_272k=4e-06,
|
||||
cache_write=2.5e-06,
|
||||
cache_write_above_272k=5e-06,
|
||||
cache_read=2e-07,
|
||||
cache_read_above_272k=4e-07,
|
||||
output_cost=1.2e-05,
|
||||
output_cost_above_272k=1.8e-05,
|
||||
),
|
||||
GptProfile(
|
||||
model_id="us.openai.gpt-5.6-luna",
|
||||
input_cost=2.2e-07, input_cost_above_272k=4.4e-07,
|
||||
cache_write=2.75e-07, cache_write_above_272k=5.5e-07,
|
||||
cache_read=2.2e-08, cache_read_above_272k=4.4e-08,
|
||||
output_cost=1.32e-06, output_cost_above_272k=1.98e-06,
|
||||
input_cost=2.2e-07,
|
||||
input_cost_above_272k=4.4e-07,
|
||||
cache_write=2.75e-07,
|
||||
cache_write_above_272k=5.5e-07,
|
||||
cache_read=2.2e-08,
|
||||
cache_read_above_272k=4.4e-08,
|
||||
output_cost=1.32e-06,
|
||||
output_cost_above_272k=1.98e-06,
|
||||
),
|
||||
GptProfile(
|
||||
model_id="global.openai.gpt-5.6-luna",
|
||||
input_cost=2e-07, input_cost_above_272k=4e-07,
|
||||
cache_write=2.5e-07, cache_write_above_272k=5e-07,
|
||||
cache_read=2e-08, cache_read_above_272k=4e-08,
|
||||
output_cost=1.2e-06, output_cost_above_272k=1.8e-06,
|
||||
input_cost=2e-07,
|
||||
input_cost_above_272k=4e-07,
|
||||
cache_write=2.5e-07,
|
||||
cache_write_above_272k=5e-07,
|
||||
cache_read=2e-08,
|
||||
cache_read_above_272k=4e-08,
|
||||
output_cost=1.2e-06,
|
||||
output_cost_above_272k=1.8e-06,
|
||||
),
|
||||
]
|
||||
|
||||
|
|
@ -116,112 +134,18 @@ def _bedrock_response(model, usage):
|
|||
)
|
||||
|
||||
|
||||
def test_proxy_cost_calculation_scenario():
|
||||
"""Test exact GitHub issue scenario: proxy cost calculation"""
|
||||
model = "litellm_proxy/bedrock/us.anthropic.claude-3-5-haiku-20241022-v1:0"
|
||||
|
||||
# Test model info lookup works
|
||||
model_info = _get_model_info_helper(
|
||||
model=model, custom_llm_provider="litellm_proxy"
|
||||
)
|
||||
assert model_info is not None
|
||||
|
||||
# Test cost calculation works
|
||||
response = ModelResponse(
|
||||
id="test",
|
||||
created=1234567890,
|
||||
model=model,
|
||||
object="chat.completion",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(content="Test", role="assistant"),
|
||||
)
|
||||
],
|
||||
usage=Usage(total_tokens=150, prompt_tokens=100, completion_tokens=50),
|
||||
)
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response, model=model, custom_llm_provider="litellm_proxy"
|
||||
)
|
||||
expected_cost = (100 * 8e-07) + (50 * 4e-06)
|
||||
assert cost == expected_cost
|
||||
|
||||
|
||||
@pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id)
|
||||
def test_bedrock_gpt_5_6_profiles_route_to_converse(profile, local_model_cost_map):
|
||||
"""GPT-5.6 is served by Converse on bedrock-runtime, never by Invoke."""
|
||||
assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse"
|
||||
|
||||
|
||||
def test_bedrock_gpt_5_6_above_272k_tier_applies_to_cost(local_model_cost_map):
|
||||
"""A prompt over 272K tokens is billed at the long-context rate, not the base rate."""
|
||||
response = _bedrock_response(
|
||||
"bedrock/us.openai.gpt-5.6-sol",
|
||||
Usage(prompt_tokens=300000, completion_tokens=1000, total_tokens=301000),
|
||||
)
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model="bedrock/us.openai.gpt-5.6-sol",
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx((300000 * 8.8e-06) + (1000 * 3.3e-05), rel=1e-9)
|
||||
|
||||
|
||||
def test_bedrock_gpt_5_6_bills_cache_read_tokens(local_model_cost_map):
|
||||
"""Bedrock caches long prefixes implicitly and reports them, so a cache-read turn
|
||||
must be billed at the cache rate rather than dropped to zero."""
|
||||
usage = Usage(
|
||||
prompt_tokens=15611,
|
||||
completion_tokens=5,
|
||||
total_tokens=15616,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=15609),
|
||||
)
|
||||
response = _bedrock_response("bedrock/us.openai.gpt-5.6-sol", usage)
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model="bedrock/us.openai.gpt-5.6-sol",
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
|
||||
expected = (2 * 4.4e-06) + (15609 * 4.4e-07) + (5 * 2.2e-05)
|
||||
assert cost == pytest.approx(expected, rel=1e-9)
|
||||
# Without cache_read_input_token_cost the cached prefix bills at zero.
|
||||
assert cost > (15611 * 4.4e-06) * 0.1
|
||||
|
||||
|
||||
def test_bedrock_gpt_5_6_bills_cache_write_tokens(local_model_cost_map):
|
||||
"""The write side of the same cache cycle is billed at the 30m cache-write rate."""
|
||||
usage = Usage(
|
||||
prompt_tokens=15611,
|
||||
completion_tokens=5,
|
||||
total_tokens=15616,
|
||||
cache_creation_input_tokens=15609,
|
||||
)
|
||||
response = _bedrock_response("bedrock/us.openai.gpt-5.6-sol", usage)
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model="bedrock/us.openai.gpt-5.6-sol",
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
|
||||
expected = (2 * 4.4e-06) + (15609 * 5.5e-06) + (5 * 2.2e-05)
|
||||
assert cost == pytest.approx(expected, rel=1e-9)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id)
|
||||
def test_bedrock_gpt_5_6_offers_tools_and_reasoning_effort_but_not_thinking(profile, local_model_cost_map):
|
||||
"""GPT-5.x on Converse maps reasoning_effort to reasoning.effort, so reasoning_effort
|
||||
is offered while the Anthropic-only thinking/output_config are not, alongside the tool
|
||||
params these models accept."""
|
||||
supported = AmazonConverseConfig().get_supported_openai_params(
|
||||
model=f"bedrock/{profile.model_id}"
|
||||
)
|
||||
supported = AmazonConverseConfig().get_supported_openai_params(model=f"bedrock/{profile.model_id}")
|
||||
|
||||
assert "tools" in supported
|
||||
assert "tool_choice" in supported
|
||||
|
|
|
|||
|
|
@ -3,10 +3,8 @@ from pathlib import Path
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.llms.base_llm.ocr.transformation import OCRPage, OCRResponse, OCRUsageInfo
|
||||
|
||||
COST_PER_PAGE = 0.0015
|
||||
REPO_ROOT = Path(__file__).parents[5]
|
||||
COST_MAPS = [
|
||||
REPO_ROOT / "model_prices_and_context_window.json",
|
||||
|
|
@ -28,17 +26,3 @@ def test_model_info_resolves_ocr_mode_and_price(local_model_cost_map, model: str
|
|||
info = litellm.get_model_info(model=model, custom_llm_provider=provider)
|
||||
|
||||
assert info["mode"] == "ocr"
|
||||
assert info["ocr_cost_per_page"] == COST_PER_PAGE
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model, provider", MODELS)
|
||||
@pytest.mark.parametrize("pages_processed", [1, 3])
|
||||
def test_cost_scales_with_billed_pages(local_model_cost_map, model: str, provider: str, pages_processed: int) -> None:
|
||||
cost = completion_cost(
|
||||
completion_response=_ocr_response(model.split("/", 1)[1], pages_processed),
|
||||
model=model,
|
||||
custom_llm_provider=provider,
|
||||
call_type="ocr",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(COST_PER_PAGE * pages_processed)
|
||||
|
|
|
|||
|
|
@ -215,7 +215,6 @@ def test_every_model_without_published_cache_dbu_bills_cache_at_its_own_input_ra
|
|||
and model not in PUBLISHED_DBU_PER_MILLION
|
||||
]
|
||||
|
||||
assert len(without_published_rates) == 14
|
||||
for model in without_published_rates:
|
||||
info = _model_info(model)
|
||||
for field in CACHE_FIELDS:
|
||||
|
|
|
|||
|
|
@ -1,51 +0,0 @@
|
|||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
|
||||
def test_databricks_pricing_integrity():
|
||||
"""
|
||||
Verifies that for all Databricks models in model_prices_and_context_window.json:
|
||||
USD Price == DBU Price * 0.07
|
||||
"""
|
||||
json_path = os.path.join(
|
||||
os.path.dirname(__file__), "../../../../model_prices_and_context_window.json"
|
||||
)
|
||||
|
||||
# Verify file exists
|
||||
assert os.path.exists(
|
||||
json_path
|
||||
), f"Could not find model_prices_and_context_window.json at {json_path}"
|
||||
|
||||
with open(json_path, "r") as f:
|
||||
data = json.load(f)
|
||||
|
||||
conversion_rate = 0.07 # 1 DBU = 0.07 USD
|
||||
errors = []
|
||||
|
||||
for model, info in data.items():
|
||||
if info.get("litellm_provider") == "databricks":
|
||||
# Check Input Cost
|
||||
input_usd = info.get("input_cost_per_token")
|
||||
input_dbu = info.get("input_dbu_cost_per_token")
|
||||
|
||||
if input_usd is not None and input_dbu is not None:
|
||||
expected = input_dbu * conversion_rate
|
||||
# Allow small floating point difference
|
||||
if abs(input_usd - expected) > 1e-9:
|
||||
errors.append(
|
||||
f"{model} input mismatch: USD={input_usd}, DBU={input_dbu}, Expected={expected}"
|
||||
)
|
||||
|
||||
# Check Output Cost
|
||||
output_usd = info.get("output_cost_per_token")
|
||||
output_dbu = info.get("output_dbu_cost_per_token")
|
||||
|
||||
if output_usd is not None and output_dbu is not None:
|
||||
expected = output_dbu * conversion_rate
|
||||
if abs(output_usd - expected) > 1e-9:
|
||||
errors.append(
|
||||
f"{model} output mismatch: USD={output_usd}, DBU={output_dbu}, Expected={expected}"
|
||||
)
|
||||
|
||||
assert not errors, "\n" + "\n".join(errors)
|
||||
|
|
@ -1,10 +1,6 @@
|
|||
|
||||
import math
|
||||
from datetime import datetime, timezone
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
import litellm
|
||||
from litellm.llms.fireworks_ai.cost_calculator import cost_per_token
|
||||
from litellm.types.utils import OffPeakPricing, PromptTokensDetailsWrapper, Usage
|
||||
|
|
@ -26,49 +22,16 @@ def _usage(prompt_tokens: int, cached_tokens: int, completion_tokens: int) -> Us
|
|||
)
|
||||
|
||||
|
||||
def test_cached_prompt_tokens_billed_at_cache_read_rate():
|
||||
prompt_tokens = 7036
|
||||
cached_tokens = 7020
|
||||
completion_tokens = 8
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model=MODEL, usage=_usage(prompt_tokens, cached_tokens, completion_tokens)
|
||||
)
|
||||
|
||||
expected_prompt_cost = (prompt_tokens - cached_tokens) * INPUT_COST + cached_tokens * CACHE_READ_COST
|
||||
assert prompt_cost == pytest.approx(expected_prompt_cost)
|
||||
assert completion_cost == pytest.approx(completion_tokens * OUTPUT_COST)
|
||||
|
||||
full_rate_cost = prompt_tokens * INPUT_COST
|
||||
assert prompt_cost < full_rate_cost
|
||||
|
||||
|
||||
def test_warm_call_cheaper_than_cold_call():
|
||||
prompt_tokens = 7036
|
||||
completion_tokens = 8
|
||||
|
||||
cold_prompt_cost, _ = cost_per_token(
|
||||
model=MODEL, usage=_usage(prompt_tokens, 16, completion_tokens)
|
||||
)
|
||||
warm_prompt_cost, _ = cost_per_token(
|
||||
model=MODEL, usage=_usage(prompt_tokens, 7020, completion_tokens)
|
||||
)
|
||||
cold_prompt_cost, _ = cost_per_token(model=MODEL, usage=_usage(prompt_tokens, 16, completion_tokens))
|
||||
warm_prompt_cost, _ = cost_per_token(model=MODEL, usage=_usage(prompt_tokens, 7020, completion_tokens))
|
||||
|
||||
assert warm_prompt_cost < cold_prompt_cost
|
||||
|
||||
|
||||
def test_no_cached_tokens_matches_full_input_rate():
|
||||
prompt_tokens = 100
|
||||
completion_tokens = 10
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model=MODEL, usage=_usage(prompt_tokens, 0, completion_tokens)
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(prompt_tokens * INPUT_COST)
|
||||
assert completion_cost == pytest.approx(completion_tokens * OUTPUT_COST)
|
||||
|
||||
|
||||
OFF_PEAK_MODEL = "accounts/fireworks/models/off-peak-test"
|
||||
OFF_PEAK_WINDOW = "14:00-00:00"
|
||||
INSIDE_WINDOW = datetime(2026, 9, 3, 17, 25, tzinfo=timezone.utc)
|
||||
|
|
@ -78,7 +41,9 @@ STANDARD_OUTPUT_COST = 6e-07
|
|||
STANDARD_CACHE_READ_COST = 1.5e-08
|
||||
|
||||
|
||||
def _register_off_peak_model(off_peak_pricing: OffPeakPricing, cache_read_cost: float | None = STANDARD_CACHE_READ_COST) -> None:
|
||||
def _register_off_peak_model(
|
||||
off_peak_pricing: OffPeakPricing, cache_read_cost: float | None = STANDARD_CACHE_READ_COST
|
||||
) -> None:
|
||||
litellm.model_cost[f"fireworks_ai/{OFF_PEAK_MODEL}"] = {
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"mode": "chat",
|
||||
|
|
@ -151,7 +116,9 @@ def test_off_peak_window_bills_cached_tokens_at_the_off_peak_input_rate_without_
|
|||
def test_off_peak_defaults_to_the_current_time():
|
||||
"""The proxy's cost dispatch passes no clock, so an all-day window has to apply on the
|
||||
default current time."""
|
||||
_register_off_peak_model({"hours_utc": "00:00-00:00", "input_cost_per_token": 1e-08, "output_cost_per_token": 2e-08})
|
||||
_register_off_peak_model(
|
||||
{"hours_utc": "00:00-00:00", "input_cost_per_token": 1e-08, "output_cost_per_token": 2e-08}
|
||||
)
|
||||
usage = _usage(prompt_tokens=1000, cached_tokens=0, completion_tokens=200)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=OFF_PEAK_MODEL, usage=usage)
|
||||
|
|
|
|||
|
|
@ -1,65 +0,0 @@
|
|||
"""
|
||||
Regression test for Fireworks Kimi K2.5 / K2.6 / K2.7 context and output limits.
|
||||
|
||||
Fireworks publishes a 262144-token context window for every Kimi K2.5, K2.6 and
|
||||
K2.7 model, but caps generation well below that. A previous bulk edit had flattened
|
||||
max_output_tokens/max_tokens to 262144 (equal to the context window), which let the
|
||||
pre-call context-window check admit requests asking for a full 262144-token
|
||||
completion that Fireworks then rejects. These assertions pin the corrected per-alias
|
||||
limits so a future bulk edit can't silently flatten them again.
|
||||
"""
|
||||
|
||||
import json
|
||||
from importlib.resources import files
|
||||
|
||||
import pytest
|
||||
|
||||
CONTEXT_WINDOW = 262144
|
||||
OUTPUT_LIMIT = 32768
|
||||
|
||||
KIMI_ALIASES = (
|
||||
"fireworks_ai/kimi-k2p5",
|
||||
"fireworks_ai/kimi-k2p6",
|
||||
"fireworks_ai/kimi-k2p6-fast",
|
||||
"fireworks_ai/kimi-k2p7-code",
|
||||
"fireworks_ai/kimi-k2p7-code-fast",
|
||||
"fireworks_ai/accounts/fireworks/models/kimi-k2p5",
|
||||
"fireworks_ai/accounts/fireworks/models/kimi-k2p6",
|
||||
"fireworks_ai/accounts/fireworks/models/kimi-k2p7-code",
|
||||
"fireworks_ai/accounts/fireworks/routers/kimi-k2p6-fast",
|
||||
"fireworks_ai/accounts/fireworks/routers/kimi-k2p7-code-fast",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def use_local_model_cost_map():
|
||||
monkeypatch = pytest.MonkeyPatch()
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
|
||||
import litellm
|
||||
from litellm.utils import _invalidate_model_cost_lowercase_map
|
||||
|
||||
original_model_cost = litellm.model_cost
|
||||
litellm.model_cost = json.loads(
|
||||
files("litellm")
|
||||
.joinpath("model_prices_and_context_window_backup.json")
|
||||
.read_text(encoding="utf-8")
|
||||
)
|
||||
litellm.get_model_info.cache_clear()
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
try:
|
||||
yield litellm
|
||||
finally:
|
||||
litellm.model_cost = original_model_cost
|
||||
litellm.get_model_info.cache_clear()
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
monkeypatch.undo()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("alias", KIMI_ALIASES)
|
||||
def test_fireworks_kimi_get_model_info_limits(use_local_model_cost_map, alias):
|
||||
model_info = use_local_model_cost_map.get_model_info(model=alias)
|
||||
|
||||
assert model_info["max_input_tokens"] == CONTEXT_WINDOW
|
||||
assert model_info["max_output_tokens"] == OUTPUT_LIMIT
|
||||
assert model_info["max_tokens"] == OUTPUT_LIMIT
|
||||
|
|
@ -4,7 +4,6 @@ import json
|
|||
import httpx
|
||||
import pytest
|
||||
|
||||
|
||||
import litellm
|
||||
from litellm.llms.gemini.audio_transcription.transformation import (
|
||||
GeminiAudioTranscriptionConfig,
|
||||
|
|
@ -318,15 +317,3 @@ class TestCostRegression:
|
|||
assert live_entry["input_cost_per_token"] == 3.5e-06
|
||||
assert live_entry["output_cost_per_token"] == 2.1e-05
|
||||
assert live_entry["supported_endpoints"] == ["/v1/realtime"]
|
||||
|
||||
def test_completion_cost_bills_provider_reported_tokens(self, config, local_cost_map):
|
||||
payload = json.loads(json.dumps(COMPLETED_RESPONSE))
|
||||
payload["usage"]["total_output_tokens"] = 10
|
||||
payload["usage"]["total_tokens"] = 210
|
||||
response = config.transform_audio_transcription_response(make_response(payload))
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
model="gemini/gemini-3.5-transcribe",
|
||||
call_type="transcription",
|
||||
)
|
||||
assert cost == pytest.approx(199 * 2e-06 + 1 * 2e-06 + 10 * 1.2e-05)
|
||||
|
|
|
|||
|
|
@ -1,128 +0,0 @@
|
|||
"""
|
||||
Cost tests for Mistral OCR models against the real litellm cost map
|
||||
(no monkeypatching of get_model_info). These regress the pricing entries
|
||||
for mistral-ocr-4-0 and mistral-ocr-latest, which now both resolve to
|
||||
OCR 4 at $4 / 1000 pages.
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.cost_calculator import completion_cost
|
||||
from litellm.llms.base_llm.ocr.transformation import OCRPage, OCRResponse, OCRUsageInfo
|
||||
|
||||
OCR4_COST_PER_PAGE = 0.004
|
||||
OCR4_ANNOTATION_COST_PER_PAGE = 0.005
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[5]
|
||||
MAIN_COST_MAP = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
BACKUP_COST_MAP = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
|
||||
OCR3_MODEL = "mistral/mistral-ocr-2512"
|
||||
OCR3_COST_PER_PAGE = 0.002
|
||||
OCR3_ANNOTATION_COST_PER_PAGE = 0.003
|
||||
|
||||
AZURE_DOC_AI_MODEL = "azure_ai/mistral-document-ai-2512"
|
||||
AZURE_DOC_AI_COST_PER_PAGE = 0.003
|
||||
|
||||
|
||||
def _ocr_response(model: str, pages_processed: int) -> OCRResponse:
|
||||
return OCRResponse(
|
||||
pages=[OCRPage(index=i, markdown=f"page {i}") for i in range(pages_processed)],
|
||||
model=model,
|
||||
usage_info=OCRUsageInfo(pages_processed=pages_processed),
|
||||
)
|
||||
|
||||
|
||||
def _annotated_ocr_response(model: str, pages_processed: int | None, annotation_pages: int) -> OCRResponse:
|
||||
return OCRResponse(
|
||||
pages=[],
|
||||
model=model,
|
||||
usage_info=OCRUsageInfo(pages_processed=pages_processed, pages_processed_annotation=annotation_pages),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["mistral-ocr-4-0", "mistral-ocr-latest"])
|
||||
@pytest.mark.parametrize("pages_processed", [1, 3, 10])
|
||||
def test_ocr4_cost_scales_with_pages(model: str, pages_processed: int) -> None:
|
||||
cost = completion_cost(
|
||||
completion_response=_ocr_response(model, pages_processed),
|
||||
model=f"mistral/{model}",
|
||||
custom_llm_provider="mistral",
|
||||
call_type="ocr",
|
||||
)
|
||||
assert cost == pytest.approx(OCR4_COST_PER_PAGE * pages_processed)
|
||||
|
||||
|
||||
def test_ocr3_model_info_price(local_model_cost_map) -> None:
|
||||
info = litellm.get_model_info(model=OCR3_MODEL, custom_llm_provider="mistral")
|
||||
assert info["ocr_cost_per_page"] == OCR3_COST_PER_PAGE
|
||||
|
||||
|
||||
@pytest.mark.parametrize("pages_processed", [1, 3, 10])
|
||||
def test_ocr3_cost_scales_with_pages(local_model_cost_map, pages_processed: int) -> None:
|
||||
cost = completion_cost(
|
||||
completion_response=_ocr_response("mistral-ocr-2512", pages_processed),
|
||||
model=OCR3_MODEL,
|
||||
custom_llm_provider="mistral",
|
||||
call_type="ocr",
|
||||
)
|
||||
assert cost == pytest.approx(OCR3_COST_PER_PAGE * pages_processed)
|
||||
|
||||
|
||||
def test_ocr3_bills_ocr_and_annotation_pages_at_their_own_rates(local_model_cost_map) -> None:
|
||||
cost = completion_cost(
|
||||
completion_response=_annotated_ocr_response("mistral-ocr-2512", 2, 3),
|
||||
model=OCR3_MODEL,
|
||||
custom_llm_provider="mistral",
|
||||
call_type="ocr",
|
||||
)
|
||||
assert cost == pytest.approx(2 * OCR3_COST_PER_PAGE + 3 * OCR3_ANNOTATION_COST_PER_PAGE)
|
||||
|
||||
|
||||
def test_ocr3_bills_annotation_only_response(local_model_cost_map) -> None:
|
||||
cost = completion_cost(
|
||||
completion_response=_annotated_ocr_response("mistral-ocr-2512", 0, 3),
|
||||
model=OCR3_MODEL,
|
||||
custom_llm_provider="mistral",
|
||||
call_type="ocr",
|
||||
)
|
||||
assert cost == pytest.approx(3 * OCR3_ANNOTATION_COST_PER_PAGE)
|
||||
|
||||
|
||||
def test_ocr3_bills_annotation_pages_when_pages_processed_missing(local_model_cost_map) -> None:
|
||||
cost = completion_cost(
|
||||
completion_response=_annotated_ocr_response("mistral-ocr-2512", None, 4),
|
||||
model=OCR3_MODEL,
|
||||
custom_llm_provider="mistral",
|
||||
call_type="ocr",
|
||||
)
|
||||
assert cost == pytest.approx(4 * OCR3_ANNOTATION_COST_PER_PAGE)
|
||||
|
||||
|
||||
def test_azure_doc_ai_annotation_pages_fall_back_to_ocr_rate(local_model_cost_map) -> None:
|
||||
info = litellm.get_model_info(model=AZURE_DOC_AI_MODEL, custom_llm_provider="azure_ai")
|
||||
assert info.get("annotation_cost_per_page") is None
|
||||
assert info["ocr_cost_per_page"] == AZURE_DOC_AI_COST_PER_PAGE
|
||||
cost = completion_cost(
|
||||
completion_response=_annotated_ocr_response("mistral-document-ai-2512", 0, 1),
|
||||
model=AZURE_DOC_AI_MODEL,
|
||||
custom_llm_provider="azure_ai",
|
||||
call_type="ocr",
|
||||
)
|
||||
assert cost == pytest.approx(AZURE_DOC_AI_COST_PER_PAGE)
|
||||
|
||||
|
||||
def test_azure_ocr4_bills_ocr_and_annotation_pages_at_their_own_rates(local_model_cost_map) -> None:
|
||||
info = litellm.get_model_info(model="azure_ai/mistral-ocr-4-0", custom_llm_provider="azure_ai")
|
||||
assert info["ocr_cost_per_page"] == OCR4_COST_PER_PAGE
|
||||
assert info["annotation_cost_per_page"] == OCR4_ANNOTATION_COST_PER_PAGE
|
||||
cost = completion_cost(
|
||||
completion_response=_annotated_ocr_response("mistral-ocr-4-0", 2, 3),
|
||||
model="azure_ai/mistral-ocr-4-0",
|
||||
custom_llm_provider="azure_ai",
|
||||
call_type="ocr",
|
||||
)
|
||||
assert cost == pytest.approx(2 * OCR4_COST_PER_PAGE + 3 * OCR4_ANNOTATION_COST_PER_PAGE)
|
||||
|
|
@ -75,9 +75,3 @@ def test_shipped_per_second_models_bill_a_non_zero_cost(model, provider):
|
|||
prompt_cost, completion_cost = cost_per_second(model=model, custom_llm_provider=provider, duration=60.0)
|
||||
|
||||
assert prompt_cost + completion_cost > 0.0
|
||||
|
||||
|
||||
def test_whisper_bills_its_documented_rate_once():
|
||||
prompt_cost, completion_cost = cost_per_second(model="whisper-1", custom_llm_provider="openai", duration=30.0)
|
||||
|
||||
assert prompt_cost + completion_cost == pytest.approx(0.003)
|
||||
|
|
|
|||
|
|
@ -172,7 +172,6 @@ class TestSCXAIModelMetadata:
|
|||
assert info["supports_prompt_caching"] is True
|
||||
assert 0 < info["cache_read_input_token_cost"] < info["input_cost_per_token"]
|
||||
|
||||
assert info["max_output_tokens"] == 131072
|
||||
assert info["max_tokens"] == info["max_output_tokens"]
|
||||
assert info["max_input_tokens"] >= 1_000_000
|
||||
|
||||
|
|
|
|||
|
|
@ -14,17 +14,15 @@ from unittest.mock import patch
|
|||
import pytest
|
||||
|
||||
# Add the project root to Python path
|
||||
|
||||
import litellm
|
||||
from litellm.cost_calculator import completion_cost, cost_per_token
|
||||
from litellm.llms.perplexity.cost_calculator import (
|
||||
cost_per_token as perplexity_cost_per_token,
|
||||
)
|
||||
from litellm.types.utils import (
|
||||
CompletionTokensDetailsWrapper,
|
||||
OffPeakPricing,
|
||||
Usage,
|
||||
PromptTokensDetailsWrapper,
|
||||
Usage,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -64,167 +62,6 @@ class TestPerplexityCostCalculator:
|
|||
}
|
||||
}
|
||||
|
||||
def test_basic_cost_calculation(self):
|
||||
"""Test basic cost calculation without additional fields."""
|
||||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
# Expected costs:
|
||||
# Input: 100 tokens * $2e-6 = $0.0002
|
||||
# Output: 50 tokens * $8e-6 = $0.0004
|
||||
expected_prompt_cost = 100 * 2e-6
|
||||
expected_completion_cost = 50 * 8e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_citation_tokens_cost_calculation(self):
|
||||
"""Test cost calculation with citation tokens."""
|
||||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
|
||||
# Add citation tokens
|
||||
usage.citation_tokens = 25
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
# Expected costs:
|
||||
# Input: 100 tokens * $2e-6 = $0.0002
|
||||
# Citation: 25 tokens * $2e-6 = $0.00005
|
||||
# Total prompt cost: $0.00025
|
||||
# Output: 50 tokens * $8e-6 = $0.0004
|
||||
expected_prompt_cost = (100 * 2e-6) + (25 * 2e-6)
|
||||
expected_completion_cost = 50 * 8e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_search_queries_cost_calculation(self):
|
||||
"""Test cost calculation with search queries."""
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=3),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
# Expected costs:
|
||||
# Input: 100 tokens * $2e-6 = $0.0002
|
||||
# Output: 50 tokens * $8e-6 = $0.0004
|
||||
# Search: 3 queries * $0.005 per request = $0.015
|
||||
# Total completion cost: $0.0154
|
||||
expected_prompt_cost = 100 * 2e-6
|
||||
expected_completion_cost = (50 * 8e-6) + (3 * 0.005)
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_reasoning_tokens_from_direct_attribute(self):
|
||||
"""Test reasoning tokens cost calculation from direct attribute."""
|
||||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
|
||||
# Set reasoning tokens directly
|
||||
usage.reasoning_tokens = 20
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
# `completion_tokens` includes `reasoning_tokens` per the OpenAI/Perplexity
|
||||
# convention codified in PR #18607. Non-reasoning portion = 50 - 20 = 30.
|
||||
# Input: 100 tokens * $2e-6 = $0.0002
|
||||
# Output (text): 30 tokens * $8e-6 = $0.00024
|
||||
# Reasoning: 20 tokens * $3e-6 = $0.00006
|
||||
# Total completion cost = $0.0003
|
||||
expected_prompt_cost = 100 * 2e-6
|
||||
expected_completion_cost = ((50 - 20) * 8e-6) + (20 * 3e-6)
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_reasoning_tokens_from_completion_tokens_details(self):
|
||||
"""Test reasoning tokens cost calculation from completion_tokens_details."""
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
reasoning_tokens=20, # This should be stored in completion_tokens_details
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
# Same convention as the direct-attribute case above; reasoning is a subset of
|
||||
# completion_tokens, so non-reasoning portion = 50 - 20 = 30.
|
||||
expected_prompt_cost = 100 * 2e-6
|
||||
expected_completion_cost = ((50 - 20) * 8e-6) + (20 * 3e-6)
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_comprehensive_cost_calculation(self):
|
||||
"""Test cost calculation with all fields combined."""
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
reasoning_tokens=15,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=2),
|
||||
)
|
||||
|
||||
# Add custom fields
|
||||
usage.citation_tokens = 30
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
# Expected costs (reasoning is a subset of completion_tokens):
|
||||
# Input: 100 tokens * $2e-6 = $0.0002
|
||||
# Citation: 30 tokens * $2e-6 = $0.00006
|
||||
# Total prompt cost = $0.00026
|
||||
# Output (text): (50 - 15) tokens * $8e-6 = $0.00028
|
||||
# Reasoning: 15 tokens * $3e-6 = $0.000045
|
||||
# Search: 2 queries * $0.005 per request = $0.01
|
||||
# Total completion cost = $0.010325
|
||||
expected_prompt_cost = (100 * 2e-6) + (30 * 2e-6)
|
||||
expected_completion_cost = ((50 - 15) * 8e-6) + (15 * 3e-6) + (2 * 0.005)
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_zero_values_handling(self):
|
||||
"""Test that zero or missing values are handled correctly."""
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=0),
|
||||
)
|
||||
|
||||
# These should not raise errors and should not affect cost
|
||||
usage.citation_tokens = 0
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
# Should be same as basic calculation
|
||||
expected_prompt_cost = 100 * 2e-6
|
||||
expected_completion_cost = 50 * 8e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_missing_model_info_fields(self):
|
||||
"""Test behavior when model info is missing some fields."""
|
||||
usage = Usage(
|
||||
|
|
@ -237,18 +74,14 @@ class TestPerplexityCostCalculator:
|
|||
usage.citation_tokens = 25
|
||||
|
||||
# Mock get_model_info to return incomplete model info
|
||||
with patch(
|
||||
"litellm.llms.perplexity.cost_calculator.get_model_info"
|
||||
) as mock_get_model_info:
|
||||
with patch("litellm.llms.perplexity.cost_calculator.get_model_info") as mock_get_model_info:
|
||||
mock_get_model_info.return_value = {
|
||||
"input_cost_per_token": 2e-6,
|
||||
"output_cost_per_token": 8e-6,
|
||||
# Missing search_queries_cost_per_query
|
||||
}
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(model="sonar-deep-research", usage=usage)
|
||||
|
||||
# Should only calculate basic costs when fields are missing
|
||||
expected_prompt_cost = 100 * 2e-6
|
||||
|
|
@ -257,104 +90,6 @@ class TestPerplexityCostCalculator:
|
|||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_integration_with_main_cost_calculator(self):
|
||||
"""Test integration with the main LiteLLM cost calculator."""
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
reasoning_tokens=10,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1),
|
||||
)
|
||||
|
||||
usage.citation_tokens = 20
|
||||
|
||||
# Test main cost calculator
|
||||
prompt_cost, completion_cost_val = cost_per_token(
|
||||
model="sonar-deep-research",
|
||||
custom_llm_provider="perplexity",
|
||||
usage_object=usage,
|
||||
)
|
||||
|
||||
# Should match direct call to perplexity cost calculator
|
||||
expected_prompt, expected_completion = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost_val, expected_completion, rel_tol=1e-6)
|
||||
|
||||
def test_integration_with_completion_cost_function(self):
|
||||
"""Test integration with the completion_cost function."""
|
||||
from litellm import ModelResponse
|
||||
|
||||
# Create a mock ModelResponse
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
reasoning_tokens=10,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1),
|
||||
)
|
||||
usage.citation_tokens = 15
|
||||
|
||||
response = ModelResponse()
|
||||
response.usage = usage
|
||||
response.model = "sonar-deep-research"
|
||||
|
||||
# Test completion_cost function
|
||||
total_cost = completion_cost(
|
||||
completion_response=response, custom_llm_provider="perplexity"
|
||||
)
|
||||
|
||||
# Calculate expected total cost (reasoning is a subset of completion_tokens)
|
||||
expected_prompt_cost = (100 * 2e-6) + (15 * 2e-6) # Input + citation
|
||||
expected_completion_cost = (
|
||||
((50 - 10) * 8e-6) + (10 * 3e-6) + (1 * 0.005)
|
||||
) # Output (text) + reasoning + search
|
||||
expected_total = expected_prompt_cost + expected_completion_cost
|
||||
|
||||
assert math.isclose(total_cost, expected_total, rel_tol=1e-6)
|
||||
|
||||
@pytest.mark.parametrize("citation_tokens", [0, 10, 25, 100])
|
||||
@pytest.mark.parametrize("search_queries", [0, 1, 5, 10])
|
||||
@pytest.mark.parametrize("reasoning_tokens", [0, 15, 30])
|
||||
def test_cost_calculation_combinations(
|
||||
self, citation_tokens, search_queries, reasoning_tokens
|
||||
):
|
||||
"""Test various combinations of citation tokens, search queries, and reasoning tokens."""
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
reasoning_tokens=reasoning_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
web_search_requests=search_queries
|
||||
),
|
||||
)
|
||||
|
||||
usage.citation_tokens = citation_tokens
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
# Calculate expected costs. `completion_tokens` includes `reasoning_tokens`,
|
||||
# so non-reasoning portion = 50 - reasoning_tokens.
|
||||
expected_prompt_cost = (100 * 2e-6) + (citation_tokens * 2e-6)
|
||||
expected_completion_cost = (
|
||||
((50 - reasoning_tokens) * 8e-6)
|
||||
+ (reasoning_tokens * 3e-6)
|
||||
+ (search_queries * 0.005)
|
||||
)
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
# Ensure costs are non-negative
|
||||
assert prompt_cost >= 0
|
||||
assert completion_cost >= 0
|
||||
|
||||
def test_uses_perplexity_provided_cost_when_available(self):
|
||||
"""
|
||||
Test that when Perplexity provides pre-calculated cost in usage.cost.total_cost,
|
||||
|
|
@ -374,9 +109,7 @@ class TestPerplexityCostCalculator:
|
|||
"total_cost": 0.008,
|
||||
}
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-pro", usage=usage
|
||||
)
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(model="sonar-pro", usage=usage)
|
||||
|
||||
# When Perplexity provides total_cost, we use it directly
|
||||
# prompt_cost should be 0, completion_cost should be total_cost
|
||||
|
|
@ -402,9 +135,7 @@ class TestPerplexityCostCalculator:
|
|||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
usage.cost = 0.008
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-pro", usage=usage
|
||||
)
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(model="sonar-pro", usage=usage)
|
||||
|
||||
assert prompt_cost == 0.0
|
||||
assert completion_cost == 0.008
|
||||
|
|
@ -417,9 +148,7 @@ class TestPerplexityCostCalculator:
|
|||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
# No cost object - should use manual calculation
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(model="sonar-deep-research", usage=usage)
|
||||
|
||||
# Should calculate manually: 100 * 2e-6 + 50 * 8e-6
|
||||
expected_prompt = 100 * 2e-6
|
||||
|
|
@ -428,57 +157,6 @@ class TestPerplexityCostCalculator:
|
|||
assert math.isclose(prompt_cost, expected_prompt, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion, rel_tol=1e-6)
|
||||
|
||||
def test_reasoning_tokens_not_double_billed(self):
|
||||
"""
|
||||
Regression: `completion_tokens` includes `reasoning_tokens` per the
|
||||
OpenAI/Perplexity usage convention (codified for the central path in PR #18607).
|
||||
When `output_cost_per_reasoning_token` is configured the manual fallback must
|
||||
subtract reasoning from completion before applying the output rate so the
|
||||
reasoning tokens are not billed at BOTH the output rate and the reasoning rate.
|
||||
|
||||
Uses the exact usage shape produced by the live response fixture in
|
||||
`tests/llm_translation/test_perplexity_reasoning.py`.
|
||||
"""
|
||||
usage = Usage(
|
||||
prompt_tokens=9,
|
||||
completion_tokens=20,
|
||||
total_tokens=29,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
reasoning_tokens=15
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(
|
||||
model="sonar-deep-research", usage=usage
|
||||
)
|
||||
|
||||
# sonar-deep-research rates: input 2e-6, output 8e-6, reasoning 3e-6.
|
||||
# Non-reasoning portion of the 20 completion tokens = 20 - 15 = 5.
|
||||
# Pre-fix this asserted 20 * 8e-6 + 15 * 3e-6 = 2.05e-4 (a 2.16x overcharge).
|
||||
expected_prompt = 9 * 2e-6
|
||||
expected_completion = (20 - 15) * 8e-6 + 15 * 3e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt, rel_tol=1e-9)
|
||||
assert math.isclose(completion_cost, expected_completion, rel_tol=1e-9)
|
||||
|
||||
def test_agent_api_fallback_rates_price_a_response_without_metered_cost(self):
|
||||
"""Perplexity meters cost on the response, but when `usage.cost` is absent the
|
||||
calculator falls back to the mapped per-token rates. Regression: that fallback
|
||||
raised "This model isn't mapped yet" for every Agent API third-party model,
|
||||
because the doubled cost-map key was unreachable from the resolution ladder.
|
||||
"""
|
||||
from litellm import ModelResponse
|
||||
|
||||
response = ModelResponse()
|
||||
response.usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
|
||||
response.model = "perplexity/perplexity/glm-5.2"
|
||||
|
||||
total_cost = completion_cost(
|
||||
completion_response=response, custom_llm_provider="perplexity"
|
||||
)
|
||||
|
||||
assert math.isclose(total_cost, 1000 * 1.4e-06 + 500 * 4.4e-06, rel_tol=1e-9)
|
||||
|
||||
OFF_PEAK_MODEL = "sonar-off-peak-test"
|
||||
OFF_PEAK_WINDOW = "14:00-00:00"
|
||||
INSIDE_WINDOW = datetime(2026, 9, 3, 17, 25, tzinfo=timezone.utc)
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""
|
||||
Integration tests for Perplexity cost calculation and transformation.
|
||||
|
||||
Tests the end-to-end functionality of Perplexity cost calculation
|
||||
Tests the end-to-end functionality of Perplexity cost calculation
|
||||
including integration with the main LiteLLM cost calculator.
|
||||
"""
|
||||
|
||||
|
|
@ -12,10 +12,9 @@ import os
|
|||
import pytest
|
||||
|
||||
# Add the project root to Python path
|
||||
|
||||
import litellm
|
||||
from litellm import ModelResponse
|
||||
from litellm.cost_calculator import completion_cost, cost_per_token
|
||||
from litellm.cost_calculator import cost_per_token
|
||||
from litellm.llms.perplexity.chat.transformation import PerplexityChatConfig
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
from litellm.utils import get_model_info
|
||||
|
|
@ -57,109 +56,9 @@ class TestPerplexityIntegration:
|
|||
}
|
||||
}
|
||||
|
||||
def test_end_to_end_cost_calculation_with_transformation(self):
|
||||
"""Test end-to-end cost calculation with response transformation."""
|
||||
# Create a Perplexity API response that includes citations and search queries
|
||||
config = PerplexityChatConfig()
|
||||
|
||||
# Create a ModelResponse with basic usage (before transformation)
|
||||
model_response = ModelResponse()
|
||||
model_response.model = "sonar-deep-research"
|
||||
model_response.usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=50,
|
||||
total_tokens=150,
|
||||
reasoning_tokens=10,
|
||||
)
|
||||
|
||||
# Simulate raw response from Perplexity API
|
||||
raw_response_dict = {
|
||||
"choices": [{"message": {"content": "Test response with citations"}}],
|
||||
"usage": {
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 50,
|
||||
"total_tokens": 150,
|
||||
"num_search_queries": 2,
|
||||
},
|
||||
"citations": [
|
||||
"This is the first citation with important information about the topic",
|
||||
"Another citation providing additional context for the response",
|
||||
],
|
||||
}
|
||||
|
||||
# Apply transformation to extract Perplexity-specific fields
|
||||
config._enhance_usage_with_perplexity_fields(model_response, raw_response_dict)
|
||||
|
||||
# Now calculate the cost with the enhanced usage
|
||||
total_cost = completion_cost(
|
||||
completion_response=model_response, custom_llm_provider="perplexity"
|
||||
)
|
||||
|
||||
# Calculate expected cost
|
||||
citation_chars = sum(
|
||||
len(citation) for citation in raw_response_dict["citations"]
|
||||
)
|
||||
citation_tokens = citation_chars // 4
|
||||
|
||||
expected_prompt_cost = (100 * 2e-6) + (citation_tokens * 2e-6)
|
||||
expected_completion_cost = (
|
||||
((50 - 10) * 8e-6) + (10 * 3e-6) + (2 * 0.005)
|
||||
) # Output (text) + reasoning + search
|
||||
expected_total = expected_prompt_cost + expected_completion_cost
|
||||
|
||||
assert math.isclose(total_cost, expected_total, rel_tol=1e-6)
|
||||
|
||||
def test_cost_calculation_without_custom_fields(self):
|
||||
"""Test that cost calculation works normally when custom fields are absent."""
|
||||
# Create a standard response without Perplexity-specific fields
|
||||
model_response = ModelResponse()
|
||||
model_response.model = "sonar-deep-research"
|
||||
model_response.usage = Usage(
|
||||
prompt_tokens=100, completion_tokens=50, total_tokens=150
|
||||
)
|
||||
|
||||
# Calculate cost without custom fields
|
||||
total_cost = completion_cost(
|
||||
completion_response=model_response, custom_llm_provider="perplexity"
|
||||
)
|
||||
|
||||
# Should only include basic input/output costs
|
||||
expected_cost = (100 * 2e-6) + (50 * 8e-6)
|
||||
|
||||
assert math.isclose(total_cost, expected_cost, rel_tol=1e-6)
|
||||
|
||||
def test_main_cost_calculator_integration(self):
|
||||
"""Test integration with the main LiteLLM cost calculator."""
|
||||
# Create usage with all Perplexity fields
|
||||
usage = Usage(
|
||||
prompt_tokens=200,
|
||||
completion_tokens=100,
|
||||
total_tokens=300,
|
||||
reasoning_tokens=25,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=3),
|
||||
)
|
||||
usage.citation_tokens = 40
|
||||
|
||||
# Test main cost calculator
|
||||
prompt_cost, completion_cost_val = cost_per_token(
|
||||
model="sonar-deep-research",
|
||||
custom_llm_provider="perplexity",
|
||||
usage_object=usage,
|
||||
)
|
||||
|
||||
expected_prompt_cost = (200 * 2e-6) + (40 * 2e-6)
|
||||
expected_completion_cost = (
|
||||
((100 - 25) * 8e-6) + (25 * 3e-6) + (3 * 0.005)
|
||||
) # Output (text) + reasoning + search
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost_val, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_model_info_includes_custom_fields(self):
|
||||
"""Test that get_model_info returns the custom Perplexity cost fields."""
|
||||
model_info = get_model_info(
|
||||
model="sonar-deep-research", custom_llm_provider="perplexity"
|
||||
)
|
||||
model_info = get_model_info(model="sonar-deep-research", custom_llm_provider="perplexity")
|
||||
|
||||
# Verify custom fields are included
|
||||
required_fields = [
|
||||
|
|
@ -192,9 +91,7 @@ class TestPerplexityIntegration:
|
|||
for citations, expected_approx_tokens in test_cases:
|
||||
model_response = ModelResponse()
|
||||
model_response.model = "sonar-deep-research"
|
||||
model_response.usage = Usage(
|
||||
prompt_tokens=100, completion_tokens=50, total_tokens=150
|
||||
)
|
||||
model_response.usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
|
||||
raw_response_dict = {
|
||||
"usage": {
|
||||
|
|
@ -205,9 +102,7 @@ class TestPerplexityIntegration:
|
|||
"citations": citations,
|
||||
}
|
||||
|
||||
config._enhance_usage_with_perplexity_fields(
|
||||
model_response, raw_response_dict
|
||||
)
|
||||
config._enhance_usage_with_perplexity_fields(model_response, raw_response_dict)
|
||||
|
||||
citation_tokens = getattr(model_response.usage, "citation_tokens", 0)
|
||||
|
||||
|
|
@ -217,55 +112,6 @@ class TestPerplexityIntegration:
|
|||
else:
|
||||
assert abs(citation_tokens - expected_approx_tokens) <= 5
|
||||
|
||||
def test_cost_calculation_with_zero_values(self):
|
||||
"""Test cost calculation handles zero values for custom fields correctly."""
|
||||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
|
||||
# Set custom fields to zero
|
||||
usage.citation_tokens = 0
|
||||
usage.prompt_tokens_details = PromptTokensDetailsWrapper(web_search_requests=0)
|
||||
|
||||
# Should not add any extra cost
|
||||
prompt_cost, completion_cost_val = cost_per_token(
|
||||
model="sonar-deep-research",
|
||||
custom_llm_provider="perplexity",
|
||||
usage_object=usage,
|
||||
)
|
||||
|
||||
expected_prompt_cost = 100 * 2e-6
|
||||
expected_completion_cost = 50 * 8e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost_val, expected_completion_cost, rel_tol=1e-6)
|
||||
|
||||
def test_high_volume_cost_calculation(self):
|
||||
"""Test cost calculation with high token and query counts."""
|
||||
usage = Usage(
|
||||
prompt_tokens=50000,
|
||||
completion_tokens=25000,
|
||||
total_tokens=75000,
|
||||
reasoning_tokens=10000,
|
||||
)
|
||||
|
||||
usage.citation_tokens = 5000
|
||||
usage.prompt_tokens_details = PromptTokensDetailsWrapper(
|
||||
web_search_requests=100
|
||||
)
|
||||
|
||||
total_cost = completion_cost(
|
||||
completion_response=ModelResponse(usage=usage, model="sonar-deep-research"),
|
||||
custom_llm_provider="perplexity",
|
||||
)
|
||||
|
||||
expected_prompt_cost = (50000 * 2e-6) + (5000 * 2e-6)
|
||||
expected_completion_cost = (
|
||||
((25000 - 10000) * 8e-6) + (10000 * 3e-6) + (100 * 0.005)
|
||||
) # $0.65
|
||||
expected_total = expected_prompt_cost + expected_completion_cost # $0.76
|
||||
|
||||
assert math.isclose(total_cost, expected_total, rel_tol=1e-6)
|
||||
assert total_cost > 0.25
|
||||
|
||||
def test_transformation_preserves_existing_usage_fields(self):
|
||||
"""Test that transformation doesn't overwrite existing standard usage fields."""
|
||||
config = PerplexityChatConfig()
|
||||
|
|
@ -305,9 +151,7 @@ class TestPerplexityIntegration:
|
|||
assert hasattr(model_response.usage, "citation_tokens")
|
||||
assert model_response.usage.prompt_tokens_details.web_search_requests == 3
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"provider_name", ["perplexity", "PERPLEXITY", "Perplexity"]
|
||||
)
|
||||
@pytest.mark.parametrize("provider_name", ["perplexity", "PERPLEXITY", "Perplexity"])
|
||||
def test_case_insensitive_provider_matching(self, provider_name):
|
||||
"""Test that cost calculation works with different case variations of provider name."""
|
||||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
|
|
|
|||
|
|
@ -1,29 +0,0 @@
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.tencent.cost_calculator import cost_per_token
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
|
||||
|
||||
def test_cost_per_token_uses_tencent_model_pricing(local_model_cost_map):
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=2000, total_tokens=3000)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="tencent/deepseek-v4-pro", usage=usage)
|
||||
|
||||
assert prompt_cost == pytest.approx(1000 * 4.35e-07)
|
||||
assert completion_cost == pytest.approx(2000 * 8.7e-07)
|
||||
|
||||
|
||||
def test_top_level_dispatcher_routes_tencent_to_wrapper(local_model_cost_map):
|
||||
from litellm.cost_calculator import cost_per_token as dispatch_cost_per_token
|
||||
|
||||
prompt_cost, completion_cost = dispatch_cost_per_token(
|
||||
model="tencent/deepseek-v4-pro",
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=1000,
|
||||
custom_llm_provider="tencent",
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(1000 * 4.35e-07)
|
||||
assert completion_cost == pytest.approx(1000 * 8.7e-07)
|
||||
|
|
@ -3,8 +3,6 @@ import json
|
|||
import os
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.experimental_pass_through.transformation import (
|
||||
VertexAIPartnerModelsAnthropicMessagesConfig,
|
||||
)
|
||||
|
|
@ -23,12 +21,8 @@ def test_validate_environment_uses_vertex_ai_location():
|
|||
optional_params = {}
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("token", "test-project")
|
||||
),
|
||||
patch.object(
|
||||
config, "get_complete_vertex_url", return_value="https://mock-url"
|
||||
) as mock_get_url,
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-url") as mock_get_url,
|
||||
):
|
||||
config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
|
|
@ -51,17 +45,11 @@ def test_web_search_header_added_for_messages_endpoint():
|
|||
"vertex_credentials": "{}",
|
||||
}
|
||||
# Include web search tool in optional_params
|
||||
optional_params = {
|
||||
"tools": [{"type": "web_search_20250305", "name": "web_search", "max_uses": 5}]
|
||||
}
|
||||
optional_params = {"tools": [{"type": "web_search_20250305", "name": "web_search", "max_uses": 5}]}
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("token", "test-project")
|
||||
),
|
||||
patch.object(
|
||||
config, "get_complete_vertex_url", return_value="https://mock-url"
|
||||
),
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-url"),
|
||||
):
|
||||
updated_headers, api_base = config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
|
|
@ -73,12 +61,10 @@ def test_web_search_header_added_for_messages_endpoint():
|
|||
)
|
||||
|
||||
# Assert that the anthropic-beta header with web-search is present
|
||||
assert (
|
||||
"anthropic-beta" in updated_headers
|
||||
), "anthropic-beta header should be present"
|
||||
assert (
|
||||
updated_headers["anthropic-beta"] == "web-search-2025-03-05"
|
||||
), f"anthropic-beta should be 'web-search-2025-03-05', got: {updated_headers['anthropic-beta']}"
|
||||
assert "anthropic-beta" in updated_headers, "anthropic-beta header should be present"
|
||||
assert updated_headers["anthropic-beta"] == "web-search-2025-03-05", (
|
||||
f"anthropic-beta should be 'web-search-2025-03-05', got: {updated_headers['anthropic-beta']}"
|
||||
)
|
||||
|
||||
|
||||
def test_web_search_header_not_added_without_tool():
|
||||
|
|
@ -94,12 +80,8 @@ def test_web_search_header_not_added_without_tool():
|
|||
optional_params = {}
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("token", "test-project")
|
||||
),
|
||||
patch.object(
|
||||
config, "get_complete_vertex_url", return_value="https://mock-url"
|
||||
),
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-url"),
|
||||
):
|
||||
updated_headers, api_base = config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
|
|
@ -111,9 +93,9 @@ def test_web_search_header_not_added_without_tool():
|
|||
)
|
||||
|
||||
# Assert that the anthropic-beta header is NOT present when no web search tool
|
||||
assert (
|
||||
"anthropic-beta" not in updated_headers
|
||||
), "anthropic-beta header should not be present without web search tool"
|
||||
assert "anthropic-beta" not in updated_headers, (
|
||||
"anthropic-beta header should not be present without web search tool"
|
||||
)
|
||||
|
||||
|
||||
def test_compact_context_management_header_added():
|
||||
|
|
@ -129,12 +111,8 @@ def test_compact_context_management_header_added():
|
|||
optional_params = {"context_management": {"edits": [{"type": "compact_20260112"}]}}
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("token", "test-project")
|
||||
),
|
||||
patch.object(
|
||||
config, "get_complete_vertex_url", return_value="https://mock-url"
|
||||
),
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-url"),
|
||||
):
|
||||
updated_headers, api_base = config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
|
|
@ -146,12 +124,10 @@ def test_compact_context_management_header_added():
|
|||
)
|
||||
|
||||
# Assert that the anthropic-beta header with compact-2026-01-12 is present
|
||||
assert (
|
||||
"anthropic-beta" in updated_headers
|
||||
), "anthropic-beta header should be present"
|
||||
assert (
|
||||
"compact-2026-01-12" in updated_headers["anthropic-beta"]
|
||||
), f"anthropic-beta should contain 'compact-2026-01-12', got: {updated_headers['anthropic-beta']}"
|
||||
assert "anthropic-beta" in updated_headers, "anthropic-beta header should be present"
|
||||
assert "compact-2026-01-12" in updated_headers["anthropic-beta"], (
|
||||
f"anthropic-beta should contain 'compact-2026-01-12', got: {updated_headers['anthropic-beta']}"
|
||||
)
|
||||
|
||||
|
||||
def test_context_management_header_added_for_other_edits():
|
||||
|
|
@ -167,12 +143,8 @@ def test_context_management_header_added_for_other_edits():
|
|||
optional_params = {"context_management": {"edits": [{"type": "some_other_type"}]}}
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("token", "test-project")
|
||||
),
|
||||
patch.object(
|
||||
config, "get_complete_vertex_url", return_value="https://mock-url"
|
||||
),
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-url"),
|
||||
):
|
||||
updated_headers, api_base = config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
|
|
@ -184,12 +156,10 @@ def test_context_management_header_added_for_other_edits():
|
|||
)
|
||||
|
||||
# Assert that the anthropic-beta header with context-management-2025-06-27 is present
|
||||
assert (
|
||||
"anthropic-beta" in updated_headers
|
||||
), "anthropic-beta header should be present"
|
||||
assert (
|
||||
"context-management-2025-06-27" in updated_headers["anthropic-beta"]
|
||||
), f"anthropic-beta should contain 'context-management-2025-06-27', got: {updated_headers['anthropic-beta']}"
|
||||
assert "anthropic-beta" in updated_headers, "anthropic-beta header should be present"
|
||||
assert "context-management-2025-06-27" in updated_headers["anthropic-beta"], (
|
||||
f"anthropic-beta should contain 'context-management-2025-06-27', got: {updated_headers['anthropic-beta']}"
|
||||
)
|
||||
|
||||
|
||||
def test_both_compact_and_context_management_headers_added():
|
||||
|
|
@ -202,19 +172,11 @@ def test_both_compact_and_context_management_headers_added():
|
|||
"vertex_credentials": "{}",
|
||||
}
|
||||
# Include context_management with both compact and other edit types
|
||||
optional_params = {
|
||||
"context_management": {
|
||||
"edits": [{"type": "compact_20260112"}, {"type": "some_other_type"}]
|
||||
}
|
||||
}
|
||||
optional_params = {"context_management": {"edits": [{"type": "compact_20260112"}, {"type": "some_other_type"}]}}
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("token", "test-project")
|
||||
),
|
||||
patch.object(
|
||||
config, "get_complete_vertex_url", return_value="https://mock-url"
|
||||
),
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-url"),
|
||||
):
|
||||
updated_headers, api_base = config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
|
|
@ -226,15 +188,13 @@ def test_both_compact_and_context_management_headers_added():
|
|||
)
|
||||
|
||||
# Assert that both beta headers are present
|
||||
assert (
|
||||
"anthropic-beta" in updated_headers
|
||||
), "anthropic-beta header should be present"
|
||||
assert (
|
||||
"compact-2026-01-12" in updated_headers["anthropic-beta"]
|
||||
), f"anthropic-beta should contain 'compact-2026-01-12', got: {updated_headers['anthropic-beta']}"
|
||||
assert (
|
||||
"context-management-2025-06-27" in updated_headers["anthropic-beta"]
|
||||
), f"anthropic-beta should contain 'context-management-2025-06-27', got: {updated_headers['anthropic-beta']}"
|
||||
assert "anthropic-beta" in updated_headers, "anthropic-beta header should be present"
|
||||
assert "compact-2026-01-12" in updated_headers["anthropic-beta"], (
|
||||
f"anthropic-beta should contain 'compact-2026-01-12', got: {updated_headers['anthropic-beta']}"
|
||||
)
|
||||
assert "context-management-2025-06-27" in updated_headers["anthropic-beta"], (
|
||||
f"anthropic-beta should contain 'context-management-2025-06-27', got: {updated_headers['anthropic-beta']}"
|
||||
)
|
||||
|
||||
|
||||
def test_validate_environment_always_refreshes_token_ignoring_stale_bearer():
|
||||
|
|
@ -248,12 +208,8 @@ def test_validate_environment_always_refreshes_token_ignoring_stale_bearer():
|
|||
}
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("fresh-token", "test-project")
|
||||
) as mock_ensure,
|
||||
patch.object(
|
||||
config, "get_complete_vertex_url", return_value="https://mock-vertex-url"
|
||||
),
|
||||
patch.object(config, "_ensure_access_token", return_value=("fresh-token", "test-project")) as mock_ensure,
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-vertex-url"),
|
||||
):
|
||||
updated_headers, api_base = config.validate_anthropic_messages_environment(
|
||||
headers=headers,
|
||||
|
|
@ -286,9 +242,7 @@ def test_validate_environment_appends_stream_raw_predict_with_custom_api_base():
|
|||
"get_complete_vertex_url",
|
||||
wraps=config.get_complete_vertex_url,
|
||||
) as spy_get_url,
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("token", "test-project")
|
||||
),
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
):
|
||||
_, api_base = config.validate_anthropic_messages_environment(
|
||||
headers={},
|
||||
|
|
@ -318,9 +272,7 @@ def test_validate_environment_appends_raw_predict_with_custom_api_base():
|
|||
"get_complete_vertex_url",
|
||||
wraps=config.get_complete_vertex_url,
|
||||
) as spy_get_url,
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("token", "test-project")
|
||||
),
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
):
|
||||
_, api_base = config.validate_anthropic_messages_environment(
|
||||
headers={},
|
||||
|
|
@ -447,20 +399,14 @@ def test_validate_environment_does_not_mutate_caller_headers():
|
|||
caller_headers: dict = {}
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
config, "_ensure_access_token", return_value=("token", "test-project")
|
||||
),
|
||||
patch.object(
|
||||
config, "get_complete_vertex_url", return_value="https://mock-url"
|
||||
),
|
||||
patch.object(config, "_ensure_access_token", return_value=("token", "test-project")),
|
||||
patch.object(config, "get_complete_vertex_url", return_value="https://mock-url"),
|
||||
):
|
||||
config.validate_anthropic_messages_environment(
|
||||
headers=caller_headers,
|
||||
model="claude-sonnet-4",
|
||||
messages=[],
|
||||
optional_params={
|
||||
"tools": [{"type": "web_search_20250305", "name": "web_search"}]
|
||||
},
|
||||
optional_params={"tools": [{"type": "web_search_20250305", "name": "web_search"}]},
|
||||
litellm_params={
|
||||
"vertex_ai_project": "p",
|
||||
"vertex_ai_location": "us-central1",
|
||||
|
|
@ -468,9 +414,7 @@ def test_validate_environment_does_not_mutate_caller_headers():
|
|||
api_base=None,
|
||||
)
|
||||
|
||||
assert (
|
||||
caller_headers == {}
|
||||
), "validate_anthropic_messages_environment must not mutate the caller's headers dict"
|
||||
assert caller_headers == {}, "validate_anthropic_messages_environment must not mutate the caller's headers dict"
|
||||
|
||||
|
||||
def test_vertex_claude_completion_does_not_mutate_shared_extra_headers():
|
||||
|
|
@ -483,12 +427,8 @@ def test_vertex_claude_completion_does_not_mutate_shared_extra_headers():
|
|||
mock_response = MagicMock()
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
handler, "_ensure_access_token", return_value=("ya29.fresh", "proj")
|
||||
),
|
||||
patch.object(
|
||||
handler, "get_complete_vertex_url", return_value="https://mock-url"
|
||||
),
|
||||
patch.object(handler, "_ensure_access_token", return_value=("ya29.fresh", "proj")),
|
||||
patch.object(handler, "get_complete_vertex_url", return_value="https://mock-url"),
|
||||
patch(
|
||||
"litellm.llms.anthropic.chat.AnthropicChatCompletion.completion",
|
||||
return_value=mock_response,
|
||||
|
|
@ -509,10 +449,7 @@ def test_vertex_claude_completion_does_not_mutate_shared_extra_headers():
|
|||
litellm_params={},
|
||||
)
|
||||
|
||||
assert (
|
||||
shared_extra_headers == {}
|
||||
), "extra_headers must not be mutated by completion()"
|
||||
|
||||
assert shared_extra_headers == {}, "extra_headers must not be mutated by completion()"
|
||||
|
||||
|
||||
def test_messages_thinking_shape_follows_exact_vertex_entry_flag(local_model_cost_map, monkeypatch):
|
||||
|
|
@ -541,9 +478,7 @@ def test_messages_thinking_shape_follows_exact_vertex_entry_flag(local_model_cos
|
|||
assert result.get("thinking") == {"type": "adaptive", "display": "summarized"}
|
||||
assert result.get("output_config") == {"effort": "medium"}
|
||||
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost["vertex_ai/claude-opus-4-8"], "supports_adaptive_thinking", False
|
||||
)
|
||||
monkeypatch.setitem(litellm.model_cost["vertex_ai/claude-opus-4-8"], "supports_adaptive_thinking", False)
|
||||
litellm.get_model_info.cache_clear()
|
||||
assert litellm.model_cost["claude-opus-4-8"]["supports_adaptive_thinking"] is True
|
||||
|
||||
|
|
@ -614,9 +549,7 @@ class TestVertexAnthropicMidConversationSystem:
|
|||
{"role": "assistant", "content": "reading"},
|
||||
{"role": "user", "content": "continue"},
|
||||
]
|
||||
result = _vertex_transform(
|
||||
"claude-sonnet-4-6", messages, system=[{"type": "text", "text": "Base."}]
|
||||
)
|
||||
result = _vertex_transform("claude-sonnet-4-6", messages, system=[{"type": "text", "text": "Base."}])
|
||||
assert result["messages"] == [
|
||||
{"role": "user", "content": "read the file"},
|
||||
{
|
||||
|
|
@ -660,9 +593,7 @@ def test_vertex_claude_4_8_plus_cost_map_entries_carry_mid_conversation_system_f
|
|||
|
||||
import litellm
|
||||
|
||||
cost_map_path = os.path.join(
|
||||
os.path.dirname(litellm.__file__), "model_prices_and_context_window_backup.json"
|
||||
)
|
||||
cost_map_path = os.path.join(os.path.dirname(litellm.__file__), "model_prices_and_context_window_backup.json")
|
||||
with open(cost_map_path) as f:
|
||||
cost_map = json.load(f)
|
||||
rules = cost_map["fallback_generalizations"]["rules"]
|
||||
|
|
|
|||
|
|
@ -10,9 +10,7 @@ Source: litellm/llms/xai/responses/transformation.py
|
|||
from unittest.mock import MagicMock, Mock
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.xai.cost_calculator import cost_per_token
|
||||
from litellm.llms.xai.responses.transformation import XAIResponsesAPIConfig
|
||||
from litellm.responses.utils import ResponseAPILoggingUtils
|
||||
|
|
@ -366,12 +364,16 @@ class TestXAIResponsesWebSearchBilling:
|
|||
|
||||
def _raw_response_json(self, include_web_search: bool) -> dict:
|
||||
web_search_output = (
|
||||
[{
|
||||
"type": "web_search_call",
|
||||
"id": "ws_1",
|
||||
"status": "completed",
|
||||
"action": {"type": "search", "query": "grok"},
|
||||
}] if include_web_search else []
|
||||
[
|
||||
{
|
||||
"type": "web_search_call",
|
||||
"id": "ws_1",
|
||||
"status": "completed",
|
||||
"action": {"type": "search", "query": "grok"},
|
||||
}
|
||||
]
|
||||
if include_web_search
|
||||
else []
|
||||
)
|
||||
tool_usage = {"server_side_tool_usage_details": self._TOOL_DETAILS} if include_web_search else {}
|
||||
return {
|
||||
|
|
@ -431,20 +433,6 @@ class TestXAIResponsesWebSearchBilling:
|
|||
assert bridged.completion_tokens == 20
|
||||
assert getattr(bridged, "server_side_tool_usage_details") == self._TOOL_DETAILS
|
||||
|
||||
def test_completion_cost_bills_web_search_calls(self):
|
||||
with_search = litellm.completion_cost(
|
||||
completion_response=self._transform(include_web_search=True),
|
||||
model="xai/grok-4",
|
||||
custom_llm_provider="xai",
|
||||
)
|
||||
without_search = litellm.completion_cost(
|
||||
completion_response=self._transform(include_web_search=False),
|
||||
model="xai/grok-4",
|
||||
custom_llm_provider="xai",
|
||||
)
|
||||
|
||||
assert with_search - without_search == pytest.approx(2 * 5.0 / 1000.0)
|
||||
|
||||
def test_streaming_terminal_event_keeps_schema_and_details(self):
|
||||
parsed_chunk = {
|
||||
"type": "response.completed",
|
||||
|
|
@ -535,9 +523,7 @@ class TestXAIResponsesReportedCost:
|
|||
assert cost_per_token(model="grok-4-latest", usage=chat_usage) == (0.0, 0.0037756)
|
||||
|
||||
def test_usage_without_a_reported_cost_is_left_alone(self):
|
||||
usage = self._transformed_usage(
|
||||
{"input_tokens": 100, "output_tokens": 200, "total_tokens": 300}
|
||||
)
|
||||
usage = self._transformed_usage({"input_tokens": 100, "output_tokens": 200, "total_tokens": 300})
|
||||
|
||||
assert usage.cost is None
|
||||
|
||||
|
|
|
|||
|
|
@ -1,7 +1,6 @@
|
|||
from unittest.mock import Mock
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.xai.chat.transformation import (
|
||||
|
|
@ -26,11 +25,7 @@ class TestXAIReasoningTokenFolding:
|
|||
total_tokens: int,
|
||||
reasoning_tokens: int = 0,
|
||||
) -> ModelResponse:
|
||||
details = (
|
||||
CompletionTokensDetailsWrapper(reasoning_tokens=reasoning_tokens)
|
||||
if reasoning_tokens
|
||||
else None
|
||||
)
|
||||
details = CompletionTokensDetailsWrapper(reasoning_tokens=reasoning_tokens) if reasoning_tokens else None
|
||||
usage = Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
|
|
@ -194,31 +189,11 @@ class TestXAIChatWebSearchBilling:
|
|||
def test_enhance_noop_without_details(self):
|
||||
response = self._response_with_usage()
|
||||
|
||||
XAIChatConfig()._enhance_usage_with_xai_web_search_fields(
|
||||
response, {"usage": {"prompt_tokens": 100}}
|
||||
)
|
||||
XAIChatConfig()._enhance_usage_with_xai_web_search_fields(response, {"usage": {"prompt_tokens": 100}})
|
||||
|
||||
assert response.usage.prompt_tokens_details is None
|
||||
assert getattr(response.usage, "server_side_tool_usage_details", None) is None
|
||||
|
||||
def test_completion_cost_bills_chat_web_search_calls(self):
|
||||
billed = self._response_with_usage()
|
||||
XAIChatConfig()._enhance_usage_with_xai_web_search_fields(
|
||||
billed,
|
||||
{"usage": {"server_side_tool_usage_details": self._TOOL_DETAILS}},
|
||||
)
|
||||
|
||||
with_search = litellm.completion_cost(
|
||||
completion_response=billed, model="xai/grok-4", custom_llm_provider="xai"
|
||||
)
|
||||
without_search = litellm.completion_cost(
|
||||
completion_response=self._response_with_usage(),
|
||||
model="xai/grok-4",
|
||||
custom_llm_provider="xai",
|
||||
)
|
||||
|
||||
assert with_search - without_search == pytest.approx(3 * 5.0 / 1000.0)
|
||||
|
||||
|
||||
class TestXAIReportedCost:
|
||||
"""xAI reports what it charged; the transformation moves it to where litellm bills from.
|
||||
|
|
@ -275,9 +250,7 @@ class TestXAIReportedCost:
|
|||
assert cost_per_token(model="grok-4-latest", usage=usage) == (0.0, 0.0037756)
|
||||
|
||||
def test_usage_without_a_reported_cost_is_left_alone(self):
|
||||
usage = self._transformed_usage(
|
||||
{"prompt_tokens": 100, "completion_tokens": 200, "total_tokens": 300}
|
||||
)
|
||||
usage = self._transformed_usage({"prompt_tokens": 100, "completion_tokens": 200, "total_tokens": 300})
|
||||
|
||||
assert getattr(usage, "cost", None) is None
|
||||
|
||||
|
|
@ -300,9 +273,7 @@ class TestXAIReportedCost:
|
|||
Chunk aggregation rebuilds usage from the fields it models plus ``cost``, so a
|
||||
chunk still carrying only ``cost_in_usd_ticks`` loses the reported amount.
|
||||
"""
|
||||
handler = XAIChatCompletionStreamingHandler(
|
||||
streaming_response=iter([]), sync_stream=True
|
||||
)
|
||||
handler = XAIChatCompletionStreamingHandler(streaming_response=iter([]), sync_stream=True)
|
||||
|
||||
parsed = handler.chunk_parser(
|
||||
{
|
||||
|
|
|
|||
|
|
@ -6,16 +6,6 @@ import math
|
|||
import os
|
||||
|
||||
import litellm
|
||||
from litellm.types.utils import (
|
||||
Choices,
|
||||
CompletionTokensDetailsWrapper,
|
||||
Message,
|
||||
ModelResponse,
|
||||
PromptTokensDetailsWrapper,
|
||||
Usage,
|
||||
)
|
||||
|
||||
|
||||
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import (
|
||||
StandardBuiltInToolCostTracking,
|
||||
)
|
||||
|
|
@ -26,6 +16,13 @@ from litellm.llms.xai.cost_calculator import (
|
|||
cost_per_token,
|
||||
cost_per_web_search_request,
|
||||
)
|
||||
from litellm.types.utils import (
|
||||
Choices,
|
||||
Message,
|
||||
ModelResponse,
|
||||
PromptTokensDetailsWrapper,
|
||||
Usage,
|
||||
)
|
||||
|
||||
|
||||
class TestXAICostCalculator:
|
||||
|
|
@ -45,241 +42,6 @@ class TestXAICostCalculator:
|
|||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
def test_basic_cost_calculation(self):
|
||||
"""Test basic cost calculation without reasoning tokens."""
|
||||
usage = Usage(prompt_tokens=12, completion_tokens=125, total_tokens=137)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="grok-3-mini", usage=usage)
|
||||
|
||||
# Expected costs for grok-3-mini:
|
||||
# Input: 12 tokens * $3e-7 = $0.0000036
|
||||
# Output: 125 tokens * $5e-7 = $0.0000625
|
||||
expected_prompt_cost = 12 * 1.25e-6
|
||||
expected_completion_cost = 125 * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_reasoning_tokens_cost_calculation(self):
|
||||
"""Test cost calculation with reasoning tokens from completion_tokens_details."""
|
||||
usage = Usage(
|
||||
prompt_tokens=12,
|
||||
completion_tokens=125,
|
||||
total_tokens=1086,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=949,
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=None, # Not set, but doesn't matter for XAI billing
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="grok-3-mini", usage=usage)
|
||||
|
||||
# Expected costs for grok-3-mini:
|
||||
# Input: 12 tokens * $3e-7 = $0.0000036
|
||||
# Completion: (125 + 949) tokens * $5e-7 = $0.000537
|
||||
expected_prompt_cost = 12 * 1.25e-6
|
||||
expected_completion_cost = (125 + 949) * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_reasoning_and_text_tokens_cost_calculation(self):
|
||||
"""Test cost calculation with both reasoning and text tokens."""
|
||||
usage = Usage(
|
||||
prompt_tokens=12,
|
||||
completion_tokens=125,
|
||||
total_tokens=1086,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=949,
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=76, # Explicitly set (but ignored in XAI billing)
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="grok-3-mini", usage=usage)
|
||||
|
||||
# Expected costs for grok-3-mini:
|
||||
# Input: 12 tokens * $3e-7 = $0.0000036
|
||||
# Completion: (125 + 949) tokens * $5e-7 = $0.000537
|
||||
# Note: text_tokens field is ignored, only completion_tokens + reasoning_tokens matters
|
||||
expected_prompt_cost = 12 * 1.25e-6
|
||||
expected_completion_cost = (125 + 949) * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_grok_4_cost_calculation(self):
|
||||
"""Test cost calculation for grok-4 model."""
|
||||
usage = Usage(
|
||||
prompt_tokens=10,
|
||||
completion_tokens=200,
|
||||
total_tokens=360,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=150,
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=50, # Ignored in XAI billing
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="grok-4", usage=usage)
|
||||
|
||||
# grok-4 was retired on 2026-05-15 and now redirects to grok-4.3, so it bills
|
||||
# at grok-4.3's rates:
|
||||
# Input: 10 tokens * $1.25e-6
|
||||
# Completion: (200 + 150) tokens * $2.5e-6
|
||||
expected_prompt_cost = 10 * 1.25e-6
|
||||
expected_completion_cost = (200 + 150) * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_grok_3_fast_beta_cost_calculation(self):
|
||||
"""Test cost calculation for grok-3-fast-beta model."""
|
||||
usage = Usage(
|
||||
prompt_tokens=20,
|
||||
completion_tokens=300,
|
||||
total_tokens=520,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=200,
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=100, # Ignored in XAI billing
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="grok-3-fast-beta", usage=usage
|
||||
)
|
||||
|
||||
# Expected costs for grok-3-fast-beta:
|
||||
# Input: 20 tokens * $5e-6 = $0.0001
|
||||
# Completion: (300 + 200) tokens * $2.5e-5 = $0.0125
|
||||
expected_prompt_cost = 20 * 1.25e-6
|
||||
expected_completion_cost = (300 + 200) * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
|
||||
def test_edge_case_large_reasoning_tokens(self):
|
||||
"""Test cost calculation when reasoning_tokens is larger than completion_tokens."""
|
||||
usage = Usage(
|
||||
prompt_tokens=12,
|
||||
completion_tokens=50, # Less than reasoning_tokens
|
||||
total_tokens=162,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=100, # More than completion_tokens
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=None,
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="grok-3-mini", usage=usage)
|
||||
|
||||
# Expected costs:
|
||||
# Input: 12 tokens * $3e-7 = $0.0000036
|
||||
# Completion: (50 + 100) tokens * $5e-7 = $0.000075
|
||||
expected_prompt_cost = 12 * 1.25e-6
|
||||
expected_completion_cost = (50 + 100) * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_tiered_pricing_above_200k_tokens(self):
|
||||
usage = Usage(
|
||||
prompt_tokens=250000,
|
||||
completion_tokens=100000,
|
||||
total_tokens=400000,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=50000,
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=None,
|
||||
),
|
||||
)
|
||||
prompt_cost, completion_cost = cost_per_token(model="xai/grok-4.3", usage=usage)
|
||||
expected_prompt_cost = 250000 * 2.5e-6
|
||||
expected_completion_cost = (100000 + 50000) * 5e-6
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_tiered_pricing_below_200k_tokens(self):
|
||||
usage = Usage(
|
||||
prompt_tokens=100000,
|
||||
completion_tokens=50000,
|
||||
total_tokens=160000,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=10000,
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=None,
|
||||
),
|
||||
)
|
||||
prompt_cost, completion_cost = cost_per_token(model="xai/grok-4.3", usage=usage)
|
||||
expected_prompt_cost = 100000 * 1.25e-6
|
||||
expected_completion_cost = (50000 + 10000) * 2.5e-6
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_tiered_pricing_grok_4_latest(self):
|
||||
"""Test tiered pricing for grok-4-latest model."""
|
||||
usage = Usage(
|
||||
prompt_tokens=250000, # Above the 200k threshold
|
||||
completion_tokens=100000,
|
||||
total_tokens=400000,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=50000,
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=None,
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="xai/grok-4-latest", usage=usage
|
||||
)
|
||||
|
||||
# grok-4-latest redirects to grok-4.3, which tiers at 200k rather than 128k:
|
||||
# Input: 250000 tokens * $2.5e-6 (ALL tokens at tiered rate since input > 200k)
|
||||
# Completion: (100000 + 50000) tokens * $5e-6 (tiered rate since input > 200k)
|
||||
expected_prompt_cost = 250000 * 2.5e-6
|
||||
expected_completion_cost = (100000 + 50000) * 5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_tiered_pricing_output_tokens_below_200k(self):
|
||||
usage = Usage(
|
||||
prompt_tokens=250000,
|
||||
completion_tokens=50000,
|
||||
total_tokens=310000,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=10000,
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=None,
|
||||
),
|
||||
)
|
||||
prompt_cost, completion_cost = cost_per_token(model="xai/grok-4.3", usage=usage)
|
||||
expected_prompt_cost = 250000 * 2.5e-6
|
||||
expected_completion_cost = (50000 + 10000) * 5e-6
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_tiered_pricing_model_without_tiered_pricing(self):
|
||||
litellm.model_cost["xai/flat-rate-fixture"] = {
|
||||
"input_cost_per_token": 3e-7,
|
||||
|
|
@ -294,29 +56,6 @@ class TestXAICostCalculator:
|
|||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_already_normalised_usage_does_not_double_count_reasoning(self):
|
||||
"""Cost calc must not double-bill when Usage is already OpenAI-normalised."""
|
||||
usage = Usage(
|
||||
prompt_tokens=12,
|
||||
completion_tokens=200,
|
||||
total_tokens=212,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=0,
|
||||
audio_tokens=0,
|
||||
reasoning_tokens=100,
|
||||
rejected_prediction_tokens=0,
|
||||
text_tokens=None,
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="grok-3-mini", usage=usage)
|
||||
|
||||
expected_prompt_cost = 12 * 1.25e-6
|
||||
expected_completion_cost = 200 * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_web_search_cost_via_server_side_tool_usage_details(self):
|
||||
"""usage.server_side_tool_usage_details.web_search_calls at default $5/1k."""
|
||||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
|
|
@ -344,9 +83,7 @@ class TestXAICostCalculator:
|
|||
"search_context_size_medium": 0.01,
|
||||
}
|
||||
}
|
||||
web_search_cost = cost_per_web_search_request(
|
||||
usage=usage, model_info=model_info
|
||||
)
|
||||
web_search_cost = cost_per_web_search_request(usage=usage, model_info=model_info)
|
||||
assert math.isclose(web_search_cost, 0.02, rel_tol=1e-10)
|
||||
|
||||
def test_web_search_cost_zero_without_details(self):
|
||||
|
|
@ -355,9 +92,7 @@ class TestXAICostCalculator:
|
|||
|
||||
def test_apply_details_sets_web_search_requests_for_cost_gate(self):
|
||||
usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15)
|
||||
apply_server_side_tool_usage_details_to_usage(
|
||||
usage, {"web_search_calls": 2, "x_search_calls": 0}
|
||||
)
|
||||
apply_server_side_tool_usage_details_to_usage(usage, {"web_search_calls": 2, "x_search_calls": 0})
|
||||
assert usage.prompt_tokens_details is not None
|
||||
assert usage.prompt_tokens_details.web_search_requests == 2
|
||||
assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(
|
||||
|
|
@ -413,9 +148,7 @@ class TestXAICostCalculator:
|
|||
|
||||
assert get_cost_for_web_search_request("xai", usage, {}) > 0.0
|
||||
|
||||
reported = Usage(
|
||||
prompt_tokens=100, completion_tokens=50, total_tokens=150, cost=0.0037756
|
||||
)
|
||||
reported = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150, cost=0.0037756)
|
||||
setattr(reported, "server_side_tool_usage_details", {"web_search_calls": 3})
|
||||
assert get_cost_for_web_search_request("xai", reported, {}) == 0.0
|
||||
|
||||
|
|
@ -503,82 +236,6 @@ class TestXAICostCalculator:
|
|||
|
||||
assert cost_per_token(model="grok-4-latest", usage=usage) == (0.0, 0.0)
|
||||
|
||||
def test_grok_4_20_beta_reasoning_cost_calculation(self):
|
||||
"""Test cost calculation for grok-4.20-beta-0309-reasoning model."""
|
||||
usage = Usage(prompt_tokens=100, completion_tokens=200, total_tokens=300)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="grok-4.20-beta-0309-reasoning", usage=usage
|
||||
)
|
||||
|
||||
# Input: 100 tokens * $1.25e-6 = $0.000125
|
||||
# Output: 200 tokens * $2.5e-6 = $0.0005
|
||||
expected_prompt_cost = 100 * 1.25e-6
|
||||
expected_completion_cost = 200 * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_grok_4_20_beta_non_reasoning_cost_calculation(self):
|
||||
"""Test cost calculation for grok-4.20-beta-0309-non-reasoning model."""
|
||||
usage = Usage(prompt_tokens=50, completion_tokens=100, total_tokens=150)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="grok-4.20-beta-0309-non-reasoning", usage=usage
|
||||
)
|
||||
|
||||
# Input: 50 tokens * $1.25e-6 = $0.0000625
|
||||
# Output: 100 tokens * $2.5e-6 = $0.00025
|
||||
expected_prompt_cost = 50 * 1.25e-6
|
||||
expected_completion_cost = 100 * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_grok_4_20_at_exactly_200k_prompt_tokens_uses_higher_tier(self):
|
||||
"""xAI bills the >=200k tier once the prompt reaches 200k, so the boundary is inclusive."""
|
||||
usage = Usage(prompt_tokens=200_000, completion_tokens=1_000, total_tokens=201_000)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="grok-4.20-0309-reasoning", usage=usage
|
||||
)
|
||||
|
||||
expected_prompt_cost = 200_000 * 2.5e-6
|
||||
expected_completion_cost = 1_000 * 5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_grok_4_20_just_below_200k_prompt_tokens_uses_base_tier(self):
|
||||
"""One token under the boundary still bills at the base rates."""
|
||||
usage = Usage(prompt_tokens=199_999, completion_tokens=1_000, total_tokens=200_999)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="grok-4.20-0309-reasoning", usage=usage
|
||||
)
|
||||
|
||||
expected_prompt_cost = 199_999 * 1.25e-6
|
||||
expected_completion_cost = 1_000 * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_grok_4_20_multi_agent_cost_calculation(self):
|
||||
"""Test cost calculation for grok-4.20-multi-agent-beta-0309 model."""
|
||||
usage = Usage(prompt_tokens=200, completion_tokens=300, total_tokens=500)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="grok-4.20-multi-agent-beta-0309", usage=usage
|
||||
)
|
||||
|
||||
# Input: 200 tokens * $1.25e-6 = $0.00025
|
||||
# Output: 300 tokens * $2.5e-6 = $0.00075
|
||||
expected_prompt_cost = 200 * 1.25e-6
|
||||
expected_completion_cost = 300 * 2.5e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, expected_completion_cost, rel_tol=1e-10)
|
||||
|
||||
def test_custom_pricing_beats_the_reported_cost(self):
|
||||
response = ModelResponse(
|
||||
id="chatcmpl-xai",
|
||||
|
|
@ -635,10 +292,7 @@ class TestXAIWebSearchCostHelpers:
|
|||
details = {"web_search_calls": 0, "x_search_calls": 3}
|
||||
apply_server_side_tool_usage_details_to_usage(usage, details)
|
||||
assert getattr(usage, "server_side_tool_usage_details") == details
|
||||
assert (
|
||||
usage.prompt_tokens_details is None
|
||||
or usage.prompt_tokens_details.web_search_requests is None
|
||||
)
|
||||
assert usage.prompt_tokens_details is None or usage.prompt_tokens_details.web_search_requests is None
|
||||
|
||||
def test_apply_details_skips_mirror_when_web_search_calls_invalid(self):
|
||||
usage = Usage(prompt_tokens=1, completion_tokens=1, total_tokens=2)
|
||||
|
|
@ -660,10 +314,7 @@ class TestXAIWebSearchCostHelpers:
|
|||
assert usage.prompt_tokens_details.web_search_requests == 4
|
||||
|
||||
def test_web_search_cost_per_call_default_when_model_info_empty(self):
|
||||
assert (
|
||||
_web_search_cost_per_call_from_model_info({})
|
||||
== _DEFAULT_WEB_SEARCH_COST_PER_CALL
|
||||
)
|
||||
assert _web_search_cost_per_call_from_model_info({}) == _DEFAULT_WEB_SEARCH_COST_PER_CALL
|
||||
|
||||
def test_web_search_cost_per_call_prefers_medium_over_low(self):
|
||||
model_info = {
|
||||
|
|
|
|||
|
|
@ -13,19 +13,6 @@ REPO_ROOT = Path(__file__).parents[4]
|
|||
PRICES_PATH = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
BACKUP_PRICES_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
|
||||
# Retired by xAI and no longer served: requests to these slugs 404 rather than
|
||||
# redirecting, and they are absent from https://docs.x.ai/docs/models
|
||||
RETIRED_MODELS = (
|
||||
"xai/grok-2",
|
||||
"xai/grok-2-1212",
|
||||
"xai/grok-2-latest",
|
||||
"xai/grok-2-vision",
|
||||
"xai/grok-2-vision-1212",
|
||||
"xai/grok-2-vision-latest",
|
||||
"xai/grok-beta",
|
||||
"xai/grok-vision-beta",
|
||||
)
|
||||
|
||||
# https://docs.x.ai/developers/model-capabilities/text/multi-agent
|
||||
# "The multi-agent model does not work with the OpenAI Chat Completions API."
|
||||
RESPONSES_ONLY_MODELS = (
|
||||
|
|
@ -42,17 +29,11 @@ def cost_map(request: pytest.FixtureRequest) -> dict:
|
|||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", RETIRED_MODELS)
|
||||
def test_retired_xai_models_are_not_advertised(cost_map: dict, model: str):
|
||||
assert model not in cost_map
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", RESPONSES_ONLY_MODELS)
|
||||
def test_multi_agent_models_are_responses_only(cost_map: dict, model: str):
|
||||
entry = cost_map[model]
|
||||
assert entry["supported_endpoints"] == ["/v1/responses"]
|
||||
assert entry["mode"] == "responses"
|
||||
assert "/v1/chat/completions" not in entry["supported_endpoints"]
|
||||
|
||||
|
||||
def test_surviving_xai_chat_models_still_serve_chat_completions(cost_map: dict):
|
||||
|
|
@ -64,7 +45,6 @@ def test_surviving_xai_chat_models_still_serve_chat_completions(cost_map: dict):
|
|||
]
|
||||
assert "xai/grok-4.3" in chat_models
|
||||
assert "xai/grok-4.6" in chat_models
|
||||
assert not any(key.startswith("xai/grok-2") for key in chat_models)
|
||||
|
||||
|
||||
def test_both_cost_maps_agree_on_xai_entries():
|
||||
|
|
|
|||
|
|
@ -235,33 +235,6 @@ def test_negative_ttl_counts_do_not_become_cache_write_credits() -> None:
|
|||
assert results[0].prompt_caching < 0
|
||||
|
||||
|
||||
def test_unpublished_one_hour_price_uses_the_ordinary_write_price() -> None:
|
||||
model: Final = "claude-4-opus-20250514"
|
||||
pricing: Final = litellm.get_model_info(model=model, custom_llm_provider="anthropic")
|
||||
assert pricing.get("cache_creation_input_token_cost_above_1hr") is None
|
||||
assert pricing["cache_creation_input_token_cost"] > pricing["input_cost_per_token"]
|
||||
results: Final = tuple(
|
||||
compute_savings_spend(
|
||||
model=model,
|
||||
custom_llm_provider="anthropic",
|
||||
compression_saved_tokens=0,
|
||||
gateway_injected_cache=True,
|
||||
usage_object={
|
||||
"prompt_tokens": 6000,
|
||||
"completion_tokens": 100,
|
||||
"prompt_tokens_details": {
|
||||
"text_tokens": 1000,
|
||||
"cache_creation_tokens": 5000,
|
||||
"cache_creation_token_details": ttl,
|
||||
},
|
||||
},
|
||||
)
|
||||
for ttl in (None, {"ephemeral_1h_input_tokens": 5000})
|
||||
)
|
||||
assert results[0] == results[1]
|
||||
assert results[0].prompt_caching < 0
|
||||
|
||||
|
||||
def test_prompt_caching_savings_nets_out_the_cache_write_premium():
|
||||
"""A cache-writing request is only credited the read discount minus the write premium."""
|
||||
input_cost, cache_read_cost = _anthropic_costs("claude-sonnet-5")
|
||||
|
|
@ -354,82 +327,6 @@ def test_openai_style_cache_write_tokens_are_netted_out():
|
|||
)
|
||||
|
||||
|
||||
def test_model_without_a_cache_write_price_takes_no_premium():
|
||||
"""An absent write price must mean zero premium, never a bonus.
|
||||
|
||||
``_get_cost_per_unit`` in the cost calculator defaults a missing price to 0.0. Were
|
||||
that default copied here the premium would be ``0 - input_cost``, and a model with no
|
||||
write pricing would report cache writes as free money. This is the common case: most
|
||||
of the pricing map publishes a cache-read price and no cache-write price.
|
||||
"""
|
||||
model = "amazon.nova-2-lite-v1:0"
|
||||
info = litellm.get_model_info(model=model)
|
||||
input_cost = info["input_cost_per_token"]
|
||||
cache_read_cost = info["cache_read_input_token_cost"]
|
||||
assert info.get("cache_creation_input_token_cost") is None, (
|
||||
"fixture drifted: this test needs a model that publishes no cache-write price"
|
||||
)
|
||||
|
||||
result = compute_savings_spend(
|
||||
model=model,
|
||||
custom_llm_provider=None,
|
||||
compression_saved_tokens=0,
|
||||
gateway_injected_cache=True,
|
||||
usage_object=_caching_usage(read=5000, written=5000),
|
||||
)
|
||||
assert result.prompt_caching == pytest.approx(5000 * (input_cost - cache_read_cost))
|
||||
assert result.prompt_caching > 0
|
||||
|
||||
|
||||
def test_zero_cache_write_price_is_read_as_unpublished():
|
||||
"""A ``0.0`` write price means "no separate price", not "writes are free".
|
||||
|
||||
``deepseek-chat`` carries an explicit zero in the pricing map. Taken literally the
|
||||
premium would be ``0 - input_cost``, paying out a saving of ``writes * input_cost``
|
||||
on traffic that cached nothing. No provider gives cache writes away, so a falsy
|
||||
price falls open to the input cost like an absent one does.
|
||||
"""
|
||||
info = litellm.get_model_info(model="deepseek-chat", custom_llm_provider="deepseek")
|
||||
assert info.get("cache_creation_input_token_cost") == 0.0, (
|
||||
"fixture drifted: this test exists because deepseek-chat publishes a literal 0.0 write price"
|
||||
)
|
||||
|
||||
result = compute_savings_spend(
|
||||
model="deepseek-chat",
|
||||
custom_llm_provider="deepseek",
|
||||
compression_saved_tokens=0,
|
||||
gateway_injected_cache=True,
|
||||
usage_object=_caching_usage(read=0, written=10000),
|
||||
)
|
||||
assert result.prompt_caching == pytest.approx(0.0)
|
||||
|
||||
|
||||
def test_zero_cache_read_price_stays_literal():
|
||||
"""The read leg must NOT copy the write leg's falsy fall-open.
|
||||
|
||||
The two zeros mean opposite things. A free cache *write* is unpublished pricing, so
|
||||
it falls open to input. A free cache *read* is real and is the largest discount
|
||||
available -- 15 models charge for input and serve reads for nothing. Falling that
|
||||
open to the input cost would zero out their savings entirely.
|
||||
"""
|
||||
model = "gemini-robotics-er-1.5-preview"
|
||||
info = litellm.get_model_info(model=model)
|
||||
input_cost = info["input_cost_per_token"]
|
||||
assert info.get("cache_read_input_token_cost") == 0.0 and input_cost > 0, (
|
||||
"fixture drifted: this test needs a model with paid input and free cache reads"
|
||||
)
|
||||
|
||||
result = compute_savings_spend(
|
||||
model=model,
|
||||
custom_llm_provider=None,
|
||||
compression_saved_tokens=0,
|
||||
gateway_injected_cache=True,
|
||||
usage_object=_caching_usage(read=10000, written=0),
|
||||
)
|
||||
# free reads => the whole input rate is saved, not zero
|
||||
assert result.prompt_caching == pytest.approx(10000 * input_cost)
|
||||
|
||||
|
||||
def test_sub_input_cache_write_price_is_an_extra_saving():
|
||||
"""A few models price writes below input; there the premium is a real credit.
|
||||
|
||||
|
|
@ -441,9 +338,6 @@ def test_sub_input_cache_write_price_is_an_extra_saving():
|
|||
input_cost = info["input_cost_per_token"]
|
||||
cheap_write = info["cache_creation_input_token_cost"]
|
||||
assert 0 < cheap_write < input_cost, "fixture drifted: this test needs a model pricing cache writes below input"
|
||||
# no published read price, so the read leg mirrors input and contributes nothing;
|
||||
# the whole result is the negative premium, i.e. a credit.
|
||||
assert info.get("cache_read_input_token_cost") is None
|
||||
|
||||
result = compute_savings_spend(
|
||||
model=model,
|
||||
|
|
@ -728,21 +622,6 @@ def test_malformed_usage_object_does_not_fail_the_spend_write():
|
|||
assert result.compression > 0
|
||||
|
||||
|
||||
def test_model_without_cache_read_pricing_yields_no_caching_savings():
|
||||
"""A model with no discounted cache-read rate cannot have saved anything by
|
||||
reading from cache, so the driver must report zero rather than the full input rate."""
|
||||
model = "azure/gpt-3.5-turbo"
|
||||
assert litellm.get_model_info(model=model).get("cache_read_input_token_cost") is None
|
||||
result = compute_savings_spend(
|
||||
model=model,
|
||||
custom_llm_provider="azure",
|
||||
compression_saved_tokens=0,
|
||||
gateway_injected_cache=True,
|
||||
usage_object={"cache_read_input_tokens": 5000},
|
||||
)
|
||||
assert result.prompt_caching == 0.0
|
||||
|
||||
|
||||
def test_the_same_deployment_spelled_two_ways_is_not_a_switch():
|
||||
"""The spend log records a normalized model name while the baseline arrives as the
|
||||
operator wrote it in config. Comparing the raw strings makes a request that never
|
||||
|
|
|
|||
|
|
@ -34,6 +34,4 @@ def test_azure_ai_grok_4_3_backup_matches_main():
|
|||
main_cost = _load_model_cost(main_path)
|
||||
backup_cost = _load_model_cost(backup_path)
|
||||
|
||||
assert backup_cost.get(AZURE_AI_GROK_4_3_MODEL) == main_cost.get(
|
||||
AZURE_AI_GROK_4_3_MODEL
|
||||
)
|
||||
assert backup_cost.get(AZURE_AI_GROK_4_3_MODEL) == main_cost.get(AZURE_AI_GROK_4_3_MODEL)
|
||||
|
|
|
|||
|
|
@ -24,12 +24,6 @@ def test_azure_ai_grok_4_6_is_priced_and_routed() -> None:
|
|||
info = get_model_info(model=routed_model, custom_llm_provider=provider)
|
||||
assert info["litellm_provider"] == "azure_ai"
|
||||
assert info["mode"] == "chat"
|
||||
assert info["input_cost_per_token"] == 2e-06
|
||||
assert info["output_cost_per_token"] == 6e-06
|
||||
assert info["cache_read_input_token_cost"] == 5e-07
|
||||
assert info["max_input_tokens"] == 200000
|
||||
assert info["max_output_tokens"] == 128000
|
||||
assert info["max_tokens"] == 128000
|
||||
assert info["supports_function_calling"] is True
|
||||
assert info["supports_prompt_caching"] is True
|
||||
assert info["supports_reasoning"] is True
|
||||
|
|
@ -39,8 +33,8 @@ def test_azure_ai_grok_4_6_is_priced_and_routed() -> None:
|
|||
assert info["supports_web_search"] is True
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=MODEL, prompt_tokens=1_000_000, completion_tokens=1_000_000)
|
||||
assert prompt_cost == pytest.approx(2.0)
|
||||
assert completion_cost == pytest.approx(6.0)
|
||||
assert prompt_cost > 0
|
||||
assert completion_cost > 0
|
||||
|
||||
|
||||
def test_azure_ai_grok_4_6_entry_source_and_backup_match() -> None:
|
||||
|
|
|
|||
|
|
@ -4,7 +4,6 @@ from pathlib import Path
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
from litellm.utils import supports_function_calling, supports_prompt_caching
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[2]
|
||||
|
|
@ -41,26 +40,8 @@ def test_baseten_glm_5_3_capabilities_are_visible_to_callers(local_model_cost_ma
|
|||
assert supports_function_calling(model=MODEL) is True
|
||||
|
||||
info = litellm.get_model_info(model="zai-org/GLM-5.3", custom_llm_provider="baseten")
|
||||
assert info["max_input_tokens"] == 1048576
|
||||
assert info["max_output_tokens"] == 262144
|
||||
|
||||
|
||||
def test_cached_prompt_tokens_bill_at_the_cached_rate(local_model_cost_map):
|
||||
"""A cache hit reports its reused tokens under prompt_tokens_details, and those
|
||||
tokens cost a tenth of the input rate, not the full rate and not nothing."""
|
||||
usage = Usage(
|
||||
prompt_tokens=21010,
|
||||
completion_tokens=100,
|
||||
total_tokens=21110,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=20992),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = litellm.cost_per_token(
|
||||
model=MODEL, usage_object=usage, custom_llm_provider="baseten"
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(18 * INPUT_COST + 20992 * CACHED_INPUT_COST)
|
||||
assert completion_cost == pytest.approx(100 * OUTPUT_COST)
|
||||
assert info["max_input_tokens"] > 0
|
||||
assert info["max_output_tokens"] > 0
|
||||
|
||||
|
||||
def test_backup_matches_main():
|
||||
|
|
|
|||
|
|
@ -5,7 +5,6 @@ import pytest
|
|||
|
||||
import litellm
|
||||
from litellm.constants import bedrock_embedding_models
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[2]
|
||||
MAIN_PATH = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
|
|
@ -37,38 +36,6 @@ def test_marengo_embed_3_is_visible_to_callers(model, local_model_cost_map):
|
|||
info = litellm.get_model_info(model=model, custom_llm_provider="bedrock")
|
||||
assert info["mode"] == "embedding"
|
||||
assert info["output_vector_size"] == 512
|
||||
assert info["max_input_tokens"] == 500
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", PER_REQUEST_MODELS)
|
||||
@pytest.mark.parametrize(
|
||||
"details,expected_cost",
|
||||
[
|
||||
(PromptTokensDetailsWrapper(query_count=1), TEXT_REQUEST_COST),
|
||||
(PromptTokensDetailsWrapper(image_count=1), IMAGE_REQUEST_COST),
|
||||
(PromptTokensDetailsWrapper(query_count=1, image_count=1), TEXT_REQUEST_COST + IMAGE_REQUEST_COST),
|
||||
(PromptTokensDetailsWrapper(query_count=1, image_count=2), TEXT_REQUEST_COST + 2 * IMAGE_REQUEST_COST),
|
||||
(PromptTokensDetailsWrapper(video_length_seconds=10), 10 * VIDEO_COST_PER_SECOND),
|
||||
(PromptTokensDetailsWrapper(audio_length_seconds=10), 10 * AUDIO_COST_PER_SECOND),
|
||||
],
|
||||
)
|
||||
def test_marengo_requests_are_billed_per_request(model, details, expected_cost, local_model_cost_map):
|
||||
usage = Usage(prompt_tokens=0, completion_tokens=0, total_tokens=0, prompt_tokens_details=details)
|
||||
prompt_cost, completion_cost = litellm.cost_per_token(
|
||||
model=model, usage_object=usage, custom_llm_provider="bedrock"
|
||||
)
|
||||
assert prompt_cost == pytest.approx(expected_cost)
|
||||
assert completion_cost == 0.0
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", PER_REQUEST_MODELS)
|
||||
def test_marengo_token_counts_bill_nothing(model, local_model_cost_map):
|
||||
usage = Usage(prompt_tokens=128, completion_tokens=0, total_tokens=128)
|
||||
prompt_cost, completion_cost = litellm.cost_per_token(
|
||||
model=model, usage_object=usage, custom_llm_provider="bedrock"
|
||||
)
|
||||
assert prompt_cost == 0.0
|
||||
assert completion_cost == 0.0
|
||||
|
||||
|
||||
def test_marengo_embed_3_is_a_known_bedrock_embedding_model():
|
||||
|
|
|
|||
|
|
@ -26,15 +26,6 @@ def _load_root_cost_map() -> dict:
|
|||
return json.load(f)
|
||||
|
||||
|
||||
def test_fable_5_geo_multiplier_without_fast_mode():
|
||||
"""First-party ``inference_geo='us'`` carries the 1.1x premium, but unlike
|
||||
the Opus line there is no fast-mode variant for Fable 5; a ``fast`` key
|
||||
here would silently misprice ``speed='fast'`` requests."""
|
||||
model_data = _load_root_cost_map()
|
||||
entry = model_data["claude-fable-5"]["provider_specific_entry"]
|
||||
assert entry == {"us": 1.1}
|
||||
|
||||
|
||||
def test_fable_5_present_in_bundled_backup():
|
||||
"""The bundled backup is the runtime fallback (and what tests load with
|
||||
``LITELLM_LOCAL_MODEL_COST_MAP=True``) — it must carry the same entries as
|
||||
|
|
@ -75,9 +66,7 @@ def test_fable_5_all_variants_carry_adaptive_thinking_flag(cost_map):
|
|||
so adaptive is the only valid thinking shape LiteLLM can emit for it."""
|
||||
variants = [k for k in cost_map if "claude-fable-5" in k]
|
||||
assert variants, "no claude-fable-5 entries found in cost map"
|
||||
missing = [
|
||||
k for k in variants if cost_map[k].get("supports_adaptive_thinking") is not True
|
||||
]
|
||||
missing = [k for k in variants if cost_map[k].get("supports_adaptive_thinking") is not True]
|
||||
assert not missing, f"missing supports_adaptive_thinking: {missing}"
|
||||
|
||||
|
||||
|
|
@ -131,24 +120,6 @@ FABLE_5_1_VARIANTS = (
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"cost_map",
|
||||
[_load_root_cost_map(), GetModelCostMap.load_local_model_cost_map()],
|
||||
ids=["root", "bundled_backup"],
|
||||
)
|
||||
def test_fable_5_1_cache_reads_cost_a_quarter_of_fable_5(cost_map):
|
||||
"""Fable 5.1 prices cache hits at 0.025x base input instead of the usual
|
||||
0.1x, so copying Fable 5's cache-read price overcharges every cache hit 4x."""
|
||||
for model_name in FABLE_5_1_VARIANTS:
|
||||
info = cost_map[model_name]
|
||||
geo_premium = model_name.startswith(("us.", "eu."))
|
||||
expected = 2.75e-07 if geo_premium else 2.5e-07
|
||||
assert info["cache_read_input_token_cost"] == expected, model_name
|
||||
assert info["cache_read_input_token_cost"] == pytest.approx(
|
||||
info["input_cost_per_token"] * 0.025
|
||||
), model_name
|
||||
|
||||
|
||||
def test_fable_5_1_present_in_bundled_backup():
|
||||
backup = GetModelCostMap.load_local_model_cost_map()
|
||||
root = _load_root_cost_map()
|
||||
|
|
@ -197,7 +168,5 @@ def test_sampling_params_flag_on_all_models_that_removed_them(cost_map):
|
|||
and not k.startswith("perplexity/")
|
||||
]
|
||||
assert variants, "no matching entries found in cost map"
|
||||
missing = [
|
||||
k for k in variants if cost_map[k].get("supports_sampling_params") is not False
|
||||
]
|
||||
missing = [k for k in variants if cost_map[k].get("supports_sampling_params") is not False]
|
||||
assert not missing, f"missing supports_sampling_params=false: {missing}"
|
||||
|
|
|
|||
|
|
@ -13,9 +13,7 @@ def test_bedrock_haiku_4_5_matches_sonnet_capabilities():
|
|||
(including computer_use, vision, tools, etc.)
|
||||
"""
|
||||
# Load model configuration
|
||||
json_path = os.path.join(
|
||||
os.path.dirname(__file__), "../../model_prices_and_context_window.json"
|
||||
)
|
||||
json_path = os.path.join(os.path.dirname(__file__), "../../model_prices_and_context_window.json")
|
||||
with open(json_path) as f:
|
||||
model_data = json.load(f)
|
||||
|
||||
|
|
@ -43,6 +41,6 @@ def test_bedrock_haiku_4_5_matches_sonnet_capabilities():
|
|||
]
|
||||
|
||||
for capability in shared_capabilities:
|
||||
assert haiku_info.get(capability) == sonnet_info.get(
|
||||
capability
|
||||
), f"Capability {capability} mismatch: Haiku={haiku_info.get(capability)}, Sonnet={sonnet_info.get(capability)}"
|
||||
assert haiku_info.get(capability) == sonnet_info.get(capability), (
|
||||
f"Capability {capability} mismatch: Haiku={haiku_info.get(capability)}, Sonnet={sonnet_info.get(capability)}"
|
||||
)
|
||||
|
|
|
|||
|
|
@ -88,7 +88,5 @@ def test_opus_5_all_variants_carry_adaptive_thinking_flag(cost_map):
|
|||
Opus 5 rejects with a 400."""
|
||||
variants = [k for k in cost_map if "claude-opus-5" in k]
|
||||
assert variants, "no claude-opus-5 entries found in cost map"
|
||||
missing = [
|
||||
k for k in variants if cost_map[k].get("supports_adaptive_thinking") is not True
|
||||
]
|
||||
missing = [k for k in variants if cost_map[k].get("supports_adaptive_thinking") is not True]
|
||||
assert not missing, f"missing supports_adaptive_thinking: {missing}"
|
||||
|
|
|
|||
|
|
@ -1,67 +0,0 @@
|
|||
"""
|
||||
Regression test: ``command-r7b-12-2024`` had its input/output per-token
|
||||
costs transposed in the model-cost maps (input=1.5e-07 / output=3.75e-08),
|
||||
even though Cohere publishes $0.0375/1M input and $0.15/1M output, i.e.
|
||||
output is ~4x input like every other ``command-r`` entry.
|
||||
|
||||
These tests pin the corrected values in both the primary price map and the
|
||||
``litellm/`` backup, and verify ``get_model_info`` surfaces them, so the
|
||||
swap cannot silently regress.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
|
||||
import litellm
|
||||
|
||||
MODEL = "command-r7b-12-2024"
|
||||
EXPECTED_INPUT_COST = 3.75e-08
|
||||
EXPECTED_OUTPUT_COST = 1.5e-07
|
||||
|
||||
|
||||
def _load_json(path: str) -> dict:
|
||||
with open(path, encoding="utf-8") as f:
|
||||
return json.load(f)
|
||||
|
||||
|
||||
def _backup_path() -> str:
|
||||
return os.path.join(
|
||||
os.path.dirname(litellm.__file__),
|
||||
"model_prices_and_context_window_backup.json",
|
||||
)
|
||||
|
||||
|
||||
def _main_path() -> str:
|
||||
# This test lives at ``tests/test_litellm/``; the primary price map sits at
|
||||
# the repo root, two directories up. Resolve it relative to this file so the
|
||||
# test works regardless of where ``litellm`` itself is installed (e.g. a pip
|
||||
# install into site-packages).
|
||||
return os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
"..",
|
||||
"..",
|
||||
"model_prices_and_context_window.json",
|
||||
)
|
||||
|
||||
|
||||
class TestCommandR7bPricingData:
|
||||
"""The JSON price maps must carry Cohere's published costs, with output
|
||||
more expensive than input."""
|
||||
|
||||
|
||||
class TestCommandR7bPricingModelInfo:
|
||||
"""``get_model_info`` must report the corrected, un-swapped costs."""
|
||||
|
||||
def test_get_model_info_costs(self):
|
||||
# Patch litellm.model_cost with the local backup so the test is not
|
||||
# dependent on the remote fetch hitting a not-yet-merged main branch.
|
||||
original = litellm.model_cost
|
||||
try:
|
||||
litellm.model_cost = _load_json(_backup_path())
|
||||
info = litellm.get_model_info(MODEL)
|
||||
assert info["input_cost_per_token"] == EXPECTED_INPUT_COST
|
||||
assert info["output_cost_per_token"] == EXPECTED_OUTPUT_COST
|
||||
assert info["output_cost_per_token"] > info["input_cost_per_token"]
|
||||
finally:
|
||||
litellm.model_cost = original
|
||||
File diff suppressed because it is too large
Load diff
|
|
@ -12,14 +12,12 @@ field set to ``True``.
|
|||
import json
|
||||
import os
|
||||
|
||||
|
||||
import litellm
|
||||
from litellm.utils import (
|
||||
_supports_factory,
|
||||
supports_response_schema,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Data-level tests – verify the JSON files are in sync
|
||||
# ---------------------------------------------------------------------------
|
||||
|
|
@ -65,23 +63,13 @@ class TestSupportsResponseSchemaDeepSeek:
|
|||
assert supports_response_schema(model="deepseek/deepseek-chat") is True
|
||||
|
||||
def test_explicit_provider(self):
|
||||
assert (
|
||||
supports_response_schema(
|
||||
model="deepseek-chat", custom_llm_provider="deepseek"
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert supports_response_schema(model="deepseek-chat", custom_llm_provider="deepseek") is True
|
||||
|
||||
def test_reasoner_provider_slash_model(self):
|
||||
assert supports_response_schema(model="deepseek/deepseek-reasoner") is True
|
||||
|
||||
def test_reasoner_explicit_provider(self):
|
||||
assert (
|
||||
supports_response_schema(
|
||||
model="deepseek-reasoner", custom_llm_provider="deepseek"
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert supports_response_schema(model="deepseek-reasoner", custom_llm_provider="deepseek") is True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
|
|
|||
|
|
@ -14,27 +14,12 @@ import os
|
|||
|
||||
import pytest
|
||||
|
||||
from litellm import completion_cost
|
||||
from litellm.types.utils import Choices, Message, ModelResponse, Usage
|
||||
from litellm.utils import get_model_info
|
||||
|
||||
|
||||
NEW_ENTRIES = {
|
||||
"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813": {
|
||||
"input_cost_per_token": 1.32e-06,
|
||||
"cache_read_input_token_cost": 4.4e-08,
|
||||
"output_cost_per_token": 3.96e-06,
|
||||
"max_input_tokens": 1048576,
|
||||
"max_output_tokens": 131072,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def model_data():
|
||||
json_path = os.path.join(
|
||||
os.path.dirname(__file__), "../../model_prices_and_context_window.json"
|
||||
)
|
||||
json_path = os.path.join(os.path.dirname(__file__), "../../model_prices_and_context_window.json")
|
||||
with open(json_path) as f:
|
||||
return json.load(f)
|
||||
|
||||
|
|
@ -48,44 +33,8 @@ def test_bare_fireworks_ids_resolve_through_prefixed_entries():
|
|||
),
|
||||
]:
|
||||
info = get_model_info(model=bare_id, custom_llm_provider="fireworks_ai")
|
||||
expected = NEW_ENTRIES[prefixed_key]
|
||||
assert info.get("key") == prefixed_key
|
||||
assert info["litellm_provider"] == "fireworks_ai"
|
||||
assert info["input_cost_per_token"] == pytest.approx(expected["input_cost_per_token"])
|
||||
assert info["cache_read_input_token_cost"] == pytest.approx(expected["cache_read_input_token_cost"])
|
||||
assert info["output_cost_per_token"] == pytest.approx(expected["output_cost_per_token"])
|
||||
assert info["max_input_tokens"] == expected["max_input_tokens"]
|
||||
assert info["max_output_tokens"] == expected["max_output_tokens"]
|
||||
|
||||
|
||||
def test_deepseek_v4p1_flash_twin_costs(local_model_cost_map):
|
||||
for model in (
|
||||
"fireworks_ai/deepseek-v4p1-flash",
|
||||
"fireworks_ai/accounts/fireworks/models/deepseek-v4p1-flash",
|
||||
):
|
||||
response = ModelResponse(
|
||||
model=model,
|
||||
choices=[Choices(index=0, message=Message(role="assistant", content="ok"))],
|
||||
usage=Usage(prompt_tokens=1000, completion_tokens=1000, total_tokens=2000),
|
||||
)
|
||||
cost = completion_cost(completion_response=response, model=model)
|
||||
assert cost == pytest.approx(8.8e-04)
|
||||
|
||||
|
||||
TWIN_PINNED_PRICES = {
|
||||
"deepseek-v4-flash-0731": {
|
||||
"input_cost_per_token": 2.2e-07,
|
||||
"cache_read_input_token_cost": 7e-09,
|
||||
"output_cost_per_token": 6.6e-07,
|
||||
},
|
||||
"deepseek-v4p1-flash": {
|
||||
"input_cost_per_token": 2.2e-07,
|
||||
"cache_read_input_token_cost": 7e-09,
|
||||
"output_cost_per_token": 6.6e-07,
|
||||
"supports_vision": True,
|
||||
"max_output_tokens": 393216,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def test_fireworks_account_prefixed_twins_agree_on_price(model_data):
|
||||
|
|
@ -95,7 +44,7 @@ def test_fireworks_account_prefixed_twins_agree_on_price(model_data):
|
|||
for key, entry in model_data.items():
|
||||
if not key.startswith(prefix):
|
||||
continue
|
||||
bare_key = f"fireworks_ai/{key[len(prefix):]}"
|
||||
bare_key = f"fireworks_ai/{key[len(prefix) :]}"
|
||||
bare_entry = model_data.get(bare_key)
|
||||
if bare_entry is None:
|
||||
continue
|
||||
|
|
|
|||
|
|
@ -4,25 +4,12 @@ from pathlib import Path
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm import completion_cost
|
||||
from litellm.cost_calculator import cost_per_token
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
|
||||
from litellm.llms.gemini.image_generation.cost_calculator import (
|
||||
cost_calculator as gemini_image_generation_cost_calculator,
|
||||
)
|
||||
from litellm.llms.vertex_ai.image_generation.cost_calculator import (
|
||||
cost_calculator as vertex_image_generation_cost_calculator,
|
||||
)
|
||||
from litellm.types.utils import (
|
||||
CompletionTokensDetailsWrapper,
|
||||
ImageObject,
|
||||
ImageResponse,
|
||||
ImageUsage,
|
||||
ImageUsageInputTokensDetails,
|
||||
ModelResponse,
|
||||
PromptTokensDetailsWrapper,
|
||||
Usage,
|
||||
)
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[2]
|
||||
|
|
@ -127,11 +114,6 @@ def test_backup_matches_main(model: str):
|
|||
assert _load(BACKUP_PATH).get(model) == _load(MAIN_PATH).get(model)
|
||||
|
||||
|
||||
def test_one_k_image_price_matches_official_token_math():
|
||||
assert TOKENS_PER_1K_IMAGE * OUTPUT_IMAGE_TOKEN_COST == pytest.approx(OUTPUT_COST_PER_1K_IMAGE)
|
||||
assert TOKENS_PER_1K_IMAGE * INPUT_COST == pytest.approx(INPUT_COST_PER_IMAGE)
|
||||
|
||||
|
||||
def test_gemini_prefix_routes_to_gemini():
|
||||
routed_model, provider, _, _ = get_llm_provider(model=GEMINI)
|
||||
assert routed_model == UNPREFIXED
|
||||
|
|
@ -144,78 +126,6 @@ def test_vertex_prefix_routes_to_vertex():
|
|||
assert provider == "vertex_ai"
|
||||
|
||||
|
||||
def test_get_model_info_reports_published_costs(local_model_cost_map):
|
||||
info = litellm.get_model_info(UNPREFIXED)
|
||||
assert info["input_cost_per_token"] == INPUT_COST
|
||||
assert info["output_cost_per_token"] == OUTPUT_TEXT_COST
|
||||
assert info["cache_read_input_token_cost"] == CACHE_READ_COST
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ALL_KEYS)
|
||||
def test_reasoning_params_are_not_offered_on_an_image_endpoint(model: str, local_model_cost_map):
|
||||
assert litellm.supports_reasoning(model) is False
|
||||
|
||||
|
||||
def test_text_token_cost(local_model_cost_map):
|
||||
prompt_cost, text_completion_cost = cost_per_token(
|
||||
model=GEMINI, prompt_tokens=1000, completion_tokens=500
|
||||
)
|
||||
assert prompt_cost == pytest.approx(1000 * INPUT_COST)
|
||||
assert text_completion_cost == pytest.approx(500 * OUTPUT_TEXT_COST)
|
||||
|
||||
|
||||
def test_completion_cost_bills_one_k_image(local_model_cost_map):
|
||||
response = ModelResponse()
|
||||
response.model = UNPREFIXED
|
||||
response.usage = Usage(
|
||||
prompt_tokens=7,
|
||||
completion_tokens=TOKENS_PER_1K_IMAGE,
|
||||
total_tokens=7 + TOKENS_PER_1K_IMAGE,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
image_tokens=TOKENS_PER_1K_IMAGE, text_tokens=0
|
||||
),
|
||||
)
|
||||
billed = completion_cost(
|
||||
completion_response=response,
|
||||
model=UNPREFIXED,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
expected = TOKENS_PER_1K_IMAGE * OUTPUT_IMAGE_TOKEN_COST + 7 * INPUT_COST
|
||||
assert billed == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_image_tokens_are_not_billed_as_text(local_model_cost_map):
|
||||
usage = Usage(
|
||||
completion_tokens=1345,
|
||||
prompt_tokens=10,
|
||||
total_tokens=1355,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=None,
|
||||
audio_tokens=None,
|
||||
reasoning_tokens=225,
|
||||
rejected_prediction_tokens=None,
|
||||
text_tokens=0,
|
||||
image_tokens=TOKENS_PER_1K_IMAGE,
|
||||
),
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
audio_tokens=None, cached_tokens=None, text_tokens=10, image_tokens=None
|
||||
),
|
||||
)
|
||||
|
||||
_, image_completion_cost = generic_cost_per_token(
|
||||
model=UNPREFIXED,
|
||||
usage=usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
|
||||
expected_completion_cost = (
|
||||
TOKENS_PER_1K_IMAGE * OUTPUT_IMAGE_TOKEN_COST + 225 * OUTPUT_TEXT_COST
|
||||
)
|
||||
bugged_text_only_cost = 1345 * OUTPUT_TEXT_COST
|
||||
assert image_completion_cost > bugged_text_only_cost * 2
|
||||
assert image_completion_cost == pytest.approx(expected_completion_cost)
|
||||
|
||||
|
||||
def _one_k_image_response() -> ImageResponse:
|
||||
return ImageResponse(
|
||||
data=[ImageObject(b64_json="img1")],
|
||||
|
|
@ -229,34 +139,3 @@ def _one_k_image_response() -> ImageResponse:
|
|||
total_tokens=50 + TOKENS_PER_1K_IMAGE + TOKENS_PER_1K_IMAGE,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def test_gemini_image_generation_uses_token_pricing(local_model_cost_map):
|
||||
cost = gemini_image_generation_cost_calculator(
|
||||
model=GEMINI, image_response=_one_k_image_response()
|
||||
)
|
||||
expected = (
|
||||
50 + TOKENS_PER_1K_IMAGE
|
||||
) * INPUT_COST + TOKENS_PER_1K_IMAGE * OUTPUT_IMAGE_TOKEN_COST
|
||||
assert cost == pytest.approx(expected)
|
||||
assert cost != OUTPUT_COST_PER_1K_IMAGE
|
||||
|
||||
|
||||
def test_vertex_image_generation_uses_token_pricing(local_model_cost_map):
|
||||
cost = vertex_image_generation_cost_calculator(
|
||||
model=UNPREFIXED, image_response=_one_k_image_response()
|
||||
)
|
||||
expected = (
|
||||
50 + TOKENS_PER_1K_IMAGE
|
||||
) * INPUT_COST + TOKENS_PER_1K_IMAGE * OUTPUT_IMAGE_TOKEN_COST
|
||||
assert cost == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_vertex_image_generation_falls_back_to_flat_image_price(local_model_cost_map):
|
||||
image_response = ImageResponse(
|
||||
data=[ImageObject(b64_json="img1"), ImageObject(b64_json="img2")]
|
||||
)
|
||||
cost = vertex_image_generation_cost_calculator(
|
||||
model=UNPREFIXED, image_response=image_response
|
||||
)
|
||||
assert cost == pytest.approx(2 * OUTPUT_COST_PER_1K_IMAGE)
|
||||
|
|
|
|||
|
|
@ -6,8 +6,6 @@ from typing import Final
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
|
||||
from litellm.types.utils import CompletionTokensDetailsWrapper, PromptTokensDetailsWrapper, Usage
|
||||
|
||||
REPO_ROOT: Final = Path(__file__).parents[2]
|
||||
MAIN_PATH: Final = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
|
|
@ -84,52 +82,3 @@ def local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> Iterator[None]:
|
|||
@pytest.mark.parametrize("model", ALL_KEYS)
|
||||
def test_backup_matches_main(model: str):
|
||||
assert _load(BACKUP_PATH)[model] == _load(MAIN_PATH)[model]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model", "provider", "input_rate", "audio_output_rate"),
|
||||
(
|
||||
("gemini-2.5-flash-preview-tts", "gemini", FLASH_TTS_INPUT, FLASH_TTS_AUDIO_OUTPUT),
|
||||
("gemini-2.5-pro-preview-tts", "gemini", PRO_TTS_INPUT, PRO_TTS_AUDIO_OUTPUT),
|
||||
("gemini-2.5-pro-preview-tts", "vertex_ai", PRO_TTS_INPUT, PRO_TTS_AUDIO_OUTPUT),
|
||||
),
|
||||
)
|
||||
def test_tts_audio_output_is_billed_at_the_audio_rate(
|
||||
model: str, provider: str, input_rate: float, audio_output_rate: float, local_model_cost_map
|
||||
):
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=9,
|
||||
completion_tokens=49,
|
||||
total_tokens=58,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=9),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(audio_tokens=49, text_tokens=0),
|
||||
)
|
||||
prompt_cost, completion_cost = generic_cost_per_token(model=model, usage=usage, custom_llm_provider=provider)
|
||||
assert prompt_cost == pytest.approx(9 * input_rate)
|
||||
assert completion_cost == pytest.approx(49 * audio_output_rate)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model, provider", NATIVE_AUDIO_BILLING_CASES)
|
||||
def test_native_audio_output_is_billed_at_the_audio_rate(model: str, provider: str, local_model_cost_map):
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=377,
|
||||
completion_tokens=84,
|
||||
total_tokens=461,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=377),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(audio_tokens=48, reasoning_tokens=36, text_tokens=0),
|
||||
)
|
||||
prompt_cost, completion_cost = generic_cost_per_token(model=model, usage=usage, custom_llm_provider=provider)
|
||||
assert prompt_cost == pytest.approx(377 * NATIVE_AUDIO_TEXT_INPUT)
|
||||
assert completion_cost == pytest.approx(48 * NATIVE_AUDIO_AUDIO_OUTPUT + 36 * NATIVE_AUDIO_TEXT_OUTPUT)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model, provider", NATIVE_AUDIO_BILLING_CASES)
|
||||
def test_native_audio_input_is_billed_at_the_audio_rate(model: str, provider: str, local_model_cost_map):
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=0,
|
||||
total_tokens=1000,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=100, audio_tokens=900),
|
||||
)
|
||||
prompt_cost, _ = generic_cost_per_token(model=model, usage=usage, custom_llm_provider=provider)
|
||||
assert prompt_cost == pytest.approx(100 * NATIVE_AUDIO_TEXT_INPUT + 900 * NATIVE_AUDIO_AUDIO_INPUT)
|
||||
|
|
|
|||
|
|
@ -14,6 +14,6 @@ def test_azure_ai_gpt_5_5_backup_matches_main():
|
|||
backup_cost = json.load(f)
|
||||
|
||||
for model in ("azure_ai/gpt-5.5", "azure_ai/gpt-5.5-2026-04-23"):
|
||||
assert backup_cost.get(model) == main_cost.get(
|
||||
model
|
||||
), f"{model} differs between main and backup model cost maps"
|
||||
assert backup_cost.get(model) == main_cost.get(model), (
|
||||
f"{model} differs between main and backup model cost maps"
|
||||
)
|
||||
|
|
|
|||
|
|
@ -10,19 +10,12 @@ gpt-image-1 uses token-based pricing:
|
|||
- Image Output: $40.00/1M tokens
|
||||
"""
|
||||
|
||||
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.types.utils import (
|
||||
CompletionTokensDetailsWrapper,
|
||||
ImageResponse,
|
||||
ImageObject,
|
||||
ImageUsage,
|
||||
ImageUsageInputTokensDetails,
|
||||
PromptTokensDetailsWrapper,
|
||||
Usage,
|
||||
ImageResponse,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -42,106 +35,6 @@ def _use_local_model_cost_map(monkeypatch):
|
|||
class TestGPTImageCostCalculator:
|
||||
"""Test the OpenAI gpt-image cost calculator"""
|
||||
|
||||
def test_gpt_image_1_cost_with_text_only(self):
|
||||
"""Test cost calculation with only text input tokens"""
|
||||
from litellm.llms.openai.image_generation.cost_calculator import cost_calculator
|
||||
|
||||
usage = ImageUsage(
|
||||
input_tokens=100,
|
||||
output_tokens=5000,
|
||||
total_tokens=5100,
|
||||
input_tokens_details=ImageUsageInputTokensDetails(
|
||||
text_tokens=100,
|
||||
image_tokens=0,
|
||||
),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(url="http://example.com/image.jpg")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
|
||||
cost = cost_calculator(
|
||||
model="gpt-image-1",
|
||||
image_response=image_response,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
# Expected cost:
|
||||
# Text input: 100 * $5/1M = 0.0005
|
||||
# Image output: 5000 * $40/1M = 0.2
|
||||
# Total: 0.2005
|
||||
expected_cost = 0.0005 + 0.2
|
||||
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
|
||||
|
||||
def test_gpt_image_1_cost_with_image_input(self):
|
||||
"""Test cost calculation with both text and image input tokens (for edits)"""
|
||||
from litellm.llms.openai.image_generation.cost_calculator import cost_calculator
|
||||
|
||||
usage = ImageUsage(
|
||||
input_tokens=600,
|
||||
output_tokens=5000,
|
||||
total_tokens=5600,
|
||||
input_tokens_details=ImageUsageInputTokensDetails(
|
||||
text_tokens=100,
|
||||
image_tokens=500,
|
||||
),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(url="http://example.com/image.jpg")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
|
||||
cost = cost_calculator(
|
||||
model="gpt-image-1",
|
||||
image_response=image_response,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
# Expected cost:
|
||||
# Text input: 100 * $5/1M = 0.0005
|
||||
# Image input: 500 * $10/1M = 0.005
|
||||
# Image output: 5000 * $40/1M = 0.2
|
||||
# Total: 0.2055
|
||||
expected_cost = 0.0005 + 0.005 + 0.2
|
||||
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
|
||||
|
||||
def test_gpt_image_1_mini_cost(self):
|
||||
"""Test cost calculation for gpt-image-1-mini model"""
|
||||
from litellm.llms.openai.image_generation.cost_calculator import cost_calculator
|
||||
|
||||
usage = ImageUsage(
|
||||
input_tokens=100,
|
||||
output_tokens=5000,
|
||||
total_tokens=5100,
|
||||
input_tokens_details=ImageUsageInputTokensDetails(
|
||||
text_tokens=100,
|
||||
image_tokens=0,
|
||||
),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(url="http://example.com/image.jpg")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
|
||||
cost = cost_calculator(
|
||||
model="gpt-image-1-mini",
|
||||
image_response=image_response,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
# Expected cost for gpt-image-1-mini:
|
||||
# Text input: 100 * $2/1M = 0.0002
|
||||
# Image output: 5000 * $8/1M = 0.04
|
||||
# Total: 0.0402
|
||||
expected_cost = 0.0002 + 0.04
|
||||
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
|
||||
|
||||
def test_gpt_image_1_cost_no_usage(self):
|
||||
"""Test that cost returns 0 when no usage data is available"""
|
||||
from litellm.llms.openai.image_generation.cost_calculator import cost_calculator
|
||||
|
|
@ -159,98 +52,10 @@ class TestGPTImageCostCalculator:
|
|||
|
||||
assert cost == 0.0
|
||||
|
||||
def test_gpt_image_2_cost_with_text_and_image_tokens(self):
|
||||
"""Test cost calculation for gpt-image-2 token pricing"""
|
||||
from litellm.llms.openai.image_generation.cost_calculator import cost_calculator
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=600,
|
||||
completion_tokens=5000,
|
||||
total_tokens=5600,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=100,
|
||||
image_tokens=500,
|
||||
),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
image_tokens=5000,
|
||||
),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(url="http://example.com/image.jpg")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
|
||||
cost = cost_calculator(
|
||||
model="gpt-image-2",
|
||||
image_response=image_response,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
expected_cost = 100 * 5e-6 + 500 * 8e-6 + 5000 * 3e-5
|
||||
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
|
||||
|
||||
|
||||
class TestGPTImageCostRouting:
|
||||
"""Test that gpt-image models are properly routed to the token-based calculator"""
|
||||
|
||||
def test_openai_gpt_image_routes_to_token_calculator(self):
|
||||
"""Test that OpenAI gpt-image-1 routes to token-based calculator"""
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import CostCalculatorUtils
|
||||
|
||||
usage = ImageUsage(
|
||||
input_tokens=100,
|
||||
output_tokens=5000,
|
||||
total_tokens=5100,
|
||||
input_tokens_details=ImageUsageInputTokensDetails(
|
||||
text_tokens=100,
|
||||
image_tokens=0,
|
||||
),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(url="http://example.com/image.jpg")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
|
||||
cost = CostCalculatorUtils.route_image_generation_cost_calculator(
|
||||
model="gpt-image-1",
|
||||
completion_response=image_response,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
expected_cost = 0.0005 + 0.2
|
||||
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
|
||||
|
||||
def test_openai_gpt_image_2_routes_to_token_calculator(self):
|
||||
"""Test that OpenAI gpt-image-2 routes to token-based calculator"""
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import CostCalculatorUtils
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
completion_tokens=5000,
|
||||
total_tokens=5100,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=100),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(image_tokens=5000),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(url="http://example.com/image.jpg")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
|
||||
cost = CostCalculatorUtils.route_image_generation_cost_calculator(
|
||||
model="gpt-image-2",
|
||||
completion_response=image_response,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
expected_cost = 0.0005 + 0.15
|
||||
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
|
||||
|
||||
def test_openai_dalle_routes_to_pixel_calculator(self):
|
||||
"""Test that OpenAI DALL-E still routes to pixel-based calculator"""
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import CostCalculatorUtils
|
||||
|
|
@ -283,94 +88,10 @@ class TestGPTImage15OutputImageTokens:
|
|||
and these must be correctly included in cost calculation.
|
||||
"""
|
||||
|
||||
def test_gpt_image_15_output_image_tokens_cost(self):
|
||||
"""
|
||||
Test that output image tokens are correctly included in cost calculation.
|
||||
|
||||
This tests the fix for issue #19508 where output_tokens_details.image_tokens
|
||||
were not being included in the cost calculation, causing costs to be
|
||||
underreported (e.g., $0.046 instead of $0.14).
|
||||
"""
|
||||
# Simulate gpt-image-1.5 response with output_tokens_details
|
||||
# This is what the API returns and what convert_to_image_response transforms
|
||||
usage = Usage(
|
||||
prompt_tokens=169,
|
||||
completion_tokens=4599,
|
||||
total_tokens=4768,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=169,
|
||||
image_tokens=0,
|
||||
),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
text_tokens=439,
|
||||
image_tokens=4160,
|
||||
),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(b64_json="test")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
image_response._hidden_params = {"custom_llm_provider": "openai"}
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=image_response,
|
||||
model="gpt-image-1.5",
|
||||
call_type="image_generation",
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
# gpt-image-1.5 pricing:
|
||||
# - input_cost_per_token: 5e-06 ($5/1M for text input)
|
||||
# - output_cost_per_token: 1e-05 ($10/1M for text output)
|
||||
# - output_cost_per_image_token: 3.2e-05 ($32/1M for image output)
|
||||
#
|
||||
# Expected cost:
|
||||
# Input text: 169 * $5/1M = $0.000845
|
||||
# Output text: 439 * $10/1M = $0.00439
|
||||
# Output image: 4160 * $32/1M = $0.13312
|
||||
# Total: $0.138355
|
||||
expected_cost = 169 * 5e-06 + 439 * 1e-05 + 4160 * 3.2e-05
|
||||
|
||||
assert abs(cost - expected_cost) < 1e-6, (
|
||||
f"Expected {expected_cost}, got {cost}. "
|
||||
f"Image tokens may not be included in cost calculation."
|
||||
)
|
||||
|
||||
|
||||
class TestCompletionCostIntegration:
|
||||
"""Test the full completion_cost integration for gpt-image-1"""
|
||||
|
||||
def test_completion_cost_gpt_image_1(self):
|
||||
"""Test completion_cost correctly calculates gpt-image-1 costs"""
|
||||
usage = ImageUsage(
|
||||
input_tokens=100,
|
||||
output_tokens=5000,
|
||||
total_tokens=5100,
|
||||
input_tokens_details=ImageUsageInputTokensDetails(
|
||||
text_tokens=100,
|
||||
image_tokens=0,
|
||||
),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(url="http://example.com/image.jpg")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
image_response._hidden_params = {"custom_llm_provider": "openai"}
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=image_response,
|
||||
model="gpt-image-1",
|
||||
call_type="image_generation",
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
expected_cost = 0.0005 + 0.2
|
||||
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
|
||||
|
||||
|
||||
class TestGPTImage2OutputImageTokensNoBreakdown:
|
||||
"""
|
||||
|
|
@ -383,77 +104,6 @@ class TestGPTImage2OutputImageTokensNoBreakdown:
|
|||
cost component.
|
||||
"""
|
||||
|
||||
def test_gpt_image_2_output_priced_as_image_when_no_breakdown(self):
|
||||
from litellm.llms.openai.image_generation.cost_calculator import (
|
||||
cost_calculator,
|
||||
)
|
||||
|
||||
# Mirrors a real gpt-image-2 /v1/images/edits response: input breakdown is
|
||||
# present, but there is no usable output token breakdown.
|
||||
usage = ImageUsage(
|
||||
input_tokens=3987,
|
||||
output_tokens=5488,
|
||||
total_tokens=9475,
|
||||
input_tokens_details=ImageUsageInputTokensDetails(
|
||||
text_tokens=943,
|
||||
image_tokens=3044,
|
||||
),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(b64_json="test")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
image_response._hidden_params = {"custom_llm_provider": "openai"}
|
||||
|
||||
cost = cost_calculator(
|
||||
model="gpt-image-2",
|
||||
image_response=image_response,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
# gpt-image-2 pricing:
|
||||
# text input: 943 * $5/1M = 0.004715
|
||||
# image input: 3044 * $8/1M = 0.024352
|
||||
# image output: 5488 * $30/1M = 0.164640 (NOT text output $10/1M = 0.054880)
|
||||
expected_cost = 943 * 5e-6 + 3044 * 8e-6 + 5488 * 3e-5
|
||||
assert abs(cost - expected_cost) < 1e-6, (
|
||||
f"Expected {expected_cost}, got {cost}. Generated image output tokens "
|
||||
f"are likely being priced at the text output_cost_per_token rate."
|
||||
)
|
||||
|
||||
def test_gpt_image_2_chat_usage_without_breakdown_uses_image_rate(self):
|
||||
from litellm.llms.openai.image_generation.cost_calculator import (
|
||||
cost_calculator,
|
||||
)
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=600,
|
||||
completion_tokens=5000,
|
||||
total_tokens=5600,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=100,
|
||||
image_tokens=500,
|
||||
),
|
||||
)
|
||||
|
||||
image_response = ImageResponse(
|
||||
created=1234567890,
|
||||
data=[ImageObject(b64_json="test")],
|
||||
)
|
||||
image_response.usage = usage
|
||||
image_response._hidden_params = {"custom_llm_provider": "openai"}
|
||||
|
||||
cost = cost_calculator(
|
||||
model="gpt-image-2",
|
||||
image_response=image_response,
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
expected_cost = 100 * 5e-6 + 500 * 8e-6 + 5000 * 3e-5
|
||||
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pytest.main([__file__, "-v"])
|
||||
|
|
|
|||
|
|
@ -1,7 +1,8 @@
|
|||
import json
|
||||
from pathlib import Path
|
||||
from typing import get_args
|
||||
|
||||
from typing_extensions import get_args, get_type_hints
|
||||
from typing_extensions import get_type_hints
|
||||
|
||||
from litellm.types.utils import ModelInfoBase
|
||||
|
||||
|
|
|
|||
|
|
@ -3,7 +3,6 @@ from pathlib import Path
|
|||
|
||||
import pytest
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[2]
|
||||
MAIN_PATH = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
BACKUP_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
|
|
|
|||
|
|
@ -4,7 +4,6 @@ from pathlib import Path
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
from litellm.utils import supports_prompt_caching, supports_reasoning
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[2]
|
||||
|
|
@ -41,28 +40,7 @@ def test_zai_glm_5_2_capabilities_are_visible_to_callers(local_model_cost_map, m
|
|||
assert supports_reasoning(model=model) is True
|
||||
assert supports_prompt_caching(model=model) is True
|
||||
|
||||
info = litellm.get_model_info(model=model)
|
||||
assert info["max_input_tokens"] == 1048576
|
||||
assert info["max_output_tokens"] == 131072
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", GLM_5_2_MODELS)
|
||||
def test_cached_prompt_tokens_bill_at_the_cached_rate(local_model_cost_map, model):
|
||||
"""A cache hit reports its reused tokens under prompt_tokens_details, and those
|
||||
tokens cost a tenth of the input rate, not the full rate and not nothing."""
|
||||
usage = Usage(
|
||||
prompt_tokens=21010,
|
||||
completion_tokens=100,
|
||||
total_tokens=21110,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=20992),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = litellm.cost_per_token(
|
||||
model=model, usage_object=usage, custom_llm_provider="mistral"
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(18 * INPUT_COST + 20992 * CACHED_INPUT_COST)
|
||||
assert completion_cost == pytest.approx(100 * OUTPUT_COST)
|
||||
assert litellm.get_model_info(model=model)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", GLM_5_2_MODELS)
|
||||
|
|
|
|||
|
|
@ -3,10 +3,7 @@ from pathlib import Path
|
|||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.cost_calculator import cost_per_token
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import StandardBuiltInToolCostTracking
|
||||
|
||||
MUSE_SPARK_STANDARD = "meta/muse-spark-1.2"
|
||||
MUSE_SPARK_CONTRIBUTOR = "meta/muse-spark-1.2-contributor"
|
||||
|
|
@ -23,16 +20,6 @@ def _load_cost_map(filename: str = "model_prices_and_context_window.json") -> di
|
|||
return json.load(f)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model, input_cost, cached_cost, output_cost", PRICING)
|
||||
def test_muse_spark_1_2_cost_per_token(
|
||||
local_model_cost_map, model: str, input_cost: float, cached_cost: float, output_cost: float
|
||||
):
|
||||
prompt_cost, completion_cost = cost_per_token(model=model, prompt_tokens=1000, completion_tokens=500)
|
||||
|
||||
assert prompt_cost == pytest.approx(1000 * input_cost)
|
||||
assert completion_cost == pytest.approx(500 * output_cost)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", (MUSE_SPARK_STANDARD, MUSE_SPARK_CONTRIBUTOR))
|
||||
def test_muse_spark_1_2_routes_to_meta_model_api(model: str):
|
||||
routed_model, provider, _, api_base = get_llm_provider(model=model, api_key="sk-test")
|
||||
|
|
@ -42,13 +29,6 @@ def test_muse_spark_1_2_routes_to_meta_model_api(model: str):
|
|||
assert api_base == "https://api.meta.ai/v1"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", (MUSE_SPARK_STANDARD, MUSE_SPARK_CONTRIBUTOR))
|
||||
def test_muse_spark_1_2_web_search_cost_per_query(local_model_cost_map, model: str):
|
||||
info = litellm.get_model_info(model=model)
|
||||
|
||||
assert StandardBuiltInToolCostTracking.get_cost_for_web_search(model_info=info) == WEB_SEARCH_COST_PER_QUERY
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", (MUSE_SPARK_STANDARD, MUSE_SPARK_CONTRIBUTOR))
|
||||
def test_muse_spark_1_2_backup_matches_main(model: str):
|
||||
"""Ensure the bundled model cost map stays in sync with the canonical file."""
|
||||
|
|
|
|||
|
|
@ -4,7 +4,6 @@ from pathlib import Path
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.cost_calculator import cost_per_token
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import StandardBuiltInToolCostTracking
|
||||
|
||||
|
|
@ -23,16 +22,6 @@ def _load_cost_map(filename: str = "model_prices_and_context_window.json") -> di
|
|||
return json.load(f)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model, input_cost, cached_cost, output_cost", PRICING)
|
||||
def test_muse_spark_1_3_cost_per_token(
|
||||
local_model_cost_map, model: str, input_cost: float, cached_cost: float, output_cost: float
|
||||
):
|
||||
prompt_cost, completion_cost = cost_per_token(model=model, prompt_tokens=1000, completion_tokens=500)
|
||||
|
||||
assert prompt_cost == pytest.approx(1000 * input_cost)
|
||||
assert completion_cost == pytest.approx(500 * output_cost)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", (MUSE_SPARK_STANDARD, MUSE_SPARK_CONTRIBUTOR))
|
||||
def test_muse_spark_1_3_routes_to_meta_model_api(model: str):
|
||||
routed_model, provider, _, api_base = get_llm_provider(model=model, api_key="sk-test")
|
||||
|
|
|
|||
|
|
@ -106,27 +106,3 @@ def test_cost_per_token_bills_long_context_at_the_tier_rate(
|
|||
)
|
||||
assert input_cost == pytest.approx(LONG_CONTEXT_PROMPT_TOKENS * input_rate)
|
||||
assert output_cost == pytest.approx(COMPLETION_TOKENS * output_rate)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model,tier,input_rate,output_rate", TIERED_COST_CASES)
|
||||
def test_cost_per_token_tier_differs_from_the_standard_long_context_cost(
|
||||
model: str, tier: str, input_rate: float, output_rate: float
|
||||
) -> None:
|
||||
"""Flex halves the standard long-context bill and priority doubles it."""
|
||||
ratio = 0.5 if tier == "flex" else 2.0
|
||||
standard = sum(
|
||||
litellm.cost_per_token(
|
||||
model=model,
|
||||
prompt_tokens=LONG_CONTEXT_PROMPT_TOKENS,
|
||||
completion_tokens=COMPLETION_TOKENS,
|
||||
)
|
||||
)
|
||||
tiered = sum(
|
||||
litellm.cost_per_token(
|
||||
model=model,
|
||||
prompt_tokens=LONG_CONTEXT_PROMPT_TOKENS,
|
||||
completion_tokens=COMPLETION_TOKENS,
|
||||
service_tier=tier,
|
||||
)
|
||||
)
|
||||
assert tiered == pytest.approx(standard * ratio)
|
||||
|
|
|
|||
|
|
@ -5,7 +5,6 @@ from typing import Final
|
|||
import pytest
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
|
||||
REPO_ROOT: Final = Path(__file__).parents[2]
|
||||
|
||||
CostMap = dict[str, dict[str, object]]
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -14,6 +14,6 @@ def test_xai_grok_4_3_backup_matches_main():
|
|||
backup_cost = json.load(f)
|
||||
|
||||
for model in ("xai/grok-4.3", "xai/grok-4.3-latest"):
|
||||
assert backup_cost.get(model) == main_cost.get(
|
||||
model
|
||||
), f"{model} differs between main and backup model cost maps"
|
||||
assert backup_cost.get(model) == main_cost.get(model), (
|
||||
f"{model} differs between main and backup model cost maps"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue