mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cost): price batch image output tokens at the batch image rate (#44897)
* fix(cost): price batch image completion tokens at image batch rate Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): cover default vertex batch output transformation for image cost Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost): sync model prices schema Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(cost): simplify batch completion cost with rate helpers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost): keep global batch pricing fallback for image-only deployment rates Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(test): rename connect tunnel helper to https redirect Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(cost): merge deployment batch rates over global pricing Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(model_prices): put flash-image batch image rate on the GA row Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(cost): drop redundant comment in batch fallback Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(cost): use _batch_or_half for the batch image rate Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(cost): price nano banana 2.1 fixtures off the real cost map rows Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(cost): use the vertex 4k image token count in the batch cost test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
7566f9bc54
commit
191967c207
12 changed files with 1412 additions and 18 deletions
|
|
@ -2568,6 +2568,20 @@ def _batch_rate(
|
|||
return fallback if rate is None else rate
|
||||
|
||||
|
||||
def _batch_or_half(batch_rate: float | None, standard_rate: float | None, default: float = 0.0) -> float:
|
||||
if batch_rate is not None:
|
||||
return batch_rate
|
||||
if standard_rate is not None:
|
||||
return standard_rate / 2
|
||||
return default
|
||||
|
||||
|
||||
def _completion_image_tokens(usage: Usage) -> int:
|
||||
details: Final = usage.completion_tokens_details
|
||||
image_tokens: Final = (details.image_tokens or 0) if details is not None else 0
|
||||
return min(image_tokens, usage.completion_tokens)
|
||||
|
||||
|
||||
def batch_cost_calculator(
|
||||
usage: Usage,
|
||||
model: str,
|
||||
|
|
@ -2607,13 +2621,14 @@ def batch_cost_calculator(
|
|||
"output_cost_per_token",
|
||||
)
|
||||
):
|
||||
# model_info was provided (e.g. deployment metadata with only id/db_model)
|
||||
# but carries no pricing fields. Fall back to the global pricing table so
|
||||
# that standard model pricing is used instead of silently returning $0.
|
||||
deployment_info: Final = model_info
|
||||
try:
|
||||
global_info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider)
|
||||
if global_info:
|
||||
model_info = global_info
|
||||
model_info = {
|
||||
**global_info,
|
||||
**{key: value for key, value in deployment_info.items() if value is not None},
|
||||
}
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
|
@ -2647,12 +2662,15 @@ def batch_cost_calculator(
|
|||
|
||||
cache_creation_cost: Final = model_info.get("cache_creation_input_token_cost") or input_cost_per_token
|
||||
total_prompt_cost += cache_creation_tokens * cache_creation_cost / 2
|
||||
if batch_rates.output is not None:
|
||||
total_completion_cost = usage.completion_tokens * batch_rates.output
|
||||
elif output_cost_per_token:
|
||||
total_completion_cost = (
|
||||
usage.completion_tokens * (output_cost_per_token) / 2
|
||||
) # batch cost is usually half of the regular token cost
|
||||
text_rate: Final = _batch_or_half(batch_rates.output, output_cost_per_token)
|
||||
image_rate: Final = _batch_or_half(
|
||||
model_info.get("output_cost_per_image_token_batches"),
|
||||
model_info.get("output_cost_per_image_token"),
|
||||
default=text_rate,
|
||||
)
|
||||
image_tokens: Final = _completion_image_tokens(usage)
|
||||
text_tokens: Final = usage.completion_tokens - image_tokens
|
||||
total_completion_cost = text_tokens * text_rate + image_tokens * image_rate
|
||||
|
||||
uplift: Final = _get_regional_uplift_multiplier(model_info, data_residency)
|
||||
if uplift != 1.0:
|
||||
|
|
|
|||
|
|
@ -29288,6 +29288,7 @@
|
|||
"mode": "image_generation",
|
||||
"output_cost_per_image": 0.134,
|
||||
"output_cost_per_image_token": 0.00012,
|
||||
"output_cost_per_image_token_batches": 6e-05,
|
||||
"output_cost_per_token": 1.2e-05,
|
||||
"rpm": 1000,
|
||||
"tpm": 4000000,
|
||||
|
|
@ -29379,6 +29380,7 @@
|
|||
"mode": "image_generation",
|
||||
"output_cost_per_image": 0.045,
|
||||
"output_cost_per_image_token": 6e-05,
|
||||
"output_cost_per_image_token_batches": 3e-05,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"output_cost_per_token_batches": 1.5e-06,
|
||||
"rpm": 1000,
|
||||
|
|
|
|||
|
|
@ -376,6 +376,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
|
|||
output_cost_per_image: float | None
|
||||
output_cost_per_pixel: ReadOnly[float | None]
|
||||
output_cost_per_image_token: float | None
|
||||
output_cost_per_image_token_batches: ReadOnly[float | None]
|
||||
output_cost_per_video_token: float | None # for gemini omni models with video output
|
||||
output_vector_size: int | None
|
||||
output_cost_per_reasoning_token: float | None
|
||||
|
|
@ -3820,6 +3821,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
output_cost_per_character_above_128k_tokens: float | None = None
|
||||
output_cost_per_image: float | None = None
|
||||
output_cost_per_image_token: float | None = None
|
||||
output_cost_per_image_token_batches: float | None = None
|
||||
output_cost_per_video_token: float | None = None
|
||||
output_cost_per_reasoning_token: float | None = None
|
||||
output_cost_per_reasoning_token_flex: float | None = None
|
||||
|
|
|
|||
|
|
@ -6271,6 +6271,7 @@ def _get_model_info_helper(
|
|||
output_cost_per_image=_model_info.get("output_cost_per_image", None),
|
||||
output_cost_per_pixel=_model_info.get("output_cost_per_pixel", None),
|
||||
output_cost_per_image_token=_model_info.get("output_cost_per_image_token", None),
|
||||
output_cost_per_image_token_batches=_model_info.get("output_cost_per_image_token_batches", None),
|
||||
output_cost_per_video_token=_model_info.get("output_cost_per_video_token", None),
|
||||
output_vector_size=_model_info.get("output_vector_size", None),
|
||||
citation_cost_per_token=_model_info.get("citation_cost_per_token", None),
|
||||
|
|
|
|||
|
|
@ -29288,6 +29288,7 @@
|
|||
"mode": "image_generation",
|
||||
"output_cost_per_image": 0.134,
|
||||
"output_cost_per_image_token": 0.00012,
|
||||
"output_cost_per_image_token_batches": 6e-05,
|
||||
"output_cost_per_token": 1.2e-05,
|
||||
"rpm": 1000,
|
||||
"tpm": 4000000,
|
||||
|
|
@ -29379,6 +29380,7 @@
|
|||
"mode": "image_generation",
|
||||
"output_cost_per_image": 0.045,
|
||||
"output_cost_per_image_token": 6e-05,
|
||||
"output_cost_per_image_token_batches": 3e-05,
|
||||
"output_cost_per_token": 3e-06,
|
||||
"output_cost_per_token_batches": 1.5e-06,
|
||||
"rpm": 1000,
|
||||
|
|
|
|||
|
|
@ -710,6 +710,10 @@
|
|||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"output_cost_per_image_token_batches": {
|
||||
"type": "number",
|
||||
"minimum": 0
|
||||
},
|
||||
"output_cost_per_pixel": {
|
||||
"type": "number",
|
||||
"minimum": 0
|
||||
|
|
|
|||
|
|
@ -49,6 +49,7 @@ class CostMapEntry(BaseModel):
|
|||
output_cost_per_token: float | None = None
|
||||
input_cost_per_token_batches: float | None = None
|
||||
output_cost_per_token_batches: float | None = None
|
||||
output_cost_per_image_token_batches: float | None = None
|
||||
input_cost_per_token_above_128k_tokens: float | None = None
|
||||
output_cost_per_token_above_128k_tokens: float | None = None
|
||||
output_vector_size: int | None = None
|
||||
|
|
@ -583,14 +584,24 @@ def data_errors() -> tuple[str, ...]:
|
|||
name for name in {case.name for case in _ALL_CASES} if sum(case.name == name for case in _ALL_CASES) > 1
|
||||
)
|
||||
input_rates: Final = tuple(
|
||||
(entry.input_cost_per_token, model)
|
||||
(entry.litellm_provider, entry.input_cost_per_token, model)
|
||||
for model, entry in COST_MAP.items()
|
||||
if entry.mode != "realtime"
|
||||
)
|
||||
shared_input_rate_details: Final = tuple(
|
||||
(
|
||||
provider,
|
||||
rate,
|
||||
tuple(
|
||||
model
|
||||
for candidate_provider, value, model in input_rates
|
||||
if candidate_provider == provider and value == rate
|
||||
),
|
||||
)
|
||||
for provider, rate in frozenset((provider, rate) for provider, rate, _ in input_rates if rate is not None)
|
||||
)
|
||||
shared_input_rates: Final = sorted(
|
||||
f"{rate}: {tuple(model for value, model in input_rates if value == rate)}"
|
||||
for rate in {value for value, _ in input_rates if value is not None}
|
||||
if sum(value == rate for value, _ in input_rates) > 1
|
||||
f"{rate}: {models}" for _, rate, models in shared_input_rate_details if len(models) > 1
|
||||
)
|
||||
recount_mismatches: Final = sorted(
|
||||
case.name
|
||||
|
|
|
|||
|
|
@ -750,6 +750,25 @@
|
|||
"output_cost_per_token": 2.16e-06,
|
||||
"litellm_provider": "together_ai",
|
||||
"mode": "completion"
|
||||
},
|
||||
"gemini/gemini-nano-banana-2.1": {
|
||||
"input_cost_per_token": 1.5e-06,
|
||||
"input_cost_per_token_batches": 7.5e-07,
|
||||
"litellm_provider": "gemini",
|
||||
"max_input_tokens": 131072,
|
||||
"max_output_tokens": 32768,
|
||||
"max_tokens": 32768,
|
||||
"mode": "image_generation",
|
||||
"output_cost_per_image": 0.0336,
|
||||
"output_cost_per_image_token": 3e-05,
|
||||
"output_cost_per_token": 7.5e-06,
|
||||
"output_cost_per_token_batches": 3.75e-06,
|
||||
"search_context_cost_per_query": {
|
||||
"search_context_size_high": 0.014,
|
||||
"search_context_size_low": 0.014,
|
||||
"search_context_size_medium": 0.014
|
||||
},
|
||||
"web_search_billing_unit": "per_query"
|
||||
}
|
||||
},
|
||||
"cases": [
|
||||
|
|
@ -31279,6 +31298,810 @@
|
|||
"min_completion_tokens": 9,
|
||||
"max_completion_tokens": 30
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-input_text",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "8e41cbb0a701 summarize the attached material in one line and name the city weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"allowed_openai_params": []
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": "scripted answer 8e41cbb0a701"
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 2000,
|
||||
"candidatesTokenCount": 100,
|
||||
"totalTokenCount": 2100,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 2000
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.00375,
|
||||
"input_cost": 0.003,
|
||||
"output_cost": 0.00075,
|
||||
"prompt_tokens": 2000,
|
||||
"completion_tokens": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-image_input",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "3f6c2d91a40e summarize the attached material in one line and name the city weather"
|
||||
},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+yVWQAAAAASUVORK5CYII=",
|
||||
"detail": "high"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"allowed_openai_params": []
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": "scripted answer 3f6c2d91a40e"
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 1680,
|
||||
"candidatesTokenCount": 100,
|
||||
"totalTokenCount": 1780,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 560
|
||||
},
|
||||
{
|
||||
"modality": "IMAGE",
|
||||
"tokenCount": 1120
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.00327,
|
||||
"input_cost": 0.00252,
|
||||
"output_cost": 0.00075,
|
||||
"prompt_tokens": 1680,
|
||||
"completion_tokens": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-output_text",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "d05a38c7e219 summarize the attached material in one line and name the city weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"allowed_openai_params": []
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": "scripted answer d05a38c7e219"
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 100,
|
||||
"candidatesTokenCount": 4000,
|
||||
"totalTokenCount": 4100,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 4000
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.03015,
|
||||
"input_cost": 0.00015,
|
||||
"output_cost": 0.03,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 4000
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-reasoning",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "c71e4f0092ba summarize the attached material in one line and name the city weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"reasoning_effort": "medium",
|
||||
"allowed_openai_params": [
|
||||
"reasoning_effort"
|
||||
]
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": "scripted answer c71e4f0092ba"
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 100,
|
||||
"candidatesTokenCount": 200,
|
||||
"thoughtsTokenCount": 1800,
|
||||
"totalTokenCount": 2100,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 200
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.01515,
|
||||
"input_cost": 0.00015,
|
||||
"output_cost": 0.015,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 2000
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-image_output_1k",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "6a9f2c13b847 summarize the attached material in one line and name the city weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"allowed_openai_params": []
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"inlineData": {
|
||||
"mimeType": "image/png",
|
||||
"data": "aGVsbG8="
|
||||
}
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 100,
|
||||
"candidatesTokenCount": 1120,
|
||||
"totalTokenCount": 1220,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "IMAGE",
|
||||
"tokenCount": 1120
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.03375,
|
||||
"input_cost": 0.00015,
|
||||
"output_cost": 0.0336,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 1120
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-image_output_2k",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "24df708bc351 summarize the attached material in one line and name the city weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"allowed_openai_params": []
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"inlineData": {
|
||||
"mimeType": "image/png",
|
||||
"data": "aGVsbG8="
|
||||
}
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 100,
|
||||
"candidatesTokenCount": 1680,
|
||||
"totalTokenCount": 1780,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "IMAGE",
|
||||
"tokenCount": 1680
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.05055,
|
||||
"input_cost": 0.00015,
|
||||
"output_cost": 0.0504,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 1680
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-image_output_4k",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "b8a315e4706c summarize the attached material in one line and name the city weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"allowed_openai_params": []
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"inlineData": {
|
||||
"mimeType": "image/png",
|
||||
"data": "aGVsbG8="
|
||||
}
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 100,
|
||||
"candidatesTokenCount": 2520,
|
||||
"totalTokenCount": 2620,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "IMAGE",
|
||||
"tokenCount": 2520
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.07575,
|
||||
"input_cost": 0.00015,
|
||||
"output_cost": 0.0756,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 2520
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-mixed_text_image_output",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "91ce62a50d3b summarize the attached material in one line and name the city weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"allowed_openai_params": []
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": "scripted answer 91ce62a50d3b"
|
||||
},
|
||||
{
|
||||
"inlineData": {
|
||||
"mimeType": "image/png",
|
||||
"data": "aGVsbG8="
|
||||
}
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 100,
|
||||
"candidatesTokenCount": 1220,
|
||||
"thoughtsTokenCount": 300,
|
||||
"totalTokenCount": 1620,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
},
|
||||
{
|
||||
"modality": "IMAGE",
|
||||
"tokenCount": 1120
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.03675,
|
||||
"input_cost": 0.00015,
|
||||
"output_cost": 0.0366,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 1520
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-web_search_single_query",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "f407b19d8e26 summarize the attached material in one line and name the city weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"tools": [
|
||||
{
|
||||
"googleSearch": {}
|
||||
}
|
||||
],
|
||||
"allowed_openai_params": [
|
||||
"web_search_options"
|
||||
]
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": "scripted answer f407b19d8e26"
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0,
|
||||
"groundingMetadata": {
|
||||
"webSearchQueries": [
|
||||
"query 0"
|
||||
]
|
||||
}
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 100,
|
||||
"candidatesTokenCount": 100,
|
||||
"totalTokenCount": 200,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.0149,
|
||||
"input_cost": 0.00015,
|
||||
"output_cost": 0.00075,
|
||||
"tool_usage_cost": 0.014,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 100
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "gemini-nano-banana-2.1-web_search_multiple_queries",
|
||||
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
|
||||
"model": "gemini/gemini-nano-banana-2.1",
|
||||
"request": {
|
||||
"model": "$MODEL",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "a36d5f80c297 summarize the attached material in one line and name the city weather"
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"stream": false,
|
||||
"tools": [
|
||||
{
|
||||
"googleSearch": {}
|
||||
}
|
||||
],
|
||||
"allowed_openai_params": [
|
||||
"web_search_options"
|
||||
]
|
||||
},
|
||||
"response": {
|
||||
"content_type": "application/json",
|
||||
"body": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": "scripted answer a36d5f80c297"
|
||||
}
|
||||
],
|
||||
"role": "model"
|
||||
},
|
||||
"finishReason": "STOP",
|
||||
"index": 0,
|
||||
"groundingMetadata": {
|
||||
"webSearchQueries": [
|
||||
"query 0",
|
||||
"query 1",
|
||||
"query 2",
|
||||
"query 3"
|
||||
]
|
||||
}
|
||||
}
|
||||
],
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 100,
|
||||
"candidatesTokenCount": 100,
|
||||
"totalTokenCount": 200,
|
||||
"promptTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
],
|
||||
"candidatesTokensDetails": [
|
||||
{
|
||||
"modality": "TEXT",
|
||||
"tokenCount": 100
|
||||
}
|
||||
]
|
||||
},
|
||||
"modelVersion": "gemini-nano-banana-2.1"
|
||||
}
|
||||
},
|
||||
"expected": {
|
||||
"spend": 0.0569,
|
||||
"input_cost": 0.00015,
|
||||
"output_cost": 0.00075,
|
||||
"tool_usage_cost": 0.056,
|
||||
"prompt_tokens": 100,
|
||||
"completion_tokens": 100
|
||||
}
|
||||
}
|
||||
],
|
||||
"batch_cases": [
|
||||
|
|
|
|||
446
tests/integration/sdk/test_vertex_gemini_image_batch_cost_sdk.py
Normal file
446
tests/integration/sdk/test_vertex_gemini_image_batch_cost_sdk.py
Normal file
|
|
@ -0,0 +1,446 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import textwrap
|
||||
import threading
|
||||
from collections.abc import Generator
|
||||
from contextlib import contextmanager
|
||||
from dataclasses import dataclass
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import pytest
|
||||
from integration._support.tls import server_context, write_self_signed_cert
|
||||
from integration._support.vertex import service_account_json
|
||||
from integration._support.wire import Reply, Request, Wire, wire_server
|
||||
from pydantic import JsonValue, TypeAdapter
|
||||
|
||||
pytestmark: Final = pytest.mark.timeout(180)
|
||||
|
||||
_PROJECT: Final = "scripted-gemini-batch-project"
|
||||
_MODEL: Final = "gemini-nano-banana-2.1"
|
||||
_BATCH_ID: Final = "scripted-batch"
|
||||
_CALLBACK_PREFIX: Final = "CALLBACK:"
|
||||
_REGISTRY_PREFIX: Final = "REGISTRY:"
|
||||
_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue])
|
||||
_IMAGE_PART: Final = {
|
||||
"inlineData": {
|
||||
"mimeType": "image/png",
|
||||
"data": "aGVsbG8=",
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
def _pipe(source: socket.socket, destination: socket.socket) -> None:
|
||||
try:
|
||||
for payload in iter(lambda: source.recv(65536), b""):
|
||||
destination.sendall(payload)
|
||||
except OSError:
|
||||
return
|
||||
finally:
|
||||
try:
|
||||
destination.shutdown(socket.SHUT_WR)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
@contextmanager
|
||||
def _redirect_https_host(host: str, port: int, *, to_port: int) -> Generator[str, None, None]:
|
||||
"""Yields an HTTPS_PROXY url that sends host:port to 127.0.0.1:to_port and refuses anything else."""
|
||||
authority: Final = f"{host}:{port}"
|
||||
|
||||
class _RedirectHandler(BaseHTTPRequestHandler):
|
||||
protocol_version = "HTTP/1.1"
|
||||
timeout = 30
|
||||
|
||||
def do_CONNECT(self) -> None:
|
||||
if self.path != authority:
|
||||
self.send_error(403)
|
||||
return
|
||||
|
||||
self.send_response(200, "Connection Established")
|
||||
self.end_headers()
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", to_port), timeout=10) as upstream:
|
||||
client_to_upstream: Final = threading.Thread(
|
||||
target=_pipe, args=(self.connection, upstream), daemon=True
|
||||
)
|
||||
upstream_to_client: Final = threading.Thread(
|
||||
target=_pipe, args=(upstream, self.connection), daemon=True
|
||||
)
|
||||
client_to_upstream.start()
|
||||
upstream_to_client.start()
|
||||
client_to_upstream.join(timeout=30)
|
||||
upstream_to_client.join(timeout=30)
|
||||
except OSError:
|
||||
return
|
||||
|
||||
def log_message(self, format: str, *args: object) -> None:
|
||||
pass
|
||||
|
||||
class _RedirectProxy(ThreadingHTTPServer):
|
||||
daemon_threads = True
|
||||
request_queue_size = 16
|
||||
|
||||
with _RedirectProxy(("127.0.0.1", 0), _RedirectHandler) as server:
|
||||
thread: Final = threading.Thread(target=server.serve_forever, kwargs={"poll_interval": 0.05})
|
||||
thread.start()
|
||||
try:
|
||||
yield f"http://127.0.0.1:{server.server_port}"
|
||||
finally:
|
||||
server.shutdown()
|
||||
thread.join(timeout=6)
|
||||
assert not thread.is_alive(), "CONNECT tunnel server survived cleanup"
|
||||
server.server_close()
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class _BatchScenario:
|
||||
name: str
|
||||
prompt_tokens: int
|
||||
prompt_details: tuple[tuple[str, int], ...]
|
||||
candidate_tokens: int
|
||||
candidate_details: tuple[tuple[str, int], ...]
|
||||
thoughts_tokens: int
|
||||
expected_prompt_cost: float
|
||||
expected_completion_cost: float
|
||||
use_explicit_image_batch_rate_override: bool = False
|
||||
|
||||
|
||||
_SCENARIOS: Final = (
|
||||
_BatchScenario(
|
||||
name="batch_input_text_and_image",
|
||||
prompt_tokens=1680,
|
||||
prompt_details=(("TEXT", 560), ("IMAGE", 1120)),
|
||||
candidate_tokens=100,
|
||||
candidate_details=(("TEXT", 100),),
|
||||
thoughts_tokens=0,
|
||||
expected_prompt_cost=0.00126,
|
||||
expected_completion_cost=0.000375,
|
||||
),
|
||||
_BatchScenario(
|
||||
name="batch_text_and_thinking_output",
|
||||
prompt_tokens=100,
|
||||
prompt_details=(("TEXT", 100),),
|
||||
candidate_tokens=400,
|
||||
candidate_details=(("TEXT", 400),),
|
||||
thoughts_tokens=600,
|
||||
expected_prompt_cost=0.000075,
|
||||
expected_completion_cost=0.00375,
|
||||
),
|
||||
_BatchScenario(
|
||||
name="batch_image_output_1k",
|
||||
prompt_tokens=100,
|
||||
prompt_details=(("TEXT", 100),),
|
||||
candidate_tokens=1120,
|
||||
candidate_details=(("IMAGE", 1120),),
|
||||
thoughts_tokens=0,
|
||||
expected_prompt_cost=0.000075,
|
||||
expected_completion_cost=0.0168,
|
||||
),
|
||||
_BatchScenario(
|
||||
name="batch_image_output_2k",
|
||||
prompt_tokens=100,
|
||||
prompt_details=(("TEXT", 100),),
|
||||
candidate_tokens=1680,
|
||||
candidate_details=(("IMAGE", 1680),),
|
||||
thoughts_tokens=0,
|
||||
expected_prompt_cost=0.000075,
|
||||
expected_completion_cost=0.0252,
|
||||
),
|
||||
_BatchScenario(
|
||||
name="batch_image_output_4k",
|
||||
prompt_tokens=100,
|
||||
prompt_details=(("TEXT", 100),),
|
||||
candidate_tokens=3780,
|
||||
candidate_details=(("IMAGE", 3780),),
|
||||
thoughts_tokens=0,
|
||||
expected_prompt_cost=0.000075,
|
||||
expected_completion_cost=0.0567,
|
||||
),
|
||||
_BatchScenario(
|
||||
name="batch_mixed_text_image_output",
|
||||
prompt_tokens=100,
|
||||
prompt_details=(("TEXT", 100),),
|
||||
candidate_tokens=1220,
|
||||
candidate_details=(("TEXT", 100), ("IMAGE", 1120)),
|
||||
thoughts_tokens=300,
|
||||
expected_prompt_cost=0.000075,
|
||||
expected_completion_cost=0.0183,
|
||||
),
|
||||
_BatchScenario(
|
||||
name="batch_image_rate_uses_explicit_override",
|
||||
prompt_tokens=100,
|
||||
prompt_details=(("TEXT", 100),),
|
||||
candidate_tokens=1120,
|
||||
candidate_details=(("IMAGE", 1120),),
|
||||
thoughts_tokens=0,
|
||||
expected_prompt_cost=0.000075,
|
||||
expected_completion_cost=0.0224,
|
||||
use_explicit_image_batch_rate_override=True,
|
||||
),
|
||||
)
|
||||
|
||||
_SDK_SCRIPT: Final = textwrap.dedent(
|
||||
"""
|
||||
import asyncio, json, os
|
||||
from typing import Final
|
||||
|
||||
import litellm
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
|
||||
from litellm.types.utils import ModelInfo
|
||||
|
||||
model: Final = "gemini-nano-banana-2.1"
|
||||
|
||||
class CaptureLogger(CustomLogger):
|
||||
async def async_log_success_event(
|
||||
self,
|
||||
kwargs: dict[str, object],
|
||||
response_obj: object,
|
||||
start_time: object,
|
||||
end_time: object,
|
||||
) -> None:
|
||||
standard: Final = kwargs.get("standard_logging_object")
|
||||
hidden: Final = getattr(response_obj, "_hidden_params", None)
|
||||
if not isinstance(standard, dict):
|
||||
raise RuntimeError("success callback omitted standard_logging_object")
|
||||
if standard.get("call_type") != "aretrieve_batch":
|
||||
return
|
||||
print(
|
||||
"CALLBACK:"
|
||||
+ json.dumps(
|
||||
{
|
||||
"response_cost": standard.get("response_cost"),
|
||||
"cost_breakdown": standard.get("cost_breakdown"),
|
||||
"batch_response_cost": hidden.get("response_cost")
|
||||
if isinstance(hidden, dict)
|
||||
else None,
|
||||
}
|
||||
),
|
||||
flush=True,
|
||||
)
|
||||
|
||||
async def main() -> None:
|
||||
has_explicit_image_batch_rate_override: Final = (
|
||||
os.environ["USE_EXPLICIT_IMAGE_BATCH_RATE_OVERRIDE"] == "1"
|
||||
)
|
||||
if has_explicit_image_batch_rate_override:
|
||||
pricing: Final[ModelInfo] = {
|
||||
**litellm.model_cost["vertex_ai/gemini-nano-banana-2.1"],
|
||||
"output_cost_per_image_token_batches": 2e-5,
|
||||
}
|
||||
litellm.register_model({f"vertex_ai/{model}": pricing}, persist_across_reloads=False)
|
||||
litellm.user_url_validation = False
|
||||
litellm.disable_vertex_batch_output_transformation = os.environ["TRANSFORM_OUTPUT"] != "1"
|
||||
info: Final = litellm.get_model_info(model=model, custom_llm_provider="vertex_ai")
|
||||
assert info["key"] == f"vertex_ai/{model}", info
|
||||
assert info["litellm_provider"] == "vertex_ai-language-models", info
|
||||
print(
|
||||
"REGISTRY:"
|
||||
+ json.dumps({"key": info["key"], "provider": info["litellm_provider"]}),
|
||||
flush=True,
|
||||
)
|
||||
logger: Final = CaptureLogger()
|
||||
litellm.logging_callback_manager.add_litellm_async_success_callback(logger)
|
||||
await litellm.aretrieve_batch(
|
||||
"scripted-batch",
|
||||
custom_llm_provider="vertex_ai",
|
||||
model=model,
|
||||
api_base=os.environ["VERTEX_API_BASE"],
|
||||
vertex_project=os.environ["VERTEX_PROJECT"],
|
||||
vertex_location="us-central1",
|
||||
vertex_credentials=os.environ["VERTEX_CREDENTIALS"],
|
||||
gcs_bucket_name="scripted-bucket",
|
||||
num_retries=0,
|
||||
)
|
||||
await asyncio.sleep(0)
|
||||
await GLOBAL_LOGGING_WORKER.flush()
|
||||
|
||||
asyncio.run(main())
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
def _prediction_jsonl(scenario: _BatchScenario) -> bytes:
|
||||
prompt_details: Final = tuple(
|
||||
{"modality": modality, "tokenCount": count} for modality, count in scenario.prompt_details
|
||||
)
|
||||
candidate_details: Final = tuple(
|
||||
{"modality": modality, "tokenCount": count} for modality, count in scenario.candidate_details
|
||||
)
|
||||
candidate_parts: Final = tuple(
|
||||
{"text": "scripted batch output"} if modality == "TEXT" else _IMAGE_PART
|
||||
for modality, _ in scenario.candidate_details
|
||||
)
|
||||
usage_metadata: Final[dict[str, JsonValue]] = {
|
||||
"promptTokenCount": scenario.prompt_tokens,
|
||||
"candidatesTokenCount": scenario.candidate_tokens,
|
||||
"totalTokenCount": scenario.prompt_tokens + scenario.candidate_tokens + scenario.thoughts_tokens,
|
||||
"promptTokensDetails": prompt_details,
|
||||
"candidatesTokensDetails": candidate_details,
|
||||
**({"thoughtsTokenCount": scenario.thoughts_tokens} if scenario.thoughts_tokens else {}),
|
||||
}
|
||||
row: Final[dict[str, JsonValue]] = {
|
||||
"request": {
|
||||
"contents": [
|
||||
{
|
||||
"role": "user",
|
||||
"parts": [{"text": "Scripted Vertex Gemini batch request"}],
|
||||
}
|
||||
]
|
||||
},
|
||||
"status": "",
|
||||
"response": {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {"parts": candidate_parts, "role": "model"},
|
||||
"finishReason": "STOP",
|
||||
"index": 0,
|
||||
}
|
||||
],
|
||||
"usageMetadata": usage_metadata,
|
||||
"modelVersion": _MODEL,
|
||||
},
|
||||
}
|
||||
return json.dumps(row, separators=(",", ":")).encode("utf-8") + b"\n"
|
||||
|
||||
|
||||
def _api_reply(request: Request) -> Reply:
|
||||
if request.method == "POST" and request.target == "/_oauth/token":
|
||||
return Reply(body=b'{"access_token":"scripted-token","expires_in":3600,"token_type":"Bearer"}')
|
||||
if request.method == "GET" and request.target.endswith(f"/batchPredictionJobs/{_BATCH_ID}"):
|
||||
return Reply(
|
||||
body=json.dumps(
|
||||
{
|
||||
"name": (f"projects/{_PROJECT}/locations/us-central1/batchPredictionJobs/{_BATCH_ID}"),
|
||||
"state": "JOB_STATE_SUCCEEDED",
|
||||
"outputInfo": {
|
||||
"gcsOutputDirectory": (
|
||||
"gs://scripted-bucket/litellm-vertex-files/"
|
||||
"publishers/google/models/gemini-nano-banana-2.1/scripted-prefix"
|
||||
)
|
||||
},
|
||||
"createTime": "2026-10-06T00:00:00.000Z",
|
||||
}
|
||||
).encode("utf-8")
|
||||
)
|
||||
return Reply(status=404, body=b"{}")
|
||||
|
||||
|
||||
def _gcs_reply(content: bytes):
|
||||
def respond(request: Request) -> Reply:
|
||||
if request.method == "GET" and request.target.endswith("predictions.jsonl?alt=media"):
|
||||
return Reply(body=content, content_type="application/jsonl")
|
||||
return Reply(status=404, body=b"{}")
|
||||
|
||||
return respond
|
||||
|
||||
|
||||
def _subprocess_environment(
|
||||
api: Wire,
|
||||
certificate: Path,
|
||||
credentials: str,
|
||||
proxy_url: str,
|
||||
scenario: _BatchScenario,
|
||||
transform_output: bool,
|
||||
) -> dict[str, str]:
|
||||
repo_root: Final = Path(__file__).resolve().parents[3]
|
||||
python_path: Final = os.pathsep.join(path for path in (str(repo_root), os.environ.get("PYTHONPATH")) if path)
|
||||
return {
|
||||
**os.environ,
|
||||
"HTTPS_PROXY": proxy_url,
|
||||
"https_proxy": proxy_url,
|
||||
"HTTP_PROXY": "",
|
||||
"http_proxy": "",
|
||||
"SSL_CERT_FILE": str(certificate),
|
||||
"NO_PROXY": "127.0.0.1,localhost",
|
||||
"no_proxy": "127.0.0.1,localhost",
|
||||
"VERTEX_API_BASE": api.url,
|
||||
"VERTEX_CREDENTIALS": credentials,
|
||||
"VERTEX_PROJECT": _PROJECT,
|
||||
"USE_EXPLICIT_IMAGE_BATCH_RATE_OVERRIDE": (
|
||||
"1" if scenario.use_explicit_image_batch_rate_override else "0"
|
||||
),
|
||||
"TRANSFORM_OUTPUT": "1" if transform_output else "0",
|
||||
"LITELLM_LOCAL_MODEL_COST_MAP": "True",
|
||||
"PYTHONPATH": python_path,
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("transform_output", (False, True), ids=("untransformed", "transformed"))
|
||||
@pytest.mark.parametrize("scenario", _SCENARIOS, ids=tuple(case.name for case in _SCENARIOS))
|
||||
def test_aretrieve_batch_costs_native_gemini_image_tokens(
|
||||
scenario: _BatchScenario, transform_output: bool, tmp_path: Path
|
||||
) -> None:
|
||||
certificate, key = write_self_signed_cert(tmp_path, names=("storage.googleapis.com",))
|
||||
tls: Final = server_context(certificate, key)
|
||||
row: Final = _prediction_jsonl(scenario)
|
||||
|
||||
with wire_server(_api_reply) as api:
|
||||
credentials: Final = service_account_json(_PROJECT, api.url)
|
||||
with wire_server(_gcs_reply(row), tls=tls) as gcs:
|
||||
gcs_port: Final = urlsplit(gcs.url).port
|
||||
assert gcs_port is not None
|
||||
with _redirect_https_host("storage.googleapis.com", 443, to_port=gcs_port) as proxy_url:
|
||||
environment: Final = _subprocess_environment(
|
||||
api, certificate, credentials, proxy_url, scenario, transform_output
|
||||
)
|
||||
outcome: Final = subprocess.run(
|
||||
[sys.executable, "-P", "-c", _SDK_SCRIPT],
|
||||
env=environment,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=60,
|
||||
check=False,
|
||||
)
|
||||
|
||||
assert outcome.returncode == 0, (outcome.stdout, outcome.stderr)
|
||||
registry_events: Final = tuple(
|
||||
_JSON_OBJECT.validate_json(line[len(_REGISTRY_PREFIX) :])
|
||||
for line in outcome.stdout.splitlines()
|
||||
if line.startswith(_REGISTRY_PREFIX)
|
||||
)
|
||||
callback_events: Final = tuple(
|
||||
_JSON_OBJECT.validate_json(line[len(_CALLBACK_PREFIX) :])
|
||||
for line in outcome.stdout.splitlines()
|
||||
if line.startswith(_CALLBACK_PREFIX)
|
||||
)
|
||||
expected_registry_key: Final = f"vertex_ai/{_MODEL}"
|
||||
assert registry_events == (
|
||||
{"key": expected_registry_key, "provider": "vertex_ai-language-models"},
|
||||
), outcome.stdout
|
||||
assert len(callback_events) == 1, (outcome.stdout, outcome.stderr)
|
||||
event: Final = callback_events[0]
|
||||
assert event["response_cost"] == pytest.approx(
|
||||
scenario.expected_prompt_cost + scenario.expected_completion_cost
|
||||
), event
|
||||
assert event["batch_response_cost"] == pytest.approx(
|
||||
scenario.expected_prompt_cost + scenario.expected_completion_cost
|
||||
), event
|
||||
breakdown: Final = event["cost_breakdown"]
|
||||
assert isinstance(breakdown, dict), event
|
||||
assert breakdown["input_cost"] == pytest.approx(scenario.expected_prompt_cost), event
|
||||
assert breakdown["output_cost"] == pytest.approx(scenario.expected_completion_cost), event
|
||||
assert breakdown["total_cost"] == pytest.approx(
|
||||
scenario.expected_prompt_cost + scenario.expected_completion_cost
|
||||
), event
|
||||
|
||||
gcs_requests: Final = gcs.drain()
|
||||
assert any(
|
||||
request.method == "GET" and request.target.endswith("predictions.jsonl?alt=media")
|
||||
for request in gcs_requests
|
||||
), tuple((request.method, request.target) for request in gcs_requests)
|
||||
api_requests: Final = api.drain()
|
||||
assert any(
|
||||
request.method == "GET"
|
||||
and request.target.endswith(f"/batchPredictionJobs/{_BATCH_ID}")
|
||||
and request.headers.get("authorization") == "Bearer scripted-token"
|
||||
for request in api_requests
|
||||
), tuple((request.method, request.target) for request in api_requests)
|
||||
|
|
@ -13,6 +13,7 @@ from litellm.cost_calculator import (
|
|||
BaseTokenUsageProcessor,
|
||||
RealtimeAPITokenUsageProcessor,
|
||||
ResponsesWebSocketTokenUsageProcessor,
|
||||
batch_cost_calculator,
|
||||
completion_cost,
|
||||
cost_per_token,
|
||||
handle_realtime_stream_cost_calculation,
|
||||
|
|
@ -27,6 +28,7 @@ from litellm.types.utils import (
|
|||
CacheCreationTokenDetails,
|
||||
CallTypes,
|
||||
Choices,
|
||||
CompletionTokensDetailsWrapper,
|
||||
EmbeddingResponse,
|
||||
ImageObject,
|
||||
ImageResponse,
|
||||
|
|
@ -3448,8 +3450,6 @@ def _batch_cache_usage() -> Usage:
|
|||
|
||||
|
||||
def test_batch_cost_calculator_prices_multimodal_tokens_at_modality_rates():
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
model_info: ModelInfo = {
|
||||
"input_cost_per_token_batches": 1e-7,
|
||||
"input_cost_per_audio_token_batches": 3.25e-6,
|
||||
|
|
@ -3478,8 +3478,6 @@ def test_batch_cost_calculator_prices_multimodal_tokens_at_modality_rates():
|
|||
|
||||
|
||||
def test_batch_cost_calculator_falls_back_to_text_batch_rate_for_modalities():
|
||||
from litellm.cost_calculator import batch_cost_calculator
|
||||
|
||||
model_info: ModelInfo = {"input_cost_per_token_batches": 1e-7}
|
||||
usage = Usage(
|
||||
prompt_tokens=100,
|
||||
|
|
@ -3498,6 +3496,87 @@ def test_batch_cost_calculator_falls_back_to_text_batch_rate_for_modalities():
|
|||
assert prompt_cost == pytest.approx(100 * 1e-7)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("image_batch_rate", "image_tokens", "expected_completion_cost"),
|
||||
(
|
||||
(5e-5, 80, 0.00408),
|
||||
(None, 80, 0.00328),
|
||||
(5e-5, 140, 0.006),
|
||||
),
|
||||
)
|
||||
def test_batch_cost_calculator_prices_image_completion_tokens_at_image_batch_rate(
|
||||
image_batch_rate: float | None, image_tokens: int, expected_completion_cost: float
|
||||
) -> None:
|
||||
model_info: Final[ModelInfo] = (
|
||||
{
|
||||
"output_cost_per_token_batches": 2e-6,
|
||||
"output_cost_per_image_token": 8e-5,
|
||||
"output_cost_per_image_token_batches": image_batch_rate,
|
||||
}
|
||||
if image_batch_rate is not None
|
||||
else {
|
||||
"output_cost_per_token_batches": 2e-6,
|
||||
"output_cost_per_image_token": 8e-5,
|
||||
}
|
||||
)
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=0,
|
||||
completion_tokens=120,
|
||||
total_tokens=120,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(image_tokens=image_tokens),
|
||||
)
|
||||
|
||||
costs: Final = batch_cost_calculator(
|
||||
usage=usage,
|
||||
model="gemini-nano-banana-2.1",
|
||||
custom_llm_provider="vertex_ai",
|
||||
model_info=model_info,
|
||||
)
|
||||
|
||||
assert costs[1] == pytest.approx(expected_completion_cost)
|
||||
|
||||
|
||||
def test_batch_cost_calculator_merges_image_only_deployment_rate_with_global_pricing(
|
||||
_local_model_cost_map: None,
|
||||
) -> None:
|
||||
model: Final = "gemini/gemini-3-pro-image"
|
||||
global_model_info: Final = litellm.get_model_info(model=model, custom_llm_provider="gemini")
|
||||
input_batch_rate: Final = global_model_info["input_cost_per_token_batches"]
|
||||
output_batch_rate: Final = global_model_info["output_cost_per_token_batches"]
|
||||
global_image_batch_rate: Final = global_model_info["output_cost_per_image_token_batches"]
|
||||
assert input_batch_rate is not None
|
||||
assert output_batch_rate is not None
|
||||
assert global_image_batch_rate is not None
|
||||
|
||||
deployment_image_batch_rate: Final = 1e-6
|
||||
assert deployment_image_batch_rate != global_image_batch_rate
|
||||
|
||||
text_tokens: Final = 300
|
||||
image_tokens: Final = 200
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=1_000,
|
||||
completion_tokens=text_tokens + image_tokens,
|
||||
total_tokens=1_500,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(image_tokens=image_tokens),
|
||||
)
|
||||
model_info: Final = ModelInfo(output_cost_per_image_token_batches=deployment_image_batch_rate)
|
||||
|
||||
prompt_cost, completion_cost = batch_cost_calculator(
|
||||
usage=usage,
|
||||
model=model,
|
||||
custom_llm_provider="gemini",
|
||||
model_info=model_info,
|
||||
)
|
||||
|
||||
assert prompt_cost > 0
|
||||
assert completion_cost > 0
|
||||
assert prompt_cost == pytest.approx(usage.prompt_tokens * input_batch_rate)
|
||||
expected_completion_cost: Final = text_tokens * output_batch_rate + image_tokens * deployment_image_batch_rate
|
||||
global_rate_completion_cost: Final = text_tokens * output_batch_rate + image_tokens * global_image_batch_rate
|
||||
assert completion_cost == pytest.approx(expected_completion_cost)
|
||||
assert completion_cost != pytest.approx(global_rate_completion_cost)
|
||||
|
||||
|
||||
def test_batch_cost_calculator_prices_cache_creation_tokens_at_cache_write_rate():
|
||||
"""
|
||||
LIT-4008 regression: anthropic batch usage is dominated by cache tokens.
|
||||
|
|
|
|||
|
|
@ -665,6 +665,7 @@ def validate_model_cost_values(model_data, exceptions=None):
|
|||
"input_cost_per_audio_token",
|
||||
"output_cost_per_audio_token",
|
||||
"output_cost_per_image_token",
|
||||
"output_cost_per_image_token_batches",
|
||||
"input_cost_per_video_token",
|
||||
"output_cost_per_video_token",
|
||||
"input_cost_per_audio_per_second",
|
||||
|
|
@ -903,6 +904,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
|
|||
"output_cost_per_image_2K": {"type": "number"},
|
||||
"output_cost_per_image_4K": {"type": "number"},
|
||||
"output_cost_per_image_token": {"type": "number"},
|
||||
"output_cost_per_image_token_batches": {"type": "number"},
|
||||
"output_cost_per_video_token": {"type": "number"},
|
||||
"output_cost_per_pixel": {"type": "number"},
|
||||
"output_cost_per_second": {"type": "number"},
|
||||
|
|
|
|||
4
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
4
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -35445,6 +35445,8 @@ export interface components {
|
|||
output_cost_per_image_512?: number | null;
|
||||
/** Output Cost Per Image Token */
|
||||
output_cost_per_image_token?: number | null;
|
||||
/** Output Cost Per Image Token Batches */
|
||||
output_cost_per_image_token_batches?: number | null;
|
||||
/** Output Cost Per Pixel */
|
||||
output_cost_per_pixel?: number | null;
|
||||
/** Output Cost Per Reasoning Token */
|
||||
|
|
@ -50281,6 +50283,8 @@ export interface components {
|
|||
output_cost_per_image_512?: number | null;
|
||||
/** Output Cost Per Image Token */
|
||||
output_cost_per_image_token?: number | null;
|
||||
/** Output Cost Per Image Token Batches */
|
||||
output_cost_per_image_token_batches?: number | null;
|
||||
/** Output Cost Per Pixel */
|
||||
output_cost_per_pixel?: number | null;
|
||||
/** Output Cost Per Reasoning Token */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue