fix(cost): price batch image output tokens at the batch image rate (#44897)

* fix(cost): price batch image completion tokens at image batch rate

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(integration): cover default vertex batch output transformation for image cost

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(cost): sync model prices schema

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* refactor(cost): simplify batch completion cost with rate helpers

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(cost): keep global batch pricing fallback for image-only deployment rates

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* refactor(test): rename connect tunnel helper to https redirect

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(cost): merge deployment batch rates over global pricing

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(model_prices): put flash-image batch image rate on the GA row

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* chore(cost): drop redundant comment in batch fallback

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* refactor(cost): use _batch_or_half for the batch image rate

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(cost): price nano banana 2.1 fixtures off the real cost map rows

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test(cost): use the vertex 4k image token count in the batch cost test

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: kerry <kerry@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-06 22:18:28 +00:00 • committed by GitHub
parent 7566f9bc54
commit 191967c207
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
12 changed files with 1412 additions and 18 deletions

View file

@ -2568,6 +2568,20 @@ def _batch_rate(
return fallback if rate is None else rate
def _batch_or_half(batch_rate: float | None, standard_rate: float | None, default: float = 0.0) -> float:
if batch_rate is not None:
return batch_rate
if standard_rate is not None:
return standard_rate / 2
return default
def _completion_image_tokens(usage: Usage) -> int:
details: Final = usage.completion_tokens_details
image_tokens: Final = (details.image_tokens or 0) if details is not None else 0
return min(image_tokens, usage.completion_tokens)
def batch_cost_calculator(
usage: Usage,
model: str,
@ -2607,13 +2621,14 @@ def batch_cost_calculator(
"output_cost_per_token",
)
):
# model_info was provided (e.g. deployment metadata with only id/db_model)
# but carries no pricing fields. Fall back to the global pricing table so
# that standard model pricing is used instead of silently returning $0.
deployment_info: Final = model_info
try:
global_info: Final = litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider)
if global_info:
model_info = global_info
model_info = {
**global_info,
**{key: value for key, value in deployment_info.items() if value is not None},
}
except Exception:
pass
@ -2647,12 +2662,15 @@ def batch_cost_calculator(
cache_creation_cost: Final = model_info.get("cache_creation_input_token_cost") or input_cost_per_token
total_prompt_cost += cache_creation_tokens * cache_creation_cost / 2
if batch_rates.output is not None:
total_completion_cost = usage.completion_tokens * batch_rates.output
elif output_cost_per_token:
total_completion_cost = (
usage.completion_tokens * (output_cost_per_token) / 2
) # batch cost is usually half of the regular token cost
text_rate: Final = _batch_or_half(batch_rates.output, output_cost_per_token)
image_rate: Final = _batch_or_half(
model_info.get("output_cost_per_image_token_batches"),
model_info.get("output_cost_per_image_token"),
default=text_rate,
)
image_tokens: Final = _completion_image_tokens(usage)
text_tokens: Final = usage.completion_tokens - image_tokens
total_completion_cost = text_tokens * text_rate + image_tokens * image_rate
uplift: Final = _get_regional_uplift_multiplier(model_info, data_residency)
if uplift != 1.0:

View file

@ -29288,6 +29288,7 @@
"mode": "image_generation",
"output_cost_per_image": 0.134,
"output_cost_per_image_token": 0.00012,
"output_cost_per_image_token_batches": 6e-05,
"output_cost_per_token": 1.2e-05,
"rpm": 1000,
"tpm": 4000000,
@ -29379,6 +29380,7 @@
"mode": "image_generation",
"output_cost_per_image": 0.045,
"output_cost_per_image_token": 6e-05,
"output_cost_per_image_token_batches": 3e-05,
"output_cost_per_token": 3e-06,
"output_cost_per_token_batches": 1.5e-06,
"rpm": 1000,

View file

@ -376,6 +376,7 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False):
output_cost_per_image: float | None
output_cost_per_pixel: ReadOnly[float | None]
output_cost_per_image_token: float | None
output_cost_per_image_token_batches: ReadOnly[float | None]
output_cost_per_video_token: float | None # for gemini omni models with video output
output_vector_size: int | None
output_cost_per_reasoning_token: float | None
@ -3820,6 +3821,7 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
output_cost_per_character_above_128k_tokens: float | None = None
output_cost_per_image: float | None = None
output_cost_per_image_token: float | None = None
output_cost_per_image_token_batches: float | None = None
output_cost_per_video_token: float | None = None
output_cost_per_reasoning_token: float | None = None
output_cost_per_reasoning_token_flex: float | None = None

View file

@ -6271,6 +6271,7 @@ def _get_model_info_helper(
output_cost_per_image=_model_info.get("output_cost_per_image", None),
output_cost_per_pixel=_model_info.get("output_cost_per_pixel", None),
output_cost_per_image_token=_model_info.get("output_cost_per_image_token", None),
output_cost_per_image_token_batches=_model_info.get("output_cost_per_image_token_batches", None),
output_cost_per_video_token=_model_info.get("output_cost_per_video_token", None),
output_vector_size=_model_info.get("output_vector_size", None),
citation_cost_per_token=_model_info.get("citation_cost_per_token", None),

View file

@ -29288,6 +29288,7 @@
"mode": "image_generation",
"output_cost_per_image": 0.134,
"output_cost_per_image_token": 0.00012,
"output_cost_per_image_token_batches": 6e-05,
"output_cost_per_token": 1.2e-05,
"rpm": 1000,
"tpm": 4000000,
@ -29379,6 +29380,7 @@
"mode": "image_generation",
"output_cost_per_image": 0.045,
"output_cost_per_image_token": 6e-05,
"output_cost_per_image_token_batches": 3e-05,
"output_cost_per_token": 3e-06,
"output_cost_per_token_batches": 1.5e-06,
"rpm": 1000,

View file

@ -710,6 +710,10 @@
"type": "number",
"minimum": 0
},
"output_cost_per_image_token_batches": {
"type": "number",
"minimum": 0
},
"output_cost_per_pixel": {
"type": "number",
"minimum": 0

View file

@ -49,6 +49,7 @@ class CostMapEntry(BaseModel):
output_cost_per_token: float | None = None
input_cost_per_token_batches: float | None = None
output_cost_per_token_batches: float | None = None
output_cost_per_image_token_batches: float | None = None
input_cost_per_token_above_128k_tokens: float | None = None
output_cost_per_token_above_128k_tokens: float | None = None
output_vector_size: int | None = None
@ -583,14 +584,24 @@ def data_errors() -> tuple[str, ...]:
name for name in {case.name for case in _ALL_CASES} if sum(case.name == name for case in _ALL_CASES) > 1
)
input_rates: Final = tuple(
(entry.input_cost_per_token, model)
(entry.litellm_provider, entry.input_cost_per_token, model)
for model, entry in COST_MAP.items()
if entry.mode != "realtime"
)
shared_input_rate_details: Final = tuple(
(
provider,
rate,
tuple(
model
for candidate_provider, value, model in input_rates
if candidate_provider == provider and value == rate
),
)
for provider, rate in frozenset((provider, rate) for provider, rate, _ in input_rates if rate is not None)
)
shared_input_rates: Final = sorted(
f"{rate}: {tuple(model for value, model in input_rates if value == rate)}"
for rate in {value for value, _ in input_rates if value is not None}
if sum(value == rate for value, _ in input_rates) > 1
f"{rate}: {models}" for _, rate, models in shared_input_rate_details if len(models) > 1
)
recount_mismatches: Final = sorted(
case.name

View file

@ -750,6 +750,25 @@
"output_cost_per_token": 2.16e-06,
"litellm_provider": "together_ai",
"mode": "completion"
},
"gemini/gemini-nano-banana-2.1": {
"input_cost_per_token": 1.5e-06,
"input_cost_per_token_batches": 7.5e-07,
"litellm_provider": "gemini",
"max_input_tokens": 131072,
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "image_generation",
"output_cost_per_image": 0.0336,
"output_cost_per_image_token": 3e-05,
"output_cost_per_token": 7.5e-06,
"output_cost_per_token_batches": 3.75e-06,
"search_context_cost_per_query": {
"search_context_size_high": 0.014,
"search_context_size_low": 0.014,
"search_context_size_medium": 0.014
},
"web_search_billing_unit": "per_query"
}
},
"cases": [
@ -31279,6 +31298,810 @@
"min_completion_tokens": 9,
"max_completion_tokens": 30
}
},
{
"name": "gemini-nano-banana-2.1-input_text",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "8e41cbb0a701 summarize the attached material in one line and name the city weather"
}
]
}
],
"stream": false,
"allowed_openai_params": []
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"text": "scripted answer 8e41cbb0a701"
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0
}
],
"usageMetadata": {
"promptTokenCount": 2000,
"candidatesTokenCount": 100,
"totalTokenCount": 2100,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 2000
}
],
"candidatesTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.00375,
"input_cost": 0.003,
"output_cost": 0.00075,
"prompt_tokens": 2000,
"completion_tokens": 100
}
},
{
"name": "gemini-nano-banana-2.1-image_input",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "3f6c2d91a40e summarize the attached material in one line and name the city weather"
},
{
"type": "image_url",
"image_url": {
"url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+yVWQAAAAASUVORK5CYII=",
"detail": "high"
}
}
]
}
],
"stream": false,
"allowed_openai_params": []
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"text": "scripted answer 3f6c2d91a40e"
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0
}
],
"usageMetadata": {
"promptTokenCount": 1680,
"candidatesTokenCount": 100,
"totalTokenCount": 1780,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 560
},
{
"modality": "IMAGE",
"tokenCount": 1120
}
],
"candidatesTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.00327,
"input_cost": 0.00252,
"output_cost": 0.00075,
"prompt_tokens": 1680,
"completion_tokens": 100
}
},
{
"name": "gemini-nano-banana-2.1-output_text",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "d05a38c7e219 summarize the attached material in one line and name the city weather"
}
]
}
],
"stream": false,
"allowed_openai_params": []
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"text": "scripted answer d05a38c7e219"
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0
}
],
"usageMetadata": {
"promptTokenCount": 100,
"candidatesTokenCount": 4000,
"totalTokenCount": 4100,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
],
"candidatesTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 4000
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.03015,
"input_cost": 0.00015,
"output_cost": 0.03,
"prompt_tokens": 100,
"completion_tokens": 4000
}
},
{
"name": "gemini-nano-banana-2.1-reasoning",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "c71e4f0092ba summarize the attached material in one line and name the city weather"
}
]
}
],
"stream": false,
"reasoning_effort": "medium",
"allowed_openai_params": [
"reasoning_effort"
]
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"text": "scripted answer c71e4f0092ba"
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0
}
],
"usageMetadata": {
"promptTokenCount": 100,
"candidatesTokenCount": 200,
"thoughtsTokenCount": 1800,
"totalTokenCount": 2100,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
],
"candidatesTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 200
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.01515,
"input_cost": 0.00015,
"output_cost": 0.015,
"prompt_tokens": 100,
"completion_tokens": 2000
}
},
{
"name": "gemini-nano-banana-2.1-image_output_1k",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "6a9f2c13b847 summarize the attached material in one line and name the city weather"
}
]
}
],
"stream": false,
"allowed_openai_params": []
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"inlineData": {
"mimeType": "image/png",
"data": "aGVsbG8="
}
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0
}
],
"usageMetadata": {
"promptTokenCount": 100,
"candidatesTokenCount": 1120,
"totalTokenCount": 1220,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
],
"candidatesTokensDetails": [
{
"modality": "IMAGE",
"tokenCount": 1120
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.03375,
"input_cost": 0.00015,
"output_cost": 0.0336,
"prompt_tokens": 100,
"completion_tokens": 1120
}
},
{
"name": "gemini-nano-banana-2.1-image_output_2k",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "24df708bc351 summarize the attached material in one line and name the city weather"
}
]
}
],
"stream": false,
"allowed_openai_params": []
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"inlineData": {
"mimeType": "image/png",
"data": "aGVsbG8="
}
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0
}
],
"usageMetadata": {
"promptTokenCount": 100,
"candidatesTokenCount": 1680,
"totalTokenCount": 1780,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
],
"candidatesTokensDetails": [
{
"modality": "IMAGE",
"tokenCount": 1680
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.05055,
"input_cost": 0.00015,
"output_cost": 0.0504,
"prompt_tokens": 100,
"completion_tokens": 1680
}
},
{
"name": "gemini-nano-banana-2.1-image_output_4k",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "b8a315e4706c summarize the attached material in one line and name the city weather"
}
]
}
],
"stream": false,
"allowed_openai_params": []
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"inlineData": {
"mimeType": "image/png",
"data": "aGVsbG8="
}
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0
}
],
"usageMetadata": {
"promptTokenCount": 100,
"candidatesTokenCount": 2520,
"totalTokenCount": 2620,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
],
"candidatesTokensDetails": [
{
"modality": "IMAGE",
"tokenCount": 2520
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.07575,
"input_cost": 0.00015,
"output_cost": 0.0756,
"prompt_tokens": 100,
"completion_tokens": 2520
}
},
{
"name": "gemini-nano-banana-2.1-mixed_text_image_output",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "91ce62a50d3b summarize the attached material in one line and name the city weather"
}
]
}
],
"stream": false,
"allowed_openai_params": []
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"text": "scripted answer 91ce62a50d3b"
},
{
"inlineData": {
"mimeType": "image/png",
"data": "aGVsbG8="
}
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0
}
],
"usageMetadata": {
"promptTokenCount": 100,
"candidatesTokenCount": 1220,
"thoughtsTokenCount": 300,
"totalTokenCount": 1620,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
],
"candidatesTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
},
{
"modality": "IMAGE",
"tokenCount": 1120
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.03675,
"input_cost": 0.00015,
"output_cost": 0.0366,
"prompt_tokens": 100,
"completion_tokens": 1520
}
},
{
"name": "gemini-nano-banana-2.1-web_search_single_query",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "f407b19d8e26 summarize the attached material in one line and name the city weather"
}
]
}
],
"stream": false,
"tools": [
{
"googleSearch": {}
}
],
"allowed_openai_params": [
"web_search_options"
]
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"text": "scripted answer f407b19d8e26"
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0,
"groundingMetadata": {
"webSearchQueries": [
"query 0"
]
}
}
],
"usageMetadata": {
"promptTokenCount": 100,
"candidatesTokenCount": 100,
"totalTokenCount": 200,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
],
"candidatesTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.0149,
"input_cost": 0.00015,
"output_cost": 0.00075,
"tool_usage_cost": 0.014,
"prompt_tokens": 100,
"completion_tokens": 100
}
},
{
"name": "gemini-nano-banana-2.1-web_search_multiple_queries",
"covers": "quota_management.spend_tracking.cost_matrix.logs_cost",
"model": "gemini/gemini-nano-banana-2.1",
"request": {
"model": "$MODEL",
"messages": [
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are a deterministic pricing-harness assistant. Keep answers to a single short line."
}
]
},
{
"role": "user",
"content": [
{
"type": "text",
"text": "a36d5f80c297 summarize the attached material in one line and name the city weather"
}
]
}
],
"stream": false,
"tools": [
{
"googleSearch": {}
}
],
"allowed_openai_params": [
"web_search_options"
]
},
"response": {
"content_type": "application/json",
"body": {
"candidates": [
{
"content": {
"parts": [
{
"text": "scripted answer a36d5f80c297"
}
],
"role": "model"
},
"finishReason": "STOP",
"index": 0,
"groundingMetadata": {
"webSearchQueries": [
"query 0",
"query 1",
"query 2",
"query 3"
]
}
}
],
"usageMetadata": {
"promptTokenCount": 100,
"candidatesTokenCount": 100,
"totalTokenCount": 200,
"promptTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
],
"candidatesTokensDetails": [
{
"modality": "TEXT",
"tokenCount": 100
}
]
},
"modelVersion": "gemini-nano-banana-2.1"
}
},
"expected": {
"spend": 0.0569,
"input_cost": 0.00015,
"output_cost": 0.00075,
"tool_usage_cost": 0.056,
"prompt_tokens": 100,
"completion_tokens": 100
}
}
],
"batch_cases": [

View file

@ -0,0 +1,446 @@
from __future__ import annotations
import json
import os
import socket
import subprocess
import sys
import textwrap
import threading
from collections.abc import Generator
from contextlib import contextmanager
from dataclasses import dataclass
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from typing import Final
from urllib.parse import urlsplit
import pytest
from integration._support.tls import server_context, write_self_signed_cert
from integration._support.vertex import service_account_json
from integration._support.wire import Reply, Request, Wire, wire_server
from pydantic import JsonValue, TypeAdapter
pytestmark: Final = pytest.mark.timeout(180)
_PROJECT: Final = "scripted-gemini-batch-project"
_MODEL: Final = "gemini-nano-banana-2.1"
_BATCH_ID: Final = "scripted-batch"
_CALLBACK_PREFIX: Final = "CALLBACK:"
_REGISTRY_PREFIX: Final = "REGISTRY:"
_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue])
_IMAGE_PART: Final = {
"inlineData": {
"mimeType": "image/png",
"data": "aGVsbG8=",
}
}
def _pipe(source: socket.socket, destination: socket.socket) -> None:
try:
for payload in iter(lambda: source.recv(65536), b""):
destination.sendall(payload)
except OSError:
return
finally:
try:
destination.shutdown(socket.SHUT_WR)
except OSError:
pass
@contextmanager
def _redirect_https_host(host: str, port: int, *, to_port: int) -> Generator[str, None, None]:
"""Yields an HTTPS_PROXY url that sends host:port to 127.0.0.1:to_port and refuses anything else."""
authority: Final = f"{host}:{port}"
class _RedirectHandler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
timeout = 30
def do_CONNECT(self) -> None:
if self.path != authority:
self.send_error(403)
return
self.send_response(200, "Connection Established")
self.end_headers()
try:
with socket.create_connection(("127.0.0.1", to_port), timeout=10) as upstream:
client_to_upstream: Final = threading.Thread(
target=_pipe, args=(self.connection, upstream), daemon=True
)
upstream_to_client: Final = threading.Thread(
target=_pipe, args=(upstream, self.connection), daemon=True
)
client_to_upstream.start()
upstream_to_client.start()
client_to_upstream.join(timeout=30)
upstream_to_client.join(timeout=30)
except OSError:
return
def log_message(self, format: str, *args: object) -> None:
pass
class _RedirectProxy(ThreadingHTTPServer):
daemon_threads = True
request_queue_size = 16
with _RedirectProxy(("127.0.0.1", 0), _RedirectHandler) as server:
thread: Final = threading.Thread(target=server.serve_forever, kwargs={"poll_interval": 0.05})
thread.start()
try:
yield f"http://127.0.0.1:{server.server_port}"
finally:
server.shutdown()
thread.join(timeout=6)
assert not thread.is_alive(), "CONNECT tunnel server survived cleanup"
server.server_close()
@dataclass(frozen=True, slots=True)
class _BatchScenario:
name: str
prompt_tokens: int
prompt_details: tuple[tuple[str, int], ...]
candidate_tokens: int
candidate_details: tuple[tuple[str, int], ...]
thoughts_tokens: int
expected_prompt_cost: float
expected_completion_cost: float
use_explicit_image_batch_rate_override: bool = False
_SCENARIOS: Final = (
_BatchScenario(
name="batch_input_text_and_image",
prompt_tokens=1680,
prompt_details=(("TEXT", 560), ("IMAGE", 1120)),
candidate_tokens=100,
candidate_details=(("TEXT", 100),),
thoughts_tokens=0,
expected_prompt_cost=0.00126,
expected_completion_cost=0.000375,
),
_BatchScenario(
name="batch_text_and_thinking_output",
prompt_tokens=100,
prompt_details=(("TEXT", 100),),
candidate_tokens=400,
candidate_details=(("TEXT", 400),),
thoughts_tokens=600,
expected_prompt_cost=0.000075,
expected_completion_cost=0.00375,
),
_BatchScenario(
name="batch_image_output_1k",
prompt_tokens=100,
prompt_details=(("TEXT", 100),),
candidate_tokens=1120,
candidate_details=(("IMAGE", 1120),),
thoughts_tokens=0,
expected_prompt_cost=0.000075,
expected_completion_cost=0.0168,
),
_BatchScenario(
name="batch_image_output_2k",
prompt_tokens=100,
prompt_details=(("TEXT", 100),),
candidate_tokens=1680,
candidate_details=(("IMAGE", 1680),),
thoughts_tokens=0,
expected_prompt_cost=0.000075,
expected_completion_cost=0.0252,
),
_BatchScenario(
name="batch_image_output_4k",
prompt_tokens=100,
prompt_details=(("TEXT", 100),),
candidate_tokens=3780,
candidate_details=(("IMAGE", 3780),),
thoughts_tokens=0,
expected_prompt_cost=0.000075,
expected_completion_cost=0.0567,
),
_BatchScenario(
name="batch_mixed_text_image_output",
prompt_tokens=100,
prompt_details=(("TEXT", 100),),
candidate_tokens=1220,
candidate_details=(("TEXT", 100), ("IMAGE", 1120)),
thoughts_tokens=300,
expected_prompt_cost=0.000075,
expected_completion_cost=0.0183,
),
_BatchScenario(
name="batch_image_rate_uses_explicit_override",
prompt_tokens=100,
prompt_details=(("TEXT", 100),),
candidate_tokens=1120,
candidate_details=(("IMAGE", 1120),),
thoughts_tokens=0,
expected_prompt_cost=0.000075,
expected_completion_cost=0.0224,
use_explicit_image_batch_rate_override=True,
),
)
_SDK_SCRIPT: Final = textwrap.dedent(
"""
import asyncio, json, os
from typing import Final
import litellm
from litellm.integrations.custom_logger import CustomLogger
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
from litellm.types.utils import ModelInfo
model: Final = "gemini-nano-banana-2.1"
class CaptureLogger(CustomLogger):
async def async_log_success_event(
self,
kwargs: dict[str, object],
response_obj: object,
start_time: object,
end_time: object,
) -> None:
standard: Final = kwargs.get("standard_logging_object")
hidden: Final = getattr(response_obj, "_hidden_params", None)
if not isinstance(standard, dict):
raise RuntimeError("success callback omitted standard_logging_object")
if standard.get("call_type") != "aretrieve_batch":
return
print(
"CALLBACK:"
+ json.dumps(
{
"response_cost": standard.get("response_cost"),
"cost_breakdown": standard.get("cost_breakdown"),
"batch_response_cost": hidden.get("response_cost")
if isinstance(hidden, dict)
else None,
}
),
flush=True,
)
async def main() -> None:
has_explicit_image_batch_rate_override: Final = (
os.environ["USE_EXPLICIT_IMAGE_BATCH_RATE_OVERRIDE"] == "1"
)
if has_explicit_image_batch_rate_override:
pricing: Final[ModelInfo] = {
**litellm.model_cost["vertex_ai/gemini-nano-banana-2.1"],
"output_cost_per_image_token_batches": 2e-5,
}
litellm.register_model({f"vertex_ai/{model}": pricing}, persist_across_reloads=False)
litellm.user_url_validation = False
litellm.disable_vertex_batch_output_transformation = os.environ["TRANSFORM_OUTPUT"] != "1"
info: Final = litellm.get_model_info(model=model, custom_llm_provider="vertex_ai")
assert info["key"] == f"vertex_ai/{model}", info
assert info["litellm_provider"] == "vertex_ai-language-models", info
print(
"REGISTRY:"
+ json.dumps({"key": info["key"], "provider": info["litellm_provider"]}),
flush=True,
)
logger: Final = CaptureLogger()
litellm.logging_callback_manager.add_litellm_async_success_callback(logger)
await litellm.aretrieve_batch(
"scripted-batch",
custom_llm_provider="vertex_ai",
model=model,
api_base=os.environ["VERTEX_API_BASE"],
vertex_project=os.environ["VERTEX_PROJECT"],
vertex_location="us-central1",
vertex_credentials=os.environ["VERTEX_CREDENTIALS"],
gcs_bucket_name="scripted-bucket",
num_retries=0,
)
await asyncio.sleep(0)
await GLOBAL_LOGGING_WORKER.flush()
asyncio.run(main())
"""
)
def _prediction_jsonl(scenario: _BatchScenario) -> bytes:
prompt_details: Final = tuple(
{"modality": modality, "tokenCount": count} for modality, count in scenario.prompt_details
)
candidate_details: Final = tuple(
{"modality": modality, "tokenCount": count} for modality, count in scenario.candidate_details
)
candidate_parts: Final = tuple(
{"text": "scripted batch output"} if modality == "TEXT" else _IMAGE_PART
for modality, _ in scenario.candidate_details
)
usage_metadata: Final[dict[str, JsonValue]] = {
"promptTokenCount": scenario.prompt_tokens,
"candidatesTokenCount": scenario.candidate_tokens,
"totalTokenCount": scenario.prompt_tokens + scenario.candidate_tokens + scenario.thoughts_tokens,
"promptTokensDetails": prompt_details,
"candidatesTokensDetails": candidate_details,
**({"thoughtsTokenCount": scenario.thoughts_tokens} if scenario.thoughts_tokens else {}),
}
row: Final[dict[str, JsonValue]] = {
"request": {
"contents": [
{
"role": "user",
"parts": [{"text": "Scripted Vertex Gemini batch request"}],
}
]
},
"status": "",
"response": {
"candidates": [
{
"content": {"parts": candidate_parts, "role": "model"},
"finishReason": "STOP",
"index": 0,
}
],
"usageMetadata": usage_metadata,
"modelVersion": _MODEL,
},
}
return json.dumps(row, separators=(",", ":")).encode("utf-8") + b"\n"
def _api_reply(request: Request) -> Reply:
if request.method == "POST" and request.target == "/_oauth/token":
return Reply(body=b'{"access_token":"scripted-token","expires_in":3600,"token_type":"Bearer"}')
if request.method == "GET" and request.target.endswith(f"/batchPredictionJobs/{_BATCH_ID}"):
return Reply(
body=json.dumps(
{
"name": (f"projects/{_PROJECT}/locations/us-central1/batchPredictionJobs/{_BATCH_ID}"),
"state": "JOB_STATE_SUCCEEDED",
"outputInfo": {
"gcsOutputDirectory": (
"gs://scripted-bucket/litellm-vertex-files/"
"publishers/google/models/gemini-nano-banana-2.1/scripted-prefix"
)
},
"createTime": "2026-10-06T00:00:00.000Z",
}
).encode("utf-8")
)
return Reply(status=404, body=b"{}")
def _gcs_reply(content: bytes):
def respond(request: Request) -> Reply:
if request.method == "GET" and request.target.endswith("predictions.jsonl?alt=media"):
return Reply(body=content, content_type="application/jsonl")
return Reply(status=404, body=b"{}")
return respond
def _subprocess_environment(
api: Wire,
certificate: Path,
credentials: str,
proxy_url: str,
scenario: _BatchScenario,
transform_output: bool,
) -> dict[str, str]:
repo_root: Final = Path(__file__).resolve().parents[3]
python_path: Final = os.pathsep.join(path for path in (str(repo_root), os.environ.get("PYTHONPATH")) if path)
return {
**os.environ,
"HTTPS_PROXY": proxy_url,
"https_proxy": proxy_url,
"HTTP_PROXY": "",
"http_proxy": "",
"SSL_CERT_FILE": str(certificate),
"NO_PROXY": "127.0.0.1,localhost",
"no_proxy": "127.0.0.1,localhost",
"VERTEX_API_BASE": api.url,
"VERTEX_CREDENTIALS": credentials,
"VERTEX_PROJECT": _PROJECT,
"USE_EXPLICIT_IMAGE_BATCH_RATE_OVERRIDE": (
"1" if scenario.use_explicit_image_batch_rate_override else "0"
),
"TRANSFORM_OUTPUT": "1" if transform_output else "0",
"LITELLM_LOCAL_MODEL_COST_MAP": "True",
"PYTHONPATH": python_path,
}
@pytest.mark.parametrize("transform_output", (False, True), ids=("untransformed", "transformed"))
@pytest.mark.parametrize("scenario", _SCENARIOS, ids=tuple(case.name for case in _SCENARIOS))
def test_aretrieve_batch_costs_native_gemini_image_tokens(
scenario: _BatchScenario, transform_output: bool, tmp_path: Path
) -> None:
certificate, key = write_self_signed_cert(tmp_path, names=("storage.googleapis.com",))
tls: Final = server_context(certificate, key)
row: Final = _prediction_jsonl(scenario)
with wire_server(_api_reply) as api:
credentials: Final = service_account_json(_PROJECT, api.url)
with wire_server(_gcs_reply(row), tls=tls) as gcs:
gcs_port: Final = urlsplit(gcs.url).port
assert gcs_port is not None
with _redirect_https_host("storage.googleapis.com", 443, to_port=gcs_port) as proxy_url:
environment: Final = _subprocess_environment(
api, certificate, credentials, proxy_url, scenario, transform_output
)
outcome: Final = subprocess.run(
[sys.executable, "-P", "-c", _SDK_SCRIPT],
env=environment,
capture_output=True,
text=True,
timeout=60,
check=False,
)
assert outcome.returncode == 0, (outcome.stdout, outcome.stderr)
registry_events: Final = tuple(
_JSON_OBJECT.validate_json(line[len(_REGISTRY_PREFIX) :])
for line in outcome.stdout.splitlines()
if line.startswith(_REGISTRY_PREFIX)
)
callback_events: Final = tuple(
_JSON_OBJECT.validate_json(line[len(_CALLBACK_PREFIX) :])
for line in outcome.stdout.splitlines()
if line.startswith(_CALLBACK_PREFIX)
)
expected_registry_key: Final = f"vertex_ai/{_MODEL}"
assert registry_events == (
{"key": expected_registry_key, "provider": "vertex_ai-language-models"},
), outcome.stdout
assert len(callback_events) == 1, (outcome.stdout, outcome.stderr)
event: Final = callback_events[0]
assert event["response_cost"] == pytest.approx(
scenario.expected_prompt_cost + scenario.expected_completion_cost
), event
assert event["batch_response_cost"] == pytest.approx(
scenario.expected_prompt_cost + scenario.expected_completion_cost
), event
breakdown: Final = event["cost_breakdown"]
assert isinstance(breakdown, dict), event
assert breakdown["input_cost"] == pytest.approx(scenario.expected_prompt_cost), event
assert breakdown["output_cost"] == pytest.approx(scenario.expected_completion_cost), event
assert breakdown["total_cost"] == pytest.approx(
scenario.expected_prompt_cost + scenario.expected_completion_cost
), event
gcs_requests: Final = gcs.drain()
assert any(
request.method == "GET" and request.target.endswith("predictions.jsonl?alt=media")
for request in gcs_requests
), tuple((request.method, request.target) for request in gcs_requests)
api_requests: Final = api.drain()
assert any(
request.method == "GET"
and request.target.endswith(f"/batchPredictionJobs/{_BATCH_ID}")
and request.headers.get("authorization") == "Bearer scripted-token"
for request in api_requests
), tuple((request.method, request.target) for request in api_requests)

View file

@ -13,6 +13,7 @@ from litellm.cost_calculator import (
BaseTokenUsageProcessor,
RealtimeAPITokenUsageProcessor,
ResponsesWebSocketTokenUsageProcessor,
batch_cost_calculator,
completion_cost,
cost_per_token,
handle_realtime_stream_cost_calculation,
@ -27,6 +28,7 @@ from litellm.types.utils import (
CacheCreationTokenDetails,
CallTypes,
Choices,
CompletionTokensDetailsWrapper,
EmbeddingResponse,
ImageObject,
ImageResponse,
@ -3448,8 +3450,6 @@ def _batch_cache_usage() -> Usage:
def test_batch_cost_calculator_prices_multimodal_tokens_at_modality_rates():
from litellm.cost_calculator import batch_cost_calculator
model_info: ModelInfo = {
"input_cost_per_token_batches": 1e-7,
"input_cost_per_audio_token_batches": 3.25e-6,
@ -3478,8 +3478,6 @@ def test_batch_cost_calculator_prices_multimodal_tokens_at_modality_rates():
def test_batch_cost_calculator_falls_back_to_text_batch_rate_for_modalities():
from litellm.cost_calculator import batch_cost_calculator
model_info: ModelInfo = {"input_cost_per_token_batches": 1e-7}
usage = Usage(
prompt_tokens=100,
@ -3498,6 +3496,87 @@ def test_batch_cost_calculator_falls_back_to_text_batch_rate_for_modalities():
assert prompt_cost == pytest.approx(100 * 1e-7)
@pytest.mark.parametrize(
("image_batch_rate", "image_tokens", "expected_completion_cost"),
(
(5e-5, 80, 0.00408),
(None, 80, 0.00328),
(5e-5, 140, 0.006),
),
)
def test_batch_cost_calculator_prices_image_completion_tokens_at_image_batch_rate(
image_batch_rate: float | None, image_tokens: int, expected_completion_cost: float
) -> None:
model_info: Final[ModelInfo] = (
{
"output_cost_per_token_batches": 2e-6,
"output_cost_per_image_token": 8e-5,
"output_cost_per_image_token_batches": image_batch_rate,
}
if image_batch_rate is not None
else {
"output_cost_per_token_batches": 2e-6,
"output_cost_per_image_token": 8e-5,
}
)
usage: Final = Usage(
prompt_tokens=0,
completion_tokens=120,
total_tokens=120,
completion_tokens_details=CompletionTokensDetailsWrapper(image_tokens=image_tokens),
)
costs: Final = batch_cost_calculator(
usage=usage,
model="gemini-nano-banana-2.1",
custom_llm_provider="vertex_ai",
model_info=model_info,
)
assert costs[1] == pytest.approx(expected_completion_cost)
def test_batch_cost_calculator_merges_image_only_deployment_rate_with_global_pricing(
_local_model_cost_map: None,
) -> None:
model: Final = "gemini/gemini-3-pro-image"
global_model_info: Final = litellm.get_model_info(model=model, custom_llm_provider="gemini")
input_batch_rate: Final = global_model_info["input_cost_per_token_batches"]
output_batch_rate: Final = global_model_info["output_cost_per_token_batches"]
global_image_batch_rate: Final = global_model_info["output_cost_per_image_token_batches"]
assert input_batch_rate is not None
assert output_batch_rate is not None
assert global_image_batch_rate is not None
deployment_image_batch_rate: Final = 1e-6
assert deployment_image_batch_rate != global_image_batch_rate
text_tokens: Final = 300
image_tokens: Final = 200
usage: Final = Usage(
prompt_tokens=1_000,
completion_tokens=text_tokens + image_tokens,
total_tokens=1_500,
completion_tokens_details=CompletionTokensDetailsWrapper(image_tokens=image_tokens),
)
model_info: Final = ModelInfo(output_cost_per_image_token_batches=deployment_image_batch_rate)
prompt_cost, completion_cost = batch_cost_calculator(
usage=usage,
model=model,
custom_llm_provider="gemini",
model_info=model_info,
)
assert prompt_cost > 0
assert completion_cost > 0
assert prompt_cost == pytest.approx(usage.prompt_tokens * input_batch_rate)
expected_completion_cost: Final = text_tokens * output_batch_rate + image_tokens * deployment_image_batch_rate
global_rate_completion_cost: Final = text_tokens * output_batch_rate + image_tokens * global_image_batch_rate
assert completion_cost == pytest.approx(expected_completion_cost)
assert completion_cost != pytest.approx(global_rate_completion_cost)
def test_batch_cost_calculator_prices_cache_creation_tokens_at_cache_write_rate():
"""
LIT-4008 regression: anthropic batch usage is dominated by cache tokens.

View file

@ -665,6 +665,7 @@ def validate_model_cost_values(model_data, exceptions=None):
"input_cost_per_audio_token",
"output_cost_per_audio_token",
"output_cost_per_image_token",
"output_cost_per_image_token_batches",
"input_cost_per_video_token",
"output_cost_per_video_token",
"input_cost_per_audio_per_second",
@ -903,6 +904,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"output_cost_per_image_2K": {"type": "number"},
"output_cost_per_image_4K": {"type": "number"},
"output_cost_per_image_token": {"type": "number"},
"output_cost_per_image_token_batches": {"type": "number"},
"output_cost_per_video_token": {"type": "number"},
"output_cost_per_pixel": {"type": "number"},
"output_cost_per_second": {"type": "number"},

View file

@ -35445,6 +35445,8 @@ export interface components {
output_cost_per_image_512?: number | null;
/** Output Cost Per Image Token */
output_cost_per_image_token?: number | null;
/** Output Cost Per Image Token Batches */
output_cost_per_image_token_batches?: number | null;
/** Output Cost Per Pixel */
output_cost_per_pixel?: number | null;
/** Output Cost Per Reasoning Token */
@ -50281,6 +50283,8 @@ export interface components {
output_cost_per_image_512?: number | null;
/** Output Cost Per Image Token */
output_cost_per_image_token?: number | null;
/** Output Cost Per Image Token Batches */
output_cost_per_image_token_batches?: number | null;
/** Output Cost Per Pixel */
output_cost_per_pixel?: number | null;
/** Output Cost Per Reasoning Token */