fix(cost): price gpt-image generated output tokens as image tokens (#31147)

The OpenAI Images endpoints (/v1/images/generations, /v1/images/edits) return
usage with no output token breakdown — litellm's `ImageUsage` has no
`output_tokens_details` field — so generated-image OUTPUT tokens were priced at
the text rate (`output_cost_per_token`) instead of the image rate
(`output_cost_per_image_token`). For gpt-image-2 that is $10/1M vs $30/1M, a ~3x
undercount on the dominant cost component (image output is ~74% of spend). This
also affects azure gpt-image, which shares this calculator.

The OpenAI gpt-image cost calculator re-implemented usage handling instead of
reusing `calculate_image_response_cost_from_usage`, the shared helper that
azure_ai/gemini/vertex_ai already use. That helper classifies generated output
tokens as image tokens when the provider does not itemize output, and splits
text/image when it does.

Fix: route the ImageUsage path through `calculate_image_response_cost_from_usage`
(pre-transformed chat Usage objects are still costed directly). Adds a regression
test for the no-breakdown ImageUsage case (gpt-image-2).
This commit is contained in:
hayden 2026-06-24 19:34:49 +09:00 • committed by GitHub
parent e0c8a6b483
commit 63741e9698
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 117 additions and 40 deletions

View file

@ -7,7 +7,10 @@ These models use token-based pricing instead of pixel-based pricing like DALL-E.
from typing import Optional
from litellm import verbose_logger
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
from litellm.litellm_core_utils.llm_cost_calc.utils import (
calculate_image_response_cost_from_usage,
generic_cost_per_token,
)
from litellm.types.utils import ImageResponse, Usage
@ -16,54 +19,40 @@ def cost_calculator(
image_response: ImageResponse,
custom_llm_provider: Optional[str] = None,
) -> float:
"""
Calculate cost for OpenAI gpt-image models.
Uses the same usage format as Responses API, so we reuse the helper
to transform to chat completion format and use generic_cost_per_token.
Args:
model: The model name (e.g., "gpt-image-1", "gpt-image-2")
image_response: The ImageResponse containing usage data
custom_llm_provider: Optional provider name
Returns:
float: Total cost in USD
"""
"""Calculate cost for OpenAI gpt-image models (token-based pricing)."""
usage = getattr(image_response, "usage", None)
if usage is None:
verbose_logger.debug(
f"No usage data available for {model}, cannot calculate token-based cost"
)
return 0.0
# If usage is already a Usage object with completion_tokens_details set,
# use it directly (it was already transformed in convert_to_image_response)
provider = custom_llm_provider or "openai"
# A chat Usage with an explicit output breakdown: cost via generic_cost_per_token.
if isinstance(usage, Usage) and usage.completion_tokens_details is not None:
chat_usage = usage
else:
# Transform ImageUsage to Usage using the existing helper
# ImageUsage has the same format as ResponseAPIUsage
from litellm.responses.utils import ResponseAPILoggingUtils
chat_usage = (
ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(usage)
prompt_cost, completion_cost = generic_cost_per_token(
model=model, usage=usage, custom_llm_provider=provider
)
return prompt_cost + completion_cost
# Use generic_cost_per_token for cost calculation
prompt_cost, completion_cost = generic_cost_per_token(
model=model,
usage=chat_usage,
custom_llm_provider=custom_llm_provider or "openai",
)
# ImageUsage / ResponseAPIUsage: reuse the shared helper (same path as
# azure_ai/gemini/vertex_ai). It prices generated output tokens at
# output_cost_per_image_token, classifying them as image tokens when the provider
# does not itemize output and splitting text/image when it does.
if getattr(usage, "input_tokens", None) is not None:
token_based_cost = calculate_image_response_cost_from_usage(
model=model, image_response=image_response, custom_llm_provider=provider
)
if token_based_cost is not None:
return token_based_cost
total_cost = prompt_cost + completion_cost
# Fallback: a Usage with no output breakdown that the image helper can't read —
# cost via generic_cost_per_token (text rate) instead of returning 0.0.
if isinstance(usage, Usage):
prompt_cost, completion_cost = generic_cost_per_token(
model=model, usage=usage, custom_llm_provider=provider
)
return prompt_cost + completion_cost
verbose_logger.debug(
f"OpenAI gpt-image cost calculation for {model}: "
f"prompt_cost=${prompt_cost:.6f}, completion_cost=${completion_cost:.6f}, "
f"total=${total_cost:.6f}"
)
return total_cost
return 0.0

View file

@ -381,5 +381,93 @@ class TestCompletionCostIntegration:
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
class TestGPTImage2OutputImageTokensNoBreakdown:
"""
Regression test: the OpenAI Images endpoints (/v1/images/generations and
/v1/images/edits) return usage with NO output token breakdown — litellm's
ImageUsage has no ``output_tokens_details`` field. Before the fix, the
generated-image OUTPUT tokens were priced at the text rate
(``output_cost_per_token`` = $10/1M for gpt-image-2) instead of the image rate
(``output_cost_per_image_token`` = $30/1M), a ~3x undercount on the dominant
cost component.
"""
def test_gpt_image_2_output_priced_as_image_when_no_breakdown(self):
from litellm.llms.openai.image_generation.cost_calculator import (
cost_calculator,
)
# Mirrors a real gpt-image-2 /v1/images/edits response: input breakdown is
# present, but there is no usable output token breakdown.
usage = ImageUsage(
input_tokens=3987,
output_tokens=5488,
total_tokens=9475,
input_tokens_details=ImageUsageInputTokensDetails(
text_tokens=943,
image_tokens=3044,
),
)
image_response = ImageResponse(
created=1234567890,
data=[ImageObject(b64_json="test")],
)
image_response.usage = usage
image_response._hidden_params = {"custom_llm_provider": "openai"}
cost = cost_calculator(
model="gpt-image-2",
image_response=image_response,
custom_llm_provider="openai",
)
# gpt-image-2 pricing:
# text input: 943 * $5/1M = 0.004715
# image input: 3044 * $8/1M = 0.024352
# image output: 5488 * $30/1M = 0.164640 (NOT text output $10/1M = 0.054880)
expected_cost = 943 * 5e-6 + 3044 * 8e-6 + 5488 * 3e-5
assert abs(cost - expected_cost) < 1e-6, (
f"Expected {expected_cost}, got {cost}. Generated image output tokens "
f"are likely being priced at the text output_cost_per_token rate."
)
def test_gpt_image_2_chat_usage_without_breakdown_is_costed_not_zero(self):
"""A chat ``Usage`` with ``completion_tokens_details=None`` must still be
costed via ``generic_cost_per_token`` (output at the text rate) rather than
erroring or silently returning 0.0."""
from litellm.llms.openai.image_generation.cost_calculator import (
cost_calculator,
)
usage = Usage(
prompt_tokens=600,
completion_tokens=5000,
total_tokens=5600,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=100,
image_tokens=500,
),
)
image_response = ImageResponse(
created=1234567890,
data=[ImageObject(b64_json="test")],
)
image_response.usage = usage
image_response._hidden_params = {"custom_llm_provider": "openai"}
cost = cost_calculator(
model="gpt-image-2",
image_response=image_response,
custom_llm_provider="openai",
)
# No output breakdown -> output priced at the text rate (output_cost_per_token):
# text in 100*$5/1M + image in 500*$8/1M + output 5000*$10/1M
expected_cost = 100 * 5e-6 + 500 * 8e-6 + 5000 * 1e-5
assert abs(cost - expected_cost) < 1e-6, f"Expected {expected_cost}, got {cost}"
if __name__ == "__main__":
pytest.main([__file__, "-v"])