fix(cost-calc): use per-image pricing for Bedrock multimodal embeddings (#21646)

Bedrock multimodal embedding models (Titan and Nova) were being costed
using the per-token text rate instead of the correct flat per-image rate
($0.00006/image). The pricing data was correct but never applied because
image_count was never populated in prompt_tokens_details.

Pass batch_data to Titan/Nova response transformers so they can count
image inputs and set PromptTokensDetailsWrapper(image_count=N) on Usage,
mirroring the existing Vertex AI pattern from PR #9623. Also fix the
text_tokens fallback in generic_cost_per_token to not override
text_tokens=0 when image_count > 0 (image-only requests).
This commit is contained in:
michelligabriele 2026-02-20 17:51:21 +01:00 • committed by GitHub
parent 048f734168
commit a395a25705
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 323 additions and 14 deletions

View file

@ -602,7 +602,7 @@ def generic_cost_per_token( # noqa: PLR0915
total_details = text_tokens + cache_hit + audio_tokens + cache_creation + image_tokens
has_double_counting = cache_hit > 0 and total_details > usage.prompt_tokens
if text_tokens == 0 or has_double_counting:
if (text_tokens == 0 and prompt_tokens_details["image_count"] == 0) or has_double_counting:
text_tokens = (
usage.prompt_tokens
- cache_hit

View file

@ -14,7 +14,7 @@ Docs - https://docs.aws.amazon.com/bedrock/latest/userguide/nova-embed.html
from typing import List, Optional
from litellm.types.utils import Embedding, EmbeddingResponse, Usage
from litellm.types.utils import Embedding, EmbeddingResponse, PromptTokensDetailsWrapper, Usage
class AmazonNovaEmbeddingConfig:
@ -244,11 +244,14 @@ class AmazonNovaEmbeddingConfig:
}
def _transform_response(
self, response_list: List[dict], model: str
self,
response_list: List[dict],
model: str,
batch_data: Optional[List[dict]] = None,
) -> EmbeddingResponse:
"""
Transform Nova response to OpenAI format.
Nova response format:
{
"embeddings": [
@ -262,7 +265,7 @@ class AmazonNovaEmbeddingConfig:
"""
embeddings: List[Embedding] = []
total_tokens = 0
for response in response_list:
# Nova response has an "embeddings" array
if "embeddings" in response and isinstance(response["embeddings"], list):
@ -274,7 +277,7 @@ class AmazonNovaEmbeddingConfig:
object="embedding",
)
embeddings.append(embedding)
# Estimate token count
# For text, use truncatedCharLength if available
if "truncatedCharLength" in item:
@ -291,9 +294,31 @@ class AmazonNovaEmbeddingConfig:
)
embeddings.append(embedding)
total_tokens += len(response["embedding"]) // 4
usage = Usage(prompt_tokens=total_tokens, total_tokens=total_tokens)
# Count images from original requests for cost calculation
image_count = 0
if batch_data:
for request_data in batch_data:
# Nova wraps params in singleEmbeddingParams or segmentedEmbeddingParams
params = request_data.get(
"singleEmbeddingParams",
request_data.get("segmentedEmbeddingParams", {}),
)
if "image" in params:
image_count += 1
prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None
if image_count > 0:
prompt_tokens_details = PromptTokensDetailsWrapper(
image_count=image_count,
)
usage = Usage(
prompt_tokens=total_tokens,
total_tokens=total_tokens,
prompt_tokens_details=prompt_tokens_details,
)
return EmbeddingResponse(data=embeddings, model=model, usage=usage)
def _transform_async_invoke_response(

View file

@ -6,14 +6,14 @@ Why separate file? Make it easy to see how transformation works
Docs - https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-titan-embed-mm.html
"""
from typing import List
from typing import List, Optional
from litellm.types.llms.bedrock import (
AmazonTitanMultimodalEmbeddingConfig,
AmazonTitanMultimodalEmbeddingRequest,
AmazonTitanMultimodalEmbeddingResponse,
)
from litellm.types.utils import Embedding, EmbeddingResponse, Usage
from litellm.types.utils import Embedding, EmbeddingResponse, PromptTokensDetailsWrapper, Usage
from litellm.utils import get_base64_str, is_base64_encoded
@ -56,7 +56,10 @@ class AmazonTitanMultimodalEmbeddingG1Config:
return transformed_request
def _transform_response(
self, response_list: List[dict], model: str
self,
response_list: List[dict],
model: str,
batch_data: Optional[List[dict]] = None,
) -> EmbeddingResponse:
total_prompt_tokens = 0
transformed_responses: List[Embedding] = []
@ -71,9 +74,23 @@ class AmazonTitanMultimodalEmbeddingG1Config:
)
total_prompt_tokens += _parsed_response["inputTextTokenCount"]
# Count images from original requests for cost calculation
image_count = 0
if batch_data:
for request_data in batch_data:
if "inputImage" in request_data:
image_count += 1
prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None
if image_count > 0:
prompt_tokens_details = PromptTokensDetailsWrapper(
image_count=image_count,
)
usage = Usage(
prompt_tokens=total_prompt_tokens,
completion_tokens=0,
total_tokens=total_prompt_tokens,
prompt_tokens_details=prompt_tokens_details,
)
return EmbeddingResponse(model=model, usage=usage, data=transformed_responses)

View file

@ -158,6 +158,7 @@ class BedrockEmbedding(BaseAWSLLM):
model: str,
provider: BEDROCK_EMBEDDING_PROVIDERS_LITERAL,
is_async_invoke: Optional[bool] = False,
batch_data: Optional[List[dict]] = None,
) -> Optional[EmbeddingResponse]:
"""
Transforms the response from the Bedrock embedding provider to the OpenAI format.
@ -212,7 +213,7 @@ class BedrockEmbedding(BaseAWSLLM):
if model == "amazon.titan-embed-image-v1":
returned_response = (
AmazonTitanMultimodalEmbeddingG1Config()._transform_response(
response_list=response_list, model=model
response_list=response_list, model=model, batch_data=batch_data
)
)
elif model == "amazon.titan-embed-text-v1":
@ -231,7 +232,7 @@ class BedrockEmbedding(BaseAWSLLM):
)
elif provider == "nova":
returned_response = AmazonNovaEmbeddingConfig()._transform_response(
response_list=response_list, model=model
response_list=response_list, model=model, batch_data=batch_data
)
##########################################################
@ -310,6 +311,7 @@ class BedrockEmbedding(BaseAWSLLM):
model=model,
provider=provider,
is_async_invoke=is_async_invoke,
batch_data=batch_data,
)
async def _async_single_func_embeddings(
@ -379,6 +381,7 @@ class BedrockEmbedding(BaseAWSLLM):
model=model,
provider=provider,
is_async_invoke=is_async_invoke,
batch_data=batch_data,
)
def embeddings( # noqa: PLR0915

View file

@ -390,6 +390,109 @@ class TestNovaTransformationResponse:
assert result.data[0].embedding == [0.1, 0.2, 0.3]
assert result.data[1].embedding == [0.4, 0.5, 0.6]
def test_image_embedding_response_with_image_count(self):
"""Test that Nova image embedding response populates image_count for cost tracking."""
config = AmazonNovaEmbeddingConfig()
response_list = [
{
"embeddings": [
{
"embeddingType": "IMAGE",
"embedding": [0.1, 0.2, 0.3],
}
]
}
]
# Simulate batch_data with image in singleEmbeddingParams
batch_data = [
{
"schemaVersion": "nova-multimodal-embed-v1",
"taskType": "SINGLE_EMBEDDING",
"singleEmbeddingParams": {
"embeddingPurpose": "GENERIC_INDEX",
"embeddingDimension": 3072,
"image": {
"format": "jpeg",
"source": {"bytes": "/9j/4AAQSkZJRg=="},
},
},
}
]
result = config._transform_response(
response_list=response_list,
model="amazon.nova-2-multimodal-embeddings-v1:0",
batch_data=batch_data,
)
assert result.usage is not None
assert result.usage.prompt_tokens_details is not None
assert result.usage.prompt_tokens_details.image_count == 1
def test_text_embedding_response_no_image_count(self):
"""Test that Nova text embedding response does not set image_count."""
config = AmazonNovaEmbeddingConfig()
response_list = [
{
"embeddings": [
{
"embeddingType": "TEXT",
"embedding": [0.1, 0.2, 0.3],
"truncatedCharLength": 20,
}
]
}
]
batch_data = [
{
"schemaVersion": "nova-multimodal-embed-v1",
"taskType": "SINGLE_EMBEDDING",
"singleEmbeddingParams": {
"embeddingPurpose": "GENERIC_INDEX",
"embeddingDimension": 3072,
"text": {"value": "hello world", "truncationMode": "END"},
},
}
]
result = config._transform_response(
response_list=response_list,
model="amazon.nova-2-multimodal-embeddings-v1:0",
batch_data=batch_data,
)
assert result.usage is not None
assert result.usage.prompt_tokens_details is None
def test_nova_embedding_backward_compat_no_batch_data(self):
"""Test that Nova transformer works without batch_data (backward compatibility)."""
config = AmazonNovaEmbeddingConfig()
response_list = [
{
"embeddings": [
{
"embeddingType": "TEXT",
"embedding": [0.1, 0.2, 0.3, 0.4, 0.5],
}
]
}
]
# Call without batch_data — should not break
result = config._transform_response(
response_list=response_list,
model="amazon.nova-2-multimodal-embeddings-v1:0",
)
assert result.usage is not None
assert result.usage.total_tokens > 0
assert result.usage.prompt_tokens_details is None
def test_async_invoke_response(self):
"""Test async invoke response transformation."""
config = AmazonNovaEmbeddingConfig()

View file

@ -862,3 +862,42 @@ def test_reasoning_tokens_without_text_tokens_gpt5_nano():
wrong_cost = 768 * 0.40 / 1_000_000 # Only reasoning tokens
assert abs(completion_cost - wrong_cost) > 1e-6, \
"Bug detected: Cost calculation is using only reasoning_tokens instead of all completion_tokens!"
def test_image_count_prevents_text_tokens_fallback():
"""
Test that the text_tokens fallback in generic_cost_per_token does not
override text_tokens=0 when image_count > 0.
Regression test for: Bedrock image embedding double-charging bug.
When image_count > 0, text_tokens=0 is intentional (image-only request),
not "text_tokens not set by provider."
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Simulate Nova image-only embedding: prompt_tokens estimated from
# embedding dimensions (768 for 3072-dim), image_count=1
usage = Usage(
prompt_tokens=768,
completion_tokens=0,
total_tokens=768,
prompt_tokens_details=PromptTokensDetailsWrapper(
image_count=1,
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="amazon.nova-2-multimodal-embeddings-v1:0",
usage=usage,
custom_llm_provider="bedrock",
)
# Cost should be 1 * input_cost_per_image ($6e-05) = $0.00006
# NOT 768 * input_cost_per_token ($1.35e-07) + $0.00006 = $0.000164
expected_image_cost = 1 * 6e-05
assert prompt_cost == expected_image_cost, (
f"Expected prompt_cost={expected_image_cost} (image-only), "
f"got {prompt_cost}. text_tokens fallback may be double-charging."
)
assert completion_cost == 0.0

View file

@ -833,3 +833,125 @@ async def test_bedrock_embedding_custom_headers_with_iam_role_and_custom_api_bas
except Exception as e:
pytest.fail(f"Failed to forward headers with IAM role + custom api_base (async): {str(e)}")
def test_titan_multimodal_embedding_image_cost_tracking():
"""Test that Titan multimodal embedding with image input populates image_count in Usage."""
from litellm.llms.bedrock.embed.amazon_titan_multimodal_transformation import (
AmazonTitanMultimodalEmbeddingG1Config,
)
config = AmazonTitanMultimodalEmbeddingG1Config()
# Simulate response from AWS Bedrock
response_list = [
{
"embedding": [0.1, 0.2, 0.3],
"inputTextTokenCount": 0,
}
]
# Simulate batch_data with an image request (inputImage key set by _transform_request)
batch_data = [
{"inputImage": "/9j/4AAQSkZJRg=="}
]
result = config._transform_response(
response_list=response_list,
model="amazon.titan-embed-image-v1",
batch_data=batch_data,
)
assert result.usage is not None
assert result.usage.prompt_tokens_details is not None
assert result.usage.prompt_tokens_details.image_count == 1
def test_titan_multimodal_embedding_text_no_image_count():
"""Test that Titan multimodal embedding with text-only input does not set image_count."""
from litellm.llms.bedrock.embed.amazon_titan_multimodal_transformation import (
AmazonTitanMultimodalEmbeddingG1Config,
)
config = AmazonTitanMultimodalEmbeddingG1Config()
response_list = [
{
"embedding": [0.1, 0.2, 0.3],
"inputTextTokenCount": 5,
}
]
# Text-only request — no inputImage key
batch_data = [
{"inputText": "hello world"}
]
result = config._transform_response(
response_list=response_list,
model="amazon.titan-embed-image-v1",
batch_data=batch_data,
)
assert result.usage is not None
# prompt_tokens_details should be None for text-only (no image_count to report)
assert result.usage.prompt_tokens_details is None
def test_titan_multimodal_embedding_backward_compat_no_batch_data():
"""Test that Titan transformer works without batch_data (backward compatibility)."""
from litellm.llms.bedrock.embed.amazon_titan_multimodal_transformation import (
AmazonTitanMultimodalEmbeddingG1Config,
)
config = AmazonTitanMultimodalEmbeddingG1Config()
response_list = [
{
"embedding": [0.1, 0.2, 0.3],
"inputTextTokenCount": 5,
}
]
# Call without batch_data — should not break
result = config._transform_response(
response_list=response_list,
model="amazon.titan-embed-image-v1",
)
assert result.usage is not None
assert result.usage.prompt_tokens == 5
assert result.usage.prompt_tokens_details is None
def test_titan_image_embedding_cost_uses_per_image_rate():
"""
End-to-end test: Titan image embedding with mocked AWS response
should populate image_count for correct per-image cost calculation.
"""
client = HTTPHandler()
with patch.object(client, "post") as mock_post:
mock_response = Mock()
mock_response.status_code = 200
embed_response = {
"embedding": [0.1] * 1024,
"inputTextTokenCount": 0,
}
mock_response.text = json.dumps(embed_response)
mock_response.json = lambda: json.loads(mock_response.text)
mock_post.return_value = mock_response
response = litellm.embedding(
model="bedrock/amazon.titan-embed-image-v1",
input=["data:image/png;base64,iVBORw0KGgoAAAANSUhEUg=="],
client=client,
aws_access_key_id="fake",
aws_secret_access_key="fake",
aws_region_name="us-east-1",
)
assert isinstance(response, litellm.EmbeddingResponse)
assert response.usage is not None
assert response.usage.prompt_tokens_details is not None
assert response.usage.prompt_tokens_details.image_count == 1