mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(cost-calc): use per-image pricing for Bedrock multimodal embeddings (#21646)
Bedrock multimodal embedding models (Titan and Nova) were being costed using the per-token text rate instead of the correct flat per-image rate ($0.00006/image). The pricing data was correct but never applied because image_count was never populated in prompt_tokens_details. Pass batch_data to Titan/Nova response transformers so they can count image inputs and set PromptTokensDetailsWrapper(image_count=N) on Usage, mirroring the existing Vertex AI pattern from PR #9623. Also fix the text_tokens fallback in generic_cost_per_token to not override text_tokens=0 when image_count > 0 (image-only requests).
This commit is contained in:
parent
048f734168
commit
a395a25705
7 changed files with 323 additions and 14 deletions
|
|
@ -602,7 +602,7 @@ def generic_cost_per_token( # noqa: PLR0915
|
|||
total_details = text_tokens + cache_hit + audio_tokens + cache_creation + image_tokens
|
||||
has_double_counting = cache_hit > 0 and total_details > usage.prompt_tokens
|
||||
|
||||
if text_tokens == 0 or has_double_counting:
|
||||
if (text_tokens == 0 and prompt_tokens_details["image_count"] == 0) or has_double_counting:
|
||||
text_tokens = (
|
||||
usage.prompt_tokens
|
||||
- cache_hit
|
||||
|
|
|
|||
|
|
@ -14,7 +14,7 @@ Docs - https://docs.aws.amazon.com/bedrock/latest/userguide/nova-embed.html
|
|||
|
||||
from typing import List, Optional
|
||||
|
||||
from litellm.types.utils import Embedding, EmbeddingResponse, Usage
|
||||
from litellm.types.utils import Embedding, EmbeddingResponse, PromptTokensDetailsWrapper, Usage
|
||||
|
||||
|
||||
class AmazonNovaEmbeddingConfig:
|
||||
|
|
@ -244,11 +244,14 @@ class AmazonNovaEmbeddingConfig:
|
|||
}
|
||||
|
||||
def _transform_response(
|
||||
self, response_list: List[dict], model: str
|
||||
self,
|
||||
response_list: List[dict],
|
||||
model: str,
|
||||
batch_data: Optional[List[dict]] = None,
|
||||
) -> EmbeddingResponse:
|
||||
"""
|
||||
Transform Nova response to OpenAI format.
|
||||
|
||||
|
||||
Nova response format:
|
||||
{
|
||||
"embeddings": [
|
||||
|
|
@ -262,7 +265,7 @@ class AmazonNovaEmbeddingConfig:
|
|||
"""
|
||||
embeddings: List[Embedding] = []
|
||||
total_tokens = 0
|
||||
|
||||
|
||||
for response in response_list:
|
||||
# Nova response has an "embeddings" array
|
||||
if "embeddings" in response and isinstance(response["embeddings"], list):
|
||||
|
|
@ -274,7 +277,7 @@ class AmazonNovaEmbeddingConfig:
|
|||
object="embedding",
|
||||
)
|
||||
embeddings.append(embedding)
|
||||
|
||||
|
||||
# Estimate token count
|
||||
# For text, use truncatedCharLength if available
|
||||
if "truncatedCharLength" in item:
|
||||
|
|
@ -291,9 +294,31 @@ class AmazonNovaEmbeddingConfig:
|
|||
)
|
||||
embeddings.append(embedding)
|
||||
total_tokens += len(response["embedding"]) // 4
|
||||
|
||||
usage = Usage(prompt_tokens=total_tokens, total_tokens=total_tokens)
|
||||
|
||||
|
||||
# Count images from original requests for cost calculation
|
||||
image_count = 0
|
||||
if batch_data:
|
||||
for request_data in batch_data:
|
||||
# Nova wraps params in singleEmbeddingParams or segmentedEmbeddingParams
|
||||
params = request_data.get(
|
||||
"singleEmbeddingParams",
|
||||
request_data.get("segmentedEmbeddingParams", {}),
|
||||
)
|
||||
if "image" in params:
|
||||
image_count += 1
|
||||
|
||||
prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None
|
||||
if image_count > 0:
|
||||
prompt_tokens_details = PromptTokensDetailsWrapper(
|
||||
image_count=image_count,
|
||||
)
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=total_tokens,
|
||||
total_tokens=total_tokens,
|
||||
prompt_tokens_details=prompt_tokens_details,
|
||||
)
|
||||
|
||||
return EmbeddingResponse(data=embeddings, model=model, usage=usage)
|
||||
|
||||
def _transform_async_invoke_response(
|
||||
|
|
|
|||
|
|
@ -6,14 +6,14 @@ Why separate file? Make it easy to see how transformation works
|
|||
Docs - https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-titan-embed-mm.html
|
||||
"""
|
||||
|
||||
from typing import List
|
||||
from typing import List, Optional
|
||||
|
||||
from litellm.types.llms.bedrock import (
|
||||
AmazonTitanMultimodalEmbeddingConfig,
|
||||
AmazonTitanMultimodalEmbeddingRequest,
|
||||
AmazonTitanMultimodalEmbeddingResponse,
|
||||
)
|
||||
from litellm.types.utils import Embedding, EmbeddingResponse, Usage
|
||||
from litellm.types.utils import Embedding, EmbeddingResponse, PromptTokensDetailsWrapper, Usage
|
||||
from litellm.utils import get_base64_str, is_base64_encoded
|
||||
|
||||
|
||||
|
|
@ -56,7 +56,10 @@ class AmazonTitanMultimodalEmbeddingG1Config:
|
|||
return transformed_request
|
||||
|
||||
def _transform_response(
|
||||
self, response_list: List[dict], model: str
|
||||
self,
|
||||
response_list: List[dict],
|
||||
model: str,
|
||||
batch_data: Optional[List[dict]] = None,
|
||||
) -> EmbeddingResponse:
|
||||
total_prompt_tokens = 0
|
||||
transformed_responses: List[Embedding] = []
|
||||
|
|
@ -71,9 +74,23 @@ class AmazonTitanMultimodalEmbeddingG1Config:
|
|||
)
|
||||
total_prompt_tokens += _parsed_response["inputTextTokenCount"]
|
||||
|
||||
# Count images from original requests for cost calculation
|
||||
image_count = 0
|
||||
if batch_data:
|
||||
for request_data in batch_data:
|
||||
if "inputImage" in request_data:
|
||||
image_count += 1
|
||||
|
||||
prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None
|
||||
if image_count > 0:
|
||||
prompt_tokens_details = PromptTokensDetailsWrapper(
|
||||
image_count=image_count,
|
||||
)
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=total_prompt_tokens,
|
||||
completion_tokens=0,
|
||||
total_tokens=total_prompt_tokens,
|
||||
prompt_tokens_details=prompt_tokens_details,
|
||||
)
|
||||
return EmbeddingResponse(model=model, usage=usage, data=transformed_responses)
|
||||
|
|
|
|||
|
|
@ -158,6 +158,7 @@ class BedrockEmbedding(BaseAWSLLM):
|
|||
model: str,
|
||||
provider: BEDROCK_EMBEDDING_PROVIDERS_LITERAL,
|
||||
is_async_invoke: Optional[bool] = False,
|
||||
batch_data: Optional[List[dict]] = None,
|
||||
) -> Optional[EmbeddingResponse]:
|
||||
"""
|
||||
Transforms the response from the Bedrock embedding provider to the OpenAI format.
|
||||
|
|
@ -212,7 +213,7 @@ class BedrockEmbedding(BaseAWSLLM):
|
|||
if model == "amazon.titan-embed-image-v1":
|
||||
returned_response = (
|
||||
AmazonTitanMultimodalEmbeddingG1Config()._transform_response(
|
||||
response_list=response_list, model=model
|
||||
response_list=response_list, model=model, batch_data=batch_data
|
||||
)
|
||||
)
|
||||
elif model == "amazon.titan-embed-text-v1":
|
||||
|
|
@ -231,7 +232,7 @@ class BedrockEmbedding(BaseAWSLLM):
|
|||
)
|
||||
elif provider == "nova":
|
||||
returned_response = AmazonNovaEmbeddingConfig()._transform_response(
|
||||
response_list=response_list, model=model
|
||||
response_list=response_list, model=model, batch_data=batch_data
|
||||
)
|
||||
|
||||
##########################################################
|
||||
|
|
@ -310,6 +311,7 @@ class BedrockEmbedding(BaseAWSLLM):
|
|||
model=model,
|
||||
provider=provider,
|
||||
is_async_invoke=is_async_invoke,
|
||||
batch_data=batch_data,
|
||||
)
|
||||
|
||||
async def _async_single_func_embeddings(
|
||||
|
|
@ -379,6 +381,7 @@ class BedrockEmbedding(BaseAWSLLM):
|
|||
model=model,
|
||||
provider=provider,
|
||||
is_async_invoke=is_async_invoke,
|
||||
batch_data=batch_data,
|
||||
)
|
||||
|
||||
def embeddings( # noqa: PLR0915
|
||||
|
|
|
|||
|
|
@ -390,6 +390,109 @@ class TestNovaTransformationResponse:
|
|||
assert result.data[0].embedding == [0.1, 0.2, 0.3]
|
||||
assert result.data[1].embedding == [0.4, 0.5, 0.6]
|
||||
|
||||
def test_image_embedding_response_with_image_count(self):
|
||||
"""Test that Nova image embedding response populates image_count for cost tracking."""
|
||||
config = AmazonNovaEmbeddingConfig()
|
||||
|
||||
response_list = [
|
||||
{
|
||||
"embeddings": [
|
||||
{
|
||||
"embeddingType": "IMAGE",
|
||||
"embedding": [0.1, 0.2, 0.3],
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
|
||||
# Simulate batch_data with image in singleEmbeddingParams
|
||||
batch_data = [
|
||||
{
|
||||
"schemaVersion": "nova-multimodal-embed-v1",
|
||||
"taskType": "SINGLE_EMBEDDING",
|
||||
"singleEmbeddingParams": {
|
||||
"embeddingPurpose": "GENERIC_INDEX",
|
||||
"embeddingDimension": 3072,
|
||||
"image": {
|
||||
"format": "jpeg",
|
||||
"source": {"bytes": "/9j/4AAQSkZJRg=="},
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
result = config._transform_response(
|
||||
response_list=response_list,
|
||||
model="amazon.nova-2-multimodal-embeddings-v1:0",
|
||||
batch_data=batch_data,
|
||||
)
|
||||
|
||||
assert result.usage is not None
|
||||
assert result.usage.prompt_tokens_details is not None
|
||||
assert result.usage.prompt_tokens_details.image_count == 1
|
||||
|
||||
def test_text_embedding_response_no_image_count(self):
|
||||
"""Test that Nova text embedding response does not set image_count."""
|
||||
config = AmazonNovaEmbeddingConfig()
|
||||
|
||||
response_list = [
|
||||
{
|
||||
"embeddings": [
|
||||
{
|
||||
"embeddingType": "TEXT",
|
||||
"embedding": [0.1, 0.2, 0.3],
|
||||
"truncatedCharLength": 20,
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
|
||||
batch_data = [
|
||||
{
|
||||
"schemaVersion": "nova-multimodal-embed-v1",
|
||||
"taskType": "SINGLE_EMBEDDING",
|
||||
"singleEmbeddingParams": {
|
||||
"embeddingPurpose": "GENERIC_INDEX",
|
||||
"embeddingDimension": 3072,
|
||||
"text": {"value": "hello world", "truncationMode": "END"},
|
||||
},
|
||||
}
|
||||
]
|
||||
|
||||
result = config._transform_response(
|
||||
response_list=response_list,
|
||||
model="amazon.nova-2-multimodal-embeddings-v1:0",
|
||||
batch_data=batch_data,
|
||||
)
|
||||
|
||||
assert result.usage is not None
|
||||
assert result.usage.prompt_tokens_details is None
|
||||
|
||||
def test_nova_embedding_backward_compat_no_batch_data(self):
|
||||
"""Test that Nova transformer works without batch_data (backward compatibility)."""
|
||||
config = AmazonNovaEmbeddingConfig()
|
||||
|
||||
response_list = [
|
||||
{
|
||||
"embeddings": [
|
||||
{
|
||||
"embeddingType": "TEXT",
|
||||
"embedding": [0.1, 0.2, 0.3, 0.4, 0.5],
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
|
||||
# Call without batch_data — should not break
|
||||
result = config._transform_response(
|
||||
response_list=response_list,
|
||||
model="amazon.nova-2-multimodal-embeddings-v1:0",
|
||||
)
|
||||
|
||||
assert result.usage is not None
|
||||
assert result.usage.total_tokens > 0
|
||||
assert result.usage.prompt_tokens_details is None
|
||||
|
||||
def test_async_invoke_response(self):
|
||||
"""Test async invoke response transformation."""
|
||||
config = AmazonNovaEmbeddingConfig()
|
||||
|
|
|
|||
|
|
@ -862,3 +862,42 @@ def test_reasoning_tokens_without_text_tokens_gpt5_nano():
|
|||
wrong_cost = 768 * 0.40 / 1_000_000 # Only reasoning tokens
|
||||
assert abs(completion_cost - wrong_cost) > 1e-6, \
|
||||
"Bug detected: Cost calculation is using only reasoning_tokens instead of all completion_tokens!"
|
||||
|
||||
|
||||
def test_image_count_prevents_text_tokens_fallback():
|
||||
"""
|
||||
Test that the text_tokens fallback in generic_cost_per_token does not
|
||||
override text_tokens=0 when image_count > 0.
|
||||
|
||||
Regression test for: Bedrock image embedding double-charging bug.
|
||||
When image_count > 0, text_tokens=0 is intentional (image-only request),
|
||||
not "text_tokens not set by provider."
|
||||
"""
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
# Simulate Nova image-only embedding: prompt_tokens estimated from
|
||||
# embedding dimensions (768 for 3072-dim), image_count=1
|
||||
usage = Usage(
|
||||
prompt_tokens=768,
|
||||
completion_tokens=0,
|
||||
total_tokens=768,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
image_count=1,
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model="amazon.nova-2-multimodal-embeddings-v1:0",
|
||||
usage=usage,
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
|
||||
# Cost should be 1 * input_cost_per_image ($6e-05) = $0.00006
|
||||
# NOT 768 * input_cost_per_token ($1.35e-07) + $0.00006 = $0.000164
|
||||
expected_image_cost = 1 * 6e-05
|
||||
assert prompt_cost == expected_image_cost, (
|
||||
f"Expected prompt_cost={expected_image_cost} (image-only), "
|
||||
f"got {prompt_cost}. text_tokens fallback may be double-charging."
|
||||
)
|
||||
assert completion_cost == 0.0
|
||||
|
|
|
|||
|
|
@ -833,3 +833,125 @@ async def test_bedrock_embedding_custom_headers_with_iam_role_and_custom_api_bas
|
|||
|
||||
except Exception as e:
|
||||
pytest.fail(f"Failed to forward headers with IAM role + custom api_base (async): {str(e)}")
|
||||
|
||||
|
||||
def test_titan_multimodal_embedding_image_cost_tracking():
|
||||
"""Test that Titan multimodal embedding with image input populates image_count in Usage."""
|
||||
from litellm.llms.bedrock.embed.amazon_titan_multimodal_transformation import (
|
||||
AmazonTitanMultimodalEmbeddingG1Config,
|
||||
)
|
||||
|
||||
config = AmazonTitanMultimodalEmbeddingG1Config()
|
||||
|
||||
# Simulate response from AWS Bedrock
|
||||
response_list = [
|
||||
{
|
||||
"embedding": [0.1, 0.2, 0.3],
|
||||
"inputTextTokenCount": 0,
|
||||
}
|
||||
]
|
||||
|
||||
# Simulate batch_data with an image request (inputImage key set by _transform_request)
|
||||
batch_data = [
|
||||
{"inputImage": "/9j/4AAQSkZJRg=="}
|
||||
]
|
||||
|
||||
result = config._transform_response(
|
||||
response_list=response_list,
|
||||
model="amazon.titan-embed-image-v1",
|
||||
batch_data=batch_data,
|
||||
)
|
||||
|
||||
assert result.usage is not None
|
||||
assert result.usage.prompt_tokens_details is not None
|
||||
assert result.usage.prompt_tokens_details.image_count == 1
|
||||
|
||||
|
||||
def test_titan_multimodal_embedding_text_no_image_count():
|
||||
"""Test that Titan multimodal embedding with text-only input does not set image_count."""
|
||||
from litellm.llms.bedrock.embed.amazon_titan_multimodal_transformation import (
|
||||
AmazonTitanMultimodalEmbeddingG1Config,
|
||||
)
|
||||
|
||||
config = AmazonTitanMultimodalEmbeddingG1Config()
|
||||
|
||||
response_list = [
|
||||
{
|
||||
"embedding": [0.1, 0.2, 0.3],
|
||||
"inputTextTokenCount": 5,
|
||||
}
|
||||
]
|
||||
|
||||
# Text-only request — no inputImage key
|
||||
batch_data = [
|
||||
{"inputText": "hello world"}
|
||||
]
|
||||
|
||||
result = config._transform_response(
|
||||
response_list=response_list,
|
||||
model="amazon.titan-embed-image-v1",
|
||||
batch_data=batch_data,
|
||||
)
|
||||
|
||||
assert result.usage is not None
|
||||
# prompt_tokens_details should be None for text-only (no image_count to report)
|
||||
assert result.usage.prompt_tokens_details is None
|
||||
|
||||
|
||||
def test_titan_multimodal_embedding_backward_compat_no_batch_data():
|
||||
"""Test that Titan transformer works without batch_data (backward compatibility)."""
|
||||
from litellm.llms.bedrock.embed.amazon_titan_multimodal_transformation import (
|
||||
AmazonTitanMultimodalEmbeddingG1Config,
|
||||
)
|
||||
|
||||
config = AmazonTitanMultimodalEmbeddingG1Config()
|
||||
|
||||
response_list = [
|
||||
{
|
||||
"embedding": [0.1, 0.2, 0.3],
|
||||
"inputTextTokenCount": 5,
|
||||
}
|
||||
]
|
||||
|
||||
# Call without batch_data — should not break
|
||||
result = config._transform_response(
|
||||
response_list=response_list,
|
||||
model="amazon.titan-embed-image-v1",
|
||||
)
|
||||
|
||||
assert result.usage is not None
|
||||
assert result.usage.prompt_tokens == 5
|
||||
assert result.usage.prompt_tokens_details is None
|
||||
|
||||
|
||||
def test_titan_image_embedding_cost_uses_per_image_rate():
|
||||
"""
|
||||
End-to-end test: Titan image embedding with mocked AWS response
|
||||
should populate image_count for correct per-image cost calculation.
|
||||
"""
|
||||
client = HTTPHandler()
|
||||
|
||||
with patch.object(client, "post") as mock_post:
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 200
|
||||
embed_response = {
|
||||
"embedding": [0.1] * 1024,
|
||||
"inputTextTokenCount": 0,
|
||||
}
|
||||
mock_response.text = json.dumps(embed_response)
|
||||
mock_response.json = lambda: json.loads(mock_response.text)
|
||||
mock_post.return_value = mock_response
|
||||
|
||||
response = litellm.embedding(
|
||||
model="bedrock/amazon.titan-embed-image-v1",
|
||||
input=["data:image/png;base64,iVBORw0KGgoAAAANSUhEUg=="],
|
||||
client=client,
|
||||
aws_access_key_id="fake",
|
||||
aws_secret_access_key="fake",
|
||||
aws_region_name="us-east-1",
|
||||
)
|
||||
|
||||
assert isinstance(response, litellm.EmbeddingResponse)
|
||||
assert response.usage is not None
|
||||
assert response.usage.prompt_tokens_details is not None
|
||||
assert response.usage.prompt_tokens_details.image_count == 1
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue