From a395a25705d70790519fc5813435ad6227d8d5ca Mon Sep 17 00:00:00 2001 From: michelligabriele Date: Fri, 20 Feb 2026 17:51:21 +0100 Subject: [PATCH] fix(cost-calc): use per-image pricing for Bedrock multimodal embeddings (#21646) Bedrock multimodal embedding models (Titan and Nova) were being costed using the per-token text rate instead of the correct flat per-image rate ($0.00006/image). The pricing data was correct but never applied because image_count was never populated in prompt_tokens_details. Pass batch_data to Titan/Nova response transformers so they can count image inputs and set PromptTokensDetailsWrapper(image_count=N) on Usage, mirroring the existing Vertex AI pattern from PR #9623. Also fix the text_tokens fallback in generic_cost_per_token to not override text_tokens=0 when image_count > 0 (image-only requests). --- .../litellm_core_utils/llm_cost_calc/utils.py | 2 +- .../embed/amazon_nova_transformation.py | 41 ++++-- .../amazon_titan_multimodal_transformation.py | 23 +++- litellm/llms/bedrock/embed/embedding.py | 7 +- .../test_bedrock_nova_embedding.py | 103 +++++++++++++++ .../llm_cost_calc/test_llm_cost_calc_utils.py | 39 ++++++ .../bedrock/embed/test_bedrock_embedding.py | 122 ++++++++++++++++++ 7 files changed, 323 insertions(+), 14 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/utils.py b/litellm/litellm_core_utils/llm_cost_calc/utils.py index 2308dc7beca..81e5e97d3cc 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/utils.py +++ b/litellm/litellm_core_utils/llm_cost_calc/utils.py @@ -602,7 +602,7 @@ def generic_cost_per_token( # noqa: PLR0915 total_details = text_tokens + cache_hit + audio_tokens + cache_creation + image_tokens has_double_counting = cache_hit > 0 and total_details > usage.prompt_tokens - if text_tokens == 0 or has_double_counting: + if (text_tokens == 0 and prompt_tokens_details["image_count"] == 0) or has_double_counting: text_tokens = ( usage.prompt_tokens - cache_hit diff --git a/litellm/llms/bedrock/embed/amazon_nova_transformation.py b/litellm/llms/bedrock/embed/amazon_nova_transformation.py index 3e5686c46fb..40d2a21e1c7 100644 --- a/litellm/llms/bedrock/embed/amazon_nova_transformation.py +++ b/litellm/llms/bedrock/embed/amazon_nova_transformation.py @@ -14,7 +14,7 @@ Docs - https://docs.aws.amazon.com/bedrock/latest/userguide/nova-embed.html from typing import List, Optional -from litellm.types.utils import Embedding, EmbeddingResponse, Usage +from litellm.types.utils import Embedding, EmbeddingResponse, PromptTokensDetailsWrapper, Usage class AmazonNovaEmbeddingConfig: @@ -244,11 +244,14 @@ class AmazonNovaEmbeddingConfig: } def _transform_response( - self, response_list: List[dict], model: str + self, + response_list: List[dict], + model: str, + batch_data: Optional[List[dict]] = None, ) -> EmbeddingResponse: """ Transform Nova response to OpenAI format. - + Nova response format: { "embeddings": [ @@ -262,7 +265,7 @@ class AmazonNovaEmbeddingConfig: """ embeddings: List[Embedding] = [] total_tokens = 0 - + for response in response_list: # Nova response has an "embeddings" array if "embeddings" in response and isinstance(response["embeddings"], list): @@ -274,7 +277,7 @@ class AmazonNovaEmbeddingConfig: object="embedding", ) embeddings.append(embedding) - + # Estimate token count # For text, use truncatedCharLength if available if "truncatedCharLength" in item: @@ -291,9 +294,31 @@ class AmazonNovaEmbeddingConfig: ) embeddings.append(embedding) total_tokens += len(response["embedding"]) // 4 - - usage = Usage(prompt_tokens=total_tokens, total_tokens=total_tokens) - + + # Count images from original requests for cost calculation + image_count = 0 + if batch_data: + for request_data in batch_data: + # Nova wraps params in singleEmbeddingParams or segmentedEmbeddingParams + params = request_data.get( + "singleEmbeddingParams", + request_data.get("segmentedEmbeddingParams", {}), + ) + if "image" in params: + image_count += 1 + + prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None + if image_count > 0: + prompt_tokens_details = PromptTokensDetailsWrapper( + image_count=image_count, + ) + + usage = Usage( + prompt_tokens=total_tokens, + total_tokens=total_tokens, + prompt_tokens_details=prompt_tokens_details, + ) + return EmbeddingResponse(data=embeddings, model=model, usage=usage) def _transform_async_invoke_response( diff --git a/litellm/llms/bedrock/embed/amazon_titan_multimodal_transformation.py b/litellm/llms/bedrock/embed/amazon_titan_multimodal_transformation.py index 338029adc35..e59d3cbf776 100644 --- a/litellm/llms/bedrock/embed/amazon_titan_multimodal_transformation.py +++ b/litellm/llms/bedrock/embed/amazon_titan_multimodal_transformation.py @@ -6,14 +6,14 @@ Why separate file? Make it easy to see how transformation works Docs - https://docs.aws.amazon.com/bedrock/latest/userguide/model-parameters-titan-embed-mm.html """ -from typing import List +from typing import List, Optional from litellm.types.llms.bedrock import ( AmazonTitanMultimodalEmbeddingConfig, AmazonTitanMultimodalEmbeddingRequest, AmazonTitanMultimodalEmbeddingResponse, ) -from litellm.types.utils import Embedding, EmbeddingResponse, Usage +from litellm.types.utils import Embedding, EmbeddingResponse, PromptTokensDetailsWrapper, Usage from litellm.utils import get_base64_str, is_base64_encoded @@ -56,7 +56,10 @@ class AmazonTitanMultimodalEmbeddingG1Config: return transformed_request def _transform_response( - self, response_list: List[dict], model: str + self, + response_list: List[dict], + model: str, + batch_data: Optional[List[dict]] = None, ) -> EmbeddingResponse: total_prompt_tokens = 0 transformed_responses: List[Embedding] = [] @@ -71,9 +74,23 @@ class AmazonTitanMultimodalEmbeddingG1Config: ) total_prompt_tokens += _parsed_response["inputTextTokenCount"] + # Count images from original requests for cost calculation + image_count = 0 + if batch_data: + for request_data in batch_data: + if "inputImage" in request_data: + image_count += 1 + + prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None + if image_count > 0: + prompt_tokens_details = PromptTokensDetailsWrapper( + image_count=image_count, + ) + usage = Usage( prompt_tokens=total_prompt_tokens, completion_tokens=0, total_tokens=total_prompt_tokens, + prompt_tokens_details=prompt_tokens_details, ) return EmbeddingResponse(model=model, usage=usage, data=transformed_responses) diff --git a/litellm/llms/bedrock/embed/embedding.py b/litellm/llms/bedrock/embed/embedding.py index 56900d296a5..783345d78da 100644 --- a/litellm/llms/bedrock/embed/embedding.py +++ b/litellm/llms/bedrock/embed/embedding.py @@ -158,6 +158,7 @@ class BedrockEmbedding(BaseAWSLLM): model: str, provider: BEDROCK_EMBEDDING_PROVIDERS_LITERAL, is_async_invoke: Optional[bool] = False, + batch_data: Optional[List[dict]] = None, ) -> Optional[EmbeddingResponse]: """ Transforms the response from the Bedrock embedding provider to the OpenAI format. @@ -212,7 +213,7 @@ class BedrockEmbedding(BaseAWSLLM): if model == "amazon.titan-embed-image-v1": returned_response = ( AmazonTitanMultimodalEmbeddingG1Config()._transform_response( - response_list=response_list, model=model + response_list=response_list, model=model, batch_data=batch_data ) ) elif model == "amazon.titan-embed-text-v1": @@ -231,7 +232,7 @@ class BedrockEmbedding(BaseAWSLLM): ) elif provider == "nova": returned_response = AmazonNovaEmbeddingConfig()._transform_response( - response_list=response_list, model=model + response_list=response_list, model=model, batch_data=batch_data ) ########################################################## @@ -310,6 +311,7 @@ class BedrockEmbedding(BaseAWSLLM): model=model, provider=provider, is_async_invoke=is_async_invoke, + batch_data=batch_data, ) async def _async_single_func_embeddings( @@ -379,6 +381,7 @@ class BedrockEmbedding(BaseAWSLLM): model=model, provider=provider, is_async_invoke=is_async_invoke, + batch_data=batch_data, ) def embeddings( # noqa: PLR0915 diff --git a/tests/llm_translation/test_bedrock_nova_embedding.py b/tests/llm_translation/test_bedrock_nova_embedding.py index 23064a3389a..aeb86dcdb9c 100644 --- a/tests/llm_translation/test_bedrock_nova_embedding.py +++ b/tests/llm_translation/test_bedrock_nova_embedding.py @@ -390,6 +390,109 @@ class TestNovaTransformationResponse: assert result.data[0].embedding == [0.1, 0.2, 0.3] assert result.data[1].embedding == [0.4, 0.5, 0.6] + def test_image_embedding_response_with_image_count(self): + """Test that Nova image embedding response populates image_count for cost tracking.""" + config = AmazonNovaEmbeddingConfig() + + response_list = [ + { + "embeddings": [ + { + "embeddingType": "IMAGE", + "embedding": [0.1, 0.2, 0.3], + } + ] + } + ] + + # Simulate batch_data with image in singleEmbeddingParams + batch_data = [ + { + "schemaVersion": "nova-multimodal-embed-v1", + "taskType": "SINGLE_EMBEDDING", + "singleEmbeddingParams": { + "embeddingPurpose": "GENERIC_INDEX", + "embeddingDimension": 3072, + "image": { + "format": "jpeg", + "source": {"bytes": "/9j/4AAQSkZJRg=="}, + }, + }, + } + ] + + result = config._transform_response( + response_list=response_list, + model="amazon.nova-2-multimodal-embeddings-v1:0", + batch_data=batch_data, + ) + + assert result.usage is not None + assert result.usage.prompt_tokens_details is not None + assert result.usage.prompt_tokens_details.image_count == 1 + + def test_text_embedding_response_no_image_count(self): + """Test that Nova text embedding response does not set image_count.""" + config = AmazonNovaEmbeddingConfig() + + response_list = [ + { + "embeddings": [ + { + "embeddingType": "TEXT", + "embedding": [0.1, 0.2, 0.3], + "truncatedCharLength": 20, + } + ] + } + ] + + batch_data = [ + { + "schemaVersion": "nova-multimodal-embed-v1", + "taskType": "SINGLE_EMBEDDING", + "singleEmbeddingParams": { + "embeddingPurpose": "GENERIC_INDEX", + "embeddingDimension": 3072, + "text": {"value": "hello world", "truncationMode": "END"}, + }, + } + ] + + result = config._transform_response( + response_list=response_list, + model="amazon.nova-2-multimodal-embeddings-v1:0", + batch_data=batch_data, + ) + + assert result.usage is not None + assert result.usage.prompt_tokens_details is None + + def test_nova_embedding_backward_compat_no_batch_data(self): + """Test that Nova transformer works without batch_data (backward compatibility).""" + config = AmazonNovaEmbeddingConfig() + + response_list = [ + { + "embeddings": [ + { + "embeddingType": "TEXT", + "embedding": [0.1, 0.2, 0.3, 0.4, 0.5], + } + ] + } + ] + + # Call without batch_data — should not break + result = config._transform_response( + response_list=response_list, + model="amazon.nova-2-multimodal-embeddings-v1:0", + ) + + assert result.usage is not None + assert result.usage.total_tokens > 0 + assert result.usage.prompt_tokens_details is None + def test_async_invoke_response(self): """Test async invoke response transformation.""" config = AmazonNovaEmbeddingConfig() diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index 9e70f3e08d5..cee5392261e 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -862,3 +862,42 @@ def test_reasoning_tokens_without_text_tokens_gpt5_nano(): wrong_cost = 768 * 0.40 / 1_000_000 # Only reasoning tokens assert abs(completion_cost - wrong_cost) > 1e-6, \ "Bug detected: Cost calculation is using only reasoning_tokens instead of all completion_tokens!" + + +def test_image_count_prevents_text_tokens_fallback(): + """ + Test that the text_tokens fallback in generic_cost_per_token does not + override text_tokens=0 when image_count > 0. + + Regression test for: Bedrock image embedding double-charging bug. + When image_count > 0, text_tokens=0 is intentional (image-only request), + not "text_tokens not set by provider." + """ + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + # Simulate Nova image-only embedding: prompt_tokens estimated from + # embedding dimensions (768 for 3072-dim), image_count=1 + usage = Usage( + prompt_tokens=768, + completion_tokens=0, + total_tokens=768, + prompt_tokens_details=PromptTokensDetailsWrapper( + image_count=1, + ), + ) + + prompt_cost, completion_cost = generic_cost_per_token( + model="amazon.nova-2-multimodal-embeddings-v1:0", + usage=usage, + custom_llm_provider="bedrock", + ) + + # Cost should be 1 * input_cost_per_image ($6e-05) = $0.00006 + # NOT 768 * input_cost_per_token ($1.35e-07) + $0.00006 = $0.000164 + expected_image_cost = 1 * 6e-05 + assert prompt_cost == expected_image_cost, ( + f"Expected prompt_cost={expected_image_cost} (image-only), " + f"got {prompt_cost}. text_tokens fallback may be double-charging." + ) + assert completion_cost == 0.0 diff --git a/tests/test_litellm/llms/bedrock/embed/test_bedrock_embedding.py b/tests/test_litellm/llms/bedrock/embed/test_bedrock_embedding.py index 2aa297ad219..a38a6612f79 100644 --- a/tests/test_litellm/llms/bedrock/embed/test_bedrock_embedding.py +++ b/tests/test_litellm/llms/bedrock/embed/test_bedrock_embedding.py @@ -833,3 +833,125 @@ async def test_bedrock_embedding_custom_headers_with_iam_role_and_custom_api_bas except Exception as e: pytest.fail(f"Failed to forward headers with IAM role + custom api_base (async): {str(e)}") + + +def test_titan_multimodal_embedding_image_cost_tracking(): + """Test that Titan multimodal embedding with image input populates image_count in Usage.""" + from litellm.llms.bedrock.embed.amazon_titan_multimodal_transformation import ( + AmazonTitanMultimodalEmbeddingG1Config, + ) + + config = AmazonTitanMultimodalEmbeddingG1Config() + + # Simulate response from AWS Bedrock + response_list = [ + { + "embedding": [0.1, 0.2, 0.3], + "inputTextTokenCount": 0, + } + ] + + # Simulate batch_data with an image request (inputImage key set by _transform_request) + batch_data = [ + {"inputImage": "/9j/4AAQSkZJRg=="} + ] + + result = config._transform_response( + response_list=response_list, + model="amazon.titan-embed-image-v1", + batch_data=batch_data, + ) + + assert result.usage is not None + assert result.usage.prompt_tokens_details is not None + assert result.usage.prompt_tokens_details.image_count == 1 + + +def test_titan_multimodal_embedding_text_no_image_count(): + """Test that Titan multimodal embedding with text-only input does not set image_count.""" + from litellm.llms.bedrock.embed.amazon_titan_multimodal_transformation import ( + AmazonTitanMultimodalEmbeddingG1Config, + ) + + config = AmazonTitanMultimodalEmbeddingG1Config() + + response_list = [ + { + "embedding": [0.1, 0.2, 0.3], + "inputTextTokenCount": 5, + } + ] + + # Text-only request — no inputImage key + batch_data = [ + {"inputText": "hello world"} + ] + + result = config._transform_response( + response_list=response_list, + model="amazon.titan-embed-image-v1", + batch_data=batch_data, + ) + + assert result.usage is not None + # prompt_tokens_details should be None for text-only (no image_count to report) + assert result.usage.prompt_tokens_details is None + + +def test_titan_multimodal_embedding_backward_compat_no_batch_data(): + """Test that Titan transformer works without batch_data (backward compatibility).""" + from litellm.llms.bedrock.embed.amazon_titan_multimodal_transformation import ( + AmazonTitanMultimodalEmbeddingG1Config, + ) + + config = AmazonTitanMultimodalEmbeddingG1Config() + + response_list = [ + { + "embedding": [0.1, 0.2, 0.3], + "inputTextTokenCount": 5, + } + ] + + # Call without batch_data — should not break + result = config._transform_response( + response_list=response_list, + model="amazon.titan-embed-image-v1", + ) + + assert result.usage is not None + assert result.usage.prompt_tokens == 5 + assert result.usage.prompt_tokens_details is None + + +def test_titan_image_embedding_cost_uses_per_image_rate(): + """ + End-to-end test: Titan image embedding with mocked AWS response + should populate image_count for correct per-image cost calculation. + """ + client = HTTPHandler() + + with patch.object(client, "post") as mock_post: + mock_response = Mock() + mock_response.status_code = 200 + embed_response = { + "embedding": [0.1] * 1024, + "inputTextTokenCount": 0, + } + mock_response.text = json.dumps(embed_response) + mock_response.json = lambda: json.loads(mock_response.text) + mock_post.return_value = mock_response + + response = litellm.embedding( + model="bedrock/amazon.titan-embed-image-v1", + input=["data:image/png;base64,iVBORw0KGgoAAAANSUhEUg=="], + client=client, + aws_access_key_id="fake", + aws_secret_access_key="fake", + aws_region_name="us-east-1", + ) + + assert isinstance(response, litellm.EmbeddingResponse) + assert response.usage is not None + assert response.usage.prompt_tokens_details is not None + assert response.usage.prompt_tokens_details.image_count == 1