fix(vertex): only bill image rate without modality details when every input is an image

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
kerry 2026-09-15 01:16:00 +00:00
parent 6cbed7b4c0
commit 6a18105275
2 changed files with 28 additions and 4 deletions

View file

@ -337,11 +337,12 @@ def _is_image_element(
return False
def _count_input_images(
def _is_image_only_input(
input: GeminiEmbeddingInput,
resolved_files: Mapping[str, Mapping[str, str]],
) -> int:
return sum(1 for element in _flatten_input(input) if _is_image_element(element, resolved_files))
) -> bool:
elements: Final = _flatten_input(input)
return bool(elements) and all(_is_image_element(element, resolved_files) for element in elements)
def _tokens_for_modality(details: Sequence[PromptTokensDetails], modality: str) -> int:
@ -376,7 +377,7 @@ def _usage_from_embed_content_response(
total_tokens=total_tokens,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=0,
image_tokens=prompt_tokens if _count_input_images(input, resolved_files) else 0,
image_tokens=prompt_tokens if _is_image_only_input(input, resolved_files) else 0,
),
)

View file

@ -524,6 +524,29 @@ class TestProcessEmbedContentResponseUsage:
)
assert prompt_cost == pytest.approx(258 * 4.5e-7)
def test_mixed_text_and_image_without_modality_details_not_billed_as_image(self):
response_json = {
"embedding": {"values": [0.1]},
"usageMetadata": {
"promptTokenCount": 270,
"totalTokenCount": 270,
},
}
result = process_embed_content_response(
input=["a short caption", IMAGE_DATA_URI],
model_response=EmbeddingResponse(),
model=self.MODEL,
response_json=response_json,
)
assert result.usage.prompt_tokens_details.image_tokens == 0
prompt_cost, _ = generic_cost_per_token(
model=self.MODEL,
usage=result.usage,
custom_llm_provider="vertex_ai",
)
assert prompt_cost == pytest.approx(270 * 2e-7)
def test_text_without_modality_details_uses_text_rate(self):
response_json = {
"embedding": {"values": [0.1]},