diff --git a/litellm/litellm_core_utils/token_counter.py b/litellm/litellm_core_utils/token_counter.py index 33704e77de1..1ca70f378cd 100644 --- a/litellm/litellm_core_utils/token_counter.py +++ b/litellm/litellm_core_utils/token_counter.py @@ -724,12 +724,7 @@ def _count_content_list( image_url, use_default_image_token_count ) elif c["type"] == "video_url": - video_url = c.get("video_url", "") - if isinstance(video_url, dict): - video_url = video_url.get("url") - if video_url is None: - video_url = "" - num_tokens += DEFAULT_IMAGE_TOKEN_COUNT + count_function(str(video_url)) + num_tokens += DEFAULT_IMAGE_TOKEN_COUNT elif c["type"] in ("tool_use", "tool_result"): num_tokens += _count_anthropic_content( c, diff --git a/tests/test_litellm/litellm_core_utils/test_token_counter.py b/tests/test_litellm/litellm_core_utils/test_token_counter.py index 4ffb70f6ed8..21039d2d419 100644 --- a/tests/test_litellm/litellm_core_utils/test_token_counter.py +++ b/tests/test_litellm/litellm_core_utils/test_token_counter.py @@ -948,6 +948,21 @@ def test_token_counter_with_video_url(): tokens_str > DEFAULT_IMAGE_TOKEN_COUNT ), f"Expected default video token budget, got {tokens_str}" + messages_base64_str = [ + { + "role": "user", + "content": [ + { + "type": "video_url", + "video_url": "data:video/mp4;base64," + ("A" * 4000), + } + ], + } + ] + assert ( + token_counter(model="gpt-4o", messages=messages_base64_str) == tokens_str + ) + messages_empty_url = [ { "role": "user",