diff --git a/litellm/litellm_core_utils/token_counter.py b/litellm/litellm_core_utils/token_counter.py index e6a68de07e9..33704e77de1 100644 --- a/litellm/litellm_core_utils/token_counter.py +++ b/litellm/litellm_core_utils/token_counter.py @@ -723,6 +723,13 @@ def _count_content_list( num_tokens += _count_image_tokens( image_url, use_default_image_token_count ) + elif c["type"] == "video_url": + video_url = c.get("video_url", "") + if isinstance(video_url, dict): + video_url = video_url.get("url") + if video_url is None: + video_url = "" + num_tokens += DEFAULT_IMAGE_TOKEN_COUNT + count_function(str(video_url)) elif c["type"] in ("tool_use", "tool_result"): num_tokens += _count_anthropic_content( c, @@ -744,7 +751,7 @@ def _count_content_list( ) raise ValueError( f"Invalid content item type: {content_type}. " - f"Expected str or dict with 'type' field (text, image_url, tool_use, tool_result, thinking)." + f"Expected str or dict with 'type' field (text, image_url, video_url, tool_use, tool_result, thinking)." ) return num_tokens except Exception as e: diff --git a/tests/test_litellm/litellm_core_utils/test_token_counter.py b/tests/test_litellm/litellm_core_utils/test_token_counter.py index 3aa5f012467..4ffb70f6ed8 100644 --- a/tests/test_litellm/litellm_core_utils/test_token_counter.py +++ b/tests/test_litellm/litellm_core_utils/test_token_counter.py @@ -16,6 +16,7 @@ from unittest.mock import AsyncMock, MagicMock, patch import litellm from litellm import create_pretrained_tokenizer, decode, encode, get_modified_max_tokens from litellm import token_counter as token_counter_old +from litellm.constants import DEFAULT_IMAGE_TOKEN_COUNT from litellm.litellm_core_utils.token_counter import token_counter as token_counter_new from tests.large_text import text from tests.test_litellm.litellm_core_utils.messages_with_counts import ( @@ -908,6 +909,75 @@ def test_token_counter_with_image_url(): ), f"Expected detail validation error, got: {e}" +def test_token_counter_with_video_url(): + messages_dict = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "Please describe this video."}, + { + "type": "video_url", + "video_url": { + "url": "data:video/mp4;base64,AAAA", + }, + }, + ], + } + ] + + tokens_dict = token_counter(model="gpt-4o", messages=messages_dict) + assert tokens_dict > 0, f"Expected positive token count, got {tokens_dict}" + + messages_str = [ + { + "role": "user", + "content": [ + { + "type": "video_url", + "video_url": "https://example.com/video.mp4", + } + ], + } + ] + + tokens_str = token_counter(model="gpt-4o", messages=messages_str) + assert ( + tokens_str > 0 + ), f"Expected positive token count for string video_url, got {tokens_str}" + assert ( + tokens_str > DEFAULT_IMAGE_TOKEN_COUNT + ), f"Expected default video token budget, got {tokens_str}" + + messages_empty_url = [ + { + "role": "user", + "content": [{"type": "video_url", "video_url": ""}], + } + ] + messages_none_url = [ + { + "role": "user", + "content": [{"type": "video_url", "video_url": None}], + } + ] + messages_none_nested_url = [ + { + "role": "user", + "content": [{"type": "video_url", "video_url": {"url": None}}], + } + ] + + tokens_empty_url = token_counter(model="gpt-4o", messages=messages_empty_url) + assert tokens_empty_url > DEFAULT_IMAGE_TOKEN_COUNT + assert ( + token_counter(model="gpt-4o", messages=messages_none_url) == tokens_empty_url + ) + assert ( + token_counter(model="gpt-4o", messages=messages_none_nested_url) + == tokens_empty_url + ) + + def test_token_counter_with_thinking_content(): """ Test that _count_content_list() correctly handles Claude's extended thinking content blocks.