From 777ae4f530303522d69b82c2c0ee7392c1b76642 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Sat, 10 Jan 2026 00:57:11 +0530 Subject: [PATCH 1/2] Fix :test_count_tokens_caching --- tests/test_litellm/test_utils_custom.py | 31 ++++++++++++++----------- 1 file changed, 17 insertions(+), 14 deletions(-) diff --git a/tests/test_litellm/test_utils_custom.py b/tests/test_litellm/test_utils_custom.py index 292da4132b9..3e924e9c719 100644 --- a/tests/test_litellm/test_utils_custom.py +++ b/tests/test_litellm/test_utils_custom.py @@ -1,4 +1,5 @@ import pytest +import sys from unittest.mock import MagicMock, patch, AsyncMock from litellm.proxy.utils import count_tokens_with_anthropic_api, _anthropic_async_clients @@ -14,29 +15,31 @@ async def test_count_tokens_caching(): messages = [{"role": "user", "content": "hello"}] model = "claude-3-opus-20240229" - # Mock anthropic - with patch("anthropic.AsyncAnthropic") as mock_cls: - mock_client = MagicMock() - mock_cls.return_value = mock_client - - # Mock response - mock_response = MagicMock() - mock_response.input_tokens = 10 - - # Setup async return for count_tokens - mock_client.beta.messages.count_tokens = AsyncMock(return_value=mock_response) - + # Create a mock anthropic module + mock_anthropic = MagicMock() + mock_client = MagicMock() + mock_anthropic.AsyncAnthropic.return_value = mock_client + + # Mock response + mock_response = MagicMock() + mock_response.input_tokens = 10 + + # Setup async return for count_tokens + mock_client.beta.messages.count_tokens = AsyncMock(return_value=mock_response) + + # Patch sys.modules to ensure our mock is used when anthropic is imported + with patch.dict(sys.modules, {"anthropic": mock_anthropic}): # First call with patch.dict("os.environ", {"ANTHROPIC_API_KEY": api_key}): await count_tokens_with_anthropic_api(model, messages) assert api_key in _anthropic_async_clients assert _anthropic_async_clients[api_key] == mock_client - mock_cls.assert_called_once() # Should be called once + mock_anthropic.AsyncAnthropic.assert_called_once() # Should be called once # Second call with patch.dict("os.environ", {"ANTHROPIC_API_KEY": api_key}): await count_tokens_with_anthropic_api(model, messages) # Should still be called once (cached) - mock_cls.assert_called_once() + mock_anthropic.AsyncAnthropic.assert_called_once() From aba7dcea9ca13fdc18c867bcddf343d5f0291ab5 Mon Sep 17 00:00:00 2001 From: Sameer Kankute Date: Sat, 10 Jan 2026 01:03:50 +0530 Subject: [PATCH 2/2] Fix : litellm import error --- litellm/llms/bedrock/count_tokens/handler.py | 17 ++++++++--------- .../llm_passthrough_endpoints.py | 7 +++++++ 2 files changed, 15 insertions(+), 9 deletions(-) diff --git a/litellm/llms/bedrock/count_tokens/handler.py b/litellm/llms/bedrock/count_tokens/handler.py index 60ace7f3369..e8366165b65 100644 --- a/litellm/llms/bedrock/count_tokens/handler.py +++ b/litellm/llms/bedrock/count_tokens/handler.py @@ -6,10 +6,9 @@ Simplified handler leveraging existing LiteLLM Bedrock infrastructure. from typing import Any, Dict -from fastapi import HTTPException - import litellm from litellm._logging import verbose_logger +from litellm.llms.bedrock.common_utils import BedrockError from litellm.llms.bedrock.count_tokens.transformation import BedrockCountTokensConfig from litellm.llms.custom_httpx.http_handler import get_async_httpx_client @@ -97,9 +96,9 @@ class BedrockCountTokensHandler(BedrockCountTokensConfig): if response.status_code != 200: error_text = response.text verbose_logger.error(f"AWS Bedrock error: {error_text}") - raise HTTPException( - status_code=400, - detail={"error": f"AWS Bedrock error: {error_text}"}, + raise BedrockError( + status_code=response.status_code, + message=f"AWS Bedrock error: {error_text}", ) bedrock_response = response.json() @@ -115,12 +114,12 @@ class BedrockCountTokensHandler(BedrockCountTokensConfig): return final_response - except HTTPException: - # Re-raise HTTP exceptions as-is + except BedrockError: + # Re-raise Bedrock exceptions as-is raise except Exception as e: verbose_logger.error(f"Error in CountTokens handler: {str(e)}") - raise HTTPException( + raise BedrockError( status_code=500, - detail={"error": f"CountTokens processing error: {str(e)}"}, + message=f"CountTokens processing error: {str(e)}", ) diff --git a/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py b/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py index 84550092d2e..d9798dae690 100644 --- a/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py +++ b/litellm/proxy/pass_through_endpoints/llm_passthrough_endpoints.py @@ -776,6 +776,7 @@ async def handle_bedrock_count_tokens( - /v1/messages/count_tokens - /v1/messages/count-tokens """ + from litellm.llms.bedrock.common_utils import BedrockError from litellm.llms.bedrock.count_tokens.handler import BedrockCountTokensHandler from litellm.proxy.proxy_server import llm_router @@ -822,6 +823,12 @@ async def handle_bedrock_count_tokens( return result + except BedrockError as e: + # Convert BedrockError to HTTPException for FastAPI + verbose_proxy_logger.error(f"BedrockError in handle_bedrock_count_tokens: {str(e)}") + raise HTTPException( + status_code=e.status_code, detail={"error": e.message} + ) except HTTPException: # Re-raise HTTP exceptions as-is raise