From 0be571296157c5dcda844e9c7e062b1355185c61 Mon Sep 17 00:00:00 2001 From: Harshit Jain Date: Sat, 10 Jan 2026 00:15:13 +0530 Subject: [PATCH] add resonable max_tokens_estimate & removed token_counter --- .../proxy/hooks/parallel_request_limiter_v3.py | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/hooks/parallel_request_limiter_v3.py b/litellm/proxy/hooks/parallel_request_limiter_v3.py index de3be1e714b..34c15d3ea4e 100644 --- a/litellm/proxy/hooks/parallel_request_limiter_v3.py +++ b/litellm/proxy/hooks/parallel_request_limiter_v3.py @@ -161,7 +161,7 @@ REDIS_NODE_HASHTAG_NAME = "all_keys" # TPM Token Reservation Constants # When max_tokens is not specified in the request, use this default for estimation -DEFAULT_MAX_TOKENS_ESTIMATE = 256 +DEFAULT_MAX_TOKENS_ESTIMATE = 4096 # Fallback: approximate characters per token for rough estimation DEFAULT_CHARS_PER_TOKEN = 4 # Metadata key for storing reserved tokens in request data @@ -286,12 +286,12 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger): if messages: # Chat completions try: - from litellm import token_counter - - estimated_input_tokens = token_counter( - model=model or "", - messages=messages, + total_chars = sum( + len(str(m.get("content", ""))) + for m in messages + if isinstance(m, dict) ) + estimated_input_tokens = total_chars // DEFAULT_CHARS_PER_TOKEN except Exception as e: verbose_proxy_logger.debug( f"Token counting failed, using fallback estimation: {str(e)}" @@ -2010,9 +2010,9 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger): from litellm.types.caching import RedisPipelineIncrementOperation try: - litellm_parent_otel_span: Union[ - Span, None - ] = _get_parent_otel_span_from_kwargs(kwargs) + litellm_parent_otel_span: Union[Span, None] = ( + _get_parent_otel_span_from_kwargs(kwargs) + ) # Get metadata from standard_logging_object - this correctly handles both # 'metadata' and 'litellm_metadata' fields from litellm_params standard_logging_object = kwargs.get("standard_logging_object") or {}