fix(router): count deployment itpm reservations off the event loop

This commit is contained in:
mateo-berri 2026-09-07 21:29:26 -07:00
parent dcd38ab9f0
commit 83d500bac9
3 changed files with 32 additions and 3 deletions

View file

@ -315,12 +315,12 @@ def calculate_img_tokens(
TokenCounterFunction = Callable[[str], int]
EXTRAPOLATION_SAMPLES: Final = 16
"""
Type for a function that counts tokens in a string.
"""
EXTRAPOLATION_SAMPLES: Final = 16
def _get_tiktoken_count_function(
encode_length: Callable[[str], int],

View file

@ -21,6 +21,7 @@ import litellm
from litellm import token_counter
from litellm._logging import verbose_router_logger
from litellm.caching.dual_cache import DualCache
from litellm.litellm_core_utils.asyncify import asyncify
from litellm.types.router import RouterCacheEnum, RouterErrors
from litellm.utils import get_utc_datetime
@ -466,7 +467,7 @@ async def async_io_token_pre_call_check(
request_kwargs: Final = get_io_token_rate_limit_request_kwargs()
_model: Final = (deployment.get("litellm_params") or {}).get("model") or ""
estimated_input: Final = _estimate_input_tokens(request_kwargs, model=_model)
estimated_input: Final = await asyncify(_estimate_input_tokens)(request_kwargs, model=_model)
max_tokens: Final = _resolve_max_tokens(request_kwargs, deployment)
dt: Final = get_utc_datetime()

View file

@ -1039,3 +1039,31 @@ class TestContextSlotRetention:
assert deployment is not None
router._update_kwargs_with_deployment(deployment=deployment.model_dump(), kwargs=kwargs)
assert get_io_token_rate_limit_request_kwargs() is kwargs
@pytest.mark.asyncio
async def test_the_deployment_itpm_reservation_counts_the_request_off_the_event_loop():
from litellm.utils import get_utc_datetime
from tests.large_text import text
from tests.test_litellm.litellm_core_utils.event_loop_lag import (
assert_loop_stayed_free,
timed_with_loop_lags,
warm_tokenizer,
)
dual_cache = DualCache()
check = ModelRateLimitingCheck(dual_cache=dual_cache)
warm_tokenizer("anthropic/claude-fable-5")
deployment = {
"litellm_params": {"model": "anthropic/claude-fable-5", "itpm": 10_000_000},
"model_info": {"id": "io-loop-id"},
"model_name": "claude",
}
set_io_token_rate_limit_request_kwargs({"messages": [{"role": "user", "content": text * 100}], "metadata": {}})
_, took, lags = await timed_with_loop_lags(lambda: check.async_pre_call_check(deployment))
minute = get_utc_datetime().strftime("%H-%M")
reserved = await dual_cache.async_get_cache(key=f"global_router:io-loop-id:anthropic/claude-fable-5:itpm:{minute}")
assert reserved > 100_000
assert_loop_stayed_free(took, lags)