mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(router): count deployment itpm reservations off the event loop
This commit is contained in:
parent
dcd38ab9f0
commit
83d500bac9
3 changed files with 32 additions and 3 deletions
|
|
@ -315,12 +315,12 @@ def calculate_img_tokens(
|
|||
|
||||
|
||||
TokenCounterFunction = Callable[[str], int]
|
||||
|
||||
EXTRAPOLATION_SAMPLES: Final = 16
|
||||
"""
|
||||
Type for a function that counts tokens in a string.
|
||||
"""
|
||||
|
||||
EXTRAPOLATION_SAMPLES: Final = 16
|
||||
|
||||
|
||||
def _get_tiktoken_count_function(
|
||||
encode_length: Callable[[str], int],
|
||||
|
|
|
|||
|
|
@ -21,6 +21,7 @@ import litellm
|
|||
from litellm import token_counter
|
||||
from litellm._logging import verbose_router_logger
|
||||
from litellm.caching.dual_cache import DualCache
|
||||
from litellm.litellm_core_utils.asyncify import asyncify
|
||||
from litellm.types.router import RouterCacheEnum, RouterErrors
|
||||
from litellm.utils import get_utc_datetime
|
||||
|
||||
|
|
@ -466,7 +467,7 @@ async def async_io_token_pre_call_check(
|
|||
|
||||
request_kwargs: Final = get_io_token_rate_limit_request_kwargs()
|
||||
_model: Final = (deployment.get("litellm_params") or {}).get("model") or ""
|
||||
estimated_input: Final = _estimate_input_tokens(request_kwargs, model=_model)
|
||||
estimated_input: Final = await asyncify(_estimate_input_tokens)(request_kwargs, model=_model)
|
||||
max_tokens: Final = _resolve_max_tokens(request_kwargs, deployment)
|
||||
|
||||
dt: Final = get_utc_datetime()
|
||||
|
|
|
|||
|
|
@ -1039,3 +1039,31 @@ class TestContextSlotRetention:
|
|||
assert deployment is not None
|
||||
router._update_kwargs_with_deployment(deployment=deployment.model_dump(), kwargs=kwargs)
|
||||
assert get_io_token_rate_limit_request_kwargs() is kwargs
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_the_deployment_itpm_reservation_counts_the_request_off_the_event_loop():
|
||||
from litellm.utils import get_utc_datetime
|
||||
from tests.large_text import text
|
||||
from tests.test_litellm.litellm_core_utils.event_loop_lag import (
|
||||
assert_loop_stayed_free,
|
||||
timed_with_loop_lags,
|
||||
warm_tokenizer,
|
||||
)
|
||||
|
||||
dual_cache = DualCache()
|
||||
check = ModelRateLimitingCheck(dual_cache=dual_cache)
|
||||
warm_tokenizer("anthropic/claude-fable-5")
|
||||
deployment = {
|
||||
"litellm_params": {"model": "anthropic/claude-fable-5", "itpm": 10_000_000},
|
||||
"model_info": {"id": "io-loop-id"},
|
||||
"model_name": "claude",
|
||||
}
|
||||
set_io_token_rate_limit_request_kwargs({"messages": [{"role": "user", "content": text * 100}], "metadata": {}})
|
||||
|
||||
_, took, lags = await timed_with_loop_lags(lambda: check.async_pre_call_check(deployment))
|
||||
|
||||
minute = get_utc_datetime().strftime("%H-%M")
|
||||
reserved = await dual_cache.async_get_cache(key=f"global_router:io-loop-id:anthropic/claude-fable-5:itpm:{minute}")
|
||||
assert reserved > 100_000
|
||||
assert_loop_stayed_free(took, lags)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue