mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(ptu): reserve an uncapped request against the cap the proxy writes, not its prompt length
This commit is contained in:
parent
e65e12d728
commit
495cf74ccd
2 changed files with 24 additions and 1 deletions
|
|
@ -4236,9 +4236,10 @@ class _PROXY_MaxParallelRequestsHandler_v3(CustomLogger):
|
|||
min_configured_tpm_limit,
|
||||
)
|
||||
|
||||
capped_request: Final = _REQUEST_RATE_LIMIT_DATA.validate_python(data)
|
||||
ptu_estimated_tokens: Final = self._estimate_ptu_tokens_for_request(
|
||||
ceiling=stash.ptu_ceiling,
|
||||
data=request_data,
|
||||
data=capped_request,
|
||||
min_configured_tpm_limit=min_configured_tpm_limit,
|
||||
call_type=call_type,
|
||||
configured_output_tokens=configured_output_tokens,
|
||||
|
|
|
|||
|
|
@ -7561,6 +7561,28 @@ async def test_a_one_ptu_share_admits_four_uncapped_requests_a_minute_and_reject
|
|||
assert all(data["max_tokens"] * 4 <= 3000 // 4 for data in admitted)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_an_uncapped_prompt_longer_than_the_output_floor_reserves_the_cap_the_proxy_writes():
|
||||
"""Without a cap the output budget defaults to the prompt's own length, so a long prompt would be
|
||||
reserved twice, weighted 4:1, and refused on an empty window. The ceiling counts the cap the proxy
|
||||
writes into the request instead, the same output the deployment can produce."""
|
||||
cache = DualCache()
|
||||
resolve, _ = _ptu_ceiling_for("t", "test-model", tpm_limit=3000, ratio=4.0)
|
||||
handler = _PROXY_MaxParallelRequestsHandler(
|
||||
internal_usage_cache=InternalUsageCache(cache), ptu_team_ceiling_resolver=resolve
|
||||
)
|
||||
key = UserAPIKeyAuth(api_key=hash_token("sk-ptu"), team_id="t")
|
||||
data = {"model": "test-model", "messages": [{"role": "user", "content": "word " * 1200}]}
|
||||
|
||||
await handler.async_pre_call_hook(user_api_key_dict=key, cache=cache, data=data, call_type="acompletion")
|
||||
|
||||
stash = get_request_stash()
|
||||
assert stash is not None
|
||||
assert data["max_tokens"] * 4 <= 3000 // 4
|
||||
assert stash.ptu_reserved_tokens == stash.reserved_tokens + 3 * data["max_tokens"]
|
||||
assert stash.ptu_reserved_tokens <= 3000
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_the_ptu_counter_holds_the_normalized_reservation_beside_the_raw_one():
|
||||
cache = DualCache()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue