From e70b9cb01ff2b932b21056652b2b28185867c805 Mon Sep 17 00:00:00 2001 From: sinksilk <785976238@qq.com> Date: Fri, 2 Oct 2026 01:18:40 +0800 Subject: [PATCH] test(proxy): cover summary quota fallbacks and bump pypdf to 6.19.0 Signed-off-by: sinksilk <785976238@qq.com> --- tests/unit/proxy/hooks/test_tpm_concurrent.py | 153 ++++++++++++++++++ 1 file changed, 153 insertions(+) diff --git a/tests/unit/proxy/hooks/test_tpm_concurrent.py b/tests/unit/proxy/hooks/test_tpm_concurrent.py index 354e86bee81..f4da0d00223 100644 --- a/tests/unit/proxy/hooks/test_tpm_concurrent.py +++ b/tests/unit/proxy/hooks/test_tpm_concurrent.py @@ -30,6 +30,7 @@ from litellm.proxy.hooks.parallel_request_limiter_v3 import ( _PROXY_MaxParallelRequestsHandler_v3 as RateLimitHandler, ) from litellm.proxy.hooks.parallel_request_limiter_v3 import ( + _as_str_object_dict, _call_id_from_callback_kwargs, _request_stash, get_or_create_request_stash, @@ -3876,5 +3877,157 @@ async def test_summary_subrequest_honors_project_itpm_otpm(rate_limiter): ] = previous_hook +@pytest.mark.asyncio +async def test_unreserved_project_io_covers_empty_and_fallback_usage(rate_limiter): + """Cover the unreserved ITPM/OTPM branches the happy-path summary test skips.""" + from typing import Final + + handler, _cache = rate_limiter + assert _as_str_object_dict("nope") is None + echoed: Final = _as_str_object_dict({"a": 1}) + assert echoed is not None + assert echoed["a"] == 1 + + assert handler._build_unreserved_project_io_token_ops("nope", {}) == () + assert handler._build_unreserved_project_io_token_ops({}, {}) == () + assert ( + handler._build_unreserved_project_io_token_ops( + {"standard_logging_object": "x"}, + {}, + ) + == () + ) + assert ( + handler._build_unreserved_project_io_token_ops( + {"standard_logging_object": {"metadata": "x"}}, + {}, + ) + == () + ) + assert ( + handler._build_unreserved_project_io_token_ops( + { + "standard_logging_object": { + "metadata": {"user_api_key_project_id": "proj"}, + "model_group": 1, + } + }, + {}, + ) + == () + ) + + combined_ops: Final = handler._build_unreserved_project_io_token_ops( + { + "standard_logging_object": { + "metadata": {"user_api_key_project_id": "proj"}, + "model_group": "gpt-4o-mini", + }, + "combined_usage_object": {"prompt_tokens": 5, "completion_tokens": 0}, + }, + {}, + ) + assert len(combined_ops) == 1 + assert combined_ops[0]["increment_value"] == 5 + + aggregate_ops: Final = handler._build_unreserved_project_io_token_ops( + { + "standard_logging_object": { + "metadata": {"user_api_key_project_id": "proj"}, + "model_group": "gpt-4o-mini", + }, + }, + {"total_tokens": 9}, + ) + assert len(aggregate_ops) == 2 + assert {op["increment_value"] for op in aggregate_ops} == {9} + + await handler._seed_rate_limit_window_if_absent(window_key="already-open", ttl=None) + await handler._seed_rate_limit_window_if_absent(window_key="already-open", ttl=None) + await handler._apply_one_reservation_aware_token_increment( + operation={ + "key": "plain-counter", + "increment_value": 3, + "ttl": 60, + }, + parent_otel_span=None, + ) + assert int(await handler.internal_usage_cache.async_get_cache(key="plain-counter", litellm_parent_otel_span=None) or 0) == 3 + + +@pytest.mark.asyncio +async def test_summary_token_estimate_uses_counter_or_falls_back(monkeypatch): + from typing import Final + + from litellm.llms.anthropic.pass_through.context_management.editors.compact import ( + _check_summary_model_rate_limit, + _estimate_summary_input_tokens, + ) + from litellm.proxy import proxy_server + + messages: Final = ({"role": "user", "content": "hi"},) + + def fail_counter(**_kwargs: object) -> int: + raise RuntimeError("counter down") + + monkeypatch.setattr("litellm.token_counter", fail_counter) + assert ( + await _estimate_summary_input_tokens( + summary_model="gpt-4o-mini", + summary_messages=messages, + fallback_tokens=42, + ) + == 42 + ) + + def fixed_counter(**_kwargs: object) -> int: + return 17 + + monkeypatch.setattr("litellm.token_counter", fixed_counter) + assert ( + await _estimate_summary_input_tokens( + summary_model="gpt-4o-mini", + summary_messages=messages, + fallback_tokens=42, + ) + == 17 + ) + + handler: Final = RateLimitHandler(internal_usage_cache=InternalUsageCache(DualCache())) + + async def junk_statuses(self, descriptors, parent_otel_span=None, read_only=False, **_kwargs): + return { + "overall_code": "OK", + "statuses": [ + "not-a-status", + {"descriptor_key": "model_per_project_itpm", "limit_remaining": "lots"}, + {"descriptor_key": "model_per_project_itpm", "limit_remaining": 1000}, + ], + } + + monkeypatch.setattr(RateLimitHandler, "should_rate_limit", junk_statuses) + previous_hook: Final = proxy_server.proxy_logging_obj.proxy_hook_mapping.get("parallel_request_limiter") + proxy_server.proxy_logging_obj.proxy_hook_mapping["parallel_request_limiter"] = handler + try: + assert ( + await _check_summary_model_rate_limit( + user_api_key_auth=UserAPIKeyAuth( + api_key="sk-proj-key", + project_id="proj-summary-io", + project_metadata={"model_itpm_limit": {"gpt-4o-mini": 2000}}, + ), + summary_model="gpt-4o-mini", + estimated_input_tokens=10, + estimated_output_tokens=1, + ) + is True + ) + finally: + if previous_hook is None: + proxy_server.proxy_logging_obj.proxy_hook_mapping.pop("parallel_request_limiter", None) + else: + proxy_server.proxy_logging_obj.proxy_hook_mapping["parallel_request_limiter"] = previous_hook + + if __name__ == "__main__": pytest.main([__file__, "-v", "-s"])