litellm/tests/unit/integrations/otel/test_runtime.py
devin-ai-integration[bot] 4c6c84afb7
perf(proxy): stop prompt-cache eligibility from tokenizing the whole conversation (#44221)
is_prompt_caching_valid_prompt ran the full Python token_counter over every message to compare against the deployment's prompt cache minimum, 500 to 1000 ms at 440k to 740k tokens on every request that reaches the prompt_caching pre-call check, Rust on or off. messages_reach_token_count does the same arithmetic as token_counter(...) >= threshold and stops at the first message that reaches the threshold. Groups with one healthy deployment skip the prefix hash and pin lookup, which cannot change the result for them

Four fixed name span events make the pre-LLM phases measurable with OTel v2: litellm.request.body_received (with body_bytes) once per body read before parsing, on the JSON, binary and form branches, body_parsed, pre_call_completed, and deployment_selected emitted once per pick inside Router.async_get_available_deployment and get_available_deployment with attempt, reason and model group, so every router surface, retry and fallback is covered. Measured locally on /v1/chat/completions, /v1/messages and /v1/responses at 440k tokens with Rust on and off against a fake upstream

Co-authored-by: yassin <yassin@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
2026-10-03 09:20:28 -07:00

71 lines
2.3 KiB
Python

"""Regression tests for the SDK-free OTel runtime shim.
The proxy auth hot path calls ``phase_span`` and ``seed_request_identity`` on
every request. These wrappers resolve the SDK-backed implementations with a
lazy import. CPython never caches a failed import, so before memoization an
absent OTel SDK made every request re-scan ``sys.path`` and contend on the
import lock. These tests pin the import to a single resolution.
"""
import builtins
import litellm.integrations.otel.runtime as runtime
def test_logger_not_reimported_after_first_resolution(monkeypatch):
runtime._otel_runtime.cache_clear()
counts = {"n": 0}
real_import = builtins.__import__
def counting_import(name, globals=None, locals=None, fromlist=(), level=0):
if name == "litellm.integrations.otel" and fromlist and "logger" in fromlist:
counts["n"] += 1
return real_import(name, globals, locals, fromlist, level)
monkeypatch.setattr(builtins, "__import__", counting_import)
with runtime.phase_span("auth /v1/chat/completions"):
pass
after_first = counts["n"]
for _ in range(49):
with runtime.phase_span("auth /v1/chat/completions"):
pass
assert counts["n"] == after_first, (
f"otel.logger re-imported {counts['n'] - after_first} times after the first "
"resolution; it must be memoized so it does not re-scan sys.path per request"
)
runtime._otel_runtime.cache_clear()
def test_resolution_is_memoized():
runtime._otel_runtime.cache_clear()
for _ in range(25):
with runtime.phase_span("p"):
pass
info = runtime._otel_runtime.cache_info()
assert info.misses == 1
assert info.hits >= 24
runtime._otel_runtime.cache_clear()
def test_wrappers_no_op_when_runtime_absent(monkeypatch):
monkeypatch.setattr(runtime, "_otel_runtime", lambda: None)
with runtime.phase_span("auth") as span:
assert span is None
assert runtime.seed_request_identity({"token": "sk-x"}, model="gpt-4o") is None
def test_phase_event_no_ops_when_runtime_absent(monkeypatch):
monkeypatch.setattr(runtime, "_otel_runtime", lambda: None)
assert runtime.phase_event("litellm.request.body_parsed") is None
assert runtime.phase_event("litellm.request.body_received", {"litellm.request.body_bytes": 3}) is None