mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
is_prompt_caching_valid_prompt ran the full Python token_counter over every message to compare against the deployment's prompt cache minimum, 500 to 1000 ms at 440k to 740k tokens on every request that reaches the prompt_caching pre-call check, Rust on or off. messages_reach_token_count does the same arithmetic as token_counter(...) >= threshold and stops at the first message that reaches the threshold. Groups with one healthy deployment skip the prefix hash and pin lookup, which cannot change the result for them Four fixed name span events make the pre-LLM phases measurable with OTel v2: litellm.request.body_received (with body_bytes) once per body read before parsing, on the JSON, binary and form branches, body_parsed, pre_call_completed, and deployment_selected emitted once per pick inside Router.async_get_available_deployment and get_available_deployment with attempt, reason and model group, so every router surface, retry and fallback is covered. Measured locally on /v1/chat/completions, /v1/messages and /v1/responses at 440k tokens with Rust on and off against a fake upstream Co-authored-by: yassin <yassin@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
71 lines
2.3 KiB
Python
71 lines
2.3 KiB
Python
"""Regression tests for the SDK-free OTel runtime shim.
|
|
|
|
The proxy auth hot path calls ``phase_span`` and ``seed_request_identity`` on
|
|
every request. These wrappers resolve the SDK-backed implementations with a
|
|
lazy import. CPython never caches a failed import, so before memoization an
|
|
absent OTel SDK made every request re-scan ``sys.path`` and contend on the
|
|
import lock. These tests pin the import to a single resolution.
|
|
"""
|
|
|
|
import builtins
|
|
|
|
import litellm.integrations.otel.runtime as runtime
|
|
|
|
|
|
def test_logger_not_reimported_after_first_resolution(monkeypatch):
|
|
runtime._otel_runtime.cache_clear()
|
|
|
|
counts = {"n": 0}
|
|
real_import = builtins.__import__
|
|
|
|
def counting_import(name, globals=None, locals=None, fromlist=(), level=0):
|
|
if name == "litellm.integrations.otel" and fromlist and "logger" in fromlist:
|
|
counts["n"] += 1
|
|
return real_import(name, globals, locals, fromlist, level)
|
|
|
|
monkeypatch.setattr(builtins, "__import__", counting_import)
|
|
|
|
with runtime.phase_span("auth /v1/chat/completions"):
|
|
pass
|
|
after_first = counts["n"]
|
|
|
|
for _ in range(49):
|
|
with runtime.phase_span("auth /v1/chat/completions"):
|
|
pass
|
|
|
|
assert counts["n"] == after_first, (
|
|
f"otel.logger re-imported {counts['n'] - after_first} times after the first "
|
|
"resolution; it must be memoized so it does not re-scan sys.path per request"
|
|
)
|
|
|
|
runtime._otel_runtime.cache_clear()
|
|
|
|
|
|
def test_resolution_is_memoized():
|
|
runtime._otel_runtime.cache_clear()
|
|
|
|
for _ in range(25):
|
|
with runtime.phase_span("p"):
|
|
pass
|
|
|
|
info = runtime._otel_runtime.cache_info()
|
|
assert info.misses == 1
|
|
assert info.hits >= 24
|
|
|
|
runtime._otel_runtime.cache_clear()
|
|
|
|
|
|
def test_wrappers_no_op_when_runtime_absent(monkeypatch):
|
|
monkeypatch.setattr(runtime, "_otel_runtime", lambda: None)
|
|
|
|
with runtime.phase_span("auth") as span:
|
|
assert span is None
|
|
|
|
assert runtime.seed_request_identity({"token": "sk-x"}, model="gpt-4o") is None
|
|
|
|
|
|
def test_phase_event_no_ops_when_runtime_absent(monkeypatch):
|
|
monkeypatch.setattr(runtime, "_otel_runtime", lambda: None)
|
|
|
|
assert runtime.phase_event("litellm.request.body_parsed") is None
|
|
assert runtime.phase_event("litellm.request.body_received", {"litellm.request.body_bytes": 3}) is None
|