mirror of
https://github.com/BerriAI/litellm.git
synced 2026-08-28 05:25:59 +00:00
* test(e2e): send no-cache on every cacheable request body, opt in only where a hit is the assertion
The e2e proxy runs with the response cache on, so any test that re-sends an
identical chat, messages, responses, completions, embeddings or rerank body
reads back a redis copy of an earlier call instead of reaching the provider.
Five tests in the last week failed that way. Default cache: {"no-cache": true}
on those request models and pass cache=None only in the two tests whose
assertion is the cache hit itself.
* test(e2e): give image edits and OCR a 180s client timeout
Both routes wait on providers that can legitimately take longer than the
60s transport-wide request timeout (gpt-image edits, Azure Document
Intelligence), and a client-side read timeout there fails a green request.
post/upload now accept a per-call timeout like get already does; only those
two call sites use it.
* test(e2e): rerun once on network errors and upstream 5xx only
Assertion failures still fail on the first attempt; only an outcome whose
error string carries the e2e_http network kind or a 5xx status gets one
more try. Test Engine records every attempt, so the flake rate stays
visible while a single provider blip no longer reds the rc run.
* test(e2e): let the reseed burst survive one upstream failure and print why
The burst is the precondition, not the property: one 5xx among six
concurrent calls still leaves five workers racing the cold counter, which
is what the reseed assertion measures. Two or more failures still abort,
and the failing bodies are now in the message instead of only the status
codes.
* test(e2e): keep polling Jaeger through a transient query failure
poll_traces_for_call already waits up to POLL_TIMEOUT for spans to land,
but a single refused connection to the query API failed the test on the
spot. Jaeger restarted twice during today's gate runs (19:05 and 19:41
UTC, each under a minute) and took ten and three otel tests with it while
the same tests passed on the rc build minutes later. A network failure
now counts as not-yet inside the same deadline; if Jaeger is still
unreachable when the deadline passes the test fails with that error, and
any non-network failure still fails immediately.
155 lines
5.5 KiB
Python
155 lines
5.5 KiB
Python
"""Jaeger read-back for the OTEL trace-completeness tests: typed models over the
|
|
Jaeger query API (the destination's own API - completeness is judged on what the
|
|
backend actually holds, never on "export succeeded" proxy-side).
|
|
|
|
Traces are fetched server-side by the ``litellm.call_id`` tag the gen-AI span
|
|
carries (the request's x-litellm-call-id response header), so read-back is
|
|
immune to the query page filling up with unrelated traffic (background jobs,
|
|
other suites sharing the stack). Jaeger returns every span of a matching trace,
|
|
so the completeness assertions see the whole tree. A failed query is a hard
|
|
failure, never an empty result - an unreachable destination must not read as
|
|
"the trace never arrived".
|
|
|
|
External reads go through ``e2e_http`` (the only module allowed to call
|
|
``requests.*``).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import time
|
|
from dataclasses import dataclass
|
|
|
|
import pytest
|
|
from pydantic import BaseModel, ConfigDict, Field
|
|
|
|
from e2e_config import OTEL_QUERY_URL, POLL_INTERVAL, POLL_TIMEOUT
|
|
from e2e_http import URL, NetworkError, NoBody, Result, Success, get
|
|
|
|
#: OTEL resource service.name the proxy exports under (OTEL_SERVICE_NAME default).
|
|
JAEGER_SERVICE = "litellm"
|
|
#: Span tag carrying the request's x-litellm-call-id (stamped on the gen-AI span).
|
|
CALL_ID_TAG = "litellm.call_id"
|
|
|
|
|
|
class JaegerTag(BaseModel):
|
|
model_config = ConfigDict(extra="ignore")
|
|
|
|
key: str
|
|
value: str | int | float | bool | None = None
|
|
|
|
|
|
class JaegerReference(BaseModel):
|
|
model_config = ConfigDict(extra="ignore", populate_by_name=True)
|
|
|
|
ref_type: str = Field(alias="refType")
|
|
trace_id: str = Field(alias="traceID")
|
|
span_id: str = Field(alias="spanID")
|
|
|
|
|
|
class JaegerSpan(BaseModel):
|
|
model_config = ConfigDict(extra="ignore", populate_by_name=True)
|
|
|
|
span_id: str = Field(alias="spanID")
|
|
operation_name: str = Field(alias="operationName")
|
|
start_time: int = Field(default=0, alias="startTime")
|
|
#: Span duration in microseconds, as reported by the Jaeger query API.
|
|
duration: int = 0
|
|
references: list[JaegerReference] = []
|
|
tags: list[JaegerTag] = []
|
|
|
|
@property
|
|
def kind(self) -> str:
|
|
for tag in self.tags:
|
|
if tag.key == "span.kind":
|
|
return str(tag.value)
|
|
return ""
|
|
|
|
|
|
class JaegerTrace(BaseModel):
|
|
model_config = ConfigDict(extra="ignore", populate_by_name=True)
|
|
|
|
trace_id: str = Field(alias="traceID")
|
|
spans: list[JaegerSpan] = []
|
|
|
|
def span_names(self) -> list[str]:
|
|
return sorted(span.operation_name for span in self.spans)
|
|
|
|
|
|
class JaegerTracesPage(BaseModel):
|
|
model_config = ConfigDict(extra="ignore")
|
|
|
|
data: list[JaegerTrace] = []
|
|
|
|
|
|
class _TracesQuery(BaseModel):
|
|
service: str
|
|
tags: str
|
|
limit: int = 20
|
|
lookback: str = "1h"
|
|
|
|
|
|
def _settled(trace: JaegerTrace, names: set[str], prefixes: set[str]) -> bool:
|
|
present = set(trace.span_names())
|
|
return names.issubset(present) and all(
|
|
any(name.startswith(prefix) for name in present) for prefix in prefixes
|
|
)
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class OtelReader:
|
|
query_url: str
|
|
|
|
def _query_traces(self, call_id: str) -> Result[JaegerTracesPage]:
|
|
return get(
|
|
URL(f"{self.query_url}/api/traces"),
|
|
headers=NoBody(),
|
|
params=_TracesQuery(service=JAEGER_SERVICE, tags=json.dumps({CALL_ID_TAG: call_id})),
|
|
response_type=JaegerTracesPage,
|
|
timeout=30.0,
|
|
)
|
|
|
|
def traces_for_call(self, call_id: str) -> list[JaegerTrace]:
|
|
"""Every trace holding a span tagged with this call id. Jaeger matches
|
|
spans server-side and returns their full traces; more than one hit for
|
|
one call IS the split-trace bug, so this never collapses to one."""
|
|
match self._query_traces(call_id):
|
|
case Success(data=page):
|
|
return page.data
|
|
case failure:
|
|
pytest.fail(f"Jaeger query API at {self.query_url} failed: {failure}")
|
|
|
|
def poll_traces_for_call(
|
|
self, *, call_id: str, settled_names: set[str], settled_prefixes: set[str]
|
|
) -> list[JaegerTrace]:
|
|
"""Poll until exactly one trace holds the call and it carries every span
|
|
name in ``settled_names`` plus at least one name per prefix in
|
|
``settled_prefixes`` (spans flush in batches, the cost write lands after
|
|
the response), then return the hits. At the deadline the last hits are
|
|
returned as-is so the caller's assertions report the real final state -
|
|
on a split trace this never settles and the orphan comes back."""
|
|
deadline = time.monotonic() + POLL_TIMEOUT
|
|
hits: list[JaegerTrace] = []
|
|
unreachable: NetworkError | None = None
|
|
while time.monotonic() < deadline:
|
|
match self._query_traces(call_id):
|
|
case Success(data=page):
|
|
unreachable = None
|
|
hits = page.data
|
|
if len(hits) == 1 and _settled(hits[0], settled_names, settled_prefixes):
|
|
return hits
|
|
case NetworkError() as failure:
|
|
unreachable = failure
|
|
case failure:
|
|
pytest.fail(f"Jaeger query API at {self.query_url} failed: {failure}")
|
|
time.sleep(POLL_INTERVAL)
|
|
if unreachable is not None:
|
|
pytest.fail(
|
|
f"Jaeger query API at {self.query_url} stayed unreachable until the "
|
|
f"{POLL_TIMEOUT}s poll deadline: {unreachable}"
|
|
)
|
|
return hits
|
|
|
|
|
|
def build_otel_reader() -> OtelReader:
|
|
return OtelReader(query_url=OTEL_QUERY_URL)
|