mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
* test(e2e): datadog log delivery for successful chat, messages, and responses Covers logging.datadog.success.exports_metric on all three routes: one successful non-streaming call must reach the DataDog logs intake as exactly one log event whose StandardLoggingPayload message carries the model group, real token counts, and a response cost equal to the x-litellm-response-cost header of the same response. Delivery is judged at the intake: the compose stack gains a dd-sink service recording every batch the datadog callback ships via the DD_BASE_URL testing override, and a typed reader replays it. Writing these caught a live product bug: /v1/messages double-logs every success (two byte-identical events per call), filed as LIT-4447; the messages test tolerates byte-identical duplicates of the one event until it lands, while a second differing event still fails * test(e2e): address review findings on the datadog delivery suite Consolidates the fresh-key first_ok helper into logging_client now that the otel PR it mirrored has merged (both test files use the shared copy), moves intake batch parsing into a helper so no path can leave the batch unbound, and gives the sink's /health endpoint a truthful text/plain content type * test(e2e): tolerate same-logical-event duplicates by call id, not byte identity A clean LIT-4447 repro showed the duplicated payload is built twice and can mint a fresh synthetic completion id per emission, arriving as two separate intake POSTs with the same litellm_call_id and identical substantive fields. Byte-identity was therefore a flaky criterion; duplicates now qualify only when they share the call id, call type, model group, tokens, and cost, and a second differing event still fails * test(e2e): assert the scenario strictly; the messages test is the LIT-4447 regression pin Per review direction the tests now assert exactly what the scenario promises: exactly one DataDog log event per successful call, on every route. The /v1/messages test therefore fails on current code against the known double-log (LIT-4447) and is its regression pin; it goes green when the fix lands. The duplicate-tolerance machinery is removed * Simplify docstrings for DataDog log tests Removed redundant phrasing about cost cross-checking in docstrings. * Update test_datadog_log_e2e.py
162 lines
7.4 KiB
Python
162 lines
7.4 KiB
Python
"""Live e2e: DataDog log delivery for successful non-streaming calls.
|
|
|
|
Covers logging.datadog.success.exports_metric: one successful call on each
|
|
route must reach the DataDog logs intake as EXACTLY ONE log event whose
|
|
message (the StandardLoggingPayload) carries the model, the token counts, and
|
|
the response cost. Delivery is judged on what the intake actually received:
|
|
the compose stack's dd-sink service records every batch the datadog callback
|
|
ships (DD_BASE_URL override) and the tests read it back, so a dropped event, a
|
|
duplicated event, or a payload missing the cost all fail here.
|
|
|
|
Both halves of the contract are asserted: the recorded state (the proxy
|
|
reports the DataDogLogger callback active via /health/readiness/details) and
|
|
the enforced behavior (the event at the intake, with the cost cross-checked
|
|
exactly against the x-litellm-response-cost header of the very response the
|
|
caller received).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
from pydantic import BaseModel, ConfigDict
|
|
|
|
from datadog_sink import DdLogEvent, DdSinkReader
|
|
from e2e_config import CHEAP_ANTHROPIC_MODEL, CHEAP_OPENAI_MODEL, unique_marker
|
|
from e2e_http import NoBody, StreamingResponse
|
|
from lifecycle import ResourceManager
|
|
from logging_client import LoggingClient, first_ok
|
|
|
|
pytestmark = pytest.mark.e2e
|
|
|
|
#: The active DataDog callback's name in /health/readiness/details success_callbacks.
|
|
DD_LOGGER_NAME = "DataDogLogger"
|
|
|
|
|
|
class _DdMessagePayload(BaseModel):
|
|
"""The fields of the StandardLoggingPayload the scenario pins."""
|
|
|
|
model_config = ConfigDict(extra="ignore")
|
|
|
|
model_group: str
|
|
total_tokens: int
|
|
response_cost: float
|
|
status: str
|
|
call_type: str
|
|
|
|
|
|
def _assert_datadog_configured(client: LoggingClient) -> None:
|
|
"""Recorded state: the proxy reports the DataDog callback among its active
|
|
callbacks, so a missing destination config fails here, before any
|
|
delivery-based assertion can time out confusingly."""
|
|
result = client.gateway.probe("/health/readiness/details", params=NoBody())
|
|
assert result.status_code == 200, (
|
|
f"/health/readiness/details must answer 200, got {result.status_code}: {result.body[:300]}"
|
|
)
|
|
assert DD_LOGGER_NAME in result.body, (
|
|
f"the proxy must report the {DD_LOGGER_NAME} callback active "
|
|
f"(callbacks + DD_* env in the compose config); got: {result.body[:400]}"
|
|
)
|
|
|
|
|
|
def _assert_exactly_one_event(
|
|
events: list[DdLogEvent], *, model_group: str, call_type: str, outcome: StreamingResponse
|
|
) -> None:
|
|
"""The enforced behavior: the intake holds exactly one event for the call,
|
|
sourced from litellm, whose payload names the model group and call type,
|
|
counts real tokens, and carries the same cost the response header reported."""
|
|
assert events, "no DataDog log event for this call reached the intake within the deadline"
|
|
assert len(events) == 1, (
|
|
f"expected exactly ONE DataDog log event for the call, got {len(events)} - "
|
|
"more than one event for one call is the duplicate-delivery bug (see LIT-4447 "
|
|
"for the currently known /v1/messages instance)"
|
|
)
|
|
event = events[0]
|
|
assert event.ddsource == "litellm", f"event ddsource must be litellm, got {event.ddsource!r}"
|
|
assert event.status == "info", f"success events ship at status info, got {event.status!r}"
|
|
|
|
payload = _DdMessagePayload.model_validate_json(event.message)
|
|
assert payload.status == "success", f"payload status must be success, got {payload.status!r}"
|
|
assert payload.model_group == model_group, (
|
|
f"payload model_group must be {model_group!r}, got {payload.model_group!r}"
|
|
)
|
|
assert payload.call_type == call_type, (
|
|
f"payload call_type must be {call_type!r}, got {payload.call_type!r}"
|
|
)
|
|
assert payload.total_tokens > 0, f"payload must count real tokens, got {payload.total_tokens}"
|
|
assert outcome.response_cost is not None and outcome.response_cost > 0, (
|
|
f"the response must report x-litellm-response-cost, got {outcome.response_cost!r}"
|
|
)
|
|
assert abs(payload.response_cost - outcome.response_cost) < 1e-12, (
|
|
f"payload response_cost {payload.response_cost} must equal the response header "
|
|
f"cost {outcome.response_cost}"
|
|
)
|
|
|
|
|
|
class TestDataDogLogDelivery:
|
|
@pytest.mark.covers("logging.datadog.success.exports_metric", exercised_on=["chat_completions"])
|
|
def test_chat_completions_emits_one_log_event(
|
|
self, client: LoggingClient, dd_sink: DdSinkReader, resources: ResourceManager
|
|
) -> None:
|
|
"""One successful non-streaming /chat/completions call must reach the
|
|
DataDog logs intake as exactly one log event whose payload carries the
|
|
model, the token counts, and the response cost."""
|
|
_assert_datadog_configured(client)
|
|
|
|
key = client.key_with_alias(f"dd-chat-{unique_marker()}", models=[CHEAP_ANTHROPIC_MODEL])
|
|
resources.defer(lambda: client.delete_key(key))
|
|
|
|
marker = unique_marker()
|
|
outcome = first_ok(
|
|
client,
|
|
lambda: client.chat_raw(key, CHEAP_ANTHROPIC_MODEL, f"reply with one word {marker}", max_tokens=16),
|
|
)
|
|
events = dd_sink.poll_events_for_marker(marker)
|
|
_assert_exactly_one_event(
|
|
events, model_group=CHEAP_ANTHROPIC_MODEL, call_type="acompletion", outcome=outcome
|
|
)
|
|
|
|
@pytest.mark.covers("logging.datadog.success.exports_metric", exercised_on=["messages"])
|
|
def test_messages_emits_one_log_event(
|
|
self, client: LoggingClient, dd_sink: DdSinkReader, resources: ResourceManager
|
|
) -> None:
|
|
"""One successful non-streaming /v1/messages call must reach the
|
|
DataDog logs intake as exactly one log event whose payload carries the
|
|
model, the token counts, and the response cost.
|
|
|
|
This currently fails on the known /v1/messages double-log (LIT-4447); it goes green when the fix lands."""
|
|
_assert_datadog_configured(client)
|
|
|
|
key = client.key_with_alias(f"dd-messages-{unique_marker()}", models=[CHEAP_ANTHROPIC_MODEL])
|
|
resources.defer(lambda: client.delete_key(key))
|
|
|
|
marker = unique_marker()
|
|
outcome = first_ok(
|
|
client,
|
|
lambda: client.messages_raw(key, CHEAP_ANTHROPIC_MODEL, f"reply with one word {marker}", max_tokens=16),
|
|
)
|
|
events = dd_sink.poll_events_for_marker(marker)
|
|
_assert_exactly_one_event(
|
|
events, model_group=CHEAP_ANTHROPIC_MODEL, call_type="anthropic_messages", outcome=outcome
|
|
)
|
|
|
|
@pytest.mark.covers("logging.datadog.success.exports_metric", exercised_on=["responses"])
|
|
def test_responses_emits_one_log_event(
|
|
self, client: LoggingClient, dd_sink: DdSinkReader, resources: ResourceManager
|
|
) -> None:
|
|
"""One successful non-streaming /v1/responses call must reach the
|
|
DataDog logs intake as exactly one log event whose payload carries the
|
|
model, the token counts, and the response cost."""
|
|
_assert_datadog_configured(client)
|
|
|
|
key = client.key_with_alias(f"dd-responses-{unique_marker()}", models=[CHEAP_OPENAI_MODEL])
|
|
resources.defer(lambda: client.delete_key(key))
|
|
|
|
marker = unique_marker()
|
|
outcome = first_ok(
|
|
client,
|
|
lambda: client.responses_raw(key, CHEAP_OPENAI_MODEL, f"reply with one word {marker}"),
|
|
)
|
|
events = dd_sink.poll_events_for_marker(marker)
|
|
_assert_exactly_one_event(
|
|
events, model_group=CHEAP_OPENAI_MODEL, call_type="aresponses", outcome=outcome
|
|
)
|