mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-30 01:52:18 +00:00
The image validates the published PyPI artifact, whose langfuse callback still reads langfuse.version, so the 4.15.2 pin broke that callback. The pins move together with the next LITELLM_VERSION bump. Also rewords the trace_version precedence test docstring: v2 carried two version fields, v4 has one per span Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
2471 lines
100 KiB
Python
2471 lines
100 KiB
Python
import datetime
|
|
import json
|
|
import logging
|
|
import threading
|
|
import time
|
|
import types
|
|
import unittest
|
|
from typing import Final, Optional
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
import pytest
|
|
|
|
import litellm
|
|
from litellm.integrations.langfuse import langfuse as langfuse_module
|
|
from litellm.integrations.langfuse.langfuse import LangFuseLogger
|
|
from litellm.integrations.langfuse.langfuse_sdk import resolve_trace_id
|
|
|
|
|
|
# Import LangfuseUsageDetails directly from the module where it's defined
|
|
from litellm.types.integrations.langfuse import *
|
|
|
|
|
|
class TestLangfuseUsageDetails(unittest.TestCase):
|
|
def setUp(self):
|
|
# Save global Langfuse client counter to restore after test
|
|
self._original_langfuse_clients_count = litellm.initialized_langfuse_clients
|
|
|
|
# Set up environment variables for testing
|
|
self.env_patcher = patch.dict(
|
|
"os.environ",
|
|
{
|
|
"LANGFUSE_SECRET_KEY": "test-secret-key",
|
|
"LANGFUSE_PUBLIC_KEY": "test-public-key",
|
|
"LANGFUSE_HOST": "https://test.langfuse.com",
|
|
},
|
|
)
|
|
self.env_patcher.start()
|
|
|
|
from opentelemetry.sdk.trace import TracerProvider
|
|
from opentelemetry.sdk.trace.export import SimpleSpanProcessor
|
|
from opentelemetry.sdk.trace.export.in_memory_span_exporter import (
|
|
InMemorySpanExporter,
|
|
)
|
|
|
|
self.span_exporter = InMemorySpanExporter()
|
|
self.real_provider = TracerProvider()
|
|
self.real_provider.add_span_processor(SimpleSpanProcessor(self.span_exporter))
|
|
|
|
# the host above is unreachable, so the REST client is cheap to build
|
|
# and each test swaps in the export channel it wants
|
|
self.logger = LangFuseLogger()
|
|
|
|
# Add the log_event_on_langfuse method to the instance
|
|
def log_event_on_langfuse(
|
|
self,
|
|
kwargs,
|
|
response_obj,
|
|
start_time=None,
|
|
end_time=None,
|
|
user_id=None,
|
|
level="DEFAULT",
|
|
status_message=None,
|
|
):
|
|
# This implementation calls _log_langfuse_v2 directly
|
|
return self._log_langfuse_v2(
|
|
user_id=user_id,
|
|
metadata=kwargs.get("litellm_params", {}).get("metadata", {}),
|
|
litellm_params=kwargs.get("litellm_params", {}),
|
|
output=None,
|
|
start_time=start_time,
|
|
end_time=end_time,
|
|
kwargs=kwargs,
|
|
optional_params=kwargs.get("optional_params", {}),
|
|
input=None,
|
|
response_obj=response_obj,
|
|
level=level,
|
|
litellm_call_id=kwargs.get("litellm_call_id", None),
|
|
)
|
|
|
|
# Bind the method to the instance
|
|
self.logger.log_event_on_langfuse = types.MethodType(log_event_on_langfuse, self.logger)
|
|
|
|
def tearDown(self):
|
|
# Clean up logger instance to prevent state leakage
|
|
if hasattr(self, "logger"):
|
|
del self.logger
|
|
|
|
# Restore global Langfuse client counter to prevent cross-test pollution
|
|
litellm.initialized_langfuse_clients = self._original_langfuse_clients_count
|
|
|
|
self.env_patcher.stop()
|
|
|
|
def use_real_langfuse_client(self):
|
|
"""Point the logger at an export channel whose spans land in memory."""
|
|
from opentelemetry.sdk.trace.export.in_memory_span_exporter import (
|
|
InMemorySpanExporter,
|
|
)
|
|
|
|
from litellm.integrations.langfuse.langfuse_sdk import build_langfuse_tracing
|
|
|
|
self.span_exporter = InMemorySpanExporter()
|
|
self.logger.tracing = build_langfuse_tracing(
|
|
exporter=self.span_exporter,
|
|
environment=None,
|
|
release=None,
|
|
sample_rate=1.0,
|
|
flush_interval_millis=10,
|
|
)
|
|
self.real_provider = self.logger.tracing.provider
|
|
return self.logger.tracing
|
|
|
|
def exported_generation(self):
|
|
self.logger.tracing.flush()
|
|
spans = [s for s in self.span_exporter.get_finished_spans()]
|
|
assert spans, "no spans were exported"
|
|
return spans[-1]
|
|
|
|
@staticmethod
|
|
def span_trace_id(span):
|
|
return format(span.context.trace_id, "032x")
|
|
|
|
def test_langfuse_usage_details_type(self):
|
|
"""Test that LangfuseUsageDetails TypedDict is properly defined with the correct fields"""
|
|
# Create an instance of LangfuseUsageDetails
|
|
usage_details: LangfuseUsageDetails = {
|
|
"input": 10,
|
|
"output": 20,
|
|
"total": 30,
|
|
"cache_creation_input_tokens": 5,
|
|
"cache_read_input_tokens": 3,
|
|
}
|
|
|
|
# Verify all fields are present
|
|
self.assertEqual(usage_details["input"], 10)
|
|
self.assertEqual(usage_details["output"], 20)
|
|
self.assertEqual(usage_details["total"], 30)
|
|
self.assertEqual(usage_details["cache_creation_input_tokens"], 5)
|
|
self.assertEqual(usage_details["cache_read_input_tokens"], 3)
|
|
|
|
# Test with all fields (all fields are required in TypedDict by default)
|
|
minimal_usage_details: LangfuseUsageDetails = {
|
|
"input": 10,
|
|
"output": 20,
|
|
"total": 30,
|
|
"cache_creation_input_tokens": 0,
|
|
"cache_read_input_tokens": 0,
|
|
}
|
|
|
|
self.assertEqual(minimal_usage_details["input"], 10)
|
|
self.assertEqual(minimal_usage_details["output"], 20)
|
|
self.assertEqual(minimal_usage_details["total"], 30)
|
|
|
|
def test_log_langfuse_v2_usage_details(self):
|
|
"""Test that usage_details in _log_langfuse_v2 is correctly typed and assigned"""
|
|
# Create a mock response object with usage information
|
|
response_obj = MagicMock()
|
|
response_obj.usage = MagicMock()
|
|
response_obj.usage.prompt_tokens = 15
|
|
response_obj.usage.completion_tokens = 25
|
|
|
|
# Add the cache token attributes using get method
|
|
def mock_get(key, default=None):
|
|
if key == "cache_creation_input_tokens":
|
|
return 7
|
|
elif key == "cache_read_input_tokens":
|
|
return 4
|
|
return default
|
|
|
|
response_obj.usage.get = mock_get
|
|
|
|
# Create kwargs for the log_event method
|
|
kwargs = {
|
|
"model": "gpt-4",
|
|
"messages": [{"role": "user", "content": "Hello"}],
|
|
"litellm_params": {"metadata": {}},
|
|
}
|
|
|
|
# Create start and end times
|
|
start_time = datetime.datetime.now()
|
|
end_time = start_time + datetime.timedelta(seconds=1)
|
|
|
|
# Call the log_event method
|
|
with patch.object(self.logger, "_log_langfuse_v2") as mock_log_langfuse_v2:
|
|
self.logger.log_event_on_langfuse(
|
|
kwargs=kwargs,
|
|
response_obj=response_obj,
|
|
start_time=start_time,
|
|
end_time=end_time,
|
|
)
|
|
|
|
# Check if _log_langfuse_v2 was called
|
|
mock_log_langfuse_v2.assert_called_once()
|
|
|
|
# Get the arguments passed to _log_langfuse_v2
|
|
call_args = mock_log_langfuse_v2.call_args[1]
|
|
|
|
# Verify response_obj was passed correctly
|
|
self.assertEqual(call_args["response_obj"], response_obj)
|
|
|
|
def test_langfuse_usage_details_optional_fields(self):
|
|
"""Test that LangfuseUsageDetails fields are properly defined as Optional"""
|
|
# Create an instance with None values for optional fields
|
|
usage_details: LangfuseUsageDetails = {
|
|
"input": 10,
|
|
"output": 20,
|
|
"total": 30,
|
|
"cache_creation_input_tokens": None,
|
|
"cache_read_input_tokens": None,
|
|
}
|
|
|
|
# Verify fields can be None
|
|
self.assertEqual(usage_details["input"], 10)
|
|
self.assertEqual(usage_details["output"], 20)
|
|
self.assertEqual(usage_details["total"], 30)
|
|
self.assertIsNone(usage_details["cache_creation_input_tokens"])
|
|
self.assertIsNone(usage_details["cache_read_input_tokens"])
|
|
|
|
def test_langfuse_usage_details_structure(self):
|
|
"""Test that LangfuseUsageDetails has the correct structure as defined in the commit"""
|
|
# This test directly verifies the structure of the TypedDict
|
|
# without relying on the LangFuseLogger class
|
|
|
|
# Create a dictionary that matches the LangfuseUsageDetails structure
|
|
usage_details = {
|
|
"input": 15,
|
|
"output": 25,
|
|
"total": 40,
|
|
"cache_creation_input_tokens": 7,
|
|
"cache_read_input_tokens": 4,
|
|
}
|
|
|
|
# Verify the structure matches what we expect
|
|
self.assertIn("input", usage_details)
|
|
self.assertIn("output", usage_details)
|
|
self.assertIn("total", usage_details)
|
|
self.assertIn("cache_creation_input_tokens", usage_details)
|
|
self.assertIn("cache_read_input_tokens", usage_details)
|
|
|
|
# Verify the values
|
|
self.assertEqual(usage_details["input"], 15)
|
|
self.assertEqual(usage_details["output"], 25)
|
|
self.assertEqual(usage_details["total"], 40)
|
|
self.assertEqual(usage_details["cache_creation_input_tokens"], 7)
|
|
self.assertEqual(usage_details["cache_read_input_tokens"], 4)
|
|
|
|
def test_log_langfuse_v2_handles_null_usage_values(self):
|
|
"""
|
|
Test that _log_langfuse_v2 correctly handles None values in the usage object
|
|
by converting them to 0, preventing validation errors.
|
|
"""
|
|
self.use_real_langfuse_client()
|
|
|
|
with (
|
|
patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kwargs: generation_params,
|
|
create=True,
|
|
) as mock_add_prompt_params,
|
|
):
|
|
# Create a mock response object with usage information containing None values
|
|
response_obj = MagicMock()
|
|
response_obj.usage = MagicMock()
|
|
response_obj.usage.prompt_tokens = None
|
|
response_obj.usage.completion_tokens = None
|
|
response_obj.usage.total_tokens = None
|
|
|
|
# Mock the .get() method to return None for cache-related fields
|
|
def mock_get(key, default=None):
|
|
if key in ["cache_creation_input_tokens", "cache_read_input_tokens"]:
|
|
return None
|
|
return default
|
|
|
|
response_obj.usage.get = mock_get
|
|
|
|
# Prepare standard kwargs for the call
|
|
kwargs = {
|
|
"model": "gpt-4-null-usage",
|
|
"messages": [{"role": "user", "content": "Test"}],
|
|
"litellm_params": {"metadata": {}},
|
|
"optional_params": {},
|
|
"litellm_call_id": "test-call-id-null-usage",
|
|
"standard_logging_object": self._build_standard_logging_payload(),
|
|
"response_cost": 0.0,
|
|
}
|
|
|
|
# Use fixed timestamps to avoid timing-related flakiness
|
|
fixed_time = datetime.datetime(2024, 1, 1, 12, 0, 0)
|
|
|
|
# Call the method under test
|
|
try:
|
|
self.logger._log_langfuse_v2(
|
|
user_id="test-user",
|
|
metadata={},
|
|
litellm_params=kwargs["litellm_params"],
|
|
output={"role": "assistant", "content": "Response"},
|
|
start_time=fixed_time,
|
|
end_time=fixed_time + datetime.timedelta(seconds=1),
|
|
kwargs=kwargs,
|
|
optional_params=kwargs["optional_params"],
|
|
input={"messages": kwargs["messages"]},
|
|
response_obj=response_obj,
|
|
level="DEFAULT",
|
|
litellm_call_id=kwargs["litellm_call_id"],
|
|
)
|
|
except Exception as e:
|
|
self.fail(f"_log_langfuse_v2 raised an exception: {e}")
|
|
|
|
usage_details = json.loads(self.exported_generation().attributes["langfuse.observation.usage_details"])
|
|
assert usage_details["input"] == 0
|
|
assert usage_details["output"] == 0
|
|
assert usage_details["total"] == 0
|
|
assert usage_details["cache_creation_input_tokens"] == 0
|
|
assert usage_details["cache_read_input_tokens"] == 0
|
|
|
|
mock_add_prompt_params.assert_called_once()
|
|
|
|
def _build_standard_logging_payload(self, trace_id: Optional[str] = None):
|
|
payload = {
|
|
"id": "payload-id",
|
|
"call_type": "completion",
|
|
"response_cost": 0.0,
|
|
"status": "success",
|
|
"total_tokens": 0,
|
|
"prompt_tokens": 0,
|
|
"completion_tokens": 0,
|
|
"startTime": 0.0,
|
|
"endTime": 0.0,
|
|
"completionStartTime": 0.0,
|
|
"model": "gpt-4",
|
|
"model_id": "model-123",
|
|
"model_group": "openai",
|
|
"api_base": "https://api.openai.com",
|
|
# only real StandardLoggingMetadata fields: session_id, trace_name,
|
|
# headers and friends are request-metadata keys the allowlist drops,
|
|
# so a payload carrying them cannot occur in production
|
|
"metadata": {
|
|
"user_api_key_end_user_id": None,
|
|
"prompt_management_metadata": None,
|
|
"user_api_key_hash": "hashed-key",
|
|
"user_api_key_alias": "canary-alias",
|
|
},
|
|
"hidden_params": {},
|
|
"request_tags": [],
|
|
"messages": [],
|
|
"response": {"id": "resp"},
|
|
"model_parameters": {},
|
|
"guardrail_information": None,
|
|
"standard_built_in_tools_params": None,
|
|
}
|
|
if trace_id is not None:
|
|
payload["trace_id"] = trace_id
|
|
return payload
|
|
|
|
def _build_langfuse_kwargs(self, standard_logging_payload):
|
|
return {
|
|
"standard_logging_object": standard_logging_payload,
|
|
"model": standard_logging_payload["model"],
|
|
"call_type": standard_logging_payload["call_type"],
|
|
"cache_hit": False,
|
|
"messages": [],
|
|
}
|
|
|
|
def test_log_langfuse_v2_uses_standard_trace_id_when_available(self):
|
|
payload = self._build_standard_logging_payload(trace_id="std-trace-id")
|
|
kwargs = self._build_langfuse_kwargs(payload)
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kwargs: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata={},
|
|
litellm_params={"metadata": {}},
|
|
output=None,
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=None,
|
|
level="INFO",
|
|
litellm_call_id="call-id-xyz",
|
|
)
|
|
|
|
assert self.span_trace_id(self.exported_generation()) == resolve_trace_id("std-trace-id")
|
|
|
|
def test_log_langfuse_v2_defaults_to_call_id_without_standard_trace_id(self):
|
|
payload = self._build_standard_logging_payload()
|
|
kwargs = self._build_langfuse_kwargs(payload)
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kwargs: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata={},
|
|
litellm_params={"metadata": {}},
|
|
output=None,
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=None,
|
|
level="INFO",
|
|
litellm_call_id="call-id-xyz",
|
|
)
|
|
|
|
assert self.span_trace_id(self.exported_generation()) == resolve_trace_id("call-id-xyz")
|
|
|
|
def test_log_langfuse_v2_uses_litellm_trace_id_fallback_over_call_id(self):
|
|
"""
|
|
When standard_logging_object has no trace_id, but kwargs contains
|
|
litellm_trace_id (the same ID the DB stores as Session ID), Langfuse
|
|
should use litellm_trace_id — NOT litellm_call_id. This ensures the
|
|
trace_id in Langfuse matches the Session ID shown in LiteLLM logs.
|
|
"""
|
|
payload = self._build_standard_logging_payload() # no trace_id
|
|
kwargs = self._build_langfuse_kwargs(payload)
|
|
kwargs["litellm_trace_id"] = "trace-id-from-kwargs"
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kwargs: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata={},
|
|
litellm_params={"metadata": {}},
|
|
output=None,
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=None,
|
|
level="ERROR",
|
|
litellm_call_id="call-id-xyz",
|
|
)
|
|
|
|
# litellm_trace_id should be preferred over litellm_call_id
|
|
assert self.span_trace_id(self.exported_generation()) == resolve_trace_id("trace-id-from-kwargs")
|
|
|
|
CANARY = "sk-lf-canary-SECRET-d4e5f6"
|
|
|
|
def _canary_request_metadata(self):
|
|
"""Raw request metadata shaped like the proxy builds it, credentials included."""
|
|
from litellm.proxy._types import UserAPIKeyAuth
|
|
|
|
team_logging = [
|
|
{
|
|
"callback_name": "langfuse",
|
|
"callback_vars": {"langfuse_secret_key": self.CANARY},
|
|
}
|
|
]
|
|
return {
|
|
"user_api_key_auth": UserAPIKeyAuth(
|
|
api_key="hashed-key",
|
|
team_metadata={"logging": team_logging},
|
|
),
|
|
"user_api_key_team_metadata": {"logging": team_logging},
|
|
"user_api_key_metadata": {"secret_manager_settings": {"vault_token": self.CANARY}},
|
|
"session_id": "canary-session",
|
|
"trace_name": "canary-trace",
|
|
"first_custom": "keep-first",
|
|
"second_custom": "keep-second",
|
|
"endpoint": "/v1/chat/completions",
|
|
"headers": {"authorization": f"Bearer {self.CANARY}"},
|
|
}
|
|
|
|
def _emitted_payload_text(self):
|
|
"""Every attribute this logger exported to langfuse, as one searchable string."""
|
|
import json
|
|
|
|
self.logger.tracing.flush()
|
|
return json.dumps(
|
|
[dict(span.attributes or {}) for span in self.span_exporter.get_finished_spans()],
|
|
default=repr,
|
|
)
|
|
|
|
def exported_generation_metadata(self):
|
|
"""The generation's metadata as langfuse receives it, one attribute per key.
|
|
|
|
v4 serializes each value onto the span, so they are decoded back here to
|
|
keep these assertions about what litellm emitted rather than about the
|
|
SDK's wire encoding.
|
|
"""
|
|
import json
|
|
|
|
prefix = "langfuse.observation.metadata."
|
|
|
|
def decoded(raw):
|
|
try:
|
|
return json.loads(raw)
|
|
except (TypeError, ValueError):
|
|
return raw
|
|
|
|
return {
|
|
key[len(prefix) :]: decoded(value)
|
|
for key, value in (self.exported_generation().attributes or {}).items()
|
|
if key.startswith(prefix)
|
|
}
|
|
|
|
def exported_spans_named(self, name):
|
|
self.logger.tracing.flush()
|
|
return [span for span in self.span_exporter.get_finished_spans() if span.name == name]
|
|
|
|
def _drive_with_canary(self, extra_metadata=None, hidden_params=None, guardrail_information=None):
|
|
metadata = {**self._canary_request_metadata(), **(extra_metadata or {})}
|
|
payload = self._build_standard_logging_payload(trace_id="canary-trace-id")
|
|
if hidden_params is not None:
|
|
payload["hidden_params"] = hidden_params
|
|
if guardrail_information is not None:
|
|
payload["guardrail_information"] = guardrail_information
|
|
kwargs = {**self._build_langfuse_kwargs(payload), "response_cost": 0.25}
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kw: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata=metadata,
|
|
litellm_params={"metadata": metadata},
|
|
output=None,
|
|
start_time=datetime.datetime(2024, 1, 1, 12, 0, 0),
|
|
end_time=datetime.datetime(2024, 1, 1, 12, 0, 1),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=None,
|
|
level="INFO",
|
|
litellm_call_id="canary-call-id",
|
|
)
|
|
return self.exported_generation_metadata()
|
|
|
|
def test_team_callback_credentials_never_reach_langfuse(self):
|
|
"""
|
|
Regression for the credential leak: request metadata carries the whole
|
|
UserAPIKeyAuth object, whose team_metadata holds the customer's own langfuse
|
|
keys. The emitted blob is sourced from StandardLoggingPayload, so none of the
|
|
three credential carriers can ride along.
|
|
"""
|
|
generation_metadata = self._drive_with_canary()
|
|
|
|
assert self.CANARY not in self._emitted_payload_text()
|
|
for leaked_key in (
|
|
"user_api_key_auth",
|
|
"user_api_key_team_metadata",
|
|
"user_api_key_metadata",
|
|
):
|
|
assert leaked_key not in generation_metadata
|
|
|
|
def test_debug_langfuse_dump_carries_no_credentials(self):
|
|
"""
|
|
debug_langfuse dumps request metadata into the trace as a second emit site.
|
|
It must be sourced from the allowlisted payload too.
|
|
"""
|
|
import json
|
|
|
|
self._drive_with_canary(extra_metadata={"debug_langfuse": True})
|
|
dumped = json.loads(self.exported_generation().attributes["langfuse.trace.metadata.metadata_passed_to_litellm"])
|
|
|
|
assert "user_api_key_auth" not in dumped
|
|
assert dumped["first_custom"] == "keep-first"
|
|
assert self.CANARY not in self._emitted_payload_text()
|
|
|
|
def test_raw_request_metadata_reaches_the_emitted_blob_through_no_key(self):
|
|
"""
|
|
The emitted blob is the allowlist plus litellm enrichments, nothing else.
|
|
Nothing from raw request metadata is copied across, whatever its type, which
|
|
is what makes the credential exclusion structural rather than a filter that
|
|
has to be kept correct. Proxy callers keep their own metadata under the
|
|
allowlisted requester_metadata key.
|
|
"""
|
|
generation_metadata = self._drive_with_canary()
|
|
|
|
for caller_key in ("first_custom", "second_custom", "session_id", "trace_name"):
|
|
assert caller_key not in generation_metadata
|
|
|
|
def test_provider_specific_span_receives_the_emitted_blob(self):
|
|
"""
|
|
The provider span reads hidden_params, which is an enrichment on the emitted
|
|
blob rather than a key of request metadata. Handing it the steering dict
|
|
instead would silently stop emitting vertex grounding spans.
|
|
"""
|
|
self._drive_with_canary(hidden_params={"vertex_ai_grounding_metadata": ["ground-a", "ground-b"]})
|
|
|
|
span_inputs = [
|
|
span.attributes.get("langfuse.observation.input")
|
|
for span in self.exported_spans_named("vertex_ai_grounding_metadata")
|
|
]
|
|
assert span_inputs == ["ground-a", "ground-b"]
|
|
assert self.CANARY not in self._emitted_payload_text()
|
|
|
|
def test_only_the_generation_claims_the_trace_root(self):
|
|
"""
|
|
Langfuse derives trace name and I/O from the root observation, and with several
|
|
roots the one with the latest start wins. A post-call guardrail starts after the
|
|
model call, so it must nest under the generation instead of being a root itself, or
|
|
the trace shows the guardrail's request instead of the model's.
|
|
"""
|
|
self._drive_with_canary(
|
|
hidden_params={"vertex_ai_grounding_metadata": ["ground-a"]},
|
|
guardrail_information=[
|
|
{
|
|
"guardrail_name": "pii-post",
|
|
"guardrail_mode": "post_call",
|
|
"guardrail_request": {"texts": ["post-call scan"]},
|
|
"guardrail_response": {"flagged": False},
|
|
"start_time": 1704110402.0,
|
|
"end_time": 1704110403.0,
|
|
}
|
|
],
|
|
)
|
|
|
|
[generation] = [span for span in self.span_exporter.get_finished_spans() if span.name.startswith("litellm-")]
|
|
[guardrail] = self.exported_spans_named("guardrail")
|
|
[grounding] = self.exported_spans_named("vertex_ai_grounding_metadata")
|
|
assert generation.parent is None
|
|
assert generation.attributes["langfuse.trace.name"] == "canary-trace"
|
|
for child in (guardrail, grounding):
|
|
assert child.parent.span_id == generation.context.span_id
|
|
assert child.context.trace_id == generation.context.trace_id
|
|
assert "langfuse.trace.name" not in child.attributes
|
|
|
|
def test_generation_is_exported_when_a_child_span_fails(self):
|
|
"""v2 buffered the generation in one call, so a bad guardrail entry could not lose it;
|
|
the OTel generation is open until ``end()`` and must still be ended when a child raises."""
|
|
self._drive_with_canary(
|
|
guardrail_information=[
|
|
{
|
|
"guardrail_name": "pii-post",
|
|
"guardrail_mode": "post_call",
|
|
"start_time": "not-a-timestamp",
|
|
"end_time": 1704110403.0,
|
|
}
|
|
],
|
|
)
|
|
|
|
[generation] = [span for span in self.span_exporter.get_finished_spans() if span.name.startswith("litellm-")]
|
|
assert generation.attributes["langfuse.trace.name"] == "canary-trace"
|
|
assert self.exported_spans_named("guardrail") == []
|
|
|
|
def test_caller_cannot_spoof_an_allowlisted_identity_field(self):
|
|
"""
|
|
Request metadata never reaches the blob, so a caller naming user_api_key_alias
|
|
cannot have their value emitted in place of the proxy-resolved one.
|
|
"""
|
|
generation_metadata = self._drive_with_canary(extra_metadata={"user_api_key_alias": "spoofed-by-caller"})
|
|
|
|
assert generation_metadata["user_api_key_alias"] == "canary-alias"
|
|
|
|
def test_caller_nested_metadata_cannot_erase_a_litellm_enrichment(self):
|
|
"""
|
|
log_requester_metadata drops any top-level key whose name also appears inside
|
|
requester_metadata. Sourcing the blob from the allowlist populates that nested
|
|
dict for real, so a caller naming a key litellm_response_cost would otherwise
|
|
blank out the cost litellm computed. Enrichments are layered after the dedupe.
|
|
"""
|
|
payload = self._build_standard_logging_payload(trace_id="canary-trace-id")
|
|
payload["metadata"]["requester_metadata"] = {"litellm_response_cost": "caller-value", "api_base": "caller"}
|
|
kwargs = {**self._build_langfuse_kwargs(payload), "response_cost": 0.25}
|
|
metadata = self._canary_request_metadata()
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kw: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata=metadata,
|
|
litellm_params={"metadata": metadata, "api_base": "https://real-api-base"},
|
|
output=None,
|
|
start_time=datetime.datetime(2024, 1, 1, 12, 0, 0),
|
|
end_time=datetime.datetime(2024, 1, 1, 12, 0, 1),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=None,
|
|
level="INFO",
|
|
litellm_call_id="canary-call-id",
|
|
)
|
|
|
|
generation_metadata = self.exported_generation_metadata()
|
|
assert generation_metadata["litellm_response_cost"] == 0.25
|
|
assert generation_metadata["api_base"] == "https://real-api-base"
|
|
|
|
def test_generation_metadata_carries_the_call_id_and_response_id(self):
|
|
"""
|
|
v2's generation id was ``time-<hh-mm-ss-us>_<response id>``, so a generation could
|
|
be found from the provider response id. v4 hashes that string onto 16 hex chars,
|
|
which leaves nothing searchable unless both ids are emitted as metadata.
|
|
"""
|
|
payload = self._build_standard_logging_payload(trace_id="canary-trace-id")
|
|
kwargs = {**self._build_langfuse_kwargs(payload), "response_cost": 0.25}
|
|
metadata = self._canary_request_metadata()
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kw: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata=metadata,
|
|
litellm_params={"metadata": metadata},
|
|
output=None,
|
|
start_time=datetime.datetime(2024, 1, 1, 12, 0, 0),
|
|
end_time=datetime.datetime(2024, 1, 1, 12, 0, 1),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=litellm.ModelResponse(
|
|
id="chatcmpl-canary-response", choices=[{"message": {"role": "assistant", "content": "OK"}}]
|
|
),
|
|
level="DEFAULT",
|
|
litellm_call_id="canary-call-id",
|
|
)
|
|
|
|
generation_metadata = self.exported_generation_metadata()
|
|
assert generation_metadata["litellm_call_id"] == "canary-call-id"
|
|
assert generation_metadata["response_id"] == "chatcmpl-canary-response"
|
|
assert "chatcmpl-canary-response" in self._emitted_payload_text()
|
|
|
|
def test_denied_steering_keys_and_enrichments(self):
|
|
"""
|
|
endpoint is a plain string, so without the deny-list it would ride the
|
|
string re-injection straight into the emitted blob. The enrichments are
|
|
litellm-computed and must survive the move off clean_metadata.
|
|
"""
|
|
generation_metadata = self._drive_with_canary()
|
|
|
|
assert "endpoint" not in generation_metadata
|
|
assert "headers" not in generation_metadata
|
|
assert generation_metadata["litellm_response_cost"] == 0.25
|
|
assert "hidden_params" in generation_metadata
|
|
|
|
def test_cache_hit_is_normalized_on_the_shared_kwargs(self):
|
|
"""
|
|
kwargs here is the shared model_call_details dict. Callbacks that run after
|
|
langfuse read cache_hit off it and copy it into their own payloads, so
|
|
dropping the None to False normalization records None for datadog, logfire,
|
|
generic_api and spend tracking.
|
|
"""
|
|
metadata = self._canary_request_metadata()
|
|
payload = self._build_standard_logging_payload(trace_id="canary-trace-id")
|
|
kwargs = {**self._build_langfuse_kwargs(payload), "cache_hit": None}
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kw: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata=metadata,
|
|
litellm_params={"metadata": metadata},
|
|
output=None,
|
|
start_time=datetime.datetime(2024, 1, 1, 12, 0, 0),
|
|
end_time=datetime.datetime(2024, 1, 1, 12, 0, 1),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=None,
|
|
level="INFO",
|
|
litellm_call_id="canary-call-id",
|
|
)
|
|
|
|
assert kwargs["cache_hit"] is False
|
|
|
|
def test_redact_user_api_key_info_still_strips_the_emitted_blob(self):
|
|
"""
|
|
The flag used to act on the raw-derived blob. That blob is now sourced from
|
|
StandardLoggingPayload, which is where the user_api_key_* fields live, so the
|
|
redaction has to run on the assembled payload or the flag silently stops working.
|
|
"""
|
|
with patch.object(litellm, "redact_user_api_key_info", True):
|
|
generation_metadata = self._drive_with_canary()
|
|
|
|
assert not [key for key in generation_metadata if key.startswith("user_api_key")]
|
|
|
|
def test_steering_keys_still_read_from_raw_metadata(self):
|
|
"""
|
|
Only the emitted payload moves to StandardLoggingPayload. The control fields
|
|
keep reading raw metadata, which is what Braintrust's migration got wrong.
|
|
"""
|
|
self._drive_with_canary()
|
|
|
|
generation = self.exported_generation()
|
|
assert generation.attributes["session.id"] == "canary-session"
|
|
assert generation.attributes["langfuse.trace.name"] == "canary-trace"
|
|
|
|
def test_failure_trace_survives_a_missing_standard_logging_object(self):
|
|
"""
|
|
get_standard_logging_object_payload is fail-open and returns None on any
|
|
exception, which is exactly the failed-request case Langfuse most needs to
|
|
show. The trace is still emitted with the litellm_trace_id fallback, and the
|
|
blob degrades to caller strings plus enrichments rather than falling back to
|
|
raw metadata, which would ship the UserAPIKeyAuth object.
|
|
"""
|
|
metadata = self._canary_request_metadata()
|
|
kwargs = {
|
|
"standard_logging_object": None,
|
|
"model": "gpt-4",
|
|
"call_type": "completion",
|
|
"cache_hit": False,
|
|
"messages": [],
|
|
"litellm_trace_id": "trace-id-failure",
|
|
}
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kwargs: generation_params,
|
|
create=True,
|
|
):
|
|
trace_id, _ = self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata=metadata,
|
|
litellm_params={"metadata": metadata},
|
|
output=None,
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=None,
|
|
level="ERROR",
|
|
litellm_call_id="call-id-different",
|
|
)
|
|
|
|
import json
|
|
|
|
# Must use litellm_trace_id, not litellm_call_id. v4 addresses a trace by a
|
|
# 32-hex id, so the callback returns the resolved form, which is what makes
|
|
# the alerting deep link point at a trace langfuse can actually open
|
|
assert trace_id == resolve_trace_id("trace-id-failure")
|
|
assert self.span_trace_id(self.exported_generation()) == trace_id
|
|
generation_metadata = self.exported_generation_metadata()
|
|
assert "user_api_key_auth" not in generation_metadata
|
|
assert self.CANARY not in self._emitted_payload_text()
|
|
assert "first_custom" not in generation_metadata
|
|
# hidden_params comes off the payload, so it is omitted rather than emitted
|
|
# as an unserializable placeholder
|
|
assert "hidden_params" not in generation_metadata
|
|
json.dumps(generation_metadata)
|
|
|
|
def test_log_langfuse_v2_session_id_passed_as_trace_session_id(self):
|
|
"""
|
|
Test that metadata.session_id is correctly passed as trace_params["session_id"]
|
|
for Langfuse session grouping, and does NOT override trace_id.
|
|
Each LLM call should get its own unique trace_id while sharing the session_id.
|
|
"""
|
|
payload = self._build_standard_logging_payload(trace_id="std-trace-123")
|
|
kwargs = self._build_langfuse_kwargs(payload)
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kwargs: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata={"session_id": "my-session-abc"},
|
|
litellm_params={"metadata": {"session_id": "my-session-abc"}},
|
|
output=None,
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=None,
|
|
level="INFO",
|
|
litellm_call_id="call-id-456",
|
|
)
|
|
|
|
# session_id should be set for Langfuse session grouping
|
|
assert self.exported_generation().attributes["session.id"] == "my-session-abc"
|
|
# trace_id should remain the standard trace_id, NOT the session_id
|
|
assert self.span_trace_id(self.exported_generation()) == resolve_trace_id("std-trace-123")
|
|
|
|
def test_log_langfuse_v2_session_id_preserved_for_error_level(self):
|
|
"""
|
|
Test that session_id is correctly passed in trace_params even when
|
|
the log level is ERROR (failure case). This verifies the fix for
|
|
failed requests losing session_id mapping in Langfuse.
|
|
"""
|
|
payload = self._build_standard_logging_payload(trace_id="std-trace-err")
|
|
kwargs = self._build_langfuse_kwargs(payload)
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kwargs: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata={"session_id": "error-session-xyz"},
|
|
litellm_params={"metadata": {"session_id": "error-session-xyz"}},
|
|
output="BadRequestError: model not found",
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input={"messages": [{"role": "user", "content": "test"}]},
|
|
response_obj=None,
|
|
level="ERROR",
|
|
litellm_call_id="call-id-err-789",
|
|
)
|
|
|
|
# session_id must be preserved even for ERROR level logs
|
|
assert self.exported_generation().attributes["session.id"] == "error-session-xyz"
|
|
# trace_id should be the standard trace_id, not the session_id
|
|
assert self.span_trace_id(self.exported_generation()) == resolve_trace_id("std-trace-err")
|
|
# status_message should be set for error traces
|
|
assert self.exported_generation().attributes["langfuse.observation.level"] == "ERROR"
|
|
|
|
def test_log_langfuse_v2_explicit_trace_id_takes_priority_over_session_id(self):
|
|
"""
|
|
Test that when both trace_id and session_id are provided in metadata,
|
|
trace_id takes priority as the trace identifier.
|
|
"""
|
|
payload = self._build_standard_logging_payload()
|
|
kwargs = self._build_langfuse_kwargs(payload)
|
|
self.use_real_langfuse_client()
|
|
|
|
with patch(
|
|
"litellm.integrations.langfuse.langfuse._add_prompt_to_generation_params",
|
|
side_effect=lambda generation_params, **kwargs: generation_params,
|
|
create=True,
|
|
):
|
|
self.logger._log_langfuse_v2(
|
|
user_id="user-1",
|
|
metadata={
|
|
"session_id": "session-999",
|
|
"trace_id": "explicit-trace-id-777",
|
|
},
|
|
litellm_params={
|
|
"metadata": {
|
|
"session_id": "session-999",
|
|
"trace_id": "explicit-trace-id-777",
|
|
}
|
|
},
|
|
output=None,
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
kwargs=kwargs,
|
|
optional_params={},
|
|
input=None,
|
|
response_obj=None,
|
|
level="DEFAULT",
|
|
litellm_call_id="call-id-aaa",
|
|
)
|
|
|
|
# Explicit trace_id must take priority
|
|
assert self.span_trace_id(self.exported_generation()) == resolve_trace_id("explicit-trace-id-777")
|
|
# session_id must still be set for session grouping
|
|
assert self.exported_generation().attributes["session.id"] == "session-999"
|
|
|
|
|
|
def test_failure_handler_langfuse_kwargs_excludes_original_response():
|
|
"""
|
|
Test that the actual Logging.failure_handler() passes kwargs without
|
|
'original_response' to the Langfuse logger. Exercises the real code path
|
|
rather than simulating the filtering logic.
|
|
"""
|
|
import litellm
|
|
from litellm.litellm_core_utils.litellm_logging import Logging
|
|
|
|
# Create a Logging instance
|
|
logging_obj = Logging(
|
|
model="gpt-4",
|
|
messages=[{"role": "user", "content": "test"}],
|
|
stream=False,
|
|
call_type="completion",
|
|
start_time=datetime.datetime.utcnow(),
|
|
litellm_call_id="test-call-id-failure",
|
|
function_id="test-function-id",
|
|
)
|
|
|
|
# Set up model_call_details with original_response (simulates a coroutine)
|
|
mock_coroutine = MagicMock()
|
|
logging_obj.model_call_details["original_response"] = mock_coroutine
|
|
logging_obj.model_call_details["litellm_params"] = {
|
|
"metadata": {"session_id": "test-session-failure"},
|
|
"litellm_session_id": None,
|
|
}
|
|
logging_obj.model_call_details["optional_params"] = {}
|
|
|
|
# Capture what gets passed to log_event_on_langfuse
|
|
captured_kwargs = {}
|
|
mock_langfuse_logger = MagicMock()
|
|
|
|
def capture_log_event(**log_kwargs):
|
|
captured_kwargs.update(log_kwargs)
|
|
return {"trace_id": "mock-trace-id", "generation_id": "mock-gen-id"}
|
|
|
|
mock_langfuse_logger.log_event_on_langfuse.side_effect = capture_log_event
|
|
|
|
# Set "langfuse" as a failure callback so the failure_handler processes it
|
|
original_failure_callback = litellm.failure_callback
|
|
litellm.failure_callback = ["langfuse"]
|
|
|
|
try:
|
|
# Mock LangFuseHandler to return our capturing mock logger
|
|
with (
|
|
patch("litellm.litellm_core_utils.litellm_logging.LangFuseHandler") as mock_handler_class
|
|
): # test-quality-ok: route the request to the capturing logger; the real handler builds live clients
|
|
mock_handler_class.get_langfuse_logger_for_request.return_value = mock_langfuse_logger
|
|
|
|
# Call the actual failure_handler
|
|
test_exception = Exception("TestError: model not found")
|
|
logging_obj.failure_handler(
|
|
exception=test_exception,
|
|
traceback_exception="Traceback: test",
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
)
|
|
|
|
# Verify log_event_on_langfuse was actually called
|
|
assert mock_langfuse_logger.log_event_on_langfuse.called, "log_event_on_langfuse was not called"
|
|
|
|
# Verify original_response is NOT in the kwargs passed to Langfuse
|
|
langfuse_kwargs = captured_kwargs.get("kwargs", {})
|
|
assert "original_response" not in langfuse_kwargs, (
|
|
"original_response should be excluded from kwargs passed to Langfuse"
|
|
)
|
|
|
|
# Verify session_id metadata is preserved in the kwargs
|
|
langfuse_metadata = langfuse_kwargs.get("litellm_params", {}).get("metadata", {})
|
|
assert langfuse_metadata.get("session_id") == "test-session-failure", (
|
|
"session_id should be preserved in kwargs passed to Langfuse"
|
|
)
|
|
|
|
# Verify level is ERROR
|
|
assert captured_kwargs.get("level") == "ERROR"
|
|
finally:
|
|
litellm.failure_callback = original_failure_callback
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_async_log_failure_event_logs_to_langfuse():
|
|
"""
|
|
Test that LangfusePromptManagement.async_log_failure_event() calls
|
|
log_event_on_langfuse with level=ERROR even when standard_logging_object
|
|
is present. This is the code path the proxy uses for failed LLM calls.
|
|
"""
|
|
from litellm.integrations.langfuse.langfuse_prompt_management import (
|
|
LangfusePromptManagement,
|
|
)
|
|
|
|
mock_langfuse_module = MagicMock()
|
|
mock_langfuse_module.version.__version__ = "3.0.0"
|
|
|
|
with (
|
|
patch.dict(
|
|
"os.environ",
|
|
{
|
|
"LANGFUSE_SECRET_KEY": "test-secret",
|
|
"LANGFUSE_PUBLIC_KEY": "test-public",
|
|
"LANGFUSE_HOST": "https://test.langfuse.com",
|
|
},
|
|
),
|
|
patch.dict("sys.modules", {"langfuse": mock_langfuse_module}),
|
|
):
|
|
prompt_mgmt = LangfusePromptManagement()
|
|
|
|
# Mock the langfuse logger returned by get_langfuse_logger_for_request
|
|
mock_logger = MagicMock()
|
|
mock_logger.log_event_on_langfuse.return_value = {
|
|
"trace_id": "mock-trace",
|
|
"generation_id": "mock-gen",
|
|
}
|
|
|
|
with (
|
|
patch("litellm.integrations.langfuse.langfuse_prompt_management.LangFuseHandler") as mock_handler
|
|
): # test-quality-ok: route the request to the capturing logger; the real handler builds live clients
|
|
mock_handler.get_langfuse_logger_for_request.return_value = mock_logger
|
|
|
|
kwargs = {
|
|
"litellm_params": {
|
|
"metadata": {"session_id": "test-session-fail"},
|
|
},
|
|
"litellm_call_id": "call-fail-123",
|
|
"user": "test-user",
|
|
"exception": Exception("API error: model not found"),
|
|
"standard_logging_object": {
|
|
"error_str": "API error: model not found",
|
|
"trace_id": "std-trace-fail",
|
|
"metadata": {},
|
|
},
|
|
}
|
|
|
|
await prompt_mgmt.async_log_failure_event(
|
|
kwargs=kwargs,
|
|
response_obj=None,
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
)
|
|
|
|
# Verify log_event_on_langfuse was called
|
|
assert mock_logger.log_event_on_langfuse.called, "log_event_on_langfuse was not called for failure event"
|
|
call_kwargs = mock_logger.log_event_on_langfuse.call_args[1]
|
|
assert call_kwargs["level"] == "ERROR"
|
|
assert call_kwargs["status_message"] == "API error: model not found"
|
|
assert call_kwargs["response_obj"] is None
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_async_log_failure_event_works_without_standard_logging_object():
|
|
"""
|
|
Test that async_log_failure_event() still logs to Langfuse even when
|
|
standard_logging_object is None (e.g. when get_standard_logging_object_payload
|
|
threw an exception). This is the critical fix — before, it silently returned.
|
|
"""
|
|
from litellm.integrations.langfuse.langfuse_prompt_management import (
|
|
LangfusePromptManagement,
|
|
)
|
|
|
|
mock_langfuse_module = MagicMock()
|
|
mock_langfuse_module.version.__version__ = "3.0.0"
|
|
|
|
with (
|
|
patch.dict(
|
|
"os.environ",
|
|
{
|
|
"LANGFUSE_SECRET_KEY": "test-secret",
|
|
"LANGFUSE_PUBLIC_KEY": "test-public",
|
|
"LANGFUSE_HOST": "https://test.langfuse.com",
|
|
},
|
|
),
|
|
patch.dict("sys.modules", {"langfuse": mock_langfuse_module}),
|
|
):
|
|
prompt_mgmt = LangfusePromptManagement()
|
|
|
|
mock_logger = MagicMock()
|
|
mock_logger.log_event_on_langfuse.return_value = {
|
|
"trace_id": "mock-trace",
|
|
"generation_id": "mock-gen",
|
|
}
|
|
|
|
with (
|
|
patch("litellm.integrations.langfuse.langfuse_prompt_management.LangFuseHandler") as mock_handler
|
|
): # test-quality-ok: route the request to the capturing logger; the real handler builds live clients
|
|
mock_handler.get_langfuse_logger_for_request.return_value = mock_logger
|
|
|
|
kwargs = {
|
|
"litellm_params": {
|
|
"metadata": {"session_id": "test-session-no-slo"},
|
|
},
|
|
"litellm_call_id": "call-no-slo-456",
|
|
"user": "test-user",
|
|
"exception": Exception("InternalServerError: something broke"),
|
|
"standard_logging_object": None, # This is the key — it's None
|
|
}
|
|
|
|
await prompt_mgmt.async_log_failure_event(
|
|
kwargs=kwargs,
|
|
response_obj=None,
|
|
start_time=datetime.datetime.utcnow(),
|
|
end_time=datetime.datetime.utcnow(),
|
|
)
|
|
|
|
# CRITICAL: log_event_on_langfuse MUST still be called
|
|
assert mock_logger.log_event_on_langfuse.called, (
|
|
"log_event_on_langfuse was NOT called when standard_logging_object "
|
|
"is None — failure trace would be silently dropped"
|
|
)
|
|
call_kwargs = mock_logger.log_event_on_langfuse.call_args[1]
|
|
assert call_kwargs["level"] == "ERROR"
|
|
# Falls back to exception from kwargs
|
|
assert "InternalServerError" in call_kwargs["status_message"]
|
|
|
|
|
|
class _OtlpReceiver:
|
|
"""A local HTTP server that records the paths of every POST it gets, standing in for Langfuse."""
|
|
|
|
def __init__(self) -> None:
|
|
from http.server import BaseHTTPRequestHandler, HTTPServer
|
|
|
|
self.received: list[str] = []
|
|
received = self.received
|
|
|
|
class _Handler(BaseHTTPRequestHandler):
|
|
def do_POST(self):
|
|
received.append(self.path)
|
|
self.rfile.read(int(self.headers.get("Content-Length") or 0))
|
|
self.send_response(200)
|
|
self.send_header("Content-Length", "0")
|
|
self.end_headers()
|
|
|
|
def log_message(self, *args):
|
|
pass
|
|
|
|
self.server = HTTPServer(("127.0.0.1", 0), _Handler)
|
|
threading.Thread(target=self.server.serve_forever, daemon=True).start()
|
|
|
|
@property
|
|
def url(self) -> str:
|
|
return f"http://127.0.0.1:{self.server.server_port}"
|
|
|
|
def close(self) -> None:
|
|
self.server.shutdown()
|
|
|
|
|
|
def _log_one_completion(logger: LangFuseLogger) -> None:
|
|
now = datetime.datetime.now()
|
|
logger.log_event_on_langfuse(
|
|
kwargs={
|
|
"call_type": "completion",
|
|
"litellm_params": {"metadata": {}, "proxy_server_request": {"headers": {}}},
|
|
"messages": [{"role": "user", "content": "hi"}],
|
|
"optional_params": {},
|
|
},
|
|
response_obj=litellm.ModelResponse(choices=[{"message": {"role": "assistant", "content": "yo"}}]),
|
|
start_time=now,
|
|
end_time=now,
|
|
)
|
|
logger.flush()
|
|
|
|
|
|
def test_mock_mode_makes_no_network_calls(monkeypatch):
|
|
"""LANGFUSE_MOCK promises full execution without egress.
|
|
|
|
The mock intercepts httpx, but v4 ships observations over its own OTLP
|
|
exporter, so nothing stops a real request to the configured host without an
|
|
exporter that drops them.
|
|
"""
|
|
receiver = _OtlpReceiver()
|
|
monkeypatch.setenv("LANGFUSE_MOCK", "true")
|
|
monkeypatch.setenv("LANGFUSE_HOST", receiver.url)
|
|
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-mock-egress")
|
|
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-mock-egress")
|
|
|
|
try:
|
|
logger = LangFuseLogger()
|
|
assert logger.is_mock_mode is True
|
|
_log_one_completion(logger)
|
|
time.sleep(1)
|
|
finally:
|
|
receiver.close()
|
|
|
|
assert receiver.received == [], f"mock mode sent real requests: {receiver.received}"
|
|
|
|
|
|
def test_max_langfuse_clients_limit():
|
|
"""
|
|
Test that the max langfuse clients limit is respected when initializing multiple clients
|
|
"""
|
|
# Mock langfuse package to avoid triggering real import.
|
|
# The real langfuse import fails on Python 3.14 due to pydantic v1 incompatibility,
|
|
# and sys.modules["langfuse"] may be absent after other tests in the suite clean up.
|
|
mock_langfuse = MagicMock()
|
|
mock_langfuse.version.__version__ = "3.0.0"
|
|
# Set max clients to 2 for testing
|
|
original_initialized_langfuse_clients = litellm.initialized_langfuse_clients
|
|
with (
|
|
patch.dict("sys.modules", {"langfuse": mock_langfuse}),
|
|
patch.object(langfuse_module, "MAX_LANGFUSE_INITIALIZED_CLIENTS", 2),
|
|
):
|
|
# Reset the counter
|
|
litellm.initialized_langfuse_clients = 0
|
|
|
|
# First client should succeed
|
|
logger1 = LangFuseLogger(
|
|
langfuse_public_key="test_key_1",
|
|
langfuse_secret="test_secret_1",
|
|
langfuse_host="https://test1.langfuse.com",
|
|
)
|
|
assert litellm.initialized_langfuse_clients == 1
|
|
|
|
# Second client should succeed
|
|
logger2 = LangFuseLogger(
|
|
langfuse_public_key="test_key_2",
|
|
langfuse_secret="test_secret_2",
|
|
langfuse_host="https://test2.langfuse.com",
|
|
)
|
|
assert litellm.initialized_langfuse_clients == 2
|
|
|
|
# Third client should fail with exception
|
|
with pytest.raises(Exception, match="Max langfuse clients reached") as exc_info:
|
|
logger3 = LangFuseLogger(
|
|
langfuse_public_key="test_key_3",
|
|
langfuse_secret="test_secret_3",
|
|
langfuse_host="https://test3.langfuse.com",
|
|
)
|
|
|
|
# Verify the error message contains the expected text
|
|
assert "Max langfuse clients reached" in str(exc_info.value)
|
|
|
|
# Counter should still be 2 (third client failed to initialize)
|
|
assert litellm.initialized_langfuse_clients == 2
|
|
|
|
litellm.initialized_langfuse_clients = original_initialized_langfuse_clients
|
|
|
|
|
|
_UNREACHABLE_HOST: Final = "http://127.0.0.1:1"
|
|
|
|
|
|
def _build_langfuse_logger(monkeypatch, **overrides) -> LangFuseLogger:
|
|
monkeypatch.setenv("LANGFUSE_MOCK", "false")
|
|
monkeypatch.setattr(litellm, "initialized_langfuse_clients", 0)
|
|
return LangFuseLogger(
|
|
**{
|
|
"langfuse_public_key": "pk-lit5228",
|
|
"langfuse_secret": "sk-lit5228",
|
|
"langfuse_host": _UNREACHABLE_HOST,
|
|
**overrides,
|
|
}
|
|
)
|
|
|
|
|
|
def _exported_environment(logger: LangFuseLogger):
|
|
from langfuse import LangfuseOtelSpanAttributes
|
|
|
|
return logger.tracing.provider.resource.attributes.get(LangfuseOtelSpanAttributes.ENVIRONMENT)
|
|
|
|
|
|
def test_langfuse_environment_lands_on_every_exported_span(monkeypatch):
|
|
monkeypatch.delenv("LANGFUSE_TRACING_ENVIRONMENT", raising=False)
|
|
logger = _build_langfuse_logger(monkeypatch, langfuse_public_key="pk-env", langfuse_environment="staging")
|
|
assert logger.langfuse_environment == "staging"
|
|
assert _exported_environment(logger) == "staging"
|
|
|
|
|
|
def test_langfuse_environment_falls_back_to_deployment_env_var(monkeypatch):
|
|
monkeypatch.setenv("LANGFUSE_TRACING_ENVIRONMENT", "deployment-wide")
|
|
logger = _build_langfuse_logger(monkeypatch, langfuse_public_key="pk-env")
|
|
assert logger.langfuse_environment == "deployment-wide"
|
|
assert _exported_environment(logger) == "deployment-wide"
|
|
|
|
|
|
def _exported_release(logger: LangFuseLogger):
|
|
from langfuse import LangfuseOtelSpanAttributes
|
|
|
|
return logger.tracing.provider.resource.attributes.get(LangfuseOtelSpanAttributes.RELEASE)
|
|
|
|
|
|
@pytest.mark.parametrize("platform_var", ["GITHUB_SHA", "CI_COMMIT_SHA", "RENDER_GIT_COMMIT", "SOURCE_VERSION"])
|
|
def test_release_falls_back_to_the_deploy_platforms_commit_variable(monkeypatch, platform_var):
|
|
"""Deployments that never set ``LANGFUSE_RELEASE`` still got a release on every trace from the v2 SDK, which
|
|
read the CI or hosting platform's commit variable; dropping that silently blanked their release filter."""
|
|
from litellm.integrations.langfuse.langfuse_sdk import _COMMON_RELEASE_ENVS
|
|
|
|
for name in ("LANGFUSE_RELEASE", *_COMMON_RELEASE_ENVS):
|
|
monkeypatch.delenv(name, raising=False)
|
|
monkeypatch.setenv(platform_var, "deadbeef")
|
|
logger = _build_langfuse_logger(monkeypatch, langfuse_public_key=f"pk-release-{platform_var}")
|
|
assert logger.langfuse_release == "deadbeef"
|
|
assert _exported_release(logger) == "deadbeef"
|
|
|
|
|
|
def test_explicit_langfuse_release_wins_over_the_platform_commit(monkeypatch):
|
|
monkeypatch.setenv("LANGFUSE_RELEASE", "v9")
|
|
monkeypatch.setenv("GITHUB_SHA", "deadbeef")
|
|
logger = _build_langfuse_logger(monkeypatch, langfuse_public_key="pk-release-explicit")
|
|
assert _exported_release(logger) == "v9"
|
|
|
|
|
|
def test_non_string_generation_name_is_exported_as_its_text(monkeypatch):
|
|
"""v2 coerced ``generation_name`` through pydantic; a raw int would now fail OTLP encoding and lose the batch."""
|
|
rig = _steering_logger()
|
|
|
|
_, _, span = _emit(rig, metadata={"generation_name": 12345})
|
|
|
|
assert span.name == "12345"
|
|
|
|
|
|
def test_dynamic_langfuse_environment_triggers_dynamic_logger():
|
|
from litellm.integrations.langfuse.langfuse_handler import LangFuseHandler
|
|
from litellm.types.utils import StandardCallbackDynamicParams
|
|
|
|
params = StandardCallbackDynamicParams(langfuse_environment="team-a-env")
|
|
|
|
assert LangFuseHandler._dynamic_langfuse_credentials_are_passed(params) is True
|
|
|
|
config = LangFuseHandler.get_dynamic_langfuse_logging_config(standard_callback_dynamic_params=params)
|
|
assert config["langfuse_environment"] == "team-a-env"
|
|
|
|
|
|
def test_langfuse_rest_client_survives_httpx_cache_eviction(monkeypatch):
|
|
import gc
|
|
import weakref
|
|
|
|
from litellm.caching.llm_caching_handler import LLMClientCache
|
|
|
|
from litellm.llms.custom_httpx.http_handler import _get_httpx_client
|
|
|
|
monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache())
|
|
logger = _build_langfuse_logger(monkeypatch)
|
|
|
|
cached_handler = _get_httpx_client()
|
|
handler_ref = weakref.ref(cached_handler)
|
|
|
|
assert logger.langfuse_client is cached_handler.client
|
|
|
|
litellm.in_memory_llm_clients_cache = LLMClientCache()
|
|
del cached_handler
|
|
gc.collect()
|
|
|
|
assert litellm.in_memory_llm_clients_cache.get_cache("httpx_client") is None
|
|
assert handler_ref() is not None, "logger must keep the handler that owns the client behind its REST API"
|
|
assert not logger.langfuse_client.is_closed
|
|
assert logger.api_client.auth_check() is not None
|
|
|
|
|
|
def test_langfuse_logger_reuses_the_shared_cached_client(monkeypatch):
|
|
import gc
|
|
|
|
from litellm.caching.llm_caching_handler import LLMClientCache
|
|
|
|
monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache())
|
|
|
|
first = _build_langfuse_logger(monkeypatch)
|
|
second = _build_langfuse_logger(monkeypatch)
|
|
|
|
assert first.langfuse_client is second.langfuse_client
|
|
|
|
del second
|
|
gc.collect()
|
|
|
|
assert not first.langfuse_client.is_closed
|
|
|
|
|
|
_LANGFUSE_REDACTED = "redacted-by-litellm"
|
|
|
|
|
|
def _steering_logger():
|
|
"""``__new__`` skips the network setup in ``__init__``; spans land in memory."""
|
|
from opentelemetry.sdk.trace.export.in_memory_span_exporter import (
|
|
InMemorySpanExporter,
|
|
)
|
|
|
|
from litellm.integrations.langfuse.langfuse import installed_langfuse_version
|
|
from litellm.integrations.langfuse.langfuse_sdk import build_langfuse_client, build_langfuse_tracing
|
|
|
|
exporter = InMemorySpanExporter()
|
|
logger = LangFuseLogger.__new__(LangFuseLogger)
|
|
logger.tracing = build_langfuse_tracing(
|
|
exporter=exporter, environment=None, release=None, sample_rate=1.0, flush_interval_millis=10
|
|
)
|
|
logger.api_client = build_langfuse_client(
|
|
public_key="pk-steering-test", secret_key="sk-steering-test", base_url=_UNREACHABLE_HOST, httpx_client=None
|
|
)
|
|
logger.langfuse_sdk_version = installed_langfuse_version()
|
|
return logger, exporter
|
|
|
|
|
|
def test_log_event_keeps_exporting_after_the_dynamic_cache_evicts_the_logger():
|
|
"""Per-key loggers are evicted from ``DynamicLoggingCache`` while a callback may still hold them.
|
|
|
|
v2 lost that callback's events to a shut-down client; the export channel is shared per
|
|
credential set and outlives any one logger, so the events still land.
|
|
"""
|
|
from litellm.litellm_core_utils.specialty_caches.dynamic_logging_cache import LangfuseInMemoryCache
|
|
|
|
logger, exporter = _steering_logger()
|
|
cache = LangfuseInMemoryCache()
|
|
cache.set_cache("langfuse-evicted", logger)
|
|
litellm.initialized_langfuse_clients += 1
|
|
before = litellm.initialized_langfuse_clients
|
|
cache._remove_key("langfuse-evicted")
|
|
|
|
now = datetime.datetime.now()
|
|
returned = logger.log_event_on_langfuse(
|
|
kwargs={
|
|
"call_type": "completion",
|
|
"litellm_params": {"metadata": {}},
|
|
"messages": [{"role": "user", "content": "the-input"}],
|
|
"optional_params": {},
|
|
},
|
|
response_obj=litellm.ModelResponse(choices=[{"message": {"role": "assistant", "content": "the-output"}}]),
|
|
start_time=now,
|
|
end_time=now,
|
|
)
|
|
|
|
assert litellm.initialized_langfuse_clients == before - 1
|
|
assert _span_trace_id(_exported_span(logger, exporter)) == returned["trace_id"]
|
|
|
|
|
|
def _exported_span(logger, exporter):
|
|
logger.flush()
|
|
return exporter.get_finished_spans()[-1]
|
|
|
|
|
|
_TRACE_FIELD_KEYS = {
|
|
"user.id": "user_id",
|
|
"session.id": "session_id",
|
|
"langfuse.version": "version",
|
|
"langfuse.release": "release",
|
|
}
|
|
|
|
|
|
def _trace_params(span):
|
|
"""The trace-level fields of the exported span, keyed as v2's ``trace_params`` were."""
|
|
prefix = "langfuse.trace."
|
|
attributes = span.attributes or {}
|
|
return {
|
|
**{
|
|
key[len(prefix) :]: value
|
|
for key, value in attributes.items()
|
|
if key.startswith(prefix) and not key.startswith(prefix + "metadata.")
|
|
},
|
|
**{name: attributes[key] for key, name in _TRACE_FIELD_KEYS.items() if key in attributes},
|
|
}
|
|
|
|
|
|
def _span_trace_id(span):
|
|
return format(span.context.trace_id, "032x")
|
|
|
|
|
|
def _emit(rig, *, metadata=None, headers=None):
|
|
"""``log_event_on_langfuse`` is the entry point that folds ``langfuse_*`` headers into metadata.
|
|
|
|
Both the trace-level and the observation fields are read back off the span litellm exported.
|
|
"""
|
|
logger, exporter = rig
|
|
exporter.clear()
|
|
|
|
now = datetime.datetime.now()
|
|
response_obj = litellm.ModelResponse(choices=[{"message": {"role": "assistant", "content": "the-output"}}])
|
|
logger.log_event_on_langfuse(
|
|
kwargs={
|
|
"call_type": "completion",
|
|
"litellm_params": {
|
|
"metadata": dict(metadata or {}),
|
|
"proxy_server_request": {"headers": dict(headers or {})},
|
|
},
|
|
"messages": [{"role": "user", "content": "the-input"}],
|
|
"optional_params": {},
|
|
},
|
|
response_obj=response_obj,
|
|
start_time=now,
|
|
end_time=now,
|
|
)
|
|
prefix = "langfuse.observation."
|
|
span = _exported_span(logger, exporter)
|
|
generation_params = {
|
|
key[len(prefix) :]: value
|
|
for key, value in (span.attributes or {}).items()
|
|
if key.startswith(prefix) and not key.startswith(prefix + "metadata.")
|
|
}
|
|
return _trace_params(span), generation_params, span
|
|
|
|
|
|
@pytest.mark.parametrize("level", ["DEFAULT", "ERROR"])
|
|
@pytest.mark.parametrize(
|
|
"headers,metadata,expected_id",
|
|
[
|
|
({"x-litellm-session-id": "session-7125"}, {}, "call"),
|
|
({"X-Claude-Code-Session-Id": "session-7125"}, {}, "call"),
|
|
({"x-session-id": "session-7125"}, {}, "call"),
|
|
({"session-id": "session-7125", "user-agent": "codex_cli_rs/1.0"}, {}, "call"),
|
|
({"thread-id": "session-7125", "user-agent": "codex-tui"}, {}, "call"),
|
|
({"session_id": "session-7125", "user-agent": "Codex 1.0"}, {}, "call"),
|
|
({"conversation_id": "session-7125", "user-agent": "codex_vscode/1.0"}, {}, "call"),
|
|
({"x-litellm-session-id": "short"}, {}, "call"),
|
|
({"x-litellm-trace-id": "session-7125"}, {}, "session-7125"),
|
|
(
|
|
{"X-LiteLLM-Trace-Id": "session-7125", "x-litellm-session-id": "session-7125"},
|
|
{},
|
|
"session-7125",
|
|
),
|
|
(
|
|
{"x-litellm-session-id": "session-7125", "langfuse_trace_id": "session-7125"},
|
|
{},
|
|
"session-7125",
|
|
),
|
|
(
|
|
{"x-litellm-session-id": "session-7125", "langfuse_trace_id": "explicit-trace"},
|
|
{},
|
|
"explicit-trace",
|
|
),
|
|
(
|
|
{"x-litellm-session-id": "session-7125", "langfuse_existing_trace_id": "existing-trace"},
|
|
{},
|
|
"existing-trace",
|
|
),
|
|
(
|
|
{"x-litellm-session-id": "session-7125", "langfuse_session_id": "custom-session"},
|
|
{},
|
|
"call",
|
|
),
|
|
(
|
|
{"x-litellm-session-id": "short", "langfuse_session_id": "custom-session"},
|
|
{},
|
|
"call",
|
|
),
|
|
(
|
|
{"X-Claude-Code-Session-Id": "session-7125", "langfuse_session_id": "custom-session"},
|
|
{},
|
|
"call",
|
|
),
|
|
(
|
|
{"x-session-id": "session-7125", "langfuse_session_id": "custom-session"},
|
|
{},
|
|
"call",
|
|
),
|
|
(
|
|
{
|
|
"session-id": "session-7125",
|
|
"user-agent": "codex_cli_rs/1.0",
|
|
"langfuse_session_id": "custom-session",
|
|
},
|
|
{},
|
|
"call",
|
|
),
|
|
(
|
|
{
|
|
"x-litellm-session-id": "session-7125",
|
|
"langfuse_session_id": "custom-session",
|
|
"x-litellm-trace-id": "explicit-trace",
|
|
},
|
|
{},
|
|
"explicit-trace",
|
|
),
|
|
(
|
|
{
|
|
"x-litellm-session-id": "session-7125",
|
|
"langfuse_session_id": "custom-session",
|
|
"langfuse_trace_id": "explicit-trace",
|
|
},
|
|
{},
|
|
"explicit-trace",
|
|
),
|
|
(
|
|
{
|
|
"x-litellm-session-id": "session-7125",
|
|
"langfuse_session_id": "custom-session",
|
|
"langfuse_existing_trace_id": "existing-trace",
|
|
},
|
|
{},
|
|
"existing-trace",
|
|
),
|
|
({}, {"trace_id": "session-7125", "session_id": "session-7125"}, "session-7125"),
|
|
({}, {"trace_id": "explicit-trace", "session_id": "session-7125"}, "explicit-trace"),
|
|
(
|
|
{"x-vendor-session-id": "short"},
|
|
{"trace_id": "short", "session_id": "short"},
|
|
"short",
|
|
),
|
|
(
|
|
{"x-session-id": "invalid value"},
|
|
{"trace_id": "invalid value", "session_id": "invalid value"},
|
|
"invalid value",
|
|
),
|
|
(
|
|
{"session-id": "session-7125", "user-agent": "codexfoo/1.0"},
|
|
{"trace_id": "session-7125", "session_id": "session-7125"},
|
|
"session-7125",
|
|
),
|
|
(
|
|
{"x-vendor-session-id": "short"},
|
|
{"trace_id": "session-7125", "session_id": "session-7125"},
|
|
"session-7125",
|
|
),
|
|
({}, {}, "call"),
|
|
],
|
|
)
|
|
def test_session_header_trace_provenance(headers, metadata, expected_id, level):
|
|
from starlette.datastructures import Headers
|
|
|
|
from litellm.proxy.litellm_pre_call_utils import (
|
|
LiteLLMProxyRequestSetup,
|
|
clean_headers,
|
|
redact_credential_headers,
|
|
)
|
|
|
|
logger, exporter = _steering_logger()
|
|
for turn in range(2):
|
|
exporter.clear()
|
|
call_id = f"call-{turn}"
|
|
request_headers = Headers(headers)
|
|
data = LiteLLMProxyRequestSetup.add_litellm_metadata_from_request_headers(
|
|
headers=request_headers, data={"metadata": dict(metadata)}, _metadata_variable_name="metadata"
|
|
)
|
|
original_metadata = dict(data["metadata"])
|
|
now = datetime.datetime.now()
|
|
result = logger.log_event_on_langfuse(
|
|
kwargs={
|
|
"call_type": "completion",
|
|
"litellm_call_id": call_id,
|
|
"litellm_trace_id": data.get("litellm_trace_id"),
|
|
"litellm_params": {
|
|
"metadata": data["metadata"],
|
|
"proxy_server_request": {"headers": redact_credential_headers(clean_headers(request_headers))},
|
|
},
|
|
"messages": [{"role": "user", "content": f"turn {turn}"}],
|
|
"optional_params": {},
|
|
},
|
|
response_obj=(
|
|
None
|
|
if level == "ERROR"
|
|
else litellm.ModelResponse(choices=[{"message": {"role": "assistant", "content": "OK"}}])
|
|
),
|
|
start_time=now,
|
|
end_time=now,
|
|
level=level,
|
|
status_message="provider error" if level == "ERROR" else None,
|
|
)
|
|
span = _exported_span(logger, exporter)
|
|
assert _span_trace_id(span) == resolve_trace_id(call_id if expected_id == "call" else expected_id)
|
|
assert result["trace_id"] == _span_trace_id(span)
|
|
if expected_id != "existing-trace":
|
|
assert span.attributes.get("session.id") == headers.get(
|
|
"langfuse_session_id", original_metadata.get("session_id")
|
|
)
|
|
steering = {key[len("langfuse_") :]: value for key, value in headers.items() if key.startswith("langfuse_")}
|
|
assert data["metadata"] == {**original_metadata, **steering}
|
|
|
|
|
|
def test_session_header_trace_without_call_id_keeps_session_alias():
|
|
logger, exporter = _steering_logger()
|
|
now: Final = datetime.datetime.now()
|
|
|
|
result: Final = logger.log_event_on_langfuse(
|
|
kwargs={
|
|
"call_type": "completion",
|
|
"litellm_call_id": "",
|
|
"litellm_params": {
|
|
"metadata": {"trace_id": "session-7125", "session_id": "session-7125"},
|
|
"proxy_server_request": {"headers": {"x-litellm-session-id": "session-7125"}},
|
|
},
|
|
"messages": [{"role": "user", "content": "no call id"}],
|
|
"optional_params": {},
|
|
},
|
|
response_obj=litellm.ModelResponse(choices=[{"message": {"role": "assistant", "content": "OK"}}]),
|
|
start_time=now,
|
|
end_time=now,
|
|
)
|
|
|
|
assert _span_trace_id(_exported_span(logger, exporter)) == resolve_trace_id("session-7125")
|
|
assert result["trace_id"] == resolve_trace_id("session-7125")
|
|
|
|
|
|
def test_every_proxy_session_header_shape_is_classified_as_a_session_alias():
|
|
"""The classifier must cover every header shape the proxy turns into a chain id."""
|
|
from litellm.integrations.langfuse.langfuse import _is_session_header_trace
|
|
from litellm.proxy.litellm_pre_call_utils import (
|
|
_CODEX_SESSION_ID_HEADERS,
|
|
get_chain_id_from_headers,
|
|
)
|
|
|
|
session: Final = "session-7125-abcdef"
|
|
session_shapes: Final = (
|
|
{"x-litellm-session-id": session},
|
|
{"X-Claude-Code-Session-Id": session},
|
|
{"x-session-id": session},
|
|
*({header: session, "user-agent": "codex_cli_rs/1.0"} for header in _CODEX_SESSION_ID_HEADERS),
|
|
)
|
|
for headers in session_shapes:
|
|
assert get_chain_id_from_headers(dict(headers)) == session, headers
|
|
assert _is_session_header_trace(session, session, {"headers": headers}) is True, headers
|
|
|
|
explicit_trace: Final = {"x-litellm-trace-id": session, "x-litellm-session-id": session}
|
|
assert get_chain_id_from_headers(dict(explicit_trace)) == session
|
|
assert _is_session_header_trace(session, session, {"headers": explicit_trace}) is False
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"proxy_server_request",
|
|
[None, {}, {"headers": None}],
|
|
ids=["no-proxy-request", "no-headers-key", "null-headers"],
|
|
)
|
|
def test_sdk_caller_without_request_headers_keeps_its_trace(proxy_server_request):
|
|
"""A direct SDK caller has no request headers, so a session-shaped trace id stays the caller's."""
|
|
logger, exporter = _steering_logger()
|
|
now: Final = datetime.datetime.now()
|
|
|
|
result: Final = logger.log_event_on_langfuse(
|
|
kwargs={
|
|
"call_type": "completion",
|
|
"litellm_call_id": "call-0",
|
|
"litellm_params": {
|
|
"metadata": {"trace_id": "session-7125", "session_id": "session-7125"},
|
|
"proxy_server_request": proxy_server_request,
|
|
},
|
|
"messages": [{"role": "user", "content": "sdk turn"}],
|
|
"optional_params": {},
|
|
},
|
|
response_obj=litellm.ModelResponse(choices=[{"message": {"role": "assistant", "content": "OK"}}]),
|
|
start_time=now,
|
|
end_time=now,
|
|
)
|
|
|
|
assert _span_trace_id(_exported_span(logger, exporter)) == resolve_trace_id("session-7125")
|
|
assert result["trace_id"] == resolve_trace_id("session-7125")
|
|
|
|
|
|
def test_session_header_classifier_survives_non_string_header_keys():
|
|
"""A non-string header key must not cost the caller its whole trace."""
|
|
from litellm.integrations.langfuse.langfuse import _is_session_header_trace
|
|
|
|
session: Final = "session-7125-abcdef"
|
|
headers: Final = {7: "numeric key", "x-litellm-session-id": session}
|
|
assert _is_session_header_trace(session, session, {"headers": headers}) is True
|
|
assert _is_session_header_trace(session, session, {"headers": {7: "numeric key"}}) is False
|
|
|
|
|
|
def test_mask_input_header_false_keeps_the_prompt():
|
|
rig = _steering_logger()
|
|
|
|
trace_params, generation_params, _ = _emit(rig, headers={"langfuse_mask_input": "false"})
|
|
|
|
assert "input" not in trace_params
|
|
assert json.loads(generation_params["input"]) == {"messages": [{"role": "user", "content": "the-input"}]}
|
|
|
|
|
|
def test_mask_input_header_true_redacts_the_prompt():
|
|
rig = _steering_logger()
|
|
|
|
trace_params, generation_params, _ = _emit(rig, headers={"langfuse_mask_input": "true"})
|
|
|
|
assert "input" not in trace_params
|
|
assert generation_params["input"] == _LANGFUSE_REDACTED
|
|
|
|
|
|
def test_mask_output_header_false_keeps_the_completion():
|
|
rig = _steering_logger()
|
|
|
|
trace_params, generation_params, _ = _emit(rig, headers={"langfuse_mask_output": "false"})
|
|
|
|
assert "output" not in trace_params
|
|
assert "the-output" in generation_params["output"]
|
|
|
|
|
|
def test_mask_output_header_true_redacts_the_completion():
|
|
rig = _steering_logger()
|
|
|
|
trace_params, generation_params, _ = _emit(rig, headers={"langfuse_mask_output": "true"})
|
|
|
|
assert "output" not in trace_params
|
|
assert generation_params["output"] == _LANGFUSE_REDACTED
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"mask_input, expect_redacted",
|
|
[
|
|
(False, False),
|
|
(True, True),
|
|
# An unrecognised string keeps its truthiness, so existing behaviour is unchanged
|
|
("yes", True),
|
|
],
|
|
)
|
|
def test_mask_input_from_the_request_body_is_unchanged(mask_input, expect_redacted):
|
|
rig = _steering_logger()
|
|
|
|
_, generation_params, _ = _emit(rig, metadata={"mask_input": mask_input})
|
|
|
|
assert (generation_params["input"] == _LANGFUSE_REDACTED) is expect_redacted
|
|
|
|
|
|
@pytest.mark.parametrize("flag", [True, "true"])
|
|
def test_update_trace_keys_header_applies_every_key_when_enabled(flag, monkeypatch):
|
|
rig = _steering_logger()
|
|
|
|
monkeypatch.setattr(litellm, "langfuse_enable_update_trace_keys", flag)
|
|
trace_params, _, span = _emit(
|
|
rig,
|
|
headers={
|
|
"langfuse_existing_trace_id": "trace-1",
|
|
"langfuse_update_trace_keys": "trace_release, trace_tail",
|
|
"langfuse_trace_release": "v1.2.3",
|
|
"langfuse_trace_tail": "last",
|
|
},
|
|
)
|
|
|
|
assert trace_params["release"] == "v1.2.3"
|
|
assert span.attributes["langfuse.release"] == "v1.2.3"
|
|
assert not [key for key in span.attributes if key.endswith("tail")]
|
|
|
|
|
|
def test_update_trace_keys_is_off_by_default():
|
|
"""
|
|
The caller picks the key name, so while the feature is on they can name
|
|
user_api_key_auth and have the resolved auth object, including team callback
|
|
credentials, serialized onto the trace. It stays inert until an operator opts in.
|
|
"""
|
|
rig = _steering_logger()
|
|
|
|
trace_params, _, span = _emit(
|
|
rig,
|
|
metadata={
|
|
"existing_trace_id": "trace-1",
|
|
"update_trace_keys": ["user_api_key_auth", "trace_release"],
|
|
"user_api_key_auth": {"team_metadata": {"logging": [{"callback_vars": {"secret": "sk-canary"}}]}},
|
|
"trace_release": "v1.2.3",
|
|
},
|
|
)
|
|
|
|
assert "user_api_key_auth" not in trace_params
|
|
assert "release" not in trace_params
|
|
assert "sk-canary" not in json.dumps(dict(span.attributes or {}), default=repr)
|
|
|
|
|
|
def test_update_trace_keys_input_and_output_are_gated_too(monkeypatch):
|
|
rig = _steering_logger()
|
|
|
|
off, _, _ = _emit(rig, metadata={"existing_trace_id": "trace-1", "update_trace_keys": ["input", "output"]})
|
|
monkeypatch.setattr(litellm, "langfuse_enable_update_trace_keys", True)
|
|
on, _, _ = _emit(rig, metadata={"existing_trace_id": "trace-1", "update_trace_keys": ["input", "output"]})
|
|
|
|
assert "input" not in off and "output" not in off
|
|
assert "input" in on and "output" in on
|
|
|
|
|
|
def test_update_trace_keys_input_output_reach_the_trace_even_under_a_parent(monkeypatch):
|
|
"""With a real parent the generation is not the trace root, so trace-level
|
|
I/O must be stamped explicitly; v2 updated the trace object directly."""
|
|
rig = _steering_logger()
|
|
|
|
monkeypatch.setattr(litellm, "langfuse_enable_update_trace_keys", True)
|
|
_, _, span = _emit(
|
|
rig,
|
|
metadata={
|
|
"existing_trace_id": "trace-1",
|
|
"parent_observation_id": "b" * 16,
|
|
"update_trace_keys": ["input", "output"],
|
|
},
|
|
)
|
|
|
|
assert "the-input" in str(span.attributes["langfuse.trace.input"])
|
|
assert "the-output" in str(span.attributes["langfuse.trace.output"])
|
|
|
|
|
|
def test_a_fresh_trace_under_a_callers_parent_still_carries_its_own_input_and_output():
|
|
"""Langfuse copies I/O onto a trace only from its root observation; a caller's ``parent_observation_id``
|
|
makes the generation a child, so the trace-level fields v2 set on ``trace(...)`` must be stamped."""
|
|
rig = _steering_logger()
|
|
|
|
_, _, span = _emit(rig, metadata={"parent_observation_id": "0123456789abcdef"})
|
|
|
|
assert span.parent is not None
|
|
assert "the-input" in str(span.attributes["langfuse.trace.input"])
|
|
assert "the-output" in str(span.attributes["langfuse.trace.output"])
|
|
|
|
|
|
def test_a_failed_call_under_a_callers_parent_stamps_the_error_as_the_trace_output():
|
|
"""The ERROR branch used to write a trace-level ``status_message``, a field the v4 trace schema does not
|
|
have, and skip ``output``; the generation's parent is the caller's, so nothing else fills the trace."""
|
|
logger, exporter = _steering_logger()
|
|
now = datetime.datetime.now()
|
|
|
|
logger.log_event_on_langfuse(
|
|
kwargs={
|
|
"call_type": "completion",
|
|
"litellm_params": {"metadata": {"parent_observation_id": "0123456789abcdef"}},
|
|
"messages": [{"role": "user", "content": "the-input"}],
|
|
"optional_params": {},
|
|
},
|
|
response_obj=None,
|
|
start_time=now,
|
|
end_time=now,
|
|
level="ERROR",
|
|
status_message="provider said no",
|
|
)
|
|
span = _exported_span(logger, exporter)
|
|
|
|
assert span.parent is not None
|
|
assert "provider said no" in str(span.attributes["langfuse.trace.output"])
|
|
assert span.attributes["langfuse.observation.status_message"] == "provider said no"
|
|
|
|
|
|
def test_a_fresh_trace_root_leaves_the_duplicate_io_to_langfuse():
|
|
rig = _steering_logger()
|
|
|
|
_, _, span = _emit(rig, metadata={"trace_id": "a" * 32})
|
|
|
|
assert span.parent is None
|
|
assert "langfuse.trace.input" not in (span.attributes or {})
|
|
assert "the-input" in str(span.attributes["langfuse.observation.input"])
|
|
|
|
|
|
def test_existing_trace_id_appends_without_claiming_trace_root():
|
|
"""Langfuse copies a root observation's name and I/O onto the trace, so a
|
|
continuation that claimed root would rename the trace after every request;
|
|
v2 only ever touched the keys in ``update_trace_keys``."""
|
|
rig = _steering_logger()
|
|
|
|
_, _, span = _emit(rig, metadata={"existing_trace_id": "trace-1", "trace_name": "second-call"})
|
|
|
|
assert span.parent is not None
|
|
assert "langfuse.trace.name" not in (span.attributes or {})
|
|
|
|
|
|
def test_a_fresh_trace_still_claims_root_so_its_generation_names_it():
|
|
rig = _steering_logger()
|
|
|
|
_, _, span = _emit(rig, metadata={"trace_id": "a" * 32, "trace_name": "first-call"})
|
|
|
|
assert span.parent is None
|
|
assert span.attributes["langfuse.trace.name"] == "first-call"
|
|
|
|
|
|
def test_trace_io_is_not_stamped_when_update_trace_keys_does_not_ask(monkeypatch):
|
|
rig = _steering_logger()
|
|
|
|
monkeypatch.setattr(litellm, "langfuse_enable_update_trace_keys", True)
|
|
_, _, span = _emit(
|
|
rig,
|
|
metadata={
|
|
"existing_trace_id": "trace-1",
|
|
"parent_observation_id": "b" * 16,
|
|
"update_trace_keys": ["trace_release"],
|
|
},
|
|
)
|
|
|
|
assert "langfuse.trace.input" not in (span.attributes or {})
|
|
assert "langfuse.trace.output" not in (span.attributes or {})
|
|
|
|
|
|
def test_update_trace_keys_from_the_request_body_list_applies_when_enabled(monkeypatch):
|
|
rig = _steering_logger()
|
|
|
|
monkeypatch.setattr(litellm, "langfuse_enable_update_trace_keys", True)
|
|
trace_params, _, span = _emit(
|
|
rig,
|
|
metadata={
|
|
"existing_trace_id": "trace-1",
|
|
"update_trace_keys": ["trace_release"],
|
|
"trace_release": "v1.2.3",
|
|
},
|
|
)
|
|
|
|
assert trace_params["release"] == "v1.2.3"
|
|
assert span.attributes["langfuse.release"] == "v1.2.3"
|
|
|
|
|
|
def test_update_trace_keys_trace_metadata_reaches_the_trace_and_stays_off_the_generation(monkeypatch):
|
|
rig = _steering_logger()
|
|
|
|
monkeypatch.setattr(litellm, "langfuse_enable_update_trace_keys", True)
|
|
_, _, span = _emit(
|
|
rig,
|
|
metadata={
|
|
"existing_trace_id": "trace-1",
|
|
"parent_observation_id": "b" * 16,
|
|
"update_trace_keys": ["trace_metadata"],
|
|
"trace_metadata": {"step": 2, "note": "x" * 300},
|
|
},
|
|
)
|
|
|
|
assert span.attributes["langfuse.trace.metadata.step"] == 2
|
|
assert span.attributes["langfuse.trace.metadata.note"] == "x" * 300
|
|
assert "langfuse.observation.metadata.step" not in span.attributes
|
|
|
|
|
|
def test_non_mapping_trace_metadata_does_not_lose_the_event():
|
|
"""A caller who passes ``trace_metadata`` as a string still gets a generation, and the string is not spread."""
|
|
rig = _steering_logger()
|
|
|
|
trace_params, generation_params, span = _emit(rig, metadata={"trace_metadata": "just-a-note"})
|
|
|
|
assert json.loads(generation_params["output"])["content"] == "the-output"
|
|
assert trace_params["name"] == "litellm-completion"
|
|
assert not any(key.startswith("langfuse.trace.metadata.") for key in span.attributes or {})
|
|
|
|
|
|
def test_trace_metadata_is_not_propagated_when_absent():
|
|
rig = _steering_logger()
|
|
|
|
_, _, span = _emit(rig, metadata={"trace_name": "plain"})
|
|
|
|
assert not any(key.startswith("langfuse.trace.metadata.") for key in span.attributes or {})
|
|
|
|
|
|
def test_update_trace_keys_matches_whole_keys_not_substrings():
|
|
rig = _steering_logger()
|
|
|
|
trace_params, _, _ = _emit(
|
|
rig,
|
|
headers={"langfuse_existing_trace_id": "trace-1", "langfuse_update_trace_keys": "my_input"},
|
|
)
|
|
|
|
assert "input" not in trace_params
|
|
|
|
|
|
def test_langfuse_environment_is_coerced_and_validated(monkeypatch):
|
|
monkeypatch.delenv("LANGFUSE_TRACING_ENVIRONMENT", raising=False)
|
|
logger = _build_langfuse_logger(monkeypatch, langfuse_public_key="pk-env", langfuse_environment=123)
|
|
assert logger.langfuse_environment == "123"
|
|
|
|
with pytest.raises(ValueError, match="langfuse_environment"):
|
|
_build_langfuse_logger(monkeypatch, langfuse_public_key="pk-env", langfuse_environment="Production")
|
|
|
|
|
|
def test_langfuse_empty_environment_falls_back_and_is_not_dynamic(monkeypatch):
|
|
from litellm.integrations.langfuse.langfuse_handler import LangFuseHandler
|
|
from litellm.types.utils import StandardCallbackDynamicParams
|
|
|
|
monkeypatch.setenv("LANGFUSE_TRACING_ENVIRONMENT", "production")
|
|
|
|
# '' falls back to the deployment env var at init
|
|
logger = _build_langfuse_logger(monkeypatch, langfuse_public_key="pk-env", langfuse_environment="")
|
|
assert logger.langfuse_environment == "production"
|
|
|
|
# env-only params that add nothing do not select a dynamic logger
|
|
for redundant in ["", " ", "production"]:
|
|
params = StandardCallbackDynamicParams(langfuse_environment=redundant)
|
|
assert LangFuseHandler._dynamic_langfuse_credentials_are_passed(params) is False
|
|
|
|
params = StandardCallbackDynamicParams(langfuse_environment="team-a-prod")
|
|
assert LangFuseHandler._dynamic_langfuse_credentials_are_passed(params) is True
|
|
|
|
# a dynamic value equal to the logger's effective (stripped) environment is redundant
|
|
monkeypatch.setenv("LANGFUSE_TRACING_ENVIRONMENT", "production ")
|
|
stripped_redundant_params: Final = StandardCallbackDynamicParams(langfuse_environment="production")
|
|
assert LangFuseHandler._dynamic_langfuse_credentials_are_passed(stripped_redundant_params) is False
|
|
|
|
# a dynamic value repeating the raw (even invalid) deployment value is redundant, not an override
|
|
monkeypatch.setenv("LANGFUSE_TRACING_ENVIRONMENT", "Production")
|
|
raw_redundant_params: Final = StandardCallbackDynamicParams(langfuse_environment="Production")
|
|
assert LangFuseHandler._dynamic_langfuse_credentials_are_passed(raw_redundant_params) is False
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("env_value", "expected"),
|
|
(
|
|
("Production", "default"),
|
|
("EU-Prod", "default"),
|
|
("langfuse-prod", "default"),
|
|
(" ", "default"),
|
|
("production ", "production"),
|
|
("prod", "prod"),
|
|
),
|
|
)
|
|
def test_langfuse_deployment_environment_fallback_never_raises(monkeypatch, env_value, expected):
|
|
monkeypatch.setenv("LANGFUSE_MOCK", "true")
|
|
monkeypatch.setenv("LANGFUSE_TRACING_ENVIRONMENT", env_value)
|
|
monkeypatch.setattr(litellm, "initialized_langfuse_clients", 0)
|
|
logger: Final = LangFuseLogger(
|
|
langfuse_public_key="pk-env",
|
|
langfuse_secret="sk-env",
|
|
langfuse_host="https://test.langfuse.com",
|
|
)
|
|
assert logger.langfuse_environment == expected
|
|
|
|
|
|
def test_continued_trace_keeps_the_generation_version():
|
|
"""v2 set ``version`` on the generation even when the trace was not being updated."""
|
|
rig = _steering_logger()
|
|
|
|
_, _, span = _emit(rig, metadata={"existing_trace_id": "b" * 32, "version": "gen-7"})
|
|
|
|
assert span.attributes["langfuse.version"] == "gen-7"
|
|
|
|
|
|
def test_new_trace_version_takes_precedence_over_the_generation_version():
|
|
"""v4 has one ``langfuse.version`` per span, so unlike v2's separate trace and generation fields only one
|
|
value can survive; ``trace_version`` wins, matching the v4 SDK, whose propagated attributes overwrite a span's own."""
|
|
rig = _steering_logger()
|
|
|
|
captured_trace_params, _, span = _emit(rig, metadata={"trace_version": "trace-1", "version": "gen-7"})
|
|
|
|
assert captured_trace_params["version"] == "trace-1"
|
|
assert span.attributes["langfuse.version"] == "trace-1"
|
|
|
|
|
|
def test_log_event_returns_the_v2_dict_shape_for_the_alerting_trace_id_cache():
|
|
"""litellm_logging only caches the langfuse trace id off a dict with a ``trace_id`` key.
|
|
|
|
Slack alerting builds its trace URL from that cache, so a different return
|
|
shape silently breaks alert links.
|
|
"""
|
|
rig = _steering_logger()
|
|
logger, _ = rig
|
|
|
|
returned = logger.log_event_on_langfuse(
|
|
kwargs={
|
|
"call_type": "completion",
|
|
"litellm_params": {"metadata": {"trace_id": "c" * 32}},
|
|
"messages": [{"role": "user", "content": "the-input"}],
|
|
"optional_params": {},
|
|
},
|
|
response_obj=litellm.ModelResponse(choices=[{"message": {"role": "assistant", "content": "the-output"}}]),
|
|
start_time=datetime.datetime.now(),
|
|
end_time=datetime.datetime.now(),
|
|
)
|
|
|
|
assert isinstance(returned, dict)
|
|
assert returned["trace_id"] == "c" * 32
|
|
assert returned["generation_id"]
|
|
|
|
|
|
def test_parse_langfuse_debug_only_enables_on_true_strings():
|
|
"""v4 treats any truthy value as debug=on, so the raw env string "false" would enable debug."""
|
|
assert langfuse_module.parse_langfuse_debug("true") is True
|
|
assert langfuse_module.parse_langfuse_debug("True") is True
|
|
assert langfuse_module.parse_langfuse_debug("1") is True
|
|
assert langfuse_module.parse_langfuse_debug("false") is False
|
|
assert langfuse_module.parse_langfuse_debug("False") is False
|
|
assert langfuse_module.parse_langfuse_debug("") is False
|
|
assert langfuse_module.parse_langfuse_debug(None) is False
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("raw", "expected"),
|
|
[(None, 1), ("", 1), ("3", 3), ("0", 1), ("-5", 1), ("abc", 1)],
|
|
ids=["unset", "empty", "valid", "zero", "negative", "text"],
|
|
)
|
|
def test_flush_interval_env_falls_back_instead_of_failing_the_first_request(monkeypatch, raw, expected, caplog):
|
|
"""The batch scheduler rejects a non-positive delay; v2's consumer thread accepted 0, so the value must
|
|
not raise out of the lazily built logger and take Langfuse logging down for the worker."""
|
|
if raw is None:
|
|
monkeypatch.delenv("LANGFUSE_FLUSH_INTERVAL", raising=False)
|
|
else:
|
|
monkeypatch.setenv("LANGFUSE_FLUSH_INTERVAL", raw)
|
|
with caplog.at_level(logging.WARNING, logger="LiteLLM"):
|
|
assert LangFuseLogger._get_langfuse_flush_interval(1) == expected # pyright: ignore[reportPrivateUsage] # the parser under test
|
|
assert ("LANGFUSE_FLUSH_INTERVAL" in caplog.text) is (raw in ("0", "-5", "abc"))
|
|
|
|
|
|
def test_zero_flush_interval_still_builds_a_working_export_channel(monkeypatch):
|
|
receiver = _OtlpReceiver()
|
|
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-flush-zero-test")
|
|
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-flush-zero-test")
|
|
monkeypatch.delenv("LANGFUSE_MOCK", raising=False)
|
|
monkeypatch.setenv("LANGFUSE_FLUSH_INTERVAL", "0")
|
|
monkeypatch.setattr(litellm, "initialized_langfuse_clients", litellm.initialized_langfuse_clients)
|
|
|
|
try:
|
|
logger = LangFuseLogger(langfuse_host=receiver.url)
|
|
_log_one_completion(logger)
|
|
finally:
|
|
receiver.close()
|
|
|
|
assert receiver.received == ["/api/public/otel/v1/traces"]
|
|
|
|
|
|
def test_langfuse_debug_env_string_false_stays_off(monkeypatch):
|
|
"""LANGFUSE_DEBUG=false must not reach the v4 client as a truthy string.
|
|
|
|
The v4 client does ``if debug:`` and then mutates root logging via
|
|
``logging.basicConfig``, so the unparsed string "false" turns debug ON.
|
|
"""
|
|
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-debug-parse-test")
|
|
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-debug-parse-test")
|
|
monkeypatch.setenv("LANGFUSE_MOCK", "true")
|
|
monkeypatch.setenv("LANGFUSE_DEBUG", "false")
|
|
monkeypatch.setattr(litellm, "initialized_langfuse_clients", litellm.initialized_langfuse_clients)
|
|
|
|
assert LangFuseLogger().langfuse_debug is False
|
|
|
|
|
|
def test_langfuse_debug_env_true_turns_on_the_langfuse_logger(monkeypatch):
|
|
"""``LANGFUSE_DEBUG=true`` reached the v2 client as ``debug=`` and switched the SDK's logger to DEBUG;
|
|
a parsed flag that nothing reads would make the variable a silent no-op."""
|
|
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-debug-wire-test")
|
|
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-debug-wire-test")
|
|
monkeypatch.setenv("LANGFUSE_MOCK", "true")
|
|
monkeypatch.setenv("LANGFUSE_DEBUG", "true")
|
|
monkeypatch.setattr(litellm, "initialized_langfuse_clients", litellm.initialized_langfuse_clients)
|
|
langfuse_logger = logging.getLogger("langfuse")
|
|
level_before = langfuse_logger.level
|
|
langfuse_logger.setLevel(logging.WARNING)
|
|
try:
|
|
assert LangFuseLogger().langfuse_debug is True
|
|
assert langfuse_logger.level == logging.DEBUG
|
|
finally:
|
|
langfuse_logger.setLevel(level_before)
|
|
|
|
|
|
def test_explicit_langfuse_host_beats_the_v4_base_url_env(monkeypatch):
|
|
"""Per-key/per-team ``langfuse_host`` must win over LANGFUSE_BASE_URL.
|
|
|
|
v4 resolves ``base_url or $LANGFUSE_BASE_URL or host``, so a stray env var
|
|
could silently redirect every tenant's traces to one server. The proof is a
|
|
real round trip: the observation lands on the configured host.
|
|
"""
|
|
receiver = _OtlpReceiver()
|
|
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-base-url-test")
|
|
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-base-url-test")
|
|
monkeypatch.delenv("LANGFUSE_MOCK", raising=False)
|
|
monkeypatch.setenv("LANGFUSE_BASE_URL", "http://127.0.0.1:1")
|
|
monkeypatch.setenv("LANGFUSE_FLUSH_INTERVAL", "1")
|
|
monkeypatch.setattr(litellm, "initialized_langfuse_clients", litellm.initialized_langfuse_clients)
|
|
|
|
try:
|
|
logger = LangFuseLogger(langfuse_host=receiver.url)
|
|
_log_one_completion(logger)
|
|
finally:
|
|
receiver.close()
|
|
|
|
assert logger.langfuse_host == receiver.url
|
|
assert receiver.received == ["/api/public/otel/v1/traces"]
|
|
|
|
|
|
def test_resolve_credentials_falls_back_to_langfuse_base_url(monkeypatch):
|
|
"""v4's canonical env var works when LANGFUSE_HOST is unset, but never beats it."""
|
|
monkeypatch.setenv("LANGFUSE_BASE_URL", "https://from-base-url.example")
|
|
monkeypatch.delenv("LANGFUSE_HOST", raising=False)
|
|
|
|
_, _, host = langfuse_module.resolve_langfuse_credentials()
|
|
assert host == "https://from-base-url.example"
|
|
|
|
monkeypatch.setenv("LANGFUSE_HOST", "https://from-host.example")
|
|
_, _, host = langfuse_module.resolve_langfuse_credentials()
|
|
assert host == "https://from-host.example"
|
|
|
|
_, _, host = langfuse_module.resolve_langfuse_credentials(langfuse_host="https://explicit.example")
|
|
assert host == "https://explicit.example"
|
|
|
|
|
|
def test_version_gate_rejects_v5_prereleases():
|
|
""" "5.0.0rc1" sorts below "5", so a plain version comparison would admit it."""
|
|
langfuse_module.raise_if_unsupported_langfuse_version("4.7")
|
|
with pytest.raises(ImportError):
|
|
langfuse_module.raise_if_unsupported_langfuse_version("5.0.0rc1")
|
|
with pytest.raises(ImportError):
|
|
langfuse_module.raise_if_unsupported_langfuse_version("5.0.0")
|
|
|
|
|
|
def test_old_sdk_fails_with_the_upgrade_message_before_the_otel_module_is_imported(monkeypatch):
|
|
"""On a v2 install `langfuse_sdk` itself fails to import, so the version gate must run first
|
|
or the caller is told the package is missing when it only needs upgrading."""
|
|
import sys
|
|
|
|
monkeypatch.setattr(langfuse_module, "installed_langfuse_version", lambda: "2.59.7")
|
|
monkeypatch.setitem(sys.modules, "litellm.integrations.langfuse.langfuse_sdk", None)
|
|
|
|
with pytest.raises(ImportError) as raised:
|
|
_build_langfuse_logger(monkeypatch, langfuse_public_key="pk-old-sdk")
|
|
|
|
assert "2.59.7" in str(raised.value)
|
|
assert "langfuse_otel" in str(raised.value)
|
|
assert "not installed" not in str(raised.value)
|
|
|
|
|
|
def test_missing_sdk_is_reported_as_not_installed(monkeypatch):
|
|
from importlib.metadata import PackageNotFoundError
|
|
|
|
def not_installed() -> str:
|
|
raise PackageNotFoundError("langfuse")
|
|
|
|
monkeypatch.setattr(langfuse_module, "installed_langfuse_version", not_installed)
|
|
|
|
with pytest.raises(Exception, match="Langfuse not installed"):
|
|
_build_langfuse_logger(monkeypatch, langfuse_public_key="pk-no-sdk")
|
|
|
|
|
|
@pytest.mark.parametrize("raw", ["abc", "2.5", ""], ids=["text", "fraction", "empty"])
|
|
def test_prompt_cache_ttl_typo_is_named_before_the_sdk_is_imported(monkeypatch, raw):
|
|
"""The v4 SDK evaluates ``int(LANGFUSE_PROMPT_CACHE_DEFAULT_TTL_SECONDS)`` at import, so without this
|
|
gate every request failed with a bare ``invalid literal for int()`` that never named the variable."""
|
|
import sys
|
|
|
|
monkeypatch.setenv("LANGFUSE_PROMPT_CACHE_DEFAULT_TTL_SECONDS", raw)
|
|
monkeypatch.setitem(sys.modules, "litellm.integrations.langfuse.langfuse_sdk", None)
|
|
|
|
with pytest.raises(ValueError, match="LANGFUSE_PROMPT_CACHE_DEFAULT_TTL_SECONDS") as raised:
|
|
_build_langfuse_logger(monkeypatch, langfuse_public_key="pk-ttl-typo")
|
|
|
|
assert repr(raw) in str(raised.value)
|
|
|
|
|
|
@pytest.mark.parametrize("raw", ["5", " -3 ", "+0"], ids=["whole", "negative", "signed-zero"])
|
|
def test_whole_second_prompt_cache_ttl_passes_the_gate(monkeypatch, raw):
|
|
monkeypatch.setenv("LANGFUSE_PROMPT_CACHE_DEFAULT_TTL_SECONDS", raw)
|
|
assert langfuse_module.raise_if_unusable_prompt_cache_ttl() is None
|
|
|
|
|
|
def test_stopped_logger_hands_its_export_channel_back(monkeypatch):
|
|
"""`DynamicLoggingCache` calls `stop()` on expiry; the channel must be retired once every
|
|
logger that held it has stopped, or each credential rotation leaks a batch export thread."""
|
|
from litellm.integrations.langfuse.langfuse_sdk import acquire_langfuse_tracing, release_langfuse_tracing
|
|
|
|
logger = _build_langfuse_logger(monkeypatch, langfuse_public_key="pk-stop-releases")
|
|
|
|
def acquire_same_credentials():
|
|
return acquire_langfuse_tracing(
|
|
public_key="pk-stop-releases",
|
|
secret_key="sk-lit5228",
|
|
base_url=_UNREACHABLE_HOST,
|
|
environment=logger.langfuse_environment,
|
|
release=logger.langfuse_release,
|
|
flush_interval=logger.langfuse_flush_interval,
|
|
mock_mode=False,
|
|
)
|
|
|
|
logger.stop()
|
|
reacquired = acquire_same_credentials()
|
|
assert reacquired is logger.tracing, "the channel stays up while another logger still holds it"
|
|
|
|
release_langfuse_tracing(reacquired, grace_seconds=0.0)
|
|
assert acquire_same_credentials() is not logger.tracing, "stop() did not give the logger's hold back"
|
|
|
|
|
|
def test_logger_that_fails_to_build_takes_no_slot_and_no_channel(monkeypatch):
|
|
"""Each failed retry for the same dynamic credentials would otherwise eat a client slot and a
|
|
holder on the channel, so fixing the configuration could not bring Langfuse logging back."""
|
|
from litellm.integrations.langfuse.langfuse_sdk import acquire_langfuse_tracing, release_langfuse_tracing
|
|
|
|
monkeypatch.setenv("LANGFUSE_TIMEOUT", "5.5")
|
|
probe = _build_langfuse_logger(monkeypatch, langfuse_public_key="pk-failed-build")
|
|
|
|
def acquire_same_credentials():
|
|
return acquire_langfuse_tracing(
|
|
public_key="pk-failed-build",
|
|
secret_key="sk-lit5228",
|
|
base_url=_UNREACHABLE_HOST,
|
|
environment=probe.langfuse_environment,
|
|
release=probe.langfuse_release,
|
|
flush_interval=probe.langfuse_flush_interval,
|
|
mock_mode=False,
|
|
)
|
|
|
|
monkeypatch.setenv("LANGFUSE_TIMEOUT", "not-a-number")
|
|
with pytest.raises(ValueError, match="not-a-number"):
|
|
_build_langfuse_logger(monkeypatch, langfuse_public_key="pk-failed-build")
|
|
assert litellm.initialized_langfuse_clients == 0
|
|
|
|
monkeypatch.setenv("LANGFUSE_TIMEOUT", "5.5")
|
|
release_langfuse_tracing(probe.tracing, grace_seconds=0.0)
|
|
assert acquire_same_credentials() is not probe.tracing, "the failed build left a holder on the channel"
|
|
|
|
|
|
def test_int_steering_values_reach_langfuse_as_strings():
|
|
"""Langfuse models user, session and version as strings; v2's pydantic coerced ints for the caller."""
|
|
rig = _steering_logger()
|
|
|
|
_, _, span = _emit(rig, metadata={"trace_user_id": 12345, "session_id": 67, "trace_version": 3})
|
|
|
|
assert span.attributes["user.id"] == "12345"
|
|
assert span.attributes["session.id"] == "67"
|
|
assert span.attributes["langfuse.version"] == "3"
|
|
|
|
|
|
def test_long_steering_values_are_neither_capped_nor_dropped():
|
|
"""v2 sent ids of any length; the SDK's 200 character rule belongs to baggage propagation, which litellm no longer uses."""
|
|
rig = _steering_logger()
|
|
long_user: Final = "u" * 250
|
|
|
|
_, _, span = _emit(rig, metadata={"trace_user_id": long_user})
|
|
|
|
assert span.attributes["user.id"] == long_user
|
|
|
|
|
|
def test_returned_generation_id_names_the_exported_observation():
|
|
"""v4 derives observation ids from the OTel span, so a pre-computed id would name nothing."""
|
|
logger, exporter = _steering_logger()
|
|
|
|
returned = logger.log_event_on_langfuse(
|
|
kwargs={
|
|
"call_type": "completion",
|
|
"litellm_params": {"metadata": {"trace_id": "d" * 32}},
|
|
"messages": [{"role": "user", "content": "the-input"}],
|
|
"optional_params": {},
|
|
},
|
|
response_obj=litellm.ModelResponse(choices=[{"message": {"role": "assistant", "content": "the-output"}}]),
|
|
start_time=datetime.datetime.now(),
|
|
end_time=datetime.datetime.now(),
|
|
)
|
|
|
|
span = _exported_span(logger, exporter)
|
|
assert returned["generation_id"] == format(span.context.span_id, "016x")
|