From b0e5dd3e1f1a71c41d2802d85bebadf67517b8f2 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Wed, 30 Sep 2026 23:10:52 +0000 Subject: [PATCH] fix(langfuse): adapt responses-api usage regression test to langfuse v4 spans The v4 migration exports usage_details as an OTEL span attribute via an in-memory exporter instead of a mocked generation() call. Assert on the exported span's langfuse.observation.usage_details with the same token counts. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../e2e/logging/test_responses_langfuse_usage_e2e.py | 6 ++---- tests/unit/integrations/test_langfuse.py | 12 ++++-------- 2 files changed, 6 insertions(+), 12 deletions(-) diff --git a/tests/e2e/logging/test_responses_langfuse_usage_e2e.py b/tests/e2e/logging/test_responses_langfuse_usage_e2e.py index cf31de24250..37f7414e080 100644 --- a/tests/e2e/logging/test_responses_langfuse_usage_e2e.py +++ b/tests/e2e/logging/test_responses_langfuse_usage_e2e.py @@ -11,8 +11,6 @@ shows 0 input / 0 output while cost is still right. from __future__ import annotations import pytest -from pydantic import BaseModel, ConfigDict, TypeAdapter - from e2e_config import CHEAP_OPENAI_MODEL, unique_marker from lifecycle import ResourceManager from logging_client import ( @@ -27,6 +25,7 @@ from models import ( KeyMetadata, ResponsesApiResponse, ) +from pydantic import BaseModel, ConfigDict, TypeAdapter pytestmark = pytest.mark.e2e @@ -106,6 +105,5 @@ class TestResponsesLangfuseUsage: f"input_tokens {response.usage.input_tokens}" ) assert details.output == response.usage.output_tokens, ( - f"usageDetails.output {details.output} must equal the proxy's " - f"output_tokens {response.usage.output_tokens}" + f"usageDetails.output {details.output} must equal the proxy's output_tokens {response.usage.output_tokens}" ) diff --git a/tests/unit/integrations/test_langfuse.py b/tests/unit/integrations/test_langfuse.py index cbbd3985e5c..f6ed5ece5ca 100644 --- a/tests/unit/integrations/test_langfuse.py +++ b/tests/unit/integrations/test_langfuse.py @@ -5,7 +5,7 @@ import threading import time import types import unittest -from typing import Final, Optional +from typing import Final from unittest.mock import MagicMock, patch import pytest @@ -14,11 +14,10 @@ import litellm from litellm.integrations.langfuse import langfuse as langfuse_module from litellm.integrations.langfuse.langfuse import LangFuseLogger from litellm.integrations.langfuse.langfuse_sdk import resolve_trace_id -from litellm.types.llms.openai import InputTokensDetails, ResponseAPIUsage, ResponsesAPIResponse - # Import LangfuseUsageDetails directly from the module where it's defined from litellm.types.integrations.langfuse import * +from litellm.types.llms.openai import InputTokensDetails, ResponseAPIUsage, ResponsesAPIResponse class TestLangfuseUsageDetails(unittest.TestCase): @@ -370,16 +369,14 @@ class TestLangfuseUsageDetails(unittest.TestCase): except Exception as e: self.fail(f"_log_langfuse_v2 raised an exception: {e}") - usage_details = json.loads( - self.exported_generation().attributes["langfuse.observation.usage_details"] - ) + usage_details = json.loads(self.exported_generation().attributes["langfuse.observation.usage_details"]) # input is reduced by cache_read_input_tokens per Langfuse docs assert usage_details["input"] == 12 assert usage_details["output"] == 21 assert usage_details["total"] == 37 assert usage_details["cache_read_input_tokens"] == 4 - def _build_standard_logging_payload(self, trace_id: Optional[str] = None): + def _build_standard_logging_payload(self, trace_id: str | None = None): payload = { "id": "payload-id", "call_type": "completion", @@ -1464,7 +1461,6 @@ def test_langfuse_rest_client_survives_httpx_cache_eviction(monkeypatch): import weakref from litellm.caching.llm_caching_handler import LLMClientCache - from litellm.llms.custom_httpx.http_handler import _get_httpx_client monkeypatch.setattr(litellm, "in_memory_llm_clients_cache", LLMClientCache())