Merge pull request #32553 from BerriAI/litellm_backport_patch_192rc1

chore(release): backport #32256, #32277, #32236, #32416, #32405, #32524 onto patch-1.92.0rc1
This commit is contained in:
yuneng-jiang 2026-07-08 16:20:09 -07:00 committed by GitHub
commit d2b4f495ea
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
21 changed files with 437 additions and 100 deletions

View file

@ -1,10 +1,10 @@
# syntax=docker/dockerfile:1.7
# Base image for building
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
# Runtime image
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
ARG UI_BUILD_IMAGE=node:20.18-alpine3.20@sha256:3488b10bf958af7125a176419d2d8a9937d895bf124012aae811651988d2ffe6

View file

@ -1,5 +1,5 @@
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
FROM $UV_IMAGE AS uvbin

View file

@ -1,10 +1,10 @@
# syntax=docker/dockerfile:1.7
# Base image for building
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
# Runtime image
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.
ARG UI_BUILD_IMAGE=node:20.18-alpine3.20@sha256:3488b10bf958af7125a176419d2d8a9937d895bf124012aae811651988d2ffe6

View file

@ -1,8 +1,8 @@
# syntax=docker/dockerfile:1.7
# Base images
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG PROXY_EXTRAS_SOURCE=published
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
# Pinned by digest like the other base images; bump explicitly on Node upgrades.

View file

@ -1,5 +1,5 @@
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
FROM $UV_IMAGE AS uvbin

View file

@ -52,6 +52,7 @@ from litellm.integrations.otel.model.semconv import (
GenAIProvider,
JsonRpc,
LiteLLM,
LiteLLMError,
MCPMethod,
Metric,
Network,
@ -87,6 +88,7 @@ __all__ = [
"HTTP",
"JsonRpc",
"LiteLLM",
"LiteLLMError",
"MCP",
"MCPMethod",
"Metric",

View file

@ -16,9 +16,10 @@ from litellm.integrations.otel.model.payloads import (
MCPListToolsSpanData,
MCPToolCallSpanData,
ServiceSpanData,
SpanError,
)
from litellm.integrations.otel.plumbing.providers import to_otel_span_kind
from litellm.integrations.otel.model.semconv import Error, ExceptionEvent
from litellm.integrations.otel.model.semconv import Error, ExceptionEvent, LiteLLMError
from litellm.integrations.otel.model.spans import (
SPAN_REGISTRY,
SpanRole,
@ -49,6 +50,27 @@ _NAME_BUILDERS: dict[SpanRole, Callable[..., str]] = {
_DEDUP_CACHE_MAX = 10_000
def _stamp_otel_error_attributes(span: Span, error_type: str, resolved_message: str) -> None:
"""Stamp the OTel-semconv error attributes (``error.type`` + ``error.message``).
``error_type`` and ``resolved_message`` are ``finish_span``'s already-computed
fallback chains, so the pair on the status, event, and attributes stays in
lockstep."""
span.set_attribute(Error.TYPE, error_type)
span.set_attribute(Error.MESSAGE, resolved_message)
def _stamp_litellm_error_attributes(span: Span, error: SpanError) -> None:
"""Stamp litellm-specific error detail attributes. Emitted only when the
corresponding field is populated so guardrail-shape errors carrying only a
message aren't polluted with empty detail keys."""
if error.code:
span.set_attribute(LiteLLMError.CODE, error.code)
if error.stack_trace:
span.set_attribute(LiteLLMError.STACK_TRACE, error.stack_trace)
if error.llm_provider:
span.set_attribute(LiteLLMError.LLM_PROVIDER, error.llm_provider)
class SpanEmitter:
def __init__(
self,
@ -190,12 +212,13 @@ class SpanEmitter:
if error and (error.error_type or error.message):
error_type = error.error_type or "error"
message = error.message or error.error_type or "error"
span.set_attribute(Error.TYPE, error_type)
_stamp_otel_error_attributes(span, error_type, message)
_stamp_litellm_error_attributes(span, error)
span.set_status(Status(StatusCode.ERROR, message))
# Carry the full message on the standard ``exception`` event so backends
# map it as full text under ``exception.message``. Setting it as a bare
# string attribute instead lets backends like Elasticsearch dynamic-map
# it to a ``keyword`` capped at 1024 chars, truncating the message.
# Also emit the semconv ``exception`` event so backends that
# dynamic-map unknown string span attrs to ``keyword`` (e.g.
# Elasticsearch with a 1024-char ``ignore_above``) still see the
# full untruncated message on the recognized event field.
span.add_event(
ExceptionEvent.NAME,
{ExceptionEvent.TYPE: error_type, ExceptionEvent.MESSAGE: message},

View file

@ -368,7 +368,11 @@ class OpenTelemetryV2(CustomLogger):
# it (named provisionally) so it isn't leaked as an open span.
carrier.span.end(end_time=to_ns(end_time))
return None
data = LLMCallSpanData.from_standard_logging_payload(payload, capture_content=self.config.capture_span_content)
data = LLMCallSpanData.from_standard_logging_payload(
payload,
capture_content=self.config.capture_span_content,
time_to_first_chunk_seconds=call.time_to_first_chunk_seconds,
)
end_time_ns = to_ns(end_time)
if carrier.span is not None:
# Born at the boundary: stamp attributes from the typed payload, set

View file

@ -55,6 +55,7 @@ class GenAIMapper:
GenAI.RESPONSE_MODEL: lambda d: d.response_model,
GenAI.RESPONSE_ID: lambda d: d.response_id,
GenAI.RESPONSE_FINISH_REASONS: lambda d: list(d.finish_reasons) if d.finish_reasons else None,
GenAI.RESPONSE_TIME_TO_FIRST_CHUNK: lambda d: d.time_to_first_chunk_seconds,
GenAI.USAGE_INPUT_TOKENS: lambda d: d.usage.input_tokens,
GenAI.USAGE_OUTPUT_TOKENS: lambda d: d.usage.output_tokens,
Error.TYPE: lambda d: d.error.error_type if d.error else None,

View file

@ -41,7 +41,7 @@ from typing import TYPE_CHECKING, Any, Mapping, cast
from litellm.constants import LITELLM_LOGGING_NO_UPSTREAM_LLM_CALL
from litellm.integrations.otel.model.semconv import resolve_operation
from litellm.integrations.otel.model.utils import as_str
from litellm.integrations.otel.model.utils import as_str, to_seconds
if TYPE_CHECKING:
from litellm.types.utils import StandardLoggingPayload
@ -201,6 +201,7 @@ class LLMCallEvent:
# span is renamed from the typed payload at close (``finish_span``); this only
# needs to be reasonable for a span that never gets closed (a leak).
provisional_span_name: str
time_to_first_chunk_seconds: float | None
@classmethod
def from_dict(cls, kwargs: Mapping[str, Any]) -> "LLMCallEvent":
@ -214,9 +215,25 @@ class LLMCallEvent:
dynamic_params=kwargs.get("standard_callback_dynamic_params"),
is_no_upstream_call=bool(kwargs.get(LITELLM_LOGGING_NO_UPSTREAM_LLM_CALL)),
provisional_span_name=f"{operation.value} {model}".strip(),
time_to_first_chunk_seconds=time_to_first_chunk_seconds(kwargs),
)
def time_to_first_chunk_seconds(kwargs: Mapping[str, Any]) -> float | None:
"""Seconds from the upstream request being issued (``api_call_start_time``)
to the first streamed chunk (``completion_start_time``); ``None`` for
non-streaming calls, where ``completion_start_time`` is backfilled with the
end time and would not measure first-chunk latency."""
optional_params = cast(Mapping[str, Any], kwargs.get("optional_params") or {})
if not optional_params.get("stream"):
return None
api_call_start = to_seconds(kwargs.get("api_call_start_time"))
completion_start = to_seconds(kwargs.get("completion_start_time"))
if api_call_start is None or completion_start is None:
return None
return completion_start - api_call_start
def _call_id(payload: "StandardLoggingPayload | None", kwargs: Mapping[str, Any]) -> str | None:
"""The call id from the payload (when closed) or the bare kwargs (at pre_call)."""
if payload is not None:

View file

@ -141,6 +141,9 @@ class LLMCost:
class SpanError:
error_type: str | None = None
message: str | None = None
code: str | None = None
stack_trace: str | None = None
llm_provider: str | None = None
@dataclass(frozen=True)
@ -305,10 +308,14 @@ class LLMCallSpanData:
messages_in: tuple[Mapping[str, object], ...] = ()
choices_out: tuple[Mapping[str, object], ...] = ()
system_fingerprint: str | None = None
time_to_first_chunk_seconds: float | None = None
@classmethod
def from_standard_logging_payload(
cls, payload: "StandardLoggingPayload", capture_content: bool = False
cls,
payload: "StandardLoggingPayload",
capture_content: bool = False,
time_to_first_chunk_seconds: float | None = None,
) -> "LLMCallSpanData":
params = cast(Mapping[str, object], payload.get("model_parameters") or {})
# The single parse of the request's metadata — the request-vs-provider
@ -349,6 +356,7 @@ class LLMCallSpanData:
messages_in=_dicts(payload.get("messages")) if capture_content else (),
choices_out=choices_out if capture_content else (),
system_fingerprint=as_str(response.get("system_fingerprint")),
time_to_first_chunk_seconds=time_to_first_chunk_seconds,
)
@ -566,6 +574,9 @@ def _parse_error(payload: "StandardLoggingPayload") -> SpanError | None:
return SpanError(
error_type=as_str(info.get("error_class")) or as_str(info.get("error_code")),
message=as_str(info.get("error_message")) or as_str(payload.get("error_str")),
code=as_str(info.get("error_code")),
stack_trace=as_str(info.get("traceback")),
llm_provider=as_str(info.get("llm_provider")),
)

View file

@ -69,6 +69,7 @@ class GenAI:
RESPONSE_ID: Final = "gen_ai.response.id"
RESPONSE_MODEL: Final = "gen_ai.response.model"
RESPONSE_FINISH_REASONS: Final = "gen_ai.response.finish_reasons"
RESPONSE_TIME_TO_FIRST_CHUNK: Final = "gen_ai.response.time_to_first_chunk"
# usage
USAGE_INPUT_TOKENS: Final = "gen_ai.usage.input_tokens"
USAGE_OUTPUT_TOKENS: Final = "gen_ai.usage.output_tokens"
@ -143,7 +144,27 @@ class Client:
class Error:
"""OTel-defined error attribute keys, from the semconv ``error.*`` registry.
``MESSAGE`` is marked *Deprecated* upstream in favor of domain-specific
error message keys plus ``exception.message`` on the exception event, but
is still defined and stamped by litellm's v1 integration; keeping it here
for byte-for-byte parity."""
TYPE: Final = "error.type"
MESSAGE: Final = "error.message"
class LiteLLMError:
"""LiteLLM-specific error attribute keys. Emitted under the ``error.*``
namespace (not ``litellm.*``) for byte-for-byte compat with the v1
integration in ``opentelemetry.py``; consumers reading these keys on v1
spans read the same keys on v2 spans. OTel semconv does not define any of
these three, and per its extension rules a namespace may carry additional
vendor keys as long as they don't collide with defined names."""
CODE: Final = "error.code"
STACK_TRACE: Final = "error.stack_trace"
LLM_PROVIDER: Final = "error.llm_provider"
class ExceptionEvent:

View file

@ -21,6 +21,7 @@ from litellm.integrations.opentelemetry import (
_build_metric_attribute_filter,
_resolve_metric_attribute_filter,
)
from litellm.integrations.otel.model.metadata import time_to_first_chunk_seconds
from litellm.integrations.otel.model.semconv import Metric, resolve_operation
from litellm.integrations.otel.model.utils import to_seconds
from litellm.litellm_core_utils.safe_json_dumps import safe_dumps
@ -181,13 +182,10 @@ class GenAIMetricRecorder:
self._metrics.token_usage.record(usage.get("completion_tokens", 0), attributes=out_attrs)
def _record_time_to_first_token(self, kwargs: Mapping[str, Any], common_attrs: dict) -> None:
if not kwargs.get("optional_params", {}).get("stream", False):
time_to_first_chunk = time_to_first_chunk_seconds(kwargs)
if time_to_first_chunk is None:
return
api_call_start = to_seconds(kwargs.get("api_call_start_time"))
completion_start = to_seconds(kwargs.get("completion_start_time"))
if api_call_start is None or completion_start is None:
return
self._metrics.time_to_first_token.record(completion_start - api_call_start, attributes=common_attrs)
self._metrics.time_to_first_token.record(time_to_first_chunk, attributes=common_attrs)
def _record_time_per_output_token(
self,

View file

@ -1117,33 +1117,6 @@ _OPENAPI_HTTP_METHODS = {
# `_SSO_SENSITIVE_FIELDS` / `_CACHE_SENSITIVE_FIELDS` constants in the SSO
# and cache endpoint files.
_ALERTING_SENSITIVE_VARS: Set[str] = {"SLACK_WEBHOOK_URL", "SMTP_PASSWORD"}
_DB_LITELLM_PARAM_ENV_REF_KEYS = frozenset(
{
"api_key",
"client_secret",
"vertex_credentials",
"vertex_ai_credentials",
"aws_access_key_id",
"aws_secret_access_key",
}
)
def _db_model_is_team_scoped(model: object) -> bool:
model_info = getattr(model, "model_info", None)
if isinstance(model_info, BaseModel):
return getattr(model_info, "team_id", None) is not None
if isinstance(model_info, str):
try:
model_info = json.loads(model_info)
except (TypeError, ValueError):
model_info = None
if isinstance(model_info, dict) and model_info.get("team_id") is not None:
return True
if getattr(model_info, "team_id", None) is not None:
return True
model_name = getattr(model, "model_name", None)
return isinstance(model_name, str) and model_name.startswith("model_name_")
def _strip_operation_id_method_suffix(operation_id: str) -> str:
@ -4972,17 +4945,12 @@ class ProxyConfig:
deleted_deployments += 1
return deleted_deployments
def _resolve_db_litellm_param(self, key: str, value: object, resolve_env_refs: bool = True) -> object:
def _resolve_db_litellm_param(self, key: str, value: object) -> object:
if not isinstance(value, str):
return value
decrypted_value = decrypt_value_helper(value=value, key=key, return_original_value=True)
if (
resolve_env_refs
and key in _DB_LITELLM_PARAM_ENV_REF_KEYS
and isinstance(decrypted_value, str)
and decrypted_value.startswith("os.environ/")
):
if isinstance(decrypted_value, str) and decrypted_value.startswith("os.environ/"):
return get_secret(decrypted_value)
return decrypted_value
@ -5003,13 +4971,10 @@ class ProxyConfig:
## ADD MODEL LOGIC
for m in db_models:
_litellm_params = m.litellm_params
resolve_env_refs = not _db_model_is_team_scoped(m)
if isinstance(_litellm_params, dict):
# decrypt values
for k, v in _litellm_params.items():
_litellm_params[k] = self._resolve_db_litellm_param(
key=k, value=v, resolve_env_refs=resolve_env_refs
)
_litellm_params[k] = self._resolve_db_litellm_param(key=k, value=v)
_litellm_params = LiteLLM_Params(**_litellm_params)
else:
@ -5035,15 +5000,12 @@ class ProxyConfig:
_model_list: list = []
for m in new_models:
_litellm_params = m.litellm_params
resolve_env_refs = not _db_model_is_team_scoped(m)
if isinstance(_litellm_params, BaseModel):
_litellm_params = _litellm_params.model_dump()
if isinstance(_litellm_params, dict):
# decrypt values
for k, v in _litellm_params.items():
_litellm_params[k] = self._resolve_db_litellm_param(
key=k, value=v, resolve_env_refs=resolve_env_refs
)
_litellm_params[k] = self._resolve_db_litellm_param(key=k, value=v)
_litellm_params = LiteLLM_Params(**_litellm_params)
else:
verbose_proxy_logger.error(

View file

@ -1,5 +1,5 @@
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:c61ac6919b811ea53c4782d69f1fe05218ba3c25d53f01b6ab7892e621bd4370
ARG LITELLM_BUILD_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG LITELLM_RUNTIME_IMAGE=cgr.dev/chainguard/wolfi-base@sha256:42df77a9974d6ec8b17a5ee8bc23b532600a44d705acef2409e0933c1251b45f
ARG UV_IMAGE=ghcr.io/astral-sh/uv:0.11.7@sha256:240fb85ab0f263ef12f492d8476aa3a2e4e1e333f7d67fbdd923d00a506a516a
FROM $UV_IMAGE AS uvbin

View file

@ -579,13 +579,10 @@ def _exception_event(span):
def test_error_message_recorded_as_full_exception_event_untruncated():
"""Regression for the Elasticsearch keyword/ignore_above:1024 truncation.
A long error message must survive intact on the standard ``exception``
event under ``exception.message`` not get dropped onto a bare string
attribute that backends dynamic-map to a 1024-char ``keyword``. The SDK
must not truncate it either, so a 5000-char message stays 5000 chars.
"""
"""The ``exception`` event carries the full untruncated message under
``exception.message`` so backends that dynamic-map unknown string span
attrs to ``keyword`` (e.g. Elasticsearch with a 1024-char ``ignore_above``)
still see it in full via the semconv-recognized event field."""
from litellm.integrations.otel.model.semconv import Error, ExceptionEvent
long_message = "boom: " + "x" * 5000
@ -596,13 +593,108 @@ def test_error_message_recorded_as_full_exception_event_untruncated():
assert len(event.attributes[ExceptionEvent.MESSAGE]) == len(long_message) > 1024
assert event.attributes[ExceptionEvent.TYPE] == "litellm.APIError"
# error.type stays a low-cardinality attribute; the message does NOT become a
# bare string attribute (which is what got truncated).
# error.type stays a low-cardinality attribute; the exception EVENT field
# ``exception.message`` never becomes a bare string attribute.
assert span.attributes[Error.TYPE] == "litellm.APIError"
assert ExceptionEvent.MESSAGE not in span.attributes
assert span.status.description == long_message
def test_error_details_stamped_as_span_attributes_for_labels_ingest():
"""OTel-defined keys and litellm-specific detail keys both ride span
attributes so backends that flatten attrs into label indexes (Elastic APM
``labels.*``, Datadog span tags) render them. The exception event with the
full untruncated message stays alongside both places, matching v1's
shape."""
from litellm.integrations.otel.model.semconv import Error, ExceptionEvent, LiteLLMError
from litellm.integrations.otel.emitter import SpanEmitter
cfg = OpenTelemetryV2Config(exporter="in_memory")
provider, exporter = providers.in_memory_provider(cfg)
engine = SpanEmitter(providers.get_tracer(provider, "t"), cfg)
data = LLMCallSpanData(
operation=GenAIOperation.CHAT,
provider="openai",
request_model="gpt-4o",
response_model=None,
response_id=None,
request_params=LLMRequestParams(),
usage=LLMUsage(),
finish_reasons=(),
error=SpanError(
error_type="litellm.BadRequestError",
message="400: violated moderation policy",
code="400",
stack_trace="File proxy_server.py line 8570 ...",
llm_provider="openai",
),
response_cost=None,
server=None,
identity=RequestIdentity(call_id=None),
)
engine.emit(SpanRole.LLM_CALL, data)
(span,) = exporter.get_finished_spans()
# OTel-defined keys (from the ``error.*`` semconv registry).
assert span.attributes[Error.TYPE] == "litellm.BadRequestError"
assert span.attributes[Error.MESSAGE] == "400: violated moderation policy"
# LiteLLM-specific detail keys — vendor-namespaced under ``error.*``
# for v1-parity, not defined by OTel semconv.
assert span.attributes[LiteLLMError.CODE] == "400"
assert span.attributes[LiteLLMError.STACK_TRACE] == "File proxy_server.py line 8570 ..."
assert span.attributes[LiteLLMError.LLM_PROVIDER] == "openai"
# The exception event carries the same message on the span too.
event = _exception_event(span)
assert event.attributes[ExceptionEvent.MESSAGE] == "400: violated moderation policy"
def test_error_details_omitted_when_span_error_carries_only_message():
"""A guardrail-shape error (message only, no code/traceback/provider) must
not pollute the span with empty-string detail attributes. Only the keys
that carry real data land."""
from litellm.integrations.otel.model.semconv import Error, LiteLLMError
span = _emit_error_span("guardrail rejected", error_type="ContentFilter")
assert span.attributes[Error.TYPE] == "ContentFilter"
assert span.attributes[Error.MESSAGE] == "guardrail rejected"
# LiteLLM-specific detail keys aren't stamped when the SpanError doesn't
# carry them.
assert LiteLLMError.CODE not in span.attributes
assert LiteLLMError.STACK_TRACE not in span.attributes
assert LiteLLMError.LLM_PROVIDER not in span.attributes
def test_v2_error_attribute_keys_match_v1_error_attributes_byte_for_byte():
"""v1 (``opentelemetry.py``) and v2 (``otel/`` package) stamp identical
span-attribute keys so consumers reading ``labels.error_message`` don't
care which integration produced the span. Renaming either side is a
breaking change for downstream dashboards; this test locks the vocabulary."""
from litellm.integrations._types.open_inference import ErrorAttributes
from litellm.integrations.otel.model.semconv import Error, LiteLLMError
assert Error.TYPE == ErrorAttributes.ERROR_TYPE
assert Error.MESSAGE == ErrorAttributes.ERROR_MESSAGE
assert LiteLLMError.CODE == ErrorAttributes.ERROR_CODE
assert LiteLLMError.STACK_TRACE == ErrorAttributes.ERROR_STACK_TRACE
assert LiteLLMError.LLM_PROVIDER == ErrorAttributes.ERROR_LLM_PROVIDER
def test_error_message_falls_back_to_error_type_when_message_absent():
"""A ``SpanError(error_type=..., message=None)`` still renders on the span:
the resolved message is the error_type, and it lands on ``error.message``,
the exception event, and the span-status description in lockstep so a
single-source-of-truth view isn't inconsistent."""
from litellm.integrations.otel.model.semconv import Error, ExceptionEvent
span = _emit_error_span(message=None, error_type="RateLimitError")
assert span.attributes[Error.MESSAGE] == "RateLimitError"
assert _exception_event(span).attributes[ExceptionEvent.MESSAGE] == "RateLimitError"
assert span.status.description == "RateLimitError"
def test_success_span_records_no_exception_event():
from litellm.integrations.otel.emitter import SpanEmitter
from litellm.integrations.otel.model.semconv import ExceptionEvent

View file

@ -168,6 +168,43 @@ def test_async_log_success_event_emits_llm_call_span():
assert span.status.status_code is StatusCode.UNSET
def test_streaming_span_carries_time_to_first_chunk():
logger, exporter = _logger()
kwargs = {
**_kwargs(payload=_payload(stream=True)),
"optional_params": {"stream": True},
"api_call_start_time": datetime(2026, 5, 26, 12, 0, 0, tzinfo=timezone.utc),
"completion_start_time": datetime(2026, 5, 26, 12, 0, 0, 750000, tzinfo=timezone.utc),
}
_emit_llm(logger, kwargs)
(span,) = exporter.get_finished_spans()
assert span.attributes[GenAI.RESPONSE_TIME_TO_FIRST_CHUNK] == pytest.approx(0.75)
def test_non_streaming_span_has_no_time_to_first_chunk():
logger, exporter = _logger()
kwargs = {
**_kwargs(),
"optional_params": {},
"api_call_start_time": datetime(2026, 5, 26, 12, 0, 0, tzinfo=timezone.utc),
"completion_start_time": datetime(2026, 5, 26, 12, 0, 5, tzinfo=timezone.utc),
}
_emit_llm(logger, kwargs)
(span,) = exporter.get_finished_spans()
assert GenAI.RESPONSE_TIME_TO_FIRST_CHUNK not in span.attributes
def test_streaming_span_without_timing_omits_time_to_first_chunk():
logger, exporter = _logger()
kwargs = {
**_kwargs(payload=_payload(stream=True)),
"optional_params": {"stream": True},
}
_emit_llm(logger, kwargs)
(span,) = exporter.get_finished_spans()
assert GenAI.RESPONSE_TIME_TO_FIRST_CHUNK not in span.attributes
def test_async_log_failure_event_marks_error_status():
logger, exporter = _logger()
payload = _payload(

View file

@ -144,11 +144,13 @@ def _all_constants(cls):
def test_attribute_keys_are_unique_across_namespaces():
from litellm.integrations.otel import MCP, Client, JsonRpc, Network
from litellm.integrations.otel import MCP, Client, JsonRpc, LiteLLMError, Network
# prefixes are allowed to be substrings; exact keys must not collide.
# ``LiteLLMError`` shares the ``error.*`` prefix with ``Error`` by design
# (v1-parity); the assert below is the guarantee they never overlap.
exact = set()
for cls in (GenAI, Error, Server, HTTP, DB, MCP, JsonRpc, Network, Client):
for cls in (GenAI, Error, LiteLLMError, Server, HTTP, DB, MCP, JsonRpc, Network, Client):
for key in _all_constants(cls):
assert key not in exact, f"duplicate attribute key {key}"
exact.add(key)
@ -342,6 +344,47 @@ def test_llm_call_adapter_failure_path():
assert data.error.message == "429 slow down"
def test_llm_call_adapter_carries_error_detail_fields():
"""``_parse_error`` threads the full detail set from ``error_information``
(``error_code``, ``traceback``, ``llm_provider``) onto ``SpanError`` so the
emitter can stamp them as span attributes."""
payload = _sample_payload(
status="failure",
error_information={
"error_class": "BadRequestError",
"error_message": "400 violated moderation policy",
"error_code": "400",
"traceback": "File proxy_server.py line 8570 ...",
"llm_provider": "openai",
},
)
data = LLMCallSpanData.from_standard_logging_payload(payload)
assert data.error is not None
assert data.error.error_type == "BadRequestError"
assert data.error.message == "400 violated moderation policy"
assert data.error.code == "400"
assert data.error.stack_trace == "File proxy_server.py line 8570 ..."
assert data.error.llm_provider == "openai"
def test_llm_call_adapter_error_details_default_to_none_when_absent():
"""Guardrail-shape payloads carry only ``error_class`` + ``error_message``.
The detail fields must stay ``None`` so the emitter's ``if error.code:``
guards skip stamping empty attributes."""
payload = _sample_payload(
status="failure",
error_information={
"error_class": "ContentFilter",
"error_message": "guardrail rejected",
},
)
data = LLMCallSpanData.from_standard_logging_payload(payload)
assert data.error is not None
assert data.error.code is None
assert data.error.stack_trace is None
assert data.error.llm_provider is None
def test_adapter_is_resilient_to_minimal_payload():
data = LLMCallSpanData.from_standard_logging_payload({})
assert data.request_model == ""

View file

@ -952,6 +952,11 @@ def test_ProxyConfig__add_deployment_invalid_litellm_params_skips(monkeypatch):
def test_ProxyConfig__add_deployment_resolves_env_refs_after_db_decrypt(monkeypatch):
"""Every ``os.environ/`` value on an admin-scoped DB row resolves at
load time, regardless of the field name. Replaces the earlier
behavior where only fields in ``_DB_LITELLM_PARAM_ENV_REF_KEYS``
resolved: the whitelist has been removed so the resolver applies to
every string field."""
monkeypatch.setenv("LITELLM_DB_MODEL_API_KEY", "resolved-secret")
monkeypatch.setenv("LITELLM_MASTER_KEY", "master-secret")
monkeypatch.setattr(
@ -979,19 +984,21 @@ def test_ProxyConfig__add_deployment_resolves_env_refs_after_db_decrypt(monkeypa
assert added == 1
assert deployment.litellm_params.api_key == "resolved-secret"
assert deployment.litellm_params.api_base == "os.environ/LITELLM_MASTER_KEY"
assert deployment.litellm_params.api_base == "master-secret"
def test_ProxyConfig__add_deployment_keeps_team_env_refs_literal(monkeypatch):
def fail_on_call(secret_name, *args, **kwargs):
raise AssertionError("team DB models should not resolve env refs")
def test_ProxyConfig__add_deployment_resolves_team_env_refs(monkeypatch):
"""Team-scoped DB rows now resolve ``os.environ/`` refs the same way
admin rows do. The prior team-scoped short-circuit and the
field-by-field whitelist have both been removed; the write-side team
auth check in ``ModelManagementAuthChecks.can_user_make_model_call``
remains the single trust boundary. A literal (non-``os.environ/``)
value still passes through unchanged."""
monkeypatch.setenv("LITELLM_MASTER_KEY", "master-secret")
monkeypatch.setattr(
"litellm.proxy.proxy_server.decrypt_value_helper",
lambda value, key, return_original_value: value,
)
monkeypatch.setattr("litellm.proxy.proxy_server.get_secret", fail_on_call)
fake_router = MagicMock()
fake_router.upsert_deployment = MagicMock(return_value=True)
monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", fake_router)
@ -1003,7 +1010,7 @@ def test_ProxyConfig__add_deployment_keeps_team_env_refs_literal(monkeypatch):
litellm_params={
"model": "openai/gpt-4o-mini",
"api_key": "os.environ/LITELLM_MASTER_KEY",
"api_base": "https://attacker.example",
"api_base": "https://team.example",
},
blocked=False,
)
@ -1012,8 +1019,8 @@ def test_ProxyConfig__add_deployment_keeps_team_env_refs_literal(monkeypatch):
deployment = fake_router.upsert_deployment.call_args.kwargs["deployment"]
assert added == 1
assert deployment.litellm_params.api_key == "os.environ/LITELLM_MASTER_KEY"
assert deployment.litellm_params.api_base == "https://attacker.example"
assert deployment.litellm_params.api_key == "master-secret"
assert deployment.litellm_params.api_base == "https://team.example"
def test_ProxyConfig__resolve_db_litellm_param_skips_non_string_values(monkeypatch):
@ -1029,6 +1036,100 @@ def test_ProxyConfig__resolve_db_litellm_param_skips_non_string_values(monkeypat
assert pc._resolve_db_litellm_param(key="tpm", value=100) == 100
def test_ProxyConfig__add_deployment_resolves_env_refs_for_aws_bedrock_auth_params(
monkeypatch,
):
"""Regression: DB-stored Bedrock/SageMaker auth params like
``aws_role_name: os.environ/BEDROCK_ASSUME_ROLE_ARN`` must resolve at
DB-load time. PR #30867 removed request-time expansion in
``BaseAWSLLM.get_credentials``; without DB-load resolution the literal
string reaches STS and fails with ``ValidationError: ... is invalid``."""
aws_env = {
"aws_session_token": ("BEDROCK_SESSION_TOKEN", "resolved-session-token"),
"aws_region_name": ("BEDROCK_REGION", "us-east-1"),
"aws_session_name": ("BEDROCK_SESSION_NAME", "resolved-session"),
"aws_profile_name": ("BEDROCK_PROFILE", "resolved-profile"),
"aws_role_name": (
"BEDROCK_ASSUME_ROLE_ARN",
"arn:aws:iam::123456789012:role/resolved",
),
"aws_web_identity_token": ("BEDROCK_WEB_IDENTITY_TOKEN", "resolved-token"),
"aws_sts_endpoint": (
"BEDROCK_STS_ENDPOINT",
"https://sts.us-east-1.amazonaws.com",
),
"aws_external_id": ("BEDROCK_EXTERNAL_ID", "resolved-external-id"),
"aws_bedrock_runtime_endpoint": (
"BEDROCK_RUNTIME_ENDPOINT",
"https://bedrock-runtime.us-east-1.amazonaws.com",
),
"aws_bedrock_project_id": ("BEDROCK_PROJECT_ID", "resolved-project-id"),
"aws_batch_role_arn": (
"BEDROCK_BATCH_ROLE_ARN",
"arn:aws:iam::123456789012:role/batch",
),
"aws_workspace_id": ("BEDROCK_WORKSPACE_ID", "resolved-workspace-id"),
}
for _, (env_name, env_value) in aws_env.items():
monkeypatch.setenv(env_name, env_value)
monkeypatch.setattr(
"litellm.proxy.proxy_server.decrypt_value_helper",
lambda value, key, return_original_value: value,
)
fake_router = MagicMock()
fake_router.upsert_deployment = MagicMock(return_value=True)
monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", fake_router)
pc = ProxyConfig()
litellm_params: Dict[str, Any] = {"model": "bedrock/anthropic.claude-v2"}
for key, (env_name, _) in aws_env.items():
litellm_params[key] = f"os.environ/{env_name}"
db_model = SimpleNamespace(
model_id="model-1",
model_name="bedrock-model",
model_info={"id": "model-1"},
litellm_params=litellm_params,
blocked=False,
)
added = pc._add_deployment(db_models=[db_model])
deployment = fake_router.upsert_deployment.call_args.kwargs["deployment"]
assert added == 1
for key, (_, expected) in aws_env.items():
assert getattr(deployment.litellm_params, key) == expected, key
def test_ProxyConfig__add_deployment_resolves_env_refs_on_arbitrary_field(monkeypatch):
"""A made-up field name that was never on the removed whitelist still
resolves ``os.environ/`` refs. Pins the "no whitelist" invariant:
the resolver applies to every string field, not a curated list."""
monkeypatch.setenv("SOME_CUSTOM_ENV", "resolved-custom-value")
monkeypatch.setattr(
"litellm.proxy.proxy_server.decrypt_value_helper",
lambda value, key, return_original_value: value,
)
fake_router = MagicMock()
fake_router.upsert_deployment = MagicMock(return_value=True)
monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", fake_router)
pc = ProxyConfig()
db_model = SimpleNamespace(
model_id="model-1",
model_name="custom-field-model",
model_info={"id": "model-1"},
litellm_params={
"model": "openai/gpt-4o-mini",
"some_future_field": "os.environ/SOME_CUSTOM_ENV",
},
blocked=False,
)
added = pc._add_deployment(db_models=[db_model])
deployment = fake_router.upsert_deployment.call_args.kwargs["deployment"]
assert added == 1
assert deployment.litellm_params.some_future_field == "resolved-custom-value"
# ---------------------------------------------------------------------------
# ProxyConfig.decrypt_model_list_from_db
# ---------------------------------------------------------------------------
@ -1064,6 +1165,9 @@ def test_ProxyConfig_decrypt_model_list_from_db_returns_decrypted(monkeypatch):
def test_ProxyConfig_decrypt_model_list_from_db_resolves_env_refs_after_db_decrypt(
monkeypatch,
):
"""Path B (feeding /v2/model/info fallback and /model/info fallback)
resolves every ``os.environ/`` field on admin-scoped rows, mirroring
path A. Both paths now share the same universal-resolution shape."""
monkeypatch.setenv("LITELLM_DB_MODEL_API_KEY", "resolved-secret")
monkeypatch.setenv("LITELLM_MASTER_KEY", "master-secret")
monkeypatch.setattr(
@ -1090,15 +1194,16 @@ def test_ProxyConfig_decrypt_model_list_from_db_resolves_env_refs_after_db_decry
out = pc.decrypt_model_list_from_db(new_models=[m])
assert out[0]["litellm_params"]["api_key"] == "resolved-secret"
assert out[0]["litellm_params"]["api_base"] == "os.environ/LITELLM_MASTER_KEY"
assert out[0]["litellm_params"]["api_base"] == "master-secret"
def test_ProxyConfig_decrypt_model_list_from_db_keeps_team_env_refs_literal_after_db_decrypt(
def test_ProxyConfig_decrypt_model_list_from_db_resolves_team_env_refs_after_db_decrypt(
monkeypatch,
):
def fail_on_call(secret_name, *args, **kwargs):
raise AssertionError("team DB models should not resolve env refs")
"""Team-scoped rows on path B resolve ``os.environ/`` refs just like
admin rows do. Pairs with
``test_ProxyConfig__add_deployment_resolves_team_env_refs`` on path
A both paths now agree on the trust model."""
monkeypatch.setenv("LITELLM_MASTER_KEY", "master-secret")
monkeypatch.setattr(
"litellm.proxy.proxy_server.decrypt_value_helper",
@ -1106,7 +1211,6 @@ def test_ProxyConfig_decrypt_model_list_from_db_keeps_team_env_refs_literal_afte
"os.environ/LITELLM_MASTER_KEY" if key == "api_key" else value
),
)
monkeypatch.setattr("litellm.proxy.proxy_server.get_secret", fail_on_call)
pc = ProxyConfig()
m = SimpleNamespace(
model_id="model-1",
@ -1114,7 +1218,7 @@ def test_ProxyConfig_decrypt_model_list_from_db_keeps_team_env_refs_literal_afte
model_info={"id": "model-1", "team_id": "team-1"},
litellm_params={
"api_key": "encrypted-env-ref",
"api_base": "https://attacker.example",
"api_base": "https://team.example",
"model": "openai/gpt-4o-mini",
},
blocked=False,
@ -1122,8 +1226,8 @@ def test_ProxyConfig_decrypt_model_list_from_db_keeps_team_env_refs_literal_afte
out = pc.decrypt_model_list_from_db(new_models=[m])
assert out[0]["litellm_params"]["api_key"] == "os.environ/LITELLM_MASTER_KEY"
assert out[0]["litellm_params"]["api_base"] == "https://attacker.example"
assert out[0]["litellm_params"]["api_key"] == "master-secret"
assert out[0]["litellm_params"]["api_base"] == "https://team.example"
def test_ProxyConfig_decrypt_model_list_from_db_invalid_params_skips():

View file

@ -937,6 +937,22 @@ describe("CreateMCPServer", () => {
const reopenedUrlInput = screen.getByPlaceholderText("https://your-mcp-server.com") as HTMLInputElement;
expect(reopenedUrlInput.value).toBe("");
});
it("does not reset an in-flight OAuth resume when mounted with the modal closed (post-redirect restore)", () => {
// After the "Authorize & Fetch Token" redirect the page reloads and this
// component mounts with isModalVisible=false while useMcpOAuthFlow is still
// exchanging the authorization code. Calling reset() during that mount bumps
// the hook's reset version and the fetched token is silently discarded, so
// the user sees no Connection Status / Tool Configuration and must authorize
// again after saving.
const { rerender } = render(<CreateMCPServer {...defaultProps} isModalVisible={false} />);
expect(oauthHook.reset).not.toHaveBeenCalled();
// A real open -> closed transition must still reset (the #30000 leak fix).
rerender(<CreateMCPServer {...defaultProps} isModalVisible={true} />);
rerender(<CreateMCPServer {...defaultProps} isModalVisible={false} />);
expect(oauthHook.reset).toHaveBeenCalled();
});
});
describe("when stdio transport is selected", () => {

View file

@ -603,9 +603,15 @@ const CreateMCPServer: React.FC<CreateMCPServerProps> = ({
// Clear form, tools, and OAuth state when the modal closes so a previous server's
// authorization, credentials, or tool list never bleed into the next "Add New MCP
// Server" session, including when a parent dismisses the modal without routing
// through handleCancel or handleCreate.
// through handleCancel or handleCreate. Only a real open -> closed transition may
// trigger this: on the post-OAuth-redirect remount the modal starts closed while
// resumeOAuthFlow's token exchange is in flight, and resetting then discards the
// fetched token.
const wasModalVisibleRef = React.useRef(isModalVisible);
React.useEffect(() => {
if (!isModalVisible) {
const wasVisible = wasModalVisibleRef.current;
wasModalVisibleRef.current = isModalVisible;
if (!isModalVisible && wasVisible) {
form.resetFields();
setFormValues({});
setOauthAccessToken(null);