Merge remote-tracking branch 'origin/main' into litellm_anthropic_wif_backend

# Conflicts:
#	tests/unit/test_router/test_router.py
This commit is contained in:
mateo-berri 2026-10-03 10:47:28 -07:00
commit f43151a97d
739 changed files with 65257 additions and 12540 deletions

Binary file not shown.

After

Width:  |  Height:  |  Size: 77 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 79 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 61 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 86 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 75 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 40 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 88 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 95 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 96 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 94 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 70 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 82 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 82 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 64 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 90 KiB

View file

@ -49,6 +49,10 @@ jobs:
if: steps.changes.outputs.decision != 'skip'
run: npm ci
- name: Check UI production source types
if: steps.changes.outputs.decision != 'skip'
run: npm run typecheck
- name: Run UI type tests (Vitest)
if: steps.changes.outputs.decision != 'skip'
env:

View file

@ -5,6 +5,8 @@ on:
paths:
- "litellm-rust/**"
- "litellm/rust_bridge/**"
- "scripts/generate_trace_types.py"
- "scripts/trace_codegen/**"
- "tests/test_litellm_rust/**"
- "litellm/integrations/custom_logger.py"
- "litellm/litellm_core_utils/litellm_logging.py"
@ -32,6 +34,8 @@ on:
paths:
- "litellm-rust/**"
- "litellm/rust_bridge/**"
- "scripts/generate_trace_types.py"
- "scripts/trace_codegen/**"
- "tests/test_litellm_rust/**"
- "litellm/integrations/custom_logger.py"
- "litellm/litellm_core_utils/litellm_logging.py"
@ -85,7 +89,7 @@ jobs:
cache-on-failure: true
save-if: ${{ github.ref == 'refs/heads/main' }}
- run: cargo clippy --workspace --all-targets --locked -- -D warnings
- run: cargo clippy --workspace --all-targets --locked --features litellm-traces/schema,litellm-traces-clickhouse/schema -- -D warnings
rust-test:
runs-on: ubuntu-latest
@ -124,7 +128,11 @@ jobs:
cache-on-failure: true
save-if: ${{ github.ref == 'refs/heads/main' }}
- run: cargo nextest run --workspace --locked
- name: Check generated trace contracts
working-directory: .
run: uv run scripts/generate_trace_types.py --check
- run: cargo nextest run --workspace --locked --features litellm-traces/schema,litellm-traces-clickhouse/schema
- run: cargo test --workspace --doc --locked

View file

@ -165,6 +165,7 @@ jobs:
tests/unit/proxy/response_api_endpoints
tests/unit/proxy/image_endpoints
tests/unit/proxy/ocr_endpoints
tests/unit/proxy/search_endpoints
tests/unit/proxy/vector_store_endpoints
tests/unit/proxy/agent_endpoints
tests/unit/proxy/a2a

View file

@ -2,5 +2,6 @@ FROM python:3.12-slim
WORKDIR /app
RUN pip install --no-cache-dir httpx==0.28.1 pydantic==2.11.7
COPY litellm/proxy/lens/__init__.py litellm/proxy/lens/models.py litellm/proxy/lens/trace_store.py litellm/proxy/lens/analysis.py litellm/proxy/lens/worker.py /app/lens/
COPY litellm/proxy/lens/prompts/ /app/lens/prompts/
USER 65532:65532
CMD ["python", "-m", "lens.worker"]

View file

@ -4,5 +4,8 @@
!litellm/proxy/lens/
!litellm/proxy/lens/__init__.py
!litellm/proxy/lens/models.py
!litellm/proxy/lens/trace_store.py
!litellm/proxy/lens/analysis.py
!litellm/proxy/lens/worker.py
!litellm/proxy/lens/prompts/
!litellm/proxy/lens/prompts/**

View file

@ -7,7 +7,8 @@ services:
target: runtime
command: ["--config", "/app/tracing-config.yaml", "--port", "4000"]
environment:
LITELLM_MASTER_KEY: local-tracing-master-key
LITELLM_MASTER_KEY: sk-1234
LITELLM_DANGEROUSLY_PERMIT_WEAK_OR_UNSET_MASTER_KEY: "true"
LITELLM_SALT_KEY: sk-local-tracing-salt-key
DATABASE_URL: postgresql://litellm:litellm@db:5432/litellm
STORE_MODEL_IN_DB: "True"

View file

@ -7,15 +7,19 @@
## This accepts a list of user id's for whom calls will be rejected
from typing import Optional, Literal
import litellm
from litellm.proxy.utils import PrismaClient
from litellm.caching.caching import DualCache
from litellm.proxy._types import UserAPIKeyAuth, LiteLLM_EndUserTable
from litellm.integrations.custom_logger import CustomLogger
from litellm._logging import verbose_proxy_logger
from typing import Literal, Optional
from fastapi import HTTPException
import litellm
from litellm._internal_context import with_service_target
from litellm._logging import verbose_proxy_logger
from litellm.caching.caching import DualCache
from litellm.integrations.custom_logger import CustomLogger
from litellm.proxy._types import LiteLLM_EndUserTable, UserAPIKeyAuth
from litellm.proxy.common_utils.user_api_key_cache import AUTH_OBJECTS_TARGET
from litellm.proxy.utils import PrismaClient
class _ENTERPRISE_BlockedUserList(CustomLogger):
enforces_request_content: bool = True
@ -54,6 +58,7 @@ class _ENTERPRISE_BlockedUserList(CustomLogger):
if litellm.set_verbose is True:
print(print_statement) # noqa
@with_service_target(AUTH_OBJECTS_TARGET)
async def async_pre_call_hook(
self,
user_api_key_dict: UserAPIKeyAuth,

View file

@ -6,7 +6,7 @@ Base class for sending emails to user after creating keys or invite links
import html
import json
import os
from typing import List, Literal, Optional
from typing import Final, List, Literal, Optional
from litellm_enterprise.types.enterprise_callbacks.send_emails import (
EmailEvent,
@ -15,6 +15,7 @@ from litellm_enterprise.types.enterprise_callbacks.send_emails import (
SendKeyRotatedEmailEvent,
)
from litellm._internal_context import with_service_target
from litellm._logging import verbose_proxy_logger
from litellm.caching.caching import DualCache
from litellm.constants import (
@ -48,6 +49,8 @@ from litellm.proxy._types import (
from litellm.secret_managers.main import get_secret_bool
from litellm.types.integrations.slack_alerting import LITELLM_LOGO_URL
_BUDGET_ALERT_CLAIMS_TARGET: Final = "budget_alert_claims"
def _max_budget_alert_id(user_info: CallInfo) -> str:
if user_info.event_group == Litellm_EntityType.TEAM_MEMBER:
@ -437,6 +440,7 @@ class BaseEmailLogger(CustomLogger):
html_body=email_html_content,
)
@with_service_target(_BUDGET_ALERT_CLAIMS_TARGET)
async def budget_alerts(
self,
type: Literal[
@ -606,6 +610,7 @@ class BaseEmailLogger(CustomLogger):
await self._release_budget_alert_claim(_cache, _cache_key)
return
@with_service_target(_BUDGET_ALERT_CLAIMS_TARGET)
async def _handle_multi_threshold_max_budget_alert(
self,
user_info: CallInfo,
@ -691,6 +696,7 @@ class BaseEmailLogger(CustomLogger):
)
await self._release_budget_alert_claim(_cache, _cache_key)
@with_service_target(_BUDGET_ALERT_CLAIMS_TARGET)
async def _release_budget_alert_claim(self, cache: DualCache, cache_key: str) -> None:
try:
await cache.async_delete_cache(key=cache_key)

View file

@ -17,6 +17,7 @@ from litellm_enterprise.types.enterprise_callbacks.send_emails import (
from litellm._logging import verbose_proxy_logger
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.db.db_span import db_span
router = APIRouter()
@ -94,16 +95,17 @@ async def _save_email_settings(prisma_client, settings: Dict[str, bool]):
json_settings = json.dumps(general_settings, default=str)
# Save updated general settings
await prisma_client.db.litellm_config.upsert(
where={"param_name": "general_settings"},
data={
"create": {
"param_name": "general_settings",
"param_value": json_settings,
async with db_span("save_email_settings", "LiteLLM_Config"):
await prisma_client.db.litellm_config.upsert(
where={"param_name": "general_settings"},
data={
"create": {
"param_name": "general_settings",
"param_value": json_settings,
},
"update": {"param_value": json_settings},
},
"update": {"param_value": json_settings},
},
)
)
except Exception as e:
raise HTTPException(
status_code=500,

View file

@ -26,6 +26,7 @@ from pydantic import ValidationError
import litellm
from litellm import Router, verbose_logger
from litellm._internal_context import with_service_target
from litellm._uuid import uuid
from litellm.caching.caching import DualCache
from litellm.constants import MAX_FILE_LIST_LIMIT
@ -54,7 +55,6 @@ from litellm.proxy.litellm_pre_call_utils import LiteLLMProxyRequestSetup
from litellm.proxy.openai_files_endpoints.common_utils import (
BATCH_CREATE_HIDDEN_PARAM,
FILE_LIST_CONTINUATION_CHUNK_SIZE,
ManagedFileIdResolver,
_is_base64_encoded_unified_file_id,
apply_unified_file_ids,
decode_model_from_file_id,
@ -230,6 +230,9 @@ def _storage_metadata_of(file_object: OpenAIFileObject | None) -> Mapping[str, s
)
_MANAGED_FILES_TARGET: Final = "managed_files"
class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
# Class variables or attributes
def __init__(self, internal_usage_cache: InternalUsageCache, prisma_client: PrismaClient):
@ -243,6 +246,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
return PrometheusLogger.get_instance()
@with_service_target(_MANAGED_FILES_TARGET)
async def store_unified_file_id(
self,
file_id: str,
@ -326,6 +330,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
verbose_logger.warning(f"could not resolve org for managed object attribution: {e}")
return None
@with_service_target(_MANAGED_FILES_TARGET)
async def store_unified_object_id(
self,
unified_object_id: str,
@ -413,6 +418,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
},
)
@with_service_target(_MANAGED_FILES_TARGET)
async def get_unified_file_id(
self, file_id: str, litellm_parent_otel_span: Optional[Span] = None
) -> Optional[LiteLLM_ManagedFileTable]:
@ -435,6 +441,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints):
return LiteLLM_ManagedFileTable.model_validate(db_object.model_dump())
return None
@with_service_target(_MANAGED_FILES_TARGET)
async def delete_unified_file_id(
self, file_id: str, litellm_parent_otel_span: Optional[Span] = None
) -> OpenAIFileObject:

View file

@ -4453,15 +4453,20 @@ dependencies = [
name = "litellm-traces"
version = "0.1.0"
dependencies = [
"askama",
"criterion",
"indexmap 2.14.0",
"litellm-llms-types",
"macro_rules_attribute",
"opentelemetry-proto",
"prost",
"rstest",
"schemars 1.2.2",
"serde",
"serde_json",
"strum",
"thiserror 2.0.19",
"time",
]
[[package]]
@ -4469,15 +4474,19 @@ name = "litellm-traces-clickhouse"
version = "0.1.0"
dependencies = [
"askama",
"base64 0.22.1",
"flate2",
"futures-util",
"hmac 0.12.1",
"jsonschema",
"litellm-http",
"litellm-migrate",
"litellm-storage-clickhouse",
"litellm-traces",
"macro_rules_attribute",
"moka",
"rstest",
"schemars 1.2.2",
"serde",
"serde_json",
"sha2 0.10.9",
@ -4486,6 +4495,7 @@ dependencies = [
"thiserror 2.0.19",
"time",
"tokio",
"tracing",
"url",
"wiremock",
]

View file

@ -45,8 +45,7 @@ mod _native {
use crate::routes::token_counter::TokenCounter;
#[pymodule_export]
use crate::routes::traces::{
NativeTraceConfig, NativeTraceStorage, trace_decode_otlp, trace_encode_error,
trace_normalized_field_definitions,
NativeTraceConfig, NativeTraceStorage, trace_encode_error, trace_span_rows,
};
#[cfg(feature = "huggingface")]
#[pymodule_export]
@ -114,9 +113,8 @@ mod tests {
"NativeDiagnosticProcessor",
"NativeTraceConfig",
"NativeTraceStorage",
"trace_decode_otlp",
"trace_encode_error",
"trace_normalized_field_definitions",
"trace_span_rows",
"TokenCounter",
"Tokenizer",
"gil_stats",

View file

@ -1,14 +1,13 @@
use std::collections::BTreeMap;
use litellm_host_python::{FromPythonCache, ToPythonCache};
use litellm_http::ClientVariant;
use litellm_traces::{QueryScope, ReadQuery, Shared};
use litellm_traces::{QueryScope, ReadQuery, Tenant, query::named::ReadAccessParams};
use litellm_traces_clickhouse::{Config, Error, InsertTable, Parameter, QueryReaders};
use prost::Message;
use pyo3::{
exceptions::{PyOverflowError, PyRuntimeError, PyValueError},
prelude::*,
types::{PyBytes, PyDict, PyList, PyMapping, PyString},
types::PyBytes,
};
#[derive(Message)]
@ -36,14 +35,20 @@ fn map_error_ref(error: &Error) -> PyErr {
use litellm_storage_clickhouse::Error as StorageError;
match error {
Error::Decode(litellm_traces::Error::TooLarge) | Error::InsertTooLarge => {
PyOverflowError::new_err(error.to_string())
}
Error::InvalidRow
| Error::InvalidTable
| Error::InvalidCursor(_)
| Error::AmbiguousTrace
| Error::Decode(_)
| Error::InvalidSchema
| Error::InvalidQuery
| Error::InvalidParameters
| Error::InvalidScope => PyValueError::new_err(error.to_string()),
Error::InsertTooLarge => PyOverflowError::new_err(error.to_string()),
Error::SchemaFailed(_)
Error::Task
| Error::SchemaFailed(_)
| Error::SchemaTransport
| Error::MissingSecret
| Error::Busy
@ -87,9 +92,15 @@ pub struct NativeTraceConfig {
#[pymethods]
impl NativeTraceConfig {
#[new]
fn new(database: String, url: &str, retention_days: u32) -> PyResult<Self> {
fn new(
database: String,
url: &str,
retention_days: u32,
max_attribute_value_bytes: usize,
) -> PyResult<Self> {
Ok(Self {
inner: Config::new(database, url, retention_days).map_err(map_error)?,
inner: Config::new(database, url, retention_days, max_attribute_value_bytes)
.map_err(map_error)?,
})
}
}
@ -137,7 +148,9 @@ impl NativeTraceStorage {
&self,
py: Python<'py>,
table: &str,
#[pyo3(from_py_with = insert_rows_from_py)] rows: Vec<litellm_traces_clickhouse::InsertRow>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] rows: Vec<
BTreeMap<String, serde_json::Value>,
>,
) -> PyResult<Bound<'py, PyAny>> {
let table = InsertTable::parse(table).map_err(map_error)?;
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
@ -146,13 +159,156 @@ impl NativeTraceStorage {
crate::execution::run_async(
py,
async move {
litellm_traces_clickhouse::insert_rows(&client, &connection, &database, table, rows)
.await
},
map_error,
)
}
fn ingest<'py>(
&self,
py: Python<'py>,
payload: &[u8],
content_type: Option<String>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] tenant: Tenant,
) -> PyResult<Bound<'py, PyAny>> {
let payload = payload.to_vec();
let max_value_bytes = self.config.max_attribute_value_bytes();
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().writer().clone();
let database = self.config.storage().database().to_owned();
crate::execution::run_async(
py,
async move {
let rows = tokio::task::spawn_blocking(move || {
litellm_traces::decode_otlp(&payload, content_type.as_deref()).map(|spans| {
litellm_traces_clickhouse::span_rows(spans, &tenant, max_value_bytes)
})
})
.await
.map_err(|_| Error::Task)??;
let count = rows.len();
litellm_traces_clickhouse::insert_shared_rows(
&client,
&connection,
&database,
table,
InsertTable::OtelTraces,
rows,
)
.await?;
Ok(count)
},
map_error,
)
}
#[pyo3(signature = (scope, start_ms, end_ms, cursor, limit))]
fn list_traces<'py>(
&self,
py: Python<'py>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: ReadAccessParams,
start_ms: i64,
end_ms: i64,
cursor: Option<String>,
limit: u32,
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
crate::execution::run_async(
py,
async move {
litellm_traces_clickhouse::list_traces(
&client,
&connection,
&scope,
start_ms,
end_ms,
cursor.as_deref(),
limit,
)
.await
},
map_error,
)
}
fn get_trace<'py>(
&self,
py: Python<'py>,
trace_id: String,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: ReadAccessParams,
trace_ref: String,
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
crate::execution::run_async(
py,
async move {
litellm_traces_clickhouse::get_trace(
&client,
&connection,
&scope,
&trace_id,
&trace_ref,
)
.await
},
map_error,
)
}
fn get_span<'py>(
&self,
py: Python<'py>,
trace_id: String,
span_id: String,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: ReadAccessParams,
trace_ref: String,
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
crate::execution::run_async(
py,
async move {
litellm_traces_clickhouse::get_span(
&client,
&connection,
&scope,
&trace_id,
&span_id,
&trace_ref,
)
.await
},
map_error,
)
}
#[pyo3(signature = (trace_id, span_id, scope, trace_ref, cursor))]
fn get_span_error<'py>(
&self,
py: Python<'py>,
trace_id: String,
span_id: String,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: ReadAccessParams,
trace_ref: String,
cursor: Option<String>,
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
crate::execution::run_async(
py,
async move {
litellm_traces_clickhouse::get_span_error(
&client,
&connection,
&scope,
&trace_id,
&span_id,
&trace_ref,
cursor.as_deref(),
)
.await
},
map_error,
@ -232,101 +388,23 @@ impl NativeTraceStorage {
}
}
/// The `otel_traces` rows an export would be stored as, without writing them.
#[pyfunction]
pub fn trace_decode_otlp<'py>(
pub fn trace_span_rows<'py>(
py: Python<'py>,
body: &[u8],
content_type: Option<&str>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] tenant: Tenant,
max_attribute_value_bytes: usize,
) -> PyResult<Bound<'py, PyAny>> {
let spans = py
.detach(|| litellm_traces::decode_otlp(body, content_type))
.map_err(|error| match error {
litellm_traces::Error::TooLarge => PyOverflowError::new_err(error.to_string()),
_ => PyValueError::new_err(error.to_string()),
})?;
spans_to_py(py, &spans).map(Bound::into_any)
}
fn insert_rows_from_py(
value: &Bound<'_, PyAny>,
) -> PyResult<Vec<litellm_traces_clickhouse::InsertRow>> {
let mut resources = FromPythonCache::default();
value
.try_iter()?
.map(|row| {
let row = row?;
let mut fields = BTreeMap::new();
for item in row.cast::<PyMapping>()?.items()?.iter() {
let (key, value): (String, Bound<'_, PyAny>) = item.extract()?;
let converted = if matches!(
key.as_str(),
"ResourceAttributes" | "ScopeName" | "ScopeVersion"
) {
resources
.get_or_try_insert_with(&value, |value| {
litellm_host_python::from_py_argument::<serde_json::Value>(value)
.map(Shared::new)
})?
.clone()
} else {
Shared::new(litellm_host_python::from_py_argument(&value)?)
};
fields.insert(key, converted);
}
Ok(fields)
let rows = py
.detach(|| {
litellm_traces::decode_otlp(body, content_type).map(|spans| {
litellm_traces_clickhouse::span_rows(spans, &tenant, max_attribute_value_bytes)
})
})
.collect()
}
fn spans_to_py<'py>(
py: Python<'py>,
spans: &[litellm_traces::DecodedSpan],
) -> PyResult<Bound<'py, PyList>> {
let mut resources = ToPythonCache::default();
let mut scopes = ToPythonCache::default();
let result = PyList::empty(py);
for span in spans {
let resource = resources
.get_or_try_insert_with(span.resource_attributes.as_ref(), |value| {
litellm_host_python::Pythonized(value).into_pyobject(py)
})?;
let row = PyDict::new(py);
row.set_item("trace_id", &span.trace_id)?;
row.set_item("span_id", &span.span_id)?;
row.set_item("parent_span_id", &span.parent_span_id)?;
row.set_item("trace_state", &span.trace_state)?;
row.set_item("name", &span.name)?;
row.set_item("kind", &span.kind)?;
row.set_item("resource_attributes", resource)?;
for (key, value) in [
("scope_name", &span.scope_name),
("scope_version", &span.scope_version),
] {
let value = scopes.get_or_try_insert_with(value.as_ref(), |value| {
Ok(PyString::new(py, value).into_any())
})?;
row.set_item(key, value)?;
}
row.set_item("attributes", &span.attributes)?;
row.set_item("start_ns", span.start_ns)?;
row.set_item("end_ns", span.end_ns)?;
row.set_item("status_code", &span.status_code)?;
row.set_item("status_message", &span.status_message)?;
row.set_item(
"events",
litellm_host_python::Pythonized(&span.events).into_pyobject(py)?,
)?;
row.set_item(
"normalized",
litellm_host_python::Pythonized(&span.normalized).into_pyobject(py)?,
)?;
row.set_item(
"consumed_attributes",
litellm_host_python::Pythonized(&span.consumed_attributes).into_pyobject(py)?,
)?;
result.append(row)?;
}
Ok(result)
.map_err(|error| map_error(error.into()))?;
litellm_host_python::Pythonized(rows).into_pyobject(py)
}
#[cfg(test)]
@ -383,50 +461,20 @@ mod tests {
}
#[rstest]
fn insert_projection_preserves_identity_without_merging_equal_resources() {
#[case::decode_budget(Error::Decode(litellm_traces::Error::TooLarge), "OverflowError")]
#[case::invalid_export(Error::Decode(litellm_traces::Error::InvalidPayload), "ValueError")]
#[case::cursor(Error::InvalidCursor("trace"), "ValueError")]
#[case::ambiguous(Error::AmbiguousTrace, "ValueError")]
fn trace_read_and_ingest_failures_preserve_public_exception_types(
#[case] error: Error,
#[case] exception_name: &str,
) {
Python::initialize();
Python::attach(|py| {
let resource = PyDict::new(py);
resource.set_item("service.name", "shared").unwrap();
let equal_resource = resource.copy().unwrap();
let rows = PyList::empty(py);
for value in [&resource, &resource, &equal_resource] {
let row = PyDict::new(py);
row.set_item("ResourceAttributes", value).unwrap();
rows.append(row).unwrap();
}
let projected = insert_rows_from_py(rows.as_any()).unwrap();
assert!(Shared::shares_storage_with(
&projected[0]["ResourceAttributes"],
&projected[1]["ResourceAttributes"]
));
assert!(!Shared::shares_storage_with(
&projected[0]["ResourceAttributes"],
&projected[2]["ResourceAttributes"]
));
assert_eq!(projected[0], projected[2]);
});
}
#[rstest]
fn shared_conversion_preserves_every_decoded_field() {
Python::initialize();
Python::attach(|py| {
let spans = litellm_traces::decode_otlp(
include_bytes!("../../../../../tests/test_litellm/tracing/fixtures/langsmith_deep_agent_export.json"),
Some("application/json"),
).unwrap();
let expected = litellm_host_python::Pythonized(&spans)
.into_pyobject(py)
.unwrap();
let actual = spans_to_py(py, &spans).unwrap();
assert!(actual.eq(expected).unwrap());
assert_eq!(
map_error(error).get_type(py).name().unwrap(),
exception_name
);
});
}
}
#[pyfunction]
pub fn trace_normalized_field_definitions<'py>(py: Python<'py>) -> PyResult<Bound<'py, PyAny>> {
litellm_host_python::Pythonized(litellm_traces_clickhouse::NORMALIZED_FIELD_DEFINITIONS)
.into_pyobject(py)
}

View file

@ -5,8 +5,14 @@ edition.workspace = true
license.workspace = true
repository.workspace = true
[features]
schema = ["dep:schemars", "litellm-traces/schema"]
[dependencies]
macro_rules_attribute.workspace = true
schemars = { workspace = true, optional = true }
askama.workspace = true
base64.workspace = true
flate2.workspace = true
futures-util.workspace = true
hmac = "0.12.1"
@ -22,10 +28,17 @@ strum.workspace = true
thiserror.workspace = true
time = { workspace = true, features = ["formatting"] }
tokio.workspace = true
tracing.workspace = true
url.workspace = true
[dev-dependencies]
jsonschema = { version = "0.55.1", default-features = false }
litellm-http = { workspace = true, features = ["test-support"] }
rstest.workspace = true
testcontainers-modules = { version = "0.15.0", features = ["clickhouse"] }
wiremock.workspace = true
[[bin]]
name = "export-traces-clickhouse-schema"
path = "src/bin/export_schema.rs"
required-features = ["schema"]

View file

@ -0,0 +1,5 @@
ALTER TABLE {database}.otel_traces
ADD COLUMN IF NOT EXISTS WrapperCandidate Bool DEFAULT false AFTER ObservationType,
ADD COLUMN IF NOT EXISTS CallKeys Array(String) DEFAULT [] AFTER LiteLLMRequestId,
ADD COLUMN IF NOT EXISTS CallEvidence LowCardinality(String) DEFAULT '' AFTER CallKeys,
ADD COLUMN IF NOT EXISTS ToolCallId String DEFAULT '' AFTER Output

View file

@ -0,0 +1,2 @@
ALTER TABLE {database}.otel_traces
ADD COLUMN IF NOT EXISTS AgentMetadata String DEFAULT '{}' CODEC(ZSTD(3))

View file

@ -0,0 +1 @@
ALTER TABLE {database}.spend_logs MODIFY COLUMN spend Nullable(Float64) DEFAULT NULL

View file

@ -1,10 +1,21 @@
SELECT request_id, response_id, team_id, api_key, user, spend,
SELECT request_id, response_id, upstream_response_id, trace_id, span_id, team_id, api_key, user, spend,
toUnixTimestamp64Milli(start_time) AS start_ms
FROM spend_logs FINAL
FROM (
SELECT *,
-- A chat request served through the Responses API returns the upstream `resp_` id to the
-- client but logs LiteLLM's managed `resp_<base64>` id, which embeds it.
if(startsWith(response_id, 'resp_'),
extract(tryBase64Decode(substring(response_id, 6)), 'response_id:([^;]+)'),
'') AS upstream_response_id
FROM spend_logs FINAL
WHERE start_time >= fromUnixTimestamp64Milli({start_ms:Int64})
AND start_time < fromUnixTimestamp64Milli({end_ms:Int64})
AND ({all_teams:UInt8} = 1
OR ({user_id:String} != '' AND user = {user_id:String})
OR has({team_ids:Array(String)}, team_id))
)
WHERE response_id IN {response_ids:Array(String)}
AND start_time >= fromUnixTimestamp64Milli({start_ms:Int64})
AND start_time < fromUnixTimestamp64Milli({end_ms:Int64})
AND ({all_teams:UInt8} = 1
OR ({user_id:String} != '' AND user = {user_id:String})
OR has({team_ids:Array(String)}, team_id))
OR upstream_response_id IN {response_ids:Array(String)}
OR request_id IN {request_ids:Array(String)}
OR (trace_id != '' AND trace_id IN {trace_ids:Array(String)})
ORDER BY start_time DESC

View file

@ -0,0 +1,24 @@
SELECT o.TraceId AS trace_id, o.SpanId AS span_id, o.ParentSpanId AS parent_span_id, o.SpanName AS name,
o.ObservationType AS type, toUInt8(o.WrapperCandidate) AS wrapper_candidate, o.AgentName AS agent,
o.Framework AS framework, o.StatusCode AS status,
substringUTF8(o.StatusMessage, 1, 128) AS status_message,
lengthUTF8(o.StatusMessage) > 128 AS error_truncated,
toUnixTimestamp64Nano(o.Timestamp) AS start_ns, o.Duration AS duration_ns,
o.ServiceName AS service, o.InputPreview AS input_preview, o.Model AS model,
o.InputTokens AS input_tokens, o.OutputTokens AS output_tokens,
o.LiteLLMRequestId AS litellm_request_id,
o.CallKeys AS call_keys, o.CallEvidence AS call_evidence,
-- Rows written before ToolCallId keep the call id only in their attributes.
if(o.ToolCallId != '' OR o.ObservationType != 'tool', o.ToolCallId,
coalesce(nullIf(o.SpanAttributes['gen_ai.tool.call.id'], ''), nullIf(o.SpanAttributes['tool.id'], ''), ''))
AS tool_call_id,
o.UserId AS user_id, o.TeamId AS team_id, o.ApiKeyHash AS api_key_hash
FROM otel_traces AS o
WHERE o.Timestamp >= fromUnixTimestamp64Milli({start_ms:Int64})
AND o.Timestamp < fromUnixTimestamp64Milli({end_ms:Int64})
AND ({all_teams:UInt8} = 1
OR ({user_id:String} != '' AND o.UserId = {user_id:String})
OR has({team_ids:Array(String)}, o.TeamId))
AND hex(SHA256(concat(o.TeamId, char(0), o.ApiKeyHash, char(0), o.TraceId))) IN {trace_refs:Array(String)}
ORDER BY o.Timestamp, o.EngineReceivedMs, o.StatusMessage
LIMIT 1 BY o.TeamId, o.ApiKeyHash, o.TraceId, o.SpanId

View file

@ -1,5 +1,5 @@
SELECT o.SpanId AS span_id, o.ParentSpanId AS parent_span_id, o.SpanName AS name,
o.ObservationType AS type, o.AgentName AS agent,
SELECT o.TraceId AS trace_id, o.SpanId AS span_id, o.ParentSpanId AS parent_span_id, o.SpanName AS name,
o.ObservationType AS type, toUInt8(o.WrapperCandidate) AS wrapper_candidate, o.AgentName AS agent,
o.Framework AS framework, o.StatusCode AS status,
substringUTF8(o.StatusMessage, 1, 128) AS status_message,
lengthUTF8(o.StatusMessage) > 128 AS error_truncated,
@ -7,6 +7,11 @@ SELECT o.SpanId AS span_id, o.ParentSpanId AS parent_span_id, o.SpanName AS name
o.ServiceName AS service, o.InputPreview AS input_preview, o.Model AS model,
o.InputTokens AS input_tokens, o.OutputTokens AS output_tokens,
o.LiteLLMRequestId AS litellm_request_id,
o.CallKeys AS call_keys, o.CallEvidence AS call_evidence,
-- Rows written before ToolCallId keep the call id only in their attributes.
if(o.ToolCallId != '' OR o.ObservationType != 'tool', o.ToolCallId,
coalesce(nullIf(o.SpanAttributes['gen_ai.tool.call.id'], ''), nullIf(o.SpanAttributes['tool.id'], ''), ''))
AS tool_call_id,
o.UserId AS user_id, o.TeamId AS team_id, o.ApiKeyHash AS api_key_hash
FROM otel_traces AS o
WHERE o.TraceId = {trace_id:String}

View file

@ -0,0 +1,6 @@
fn main() {
println!(
"{}",
serde_json::to_string_pretty(&litellm_traces_clickhouse::wire_schema::schemas()).unwrap()
);
}

View file

@ -5,14 +5,21 @@ use litellm_storage_clickhouse::Storage;
pub struct Config {
storage: Storage,
retention_days: u32,
max_attribute_value_bytes: usize,
}
impl Config {
pub fn new(database: String, url: &str, retention_days: u32) -> Result<Self, Error> {
pub fn new(
database: String,
url: &str,
retention_days: u32,
max_attribute_value_bytes: usize,
) -> Result<Self, Error> {
super::schema_statements(&database, retention_days)?;
Ok(Self {
storage: Storage::new(database, url)?,
retention_days,
max_attribute_value_bytes,
})
}
@ -23,4 +30,9 @@ impl Config {
pub fn retention_days(&self) -> u32 {
self.retention_days
}
/// Stored span attribute and payload values longer than this are truncated with a marker.
pub fn max_attribute_value_bytes(&self) -> usize {
self.max_attribute_value_bytes
}
}

View file

@ -30,6 +30,14 @@ pub enum Error {
ProvisionFailed(u16),
#[error("ClickHouse reader provisioning transport failed")]
ProvisionTransport,
#[error("Invalid {0} cursor")]
InvalidCursor(&'static str),
#[error("Multiple traces have this ID; provide trace_ref")]
AmbiguousTrace,
#[error(transparent)]
Decode(#[from] litellm_traces::Error),
#[error("trace ingestion task failed")]
Task,
#[error(transparent)]
Storage(#[from] litellm_storage_clickhouse::Error),
#[error(transparent)]

View file

@ -1,11 +1,27 @@
macro_rules_attribute::attribute_alias! {
#[apply(wire_type)] =
#[derive(serde::Serialize, serde::Deserialize)]
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
#[apply(response_type)] =
#[derive(serde::Serialize)]
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
#[apply(request_type)] =
#[derive(serde::Deserialize)]
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
}
mod config;
mod error;
mod insert;
pub mod query;
mod query_access;
mod reads;
mod schema;
mod span_row;
mod sql;
mod table;
#[cfg(feature = "schema")]
pub mod wire_schema;
pub use config::Config;
pub use error::Error;
@ -14,8 +30,10 @@ pub use litellm_storage_clickhouse::{Connection, Parameter};
pub use litellm_traces::{QueryScope, ReadQuery};
pub use query::{QueryHelp, execute_read, query_help, query_sql};
pub use query_access::QueryReaders;
pub use reads::{get_span, get_span_error, get_trace, list_traces};
pub use schema::{
NORMALIZED_FIELD_DEFINITIONS, NormalizedFieldDefinition, ensure_schema, schema_statements,
};
pub use span_row::span_rows;
pub use sql::execute_named_read;
pub use table::TraceTable;

View file

@ -6,6 +6,7 @@ use futures_util::{
stream::{self, TryStreamExt},
};
use litellm_http::Client;
use litellm_traces::query::guide::{Example, QueryGuide, Section};
use serde::{Deserialize, Serialize, Serializer};
use serde_json::Value;
use strum::IntoEnumIterator;
@ -39,21 +40,24 @@ struct MetadataRow {
metadata: String,
}
#[derive(Deserialize)]
#[macro_rules_attribute::apply(request_type)]
struct AttributeRow {
key: String,
}
#[derive(Clone, Debug, Eq, Ord, PartialEq, PartialOrd, Serialize)]
#[macro_rules_attribute::apply(response_type)]
#[derive(Clone, Debug, Eq, Ord, PartialEq, PartialOrd)]
#[serde(untagged)]
enum PathPart {
Key(String),
Index(usize),
}
#[derive(Clone, Copy, Debug, Eq, Ord, PartialEq, PartialOrd, Serialize, strum::Display)]
#[macro_rules_attribute::apply(response_type)]
#[derive(Clone, Copy, Debug, Eq, Ord, PartialEq, PartialOrd, strum::Display)]
#[serde(rename_all = "lowercase")]
#[strum(serialize_all = "lowercase")]
#[cfg_attr(feature = "schema", schemars(rename = "MetadataValueType"))]
enum JsonKind {
Array,
Boolean,
@ -78,19 +82,23 @@ impl JsonKind {
}
}
#[derive(Clone, Copy, Debug, Serialize, strum::Display)]
#[macro_rules_attribute::apply(response_type)]
#[derive(Clone, Copy, Debug, strum::Display)]
enum MapValueType {
String,
}
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryMetadataField"))]
struct MetadataField {
path: Vec<PathPart>,
types: BTreeSet<JsonKind>,
expression: String,
}
#[derive(Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryColumn"))]
struct ColumnSchema {
name: String,
#[serde(rename = "type")]
@ -99,7 +107,9 @@ struct ColumnSchema {
details: BTreeMap<String, Value>,
}
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryTable"))]
struct TableSchema {
name: TraceTable,
columns: Vec<ColumnSchema>,
@ -114,6 +124,38 @@ enum Discovery<T> {
Unavailable(String),
}
#[cfg(feature = "schema")]
impl<T: schemars::JsonSchema> schemars::JsonSchema for Discovery<T> {
fn schema_name() -> std::borrow::Cow<'static, str> {
format!("Discovery{}", T::schema_name()).into()
}
fn json_schema(generator: &mut schemars::SchemaGenerator) -> schemars::Schema {
let mut schema = T::json_schema(generator);
schema
.as_object_mut()
.unwrap()
.get_mut("properties")
.unwrap()
.as_object_mut()
.unwrap()
.insert(
"error".into(),
serde_json::json!({"type": ["string", "null"], "default": null}),
);
schema
}
}
#[cfg(feature = "schema")]
pub(crate) fn help_schema() -> schemars::Schema {
schemars::generate::SchemaSettings::draft2020_12()
.for_serialize()
.with_transform(litellm_traces::schema::integer_bounds)
.into_generator()
.into_root_schema_for::<QueryHelp>()
}
impl<T: Serialize + Unobserved> Serialize for Discovery<T> {
fn serialize<S: Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
#[derive(Serialize)]
@ -133,7 +175,7 @@ impl<T: Serialize + Unobserved> Serialize for Discovery<T> {
}
}
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
struct MetadataSample {
fields: Vec<MetadataField>,
sampled_rows: usize,
@ -152,7 +194,9 @@ impl Unobserved for MetadataSample {
}
}
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryMetadata"))]
struct MetadataCatalog {
table: TraceTable,
column: &'static str,
@ -162,7 +206,9 @@ struct MetadataCatalog {
scope: &'static str,
}
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryAttributeField"))]
struct AttributeField {
key: String,
#[serde(rename = "type")]
@ -170,7 +216,7 @@ struct AttributeField {
expression: String,
}
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
struct AttributeSample {
fields: Vec<AttributeField>,
truncated: bool,
@ -185,7 +231,9 @@ impl Unobserved for AttributeSample {
}
}
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryAttributes"))]
struct AttributeCatalog {
table: TraceTable,
column: &'static str,
@ -195,7 +243,9 @@ struct AttributeCatalog {
scope: &'static str,
}
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryNormalizedField"))]
struct NormalizedField {
table: TraceTable,
name: &'static str,
@ -217,7 +267,9 @@ impl From<&NormalizedFieldDefinition> for NormalizedField {
}
}
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryRelationship"))]
struct Relationship {
left: &'static str,
right: &'static str,
@ -228,11 +280,13 @@ struct Relationship {
const RELATIONSHIPS: [Relationship; 1] = [Relationship {
left: "otel_traces.LiteLLMRequestId",
right: "spend_logs.response_id",
additional_predicates: "otel_traces.TeamId = spend_logs.team_id AND (otel_traces.TeamId != '' OR (otel_traces.UserId != '' AND otel_traces.UserId = spend_logs.user) OR (otel_traces.ApiKeyHash != '' AND otel_traces.ApiKeyHash = spend_logs.api_key))",
meaning: "The normalized ID is the response ID, not request_id. Cached requests can share response_id; joins may return multiple spend rows",
additional_predicates: "otel_traces.TeamId = spend_logs.team_id AND ((otel_traces.UserId != '' AND otel_traces.UserId = spend_logs.user) OR (otel_traces.ApiKeyHash != '' AND otel_traces.ApiKeyHash = spend_logs.api_key))",
meaning: "LiteLLMRequestId contains the first normalized request or provider response ID. This relationship matches response IDs only; CallKeys retains all typed identifiers. Cached requests can share response_id; joins may return multiple spend rows",
}];
#[derive(Serialize)]
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryHelp"))]
pub struct QueryHelp {
dialect: &'static str,
access: &'static str,
@ -242,8 +296,10 @@ pub struct QueryHelp {
metadata: MetadataCatalog,
attributes: Vec<AttributeCatalog>,
relationships: &'static [Relationship],
examples: [guide::Example; 5],
gotchas: [String; 11],
#[cfg_attr(feature = "schema", schemars(with = "Vec<Example>"))]
examples: [Example; 9],
#[cfg_attr(feature = "schema", schemars(with = "Vec<String>"))]
gotchas: [String; 13],
guide: String,
}
@ -423,13 +479,33 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result<Quer
attributes: &attributes,
limits: &READER_LIMITS,
};
let bodies = guide.sections()?;
let sections = [
"Live ClickHouse schema",
"Normalized span fields",
"Observed LLM call metadata",
"Observed span and resource attributes",
]
.into_iter()
.zip(&bodies)
.map(|(title, body)| Section { title, body })
.collect::<Vec<_>>();
let examples = guide.examples()?;
let gotchas = guide.gotchas()?;
let rendered = QueryGuide {
sections: &sections,
examples: &examples,
gotchas: &gotchas,
}
.render()
.map_err(|_| Error::InvalidResponse)?;
Ok(QueryHelp {
dialect: "ClickHouse SQL",
access: "Request-log visibility enforced by ClickHouse row policies; proxy admins see all rows, users see their own rows and permitted teams",
response: "ClickHouse JSON envelope: meta, data, rows, statistics; 64-bit integers may be strings",
examples: guide.examples()?,
gotchas: guide.gotchas()?,
guide: guide::render(&guide)?,
examples,
gotchas,
guide: rendered,
normalized_fields: NORMALIZED_FIELD_DEFINITIONS
.iter()
.map(NormalizedField::from)
@ -447,6 +523,33 @@ mod tests {
use rstest::rstest;
use serde_json::json;
#[cfg(feature = "schema")]
#[rstest]
#[case::observed(false)]
#[case::unavailable(true)]
fn discovery_serialization_matches_its_schema(#[case] unavailable: bool) {
let discovery = if unavailable {
Discovery::Unavailable("discovery failed".into())
} else {
Discovery::Observed(MetadataSample::unobserved())
};
let catalog = MetadataCatalog {
table: TraceTable::SpendLogs,
column: "metadata",
discovery,
sample_sql: METADATA_SQL,
scope: METADATA_SCOPE,
};
let schema = schemars::generate::SchemaSettings::draft2020_12()
.for_serialize()
.into_generator()
.into_root_schema_for::<MetadataCatalog>();
let serialized = serde_json::to_value(&catalog).unwrap();
assert!(jsonschema::is_valid(schema.as_value(), &serialized));
assert_eq!(serialized.get("error").is_some(), unavailable);
assert!(serialized["fields"].is_array());
}
#[rstest]
fn metadata_discovery_preserves_mixed_types_and_reports_invalid_rows() {
let sample = [

View file

@ -1,11 +1,15 @@
use askama::Template;
use serde::Serialize;
use litellm_traces::query::guide::Example;
use super::{AttributeCatalog, Discovery, MetadataCatalog, TableSchema};
use crate::{Error, NormalizedFieldDefinition, query_access::ReaderLimits};
#[derive(Template)]
#[template(path = "query_help.jinja", escape = "none", blocks = [
"live_schema",
"normalized_fields",
"metadata",
"attributes",
"recent_spans_name",
"recent_spans_sql",
"custom_metadata_name",
@ -16,6 +20,16 @@ use crate::{Error, NormalizedFieldDefinition, query_access::ReaderLimits};
"correlated_calls_sql",
"discover_keys_name",
"discover_keys_sql",
"recent_spend_name",
"recent_spend_sql",
"model_spend_name",
"model_spend_sql",
"trace_spend_name",
"trace_spend_sql",
"unmatched_spans_name",
"unmatched_spans_sql",
"missing_spend",
"partial_spend",
"time_window",
"reader_limits",
"reader_profile",
@ -36,14 +50,17 @@ pub(super) struct QueryGuide<'a> {
pub limits: &'a ReaderLimits,
}
#[derive(Serialize)]
pub(super) struct Example {
name: String,
sql: String,
}
impl QueryGuide<'_> {
pub fn examples(&self) -> Result<[Example; 5], Error> {
pub fn sections(&self) -> Result<[String; 4], Error> {
Ok([
render(&self.as_live_schema())?,
render(&self.as_normalized_fields())?,
render(&self.as_metadata())?,
render(&self.as_attributes())?,
])
}
pub fn examples(&self) -> Result<[Example; 9], Error> {
Ok([
Example {
name: render(&self.as_recent_spans_name())?,
@ -65,10 +82,26 @@ impl QueryGuide<'_> {
name: render(&self.as_discover_keys_name())?,
sql: render(&self.as_discover_keys_sql())?,
},
Example {
name: render(&self.as_recent_spend_name())?,
sql: render(&self.as_recent_spend_sql())?,
},
Example {
name: render(&self.as_model_spend_name())?,
sql: render(&self.as_model_spend_sql())?,
},
Example {
name: render(&self.as_trace_spend_name())?,
sql: render(&self.as_trace_spend_sql())?,
},
Example {
name: render(&self.as_unmatched_spans_name())?,
sql: render(&self.as_unmatched_spans_sql())?,
},
])
}
pub fn gotchas(&self) -> Result<[String; 11], Error> {
pub fn gotchas(&self) -> Result<[String; 13], Error> {
Ok([
render(&self.as_time_window())?,
render(&self.as_reader_limits())?,
@ -79,6 +112,8 @@ impl QueryGuide<'_> {
render(&self.as_literal_keys())?,
render(&self.as_time_units())?,
render(&self.as_spend_totals())?,
render(&self.as_missing_spend())?,
render(&self.as_partial_spend())?,
render(&self.as_trace_rollups())?,
render(&self.as_sampling())?,
])

View file

@ -1,27 +1,72 @@
use litellm_storage_clickhouse::Query;
use serde::{Deserialize, Serialize};
#[derive(Debug, Deserialize, Serialize)]
pub const LENS_QUERIES: [litellm_traces::ReadQuery; 5] = [
litellm_traces::ReadQuery::Availability,
litellm_traces::ReadQuery::Agents,
litellm_traces::ReadQuery::Sample,
litellm_traces::ReadQuery::Content,
litellm_traces::ReadQuery::Evidence,
];
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[serde(rename_all = "lowercase")]
pub enum ExecutionSource {
Traces,
Requests,
Both,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[serde(rename_all = "lowercase")]
pub enum ContentSource {
Traces,
Requests,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
pub struct LensAccessParams {
#[serde(deserialize_with = "super::number::deserialize")]
pub all_teams: u8,
#[serde(
deserialize_with = "super::number::boolean",
serialize_with = "litellm_traces::wire::serialize_flag"
)]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "litellm_traces::schema::flag")
)]
pub all_teams: bool,
pub team: String,
pub key_hash: String,
}
pub struct LensAvailability;
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[serde(deny_unknown_fields)]
pub struct LensAvailabilityParams {
#[serde(flatten)]
pub access: LensAccessParams,
}
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[cfg_attr(feature = "schema", schemars(rename = "ActivityAvailability"))]
pub struct LensAvailabilityRow {
#[serde(deserialize_with = "super::number::deserialize")]
#[serde(default, deserialize_with = "super::number::flag")]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "crate::wire_schema::boolean_flag")
)]
pub traces: u8,
#[serde(deserialize_with = "super::number::deserialize")]
#[serde(default, deserialize_with = "super::number::flag")]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "crate::wire_schema::boolean_flag")
)]
pub requests: u8,
}
@ -34,13 +79,17 @@ impl Query for LensAvailability {
pub struct LensAgents;
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[serde(deny_unknown_fields)]
pub struct LensAgentsParams {
#[serde(flatten)]
pub access: LensAccessParams,
}
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[cfg_attr(feature = "schema", schemars(rename = "AgentRow"))]
pub struct LensAgentsRow {
pub agent_name: String,
}
@ -54,11 +103,13 @@ impl Query for LensAgents {
pub struct LensSample;
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[serde(deny_unknown_fields)]
pub struct LensSampleParams {
#[serde(flatten)]
pub access: LensAccessParams,
pub source: String,
pub source: ExecutionSource,
#[serde(deserialize_with = "super::number::deserialize")]
pub start: u64,
#[serde(deserialize_with = "super::number::deserialize")]
@ -71,9 +122,14 @@ pub struct LensSampleParams {
pub execution_ids: Vec<String>,
#[serde(deserialize_with = "super::number::deserialize")]
pub sample_cap: u64,
#[serde(deserialize_with = "super::number::deserialize")]
#[serde(deserialize_with = "super::number::percent")]
#[cfg_attr(feature = "schema", schemars(range(min = 0, max = 100)))]
pub sample_percent: f64,
#[serde(deserialize_with = "super::number::deserialize")]
#[serde(deserialize_with = "super::number::flag")]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "litellm_traces::schema::flag")
)]
pub preview: u8,
pub after: String,
#[serde(deserialize_with = "super::number::deserialize")]
@ -82,26 +138,49 @@ pub struct LensSampleParams {
pub offset: u64,
}
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[cfg_attr(feature = "schema", schemars(rename = "ExecutionRow"))]
pub struct LensSampleRow {
pub source: String,
pub source: ContentSource,
pub trace_id: String,
pub team_id: String,
#[serde(default)]
pub trace_ref: String,
pub name: String,
pub start_time: String,
#[serde(deserialize_with = "super::number::deserialize")]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "crate::wire_schema::u64_number")
)]
pub span_count: u64,
#[serde(deserialize_with = "super::number::deserialize")]
#[serde(deserialize_with = "super::number::flag")]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "crate::wire_schema::flag_number")
)]
pub root_seen: u8,
#[serde(default)]
pub service: String,
#[serde(default)]
pub attributes: Vec<(String, String)>,
#[serde(deserialize_with = "super::number::deserialize")]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "crate::wire_schema::u64_number")
)]
pub eligible: u64,
#[serde(deserialize_with = "super::number::deserialize")]
#[cfg_attr(feature = "schema", schemars(skip))]
pub position: u64,
#[serde(deserialize_with = "super::number::deserialize")]
#[serde(default, deserialize_with = "super::number::deserialize")]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "crate::wire_schema::selected")
)]
pub selected: f64,
#[serde(default)]
pub selection_key: String,
}
@ -114,11 +193,13 @@ impl Query for LensSample {
pub struct LensContent;
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[serde(deny_unknown_fields)]
pub struct LensContentParams {
#[serde(flatten)]
pub access: LensAccessParams,
pub source: String,
pub source: ContentSource,
pub id: String,
pub record_team: String,
pub trace_ref: String,
@ -127,14 +208,20 @@ pub struct LensContentParams {
pub offset: u32,
}
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[cfg_attr(feature = "schema", schemars(rename = "PartRow"))]
pub struct LensContentRow {
pub span_id: String,
pub parent_span_id: String,
pub name: String,
pub kind: String,
pub content: String,
#[serde(deserialize_with = "super::number::deserialize")]
#[serde(deserialize_with = "super::number::flag")]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "crate::wire_schema::flag_number")
)]
pub truncated: u8,
}
@ -147,11 +234,13 @@ impl Query for LensContent {
pub struct LensEvidence;
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[serde(deny_unknown_fields)]
pub struct LensEvidenceParams {
#[serde(flatten)]
pub access: LensAccessParams,
pub source: String,
pub source: ContentSource,
pub id: String,
pub record_team: String,
pub trace_ref: String,
@ -159,9 +248,15 @@ pub struct LensEvidenceParams {
pub quote: String,
}
#[derive(Debug, Deserialize, Serialize)]
#[macro_rules_attribute::apply(wire_type)]
#[derive(Debug)]
#[cfg_attr(feature = "schema", schemars(rename = "CountRow"))]
pub struct LensEvidenceRow {
#[serde(deserialize_with = "super::number::deserialize")]
#[cfg_attr(
feature = "schema",
schemars(schema_with = "crate::wire_schema::u64_number")
)]
pub count: u64,
}

View file

@ -42,7 +42,8 @@ struct ListTracesRowEncoding {
pub name: String,
pub service: String,
pub input_preview: String,
pub status: String,
#[serde(serialize_with = "litellm_traces::wire::serialize_status")]
pub status: litellm_traces::SpanStatus,
#[serde(deserialize_with = "super::number::deserialize")]
pub start_ms: i64,
#[serde(deserialize_with = "super::number::deserialize")]
@ -79,18 +80,30 @@ pub use contracts::TraceSpansParams;
#[derive(Deserialize, Serialize)]
#[serde(remote = "contracts::TraceSpansRow")]
struct TraceSpansRowEncoding {
#[serde(default)]
pub trace_id: String,
pub span_id: String,
pub parent_span_id: String,
pub name: String,
#[serde(rename = "type")]
pub kind: String,
pub kind: litellm_traces::ObservationType,
#[serde(
default,
deserialize_with = "super::number::boolean",
serialize_with = "litellm_traces::wire::serialize_flag"
)]
pub wrapper_candidate: bool,
pub agent: String,
#[serde(default)]
pub framework: String,
pub status: String,
#[serde(serialize_with = "litellm_traces::wire::serialize_status")]
pub status: litellm_traces::SpanStatus,
pub status_message: String,
#[serde(deserialize_with = "super::number::deserialize")]
pub error_truncated: u8,
#[serde(
deserialize_with = "super::number::boolean",
serialize_with = "litellm_traces::wire::serialize_flag"
)]
pub error_truncated: bool,
#[serde(deserialize_with = "super::number::deserialize")]
pub start_ns: i64,
#[serde(deserialize_with = "super::number::deserialize")]
@ -103,6 +116,16 @@ struct TraceSpansRowEncoding {
#[serde(deserialize_with = "super::number::deserialize")]
pub output_tokens: u32,
pub litellm_request_id: String,
#[serde(default)]
pub call_keys: Vec<litellm_traces::CallKey>,
#[serde(
default,
deserialize_with = "litellm_traces::wire::evidence",
serialize_with = "litellm_traces::wire::serialize_evidence"
)]
pub call_evidence: Option<litellm_traces::CallEvidenceKind>,
#[serde(default)]
pub tool_call_id: String,
pub team_id: String,
pub api_key_hash: String,
pub user_id: String,
@ -158,6 +181,8 @@ struct SpendByResponseIdsParamsEncoding {
#[serde(flatten)]
pub access: contracts::ReadAccessParams,
pub response_ids: Vec<String>,
pub request_ids: Vec<String>,
pub trace_ids: Vec<String>,
#[serde(deserialize_with = "super::number::deserialize")]
pub start_ms: i64,
#[serde(deserialize_with = "super::number::deserialize")]
@ -180,11 +205,14 @@ impl From<contracts::SpendByResponseIdsParams> for SpendByResponseIdsParams {
struct SpendByResponseIdsRowEncoding {
pub request_id: String,
pub response_id: String,
pub upstream_response_id: String,
pub trace_id: String,
pub span_id: String,
pub team_id: String,
pub api_key: String,
pub user: String,
#[serde(deserialize_with = "super::number::deserialize")]
pub spend: f64,
#[serde(deserialize_with = "super::number::optional_finite")]
pub spend: Option<f64>,
#[serde(deserialize_with = "super::number::deserialize")]
pub start_ms: i64,
}
@ -203,6 +231,38 @@ impl Query for ListTraces {
const SQL: &'static str = include_str!("../../query/list_traces.sql");
}
#[derive(Deserialize, Serialize)]
#[serde(remote = "contracts::TracePageSpansParams")]
struct TracePageSpansParamsEncoding {
#[serde(flatten)]
pub access: contracts::ReadAccessParams,
pub trace_refs: Vec<String>,
#[serde(deserialize_with = "super::number::deserialize")]
pub start_ms: i64,
#[serde(deserialize_with = "super::number::deserialize")]
pub end_ms: i64,
}
#[derive(Debug, Deserialize, Serialize)]
pub struct TracePageSpansParams(
#[serde(with = "TracePageSpansParamsEncoding")] pub contracts::TracePageSpansParams,
);
impl From<contracts::TracePageSpansParams> for TracePageSpansParams {
fn from(value: contracts::TracePageSpansParams) -> Self {
Self(value)
}
}
pub struct TracePageSpans;
impl Query for TracePageSpans {
type Params = TracePageSpansParams;
type Row = TraceSpansRow;
const SQL: &'static str = include_str!("../../query/trace_page_spans.sql");
}
pub struct TraceSpans;
impl Query for TraceSpans {
@ -279,11 +339,11 @@ mod tests {
#[case::quoted(true)]
fn rows_decode_into_neutral_contracts(#[case] quoted: bool) {
round_trip::<ListTracesRow>(
json!({"trace_id": "trace", "trace_ref": "ref", "team_id": "team", "api_key_hash": "key", "user_id": "user", "name": "agent", "service": "service", "input_preview": "input", "status": "ok", "start_ms": -1, "duration_ms": 20, "span_count": u64::MAX, "agent_count": 1, "agent_invocations": 2, "agent_names": ["agent"], "frameworks": ["claude-agent-sdk"], "llm_calls": 3, "tool_calls": 4, "input_tokens": 5, "output_tokens": 6, "models": ["model"], "error_count": 0, "request_ids": ["request"]}),
json!({"trace_id": "trace", "trace_ref": "ref", "team_id": "team", "api_key_hash": "key", "user_id": "user", "name": "agent", "service": "service", "input_preview": "input", "status": "STATUS_CODE_OK", "start_ms": -1, "duration_ms": 20, "span_count": u64::MAX, "agent_count": 1, "agent_invocations": 2, "agent_names": ["agent"], "frameworks": ["claude-agent-sdk"], "llm_calls": 3, "tool_calls": 4, "input_tokens": 5, "output_tokens": 6, "models": ["model"], "error_count": 0, "request_ids": ["request"]}),
quoted,
);
round_trip::<TraceSpansRow>(
json!({"span_id": "span", "parent_span_id": "parent", "name": "agent", "type": "agent", "agent": "agent", "framework": "claude-agent-sdk", "status": "error", "status_message": "error", "error_truncated": 1, "start_ns": -1, "duration_ns": u64::MAX, "service": "service", "input_preview": "input", "model": "model", "input_tokens": u32::MAX, "output_tokens": 6, "litellm_request_id": "request", "team_id": "team", "api_key_hash": "key", "user_id": "user"}),
json!({"trace_id": "trace", "span_id": "span", "parent_span_id": "parent", "name": "agent", "type": "agent", "wrapper_candidate": 1, "agent": "agent", "framework": "claude-agent-sdk", "status": "STATUS_CODE_ERROR", "status_message": "error", "error_truncated": 1, "start_ns": -1, "duration_ns": u64::MAX, "service": "service", "input_preview": "input", "model": "model", "input_tokens": u32::MAX, "output_tokens": 6, "litellm_request_id": "request", "call_keys": ["provider_response:request"], "call_evidence": "complete", "tool_call_id": "call", "team_id": "team", "api_key_hash": "key", "user_id": "user"}),
quoted,
);
round_trip::<SpanDetailRow>(
@ -295,7 +355,7 @@ mod tests {
quoted,
);
round_trip::<SpendByResponseIdsRow>(
json!({"request_id": "request", "response_id": "response", "team_id": "team", "api_key": "key", "user": "user", "spend": 0.125, "start_ms": -1}),
json!({"request_id": "request", "response_id": "response", "upstream_response_id": "upstream", "trace_id": "trace", "span_id": "span", "team_id": "team", "api_key": "key", "user": "user", "spend": 0.125, "start_ms": -1}),
quoted,
);
}
@ -313,8 +373,36 @@ mod tests {
quoted,
);
round_trip::<SpendByResponseIdsParams>(
json!({"all_teams": 0, "user_id": "user", "team_ids": ["team-a", "team-b"], "response_ids": ["response"], "start_ms": -1, "end_ms": 10}),
json!({"all_teams": 0, "user_id": "user", "team_ids": ["team-a", "team-b"], "response_ids": ["response"], "request_ids": ["request"], "trace_ids": ["trace"], "start_ms": -1, "end_ms": 10}),
quoted,
);
}
#[rstest]
#[case::unknown(json!(null), None)]
#[case::free(json!(0), Some(0.0))]
#[case::paid(json!("0.125"), Some(0.125))]
fn spend_rows_preserve_unknown_and_known_cost(
#[case] cost: serde_json::Value,
#[case] expected: Option<f64>,
) {
let row: SpendByResponseIdsRow = serde_json::from_value(json!({
"request_id": "request", "response_id": "response", "upstream_response_id": "",
"trace_id": "trace", "span_id": "span", "team_id": "team", "api_key": "key",
"user": "user", "spend": cost, "start_ms": 0
}))
.unwrap();
assert_eq!(row.0.spend, expected);
}
#[rstest]
#[case::nan(json!("NaN"))]
#[case::infinity(json!("1e999"))]
#[case::boolean(json!(true))]
fn spend_rows_reject_invalid_cost(#[case] cost: serde_json::Value) {
let row = serde_json::from_value::<SpendByResponseIdsRow>(json!({
"request_id": "request", "response_id": "response", "upstream_response_id": "",
"trace_id": "trace", "span_id": "span", "team_id": "team", "api_key": "key",
"user": "user", "spend": cost, "start_ms": 0
}));
assert!(row.is_err());
}
}

View file

@ -18,11 +18,95 @@ where
.map_err(serde::de::Error::custom)
}
pub(super) fn optional_finite<'de, D: Deserializer<'de>>(
deserializer: D,
) -> Result<Option<f64>, D::Error> {
let value = Option::<serde_json::Value>::deserialize(deserializer)?;
let Some(value) = value else {
return Ok(None);
};
let number: f64 = deserialize(value).map_err(serde::de::Error::custom)?;
if number.is_finite() {
Ok(Some(number))
} else {
Err(serde::de::Error::custom("expected finite spend"))
}
}
pub(super) fn flag<'de, D: Deserializer<'de>>(deserializer: D) -> Result<u8, D::Error> {
match deserialize(deserializer)? {
value @ 0..=1 => Ok(value),
_ => Err(serde::de::Error::custom("expected 0 or 1")),
}
}
pub(super) fn percent<'de, D: Deserializer<'de>>(deserializer: D) -> Result<f64, D::Error> {
let value: f64 = deserialize(deserializer)?;
if value.is_finite() && (0.0..=100.0).contains(&value) {
Ok(value)
} else {
Err(serde::de::Error::custom(
"expected a finite percentage between 0 and 100",
))
}
}
pub(super) fn boolean<'de, D: Deserializer<'de>>(deserializer: D) -> Result<bool, D::Error> {
flag(deserializer).map(|value| value == 1)
}
#[cfg(test)]
mod tests {
use crate::query::named::SpanErrorRow;
use rstest::rstest;
#[rstest]
#[case::flag_zero(serde_json::json!(0), true)]
#[case::flag_one(serde_json::json!("1"), true)]
#[case::invalid_flag(serde_json::json!(2), false)]
fn access_rejects_non_boolean_flags(#[case] value: serde_json::Value, #[case] valid: bool) {
let parameters = serde_json::json!({"all_teams": value, "team": "team", "key_hash": ""});
assert_eq!(
serde_json::from_value::<crate::query::lens::LensAccessParams>(parameters).is_ok(),
valid
);
}
#[rstest]
#[case::zero(serde_json::json!(0), true)]
#[case::hundred(serde_json::json!("100"), true)]
#[case::negative(serde_json::json!(-0.1), false)]
#[case::too_large(serde_json::json!(100.1), false)]
#[case::nan(serde_json::json!("NaN"), false)]
fn sampling_rejects_invalid_percentages(#[case] value: serde_json::Value, #[case] valid: bool) {
let parameters = serde_json::json!({
"all_teams": 0, "team": "team", "key_hash": "", "source": "both", "start": 0, "end": 1,
"agent_name": "", "service": "", "filter_keys": [], "filter_values": [], "selected_team": "",
"execution_ids": [], "sample_cap": 0, "sample_percent": value, "preview": 0, "after": "",
"limit": 10, "offset": 0
});
assert_eq!(
serde_json::from_value::<crate::query::lens::LensSampleParams>(parameters).is_ok(),
valid
);
}
#[rstest]
#[case::trace("traces", true)]
#[case::request("requests", true)]
#[case::both("both", false)]
#[case::unknown("unknown", false)]
fn content_rejects_unsupported_sources(#[case] source: &str, #[case] valid: bool) {
let parameters = serde_json::json!({
"all_teams": 0, "team": "team", "key_hash": "", "source": source, "id": "id",
"record_team": "team", "trace_ref": "", "cursor": "", "offset": 0
});
assert_eq!(
serde_json::from_value::<crate::query::lens::LensContentParams>(parameters).is_ok(),
valid
);
}
#[rstest]
#[case::quoted_max(serde_json::json!(u64::MAX.to_string()), Some(u64::MAX))]
#[case::unquoted_max(serde_json::json!(u64::MAX), Some(u64::MAX))]

View file

@ -0,0 +1,368 @@
//! Scoped trace reads: the trace list, one trace resolved with its spend, and span payloads.
use std::collections::HashMap;
use base64::{Engine, engine::general_purpose::URL_SAFE};
use litellm_http::Client;
use litellm_storage_clickhouse::fetch;
use litellm_traces::{
SpanDetail, SpanErrorPage, SpendLookup, Trace, TracePage, listed_summary,
query::named as contracts, resolve_trace, to_ui_content,
};
use serde::{Deserialize, Serialize};
use crate::{
Connection, Error,
query::named::{
ListTraces, ListTracesParams, ReadAccessParams, SpanDetail as SpanDetailQuery,
SpanDetailParams, SpanError, SpanErrorParams, SpendByResponseIds, SpendByResponseIdsParams,
TraceIdentity, TraceIdentityParams, TracePageSpans, TracePageSpansParams, TraceSpans,
TraceSpansParams,
},
};
const NANOS_PER_MS: i64 = 1_000_000;
const SPEND_WINDOW_MS: i64 = 30 * 60 * 1000;
fn encode_cursor<T: Serialize>(position: &T) -> String {
URL_SAFE.encode(serde_json::to_vec(position).unwrap_or_default())
}
fn decode_cursor<T: for<'de> Deserialize<'de>>(
cursor: &str,
kind: &'static str,
) -> Result<T, Error> {
URL_SAFE
.decode(cursor)
.ok()
.and_then(|json| serde_json::from_slice(&json).ok())
.ok_or(Error::InvalidCursor(kind))
}
fn trace_position(cursor: Option<&str>) -> Result<(i64, String), Error> {
let Some(cursor) = cursor.filter(|cursor| !cursor.is_empty()) else {
return Ok((0, String::new()));
};
match decode_cursor::<(i64, String)>(cursor, "trace")? {
(start_ms, trace_ref) if start_ms > 0 && !trace_ref.is_empty() => Ok((start_ms, trace_ref)),
_ => Err(Error::InvalidCursor("trace")),
}
}
#[derive(Deserialize, Serialize)]
struct ErrorPosition {
offset: u64,
version: String,
}
fn error_position(cursor: Option<&str>) -> Result<Option<ErrorPosition>, Error> {
let Some(cursor) = cursor else {
return Ok(None);
};
let position = decode_cursor::<ErrorPosition>(cursor, "diagnostic")?;
let valid_version = position.version.len() == 64
&& position
.version
.bytes()
.all(|byte| byte.is_ascii_digit() || (b'A'..=b'F').contains(&byte));
if i64::try_from(position.offset).is_err() || !valid_version {
return Err(Error::InvalidCursor("diagnostic"));
}
Ok(Some(position))
}
/// The stored run a trace id names for this caller; ids can repeat across tenants and runs.
async fn reference(
client: &Client,
connection: &Connection,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
) -> Result<Option<String>, Error> {
if !trace_ref.is_empty() {
return Ok(Some(trace_ref.to_owned()));
}
let params = TraceIdentityParams {
access: access.clone(),
trace_id: trace_id.to_owned(),
};
let mut identities = fetch::<TraceIdentity>(client, connection, &params).await?;
if identities.len() > 1 {
return Err(Error::AmbiguousTrace);
}
Ok(identities.pop().map(|identity| identity.trace_ref))
}
/// Spend records behind the spans' calls. A failed lookup leaves cost unknown instead of failing
/// the read.
async fn spend(
client: &Client,
connection: &Connection,
access: &ReadAccessParams,
rows: &[contracts::TraceSpansRow],
) -> Vec<contracts::SpendByResponseIdsRow> {
let lookup = SpendLookup::new(rows);
let (Some(start_ns), Some(end_ns)) = (
rows.iter().map(|row| row.start_ns).min(),
rows.iter()
.map(|row| row.start_ns.saturating_add_unsigned(row.duration_ns))
.max(),
) else {
return Vec::new();
};
if lookup.is_empty() {
return Vec::new();
}
let params = SpendByResponseIdsParams::from(contracts::SpendByResponseIdsParams {
access: access.clone(),
response_ids: lookup.response_ids,
request_ids: lookup.request_ids,
trace_ids: lookup.trace_ids,
start_ms: start_ns.div_euclid(NANOS_PER_MS) - SPEND_WINDOW_MS,
end_ms: end_ns.div_euclid(NANOS_PER_MS) + SPEND_WINDOW_MS,
});
match fetch::<SpendByResponseIds>(client, connection, &params).await {
Ok(rows) => rows.into_iter().map(|row| row.0).collect(),
Err(error) => {
tracing::warn!(%error, "trace spend lookup unavailable");
Vec::new()
}
}
}
pub async fn list_traces(
client: &Client,
connection: &Connection,
access: &ReadAccessParams,
start_ms: i64,
end_ms: i64,
cursor: Option<&str>,
limit: u32,
) -> Result<TracePage, Error> {
let (cursor_ms, cursor_trace_id) = trace_position(cursor)?;
let params = ListTracesParams::from(contracts::ListTracesParams {
access: access.clone(),
start_ms,
end_ms,
cursor_ms,
cursor_trace_id,
limit,
});
let page: Vec<contracts::ListTracesRow> = fetch::<ListTraces>(client, connection, &params)
.await?
.into_iter()
.map(|row| row.0)
.collect();
let next_cursor = page
.last()
.filter(|_| page.len() == limit as usize)
.map(|last| encode_cursor(&(last.start_ms, &last.trace_ref)));
let (Some(page_start), Some(page_end)) = (
page.iter().map(|row| row.start_ms).min(),
page.iter().map(|row| row.start_ms + row.duration_ms).max(),
) else {
return Ok(TracePage {
data: Vec::new(),
next_cursor,
});
};
let span_params = TracePageSpansParams::from(contracts::TracePageSpansParams {
access: access.clone(),
trace_refs: page.iter().map(|row| row.trace_ref.clone()).collect(),
start_ms: page_start,
end_ms: page_end + 1,
});
let span_rows: Vec<contracts::TraceSpansRow> =
fetch::<TracePageSpans>(client, connection, &span_params)
.await?
.into_iter()
.map(|row| row.0)
.collect();
let spend_rows = spend(client, connection, access, &span_rows).await;
let mut by_trace: HashMap<(String, String, String), Vec<contracts::TraceSpansRow>> =
HashMap::new();
for span in span_rows {
let key = (
span.team_id.clone(),
span.api_key_hash.clone(),
span.trace_id.clone(),
);
by_trace.entry(key).or_default().push(span);
}
let data = page
.iter()
.map(|row| {
let spans = by_trace
.get(&(
row.team_id.clone(),
row.api_key_hash.clone(),
row.trace_id.clone(),
))
.map(Vec::as_slice)
.unwrap_or_default();
resolve_trace(&row.trace_id, &row.trace_ref, spans, &spend_rows)
.map_or_else(|| listed_summary(row), |trace| trace.summary)
})
.collect();
Ok(TracePage { data, next_cursor })
}
pub async fn get_trace(
client: &Client,
connection: &Connection,
access: &ReadAccessParams,
trace_id: &str,
trace_ref: &str,
) -> Result<Option<Trace>, Error> {
let Some(trace_ref) = reference(client, connection, access, trace_id, trace_ref).await? else {
return Ok(None);
};
let params = TraceSpansParams {
access: access.clone(),
trace_id: trace_id.to_owned(),
trace_ref: trace_ref.clone(),
};
let rows: Vec<contracts::TraceSpansRow> = fetch::<TraceSpans>(client, connection, &params)
.await?
.into_iter()
.map(|row| row.0)
.collect();
if rows.is_empty() {
return Ok(None);
}
let spend_rows = spend(client, connection, access, &rows).await;
Ok(resolve_trace(trace_id, &trace_ref, &rows, &spend_rows))
}
pub async fn get_span(
client: &Client,
connection: &Connection,
access: &ReadAccessParams,
trace_id: &str,
span_id: &str,
trace_ref: &str,
) -> Result<Option<SpanDetail>, Error> {
let Some(trace_ref) = reference(client, connection, access, trace_id, trace_ref).await? else {
return Ok(None);
};
let params = SpanDetailParams {
access: access.clone(),
trace_id: trace_id.to_owned(),
trace_ref,
span_id: span_id.to_owned(),
};
let row = fetch::<SpanDetailQuery>(client, connection, &params)
.await?
.into_iter()
.next();
Ok(row.map(|row| SpanDetail {
input_ui: to_ui_content(&row.input),
output_ui: to_ui_content(&row.output),
span_id: row.span_id,
input: row.input,
output: row.output,
attributes: row.attributes,
}))
}
pub async fn get_span_error(
client: &Client,
connection: &Connection,
access: &ReadAccessParams,
trace_id: &str,
span_id: &str,
trace_ref: &str,
cursor: Option<&str>,
) -> Result<Option<SpanErrorPage>, Error> {
let position = error_position(cursor)?;
let Some(trace_ref) = reference(client, connection, access, trace_id, trace_ref).await? else {
return Ok(None);
};
let offset = position.as_ref().map_or(0, |position| position.offset);
let params = SpanErrorParams::from(contracts::SpanErrorParams {
access: access.clone(),
trace_id: trace_id.to_owned(),
trace_ref,
span_id: span_id.to_owned(),
error_offset: offset,
error_version: position
.map(|position| position.version)
.unwrap_or_default(),
});
let Some(row) = fetch::<SpanError>(client, connection, &params)
.await?
.into_iter()
.next()
else {
return Ok(None);
};
let row = row.0;
let next_offset = offset + row.message.chars().count() as u64;
let next_cursor = (next_offset < row.total_chars).then(|| {
encode_cursor(&ErrorPosition {
offset: next_offset,
version: row.version,
})
});
Ok(Some(SpanErrorPage {
span_id: row.span_id,
message: row.message,
total_chars: row.total_chars,
next_cursor,
}))
}
#[cfg(test)]
mod tests {
use rstest::rstest;
use super::*;
#[rstest]
fn trace_cursor_round_trips_the_last_listed_run() {
let cursor = encode_cursor(&(1_790_742_989_377_i64, "4bad42b84e9de3ba46fc870185f8f023"));
assert_eq!(
trace_position(Some(&cursor)).unwrap(),
(
1_790_742_989_377,
"4bad42b84e9de3ba46fc870185f8f023".to_owned()
)
);
assert_eq!(trace_position(None).unwrap(), (0, String::new()));
assert_eq!(trace_position(Some("")).unwrap(), (0, String::new()));
}
#[rstest]
#[case::not_base64("abc")]
#[case::not_json("bm90LWpzb24=")]
#[case::numeric_reference("WzEsIDJd")]
#[case::zero_start("WzAsICJ0Il0=")]
fn malformed_trace_cursors_are_rejected(#[case] cursor: &str) {
assert!(matches!(
trace_position(Some(cursor)),
Err(Error::InvalidCursor("trace"))
));
}
#[rstest]
#[case::not_base64("garbage")]
#[case::missing_fields("e30=")]
#[case::not_an_object("WzEsMl0=")]
fn malformed_diagnostic_cursors_are_rejected(#[case] cursor: &str) {
assert!(matches!(
error_position(Some(cursor)),
Err(Error::InvalidCursor("diagnostic"))
));
}
#[rstest]
#[case::lowercase_version("a".repeat(64))]
#[case::short_version("A".repeat(63))]
fn diagnostic_cursor_requires_a_content_version(#[case] version: String) {
let cursor = encode_cursor(&ErrorPosition { offset: 1, version });
assert!(matches!(
error_position(Some(&cursor)),
Err(Error::InvalidCursor("diagnostic"))
));
}
}

View file

@ -78,12 +78,18 @@ pub struct NormalizedFieldDefinition {
pub meaning: &'static str,
}
pub const NORMALIZED_FIELD_DEFINITIONS: [NormalizedFieldDefinition; 9] = [
pub const NORMALIZED_FIELD_DEFINITIONS: [NormalizedFieldDefinition; 15] = [
NormalizedFieldDefinition {
name: "observation_type",
clickhouse_column: "ObservationType",
clickhouse_type: "LowCardinality(String)",
meaning: "Agent, LLM, tool, chain, or framework span",
meaning: "Operation recorded by the span, including agent, model, tool, retrieval and evaluation steps",
},
NormalizedFieldDefinition {
name: "wrapper_candidate",
clickhouse_column: "WrapperCandidate",
clickhouse_type: "Bool",
meaning: "Span may only wrap the operation it names; the trace graph decides",
},
NormalizedFieldDefinition {
name: "agent_name",
@ -97,12 +103,30 @@ pub const NORMALIZED_FIELD_DEFINITIONS: [NormalizedFieldDefinition; 9] = [
clickhouse_type: "LowCardinality(String)",
meaning: "Agent framework or SDK that emitted this span, e.g. claude-agent-sdk",
},
NormalizedFieldDefinition {
name: "agent_metadata",
clickhouse_column: "AgentMetadata",
clickhouse_type: "String",
meaning: "Typed agent metadata as JSON, including thread, subagent, runtime and repository identity",
},
NormalizedFieldDefinition {
name: "litellm_request_id",
clickhouse_column: "LiteLLMRequestId",
clickhouse_type: "String",
meaning: "LiteLLM response ID used to link a span to a spend log",
},
NormalizedFieldDefinition {
name: "call_keys",
clickhouse_column: "CallKeys",
clickhouse_type: "Array(String)",
meaning: "Model requests the span accounts for, as kind:id (litellm_request, provider_response, transport)",
},
NormalizedFieldDefinition {
name: "call_evidence",
clickhouse_column: "CallEvidence",
clickhouse_type: "LowCardinality(String)",
meaning: "Whether CallKeys are all of the span's requests: complete, partial or unknown",
},
NormalizedFieldDefinition {
name: "model",
clickhouse_column: "Model",
@ -127,10 +151,22 @@ pub const NORMALIZED_FIELD_DEFINITIONS: [NormalizedFieldDefinition; 9] = [
clickhouse_type: "String",
meaning: "Normalized input payload",
},
NormalizedFieldDefinition {
name: "input_preview",
clickhouse_column: "InputPreview",
clickhouse_type: "String",
meaning: "Latest user message of the input, else the input's first characters",
},
NormalizedFieldDefinition {
name: "output",
clickhouse_column: "Output",
clickhouse_type: "String",
meaning: "Normalized output payload",
},
NormalizedFieldDefinition {
name: "tool_call_id",
clickhouse_column: "ToolCallId",
clickhouse_type: "String",
meaning: "Tool call the span executes, shared by instrumentations recording the same call",
},
];

View file

@ -0,0 +1,217 @@
//! Decoded spans as `otel_traces` rows: payloads capped, the sending tenant stamped over whatever
//! the export claimed, and resource maps shared across the rows that came from one resource.
use std::collections::{BTreeMap, HashMap};
use litellm_traces::{
CallEvidence, CallKey, DecodedEvent, DecodedSpan, Shared, SharedIdentity, Tenant,
truncate_messages, truncate_value,
};
use serde::Serialize;
use serde_json::{Map, Value};
use crate::InsertRow;
/// Converts each distinct shared source once; keeping the source pins its identity.
struct SharedValues<T>(HashMap<SharedIdentity, (Shared<T>, Shared<Value>)>);
impl<T: Clone> SharedValues<T> {
fn new() -> Self {
Self(HashMap::new())
}
fn get(&mut self, source: &Shared<T>, convert: impl FnOnce(&T) -> Value) -> Shared<Value> {
self.0
.entry(source.identity())
.or_insert_with(|| (source.clone(), Shared::new(convert(source))))
.1
.clone()
}
}
fn stamped(attributes: &BTreeMap<String, String>, tenant: &Tenant) -> Value {
let mut stamped: Map<String, Value> = attributes
.iter()
.map(|(key, value)| (key.clone(), Value::from(value.as_str())))
.collect();
for (key, value) in [
("litellm.team_id", &tenant.team_id),
("litellm.api_key_hash", &tenant.api_key_hash),
("litellm.org_id", &tenant.org_id),
("litellm.user_id", &tenant.user_id),
] {
stamped.insert(key.to_owned(), Value::from(value.as_str()));
}
Value::Object(stamped)
}
fn exception_message(events: &[DecodedEvent]) -> String {
events
.iter()
.find(|event| event.name == "exception")
.and_then(|event| {
event
.attributes
.get("exception.message")
.filter(|message| !message.is_empty())
.or_else(|| event.attributes.get("exception.type"))
})
.cloned()
.unwrap_or_default()
}
fn json<T: Serialize>(value: T) -> Value {
serde_json::to_value(value).unwrap_or(Value::Null)
}
fn present_fields<T: Serialize>(value: &T) -> String {
match json(value) {
Value::Object(fields) => Value::Object(
fields
.into_iter()
.filter(|(_, value)| !value.is_null())
.collect(),
)
.to_string(),
other => other.to_string(),
}
}
pub fn span_rows(
spans: Vec<DecodedSpan>,
tenant: &Tenant,
max_value_bytes: usize,
) -> Vec<InsertRow> {
let mut resources = SharedValues::new();
let mut scopes = SharedValues::new();
spans
.into_iter()
.map(|span| {
let normalized = span.normalized;
let service = span
.resource_attributes
.get("service.name")
.cloned()
.unwrap_or_default();
let status_message = if span.status_message.is_empty() {
exception_message(&span.events)
} else {
span.status_message
};
let attributes: Map<String, Value> = span
.attributes
.into_iter()
.filter(|(key, _)| !span.consumed_attributes.contains(&key.as_str()))
.map(|(key, value)| (key, Value::String(truncate_value(value, max_value_bytes))))
.collect();
let shared = [
(
"ResourceAttributes",
resources.get(&span.resource_attributes, |attributes| {
stamped(attributes, tenant)
}),
),
(
"ScopeName",
scopes.get(&span.scope_name, |name| Value::from(name.as_str())),
),
(
"ScopeVersion",
scopes.get(&span.scope_version, |version| Value::from(version.as_str())),
),
];
let owned = [
("Timestamp", json(span.start_ns)),
("TraceId", Value::String(span.trace_id)),
("SpanId", Value::String(span.span_id)),
("ParentSpanId", Value::String(span.parent_span_id)),
("TraceState", Value::String(span.trace_state)),
("SpanName", Value::String(span.name)),
("SpanKind", Value::String(span.kind)),
("ServiceName", Value::String(service)),
("SpanAttributes", Value::Object(attributes)),
("Duration", json(span.end_ns - span.start_ns)),
("StatusCode", Value::String(span.status_code)),
("StatusMessage", Value::String(status_message)),
("TeamId", Value::from(tenant.team_id.as_str())),
("ApiKeyHash", Value::from(tenant.api_key_hash.as_str())),
("UserId", Value::from(tenant.user_id.as_str())),
("ObservationType", json(normalized.observation_type)),
(
"WrapperCandidate",
Value::Bool(normalized.wrapper_candidate),
),
(
"AgentName",
Value::String(normalized.agent_name.unwrap_or_default()),
),
(
"Framework",
Value::String(
normalized
.framework
.map(|integration| integration.to_string())
.unwrap_or_default(),
),
),
(
"AgentMetadata",
Value::String(present_fields(&normalized.agent_metadata)),
),
(
"LiteLLMRequestId",
Value::String(request_id(&normalized.calls).to_owned()),
),
(
"CallKeys",
json(
normalized
.calls
.key_set()
.into_iter()
.flatten()
.collect::<Vec<_>>(),
),
),
("CallEvidence", json(normalized.calls.kind())),
("Model", Value::String(normalized.model.unwrap_or_default())),
("InputTokens", Value::from(normalized.input_tokens)),
("OutputTokens", Value::from(normalized.output_tokens)),
(
"Input",
Value::String(truncate_messages(normalized.input, max_value_bytes)),
),
("InputPreview", Value::String(normalized.input_preview)),
(
"Output",
Value::String(truncate_value(normalized.output, max_value_bytes)),
),
(
"ToolCallId",
Value::String(normalized.tool_call_id.unwrap_or_default()),
),
];
shared
.into_iter()
.chain(
owned
.into_iter()
.map(|(column, value)| (column, Shared::new(value))),
)
.map(|(column, value)| (column.to_owned(), value))
.collect()
})
.collect()
}
fn request_id(evidence: &CallEvidence) -> &str {
evidence
.key_set()
.into_iter()
.flatten()
.find_map(|key| match key {
CallKey::LiteLlmRequest(id) | CallKey::ProviderResponse(id) => Some(id.as_str()),
CallKey::Transport => None,
})
.unwrap_or_default()
}

View file

@ -19,6 +19,9 @@ pub async fn execute_named_read(
named_json::<TraceIdentity>(client, connection, parameters).await
}
ReadQuery::TraceSpans => named_json::<TraceSpans>(client, connection, parameters).await,
ReadQuery::TracePageSpans => {
named_json::<TracePageSpans>(client, connection, parameters).await
}
ReadQuery::SpanDetail => named_json::<SpanDetail>(client, connection, parameters).await,
ReadQuery::SpanError => named_json::<SpanError>(client, connection, parameters).await,
ReadQuery::SpendByResponseIds => {

View file

@ -1,12 +1,7 @@
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(rename = "TraceTableName"))]
#[derive(
Clone,
Copy,
Debug,
serde::Serialize,
strum::Display,
strum::AsRefStr,
strum::EnumIter,
strum::IntoStaticStr,
Clone, Copy, Debug, strum::Display, strum::AsRefStr, strum::EnumIter, strum::IntoStaticStr,
)]
#[serde(rename_all = "snake_case")]
#[strum(serialize_all = "snake_case")]

View file

@ -0,0 +1,119 @@
use std::collections::BTreeMap;
use schemars::{JsonSchema, Schema, SchemaGenerator, generate::SchemaSettings};
use serde_json::json;
use crate::query::lens;
fn quoted_u64() -> Schema {
let upper = u64::MAX.to_string();
let alternatives = upper
.char_indices()
.filter_map(|(index, digit)| {
let lower = if index == 0 { '1' } else { '0' };
if digit <= lower {
return None;
}
Some(format!(
"{}[{}-{}][0-9]{{{}}}",
&upper[..index],
lower,
char::from(digit as u8 - 1),
upper.len() - index - 1
))
})
.collect::<Vec<_>>()
.join("|");
json!({
"type": "string",
"pattern": format!("^(?:0|[1-9][0-9]{{0,{}}}|{alternatives}|{upper})$", upper.len() - 2),
})
.try_into()
.unwrap()
}
fn numeric_wire(normalized: Schema, python_type: String) -> Schema {
json!({
"anyOf": [normalized, quoted_u64()],
"x-python-normalized": {"type": python_type, "minimum": 0, "maximum": u64::MAX},
})
.try_into()
.unwrap()
}
pub(crate) fn u64_number(generator: &mut SchemaGenerator) -> Schema {
numeric_wire(u64::json_schema(generator), "int".to_owned())
}
pub(crate) fn flag_number(_: &mut SchemaGenerator) -> Schema {
json!({
"anyOf": [{"type": "integer", "enum": [0, 1]}, {"type": "string", "enum": ["0", "1"]}],
"x-python-normalized": {"type": "int", "minimum": 0, "maximum": 1}
})
.try_into()
.unwrap()
}
pub(crate) fn boolean_flag(_: &mut SchemaGenerator) -> Schema {
json!({
"anyOf": [{"type": "boolean"}, {"type": "integer", "enum": [0, 1]}, {"type": "string", "enum": ["0", "1"]}],
"default": false,
"x-python-normalized": {"type": "bool"}
}).try_into().unwrap()
}
pub(crate) fn selected(generator: &mut SchemaGenerator) -> Schema {
u64_number(generator)
}
fn received<T: JsonSchema>() -> Schema {
SchemaSettings::draft2020_12()
.for_deserialize()
.with_transform(litellm_traces::schema::integer_bounds)
.into_generator()
.into_root_schema_for::<T>()
}
pub fn schemas() -> BTreeMap<&'static str, Schema> {
BTreeMap::from([
("ReadQueryName", json!({"$schema": "https://json-schema.org/draft/2020-12/schema", "title": "ReadQueryName", "type": "string", "enum": lens::LENS_QUERIES.map(|query| query.to_string())}).try_into().unwrap()),
("LensAccessParams", received::<lens::LensAccessParams>()),
("LensSampleParams", received::<lens::LensSampleParams>()),
("LensContentParams", received::<lens::LensContentParams>()),
("LensEvidenceParams", received::<lens::LensEvidenceParams>()),
(
"ActivityAvailability",
received::<lens::LensAvailabilityRow>(),
),
("ExecutionRow", received::<lens::LensSampleRow>()),
("PartRow", received::<lens::LensContentRow>()),
("CountRow", received::<lens::LensEvidenceRow>()),
("AgentRow", received::<lens::LensAgentsRow>()),
("TraceQueryHelp", crate::query::help_schema()),
])
}
#[cfg(test)]
mod tests {
use super::*;
use rstest::rstest;
#[rstest]
#[case::zero(json!(0), true)]
#[case::quoted_zero(json!("0"), true)]
#[case::maximum(json!(u64::MAX), true)]
#[case::quoted_maximum(json!(u64::MAX.to_string()), true)]
#[case::negative(json!(-1), false)]
#[case::overflow(json!((u128::from(u64::MAX) + 1).to_string()), false)]
#[case::fraction(json!(1.5), false)]
fn count_schema_enforces_the_native_range(
#[case] value: serde_json::Value,
#[case] valid: bool,
) {
let schema = received::<lens::LensEvidenceRow>();
assert_eq!(
jsonschema::is_valid(schema.as_value(), &json!({"count": value})),
valid
);
}
}

View file

@ -1,65 +1,268 @@
Trace SQL query guide
Live ClickHouse schema
{% for table in tables %}
{% block live_schema -%}
{% for table in tables -%}
{{ table.name }}
{% for column in table.columns %}{{ column.name }}: {{ column.kind }}
{% endfor %}{% endfor %}
Normalized span fields
{% for field in normalized_fields %}{{ field.name }}: otel_traces.{{ field.clickhouse_column }} ({{ field.clickhouse_type }})
{{ field.meaning }}
{% for column in table.columns -%}
{{ column.name }}: {{ column.kind }}
{% endfor %}
Observed LLM call metadata
{% endfor -%}
{%- endblock %}
{% block normalized_fields -%}
{% for field in normalized_fields -%}
{{ field.name }}: otel_traces.{{ field.clickhouse_column }} ({{ field.clickhouse_type }})
{{ field.meaning }}
{% endfor -%}
{%- endblock %}
{% block metadata -%}
{{ metadata.scope }}
{% match metadata.discovery %}{% when Discovery::Unavailable(error) %}Metadata discovery unavailable: {{ error }}
{% when Discovery::Observed(sample) %}Sampled rows: {{ sample.sampled_rows }}; invalid JSON rows: {{ sample.invalid_json_rows }}; truncated: {{ sample.truncated }}
{% if sample.fields.is_empty() %}No metadata paths found in the sampled rows
{% else %}{% for field in sample.fields %}{{ field.expression }}: {% for kind in field.types %}{{ kind }} {% endfor %}
{% endfor %}{% endif %}{% endmatch %}
Observed span and resource attributes
{% for catalog in attributes %}{{ catalog.table }}.{{ catalog.column }}
Sampling SQL:
{{ metadata.sample_sql }}
{% match metadata.discovery -%}
{% when Discovery::Unavailable(error) -%}
Metadata discovery unavailable: {{ error }}
{% when Discovery::Observed(sample) -%}
Sampled rows: {{ sample.sampled_rows }}; invalid JSON rows: {{ sample.invalid_json_rows }}; truncated: {{ sample.truncated }}
{% if sample.fields.is_empty() -%}
No metadata paths found in the sampled rows
{% else -%}
{% for field in sample.fields -%}
{{ field.expression }}: {{ field.types|join(", ") }}
{% endfor -%}
{% endif -%}
{% endmatch -%}
{%- endblock %}
{% block attributes -%}
{% for catalog in attributes -%}
{{ catalog.table }}.{{ catalog.column }}
{{ catalog.scope }}
{% match catalog.discovery %}{% when Discovery::Unavailable(error) %}Attribute discovery unavailable: {{ error }}
{% when Discovery::Observed(sample) %}{% if sample.fields.is_empty() %}No attribute keys found in the sampled spans
{% else %}{% for field in sample.fields %}{{ field.expression }}: {{ field.kind }}
{% endfor %}{% endif %}{% endmatch %}{% endfor %}
Examples
Discovery SQL:
{{ catalog.discovery_sql }}
{% match catalog.discovery -%}
{% when Discovery::Unavailable(error) -%}
Attribute discovery unavailable: {{ error }}
{% when Discovery::Observed(sample) -%}
Truncated: {{ sample.truncated }}
{% if sample.fields.is_empty() -%}
No attribute keys found in the sampled spans
{% else -%}
{% for field in sample.fields -%}
{{ field.expression }}: {{ field.kind }}
{% endfor -%}
{% endif -%}
{% endmatch %}
{% endfor -%}
{%- endblock %}
{% block recent_spans_name %}Recent normalized LLM spans{% endblock %}
{% block recent_spans_sql %}SELECT TraceId, SpanId, Model, InputTokens, OutputTokens, Duration / 1000000 AS duration_ms FROM otel_traces WHERE Timestamp >= now() - INTERVAL 1 DAY AND ObservationType = 'llm' ORDER BY Timestamp DESC LIMIT 100{% endblock %}
{% block recent_spans_name -%}
Recent normalized LLM spans
{%- endblock %}
{% block custom_metadata_name %}Find calls by custom metadata{% endblock %}
{% block custom_metadata_sql %}SELECT request_id, response_id, model, spend, JSONExtractString(metadata, 'project') AS project FROM spend_logs FINAL WHERE start_time >= now() - INTERVAL 1 DAY AND JSONHas(metadata, 'project') AND JSONExtractString(metadata, 'project') = 'example' ORDER BY start_time DESC LIMIT 100{% endblock %}
{% block recent_spans_sql -%}
SELECT
TraceId, SpanId, Model, InputTokens, OutputTokens,
Duration / 1000000 AS duration_ms
FROM otel_traces
WHERE Timestamp >= now() - INTERVAL 1 DAY
AND ObservationType = 'llm'
ORDER BY Timestamp DESC
LIMIT 100
{%- endblock %}
{% block nested_metadata_name %}Nested metadata with unknown types{% endblock %}
{% block nested_metadata_sql %}SELECT request_id, JSONType(metadata, 'labels', 'priority') AS type, JSONExtractRaw(metadata, 'labels', 'priority') AS value FROM spend_logs FINAL WHERE start_time >= now() - INTERVAL 1 DAY AND JSONHas(metadata, 'labels', 'priority') LIMIT 100{% endblock %}
{% block custom_metadata_name -%}
Find calls by custom metadata
{%- endblock %}
{% block correlated_calls_name %}Traces correlated with LLM call metadata{% endblock %}
{% block correlated_calls_sql %}SELECT t.TraceId, t.SpanId, s.request_id, s.spend, s.metadata FROM otel_traces AS t INNER JOIN (SELECT * FROM spend_logs FINAL WHERE start_time >= now() - INTERVAL 1 DAY) AS s ON t.LiteLLMRequestId = s.response_id AND t.TeamId = s.team_id AND (t.TeamId != '' OR (t.UserId != '' AND t.UserId = s.user) OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key)) WHERE t.Timestamp >= now() - INTERVAL 1 DAY AND t.LiteLLMRequestId != '' AND JSONExtractString(s.metadata, 'project') = 'example' LIMIT 100{% endblock %}
{% block custom_metadata_sql -%}
SELECT
request_id, response_id, model, spend, JSONExtractString(metadata, 'project') AS project
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
AND JSONHas(metadata, 'project')
AND JSONExtractString(metadata, 'project') = 'example'
ORDER BY start_time DESC
LIMIT 100
{%- endblock %}
{% block discover_keys_name %}Discover metadata keys over a different window{% endblock %}
{% block discover_keys_sql %}SELECT DISTINCT arrayJoin(JSONExtractKeys(metadata)) AS key FROM spend_logs FINAL WHERE start_time >= now() - INTERVAL 30 DAY ORDER BY key LIMIT 200{% endblock %}
{% block nested_metadata_name -%}
Nested metadata with unknown types
{%- endblock %}
Gotchas
{% block nested_metadata_sql -%}
SELECT
request_id,
JSONType(metadata, 'labels', 'priority') AS type,
JSONExtractRaw(metadata, 'labels', 'priority') AS value
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
AND JSONHas(metadata, 'labels', 'priority')
LIMIT 100
{%- endblock %}
{% block time_window %}Always bound Timestamp or start_time and use LIMIT; add TeamId/ApiKeyHash or team_id/api_key filters when investigating one tenant{% endblock %}
{% block correlated_calls_name -%}
Traces correlated with LLM call metadata
{%- endblock %}
{% block reader_limits %}The reader enforces {{ limits.result_rows }} result rows, {{ limits.result_mib() }} MiB response bytes, {{ limits.memory_mib() }} MiB memory and a {{ limits.execution_seconds }} second query limit; exceeding limits fails instead of returning partial results{% endblock %}
{% block correlated_calls_sql -%}
SELECT t.TraceId, t.SpanId, s.request_id, s.spend, s.metadata
FROM otel_traces AS t
INNER JOIN (
SELECT *
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
) AS s
ON t.LiteLLMRequestId = s.response_id
AND t.TeamId = s.team_id
AND ((t.UserId != '' AND t.UserId = s.user)
OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key))
WHERE t.Timestamp >= now() - INTERVAL 1 DAY
AND t.LiteLLMRequestId != ''
LIMIT 100
{%- endblock %}
{% block reader_profile %}LiteLLM provisions SELECT-only readers from the configured ClickHouse connection and enforces request-log visibility through row policies. Callers see their own user rows and permitted teams. Provisioning requires CREATE USER, ALTER USER, CREATE ROW POLICY, and GRANT SELECT permissions{% endblock %}
{% block discover_keys_name -%}
Discover metadata keys over a different window
{%- endblock %}
{% block output_format %}Do not add FORMAT clauses; the endpoint requires ClickHouse JSON output{% endblock %}
{% block discover_keys_sql -%}
SELECT
DISTINCT arrayJoin(JSONExtractKeys(metadata)) AS key
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 30 DAY
ORDER BY key
LIMIT 200
{%- endblock %}
{% block json_values %}metadata is a JSON-encoded String; use JSONHas before typed extraction to distinguish missing values from empty strings, zero and false{% endblock %}
{% block recent_spend_name -%}
Recent spend records
{%- endblock %}
{% block map_values %}SpanAttributes and ResourceAttributes are Map(String, String); missing map keys return an empty string, so use mapContains for existence checks{% endblock %}
{% block recent_spend_sql -%}
SELECT
request_id, response_id, trace_id, span_id, model, spend,
prompt_tokens, completion_tokens, status,
JSONExtractBool(metadata, 'synthetic_spend') AS synthetic_spend
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
ORDER BY start_time DESC, request_id
LIMIT 100
{%- endblock %}
{% block literal_keys %}Use the discovered path components as separate JSONExtract arguments; a dot inside a key is literal, not a path separator{% endblock %}
{% block model_spend_name -%}
Spend and tokens by model
{%- endblock %}
{% block time_units %}Duration is nanoseconds; Timestamp has nanosecond precision, spend start_time has millisecond precision{% endblock %}
{% block model_spend_sql -%}
SELECT
team_id, model, requests, unknown_cost_requests,
if(unknown_cost_requests = 0, recorded_spend, NULL) AS spend,
input_tokens, output_tokens
FROM (
SELECT
team_id, model, count() AS requests,
countIf(isNull(spend) OR NOT isFinite(spend)) AS unknown_cost_requests,
sum(spend) AS recorded_spend,
sum(prompt_tokens) AS input_tokens,
sum(completion_tokens) AS output_tokens
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
GROUP BY team_id, model
)
ORDER BY team_id, model
LIMIT 100
{%- endblock %}
{% block spend_totals %}Use spend_logs FINAL to collapse replacement rows before totals. Shared response IDs and multiple spans can multiply costs in joins; require one spend match per response ID and ownership before aggregating. Missing IDs or costs leave totals unknown{% endblock %}
{% block trace_spend_name -%}
Recorded spend by trace
{%- endblock %}
{% block trace_rollups %}agent_traces_by_key uses SimpleAggregateFunction columns; group by TeamId, ApiKeyHash and TraceId, using min(StartTs), max(EndTs), sum(SpanCount) and groupUniqArrayArray(Models). Do not use Merge combinators{% endblock %}
{% block trace_spend_sql -%}
SELECT
team_id, api_key, trace_id, count() AS requests,
countIf(isNull(spend) OR NOT isFinite(spend)) AS unknown_cost_requests,
if(unknown_cost_requests = 0, sum(spend), NULL) AS recorded_spend
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
AND trace_id != ''
GROUP BY team_id, api_key, trace_id
ORDER BY team_id, api_key, trace_id
LIMIT 100
{%- endblock %}
{% block sampling %}Discovery is sampled, contains no metadata values, and is not an exhaustive schema. Edit the supplied discovery SQL for older data or nested JSONExtractKeys(metadata, 'parent'){% endblock %}
{% block unmatched_spans_name -%}
LLM spans without a direct spend match
{%- endblock %}
{% block unmatched_spans_sql -%}
SELECT
t.TraceId, t.SpanId, t.Model, t.LiteLLMRequestId,
t.InputTokens, t.OutputTokens
FROM otel_traces AS t
LEFT ANTI JOIN (
SELECT *
FROM spend_logs FINAL
WHERE start_time >= now() - INTERVAL 1 DAY
) AS s
ON t.TeamId = s.team_id
AND ((t.UserId != '' AND t.UserId = s.user)
OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key))
AND t.LiteLLMRequestId != ''
AND (t.LiteLLMRequestId = s.response_id OR t.LiteLLMRequestId = s.request_id)
WHERE t.Timestamp >= now() - INTERVAL 1 DAY
AND t.ObservationType = 'llm'
ORDER BY t.Timestamp DESC, t.SpanId
LIMIT 100
{%- endblock %}
{% block time_window -%}
Always bound Timestamp or start_time and use LIMIT; add TeamId/ApiKeyHash or team_id/api_key filters when investigating one tenant
{%- endblock %}
{% block reader_limits -%}
The reader enforces {{ limits.result_rows }} result rows, {{ limits.result_mib() }} MiB response bytes, {{ limits.memory_mib() }} MiB memory and a {{ limits.execution_seconds }} second query limit; exceeding limits fails instead of returning partial results
{%- endblock %}
{% block reader_profile -%}
LiteLLM provisions SELECT-only readers from the configured ClickHouse connection and enforces request-log visibility through row policies. Callers see their own user rows and permitted teams. Provisioning requires CREATE USER, ALTER USER, CREATE ROW POLICY, and GRANT SELECT permissions
{%- endblock %}
{% block output_format -%}
Do not add FORMAT clauses; the endpoint requires ClickHouse JSON output
{%- endblock %}
{% block json_values -%}
metadata is a JSON-encoded String; use JSONHas before typed extraction to distinguish missing values from empty strings, zero and false
{%- endblock %}
{% block map_values -%}
SpanAttributes and ResourceAttributes are Map(String, String); missing map keys return an empty string, so use mapContains for existence checks
{%- endblock %}
{% block literal_keys -%}
Use the discovered path components as separate JSONExtract arguments; a dot inside a key is literal, not a path separator
{%- endblock %}
{% block time_units -%}
Duration is nanoseconds; Timestamp has nanosecond precision, spend start_time has millisecond precision
{%- endblock %}
{% block missing_spend -%}
Token usage does not establish billed spend. OTLP exports without companion spend_logs rows have unknown cost; synthetic fixture spend is marked by metadata.synthetic_spend
{%- endblock %}
{% block partial_spend -%}
Recorded spend by trace totals only requests whose spend_logs.trace_id is populated. Direct ID joins do not resolve every CallKeys entry, managed Responses IDs, or transport correlation. Use the trace detail API for resolved totals; unmatched spans are a starting point for investigation
{%- endblock %}
{% block spend_totals -%}
Use spend_logs FINAL to collapse replacement rows before totals. Shared response IDs and multiple spans can multiply costs in joins; require one spend match per response ID and ownership before aggregating. Missing IDs or costs leave totals unknown
{%- endblock %}
{% block trace_rollups -%}
agent_traces_by_key uses SimpleAggregateFunction columns; group by TeamId, ApiKeyHash and TraceId, using min(StartTs), max(EndTs), sum(SpanCount) and groupUniqArrayArray(Models). Do not use Merge combinators
{%- endblock %}
{% block sampling -%}
Discovery is sampled, contains no metadata values, and is not an exhaustive schema. Edit the supplied discovery SQL for older data or nested JSONExtractKeys(metadata, 'parent')
{%- endblock %}

View file

@ -8,6 +8,18 @@ Raw OTLP exports live in `crates/traces/tests/fixtures/query_*.json`. The seeded
The swarm capture has handoff spans marked ERROR with `ParentCommand` exception events and a root with UNSET status. These are exported diagnostic statuses, which do not establish a failed execution. The tests preserve incoming statuses and check root status separately from the count of error spans, deriving both from the decoded export. They do not infer an execution outcome from exception text, framework names, successful model calls, or output presence. Framework-specific interpretation of control-flow exceptions belongs in the instrumentation integration
For a local dashboard with linked requests and traces, run `bash scripts/run_tracing_proxy_local.sh --seed` from the repository root and open `http://127.0.0.1:4002/ui/`. Log in as `admin` with password `sk-1234`, matching the UI E2E harness. The launcher keeps the proxy running until Ctrl-C and leaves the database volumes intact
`deeplite_swarm_spend_logs.jsonl` pairs every LLM span in the swarm export with a ClickHouse spend row. Response IDs, trace and span IDs, token counts, input, output, and timestamps come from the export. Messages and responses use the chat completion format supported by the request viewer. Spend is synthetic, set to $0.01 per request and marked in metadata, because the export does not include actual billed costs. These rows are stored here because `traces-clickhouse` owns the spend row schema
The simple and swarm exports for all twelve SDK examples were captured on 2026-10-03 against port 4002 using `openai/gpt-6-luna`. Each export has a matching `<name>_spend_logs.jsonl` with actual proxy spend, usage, request and response IDs, messages, and timestamps. Authorization headers, provider cookies, organization and project identifiers, and local paths were redacted. OTLP identifiers and enums use their canonical JSON encodings. `metadata.fixture_capture` identifies the associated export and whether model spans contain sufficient identity to join spend
The LlamaIndex captures contain provider IDs inside `output.value.raw.id`. Regression tests require normalization to retain those call keys and trace cost resolution to count nested model spans once. The Claude captures use the SDK example's local gateway adapter, which supplies the actual Anthropic message ID in the `request-id` response header. The two `claude_agent_sdk_missing_request_id_*` exports retain the earlier behavior: real spend rows exist, but model spans contain no matching call IDs, so trace spend remains unknown
`scripts/seed_tracing_fixtures.py` replays every JSON export in `crates/traces/tests/fixtures` through `POST /v1/traces`, then inserts all companion spend rows into ClickHouse through the production storage API and into Postgres through Prisma. The Requests table reads Postgres, while trace costs and Lens read ClickHouse. It shifts each capture into the current time window, keeping span, event, and paired spend timestamps aligned. The split `query_*.json` exports share a time shift and ID namespace to preserve cross-file parent links. Other captures get separate ID namespaces to avoid collisions between fixtures. It assigns fresh linked IDs for each run, including provider IDs inside managed response IDs, and reads the authenticated tenant from the ingested spans before stamping spend rows. The command exits unsuccessfully if any trace detail API result differs from its captured spend total or expected unknown cost. Exports without companion spend rows retain missing costs and do not create Requests entries
`tests/test_litellm_rust/test_traces.py` ingests these exports and spend rows into an isolated ClickHouse container, then checks trace detail costs and spend queries through the real FastAPI endpoints. Every SQL example returned by `/v1/traces/query/help` is executed through `/v1/traces/query`, including missing costs, free requests, replacement rows, and tenant ownership cases
`spend_logs.jsonl` contains spend insert rows with millisecond timestamps, including two versions of one request. Replace this small placeholder dataset when the actual data is available. The query fixture applies production migrations, then removes TTL from its isolated database so fixed timestamps do not expire. Background merges are stopped so rollup aggregation and `FINAL` deduplication are exercised on unmerged data. Retention behavior stays covered by the migration tests
Curated SQL lives in `tests/queries/*.sql`. Each query has a matching `.expected.json` containing ordered result rows for `admin`, `team`, `key`, and `other_team` readers. Update the exports and expected results together. Add a named case in `tests/queries.rs` for each new query. Assertions compare only result data, excluding server statistics and execution timing

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View file

@ -101,7 +101,7 @@ async fn schema_supports_span_rollups_and_spend_joins(
&reader,
&litellm_traces_clickhouse::query::named::SpanDetailParams {
access: litellm_traces_clickhouse::query::named::ReadAccessParams {
all_teams: 0,
all_teams: false,
user_id: String::new(),
team_ids: vec!["team-1".into()],
},
@ -148,6 +148,8 @@ async fn schema_supports_span_rollups_and_spend_joins(
"response_ids".into(),
Parameter::Strings(vec!["response-1".into()]),
),
("request_ids".into(), Parameter::Strings(Vec::new())),
("trace_ids".into(), Parameter::Strings(Vec::new())),
("all_teams".into(), Parameter::Integer(0)),
("user_id".into(), Parameter::Text(String::new())),
("team_ids".into(), Parameter::Strings(vec!["team-1".into()])),
@ -236,6 +238,44 @@ async fn normalized_fields_match_clickhouse_catalog(
Ok(())
}
#[rstest]
#[tokio::test]
async fn agent_metadata_is_stored_and_queryable(
#[future(awt)] database: TestResult<ClickHouseDatabase>,
) -> TestResult {
let ready = database?;
ensure_schema(
&ready.client,
&Connection::writer(&ready.url)?,
"trace_test",
7,
)
.await?;
let metadata = serde_json::json!({"thread_id": "thread-1", "ls_subagent_id": "agent-1"});
let timestamp = time::OffsetDateTime::now_utc().unix_timestamp_nanos() as i64;
insert_rows(
&ready,
"otel_traces",
vec![BTreeMap::from([
("Timestamp".into(), timestamp.into()),
("TraceId".into(), "trace-1".into()),
("SpanId".into(), "span-1".into()),
("AgentMetadata".into(), metadata.to_string().into()),
])],
)
.await?;
let response = read_json(
&ready,
"SELECT JSONExtractString(AgentMetadata, 'thread_id') AS thread_id, JSONExtractString(AgentMetadata, 'ls_subagent_id') AS subagent_id FROM trace_test.otel_traces WHERE TraceId = 'trace-1'",
).await?;
assert_eq!(response["data"][0]["thread_id"], metadata["thread_id"]);
assert_eq!(
response["data"][0]["subagent_id"],
metadata["ls_subagent_id"]
);
Ok(())
}
#[rstest]
#[tokio::test]
async fn insert_rejects_unknown_columns_even_if_url_requests_skipping_them(
@ -796,7 +836,7 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
}))?]).await?;
}
let connection = Connection::configured(&database.url, "trace_test", "default", "")?;
let parameters = BTreeMap::from([
let sample_parameters = BTreeMap::from([
("source".into(), Parameter::Text("traces".into())),
("all_teams".into(), Parameter::Integer(1)),
("team".into(), Parameter::Text(String::new())),
@ -833,7 +873,7 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
&database.client,
&connection,
ReadQuery::Sample,
&parameters,
&sample_parameters,
)
.await?,
)?;
@ -856,13 +896,12 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
.await?,
)?;
assert_eq!(identities["data"].as_array().map(Vec::len), Some(2));
let user_params = identity_params
.into_iter()
.chain([
("team_ids".into(), Parameter::Strings(vec![])),
("user_id".into(), Parameter::Text("one".into())),
])
.collect();
let user_params = BTreeMap::from([
("trace_id".into(), Parameter::Text("shared".into())),
("all_teams".into(), Parameter::Integer(0)),
("user_id".into(), Parameter::Text("one".into())),
("team_ids".into(), Parameter::Strings(vec![])),
]);
let identity: serde_json::Value = serde_json::from_str(
&execute_named_read(
&database.client,
@ -878,23 +917,23 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
.any(|row| row["trace_ref"] == identity["data"][0]["trace_ref"])
);
let first_ref = rows[0]["trace_ref"].as_str().expect("reference");
let read_parameters: BTreeMap<_, _> = parameters
.into_iter()
.chain([
("id".into(), Parameter::Text("shared".into())),
("record_team".into(), Parameter::Text("team".into())),
("trace_ref".into(), Parameter::Text(first_ref.into())),
("cursor".into(), Parameter::Text(String::new())),
("offset".into(), Parameter::Integer(1)),
("span".into(), Parameter::Text("root".into())),
])
.collect();
let content_parameters = BTreeMap::from([
("all_teams".into(), Parameter::Integer(1)),
("team".into(), Parameter::Text(String::new())),
("key_hash".into(), Parameter::Text(String::new())),
("source".into(), Parameter::Text("traces".into())),
("id".into(), Parameter::Text("shared".into())),
("record_team".into(), Parameter::Text("team".into())),
("trace_ref".into(), Parameter::Text(first_ref.into())),
("cursor".into(), Parameter::Text(String::new())),
("offset".into(), Parameter::Integer(1)),
]);
let content: serde_json::Value = serde_json::from_str(
&execute_named_read(
&database.client,
&connection,
ReadQuery::Content,
&read_parameters,
&content_parameters,
)
.await?,
)?;
@ -905,10 +944,17 @@ async fn lens_filters_reads_and_evidence_keep_reused_trace_ids_separate(
} else {
"timeout"
};
let evidence_parameters = read_parameters
.into_iter()
.chain([("quote".into(), Parameter::Text(opposite.into()))])
.collect();
let evidence_parameters = BTreeMap::from([
("all_teams".into(), Parameter::Integer(1)),
("team".into(), Parameter::Text(String::new())),
("key_hash".into(), Parameter::Text(String::new())),
("source".into(), Parameter::Text("traces".into())),
("id".into(), Parameter::Text("shared".into())),
("record_team".into(), Parameter::Text("team".into())),
("trace_ref".into(), Parameter::Text(first_ref.into())),
("span".into(), Parameter::Text("root".into())),
("quote".into(), Parameter::Text(opposite.into())),
]);
let evidence: serde_json::Value = serde_json::from_str(
&execute_named_read(
&database.client,
@ -1317,7 +1363,7 @@ async fn lens_agent_discovery_and_selection_preserve_scope(
.await?;
}
let connection = Connection::configured(&database.url, "trace_test", "default", "")?;
let scope_parameters = BTreeMap::from([
let agent_parameters = BTreeMap::from([
("all_teams".into(), Parameter::Integer(0)),
("team".into(), Parameter::Text("alpha".into())),
("key_hash".into(), Parameter::Text("one".into())),
@ -1327,7 +1373,7 @@ async fn lens_agent_discovery_and_selection_preserve_scope(
&database.client,
&connection,
ReadQuery::Agents,
&scope_parameters,
&agent_parameters,
)
.await?,
)?;
@ -1337,53 +1383,58 @@ async fn lens_agent_discovery_and_selection_preserve_scope(
{"agent_name": "research_agent"}, {"agent_name": "support_agent"}
])
);
let parameters = scope_parameters
.into_iter()
.chain([
("source".into(), Parameter::Text("traces".into())),
(
"start".into(),
Parameter::Integer(timestamp / 1_000_000 - 1000),
),
(
"end".into(),
Parameter::Integer(timestamp / 1_000_000 + 1000),
),
("service".into(), Parameter::Text("shared-app".into())),
(
"agent_name".into(),
Parameter::Text("research_agent".into()),
),
("filter_keys".into(), Parameter::Strings(vec![])),
("filter_values".into(), Parameter::Strings(vec![])),
("limit".into(), Parameter::Integer(100)),
("offset".into(), Parameter::Integer(0)),
("after".into(), Parameter::Text(String::new())),
("sample_percent".into(), Parameter::Text("100".into())),
("sample_cap".into(), Parameter::Integer(0)),
("preview".into(), Parameter::Integer(1)),
("selected_team".into(), Parameter::Text(String::new())),
("execution_ids".into(), Parameter::Strings(vec![])),
])
.collect::<BTreeMap<_, _>>();
let sample_parameters = BTreeMap::from([
("all_teams".into(), Parameter::Integer(0)),
("team".into(), Parameter::Text("alpha".into())),
("key_hash".into(), Parameter::Text("one".into())),
("source".into(), Parameter::Text("traces".into())),
(
"start".into(),
Parameter::Integer(timestamp / 1_000_000 - 1000),
),
(
"end".into(),
Parameter::Integer(timestamp / 1_000_000 + 1000),
),
("service".into(), Parameter::Text("shared-app".into())),
(
"agent_name".into(),
Parameter::Text("research_agent".into()),
),
("filter_keys".into(), Parameter::Strings(vec![])),
("filter_values".into(), Parameter::Strings(vec![])),
("limit".into(), Parameter::Integer(100)),
("offset".into(), Parameter::Integer(0)),
("after".into(), Parameter::Text(String::new())),
("sample_percent".into(), Parameter::Text("100".into())),
("sample_cap".into(), Parameter::Integer(0)),
("preview".into(), Parameter::Integer(1)),
("selected_team".into(), Parameter::Text(String::new())),
("execution_ids".into(), Parameter::Strings(vec![])),
]);
let sample: serde_json::Value = serde_json::from_str(
&execute_named_read(
&database.client,
&connection,
ReadQuery::Sample,
&parameters,
&sample_parameters,
)
.await?,
)?;
assert_eq!(sample["data"].as_array().expect("rows").len(), 1);
assert_eq!(sample["data"][0]["trace_id"], "research");
assert_eq!(sample["data"][0]["span_count"], 2);
let availability_parameters = BTreeMap::from([
("all_teams".into(), Parameter::Integer(0)),
("team".into(), Parameter::Text("alpha".into())),
("key_hash".into(), Parameter::Text("one".into())),
]);
let available: serde_json::Value = serde_json::from_str(
&execute_named_read(
&database.client,
&connection,
ReadQuery::Availability,
&parameters,
&availability_parameters,
)
.await?,
)?;
@ -1427,7 +1478,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
.await?;
let metadata = serde_json::json!({
"project": "example", "labels": {"priority": 3, "enabled": true},
"dotted.key": "literal", "quote'\\key": null, "items": [{"name": "first"}],
"dotted.key": "private-metadata-value", "quote'\\key": null, "items": [{"name": "first"}],
"<custom>&{{key}}": {"nested.key": true}
});
insert_rows(
@ -1435,7 +1486,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
"spend_logs",
vec![serde_json::from_value(serde_json::json!({
"request_id": "request-1", "response_id": "response-1", "team_id": "team-1",
"api_key": "key-1", "metadata": metadata.to_string(), "spend": 0.25,
"api_key": "key-1", "trace_id": "trace-1", "metadata": metadata.to_string(), "spend": 0.25,
"start_time": timestamp / 1_000_000, "end_time": timestamp / 1_000_000 + 100
}))?],
)
@ -1501,9 +1552,31 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
)));
}
}
for gotcha in help["gotchas"].as_array().ok_or("missing gotchas")? {
assert!(guide.contains(gotcha.as_str().ok_or("gotcha text")?));
let gotchas = help["gotchas"].as_array().ok_or("missing gotchas")?;
let gotcha_positions = gotchas
.iter()
.map(|gotcha| {
guide
.find(gotcha.as_str().expect("gotcha text"))
.expect("rendered gotcha")
})
.collect::<Vec<_>>();
assert!(gotcha_positions.windows(2).all(|pair| pair[0] < pair[1]));
assert!(
guide.contains(
help["metadata"]["sample_sql"]
.as_str()
.ok_or("sampling SQL")?
)
);
for catalog in help["attributes"].as_array().ok_or("attributes")? {
assert!(guide.contains(catalog["discovery_sql"].as_str().ok_or("discovery SQL")?));
assert!(guide.contains(&format!("Truncated: {}", catalog["truncated"])));
}
assert_eq!(
guide.contains("No attribute keys found in the sampled spans"),
!populated
);
let tables = help["tables"].as_array().ok_or("missing tables")?;
assert_eq!(tables.len(), 3);
let columns = tables[0]["columns"].as_array().ok_or("missing columns")?;
@ -1557,6 +1630,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
.any(|field| field["path"] == serde_json::json!(["items", 1, "name"]))
);
assert!(guide.contains("CustomColumn: String"));
assert!(!guide.contains("private-metadata-value"));
assert!(guide.contains("JSONExtractRaw(metadata, '<custom>&{{key}}', 'nested.key')"));
assert!(guide.contains("SpanAttributes['custom.tag']"));
assert!(guide.contains("ResourceAttributes['custom.resource']"));
@ -1575,7 +1649,24 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
assert_ne!(values["data"][0]["value"], "");
}
}
for example in help["examples"].as_array().ok_or("missing examples")? {
let examples = help["examples"].as_array().ok_or("missing examples")?;
let example_positions = examples
.iter()
.map(|example| {
let rendered = format!(
"{}\n{}",
example["name"].as_str().expect("name"),
example["sql"].as_str().expect("SQL")
);
guide.find(&rendered).expect("rendered example")
})
.collect::<Vec<_>>();
assert!(example_positions.windows(2).all(|pair| pair[0] < pair[1]));
assert!(
example_positions.last().ok_or("last example")?
< gotcha_positions.first().ok_or("first gotcha")?
);
for example in examples {
let sql = example["sql"].as_str().ok_or("missing example SQL")?;
assert!(guide.contains(example["name"].as_str().ok_or("missing example name")?));
assert!(guide.contains(sql));
@ -1592,7 +1683,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
let values: serde_json::Value = serde_json::from_str(&body)?;
assert_eq!(
values["data"].as_array().ok_or("missing data")?.is_empty(),
!populated,
!populated || example["name"] == "LLM spans without a direct spend match",
"{sql}"
);
if populated && example["name"] == "Traces correlated with LLM call metadata" {
@ -1666,6 +1757,18 @@ async fn query_help_preserves_schema_and_guide_when_discovery_hits_reader_limits
guide.contains("Attribute discovery unavailable:"),
span_rows > 1
);
assert!(!guide.contains("No metadata paths found in the sampled rows"));
assert!(!guide.contains("No attribute keys found in the sampled spans"));
assert!(
guide.contains(
help["metadata"]["sample_sql"]
.as_str()
.ok_or("sampling SQL")?
)
);
for catalog in help["attributes"].as_array().ok_or("attributes")? {
assert!(guide.contains(catalog["discovery_sql"].as_str().ok_or("discovery SQL")?));
}
for (catalog, unavailable) in [
(&help["metadata"], spend_rows > 1),
(&help["attributes"][0], span_rows > 1),
@ -1681,28 +1784,86 @@ async fn query_help_preserves_schema_and_guide_when_discovery_hits_reader_limits
Ok(())
}
#[rstest]
#[tokio::test]
async fn query_help_displays_discovery_truncation(
#[future(awt)] database: TestResult<ClickHouseDatabase>,
) -> TestResult {
let database = database?;
let writer = Connection::writer(&database.url)?;
ensure_schema(&database.client, &writer, "trace_test", 7).await?;
execute_write(
&database,
"INSERT INTO trace_test.otel_traces (Timestamp, TraceId, SpanId, SpanAttributes, ResourceAttributes) \
SELECT now64(9), 'trace', 'span', \
mapFromArrays(arrayMap(x -> concat('key-', toString(x)), range(1000)), arrayMap(x -> 'value', range(1000))) AS attributes, \
attributes FROM numbers(1)",
)
.await?;
execute_write(
&database,
"INSERT INTO trace_test.spend_logs (request_id, start_time, end_time, metadata) \
SELECT toString(number), now64(3), now64(3), '{\"key\":true}' FROM numbers(1000)",
)
.await?;
let reader = Connection::configured(&database.url, "trace_test", "default", "")?;
let help = serde_json::to_value(
litellm_traces_clickhouse::query_help(&database.client, &reader).await?,
)?;
let guide = help["guide"].as_str().ok_or("guide")?;
assert_eq!(help["metadata"]["truncated"], true);
assert!(guide.contains("truncated: true"));
for catalog in help["attributes"].as_array().ok_or("attributes")? {
assert_eq!(catalog["truncated"], true);
let displayed = format!(
"{}.{}",
catalog["table"].as_str().ok_or("table")?,
catalog["column"].as_str().ok_or("column")?
);
let section = guide.split(&displayed).nth(1).ok_or("attribute section")?;
assert!(
section
.split("\n\n")
.next()
.ok_or("catalog body")?
.contains("Truncated: true")
);
for field in catalog["fields"].as_array().ok_or("fields")? {
assert!(section.contains(field["expression"].as_str().ok_or("expression")?));
}
}
Ok(())
}
#[rstest]
fn field_definitions_match_serialized_normalized_span() {
use litellm_traces::decode_otlp;
use litellm_traces::{Tenant, decode_otlp};
use litellm_traces_clickhouse::span_rows;
use std::collections::BTreeSet;
let spans = decode_otlp(
br#"{"resourceSpans":[{"scopeSpans":[{"spans":[{"traceId":"11111111111111111111111111111111","spanId":"2222222222222222","name":"root"}]}]}]}"#,
Some("application/json"),
)
.expect("valid OTLP");
let fields = &spans[0].normalized;
let serialized = serde_json::to_value(fields).expect("serializable fields");
let keys: BTreeSet<_> = serialized
let tenant = Tenant {
team_id: "team".into(),
api_key_hash: "key".into(),
..Tenant::default()
};
let rows = span_rows(spans, &tenant, 64 * 1024);
let row =
serde_json::to_value(rows.first().expect("storage row")).expect("serializable storage row");
let keys: BTreeSet<_> = row
.as_object()
.expect("field object")
.expect("storage row object")
.keys()
.map(String::as_str)
.collect();
let mapped: BTreeSet<_> = NORMALIZED_FIELD_DEFINITIONS
.iter()
.map(|field| field.name)
.map(|field| field.clickhouse_column)
.collect();
assert_eq!(keys, mapped);
assert!(mapped.is_subset(&keys));
}
#[rstest]
@ -1743,6 +1904,8 @@ async fn named_and_sql_readers_share_request_log_visibility(
"api_key_hash": legacy_key.unwrap_or_default(),
}))?,
response_ids: vec!["shared-response".into()],
request_ids: Vec::new(),
trace_ids: Vec::new(),
start_ms: timestamp / 1_000_000 - 1,
end_ms: timestamp / 1_000_000 + 1,
});
@ -1813,7 +1976,7 @@ async fn rollup_cost_completeness_preserves_missing_ids_and_fails_closed_for_his
let params = litellm_traces_clickhouse::query::named::ListTracesParams::from(
litellm_traces::query::named::ListTracesParams {
access: litellm_traces::query::named::ReadAccessParams {
all_teams: 0,
all_teams: false,
user_id: "".into(),
team_ids: vec!["team".into()],
},
@ -1835,7 +1998,7 @@ async fn rollup_cost_completeness_preserves_missing_ids_and_fails_closed_for_his
access: litellm_traces::query::named::ReadAccessParams {
user_id: "owner".into(),
team_ids: vec![],
all_teams: 0,
all_teams: false,
},
..params.0
},
@ -1932,7 +2095,7 @@ async fn agent_final_answer_preserves_visibility_and_trace_ownership(
&reader,
&SpanDetailParams {
access: ReadAccessParams {
all_teams,
all_teams: all_teams == 1,
user_id: user.into(),
team_ids: teams.into_iter().map(str::to_owned).collect(),
},
@ -1951,3 +2114,55 @@ async fn agent_final_answer_preserves_visibility_and_trace_ownership(
}
Ok(())
}
#[rstest]
#[tokio::test]
async fn nullable_spend_upgrade_preserves_existing_costs_and_unknown_new_costs(
#[future(awt)] database: TestResult<ClickHouseDatabase>,
) -> TestResult {
let database = database?;
let writer = Connection::writer(&database.url)?;
let timestamp = (time::OffsetDateTime::now_utc().unix_timestamp_nanos() / 1_000_000) as i64;
let statements = schema_statements("trace_test", 7)?;
for statement in &statements[..statements.len() - 1] {
execute_write(&database, statement).await?;
}
let legacy = serde_json::from_value(serde_json::json!({
"request_id": "legacy", "response_id": "legacy-response", "spend": 0.25,
"start_time": timestamp, "end_time": timestamp + 100
}))?;
insert_rows(&database, "spend_logs", vec![legacy]).await?;
ensure_schema(&database.client, &writer, "trace_test", 7).await?;
ensure_schema(&database.client, &writer, "trace_test", 7).await?;
let unknown = serde_json::from_value(serde_json::json!({
"request_id": "unknown", "response_id": "unknown-response", "spend": null,
"start_time": timestamp, "end_time": timestamp + 100
}))?;
let free = serde_json::from_value(serde_json::json!({
"request_id": "free", "response_id": "free-response", "spend": 0.0,
"start_time": timestamp, "end_time": timestamp + 100
}))?;
insert_rows(&database, "spend_logs", vec![unknown, free]).await?;
let result = read_json(
&database,
"SELECT request_id, spend FROM trace_test.spend_logs FINAL ORDER BY request_id",
)
.await?;
#[derive(Debug, serde::Deserialize)]
struct CostRow {
request_id: String,
spend: Option<f64>,
}
let rows: Vec<CostRow> = serde_json::from_value(result["data"].clone())?;
assert_eq!(
rows.iter()
.map(|row| (row.request_id.as_str(), row.spend))
.collect::<Vec<_>>(),
vec![
("free", Some(0.0)),
("legacy", Some(0.25)),
("unknown", None)
]
);
Ok(())
}

View file

@ -144,11 +144,11 @@ async fn typed_queries_read_normalized_spans_and_keep_trace_identities_separate(
);
assert_eq!(
(
spans[1].0.kind.as_str(),
spans[1].0.kind,
spans[1].0.input_tokens,
spans[1].0.output_tokens
),
("llm", 12, 6)
(litellm_traces::ObservationType::Llm, 12, 6)
);
assert_eq!(spans[2].0.status_message, "lookup timed out");
Ok(())
@ -225,7 +225,13 @@ async fn captured_deeplite_exports_round_trip_through_clickhouse(
.filter(|span| span.parent_span_id.is_empty())
.collect::<Vec<_>>();
assert_eq!(roots.len(), 1);
assert_eq!(traces[0].0.status, roots[0].status_code);
assert_eq!(
traces[0].0.status,
serde_json::from_value::<litellm_traces::SpanStatus>(serde_json::json!(
roots[0].status_code
))
.unwrap()
);
assert_eq!(
traces[0].0.error_count,
decoded
@ -246,7 +252,13 @@ async fn captured_deeplite_exports_round_trip_through_clickhouse(
assert_eq!(row.duration_ns, span.end_ns - span.start_ns);
assert_eq!(row.input_tokens, span.normalized.input_tokens);
assert_eq!(row.output_tokens, span.normalized.output_tokens);
assert_eq!(row.status, span.status_code);
assert_eq!(
row.status,
serde_json::from_value::<litellm_traces::SpanStatus>(serde_json::json!(
span.status_code
))
.unwrap()
);
}
Ok(())
}

View file

@ -139,11 +139,29 @@ fn span_row(span: &DecodedSpan, team: &str, key: &str) -> BTreeMap<String, Value
"ObservationType".into(),
json!(span.normalized.observation_type),
),
("AgentName".into(), json!(span.normalized.agent_name)),
("Model".into(), json!(span.normalized.model)),
(
"AgentName".into(),
json!(span.normalized.agent_name.as_deref().unwrap_or_default()),
),
(
"Model".into(),
json!(span.normalized.model.as_deref().unwrap_or_default()),
),
(
"LiteLLMRequestId".into(),
json!(span.normalized.litellm_request_id),
json!(
span.normalized
.calls
.key_set()
.into_iter()
.flatten()
.find_map(|key| match key {
litellm_traces::CallKey::LiteLlmRequest(id)
| litellm_traces::CallKey::ProviderResponse(id) => Some(id.as_str()),
litellm_traces::CallKey::Transport => None,
})
.unwrap_or_default()
),
),
("InputTokens".into(), json!(span.normalized.input_tokens)),
("OutputTokens".into(), json!(span.normalized.output_tokens)),

View file

@ -0,0 +1,216 @@
use litellm_traces::{Shared, Tenant, decode_otlp};
use litellm_traces_clickhouse::{NORMALIZED_FIELD_DEFINITIONS, span_rows};
use rstest::{fixture, rstest};
use serde_json::{Value, json};
const MAX_VALUE_BYTES: usize = 64 * 1024;
#[fixture]
fn tenant() -> Tenant {
Tenant {
team_id: "team-a".into(),
api_key_hash: "key-a".into(),
org_id: "org-a".into(),
user_id: "user-a".into(),
}
}
fn attribute(key: &str, value: &str) -> Value {
json!({"key": key, "value": {"stringValue": value}})
}
fn span(span_id: &str, attributes: Vec<Value>, extra: Value) -> Value {
let mut span = json!({
"traceId": "01".repeat(16),
"spanId": span_id,
"name": "operation",
"startTimeUnixNano": "1000",
"endTimeUnixNano": "5000",
"attributes": attributes,
});
span.as_object_mut()
.unwrap()
.extend(extra.as_object().unwrap().clone());
span
}
fn export(resources: Vec<(Vec<Value>, Vec<Value>)>) -> Vec<u8> {
let resource_spans: Vec<Value> = resources
.into_iter()
.map(|(attributes, spans)| {
json!({
"resource": {"attributes": attributes},
"scopeSpans": [{"scope": {"name": "scope", "version": "1"}, "spans": spans}],
})
})
.collect();
json!({"resourceSpans": resource_spans})
.to_string()
.into_bytes()
}
fn rows(body: &[u8], tenant: &Tenant, max_value_bytes: usize) -> Vec<Value> {
let spans = decode_otlp(body, Some("application/json")).unwrap();
span_rows(spans, tenant, max_value_bytes)
.iter()
.map(|row| serde_json::to_value(row).unwrap())
.collect()
}
#[rstest]
fn tenant_overwrites_claimed_identity_and_resources_stay_shared_per_group(tenant: Tenant) {
let spoofed = vec![
attribute("service.name", "svc"),
attribute("litellm.team_id", "spoofed-team"),
attribute("litellm.user_id", "spoofed-user"),
];
let body = export(vec![
(
spoofed.clone(),
vec![
span(&"02".repeat(8), vec![], json!({})),
span(&"03".repeat(8), vec![], json!({})),
],
),
(spoofed, vec![span(&"04".repeat(8), vec![], json!({}))]),
]);
let spans = decode_otlp(&body, Some("application/json")).unwrap();
let stored = span_rows(spans, &tenant, MAX_VALUE_BYTES);
let resource = |index: usize| &stored[index]["ResourceAttributes"];
assert!(Shared::shares_storage_with(resource(0), resource(1)));
assert!(!Shared::shares_storage_with(resource(0), resource(2)));
assert_eq!(resource(0), resource(2));
assert_eq!(
**resource(0),
json!({
"service.name": "svc",
"litellm.team_id": "team-a",
"litellm.user_id": "user-a",
"litellm.api_key_hash": "key-a",
"litellm.org_id": "org-a",
})
);
for row in &stored {
assert_eq!(
(&*row["TeamId"], &*row["ApiKeyHash"], &*row["UserId"]),
(&json!("team-a"), &json!("key-a"), &json!("user-a"))
);
assert_eq!(*row["ServiceName"], json!("svc"));
}
}
#[rstest]
#[case::exception_event("", json!("customer acme-404 not found"))]
#[case::status_message_wins("boom", json!("boom"))]
fn status_message_falls_back_to_the_exception_event(
tenant: Tenant,
#[case] status_message: &str,
#[case] expected: Value,
) {
let exported = span(
&"02".repeat(8),
vec![],
json!({
"status": {"code": 2, "message": status_message},
"events": [{"name": "exception", "timeUnixNano": "2000", "attributes": [
attribute("exception.type", "KeyError"),
attribute("exception.message", "customer acme-404 not found"),
]}],
}),
);
let row = &rows(
&export(vec![(vec![], vec![exported])]),
&tenant,
MAX_VALUE_BYTES,
)[0];
assert_eq!(row["StatusCode"], "STATUS_CODE_ERROR");
assert_eq!(row["StatusMessage"], expected);
}
#[rstest]
fn consumed_payloads_leave_span_attributes_and_long_values_are_capped(tenant: Tenant) {
let messages = json!([
{"role": "system", "content": "be brief"},
{"role": "user", "content": "x".repeat(300)},
{"role": "user", "content": "latest question"},
]);
let exported = span(
&"02".repeat(8),
vec![
attribute("gen_ai.operation.name", "chat"),
attribute("gen_ai.input.messages", &messages.to_string()),
attribute(
"gen_ai.output.messages",
&json!([{"role": "assistant", "content": "y".repeat(300)}]).to_string(),
),
attribute("custom.blob", &"z".repeat(300)),
],
json!({}),
);
let row = &rows(&export(vec![(vec![], vec![exported])]), &tenant, 200)[0];
let attributes = row["SpanAttributes"].as_object().unwrap();
assert!(!attributes.contains_key("gen_ai.input.messages"));
assert!(!attributes.contains_key("gen_ai.output.messages"));
assert_eq!(
attributes["custom.blob"],
format!("{}…[truncated 100 bytes]", "z".repeat(200))
);
let input = row["Input"].as_str().unwrap();
let kept: Vec<Value> = serde_json::from_str(input).unwrap();
assert!(input.len() <= 200);
assert_eq!(kept[0]["content"], "be brief");
assert_eq!(kept.last().unwrap()["content"], "latest question");
assert!(row["Output"].as_str().unwrap().contains("…[truncated "));
assert_eq!(row["ObservationType"], "llm");
}
#[rstest]
fn rows_carry_every_normalized_column(tenant: Tenant) {
let row = &rows(
&export(vec![(
vec![],
vec![span(&"02".repeat(8), vec![], json!({}))],
)]),
&tenant,
MAX_VALUE_BYTES,
)[0];
for field in NORMALIZED_FIELD_DEFINITIONS {
assert!(
row.get(field.clickhouse_column).is_some(),
"{}",
field.clickhouse_column
);
}
assert_eq!(row["Duration"], 4000);
assert_eq!(row["AgentMetadata"], "{}");
}
#[rstest]
fn absent_identity_fields_are_empty_only_in_storage(tenant: Tenant) {
let body = export(vec![(
vec![],
vec![span(&"02".repeat(8), vec![], json!({}))],
)]);
let decoded = decode_otlp(&body, Some("application/json")).unwrap();
let normalized = &decoded[0].normalized;
assert_eq!(normalized.agent_name, None);
assert_eq!(normalized.framework, None);
assert_eq!(normalized.model, None);
assert_eq!(normalized.tool_call_id, None);
let stored = span_rows(decoded, &tenant, MAX_VALUE_BYTES);
let row = serde_json::to_value(&stored[0]).unwrap();
assert_eq!(
[
"AgentName",
"Framework",
"Model",
"ToolCallId",
"LiteLLMRequestId"
]
.map(|column| row[column].clone()),
[""; 5].map(|value| json!(value)),
);
assert_eq!(row["CallKeys"], json!([]));
assert_eq!(row["CallEvidence"], "unknown");
}

View file

@ -5,14 +5,22 @@ edition.workspace = true
license.workspace = true
repository.workspace = true
[features]
schema = ["dep:schemars"]
[dependencies]
askama.workspace = true
macro_rules_attribute.workspace = true
schemars = { workspace = true, optional = true }
indexmap = { version = "2", features = ["serde"] }
litellm-llms-types.workspace = true
opentelemetry-proto = { workspace = true, features = ["gen-tonic-messages", "trace", "with-serde"] }
prost.workspace = true
serde = { workspace = true, features = ["rc"] }
serde_json.workspace = true
serde_json = { workspace = true, features = ["preserve_order"] }
strum.workspace = true
thiserror.workspace = true
time.workspace = true
[dev-dependencies]
criterion.workspace = true
@ -21,3 +29,8 @@ rstest.workspace = true
[[bench]]
name = "resource-fanout"
harness = false
[[bin]]
name = "export-traces-schema"
path = "src/bin/export_schema.rs"
required-features = ["schema"]

View file

@ -0,0 +1,6 @@
fn main() {
println!(
"{}",
serde_json::to_string_pretty(&litellm_traces::schema::schemas()).unwrap()
);
}

View file

@ -15,3 +15,7 @@ pub struct InvalidScope;
#[derive(Debug, thiserror::Error)]
#[error("unknown ClickHouse read query")]
pub struct InvalidQuery;
#[derive(Debug, thiserror::Error)]
#[error("invalid trace call key")]
pub struct InvalidCallKey;

View file

@ -1,13 +1,43 @@
macro_rules_attribute::attribute_alias! {
#[apply(wire_type)] =
#[derive(serde::Serialize, serde::Deserialize)]
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
#[apply(response_type)] =
#[derive(serde::Serialize)]
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
#[apply(request_type)] =
#[derive(serde::Deserialize)]
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
}
mod error;
mod normalize;
mod otlp;
pub mod query;
mod query_access;
mod resolve;
#[cfg(feature = "schema")]
pub mod schema;
mod shared;
mod tenant;
mod truncate;
mod ui;
mod view;
pub mod wire;
pub use error::{Error, InvalidQuery, InvalidScope};
pub use normalize::{NormalizedSpan, ObservationType};
pub use otlp::{DecodedSpan, decode_otlp};
pub use error::{Error, InvalidCallKey, InvalidQuery, InvalidScope};
pub use normalize::{
AgentMetadata, AgentType, CallEvidence, CallEvidenceKind, CallKey, Integration, NormalizedSpan,
ObservationType,
};
pub use otlp::{DecodedEvent, DecodedSpan, decode_otlp};
pub use query::ReadQuery;
pub use query_access::QueryScope;
pub use resolve::{SpendLookup, iso_time, listed_summary, resolve_trace};
pub use shared::{Shared, SharedIdentity};
pub use tenant::Tenant;
pub use truncate::{truncate_messages, truncate_value};
pub use ui::{ChatRole, UiContent, UiField, UiMessage, UiToolCall, to_ui_content};
pub use view::{
AgentNode, Span, SpanDetail, SpanErrorPage, SpanStatus, Trace, TracePage, TraceSummary,
};

View file

@ -0,0 +1,7 @@
- Normalize one decoded span at a time: `format/` extracts recorded facts, then `instrumentation/` applies SDK semantics
- Own normalized span types, role and call evidence, shared message conversion in `messages.rs`, and metadata extraction in `metadata.rs`
- Keep wire-format parsing in `format/` and SDK-specific interpretation in `instrumentation/`; share message helpers instead of duplicating payload parsing
- Preserve format precedence, attribute alias precedence, token validation, and consumed-attribute tracking
- Leave wrapper resolution, cross-span ownership, and spend attribution to `resolve/`; related spans can arrive in separate exports
- Keep OTLP decoding in `otlp/`, storage in `traces-clickhouse`, and Python conversion in `python-bridge`
- Test observable normalization through the public API in `tests/normalize.rs` and `tests/normalization_formats.rs`; keep private-helper tests inline

View file

@ -0,0 +1,11 @@
- Read a span's recorded convention into `Extraction`: facts, an optional display name, and consumed attributes
- Own convention detection, attribute aliases, payload shapes, model and token fields, tool-call IDs, and explicitly recorded roles
- Preserve first-match format precedence in `mod.rs`, with GenAI as the fallback; use shared alias and token helpers from the parent module
- Track the source attributes selected for payload extraction so normalization retains unconsumed data
- Reuse `../messages.rs` for canonical messages, indexed attributes, and event payloads; keep SDK behavior in `../instrumentation/`
- Leave cross-span wrapper resolution, ownership, and spend attribution to `resolve/`
- Extend `tests/normalization_formats.rs` for parsing changes, including mixed conventions, fallbacks, and malformed payloads
- Consult the convention specifications when changing mappings:
- [OpenInference](https://github.com/Arize-ai/openinference/tree/main/spec)
- [OpenTelemetry GenAI](https://opentelemetry.io/docs/specs/semconv/registry/attributes/gen-ai/index.md)
- [LangSmith OTLP](https://docs.langchain.com/langsmith/trace-with-opentelemetry.md)

View file

@ -2,14 +2,18 @@ use std::collections::BTreeMap;
use serde_json::{Map, Value, json};
use super::{NormalizedSpan, ObservationType, SpanNormalizer, attr, first, tokens};
use crate::{Error, otlp::DecodedEvent};
use super::{Extraction, Format, SpanFacts};
use crate::{
Error,
normalize::{
CLAUDE_CODE_AGENT, CLAUDE_CODE_SCOPE, CallEvidence, CallKey, ObservationType, RoleEvidence,
SpanContext, attr, present, tokens,
},
otlp::DecodedEvent,
};
pub(crate) const CLAUDE_CODE_SCOPE: &str = "com.anthropic.claude_code.tracing";
pub(crate) const CLAUDE_CODE_AGENT: &str = "claude-code";
const AGENT_SDK_FRAMEWORK: &str = "claude-agent-sdk";
pub(super) struct ClaudeCodeNormalizer;
/// Claude Code's built-in tracing, identified by its instrumentation scope.
pub(crate) struct ClaudeCode;
enum SpanType {
Interaction,
@ -33,14 +37,13 @@ fn span_type(name: &str, attributes: &BTreeMap<String, String>) -> SpanType {
}
}
fn framework(attributes: &BTreeMap<String, String>) -> &'static str {
if attr(attributes, "query_source_safe") == "sdk"
|| attr(attributes, "system_prompt_preview").contains("cc_entrypoint=sdk")
{
AGENT_SDK_FRAMEWORK
} else {
CLAUDE_CODE_AGENT
}
/// `agent:custom:search_agent` -> `search_agent`: the subagent a request ran for.
fn subagent(attributes: &BTreeMap<String, String>) -> Option<&str> {
let mut parts = attr(attributes, "query_source")
.strip_prefix("agent:")?
.splitn(2, ':');
let (_kind, name) = (parts.next()?, parts.next()?);
(!name.is_empty()).then_some(name)
}
fn split_header(text: &str) -> Option<(&str, &str)> {
@ -154,68 +157,69 @@ fn input_tokens(attributes: &BTreeMap<String, String>) -> Result<u32, Error> {
})
}
impl SpanNormalizer for ClaudeCodeNormalizer {
fn matches(&self, scope_name: &str, _attributes: &BTreeMap<String, String>) -> bool {
scope_name == CLAUDE_CODE_SCOPE
impl Format for ClaudeCode {
fn matches(&self, context: &SpanContext<'_>) -> bool {
context.scope == CLAUDE_CODE_SCOPE
}
fn consumed_attributes(&self, attributes: &BTreeMap<String, String>) -> [&'static str; 2] {
match span_type("", attributes) {
SpanType::Interaction => ["user_prompt", ""],
SpanType::LlmRequest => ["new_context", "response.model_output"],
SpanType::Tool if tool_arguments(attributes).is_some() => ["tool_input", ""],
SpanType::Tool | SpanType::Other => ["", ""],
}
}
fn display_name(&self, attributes: &BTreeMap<String, String>) -> Option<String> {
let tool_name = attr(attributes, "tool_name");
(matches!(span_type("", attributes), SpanType::Tool) && !tool_name.is_empty())
.then(|| tool_name.to_owned())
}
fn normalize(
&self,
name: &str,
_parent_span_id: &str,
attributes: &BTreeMap<String, String>,
events: &[DecodedEvent],
) -> Result<NormalizedSpan, Error> {
let base = NormalizedSpan {
observation_type: ObservationType::Framework,
agent_name: CLAUDE_CODE_AGENT.to_owned(),
framework: framework(attributes).to_owned(),
litellm_request_id: String::new(),
model: String::new(),
input_tokens: 0,
output_tokens: 0,
input: String::new(),
output: String::new(),
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
let attributes = context.attributes;
let kind = span_type(context.name, attributes);
let base = SpanFacts {
role: Some(RoleEvidence::Declared(ObservationType::Framework)),
agent_name: Some(CLAUDE_CODE_AGENT.to_owned()),
tool_call_id: present(attributes, &["gen_ai.tool.call.id"]),
..SpanFacts::default()
};
Ok(match span_type(name, attributes) {
SpanType::Interaction => NormalizedSpan {
observation_type: ObservationType::Agent,
input: user_prompt(attributes),
..base
let (facts, consumed): (SpanFacts, Vec<&'static str>) = match kind {
SpanType::Interaction => (
SpanFacts {
role: Some(RoleEvidence::Declared(ObservationType::Agent)),
input: user_prompt(attributes),
..base
},
vec!["user_prompt"],
),
SpanType::LlmRequest => (
SpanFacts {
role: Some(RoleEvidence::Declared(ObservationType::Llm)),
agent_name: Some(subagent(attributes).unwrap_or(CLAUDE_CODE_AGENT).to_owned()),
model: present(attributes, &["model", "gen_ai.request.model"]),
input_tokens: input_tokens(attributes)?,
output_tokens: tokens(attributes, "output_tokens")?,
input: llm_input(attributes),
output: llm_output(attributes),
calls: present(attributes, &["gen_ai.response.id", "request_id"])
.map_or(CallEvidence::Unknown, |id| {
CallEvidence::complete(CallKey::ProviderResponse(id))
}),
..base
},
vec!["new_context", "response.model_output"],
),
SpanType::Tool => (
SpanFacts {
role: Some(RoleEvidence::Declared(ObservationType::Tool)),
input: tool_input(attributes),
output: tool_output(attributes, context.events),
..base
},
if tool_arguments(attributes).is_some() {
vec!["tool_input"]
} else {
Vec::new()
},
),
SpanType::Other => (base, Vec::new()),
};
Ok(Extraction {
facts,
display_name: if matches!(kind, SpanType::Tool) {
present(attributes, &["tool_name"])
} else {
None
},
SpanType::LlmRequest => NormalizedSpan {
observation_type: ObservationType::Llm,
litellm_request_id: first(attributes, "gen_ai.response.id", "request_id")
.to_owned(),
model: first(attributes, "model", "gen_ai.request.model").to_owned(),
input_tokens: input_tokens(attributes)?,
output_tokens: tokens(attributes, "output_tokens")?,
input: llm_input(attributes),
output: llm_output(attributes),
..base
},
SpanType::Tool => NormalizedSpan {
observation_type: ObservationType::Tool,
input: tool_input(attributes),
output: tool_output(attributes, events),
..base
},
SpanType::Other => base,
consumed_attributes: consumed,
})
}
}
@ -227,8 +231,35 @@ mod tests {
use rstest::rstest;
use serde_json::Value;
use super::{CLAUDE_CODE_SCOPE, ClaudeCodeNormalizer, SpanNormalizer};
use crate::{Error, normalize::ObservationType, otlp::DecodedEvent};
use super::CLAUDE_CODE_SCOPE;
use crate::{
Error,
normalize::{Normalization, NormalizedSpan, ObservationType},
otlp::DecodedEvent,
};
fn normalization(
name: &str,
attributes: &BTreeMap<String, String>,
events: &[DecodedEvent],
) -> Result<Normalization, Error> {
crate::normalize::normalize(&crate::normalize::SpanContext {
scope: CLAUDE_CODE_SCOPE,
name,
parent_span_id: "parent",
attributes,
events,
resource_attributes: &BTreeMap::new(),
})
}
fn normalize(
name: &str,
attributes: &BTreeMap<String, String>,
events: &[DecodedEvent],
) -> Result<NormalizedSpan, Error> {
normalization(name, attributes, events).map(|normalization| normalization.span)
}
fn attributes(pairs: &[(&str, &str)]) -> BTreeMap<String, String> {
pairs
@ -239,19 +270,17 @@ mod tests {
#[rstest]
fn tool_without_detailed_input_lists_known_arguments() {
let span = ClaudeCodeNormalizer
.normalize(
"claude_code.tool",
"parent",
&attributes(&[
("span.type", "tool"),
("tool_name", "Bash"),
("full_command", "git status"),
("bash_argv0", "git"),
]),
&[],
)
.expect("valid span");
let span = normalize(
"claude_code.tool",
&attributes(&[
("span.type", "tool"),
("tool_name", "Bash"),
("full_command", "git status"),
("bash_argv0", "git"),
]),
&[],
)
.expect("valid span");
let input: Value = serde_json::from_str(&span.input).expect("argument object");
assert_eq!(input["command"], "git status");
assert_eq!(input["bash_argv0"], "git");
@ -266,14 +295,13 @@ mod tests {
("tool_input", "[TOOL INPUT: Read]\nnot json"),
("file_path", "/workspace/a.py"),
]);
let span = ClaudeCodeNormalizer
.normalize("claude_code.tool", "parent", &attrs, &[])
.expect("valid span");
let span = normalize("claude_code.tool", &attrs, &[]).expect("valid span");
let input: Value = serde_json::from_str(&span.input).expect("argument object");
assert_eq!(input["file_path"], "/workspace/a.py");
assert!(
!ClaudeCodeNormalizer
.consumed_attributes(&attrs)
!normalization("claude_code.tool", &attrs, &[])
.expect("valid span")
.consumed_attributes
.contains(&"tool_input")
);
}
@ -295,45 +323,40 @@ mod tests {
#[case] events: Vec<DecodedEvent>,
#[case] expected: &str,
) {
let span = ClaudeCodeNormalizer
.normalize(
"claude_code.tool",
"parent",
&attributes(&[
("span.type", "tool"),
("new_context", "[TOOL RESULT: Bash]\n{\"stdout\":\"ctx\"}"),
]),
&events,
)
.expect("valid span");
let span = normalize(
"claude_code.tool",
&attributes(&[
("span.type", "tool"),
("new_context", "[TOOL RESULT: Bash]\n{\"stdout\":\"ctx\"}"),
]),
&events,
)
.expect("valid span");
assert_eq!(span.output, expected);
}
#[rstest]
fn llm_tool_result_context_becomes_tool_message() {
let span = ClaudeCodeNormalizer
.normalize(
"claude_code.llm_request",
"parent",
&attributes(&[
("span.type", "llm_request"),
("new_context", "[TOOL RESULT: toolu_1]\n1\timport os"),
]),
&[],
)
.expect("valid span");
let span = normalize(
"claude_code.llm_request",
&attributes(&[
("span.type", "llm_request"),
("new_context", "[TOOL RESULT: toolu_1]\n1\timport os"),
]),
&[],
)
.expect("valid span");
let input: Value = serde_json::from_str(&span.input).expect("messages");
assert_eq!(input[0]["role"], "tool");
assert_eq!(input[0]["content"], "1\timport os");
assert_eq!(span.output, "");
assert_eq!(span.framework, "claude-code");
assert_eq!(span.framework, Some(crate::Integration::ClaudeCode));
}
#[rstest]
fn llm_token_sum_overflow_is_rejected() {
let result = ClaudeCodeNormalizer.normalize(
let result = normalize(
"claude_code.llm_request",
"parent",
&attributes(&[
("span.type", "llm_request"),
("input_tokens", "4294967295"),
@ -358,10 +381,7 @@ mod tests {
} else {
attributes(&[("span.type", kind)])
};
let span = ClaudeCodeNormalizer
.normalize(name, "parent", &attrs, &[])
.expect("valid span");
let span = normalize(name, &attrs, &[]).expect("valid span");
assert_eq!(span.observation_type, expected);
assert!(ClaudeCodeNormalizer.matches(CLAUDE_CODE_SCOPE, &attrs));
}
}

View file

@ -0,0 +1,132 @@
use super::{Extraction, Format, Payload, SpanFacts};
use crate::{
Error,
normalize::{
ObservationType, RoleEvidence, SpanContext, attr, messages, present, select_attribute,
usage_tokens,
},
};
/// OpenTelemetry GenAI semantic conventions: the fallback, since any span may carry `gen_ai.*`.
pub(crate) struct GenAi;
#[derive(strum::EnumString)]
#[strum(serialize_all = "snake_case")]
pub(crate) enum Operation {
CreateAgent,
InvokeAgent,
InvokeWorkflow,
Chat,
#[strum(serialize = "text_completion", serialize = "completion")]
TextCompletion,
GenerateContent,
ExecuteTool,
#[strum(serialize = "embeddings", serialize = "embedding")]
Embeddings,
Retrieval,
}
impl Operation {
pub(crate) fn from_context(context: &SpanContext<'_>) -> Option<Self> {
Self::try_from(attr(context.attributes, "gen_ai.operation.name")).ok()
}
fn role(self) -> ObservationType {
match self {
Self::InvokeAgent => ObservationType::Agent,
Self::CreateAgent => ObservationType::Framework,
Self::InvokeWorkflow => ObservationType::Chain,
Self::Chat | Self::TextCompletion | Self::GenerateContent => ObservationType::Llm,
Self::ExecuteTool => ObservationType::Tool,
Self::Embeddings => ObservationType::Embedding,
Self::Retrieval => ObservationType::Retriever,
}
}
}
const INPUT_KEYS: [&str; 4] = [
"gen_ai.input.messages",
"gen_ai.tool.call.arguments",
"gen_ai.retrieval.query.text",
"gen_ai.prompt",
];
const OUTPUT_KEYS: [&str; 4] = [
"gen_ai.output.messages",
"gen_ai.tool.call.result",
"gen_ai.retrieval.documents",
"gen_ai.completion",
];
/// The messages key comes first and is put in the common format; other payloads stay as recorded.
fn payload(context: &SpanContext<'_>, keys: &[&'static str]) -> Payload {
let Some(attribute) = select_attribute(context.attributes, keys) else {
let prefix = if keys[0] == INPUT_KEYS[0] {
"gen_ai.prompt"
} else {
"gen_ai.completion"
};
let indexed = messages::indexed(context.attributes, prefix);
return Payload {
text: indexed
.or_else(|| {
let events: Vec<_> = context
.events
.iter()
.filter_map(|event| {
let encoded = attr(&event.attributes, "gen_ai.event.content");
let value = serde_json::from_str(encoded).unwrap_or_else(|_| {
serde_json::to_value(&event.attributes).unwrap_or_default()
});
messages::event_message(&event.name, &value)
})
.collect();
messages::event_payload(&events, keys[0] == OUTPUT_KEYS[0])
})
.unwrap_or_default(),
consumed: None,
};
};
Payload {
text: if attribute.source == keys[0] {
messages::canonical(attribute.text)
} else {
attribute.text.to_owned()
},
consumed: Some(attribute.source),
}
}
impl Format for GenAi {
fn matches(&self, _context: &SpanContext<'_>) -> bool {
true
}
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
let attributes = context.attributes;
let (input_tokens, output_tokens) = usage_tokens(attributes)?;
let input = payload(context, &INPUT_KEYS);
let output = payload(context, &OUTPUT_KEYS);
Ok(Extraction {
facts: SpanFacts {
role: Operation::from_context(context)
.map(|operation| RoleEvidence::Declared(operation.role())),
model: present(
attributes,
&["gen_ai.request.model", "gen_ai.response.model"],
),
input_tokens,
output_tokens,
input: input.text,
output: output.text,
tool_call_id: present(attributes, &["gen_ai.tool.call.id"]),
..SpanFacts::default()
},
display_name: None,
consumed_attributes: [input.consumed, output.consumed]
.into_iter()
.flatten()
.collect(),
})
}
}

View file

@ -0,0 +1,267 @@
use std::collections::BTreeMap;
use serde::{
Deserialize, Deserializer,
de::{DeserializeOwned, IgnoredAny},
};
use serde_json::Value;
use super::{Extraction, Format, SpanFacts, genai::GenAi};
use crate::{
Error,
normalize::{
CallEvidence, ObservationType, RoleEvidence, SpanContext, attr,
messages::{RawMessage, encode, langchain_result},
},
};
/// LangSmith's OpenTelemetry exporter: spans carry `langsmith.span.kind`.
pub(crate) struct LangSmith;
enum MessageBatch {
Flat(Vec<RawMessage>),
Nested(Vec<Vec<RawMessage>>),
}
impl<'de> Deserialize<'de> for MessageBatch {
fn deserialize<D: Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
let value = Value::deserialize(deserializer)?;
let Value::Array(items) = value else {
return Err(serde::de::Error::custom("messages must be an array"));
};
let parse = |items: Vec<Value>| {
items
.into_iter()
.filter_map(|item| serde_json::from_value(item).ok())
.collect()
};
Ok(if items.first().is_some_and(Value::is_array) {
Self::Nested(
items
.into_iter()
.filter_map(|item| item.as_array().cloned())
.map(parse)
.collect(),
)
} else {
Self::Flat(parse(items))
})
}
}
fn lenient<'de, D: Deserializer<'de>, T: DeserializeOwned>(
deserializer: D,
) -> Result<Option<T>, D::Error> {
let value = Value::deserialize(deserializer)?;
Ok(serde_json::from_value(value).ok())
}
impl MessageBatch {
fn first_batch(&self) -> &[RawMessage] {
match self {
Self::Flat(messages) => messages,
Self::Nested(batches) => batches.first().map(Vec::as_slice).unwrap_or_default(),
}
}
}
#[derive(Default, Deserialize)]
struct Payload {
#[serde(default, deserialize_with = "lenient")]
messages: Option<MessageBatch>,
}
#[derive(Deserialize)]
struct Command {
update: CommandUpdate,
}
#[derive(Deserialize)]
struct CommandUpdate {
messages: Vec<Value>,
}
#[derive(Deserialize)]
struct ContentValue {
content: Value,
}
#[derive(Deserialize)]
struct WrappedOutput {
output: Value,
#[serde(flatten)]
_other: BTreeMap<String, IgnoredAny>,
}
struct SpanIo {
input: String,
output: String,
calls: CallEvidence,
}
fn normalized_messages(messages: &[RawMessage]) -> String {
encode(
&messages
.iter()
.map(RawMessage::normalized)
.collect::<Vec<_>>(),
)
}
fn tool_output(raw_completion: &str) -> String {
let completion = serde_json::from_str::<Value>(raw_completion).unwrap_or(Value::Null);
let raw = WrappedOutput::deserialize(&completion)
.map(|wrapped| wrapped.output)
.unwrap_or(completion);
let selected = Command::deserialize(&raw)
.ok()
.and_then(|command| command.update.messages.into_iter().last())
.unwrap_or(raw);
let output = ContentValue::deserialize(&selected)
.map(|message| message.content)
.unwrap_or(selected);
output
.as_str()
.map(str::to_owned)
.unwrap_or_else(|| encode(&output))
}
fn span_io(kind: ObservationType, attributes: &BTreeMap<String, String>) -> SpanIo {
let raw_prompt = attr(attributes, "gen_ai.prompt");
let raw_completion = attr(attributes, "gen_ai.completion");
let prompt = serde_json::from_str::<Payload>(raw_prompt).unwrap_or_default();
if kind == ObservationType::Llm
&& serde_json::from_str::<Value>(raw_completion).is_ok_and(|value| value.is_object())
{
let input = prompt.messages.as_ref().map_or_else(
|| "[]".to_owned(),
|messages| normalized_messages(messages.first_batch()),
);
let result = serde_json::from_str::<Value>(raw_completion)
.ok()
.and_then(|value| langchain_result(&value));
return match result {
Some(result) if result.first.is_some() => SpanIo {
input,
output: result.first.as_ref().map(encode).unwrap_or_default(),
calls: result.calls,
},
_ => SpanIo {
input,
output: raw_completion.to_owned(),
calls: result.map_or(CallEvidence::Unknown, |result| result.calls),
},
};
}
if kind == ObservationType::Tool {
return SpanIo {
input: raw_prompt.to_owned(),
output: tool_output(raw_completion),
calls: CallEvidence::Unknown,
};
}
SpanIo {
input: raw_prompt.to_owned(),
output: raw_completion.to_owned(),
calls: CallEvidence::Unknown,
}
}
impl Format for LangSmith {
fn matches(&self, context: &SpanContext<'_>) -> bool {
context.scope == "langsmith" || context.attributes.contains_key("langsmith.span.kind")
}
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
let attributes = context.attributes;
let base = GenAi.extract(context)?;
let observation_type = ObservationType::try_from(attr(attributes, "langsmith.span.kind"))
.unwrap_or(ObservationType::Chain);
let io = span_io(observation_type, attributes);
Ok(Extraction {
facts: SpanFacts {
role: Some(RoleEvidence::Declared(observation_type)),
input: if attr(attributes, "gen_ai.prompt").is_empty() {
String::new()
} else {
io.input
},
output: if attr(attributes, "gen_ai.completion").is_empty() {
String::new()
} else {
io.output
},
calls: io.calls,
..SpanFacts::default()
}
.or(base.facts),
display_name: None,
consumed_attributes: base.consumed_attributes,
})
}
}
#[cfg(test)]
mod tests {
use std::collections::BTreeMap;
use rstest::rstest;
use serde_json::{Value, json};
use super::{CallEvidence, ObservationType, span_io};
use crate::normalize::CallKey;
#[rstest]
fn malformed_messages_preserve_valid_input_and_response_id() {
let attributes = BTreeMap::from([
(
"gen_ai.prompt".to_owned(),
r#"{"messages":[[{"kwargs":{"type":"human","content":"hello"}},null]]}"#.to_owned(),
),
(
"gen_ai.completion".to_owned(),
r#"{"messages":"unexpected","generations":[[{"message":{"kwargs":{"type":"ai","content":"hi","response_metadata":{"id":"response-1"}}}}]]}"#.to_owned(),
),
]);
let io = span_io(ObservationType::Llm, &attributes);
let input: Value = serde_json::from_str(&io.input).expect("normalized input");
assert_eq!(input.as_array().expect("messages").len(), 1);
assert_eq!(input[0]["content"], "hello");
assert_eq!(
io.calls,
CallEvidence::complete(CallKey::ProviderResponse("response-1".to_owned()))
);
}
#[rstest]
#[case::null(r#"{"output":null}"#, Value::Null)]
#[case::string(r#""answer""#, json!("answer"))]
#[case::wrapped_string(r#"{"output":"answer","other":7}"#, json!("answer"))]
#[case::repeated_output(r#"{"output":"first","output":"last"}"#, json!("last"))]
#[case::wrapped_content(r#"{"output":{"content":"answer"}}"#, json!("answer"))]
#[case::last_command_message(r#"{"output":{"update":{"messages":[{"content":"first"},{"content":"last"}]}}}"#, json!("last"))]
#[case::direct_command(r#"{"update":{"messages":[{"content":"answer"}]}}"#, json!("answer"))]
#[case::empty_command(r#"{"update":{"messages":[]}}"#, json!({"update":{"messages":[]}}))]
#[case::arbitrary_object(r#"{"result":7}"#, json!({"result":7}))]
#[case::arbitrary_array(r#"[1,2]"#, json!([1,2]))]
#[case::malformed("not-json", Value::Null)]
fn tool_outputs_preserve_content_and_fallbacks(
#[case] completion: &str,
#[case] expected: Value,
) {
let attributes = BTreeMap::from([("gen_ai.completion".to_owned(), completion.to_owned())]);
let io = span_io(ObservationType::Tool, &attributes);
match expected {
Value::String(text) => assert_eq!(io.output, text),
value => assert_eq!(serde_json::from_str::<Value>(&io.output).unwrap(), value),
}
}
#[rstest]
fn absent_llm_messages_render_as_an_empty_list() {
let attributes = BTreeMap::from([("gen_ai.completion".to_owned(), "{}".to_owned())]);
let io = span_io(ObservationType::Llm, &attributes);
assert_eq!(io.input, "[]");
}
}

View file

@ -0,0 +1,64 @@
use serde::Deserialize;
use serde_json::Value;
use super::{Extraction, Format, SpanFacts, genai::GenAi};
use crate::{
Error,
normalize::{SpanContext, messages, select_attribute},
};
pub(crate) struct Logfire;
impl Format for Logfire {
fn matches(&self, context: &SpanContext<'_>) -> bool {
context.attributes.contains_key("all_messages_events")
|| ((context.scope.starts_with("logfire") || context.scope == "pydantic-ai")
&& (context.attributes.contains_key("events")
|| context.attributes.contains_key("prompt")))
}
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
let base = GenAi.extract(context)?;
let input = base
.facts
.input
.is_empty()
.then(|| select_attribute(context.attributes, &["prompt"]))
.flatten();
let output = base
.facts
.output
.is_empty()
.then(|| select_attribute(context.attributes, &["final_result"]))
.flatten();
let recorded = select_attribute(context.attributes, &["all_messages_events", "events"]);
let values = recorded
.as_ref()
.and_then(|value| serde_json::from_str::<Vec<Value>>(value.text).ok())
.unwrap_or_default();
let events: Vec<_> = values
.iter()
.filter_map(|value| messages::EventMessage::deserialize(value).ok()?.recorded())
.collect();
Ok(Extraction {
facts: base.facts.or(SpanFacts {
input: input
.as_ref()
.map(|value| messages::canonical(value.text))
.or_else(|| messages::event_payload(&events, false))
.unwrap_or_default(),
output: output
.as_ref()
.map(|value| value.text.to_owned())
.or_else(|| messages::event_payload(&events, true))
.unwrap_or_default(),
..SpanFacts::default()
}),
display_name: base.display_name,
consumed_attributes: base.consumed_attributes,
}
.consuming(input)
.consuming(output)
.consuming(recorded))
}
}

View file

@ -0,0 +1,133 @@
//! Step one of normalization: what a span records, read in the format it was recorded in.
use super::{AttributeText, CallEvidence, RoleEvidence, SpanContext};
use crate::Error;
pub(crate) mod claude_code;
pub(crate) mod genai;
pub(crate) mod langsmith;
pub(crate) mod logfire;
pub(crate) mod openinference;
pub(crate) mod traceloop;
pub(crate) mod vercel;
/// What a span records, read in its convention's format.
#[derive(Debug, Default)]
pub(crate) struct SpanFacts {
pub role: Option<RoleEvidence>,
pub agent_name: Option<String>,
pub model: Option<String>,
pub input_tokens: u32,
pub output_tokens: u32,
pub input: String,
pub output: String,
pub tool_call_id: Option<String>,
pub calls: CallEvidence,
/// Set when the latest user message is not simply read from `input`.
pub input_preview: Option<String>,
}
impl SpanFacts {
pub(crate) fn or(self, fallback: Self) -> Self {
Self {
role: self.role.or(fallback.role),
agent_name: self.agent_name.or(fallback.agent_name),
model: self.model.or(fallback.model),
input_tokens: if self.input_tokens == 0 {
fallback.input_tokens
} else {
self.input_tokens
},
output_tokens: if self.output_tokens == 0 {
fallback.output_tokens
} else {
self.output_tokens
},
input: if self.input.is_empty() {
fallback.input
} else {
self.input
},
output: if self.output.is_empty() {
fallback.output
} else {
self.output
},
tool_call_id: self.tool_call_id.or(fallback.tool_call_id),
calls: if self.calls == CallEvidence::Unknown {
fallback.calls
} else {
self.calls
},
input_preview: self.input_preview.or(fallback.input_preview),
}
}
}
/// A convention's complete reading of a span, including which attributes it consumed.
pub(crate) struct Extraction {
pub facts: SpanFacts,
pub display_name: Option<String>,
pub consumed_attributes: Vec<&'static str>,
}
impl Extraction {
pub(crate) fn consuming(self, attribute: Option<AttributeText<'_>>) -> Self {
Self {
consumed_attributes: self
.consumed_attributes
.into_iter()
.chain(attribute.map(|value| value.source))
.collect(),
..self
}
}
pub(crate) fn map_facts(self, adjust: impl FnOnce(SpanFacts) -> SpanFacts) -> Self {
Self {
facts: adjust(self.facts),
..self
}
}
}
/// A payload read from one attribute, which the extraction then reports as consumed.
#[derive(Default)]
pub(crate) struct Payload {
pub text: String,
pub consumed: Option<&'static str>,
}
impl From<AttributeText<'_>> for Payload {
fn from(attribute: AttributeText<'_>) -> Self {
Self {
text: attribute.text.to_owned(),
consumed: Some(attribute.source),
}
}
}
/// A span format: whether a span is recorded in it, and what the span then records.
pub(crate) trait Format {
fn matches(&self, context: &SpanContext<'_>) -> bool;
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error>;
}
/// In precedence order. `gen_ai` accepts every span, so it is last.
const FORMATS: [&dyn Format; 7] = [
&claude_code::ClaudeCode,
&langsmith::LangSmith,
&openinference::OpenInference,
&traceloop::Traceloop,
&vercel::Vercel,
&logfire::Logfire,
&genai::GenAi,
];
pub(crate) fn extract(context: &SpanContext<'_>) -> Result<Extraction, Error> {
FORMATS
.into_iter()
.find(|format| format.matches(context))
.unwrap_or(&genai::GenAi)
.extract(context)
}

View file

@ -0,0 +1,128 @@
use std::collections::BTreeMap;
use litellm_llms_types::recognized::Recognized;
use serde::{Deserialize, de::IgnoredAny};
use serde_json::Value;
use super::{Extraction, Format, Payload, SpanFacts};
use crate::{
Error,
normalize::{
CallEvidence, CallKey, ObservationType, RoleEvidence, SpanContext, attr, messages, present,
select_attribute, tokens, usage_tokens,
},
};
/// Arize OpenInference: spans carry `openinference.span.kind`.
pub(crate) struct OpenInference;
#[derive(Deserialize)]
struct ResponseIdentity {
#[serde(default, deserialize_with = "messages::present")]
id: Option<Recognized<String>>,
#[serde(flatten)]
_other: BTreeMap<String, IgnoredAny>,
}
#[derive(Deserialize)]
struct ProviderResponse {
raw: Option<Recognized<ResponseIdentity>>,
#[serde(flatten)]
response: ResponseIdentity,
}
impl ProviderResponse {
fn id(&self) -> Option<&str> {
let identity = match &self.response.id {
Some(id) => return id.known().map(String::as_str),
None => self.raw.as_ref()?.known()?,
};
identity.id.as_ref()?.known().map(String::as_str)
}
}
fn role(context: &SpanContext<'_>) -> Option<RoleEvidence> {
let root = context.parent_span_id.is_empty();
match ObservationType::try_from(attr(context.attributes, "openinference.span.kind")) {
// A root chain (crew kickoff, workflow run) may be the agent run or only wrap its agents.
Ok(ObservationType::Chain) if root => {
Some(RoleEvidence::WrapperCandidate(ObservationType::Agent))
}
Ok(kind) => Some(RoleEvidence::Declared(kind)),
_ if root => None,
_ => Some(RoleEvidence::Declared(ObservationType::Chain)),
}
}
/// LLM instrumentations record the provider response as `output.value`: a raw response is one
/// request (`id`); a LangChain `LLMResult` carries one per prompt.
fn calls(output: &str) -> CallEvidence {
let Ok(value) = serde_json::from_str::<Value>(output) else {
return CallEvidence::Unknown;
};
if let Ok(response) = ProviderResponse::deserialize(&value)
&& let Some(id) = response.id()
{
return CallEvidence::complete(CallKey::ProviderResponse(id.to_owned()));
}
messages::langchain_result(&value).map_or(CallEvidence::Unknown, |result| result.calls)
}
/// `llm.<direction>_messages.*` when the instrumentation flattened the messages, else `raw`.
fn payload(context: &SpanContext<'_>, flattened: &str, raw: &'static str) -> Payload {
if let Some(conversation) = messages::flattened(context.attributes, flattened) {
return Payload {
text: messages::encode(&conversation),
consumed: None,
};
}
select_attribute(context.attributes, &[raw])
.map(Payload::from)
.unwrap_or_default()
}
/// OpenInference's own count when recorded, else the `gen_ai.usage.*` one.
fn token_count(attributes: &BTreeMap<String, String>, key: &str, usage: u32) -> Result<u32, Error> {
if attributes.contains_key(key) {
tokens(attributes, key)
} else {
Ok(usage)
}
}
impl Format for OpenInference {
fn matches(&self, context: &SpanContext<'_>) -> bool {
context.attributes.contains_key("openinference.span.kind")
}
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
let attributes = context.attributes;
let (usage_input, usage_output) = usage_tokens(attributes)?;
let role = role(context);
let input = payload(context, "llm.input_messages", "input.value");
let output = payload(context, "llm.output_messages", "output.value");
Ok(Extraction {
facts: SpanFacts {
role,
agent_name: present(attributes, &["agent.name"]),
model: present(attributes, &["llm.model_name", "embedding.model_name"]),
input_tokens: token_count(attributes, "llm.token_count.prompt", usage_input)?,
output_tokens: token_count(attributes, "llm.token_count.completion", usage_output)?,
input: input.text,
output: output.text,
tool_call_id: present(attributes, &["tool.id"]),
calls: if role == Some(RoleEvidence::Declared(ObservationType::Llm)) {
calls(attr(attributes, "output.value"))
} else {
CallEvidence::Unknown
},
input_preview: None,
},
display_name: None,
consumed_attributes: [input.consumed, output.consumed]
.into_iter()
.flatten()
.collect(),
})
}
}

View file

@ -0,0 +1,57 @@
use super::{Extraction, Format, SpanFacts, genai::GenAi};
use crate::{
Error,
normalize::{
ObservationType, RoleEvidence, SpanContext, attr, messages, present, select_attribute,
},
};
pub(crate) struct Traceloop;
impl Format for Traceloop {
fn matches(&self, context: &SpanContext<'_>) -> bool {
context
.attributes
.keys()
.any(|key| key.starts_with("traceloop."))
}
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
let base = GenAi.extract(context)?;
let role = match attr(context.attributes, "traceloop.span.kind") {
"agent" => Some(ObservationType::Agent),
"tool" => Some(ObservationType::Tool),
"workflow" | "task" => Some(ObservationType::Chain),
_ => match present(
context.attributes,
&["traceloop.llm.request.type", "llm.request.type"],
)
.as_deref()
{
Some("embedding" | "embeddings") => Some(ObservationType::Embedding),
Some("chat" | "completion") => Some(ObservationType::Llm),
_ => None,
},
};
let input = select_attribute(context.attributes, &["traceloop.entity.input"]);
let output = select_attribute(context.attributes, &["traceloop.entity.output"]);
Ok(Extraction {
facts: SpanFacts {
role: role.map(RoleEvidence::Declared),
input: input
.as_ref()
.map_or(String::new(), |value| messages::canonical(value.text)),
output: output
.as_ref()
.map_or(String::new(), |value| messages::canonical(value.text)),
..SpanFacts::default()
}
.or(base.facts),
display_name: present(context.attributes, &["traceloop.entity.name"])
.or(base.display_name),
consumed_attributes: base.consumed_attributes,
}
.consuming(input)
.consuming(output))
}
}

View file

@ -0,0 +1,171 @@
use serde::{Deserialize, Serialize};
use serde_json::Value;
use super::{Extraction, Format, SpanFacts, genai::GenAi};
use crate::{
Error,
normalize::{
ObservationType, RoleEvidence, SpanContext, attr, messages, present, select_attribute,
token_alias,
},
};
pub(crate) struct Vercel;
#[derive(Deserialize)]
struct Prompt {
messages: Option<Value>,
prompt: Option<String>,
system: Option<String>,
}
#[derive(Deserialize, Serialize)]
struct ToolCall {
#[serde(rename(deserialize = "toolCallId"))]
id: String,
#[serde(rename(deserialize = "toolName"))]
name: String,
#[serde(alias = "args", alias = "input")]
arguments: Value,
}
fn prompt(raw: &str) -> String {
let Ok(value) = serde_json::from_str::<Prompt>(raw) else {
return messages::canonical(raw);
};
let content = value.messages.unwrap_or_else(|| {
Value::Array(
value
.prompt
.into_iter()
.map(|text| serde_json::json!({"role": "user", "content": text}))
.collect(),
)
});
let conversation: Vec<Value> = value
.system
.into_iter()
.map(|text| serde_json::json!({"role": "system", "content": text}))
.chain(content.as_array().into_iter().flatten().cloned())
.collect();
if conversation.is_empty() {
return raw.to_owned();
}
messages::canonical(&messages::encode(&conversation))
}
impl Format for Vercel {
fn matches(&self, context: &SpanContext<'_>) -> bool {
context.attributes.contains_key("ai.operationId")
|| (context.scope == "ai"
&& context.attributes.keys().any(|key| key.starts_with("ai.")))
}
fn extract(&self, context: &SpanContext<'_>) -> Result<Extraction, Error> {
let base = GenAi.extract(context)?;
let operation = attr(context.attributes, "ai.operationId");
let role = match operation {
"ai.toolCall" => Some(ObservationType::Tool),
"ai.embed" | "ai.embedMany" | "ai.embed.doEmbed" | "ai.embedMany.doEmbed" => {
Some(ObservationType::Embedding)
}
"ai.generateText"
| "ai.streamText"
| "ai.generateObject"
| "ai.streamObject"
| "ai.generateText.doGenerate"
| "ai.streamText.doStream"
| "ai.generateObject.doGenerate"
| "ai.streamObject.doStream" => Some(ObservationType::Llm),
_ => None,
};
let input = base
.facts
.input
.is_empty()
.then(|| {
select_attribute(
context.attributes,
&[
"ai.toolCall.args",
"ai.prompt.messages",
"ai.prompt",
"ai.value",
"ai.values",
],
)
})
.flatten();
let output = base
.facts
.output
.is_empty()
.then(|| {
select_attribute(
context.attributes,
&[
"ai.toolCall.result",
"ai.response.object",
"ai.response.text",
"ai.embeddings",
"ai.embedding",
],
)
})
.flatten();
let calls = base
.facts
.output
.is_empty()
.then(|| select_attribute(context.attributes, &["ai.response.toolCalls"]))
.flatten();
let response = calls
.as_ref()
.and_then(|value| serde_json::from_str::<Vec<ToolCall>>(value.text).ok());
let legacy_output = match response {
Some(calls) => messages::canonical(&messages::encode(&serde_json::json!([{
"role": "assistant", "content": output.as_ref().map_or("", |value| value.text), "tool_calls": calls,
}]))),
None => output
.as_ref()
.map_or(String::new(), |value| value.text.to_owned()),
};
Ok(Extraction {
facts: base.facts.or(SpanFacts {
role: role.map(RoleEvidence::Declared),
model: present(context.attributes, &["ai.model.id"]),
input_tokens: token_alias(
context.attributes,
&[
"gen_ai.usage.input_tokens",
"gen_ai.usage.prompt_tokens",
"ai.usage.promptTokens",
"ai.usage.tokens",
],
)?,
output_tokens: token_alias(
context.attributes,
&[
"gen_ai.usage.output_tokens",
"gen_ai.usage.completion_tokens",
"ai.usage.completionTokens",
],
)?,
input: input
.as_ref()
.map_or(String::new(), |value| match value.source {
"ai.prompt" | "ai.prompt.messages" => prompt(value.text),
_ => value.text.to_owned(),
}),
output: legacy_output,
tool_call_id: present(context.attributes, &["ai.toolCall.id"]),
..SpanFacts::default()
}),
display_name: present(context.attributes, &["ai.toolCall.name"]).or(base.display_name),
consumed_attributes: base.consumed_attributes,
}
.consuming(input)
.consuming(output)
.consuming(calls))
}
}

View file

@ -1,65 +0,0 @@
use std::collections::BTreeMap;
use super::{NormalizedSpan, ObservationType, SpanNormalizer, attr, first, usage_tokens};
use crate::{Error, otlp::DecodedEvent};
pub(super) struct GenAiNormalizer;
impl SpanNormalizer for GenAiNormalizer {
fn matches(&self, _scope_name: &str, _attributes: &BTreeMap<String, String>) -> bool {
true
}
fn consumed_attributes(&self, attributes: &BTreeMap<String, String>) -> [&'static str; 2] {
[
if attr(attributes, "gen_ai.input.messages").is_empty() {
"gen_ai.tool.call.arguments"
} else {
"gen_ai.input.messages"
},
if attr(attributes, "gen_ai.output.messages").is_empty() {
"gen_ai.tool.call.result"
} else {
"gen_ai.output.messages"
},
]
}
fn normalize(
&self,
_name: &str,
parent_span_id: &str,
attributes: &BTreeMap<String, String>,
_events: &[DecodedEvent],
) -> Result<NormalizedSpan, Error> {
let (input_tokens, output_tokens) = usage_tokens(attributes)?;
let observation_type = match attr(attributes, "gen_ai.operation.name") {
"invoke_agent" => ObservationType::Agent,
"chat" | "text_completion" | "generate_content" => ObservationType::Llm,
"execute_tool" => ObservationType::Tool,
_ if parent_span_id.is_empty() => ObservationType::Agent,
_ => ObservationType::Chain,
};
Ok(NormalizedSpan {
observation_type,
agent_name: attr(attributes, "gen_ai.agent.name").to_owned(),
framework: String::new(),
litellm_request_id: attr(attributes, "gen_ai.response.id").to_owned(),
model: first(attributes, "gen_ai.request.model", "gen_ai.response.model").to_owned(),
input_tokens,
output_tokens,
input: first(
attributes,
"gen_ai.input.messages",
"gen_ai.tool.call.arguments",
)
.to_owned(),
output: first(
attributes,
"gen_ai.output.messages",
"gen_ai.tool.call.result",
)
.to_owned(),
})
}
}

View file

@ -0,0 +1,7 @@
- Interpret extracted facts using known behavior of the SDK or instrumentor that emitted the span
- Own SDK detection, integration identity, agent naming, role adjustments, input previews, and call-evidence guarantees
- Require positive SDK evidence before applying a rule; preserve detection precedence when scopes overlap
- Mark call evidence complete only when the emitting contract guarantees which calls the span represents, never from the number of IDs found
- Keep attribute conventions and payload decoding in `../format/`; reuse `../messages.rs` for message and state conversion
- Emit role and call evidence for `resolve/`; do not infer wrappers, ownership, or spend from spans outside the current context
- Add regression cases to the existing public normalization tests for SDK behavior and ambiguous or unmatched input

View file

@ -0,0 +1,12 @@
use super::{ObservationType, RoleEvidence, SpanFacts};
pub(super) fn adjust(facts: SpanFacts) -> SpanFacts {
if facts.agent_name.as_deref() != Some("Agent") {
return facts;
}
SpanFacts {
role: Some(RoleEvidence::WrapperCandidate(ObservationType::Agent)),
agent_name: None,
..facts
}
}

Some files were not shown because too many files have changed in this diff Show more