feat(tracing): own the query API contract in Rust

This commit is contained in:
Yujong Lee 2026-10-04 20:57:58 -07:00
parent a43f111abe
commit b630654146
145 changed files with 9433 additions and 1793 deletions

View file

@ -4458,6 +4458,7 @@ dependencies = [
"base64 0.22.1",
"criterion",
"indexmap 2.14.0",
"jsonschema",
"litellm-llms-types",
"macro_rules_attribute",
"opentelemetry-proto",
@ -4469,6 +4470,7 @@ dependencies = [
"strum",
"thiserror 2.0.19",
"time",
"utoipa",
]
[[package]]
@ -7977,6 +7979,29 @@ version = "1.0.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be"
[[package]]
name = "utoipa"
version = "5.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8bde15df68e80b16c7d16b9616e80770ad158988daa56a27dccd1e55558b0160"
dependencies = [
"indexmap 2.14.0",
"serde",
"serde_json",
"utoipa-gen",
]
[[package]]
name = "utoipa-gen"
version = "5.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6ba0b99ee52df3028635d93840c797102da61f8a7bb3cf751032455895b52ef8"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.119",
]
[[package]]
name = "uuid"
version = "1.24.0"

View file

@ -83,6 +83,7 @@ pyo3-async-runtimes = { version = "0.29.0", features = ["tokio-runtime"] }
rand = "0.8"
macro_rules_attribute = "0.2.3"
schemars = "1"
utoipa = { version = "5.5.0", default-features = false, features = ["macros"] }
reqwest = { version = "0.12", default-features = false, features = ["json", "multipart", "rustls-tls", "http2", "stream"] }
qdrant-client = { version = "1.19.0", default-features = false }
uuid = { version = "1", features = ["v4"] }

View file

@ -7,6 +7,7 @@ use std::{
use litellm_http::ClientVariant;
use litellm_traces::{
QueryScope, Tenant,
api::TraceQueryWindow,
search::{RunField, RunFilter, RunSearch},
store::{RunOrder, SpanPart, TextRange},
};
@ -98,13 +99,38 @@ fn map_read_error(error: ReadError<Error>) -> PyErr {
}
}
fn run_filter(start_ms: i64, end_ms: i64, q: &str, trace_refs: Vec<String>) -> RunFilter {
RunFilter {
fn run_window(
start_ms: Option<i64>,
end_ms: Option<i64>,
as_of_ms: Option<u64>,
cursor: Option<&str>,
) -> PyResult<TraceQueryWindow> {
let now_ms = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_millis() as u64;
resolve_run_window(
start_ms,
end_ms,
search: RunSearch::parse(q),
as_of_ms,
cursor,
TraceQueryWindow {
start_ms: now_ms as i64 - 86_400_000,
end_ms: now_ms as i64,
as_of_ms: now_ms.saturating_sub(1),
},
)
.map_err(map_read_error)
}
fn run_filter(window: TraceQueryWindow, q: &str, trace_refs: Vec<String>) -> PyResult<RunFilter> {
Ok(RunFilter {
start_ms: window.start_ms,
end_ms: window.end_ms,
as_of_ms: window.as_of_ms,
search: RunSearch::parse(q).map_err(|error| PyValueError::new_err(error.to_string()))?,
trace_refs,
}
})
}
fn parsed<T: std::str::FromStr>(kind: &str, value: &str) -> PyResult<T> {
@ -278,7 +304,7 @@ impl NativeTraceStorage {
)
}
#[pyo3(signature = (scope, start_ms, end_ms, q, cursor, limit, order, trace_refs=Vec::new()))]
#[pyo3(signature = (scope, start_ms, end_ms, q, cursor, limit, order, trace_refs=Vec::new(), as_of_ms=None))]
#[expect(
clippy::too_many_arguments,
reason = "one parameter per Python argument"
@ -294,19 +320,10 @@ impl NativeTraceStorage {
limit: u32,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] order: RunOrder,
trace_refs: Vec<String>,
as_of_ms: Option<u64>,
) -> PyResult<Bound<'py, PyAny>> {
let now_ms = SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_millis() as i64;
let window = resolve_run_window(
start_ms,
end_ms,
cursor.as_deref(),
(now_ms - 86_400_000, now_ms),
)
.map_err(map_read_error)?;
let filter = run_filter(window.0, window.1, q, trace_refs);
let window = run_window(start_ms, end_ms, as_of_ms, cursor.as_deref())?;
let filter = run_filter(window, q, trace_refs)?;
let page = PageRequest { cursor, limit };
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
@ -323,17 +340,23 @@ impl NativeTraceStorage {
)
}
#[pyo3(signature = (scope, start_ms, end_ms, q, trace_refs=Vec::new()))]
#[pyo3(signature = (scope, start_ms, end_ms, q, trace_refs=Vec::new(), as_of_ms=None))]
#[expect(
clippy::too_many_arguments,
reason = "one parameter per Python argument"
)]
fn count_traces<'py>(
&self,
py: Python<'py>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
start_ms: i64,
end_ms: i64,
start_ms: Option<i64>,
end_ms: Option<i64>,
q: &str,
trace_refs: Vec<String>,
as_of_ms: Option<u64>,
) -> PyResult<Bound<'py, PyAny>> {
let filter = run_filter(start_ms, end_ms, q, trace_refs);
let window = run_window(start_ms, end_ms, as_of_ms, None)?;
let filter = run_filter(window, q, trace_refs)?;
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
@ -389,16 +412,23 @@ impl NativeTraceStorage {
)
}
#[pyo3(signature = (scope, start_ms, end_ms, q, buckets, as_of_ms=None))]
#[expect(
clippy::too_many_arguments,
reason = "one parameter per Python argument"
)]
fn trace_histogram<'py>(
&self,
py: Python<'py>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
start_ms: i64,
end_ms: i64,
start_ms: Option<i64>,
end_ms: Option<i64>,
q: &str,
buckets: u32,
as_of_ms: Option<u64>,
) -> PyResult<Bound<'py, PyAny>> {
let filter = run_filter(start_ms, end_ms, q, Vec::new());
let window = run_window(start_ms, end_ms, as_of_ms, None)?;
let filter = run_filter(window, q, Vec::new())?;
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
@ -416,21 +446,24 @@ impl NativeTraceStorage {
clippy::too_many_arguments,
reason = "one parameter per Python argument"
)]
#[pyo3(signature = (scope, start_ms, end_ms, q, field, contains, limit, as_of_ms=None))]
fn run_values<'py>(
&self,
py: Python<'py>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
start_ms: i64,
end_ms: i64,
start_ms: Option<i64>,
end_ms: Option<i64>,
q: &str,
field: &str,
contains: &str,
limit: u32,
as_of_ms: Option<u64>,
) -> PyResult<Bound<'py, PyAny>> {
let field = field
.parse::<RunField>()
.map_err(|_| PyValueError::new_err(format!("unknown run field {field}")))?;
let filter = run_filter(start_ms, end_ms, q, Vec::new());
let window = run_window(start_ms, end_ms, as_of_ms, None)?;
let filter = run_filter(window, q, Vec::new())?;
let contains = contains.to_owned();
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
@ -542,12 +575,103 @@ impl NativeTraceStorage {
)
}
fn get_trace_metadata<'py>(
&self,
py: Python<'py>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
id: String,
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
crate::execution::run_async(
py,
async move {
let store = ClickHouseTraces::new(client, connection);
reader.get_trace_metadata(&store, &scope, &id).await
},
map_read_error,
)
}
#[pyo3(signature = (scope, id, cursor, page_size))]
fn get_trace_spans<'py>(
&self,
py: Python<'py>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
id: String,
cursor: Option<String>,
page_size: u32,
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
crate::execution::run_async(
py,
async move {
let store = ClickHouseTraces::new(client, connection);
reader
.get_trace_spans(&store, &scope, &id, cursor.as_deref(), page_size)
.await
},
map_read_error,
)
}
fn get_span_by_id<'py>(
&self,
py: Python<'py>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
id: String,
span_id: String,
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
crate::execution::run_async(
py,
async move {
let store = ClickHouseTraces::new(client, connection);
reader.get_span_by_id(&store, &scope, &id, &span_id).await
},
map_read_error,
)
}
#[pyo3(signature = (scope, id, span_id, cursor=None))]
fn get_span_error_by_id<'py>(
&self,
py: Python<'py>,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
id: String,
span_id: String,
cursor: Option<String>,
) -> PyResult<Bound<'py, PyAny>> {
let client = crate::http::host_client(py, ClientVariant::NoRedirect)?;
let connection = self.config.storage().reader().clone();
let reader = Arc::clone(&self.reader);
crate::execution::run_async(
py,
async move {
let store = ClickHouseTraces::new(client, connection);
reader
.get_span_error_by_id(&store, &scope, &id, &span_id, cursor.as_deref())
.await
},
map_read_error,
)
}
fn query_sql<'py>(
&self,
py: Python<'py>,
sql: String,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope,
secret: String,
#[pyo3(from_py_with = litellm_host_python::from_py_argument)] params: BTreeMap<
String,
litellm_traces::api::SqlParameter,
>,
) -> PyResult<Bound<'py, PyAny>> {
if sql.trim().is_empty() {
return Err(map_error(
@ -561,7 +685,13 @@ impl NativeTraceStorage {
async move {
let _permit = readers.acquire()?;
let connection = readers.connection(&client, &scope, &secret).await?;
litellm_traces_clickhouse::query_sql(&client, &connection, &sql).await
litellm_traces_clickhouse::query_sql_with_params(
&client,
&connection,
&sql,
&params,
)
.await
},
map_sql_error,
)

View file

@ -25,6 +25,8 @@ pub enum Parameter {
Integer(i64),
Unsigned(u64),
Float(f64),
Boolean(bool),
Null,
Strings(Vec<String>),
}
@ -35,6 +37,8 @@ impl Parameter {
Self::Integer(value) => value.to_string(),
Self::Unsigned(value) => value.to_string(),
Self::Float(value) => value.to_string(),
Self::Boolean(value) => u8::from(*value).to_string(),
Self::Null => "\\N".to_owned(),
Self::Strings(values) => format!(
"[{}]",
values

View file

@ -38,6 +38,8 @@ struct QueryParams {
signed: i64,
unsigned: u64,
float: f64,
boolean: bool,
nullable: Option<String>,
text: String,
strings: Vec<String>,
}
@ -79,6 +81,8 @@ async fn typed_fetch_encodes_parameters_and_validates_rows(
.and(query_param("param_signed", i64::MIN.to_string()))
.and(query_param("param_unsigned", u64::MAX.to_string()))
.and(query_param("param_float", "12.5"))
.and(query_param("param_boolean", "1"))
.and(query_param("param_nullable", "\\N"))
.and(query_param("param_text", "line\\nbreak"))
.and(query_param("param_strings", "['a\\'b','雪']"))
.and(query_param("readonly", "1"))
@ -93,6 +97,8 @@ async fn typed_fetch_encodes_parameters_and_validates_rows(
signed: i64::MIN,
unsigned: u64::MAX,
float: 12.5,
boolean: true,
nullable: None,
text: "line\nbreak".into(),
strings: vec!["a'b".into(), "雪".into()],
};

View file

@ -45,7 +45,7 @@ impl SnapshotKey {
pub(crate) fn run(
source: &str,
access: &QueryScope,
run: (&str, &str, &str, &str),
run: (&str, &str, &str, &str, u64),
) -> Result<Self, Error> {
Self::digest(&("run", source, access, run))
}
@ -55,7 +55,7 @@ impl SnapshotKey {
access: &QueryScope,
filter: &RunFilter,
) -> Result<String, Error> {
Self::digest(&("run_page_v1", source, access, filter)).map(|key| key.0)
Self::digest(&("run_page_v2", source, access, filter)).map(|key| key.0)
}
pub(crate) fn scope(source: &str, access: &QueryScope) -> Result<Self, Error> {

View file

@ -1,4 +1,5 @@
use base64::{Engine, engine::general_purpose::URL_SAFE};
use litellm_traces::api::TraceQueryWindow;
use litellm_traces::store::{RunCursor, RunOrder, RunRow, SpanPart};
use serde::{Deserialize, Serialize};
@ -45,7 +46,7 @@ impl Cursor {
pub(super) struct RunPosition {
order: RunOrder,
query_scope: String,
window: (i64, i64),
window: TraceQueryWindow,
value: i64,
trace_ref: String,
}
@ -55,7 +56,7 @@ impl RunPosition {
order: RunOrder,
row: &RunRow,
query_scope: &str,
window: (i64, i64),
window: TraceQueryWindow,
) -> Self {
let RunCursor { value, trace_ref } = order.cursor(row);
Self {
@ -73,6 +74,7 @@ impl RunPosition {
pub(super) struct SpanPosition {
pub(super) trace_ref: String,
pub(super) snapshot_ms: u64,
pub(super) page_size: u32,
pub(super) offset: usize,
pub(super) version: String,
}
@ -89,7 +91,7 @@ pub(super) fn run_position<E>(
cursor: Option<&str>,
order: RunOrder,
query_scope: &str,
window: (i64, i64),
window: TraceQueryWindow,
) -> Result<Option<RunCursor>, ReadError<E>> {
let Some(cursor) = cursor.filter(|cursor| !cursor.is_empty()) else {
return Ok(None);
@ -113,15 +115,17 @@ pub(super) fn run_position<E>(
pub fn resolve_run_window<E>(
start_ms: Option<i64>,
end_ms: Option<i64>,
as_of_ms: Option<u64>,
cursor: Option<&str>,
default_window: (i64, i64),
) -> Result<(i64, i64), ReadError<E>> {
default: TraceQueryWindow,
) -> Result<TraceQueryWindow, ReadError<E>> {
let Some(cursor) = cursor.filter(|cursor| !cursor.is_empty()) else {
let window = (
start_ms.unwrap_or(default_window.0),
end_ms.unwrap_or(default_window.1),
);
return if window.0 < window.1 {
let window = TraceQueryWindow {
start_ms: start_ms.unwrap_or(default.start_ms),
end_ms: end_ms.unwrap_or(default.end_ms),
as_of_ms: as_of_ms.unwrap_or(default.as_of_ms),
};
return if window.start_ms < window.end_ms && window.as_of_ms <= default.as_of_ms {
Ok(window)
} else {
Err(ReadError::InvalidParameters)
@ -130,13 +134,16 @@ pub fn resolve_run_window<E>(
let Cursor::Run(position) = Cursor::decode(cursor, "trace")? else {
return Err(ReadError::InvalidCursor("trace"));
};
if position.window.0 >= position.window.1
|| start_ms.is_some_and(|start| start != position.window.0)
|| end_ms.is_some_and(|end| end != position.window.1)
let window = position.window;
if window.start_ms >= window.end_ms
|| window.as_of_ms > default.as_of_ms
|| start_ms.is_some_and(|start| start != window.start_ms)
|| end_ms.is_some_and(|end| end != window.end_ms)
|| as_of_ms.is_some_and(|cutoff| cutoff != window.as_of_ms)
{
return Err(ReadError::InvalidCursor("trace"));
}
Ok(position.window)
Ok(window)
}
pub(super) fn span_position<E>(cursor: &str) -> Result<SpanPosition, ReadError<E>> {
@ -174,7 +181,14 @@ mod tests {
use super::*;
const WINDOW: (i64, i64) = (10, 100);
const WINDOW: TraceQueryWindow = window(10, 100, 150);
const fn window(start_ms: i64, end_ms: i64, as_of_ms: u64) -> TraceQueryWindow {
TraceQueryWindow {
start_ms,
end_ms,
as_of_ms,
}
}
fn run(order: RunOrder, value: i64, trace_ref: &str) -> String {
Cursor::Run(RunPosition {
@ -200,6 +214,7 @@ mod tests {
Cursor::Span(SpanPosition {
trace_ref: "ref".into(),
snapshot_ms: 1,
page_size: 2,
offset: 2,
version: "A".repeat(64),
})
@ -237,11 +252,12 @@ mod tests {
#[case::order(BY_ERRORS, "query", WINDOW)]
#[case::direction(RunOrder { descending: false, ..RunOrder::NEWEST }, "query", WINDOW)]
#[case::scope(RunOrder::NEWEST, "other-query", WINDOW)]
#[case::window(RunOrder::NEWEST, "query", (11, 100))]
#[case::window(RunOrder::NEWEST, "query", window(11, 100, 150))]
#[case::cutoff(RunOrder::NEWEST, "query", window(10, 100, 151))]
fn run_cursor_rejects_a_changed_query(
#[case] order: RunOrder,
#[case] scope: &str,
#[case] window: (i64, i64),
#[case] window: TraceQueryWindow,
) {
assert!(matches!(
run_position::<std::io::Error>(
@ -290,41 +306,94 @@ mod tests {
#[case] expected: (i64, i64),
) {
assert_eq!(
resolve_run_window::<std::io::Error>(start, end, None, (0, 50)).unwrap(),
{
let resolved = resolve_run_window::<std::io::Error>(
start,
end,
None,
None,
window(0, 50, 150),
)
.unwrap();
assert_eq!(resolved.as_of_ms, 150);
(resolved.start_ms, resolved.end_ms)
},
expected
);
}
#[rstest]
#[case::omitted(None, None)]
#[case::start(Some(WINDOW.0), None)]
#[case::end(None, Some(WINDOW.1))]
#[case::explicit(Some(WINDOW.0), Some(WINDOW.1))]
#[case::start(Some(WINDOW.start_ms), None)]
#[case::end(None, Some(WINDOW.end_ms))]
#[case::explicit(Some(WINDOW.start_ms), Some(WINDOW.end_ms))]
fn cursor_keeps_its_window_when_the_default_clock_advances(
#[case] start: Option<i64>,
#[case] end: Option<i64>,
) {
let cursor = run(RunOrder::NEWEST, 1, "ref");
assert_eq!(
resolve_run_window::<std::io::Error>(start, end, Some(&cursor), (200, 300)).unwrap(),
resolve_run_window::<std::io::Error>(
start,
end,
None,
Some(&cursor),
window(200, 300, 350)
)
.unwrap(),
WINDOW
);
}
#[rstest]
#[case::start(Some(WINDOW.0 + 1), None)]
#[case::end(None, Some(WINDOW.1 + 1))]
#[case::start(Some(WINDOW.start_ms + 1), None)]
#[case::end(None, Some(WINDOW.end_ms + 1))]
fn cursor_rejects_explicit_window_changes(
#[case] start: Option<i64>,
#[case] end: Option<i64>,
) {
let cursor = run(RunOrder::NEWEST, 1, "ref");
assert!(matches!(
resolve_run_window::<std::io::Error>(start, end, Some(&cursor), (200, 300)),
resolve_run_window::<std::io::Error>(
start,
end,
None,
Some(&cursor),
window(200, 300, 350)
),
Err(ReadError::InvalidCursor("trace"))
));
}
#[rstest]
fn cursor_rejects_a_changed_ingestion_cutoff() {
let cursor = run(RunOrder::NEWEST, 1, "ref");
assert!(matches!(
resolve_run_window::<std::io::Error>(
None,
None,
Some(WINDOW.as_of_ms + 1),
Some(&cursor),
window(200, 300, 350)
),
Err(ReadError::InvalidCursor("trace"))
));
}
#[rstest]
fn first_page_rejects_a_future_ingestion_cutoff() {
assert!(matches!(
resolve_run_window::<std::io::Error>(
None,
None,
Some(WINDOW.as_of_ms + 1),
None,
WINDOW
),
Err(ReadError::InvalidParameters)
));
}
#[rstest]
fn span_cursor_round_trips() {
let position = span_position::<std::io::Error>(&span()).unwrap();

View file

@ -8,7 +8,7 @@ use litellm_traces::{
use crate::{
ReadError, SnapshotKey, TraceReader, TraceStore,
cache::{Freshness, ListedRun},
reader::{map_store_error, now_ms, spans},
reader::{map_store_error, now_ms, settle, spans},
spend::{spend, spend_window, spend_within},
store::StoreError,
};
@ -27,6 +27,7 @@ fn cache_key<E>(
source: &str,
access: &QueryScope,
row: &RunRow,
as_of_ms: u64,
) -> Result<SnapshotKey, ReadError<E>> {
Ok(SnapshotKey::run(
source,
@ -36,6 +37,7 @@ fn cache_key<E>(
&row.api_key_hash,
&row.trace_id,
&row.trace_ref,
as_of_ms,
),
)?)
}
@ -50,6 +52,8 @@ fn summary(row: &RunRow, listed: Option<&ListedRun>) -> TraceSummary {
duration_ms: row.duration_ns as f64 / 1_000_000.0,
span_count: row.span_count,
error_count: row.error_count,
has_error: row.error_count > 0,
status: row.status,
..summary
}
}
@ -61,11 +65,12 @@ pub(super) async fn list_summaries<S: TraceStore>(
store: &S,
access: &QueryScope,
runs: &[RunRow],
as_of_ms: u64,
) -> Result<Vec<TraceSummary>, ReadError<S::Error>> {
let mut keys = Vec::with_capacity(runs.len());
let mut listed = Vec::with_capacity(runs.len());
for row in runs {
let key = cache_key(store.source(), access, row)?;
let key = cache_key(store.source(), access, row, as_of_ms)?;
listed.push(reader.lists.runs.get(&key).await);
keys.push(key);
}
@ -74,7 +79,7 @@ pub(super) async fn list_summaries<S: TraceStore>(
.zip(&listed)
.filter_map(|(row, listed)| listed.is_none().then_some(row))
.collect();
let mut resolved = resolve_runs(reader, store, access, &misses)
let mut resolved = resolve_runs(reader, store, access, &misses, as_of_ms)
.await?
.into_iter();
let mut summaries = Vec::with_capacity(runs.len());
@ -101,6 +106,7 @@ async fn resolve_runs<S: TraceStore>(
store: &S,
access: &QueryScope,
runs: &[&RunRow],
snapshot_ms: u64,
) -> Result<Vec<Option<ListedRun>>, ReadError<S::Error>> {
let (Some(start_ms), Some(end_ms)) = (
runs.iter().map(|row| row.start_ms).min(),
@ -119,24 +125,23 @@ async fn resolve_runs<S: TraceStore>(
trace_refs: runs.iter().map(|row| row.trace_ref.clone()).collect(),
window: start_ms..end_ms.saturating_add(1),
};
let snapshot_ms = now_ms();
let spans = match spans(store, access, selection, snapshot_ms).await {
Ok(spans) => spans,
Err(StoreError::TooLarge) => {
let mut resolved = Vec::with_capacity(runs.len());
for row in runs {
resolved.push(resolve_run(reader, store, access, row).await?);
resolved.push(resolve_run(reader, store, access, row, snapshot_ms).await?);
}
return Ok(resolved);
}
Err(error) => return Err(map_store_error(error)),
};
let Some(spend_rows) = spend(store, access, &spans).await else {
let Some(spend_rows) = spend(store, access, &spans, snapshot_ms).await else {
// The batch's combined spend read failed; a run's own narrower window may still
// resolve, so fall back per run instead of leaving every run in the batch costless.
let mut resolved = Vec::with_capacity(runs.len());
for row in runs {
resolved.push(resolve_run(reader, store, access, row).await?);
resolved.push(resolve_run(reader, store, access, row, snapshot_ms).await?);
}
return Ok(resolved);
};
@ -148,7 +153,7 @@ async fn resolve_runs<S: TraceStore>(
&right.api_key_hash,
&right.trace_id,
))
.then(left.start_ns.cmp(&right.start_ns))
.then((left.start_ns, &left.span_id).cmp(&(right.start_ns, &right.span_id)))
});
let by_run: HashMap<_, &[SpanRow]> = spans
.chunk_by(|left, right| {
@ -174,7 +179,7 @@ async fn resolve_runs<S: TraceStore>(
resolve_trace(&row.trace_id, &row.trace_ref, spans, spend).map(|trace| {
ListedRun::Resolved(
Box::new(trace.summary),
Freshness::of(spans, true, snapshot_ms),
Freshness::of(spans, true, now_ms()),
)
})
})
@ -186,11 +191,13 @@ async fn resolve_run<S: TraceStore>(
store: &S,
access: &QueryScope,
row: &RunRow,
as_of_ms: u64,
) -> Result<Option<ListedRun>, ReadError<S::Error>> {
match reader
.current(store, access, &row.trace_id, &row.trace_ref)
.await
{
match settle(
reader
.pinned(store, access, &row.trace_id, &row.trace_ref, as_of_ms)
.await,
) {
Ok(snapshot) => Ok(snapshot.map(|snapshot| {
ListedRun::Resolved(
Box::new(snapshot.trace().summary.clone()),

View file

@ -10,7 +10,6 @@ use litellm_traces::{
CountBy, CountValue, RunCountQuery, RunOrder, RunQuery, RunSelection, SpanPart, SpanQuery,
SpanRow, SpanSelection, SpanText, SpanTextQuery, TextRange,
},
to_ui_content,
};
use crate::{
@ -50,7 +49,7 @@ impl<E> From<crate::Error> for Miss<E> {
}
}
fn settle<T, E>(result: Result<T, Arc<Miss<E>>>) -> Result<Option<T>, ReadError<E>> {
pub(super) fn settle<T, E>(result: Result<T, Arc<Miss<E>>>) -> Result<Option<T>, ReadError<E>> {
match result {
Ok(value) => Ok(Some(value)),
Err(miss) => match &*miss {
@ -87,12 +86,7 @@ impl TraceReader {
return Err(ReadError::InvalidParameters);
}
let query_scope = SnapshotKey::run_page_scope(store.source(), access, filter)?;
let after = run_position(
page.cursor.as_deref(),
order,
&query_scope,
(filter.start_ms, filter.end_ms),
)?;
let after = run_position(page.cursor.as_deref(), order, &query_scope, filter.window())?;
let scope = SnapshotKey::scope(store.source(), access)?;
let accepted = self.lists.limits.get(&scope).await.unwrap_or(u32::MAX);
let mut page_size = page.limit.min(500).min(accepted);
@ -119,18 +113,23 @@ impl TraceReader {
order,
last,
&query_scope,
(filter.start_ms, filter.end_ms),
filter.window(),
))
.encode()
});
let data = {
let mut summaries = Vec::with_capacity(rows.len());
for batch in run_batches(&rows) {
summaries.extend(list_summaries(self, store, access, batch).await?);
summaries
.extend(list_summaries(self, store, access, batch, filter.as_of_ms).await?);
}
summaries
};
Ok(TracePage { data, next_cursor })
Ok(TracePage {
data,
next_cursor,
window: filter.window(),
})
}
pub async fn histogram<S: TraceStore>(
@ -157,7 +156,7 @@ impl TraceReader {
.run_counts(access, &query)
.await
.map_err(map_store_error)?;
Ok(histogram(&rows, filter.start_ms, filter.end_ms, buckets))
Ok(histogram(&rows, filter.window(), buckets))
}
pub async fn values<S: TraceStore>(
@ -186,6 +185,7 @@ impl TraceReader {
.await
.map_err(map_store_error)?;
Ok(RunValues {
window: filter.window(),
values: rows.into_iter().map(|row| row.value).collect(),
})
}
@ -245,6 +245,76 @@ impl TraceReader {
.map_err(map_store_error)
}
pub async fn get_trace_metadata<S: TraceStore>(
&self,
store: &S,
access: &QueryScope,
id: &str,
) -> Result<Option<litellm_traces::api::TraceMetadata>, ReadError<S::Error>> {
let Some(row) = run_by_id(store, access, id).await? else {
return Ok(None);
};
Ok(self
.current(store, access, &row.trace_id, id)
.await?
.map(|snapshot| {
let trace = snapshot.trace();
litellm_traces::api::TraceMetadata {
summary: trace.summary.clone(),
agents: trace.agents.clone(),
}
}))
}
pub async fn get_trace_spans<S: TraceStore>(
&self,
store: &S,
access: &QueryScope,
id: &str,
cursor: Option<&str>,
page_size: u32,
) -> Result<Option<litellm_traces::api::TraceSpansPage>, ReadError<S::Error>> {
let Some(row) = run_by_id(store, access, id).await? else {
return Ok(None);
};
Ok(self
.get_trace_page(store, access, &row.trace_id, id, cursor, page_size)
.await?
.map(|trace| litellm_traces::api::TraceSpansPage {
data: trace.spans,
next_cursor: trace.next_cursor,
}))
}
pub async fn get_span_by_id<S: TraceStore>(
&self,
store: &S,
access: &QueryScope,
id: &str,
span_id: &str,
) -> Result<Option<SpanDetail>, ReadError<S::Error>> {
let Some(row) = run_by_id(store, access, id).await? else {
return Ok(None);
};
self.get_span(store, access, &row.trace_id, span_id, id)
.await
}
pub async fn get_span_error_by_id<S: TraceStore>(
&self,
store: &S,
access: &QueryScope,
id: &str,
span_id: &str,
cursor: Option<&str>,
) -> Result<Option<SpanErrorPage>, ReadError<S::Error>> {
let Some(row) = run_by_id(store, access, id).await? else {
return Ok(None);
};
self.get_span_error(store, access, &row.trace_id, span_id, id, cursor)
.await
}
pub async fn get_trace<S: TraceStore>(
&self,
store: &S,
@ -283,13 +353,17 @@ impl TraceReader {
let position = SpanPosition {
trace_ref,
snapshot_ms: snapshot.snapshot_ms(),
page_size,
offset: 0,
version: snapshot.version().to_owned(),
};
return page(&snapshot, &position, page_size, self.response_bytes).map(Some);
};
let position = span_position(cursor)?;
if position.trace_ref != trace_ref || position.snapshot_ms == 0 {
if position.trace_ref != trace_ref
|| position.snapshot_ms == 0
|| position.page_size != page_size
{
return Err(ReadError::InvalidCursor("span"));
}
let Some(snapshot) = settle(
@ -325,7 +399,7 @@ impl TraceReader {
)
}
async fn pinned<S: TraceStore>(
pub(super) async fn pinned<S: TraceStore>(
&self,
store: &S,
access: &QueryScope,
@ -344,7 +418,7 @@ impl TraceReader {
let rows = spans(store, access, selection, snapshot_ms)
.await
.map_err(|error| Miss::Read(map_store_error(error)))?;
let spend_rows = spend(store, access, &rows).await;
let spend_rows = spend(store, access, &rows, snapshot_ms).await;
let freshness = Freshness::of(&rows, spend_rows.is_some(), snapshot_ms);
resolve_trace(
trace_id,
@ -392,8 +466,6 @@ impl TraceReader {
output
};
Ok(Some(SpanDetail {
input_ui: to_ui_content(&input.text),
output_ui: to_ui_content(&output),
span_id: span_id.to_owned(),
input: input.text,
output,
@ -552,10 +624,38 @@ pub(super) async fn spans<S: TraceStore>(
async move { store.spans(access, &query).await }
})
.await?;
rows.sort_by_key(|row| row.start_ns);
rows.sort_by(|left, right| {
(left.start_ns, &left.span_id).cmp(&(right.start_ns, &right.span_id))
});
Ok(rows)
}
async fn run_by_id<S: TraceStore>(
store: &S,
access: &QueryScope,
id: &str,
) -> Result<Option<litellm_traces::store::RunRow>, ReadError<S::Error>> {
if id.len() != 64
|| !id
.bytes()
.all(|ch| ch.is_ascii_digit() || (b'A'..=b'F').contains(&ch))
{
return Err(ReadError::InvalidParameters);
}
let query = RunQuery {
selection: RunSelection::TraceRef(id.to_owned()),
order: RunOrder::BY_REFERENCE,
after: None,
limit: 1,
};
Ok(store
.runs(access, &query)
.await
.map_err(map_store_error)?
.into_iter()
.find(|row| row.trace_ref == id))
}
async fn reference<S: TraceStore>(
store: &S,
access: &QueryScope,
@ -595,6 +695,7 @@ fn page<E>(
Cursor::Span(SpanPosition {
trace_ref: position.trace_ref.clone(),
snapshot_ms: position.snapshot_ms,
page_size: position.page_size,
offset: end,
version: snapshot.version().to_owned(),
})

View file

@ -33,6 +33,7 @@ pub(super) async fn spend<S: TraceStore>(
store: &S,
access: &QueryScope,
rows: &[SpanRow],
as_of_ms: u64,
) -> Option<Vec<CallRow>> {
let lookup = SpendLookup::new(rows);
let Some(window) = spend_window(rows) else {
@ -43,6 +44,7 @@ pub(super) async fn spend<S: TraceStore>(
}
let calls = read_all(|after, limit| {
let query = CallQuery {
as_of_ms,
window: window.clone(),
response_ids: lookup.response_ids.clone(),
request_ids: lookup.request_ids.clone(),

View file

@ -342,7 +342,7 @@ fn window(start_ms: i64, end_ms: i64, q: &str) -> RunFilter {
RunFilter {
start_ms,
end_ms,
search: RunSearch::parse(q),
search: RunSearch::parse(q).unwrap(),
..Default::default()
}
}
@ -462,6 +462,33 @@ async fn pages_reuse_one_trace_snapshot_and_concatenate_in_order() {
assert_eq!(store.calls(Operation::TraceSpans), 1);
}
#[rstest]
#[case::smaller(1)]
#[case::larger(3)]
#[tokio::test]
async fn span_cursors_reject_changed_page_sizes(#[case] page_size: u32) {
let store = FakeStore::with_spans("ref", (0..5).map(span).collect());
let reader = TraceReader::new(usize::MAX);
let access = access();
let first = reader
.get_trace_page(&store, &access, "trace", "ref", None, 2)
.await
.unwrap()
.unwrap();
let result = reader
.get_trace_page(
&store,
&access,
"trace",
"ref",
first.next_cursor.as_deref(),
page_size,
)
.await;
assert!(matches!(result, Err(ReadError::InvalidCursor("span"))));
assert_eq!(store.calls(Operation::TraceSpans), 1);
}
#[rstest]
#[tokio::test]
async fn snapshot_versions_are_stable_across_readers_and_detect_changes() {
@ -1068,7 +1095,7 @@ async fn values_narrow_by_the_search_and_the_needle() {
runs: 1,
},
];
let filter = window(0, 10, "-status:error");
let filter = window(0, 10, "-has_error:true");
let values = TraceReader::new(usize::MAX)
.values(&store, &access(), &filter, RunField::Agent, "re_", 20)
.await
@ -1197,6 +1224,8 @@ async fn cached_run_keeps_all_canonical_metrics_current_for_every_sort(#[case] k
duration_ms: changed.duration_ns as f64 / 1_000_000.0,
span_count: changed.span_count,
error_count: changed.error_count,
has_error: changed.error_count > 0,
status: changed.status,
..first.data[0].clone()
};
assert_eq!(expected.spend, Some(1.5));

View file

@ -41,3 +41,8 @@ wiremock.workspace = true
name = "export-traces-clickhouse-schema"
path = "src/bin/export_schema.rs"
required-features = ["schema"]
[[bin]]
name = "export-traces-openapi"
path = "src/bin/export_openapi.rs"
required-features = ["schema"]

View file

@ -0,0 +1,42 @@
CREATE VIEW IF NOT EXISTS {database}.spans
SQL SECURITY INVOKER
AS SELECT
TeamId AS team_id,
ApiKeyHash AS api_key_hash,
UserId AS user_id,
TraceId AS trace_id,
hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) AS trace_ref,
SpanId AS span_id,
ParentSpanId AS parent_span_id,
SpanName AS name,
SpanKind AS span_kind,
ServiceName AS service,
Timestamp AS start_time,
Duration AS duration_ns,
Duration / 1000000.0 AS duration_ms,
multiIf(StatusCode = 'STATUS_CODE_ERROR', 'error', StatusCode = 'STATUS_CODE_OK', 'ok', 'unset') AS status,
StatusCode AS status_code,
StatusMessage AS status_message,
ObservationType AS observation_type,
AgentName AS agent_name,
Framework AS framework,
Model AS model,
InputTokens AS input_tokens,
OutputTokens AS output_tokens,
ResourceAttributes AS resource_attributes,
SpanAttributes AS span_attributes,
AgentMetadata AS agent_metadata,
Input AS input,
Output AS output,
InputPreview AS input_preview,
WrapperCandidate AS wrapper_candidate,
LiteLLMRequestId AS request_id,
CallKeys AS call_keys,
CallEvidence AS call_evidence,
ToolCallId AS tool_call_id,
EngineReceivedMs AS ingested_at_ms
FROM (
SELECT * FROM {database}.otel_traces
ORDER BY Timestamp, EngineReceivedMs, StatusMessage, Duration, StatusCode
LIMIT 1 BY TeamId, ApiKeyHash, TraceId, SpanId
)

View file

@ -0,0 +1,35 @@
CREATE VIEW IF NOT EXISTS {database}.calls
SQL SECURITY INVOKER
AS SELECT
team_id,
api_key AS api_key_hash,
user AS user_id,
request_id,
response_id,
litellm_call_id,
trace_id,
span_id,
call_type,
model,
model_group,
custom_llm_provider AS provider,
spend,
prompt_tokens AS input_tokens,
completion_tokens AS output_tokens,
total_tokens,
cache_read_tokens,
cache_write_tokens,
start_time,
end_time,
completion_start_time,
dateDiff('millisecond', start_time, end_time) AS duration_ms,
status,
error_str AS error,
cache_hit,
session_id,
request_tags,
metadata,
messages,
response,
EngineReceivedMs AS ingested_at_ms
FROM {database}.spend_logs FINAL

View file

@ -0,0 +1,38 @@
CREATE VIEW IF NOT EXISTS {database}.traces
SQL SECURITY INVOKER
AS SELECT
team_id,
api_key_hash,
trace_id,
any(trace_ref) AS id,
if(uniqExact(user_id) = 1, any(user_id), '') AS user_id,
if(countIf(parent_span_id = '') > 0,
argMinIf(s.name, tuple(s.start_time, span_id), parent_span_id = ''),
argMin(s.name, tuple(s.start_time, span_id))) AS name,
argMin(s.service, tuple(s.start_time, span_id)) AS service,
coalesce(nullIf(if(countIf(parent_span_id = '') > 0,
argMinIf(s.input_preview, tuple(s.start_time, span_id), parent_span_id = ''),
argMin(s.input_preview, tuple(s.start_time, span_id))), ''),
argMinIf(s.input_preview, tuple(s.start_time, span_id),
s.input_preview != '' AND observation_type IN ('agent', 'llm'))) AS input_preview,
if(countIf(parent_span_id = '') > 0,
argMinIf(status, tuple(s.start_time, span_id), parent_span_id = ''),
argMin(status, tuple(s.start_time, span_id))) AS root_status,
min(s.start_time) AS start_time,
max(s.start_time + toIntervalNanosecond(s.duration_ns)) AS end_time,
toUInt64(greatest(toUnixTimestamp64Nano(end_time) - toUnixTimestamp64Nano(start_time), 0)) AS duration_ns,
duration_ns / 1000000.0 AS duration_ms,
count() AS span_count,
countIf(status = 'error') AS error_count,
CAST(error_count > 0, 'Bool') AS has_error,
countIf(observation_type = 'agent') AS agent_span_count,
arraySort(groupUniqArrayIf(if(agent_name = '', s.name, agent_name), agent_name != '' OR observation_type = 'agent')) AS agent_names,
length(agent_names) AS agent_label_count,
arraySort(groupUniqArrayIf(framework, framework != '')) AS frameworks,
countIf(observation_type = 'llm') AS llm_span_count,
countIf(observation_type = 'tool') AS tool_span_count,
sum(s.input_tokens) AS span_input_tokens,
sum(s.output_tokens) AS span_output_tokens,
arraySort(groupUniqArrayIf(model, model != '')) AS models
FROM {database}.spans AS s
GROUP BY team_id, api_key_hash, trace_id

View file

@ -9,7 +9,8 @@ FROM (
extract(tryBase64Decode(substring(response_id, 6)), 'response_id:([^;]+)'),
'') AS upstream_response_id
FROM owned_calls
WHERE start_time >= fromUnixTimestamp64Milli({start_ms:Int64})
WHERE EngineReceivedMs <= {as_of_ms:UInt64}
AND start_time >= fromUnixTimestamp64Milli({start_ms:Int64})
AND start_time < fromUnixTimestamp64Milli({end_ms:Int64})
)
WHERE response_id IN {response_ids:Array(String)}

View file

@ -0,0 +1,24 @@
candidate_runs AS (
SELECT TeamId, ApiKeyHash, TraceId
FROM owned_runs
WHERE ({trace_id:String} != '' AND TraceId = {trace_id:String})
OR ({trace_ref:String} != '' AND hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) = {trace_ref:String})
GROUP BY TeamId, ApiKeyHash, TraceId
UNION ALL
SELECT TeamId, ApiKeyHash, TraceId
FROM owned_spans
WHERE {trace_id:String} = '' AND {trace_ref:String} = ''
AND EngineReceivedMs <= {as_of_ms:UInt64}
AND Timestamp >= fromUnixTimestamp64Milli({start_ms:Int64})
AND Timestamp < fromUnixTimestamp64Milli({end_ms:Int64})
GROUP BY TeamId, ApiKeyHash, TraceId
),
canonical_spans AS (
SELECT * FROM owned_spans
WHERE EngineReceivedMs <= {as_of_ms:UInt64}
AND Timestamp >= (SELECT min(StartTs) FROM owned_runs WHERE (TeamId, ApiKeyHash, TraceId) IN (SELECT TeamId, ApiKeyHash, TraceId FROM candidate_runs))
AND (TeamId, ApiKeyHash, TraceId) IN (SELECT TeamId, ApiKeyHash, TraceId FROM candidate_runs)
AND (TeamId, ApiKeyHash, TraceId) IN (SELECT TeamId, ApiKeyHash, TraceId FROM owned_runs GROUP BY TeamId, ApiKeyHash, TraceId)
ORDER BY Timestamp, EngineReceivedMs, StatusMessage, Duration, StatusCode
LIMIT 1 BY TeamId, ApiKeyHash, TraceId, SpanId
)

View file

@ -1,14 +1,15 @@
SELECT t.TraceId, t.SpanId, s.request_id, s.spend, s.metadata
FROM otel_traces AS t
SELECT t.trace_id AS trace_id, t.span_id AS span_id,
s.request_id AS request_id, s.spend AS spend, s.metadata AS metadata
FROM spans AS t
INNER JOIN (
SELECT *
FROM spend_logs FINAL
FROM calls
WHERE start_time >= now() - INTERVAL 1 DAY
) AS s
ON t.LiteLLMRequestId = s.response_id
AND t.TeamId = s.team_id
AND ((t.UserId != '' AND t.UserId = s.user)
OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key))
WHERE t.Timestamp >= now() - INTERVAL 1 DAY
AND t.LiteLLMRequestId != ''
ON t.request_id = s.response_id
AND t.team_id = s.team_id
AND ((t.user_id != '' AND t.user_id = s.user_id)
OR (t.api_key_hash != '' AND t.api_key_hash = s.api_key_hash))
WHERE t.start_time >= now() - INTERVAL 1 DAY
AND t.request_id != ''
LIMIT 100

View file

@ -1,6 +1,6 @@
SELECT
request_id, response_id, model, spend, JSONExtractString(metadata, 'project') AS project
FROM spend_logs FINAL
FROM calls
WHERE start_time >= now() - INTERVAL 1 DAY
AND JSONHas(metadata, 'project')
AND JSONExtractString(metadata, 'project') = 'example'

View file

@ -1,6 +1,6 @@
SELECT
DISTINCT arrayJoin(JSONExtractKeys(metadata)) AS key
FROM spend_logs FINAL
FROM calls
WHERE start_time >= now() - INTERVAL 30 DAY
ORDER BY key
LIMIT 200

View file

@ -1,12 +1,6 @@
SELECT TeamId AS team, ApiKeyHash AS api_key, TraceId AS trace_id,
SpanId AS span_id, StatusMessage AS message
FROM (
SELECT *
FROM otel_traces
WHERE Timestamp >= now() - INTERVAL 1 DAY
ORDER BY Timestamp, EngineReceivedMs, StatusMessage, Duration, StatusCode
LIMIT 1 BY TeamId, ApiKeyHash, TraceId, SpanId
)
WHERE StatusCode = 'STATUS_CODE_ERROR'
ORDER BY Timestamp DESC, team, api_key, trace_id, span_id
SELECT team_id AS team, api_key_hash AS api_key, trace_id,
span_id, status_message AS message
FROM spans
WHERE start_time >= now() - INTERVAL 1 DAY AND status = 'error'
ORDER BY start_time DESC, team, api_key, trace_id, span_id
LIMIT 100

View file

@ -1,6 +1,6 @@
SELECT team_id AS team, api_key, request_id, spend,
SELECT team_id AS team, api_key_hash AS api_key, request_id, spend,
JSONExtractString(metadata, 'labels', 'priority') AS priority
FROM spend_logs FINAL
FROM calls
WHERE start_time >= now() - INTERVAL 1 DAY
AND JSONHas(metadata, 'labels', 'priority')
AND JSONExtractString(metadata, 'labels', 'priority') = 'high'

View file

@ -7,9 +7,9 @@ FROM (
team_id, model, count() AS requests,
countIf(isNull(spend) OR NOT isFinite(spend)) AS unknown_cost_requests,
sum(spend) AS recorded_spend,
sum(prompt_tokens) AS input_tokens,
sum(completion_tokens) AS output_tokens
FROM spend_logs FINAL
sum(c.input_tokens) AS input_tokens,
sum(c.output_tokens) AS output_tokens
FROM calls AS c
WHERE start_time >= now() - INTERVAL 1 DAY
GROUP BY team_id, model
)

View file

@ -2,7 +2,7 @@ SELECT
request_id,
JSONType(metadata, 'labels', 'priority') AS type,
JSONExtractRaw(metadata, 'labels', 'priority') AS value
FROM spend_logs FINAL
FROM calls
WHERE start_time >= now() - INTERVAL 1 DAY
AND JSONHas(metadata, 'labels', 'priority')
LIMIT 100

View file

@ -1,8 +1,7 @@
SELECT
TraceId, SpanId, Model, InputTokens, OutputTokens,
Duration / 1000000 AS duration_ms
FROM otel_traces
WHERE Timestamp >= now() - INTERVAL 1 DAY
AND ObservationType = 'llm'
ORDER BY Timestamp DESC
trace_id, span_id, model, input_tokens, output_tokens, duration_ms
FROM spans
WHERE start_time >= now() - INTERVAL 1 DAY
AND observation_type = 'llm'
ORDER BY start_time DESC, trace_ref, span_id
LIMIT 100

View file

@ -1,7 +1,7 @@
SELECT
request_id, response_id, trace_id, span_id, model, spend,
prompt_tokens, completion_tokens, status
FROM spend_logs FINAL
input_tokens, output_tokens, status
FROM calls
WHERE start_time >= now() - INTERVAL 1 DAY
ORDER BY start_time DESC, request_id
LIMIT 100

View file

@ -1,8 +1,8 @@
SELECT
team_id, api_key, trace_id, count() AS requests,
team_id, api_key_hash AS api_key, trace_id, count() AS requests,
countIf(isNull(spend) OR NOT isFinite(spend)) AS unknown_cost_requests,
if(unknown_cost_requests = 0, sum(spend), NULL) AS recorded_spend
FROM spend_logs FINAL
FROM calls
WHERE start_time >= now() - INTERVAL 1 DAY
AND trace_id != ''
GROUP BY team_id, api_key, trace_id

View file

@ -1,12 +1,7 @@
SELECT TeamId AS team, ApiKeyHash AS api_key, TraceId AS trace_id,
ifNull(any(RootName), '') AS name,
toUInt32(sum(SpanCount)) AS spans,
toUInt32(sum(LlmCount)) AS llm_calls,
toUInt32(sum(ErrorCount)) AS errors,
toUInt32(sum(InputTokens)) AS input_tokens,
toUInt32(sum(OutputTokens)) AS output_tokens
FROM agent_traces_by_key
GROUP BY TeamId, ApiKeyHash, TraceId
HAVING min(StartTs) >= now() - INTERVAL 1 DAY
SELECT id, team_id AS team, api_key_hash AS api_key, trace_id, name,
span_count AS spans, llm_span_count, error_count AS errors,
span_input_tokens, span_output_tokens
FROM traces
WHERE start_time >= now() - INTERVAL 1 DAY
ORDER BY team, api_key, trace_id
LIMIT 100

View file

@ -1,18 +1,18 @@
SELECT
t.TraceId, t.SpanId, t.Model, t.LiteLLMRequestId,
t.InputTokens, t.OutputTokens
FROM otel_traces AS t
t.trace_id, t.span_id, t.model, t.request_id,
t.input_tokens, t.output_tokens
FROM spans AS t
LEFT ANTI JOIN (
SELECT *
FROM spend_logs FINAL
FROM calls
WHERE start_time >= now() - INTERVAL 1 DAY
) AS s
ON t.TeamId = s.team_id
AND ((t.UserId != '' AND t.UserId = s.user)
OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key))
AND t.LiteLLMRequestId != ''
AND (t.LiteLLMRequestId = s.response_id OR t.LiteLLMRequestId = s.request_id)
WHERE t.Timestamp >= now() - INTERVAL 1 DAY
AND t.ObservationType = 'llm'
ORDER BY t.Timestamp DESC, t.SpanId
ON t.team_id = s.team_id
AND ((t.user_id != '' AND t.user_id = s.user_id)
OR (t.api_key_hash != '' AND t.api_key_hash = s.api_key_hash))
AND t.request_id != ''
AND (t.request_id = s.response_id OR t.request_id = s.request_id)
WHERE t.start_time >= now() - INTERVAL 1 DAY
AND t.observation_type = 'llm'
ORDER BY t.start_time DESC, t.span_id
LIMIT 100

View file

@ -1,69 +1,42 @@
SELECT TraceId AS trace_id,
hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) AS trace_ref,
if(length(groupUniqArrayArray(UserIds)) = 1, arrayElement(groupUniqArrayArray(UserIds), 1), '') AS user_id, TeamId AS team_id, ApiKeyHash AS api_key_hash,
ifNull(any(RootName), '') AS name, any(ServiceName) AS service,
ifNull(any(RootInput), '') AS input_preview, ifNull(any(RootStatus), '') AS status,
toUnixTimestamp64Milli(min(StartTs)) AS start_ms,
if(any(metrics.span_count) > 0,
toUInt64(greatest(toUnixTimestamp64Nano(any(metrics.end_ts)) - toUnixTimestamp64Nano(any(metrics.start_ts)), 0)),
toUInt64(greatest(toUnixTimestamp64Nano(max(EndTs)) - toUnixTimestamp64Nano(min(StartTs)), 0))) AS duration_ns,
if(any(metrics.span_count) > 0, any(metrics.span_count), sum(SpanCount)) AS span_count,
sum(AgentCount) AS agent_invocations,
sum(LlmCount) AS llm_calls, sum(ToolCount) AS tool_calls,
if(length(groupUniqArray(UserId)) = 1, arrayElement(groupUniqArray(UserId), 1), '') AS user_id,
TeamId AS team_id, ApiKeyHash AS api_key_hash,
if(countIf(ParentSpanId = '') > 0, argMinIf(SpanName, tuple(Timestamp, SpanId), ParentSpanId = ''), argMin(SpanName, tuple(Timestamp, SpanId))) AS name,
argMin(ServiceName, tuple(Timestamp, SpanId)) AS service,
if(countIf(ParentSpanId = '') > 0, argMinIf(InputPreview, tuple(Timestamp, SpanId), ParentSpanId = ''), argMin(InputPreview, tuple(Timestamp, SpanId))) AS root_input,
if(root_input != '', root_input, argMinIf(InputPreview, tuple(Timestamp, SpanId), InputPreview != '' AND ObservationType IN ('agent', 'llm'))) AS input_preview,
if(countIf(ParentSpanId = '') > 0, argMinIf(StatusCode, tuple(Timestamp, SpanId), ParentSpanId = ''), argMin(StatusCode, tuple(Timestamp, SpanId))) AS status,
toUnixTimestamp64Milli(min(Timestamp)) AS start_ms,
toUInt64(greatest(toUnixTimestamp64Nano(max(Timestamp + toIntervalNanosecond(Duration))) - toUnixTimestamp64Nano(min(Timestamp)), 0)) AS duration_ns,
count() AS span_count,
countIf(ObservationType = 'agent') AS agent_invocations,
countIf(ObservationType = 'llm') AS llm_calls,
countIf(ObservationType = 'tool') AS tool_calls,
sum(InputTokens) AS input_tokens, sum(OutputTokens) AS output_tokens,
groupUniqArrayArray(Models) AS models,
if(any(metrics.span_count) > 0, any(metrics.error_count), sum(ErrorCount)) AS error_count,
arraySort(groupUniqArrayArray(AgentNames)) AS search_agents,
if(error_count > 0, 'error', 'ok') AS search_status,
length(groupUniqArrayArray(AgentIdentities)) AS agent_count,
arraySort(groupUniqArrayArray(Frameworks)) AS frameworks
FROM owned_runs
LEFT JOIN (
SELECT TeamId, ApiKeyHash, TraceId,
min(Timestamp) AS start_ts,
max(Timestamp + toIntervalNanosecond(Duration)) AS end_ts,
count() AS span_count,
countIf(StatusCode = 'STATUS_CODE_ERROR') AS error_count
FROM (
SELECT TeamId, ApiKeyHash, TraceId, SpanId, Timestamp, Duration, StatusCode
FROM owned_spans
WHERE (TeamId, ApiKeyHash, TraceId) IN (
SELECT TeamId, ApiKeyHash, TraceId
FROM owned_runs
WHERE {trace_id:String} = '' OR TraceId = {trace_id:String}
GROUP BY TeamId, ApiKeyHash, TraceId
HAVING {trace_id:String} != '' OR (
min(StartTs) >= fromUnixTimestamp64Milli({start_ms:Int64})
AND min(StartTs) < fromUnixTimestamp64Milli({end_ms:Int64}))
)
AND ({trace_id:String} != '' OR Timestamp >= fromUnixTimestamp64Milli({start_ms:Int64}))
ORDER BY Timestamp, EngineReceivedMs, StatusMessage, Duration, StatusCode
LIMIT 1 BY TeamId, ApiKeyHash, TraceId, SpanId
)
GROUP BY TeamId, ApiKeyHash, TraceId
) AS metrics USING (TeamId, ApiKeyHash, TraceId)
LEFT JOIN (
SELECT TeamId, ApiKeyHash, TraceId,
groupUniqArrayArray(arrayFilter(i -> (mapContains(ResourceAttributes, {attribute_keys:Array(String)}[i])
AND ResourceAttributes[{attribute_keys:Array(String)}[i]] ILIKE {attribute_patterns:Array(String)}[i])
OR (mapContains(SpanAttributes, {attribute_keys:Array(String)}[i])
AND SpanAttributes[{attribute_keys:Array(String)}[i]] ILIKE {attribute_patterns:Array(String)}[i]),
arrayEnumerate({attribute_keys:Array(String)}))) AS matched_attributes
FROM owned_spans
WHERE notEmpty({attribute_keys:Array(String)})
AND Timestamp >= fromUnixTimestamp64Milli({start_ms:Int64})
GROUP BY TeamId, ApiKeyHash, TraceId
) AS attributes USING (TeamId, ApiKeyHash, TraceId)
WHERE {trace_id:String} = '' OR TraceId = {trace_id:String}
groupUniqArrayIf(toString(Model), Model != '') AS models,
countIf(StatusCode = 'STATUS_CODE_ERROR') AS error_count,
arraySort(groupUniqArrayIf(if(AgentName = '', SpanName, AgentName), AgentName != '' OR ObservationType = 'agent')) AS search_agents,
multiIf(status = 'STATUS_CODE_OK', 'ok', status = 'STATUS_CODE_ERROR', 'error', 'unset') AS search_root_status,
if(error_count > 0, 'true', 'false') AS search_has_error,
length(groupUniqArrayIf(if(AgentName = '', SpanName, AgentName), ObservationType = 'agent')) AS agent_count,
arraySort(groupUniqArrayIf(toString(Framework), Framework != '')) AS frameworks,
groupUniqArrayArray(arrayFilter(i -> (mapContains(ResourceAttributes, {attribute_keys:Array(String)}[i]) AND ResourceAttributes[{attribute_keys:Array(String)}[i]] ILIKE {attribute_patterns:Array(String)}[i])
OR (mapContains(SpanAttributes, {attribute_keys:Array(String)}[i]) AND SpanAttributes[{attribute_keys:Array(String)}[i]] ILIKE {attribute_patterns:Array(String)}[i]),
arrayEnumerate({attribute_keys:Array(String)}))) AS matched_attributes
FROM canonical_spans
GROUP BY TeamId, ApiKeyHash, TraceId
HAVING {trace_id:String} != ''
OR (min(StartTs) >= fromUnixTimestamp64Milli({start_ms:Int64})
AND min(StartTs) < fromUnixTimestamp64Milli({end_ms:Int64})
HAVING ({trace_id:String} != '' AND TraceId = {trace_id:String})
OR ({trace_ref:String} != '' AND trace_ref = {trace_ref:String})
OR ({trace_id:String} = '' AND {trace_ref:String} = ''
AND min(Timestamp) >= fromUnixTimestamp64Milli({start_ms:Int64})
AND min(Timestamp) < fromUnixTimestamp64Milli({end_ms:Int64})
AND arrayAll(t -> trace_id ILIKE t OR input_preview ILIKE t OR name ILIKE t, {text:Array(String)})
AND arrayAll((f, p, m) -> (m = 'exclude') != multiIf(
f = 'name', name ILIKE p,
f = 'agent', arrayExists(a -> a ILIKE p, search_agents),
f = 'status', search_status ILIKE p,
f = 'root_status', search_root_status ILIKE p,
f = 'has_error', search_has_error ILIKE p,
f = 'model', arrayExists(x -> x ILIKE p, models),
f = 'input', input_preview ILIKE p,
f = 'trace_id', trace_id ILIKE p,
@ -71,6 +44,6 @@ HAVING {trace_id:String} != ''
f = 'team', team_id ILIKE p,
false),
{filter_fields:Array(String)}, {filter_patterns:Array(String)}, {filter_modes:Array(String)})
AND arrayAll((i, m) -> (m = 'exclude') != has(any(matched_attributes), i),
AND arrayAll((i, m) -> (m = 'exclude') != has(matched_attributes, i),
arrayEnumerate({attribute_keys:Array(String)}), {attribute_modes:Array(String)})
AND (empty({trace_refs:Array(String)}) OR trace_ref IN {trace_refs:Array(String)}))

View file

@ -5,7 +5,7 @@ FROM (
toUInt32(intDiv((runs.start_ms - {start_ms:Int64} + 1) * {buckets:UInt32} - 1, {end_ms:Int64} - {start_ms:Int64}))) AS bucket,
toUInt8({by_failed:UInt8} = 1 AND runs.error_count > 0) AS failed,
value
FROM owned_spans AS spans
FROM canonical_spans AS spans
INNER JOIN runs ON spans.TeamId = runs.team_id AND spans.ApiKeyHash = runs.api_key_hash
AND spans.TraceId = runs.trace_id
ARRAY JOIN if({attribute_key:String} = '',

View file

@ -1,5 +1,5 @@
SELECT if({buckets:UInt32} = 0, toUInt32(0),
toUInt32(intDiv((start_ms - {start_ms:Int64} + 1) * {buckets:UInt32} - 1, {end_ms:Int64} - {start_ms:Int64}))) AS bucket,
toUInt32(intDiv((toInt128(start_ms) - toInt128({start_ms:Int64}) + 1) * {buckets:UInt32} - 1, toInt128({end_ms:Int64}) - toInt128({start_ms:Int64})))) AS bucket,
toUInt8({by_failed:UInt8} = 1 AND error_count > 0) AS failed,
value,
count() AS runs
@ -9,7 +9,8 @@ ARRAY JOIN multiIf(
{value:String} = 'primary_agent', [ifNull(nullIf(arrayElement(search_agents, 1), ''), service)],
{value:String} = 'name', [name],
{value:String} = 'agent', search_agents,
{value:String} = 'status', [search_status],
{value:String} = 'root_status', [search_root_status],
{value:String} = 'has_error', [search_has_error],
{value:String} = 'model', models,
{value:String} = 'input', [input_preview],
{value:String} = 'trace_id', [trace_id],

View file

@ -1,5 +1,5 @@
page AS (
SELECT * EXCEPT (search_status),
SELECT * EXCEPT (root_input, search_root_status, search_has_error, matched_attributes),
multiIf({sort_key:String} = 'duration_ms', toInt64(least(duration_ns, toUInt64(9223372036854775807))),
{sort_key:String} = 'span_count', toInt64(least(span_count, toUInt64(9223372036854775807))),
{sort_key:String} = 'error_count', toInt64(least(error_count, toUInt64(9223372036854775807))),

View file

@ -0,0 +1,5 @@
fn main() {
let document =
litellm_traces::api::openapi::document(litellm_traces_clickhouse::wire_schema::schemas());
println!("{}", document.to_pretty_json().unwrap());
}

View file

@ -28,11 +28,11 @@ pub use error::Error;
pub use insert::{InsertRow, InsertTable, encode_rows, insert_rows, insert_shared_rows};
pub use litellm_storage_clickhouse::{Connection, Parameter};
pub use litellm_traces::QueryScope;
pub use query::{QueryHelp, execute_read, query_help, query_sql};
pub use query::{QueryHelp, execute_read, query_help, query_sql, query_sql_with_params};
pub use query_access::QueryReaders;
pub use reads::ClickHouseTraces;
pub use schema::{
NORMALIZED_FIELD_DEFINITIONS, NormalizedFieldDefinition, ensure_schema, schema_statements,
};
pub use span_row::span_rows;
pub use table::TraceTable;
pub use table::{QueryTable, TraceTable};

View file

@ -5,6 +5,7 @@ use futures_util::{
stream::{self, TryStreamExt},
};
use litellm_http::Client;
use litellm_traces::api::{SqlParameter, TraceSQLColumn};
use litellm_traces::query::guide::{Example, QueryGuide, Section};
use serde::{Deserialize, Serialize, Serializer};
use serde_json::Value;
@ -14,7 +15,7 @@ use super::{
Connection, Error, NORMALIZED_FIELD_DEFINITIONS, NormalizedFieldDefinition, Parameter,
query_access::READER_LIMITS,
};
use crate::TraceTable;
use crate::QueryTable;
mod guide;
pub mod named;
@ -23,7 +24,7 @@ mod number;
const SAMPLE_ROWS: usize = 200;
const MAX_FIELDS: usize = 200;
const MAX_DEPTH: usize = 16;
const METADATA_SQL: &str = "SELECT metadata FROM spend_logs FINAL \
const METADATA_SQL: &str = "SELECT metadata FROM calls \
WHERE start_time >= now() - INTERVAL 7 DAY AND length(metadata) <= 8192 \
LIMIT 201";
const METADATA_SCOPE: &str = "Up to 200 unordered rows from the last 7 days, excluding metadata larger than 8192 bytes; up to 200 paths and 16 levels. Missing paths may exist outside this sample. Array indexes are 1-based and describe sampled positions, not a fixed schema";
@ -96,22 +97,12 @@ struct MetadataField {
expression: String,
}
#[macro_rules_attribute::apply(wire_type)]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryColumn"))]
struct ColumnSchema {
name: String,
#[serde(rename = "type")]
kind: String,
#[serde(flatten)]
details: BTreeMap<String, Value>,
}
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryTable"))]
struct TableSchema {
name: TraceTable,
columns: Vec<ColumnSchema>,
name: QueryTable,
columns: Vec<TraceSQLColumn>,
}
trait Unobserved {
@ -197,7 +188,7 @@ impl Unobserved for MetadataSample {
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryMetadata"))]
struct MetadataCatalog {
table: TraceTable,
table: QueryTable,
column: &'static str,
#[serde(flatten)]
discovery: Discovery<MetadataSample>,
@ -234,7 +225,7 @@ impl Unobserved for AttributeSample {
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryAttributes"))]
struct AttributeCatalog {
table: TraceTable,
table: QueryTable,
column: &'static str,
#[serde(flatten)]
discovery: Discovery<AttributeSample>,
@ -246,7 +237,7 @@ struct AttributeCatalog {
#[cfg_attr(feature = "schema", schemars(deny_unknown_fields))]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryNormalizedField"))]
struct NormalizedField {
table: TraceTable,
table: QueryTable,
name: &'static str,
column: &'static str,
#[serde(rename = "type")]
@ -257,7 +248,7 @@ struct NormalizedField {
impl From<&NormalizedFieldDefinition> for NormalizedField {
fn from(field: &NormalizedFieldDefinition) -> Self {
Self {
table: TraceTable::OtelTraces,
table: QueryTable::OtelTraces,
name: field.name,
column: field.clickhouse_column,
kind: field.clickhouse_type,
@ -277,10 +268,10 @@ struct Relationship {
}
const RELATIONSHIPS: [Relationship; 1] = [Relationship {
left: "otel_traces.LiteLLMRequestId",
right: "spend_logs.response_id",
additional_predicates: "otel_traces.TeamId = spend_logs.team_id AND ((otel_traces.UserId != '' AND otel_traces.UserId = spend_logs.user) OR (otel_traces.ApiKeyHash != '' AND otel_traces.ApiKeyHash = spend_logs.api_key))",
meaning: "LiteLLMRequestId contains the first normalized request or provider response ID. This relationship matches response IDs only; CallKeys retains all typed identifiers. Cached requests can share response_id; joins may return multiple spend rows",
left: "spans.request_id",
right: "calls.response_id",
additional_predicates: "spans.team_id = calls.team_id AND ((spans.user_id != '' AND spans.user_id = calls.user_id) OR (spans.api_key_hash != '' AND spans.api_key_hash = calls.api_key_hash))",
meaning: "request_id contains the first normalized request or provider response ID. This relationship matches response IDs only; call_keys retains all typed identifiers. Cached requests can share response_id; joins may return multiple calls",
}];
#[macro_rules_attribute::apply(response_type)]
@ -321,6 +312,30 @@ pub async fn query_sql(
execute_read(client, connection, sql, &BTreeMap::new()).await
}
pub async fn query_sql_with_params(
client: &Client,
connection: &Connection,
sql: &str,
params: &BTreeMap<String, SqlParameter>,
) -> Result<String, Error> {
let parameters = params
.iter()
.map(|(name, value)| {
let parameter = match value {
SqlParameter::String(value) => Parameter::Text(value.clone()),
SqlParameter::Integer(value) => Parameter::Integer(*value),
SqlParameter::Unsigned(value) => Parameter::Unsigned(*value),
SqlParameter::Number(value) => Parameter::Float(*value),
SqlParameter::Boolean(value) => Parameter::Boolean(*value),
SqlParameter::Null => Parameter::Null,
SqlParameter::Strings(value) => Parameter::Strings(value.clone()),
};
(name.clone(), parameter)
})
.collect();
execute_read(client, connection, sql, &parameters).await
}
async fn rows<T: serde::de::DeserializeOwned>(
client: &Client,
connection: &Connection,
@ -415,11 +430,11 @@ fn metadata_sample(sample: &[MetadataRow]) -> MetadataSample {
}
pub async fn query_help(client: &Client, connection: &Connection) -> Result<QueryHelp, Error> {
let tables = stream::iter(TraceTable::iter())
let tables = stream::iter(QueryTable::iter())
.then(|table| async move {
Ok::<_, Error>(TableSchema {
name: table,
columns: rows::<ColumnSchema>(
columns: rows::<TraceSQLColumn>(
client,
connection,
&format!("DESCRIBE TABLE {table}"),
@ -430,7 +445,7 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result<Quer
.try_collect::<Vec<_>>()
.await?;
let metadata = MetadataCatalog {
table: TraceTable::SpendLogs,
table: QueryTable::Calls,
column: "metadata",
discovery: match rows::<MetadataRow>(client, connection, METADATA_SQL).await {
Ok(sample) => Discovery::Observed(metadata_sample(&sample)),
@ -439,11 +454,11 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result<Quer
sample_sql: METADATA_SQL,
scope: METADATA_SCOPE,
};
let attributes = stream::iter(["SpanAttributes", "ResourceAttributes"])
let attributes = stream::iter(["span_attributes", "resource_attributes"])
.then(|column| async move {
let sql = format!(
"SELECT DISTINCT arrayJoin(mapKeys({column})) AS key FROM \
(SELECT {column} FROM otel_traces WHERE Timestamp >= now() - INTERVAL 7 DAY \
(SELECT {column} FROM spans WHERE start_time >= now() - INTERVAL 7 DAY \
LIMIT 200) ORDER BY key LIMIT 201"
);
let discovery = match rows::<AttributeRow>(client, connection, &sql).await {
@ -462,7 +477,7 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result<Quer
Err(error) => Discovery::Unavailable(error.to_string()),
};
AttributeCatalog {
table: TraceTable::OtelTraces,
table: QueryTable::Spans,
column,
discovery,
discovery_sql: sql,
@ -484,6 +499,7 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result<Quer
"Normalized span fields",
"Observed LLM call metadata",
"Observed span and resource attributes",
"Native SQL parameters",
]
.into_iter()
.zip(&bodies)
@ -500,7 +516,7 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result<Quer
.map_err(|_| Error::InvalidResponse)?;
Ok(QueryHelp {
dialect: "ClickHouse SQL",
access: "Request-log visibility enforced by ClickHouse row policies; proxy admins see all rows, users see their own rows and permitted teams",
access: "ClickHouse row policies enforce user and permitted-team visibility. Logical traces summarize visible canonical spans; curated user-only reads require full trace ownership. Team and admin start time, duration, span count and error count agree with curated reads",
response: "ClickHouse JSON envelope: meta, data, rows, statistics; 64-bit integers may be strings",
examples,
gotchas,
@ -534,7 +550,7 @@ mod tests {
Discovery::Observed(MetadataSample::unobserved())
};
let catalog = MetadataCatalog {
table: TraceTable::SpendLogs,
table: QueryTable::Calls,
column: "metadata",
discovery,
sample_sql: METADATA_SQL,

View file

@ -10,6 +10,7 @@ use crate::{Error, NormalizedFieldDefinition, query_access::ReaderLimits};
"normalized_fields",
"metadata",
"attributes",
"parameters",
"recent_spans_name",
"recent_spans_sql",
"custom_metadata_name",
@ -57,12 +58,13 @@ pub(super) struct QueryGuide<'a> {
}
impl QueryGuide<'_> {
pub fn sections(&self) -> Result<[String; 4], Error> {
pub fn sections(&self) -> Result<[String; 5], Error> {
Ok([
render(&self.as_live_schema())?,
render(&self.as_normalized_fields())?,
render(&self.as_metadata())?,
render(&self.as_attributes())?,
render(&self.as_parameters())?,
])
}

View file

@ -79,9 +79,11 @@ impl From<&RunSearch> for SearchColumns {
}
}
#[derive(Debug, Default, Serialize)]
#[derive(Debug, Serialize)]
struct RunsFilter {
trace_id: String,
trace_ref: String,
as_of_ms: u64,
start_ms: i64,
end_ms: i64,
#[serde(flatten)]
@ -89,10 +91,26 @@ struct RunsFilter {
trace_refs: Vec<String>,
}
impl Default for RunsFilter {
fn default() -> Self {
Self {
trace_id: String::new(),
trace_ref: String::new(),
as_of_ms: u64::MAX,
start_ms: 0,
end_ms: 0,
search: SearchColumns::default(),
trace_refs: Vec::new(),
}
}
}
impl From<&RunFilter> for RunsFilter {
fn from(filter: &RunFilter) -> Self {
Self {
trace_id: String::new(),
trace_ref: String::new(),
as_of_ms: filter.as_of_ms,
start_ms: filter.start_ms,
end_ms: filter.end_ms,
search: (&filter.search).into(),
@ -105,6 +123,10 @@ impl From<&RunSelection> for RunsFilter {
fn from(selection: &RunSelection) -> Self {
match selection {
RunSelection::Matching(filter) => filter.into(),
RunSelection::TraceRef(trace_ref) => Self {
trace_ref: trace_ref.clone(),
..Self::default()
},
RunSelection::TraceId(trace_id) => Self {
trace_id: trace_id.clone(),
..Self::default()
@ -189,6 +211,8 @@ pub(crate) struct RunRowWire(#[serde(with = "RunRowEncoding")] pub RunRow);
macro_rules! over_matching_runs {
($($tail:expr),+ $(,)?) => {
owned!(
",\n",
include_str!("../../query/canonical_spans.sql"),
",\nruns AS (\n",
include_str!("../../query/matching_runs.sql"),
")",
@ -521,6 +545,7 @@ impl Query for SpanTexts {
#[derive(Debug, Serialize)]
pub(crate) struct CallsParams {
as_of_ms: u64,
#[serde(flatten)]
access: AccessParams,
start_ms: i64,
@ -540,6 +565,7 @@ impl CallsParams {
let after = query.after.clone().unwrap_or_default();
Self {
access: access.into(),
as_of_ms: query.as_of_ms,
start_ms: query.window.start,
end_ms: query.window.end,
response_ids: query.response_ids.clone(),

View file

@ -9,7 +9,7 @@ use sha2::{Digest, Sha256};
use strum::IntoEnumIterator;
use tokio::sync::{OwnedSemaphorePermit, Semaphore};
use super::{Connection, Error, TraceTable};
use super::{Connection, Error, QueryTable, TraceTable};
const MIB: u64 = 1024 * 1024;
@ -141,7 +141,7 @@ impl QueryReaders {
)
.await?;
}
for table in TraceTable::iter() {
for table in QueryTable::iter() {
self.execute(
client,
format!("GRANT SELECT ON `{database}`.{table} TO {user}"),

View file

@ -10,3 +10,17 @@ pub enum TraceTable {
AgentTracesByKey,
SpendLogs,
}
#[macro_rules_attribute::apply(response_type)]
#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryTableName"))]
#[derive(Clone, Copy, Debug, strum::Display, strum::EnumIter)]
#[serde(rename_all = "snake_case")]
#[strum(serialize_all = "snake_case")]
pub enum QueryTable {
Traces,
Spans,
Calls,
OtelTraces,
AgentTracesByKey,
SpendLogs,
}

View file

@ -1,4 +1,7 @@
{% block live_schema -%}
Use traces for canonical trace metrics, spans for one earliest copy of each span, and calls for replacement-deduplicated spend records
traces.id is the canonical trace ID used by the curated API; join spans with traces.id = spans.trace_ref
otel_traces, agent_traces_by_key and spend_logs are physical diagnostic tables. Repeated exports can inflate raw span and rollup counts
{% for table in tables -%}
{{ table.name }}
{% for column in table.columns -%}
@ -7,6 +10,12 @@
{% endfor -%}
{%- endblock %}
{% block parameters -%}
Pass values in the optional params object using ClickHouse typed placeholders. The engine binds values without rewriting SQL
{"sql":"SELECT id, name, duration_ms FROM traces WHERE start_time >= fromUnixTimestamp64Milli({start_ms:Int64}) AND has({teams:Array(String)}, team_id) ORDER BY start_time DESC, id LIMIT {limit:UInt32}","params":{"start_ms":0,"teams":["example-team"],"limit":50}}
Values support strings, integers, numbers, booleans, null and string arrays. Match each placeholder type to its value, for example {enabled:Bool} or {optional:Nullable(String)}
{%- endblock %}
{% block normalized_fields -%}
{% for field in normalized_fields -%}
{{ field.name }}: otel_traces.{{ field.clickhouse_column }} ({{ field.clickhouse_type }})
@ -152,8 +161,9 @@ Filter calls by nested metadata
{%- endblock %}
{% block time_window -%}
Always bound Timestamp or start_time and use LIMIT; add TeamId/ApiKeyHash or team_id/api_key filters when investigating one tenant
Always bound start_time and use LIMIT; add team_id/api_key_hash filters when investigating one tenant
Raw SQL returns one bounded response without a cursor. Callers own ORDER BY, LIMIT and keyset predicates for pagination
SQL views are live relations; separate queries do not freeze membership or span versions
{%- endblock %}
{% block reader_limits -%}
@ -161,7 +171,7 @@ The reader enforces {{ limits.result_rows }} result rows, {{ limits.result_mib()
{%- endblock %}
{% block reader_profile -%}
LiteLLM provisions SELECT-only readers from the configured ClickHouse connection and enforces request-log visibility through row policies. Callers see their own user rows and permitted teams. Provisioning requires CREATE USER, ALTER USER, CREATE ROW POLICY, and GRANT SELECT permissions
LiteLLM provisions SELECT-only readers and INVOKER views using ClickHouse row policies. Callers see their own user rows and permitted teams. SQL traces summarize visible spans, while curated user-only reads require full trace ownership. Provisioning requires CREATE USER, ALTER USER, CREATE ROW POLICY, and GRANT SELECT permissions
{%- endblock %}
{% block output_format -%}
@ -173,7 +183,7 @@ metadata is a JSON-encoded String; use JSONHas before typed extraction to distin
{%- endblock %}
{% block map_values -%}
SpanAttributes and ResourceAttributes are Map(String, String); missing map keys return an empty string, so use mapContains for existence checks
spans.span_attributes and spans.resource_attributes are Map(String, String); missing map keys return an empty string, so use mapContains for existence checks
{%- endblock %}
{% block literal_keys -%}
@ -181,7 +191,7 @@ Use the discovered path components as separate JSONExtract arguments; a dot insi
{%- endblock %}
{% block time_units -%}
Duration is nanoseconds; Timestamp has nanosecond precision, spend start_time has millisecond precision
spans.duration_ns and traces.duration_ns are nanoseconds; duration_ms preserves fractional milliseconds. Span and trace start_time has nanosecond precision, call start_time has millisecond precision
{%- endblock %}
{% block missing_spend -%}
@ -193,13 +203,14 @@ Recorded spend by trace totals only requests whose spend_logs.trace_id is popula
{%- endblock %}
{% block spend_totals -%}
Use spend_logs FINAL to collapse replacement rows before totals. Shared response IDs and multiple spans can multiply costs in joins; require one spend match per response ID and ownership before aggregating. Missing IDs or costs leave totals unknown
calls collapses spend replacement rows before totals. Shared response IDs and multiple spans can multiply costs in joins; require one call match per response ID and ownership before aggregating. Missing IDs or costs leave totals unknown
{%- endblock %}
{% block trace_rollups -%}
agent_traces_by_key uses SimpleAggregateFunction columns; group by TeamId, ApiKeyHash and TraceId, using min(StartTs), max(EndTs), sum(SpanCount) and groupUniqArrayArray(Models). Do not use Merge combinators
Rollup counts and raw span reads can include repeated exports. Curated trace metrics select one copy per TeamId, ApiKeyHash, TraceId and SpanId, ordered by Timestamp, EngineReceivedMs, StatusMessage, Duration and StatusCode, before filtering status or aggregating
AgentNames contains searchable agent labels, including SpanName for unnamed agents. AgentIdentities contains distinct labels from agent spans, and Frameworks contains observed framework names
traces and spans select one copy per team_id, api_key_hash, trace_id and span_id, ordered by source Timestamp, EngineReceivedMs, StatusMessage, Duration and StatusCode, before filtering status or aggregating
traces.root_status is the earliest physical root's status, or the earliest span's status when no root exists. has_error reports any canonical span error
agent_span_count, llm_span_count and tool_span_count count normalized span types; agent_label_count counts distinct agent_names. span_input_tokens and span_output_tokens sum canonical span tokens. Curated trace summaries resolve graph wrappers and active calls, so their enriched invocation and token totals can differ
The diagnostic agent_traces_by_key table uses SimpleAggregateFunction columns and counts repeated exports. Group by TeamId, ApiKeyHash and TraceId; do not use Merge combinators
{%- endblock %}
{% block sampling -%}

View file

@ -165,6 +165,7 @@ async fn schema_supports_span_rollups_and_spend_joins(
.calls(
&team,
&CallQuery {
as_of_ms: u64::MAX,
window: timestamp / 1_000_000 - 1000..timestamp / 1_000_000 + 1000,
response_ids: vec!["response-1".into()],
request_ids: Vec::new(),
@ -819,7 +820,8 @@ async fn reused_trace_ids_stay_separate_runs_through_filters_and_span_text(
end_ms: window.end,
search: litellm_traces::search::RunSearch::parse(
"service:review attr.swarm:release",
),
)
.unwrap(),
..Default::default()
}),
after: None,
@ -843,6 +845,26 @@ async fn reused_trace_ids_stay_separate_runs_through_filters_and_span_text(
let own = store.runs(&owned("one", &[]), &by_trace_id).await?;
assert_eq!(own.len(), 1);
let reader = TraceReader::new(usize::MAX);
for run in &filtered {
let metadata = reader
.get_trace_metadata(&store, &QueryScope::All, &run.trace_ref)
.await?
.unwrap();
assert_eq!(metadata.summary.trace_ref, run.trace_ref);
assert_eq!(metadata.summary.input_preview, run.input_preview);
let spans = reader
.get_trace_spans(&store, &QueryScope::All, &run.trace_ref, None, 10)
.await?
.unwrap();
assert_eq!(spans.data.len(), 1);
assert_eq!(spans.data[0].input_preview, run.input_preview);
}
assert!(
reader
.get_trace_metadata(&store, &owned("two", &[]), &own[0].trace_ref)
.await?
.is_none()
);
let read = |trace_ref: String, contains: &'static str| {
let reader = &reader;
let store = &store;
@ -1045,7 +1067,14 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
let writer = Connection::writer(&database.url)?;
ensure_schema(&database.client, &writer, "trace_test", 7).await?;
execute_write(&database, "CREATE USER help_reader").await?;
for table in ["otel_traces", "agent_traces_by_key", "spend_logs"] {
for table in [
"traces",
"spans",
"calls",
"otel_traces",
"agent_traces_by_key",
"spend_logs",
] {
execute_write(
&database,
&format!("GRANT SELECT ON trace_test.{table} TO help_reader"),
@ -1125,7 +1154,14 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
);
let guide = help["guide"].as_str().ok_or("missing rendered guide")?;
assert!(guide.starts_with("Trace SQL query guide"));
for table in ["otel_traces", "agent_traces_by_key", "spend_logs"] {
for table in [
"traces",
"spans",
"calls",
"otel_traces",
"agent_traces_by_key",
"spend_logs",
] {
let described = read_json(&database, &format!("DESCRIBE TABLE {table}")).await?;
let schema = help["tables"]
.as_array()
@ -1168,8 +1204,13 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
!populated
);
let tables = help["tables"].as_array().ok_or("missing tables")?;
assert_eq!(tables.len(), 3);
let columns = tables[0]["columns"].as_array().ok_or("missing columns")?;
assert_eq!(tables.len(), 6);
let columns = tables
.iter()
.find(|table| table["name"] == "otel_traces")
.ok_or("missing raw span table")?["columns"]
.as_array()
.ok_or("missing columns")?;
for field in NORMALIZED_FIELD_DEFINITIONS {
assert!(
columns
@ -1222,8 +1263,8 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
assert!(guide.contains("CustomColumn: String"));
assert!(!guide.contains("private-metadata-value"));
assert!(guide.contains("JSONExtractRaw(metadata, '<custom>&{{key}}', 'nested.key')"));
assert!(guide.contains("SpanAttributes['custom.tag']"));
assert!(guide.contains("ResourceAttributes['custom.resource']"));
assert!(guide.contains("span_attributes['custom.tag']"));
assert!(guide.contains("resource_attributes['custom.resource']"));
assert_eq!(help["attributes"][0]["fields"][0]["key"], "custom.tag");
assert_eq!(help["attributes"][1]["fields"][0]["key"], "custom.resource");
for field in fields {
@ -1285,7 +1326,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples(
"{sql}"
);
if populated && example["name"] == "Traces correlated with LLM call metadata" {
assert_eq!(values["data"][0]["TraceId"], "trace-1");
assert_eq!(values["data"][0]["trace_id"], "trace-1");
assert_eq!(values["data"][0]["spend"], 0.25);
}
}
@ -1310,7 +1351,14 @@ async fn query_help_preserves_schema_and_guide_when_discovery_hits_reader_limits
"CREATE USER help_reader SETTINGS max_rows_to_read = 1",
)
.await?;
for table in ["otel_traces", "agent_traces_by_key", "spend_logs"] {
for table in [
"traces",
"spans",
"calls",
"otel_traces",
"agent_traces_by_key",
"spend_logs",
] {
execute_write(
&database,
&format!("GRANT SELECT ON trace_test.{table} TO help_reader"),
@ -1336,7 +1384,7 @@ async fn query_help_preserves_schema_and_guide_when_discovery_hits_reader_limits
let help = serde_json::to_value(
litellm_traces_clickhouse::query_help(&database.client, &reader).await?,
)?;
assert_eq!(help["tables"].as_array().ok_or("tables")?.len(), 3);
assert_eq!(help["tables"].as_array().ok_or("tables")?.len(), 6);
assert!(!help["examples"].as_array().ok_or("examples")?.is_empty());
assert_eq!(
help["normalized_fields"]
@ -1495,6 +1543,7 @@ async fn trusted_and_sql_readers_share_request_log_visibility(
.calls(
&owned(user, &teams),
&CallQuery {
as_of_ms: u64::MAX,
window: timestamp / 1_000_000 - 1..timestamp / 1_000_000 + 1,
response_ids: vec!["shared-response".into()],
request_ids: Vec::new(),
@ -1577,7 +1626,7 @@ async fn rollup_cost_completeness_preserves_missing_ids_and_fails_closed_for_his
)
.await?;
let listed = listed["data"].as_array().ok_or("missing runs")?;
assert_eq!(listed.len(), 4);
assert_eq!(listed.len(), 3);
let owner = list_runs(&database, &reader, &owned("owner", &[]), window, None, 10).await?;
let owner: std::collections::BTreeSet<_> = owner["data"]
.as_array()

View file

@ -176,6 +176,155 @@ async fn documented_failed_spans_filters_status_after_selecting_the_canonical_co
Ok(())
}
#[rstest]
#[case::admin(QueryScope::All)]
#[case::team(QueryScope::Owned { user_id: String::new(), team_ids: vec!["team-a".into()] })]
#[tokio::test]
async fn logical_trace_core_metrics_agree_with_curated_metrics_after_duplicate_exports(
#[future(awt)] migrated_database: TestResult<SeededDatabase>,
fixture_clock: TestResult<u64>,
#[case] scope: QueryScope,
) -> TestResult {
let fixture = migrated_database?;
let start_ns = fixture_clock? * 1_000_000_000 + 900_000;
let rows = [
(
"key-a",
"root",
"",
start_ns,
900_000,
"STATUS_CODE_OK",
1,
12,
),
(
"key-a",
"root",
"",
start_ns,
9_000_000,
"STATUS_CODE_ERROR",
2,
999,
),
(
"key-a",
"child",
"root",
start_ns + 1_200_000,
700_000,
"STATUS_CODE_ERROR",
1,
3,
),
(
"key-a",
"child",
"root",
start_ns + 1_200_000,
100_000,
"STATUS_CODE_OK",
2,
999,
),
(
"key-alt",
"root",
"",
start_ns,
7_000_000,
"STATUS_CODE_OK",
1,
5,
),
]
.map(
|(key, span, parent, timestamp, duration, status, received, tokens)| {
BTreeMap::from([
("TeamId".into(), json!("team-a")),
("ApiKeyHash".into(), json!(key)),
("TraceId".into(), json!("duplicates")),
("SpanId".into(), json!(span)),
("ParentSpanId".into(), json!(parent)),
("SpanName".into(), json!(span)),
(
"ObservationType".into(),
json!(if parent.is_empty() { "agent" } else { "llm" }),
),
(
"InputPreview".into(),
json!(if parent.is_empty() { "" } else { "child input" }),
),
("Timestamp".into(), json!(timestamp)),
("Duration".into(), json!(duration)),
("StatusCode".into(), json!(status)),
("EngineReceivedMs".into(), json!(received)),
("InputTokens".into(), json!(tokens)),
])
},
);
let writer = litellm_traces_clickhouse::Connection::writer(&fixture.database.url)?;
let encoded = litellm_traces_clickhouse::encode_rows(rows.to_vec())?;
fixture
.database
.client
.post(writer.url().clone())
.body(format!(
"INSERT INTO {}.otel_traces FORMAT JSONEachRow\n{encoded}",
fixtures::DATABASE
))
.send()
.await?
.error_for_status()?;
let connection = fixture
.readers
.connection(&fixture.database.client, &scope, "fixture-secret")
.await?;
let logical: QueryResult = serde_json::from_str(&query_sql(
&fixture.database.client, &connection,
"SELECT id, api_key_hash, span_count, error_count, duration_ns, duration_ms, span_input_tokens, input_preview, root_status, has_error, agent_span_count, agent_label_count, llm_span_count, tool_span_count FROM traces ORDER BY id",
).await?)?;
assert_eq!(logical.data.len(), 2);
let primary = logical
.data
.iter()
.find(|row| row["api_key_hash"] == "key-a")
.ok_or("missing primary trace")?;
assert_eq!(primary["span_count"], 2);
assert_eq!(primary["error_count"], 1);
assert_eq!(primary["duration_ns"], 1_900_000);
assert_eq!(primary["duration_ms"], 1.9);
assert_eq!(primary["span_input_tokens"], 15);
assert_eq!(primary["agent_span_count"], 1);
assert_eq!(primary["agent_label_count"], 1);
assert_eq!(primary["llm_span_count"], 1);
assert_eq!(primary["tool_span_count"], 0);
assert_eq!(primary["input_preview"], "child input");
assert_eq!(primary["root_status"], "ok");
assert_eq!(primary["has_error"], true);
let alternate = logical
.data
.iter()
.find(|row| row["api_key_hash"] == "key-alt")
.ok_or("missing alternate trace")?;
assert_eq!(alternate["span_count"], 1);
let store = ClickHouseTraces::new(fixture.database.client.clone(), connection);
let curated = store.runs(&scope, &newest(10, None)).await?;
for row in curated {
let sql_row = logical
.data
.iter()
.find(|value| value["id"] == row.trace_ref)
.ok_or("missing logical trace")?;
assert_eq!(sql_row["span_count"], row.span_count);
assert_eq!(sql_row["error_count"], row.error_count);
assert_eq!(sql_row["duration_ns"], row.duration_ns);
assert_eq!(sql_row["input_preview"], row.input_preview);
}
Ok(())
}
#[fixture]
fn fixture_clock() -> TestResult<u64> {
let spans = litellm_traces::decode_otlp(

View file

@ -1,87 +1,94 @@
{
"admin": [
{
"id": "62AB4F77353031F31CA3F7DBDCDB9384D00A2C2D683F8E318DD52CEDC8F50F12",
"team": "team-a",
"api_key": "key-a",
"trace_id": "01010101010101010101010101010101",
"name": "review",
"spans": 3,
"llm_calls": 1,
"llm_span_count": 1,
"errors": 1,
"input_tokens": 12,
"output_tokens": 6
"span_input_tokens": 12,
"span_output_tokens": 6
},
{
"id": "191720068A91A381A3C4764EF56D734B47B21ECFF199F0CDECEC265D7FCC0F10",
"team": "team-a",
"api_key": "key-alt",
"trace_id": "01010101010101010101010101010101",
"name": "alternate",
"spans": 1,
"llm_calls": 0,
"llm_span_count": 0,
"errors": 0,
"input_tokens": 0,
"output_tokens": 0
"span_input_tokens": 0,
"span_output_tokens": 0
},
{
"id": "9C868E6B1426BCC5379F5D413F185B57ADBC324A0C58C93311784730A2D3FE12",
"team": "team-b",
"api_key": "key-b",
"trace_id": "01010101010101010101010101010101",
"name": "other-team",
"spans": 1,
"llm_calls": 0,
"llm_span_count": 0,
"errors": 0,
"input_tokens": 0,
"output_tokens": 0
"span_input_tokens": 0,
"span_output_tokens": 0
}
],
"team": [
{
"id": "62AB4F77353031F31CA3F7DBDCDB9384D00A2C2D683F8E318DD52CEDC8F50F12",
"team": "team-a",
"api_key": "key-a",
"trace_id": "01010101010101010101010101010101",
"name": "review",
"spans": 3,
"llm_calls": 1,
"llm_span_count": 1,
"errors": 1,
"input_tokens": 12,
"output_tokens": 6
"span_input_tokens": 12,
"span_output_tokens": 6
},
{
"id": "191720068A91A381A3C4764EF56D734B47B21ECFF199F0CDECEC265D7FCC0F10",
"team": "team-a",
"api_key": "key-alt",
"trace_id": "01010101010101010101010101010101",
"name": "alternate",
"spans": 1,
"llm_calls": 0,
"llm_span_count": 0,
"errors": 0,
"input_tokens": 0,
"output_tokens": 0
"span_input_tokens": 0,
"span_output_tokens": 0
}
],
"key": [
{
"id": "62AB4F77353031F31CA3F7DBDCDB9384D00A2C2D683F8E318DD52CEDC8F50F12",
"team": "team-a",
"api_key": "key-a",
"trace_id": "01010101010101010101010101010101",
"name": "review",
"spans": 3,
"llm_calls": 1,
"llm_span_count": 1,
"errors": 1,
"input_tokens": 12,
"output_tokens": 6
"span_input_tokens": 12,
"span_output_tokens": 6
}
],
"other_team": [
{
"id": "9C868E6B1426BCC5379F5D413F185B57ADBC324A0C58C93311784730A2D3FE12",
"team": "team-b",
"api_key": "key-b",
"trace_id": "01010101010101010101010101010101",
"name": "other-team",
"spans": 1,
"llm_calls": 0,
"llm_span_count": 0,
"errors": 0,
"input_tokens": 0,
"output_tokens": 0
"span_input_tokens": 0,
"span_output_tokens": 0
}
]
}

View file

@ -1,8 +1,10 @@
use std::collections::BTreeMap;
use litellm_http::Client;
use litellm_traces::api::SqlParameter;
use litellm_traces_clickhouse::{
Connection, Error, QueryReaders, QueryScope, ensure_schema, query_help, query_sql,
query_sql_with_params,
};
use rstest::{fixture, rstest};
use serde_json::{Value, json};
@ -69,6 +71,13 @@ async fn queries_and_help_are_scoped_by_the_database(
"SELECT id FROM (SELECT SpanId AS id FROM otel_traces UNION DISTINCT SELECT SpanId AS id FROM trace_test.otel_traces) ORDER BY id",
"SELECT t.SpanId AS id FROM otel_traces t INNER JOIN spend_logs s ON t.SpanId = s.request_id ORDER BY id",
"SELECT request_id AS id FROM spend_logs FINAL ORDER BY id",
"SELECT span_id AS id FROM spans ORDER BY id",
"SELECT span_id AS id FROM spans ORDER BY id FORMAT CSV",
"SELECT span_id AS id FROM spans ORDER BY id SETTINGS http_x_clickhouse_format_overrides_output_format = 0 FORMAT CSV",
"SELECT span_id AS id FROM trace_test.spans ORDER BY id",
"WITH visible AS (SELECT * FROM spans) SELECT span_id AS id FROM visible ORDER BY id",
"SELECT t.span_id AS id FROM spans t INNER JOIN calls c ON t.span_id = c.request_id ORDER BY id",
"SELECT request_id AS id FROM calls ORDER BY id",
];
for sql in queries {
let body: Value = serde_json::from_str(&query_sql(&database.client, &reader, sql).await?)?;
@ -92,6 +101,15 @@ async fn queries_and_help_are_scoped_by_the_database(
.await?,
)?;
assert_eq!(summary["data"][0]["count"], json!(expected.len()));
let canonical: Value = serde_json::from_str(
&query_sql(
&database.client,
&reader,
"SELECT sum(span_count) AS count FROM traces",
)
.await?,
)?;
assert_eq!(canonical["data"], summary["data"]);
let help = serde_json::to_string(&query_help(&database.client, &reader).await?)?;
assert_eq!(help.contains("secret_b"), expected.contains(&"b"));
assert_eq!(help.contains("secret-b"), expected.contains(&"b"));
@ -103,6 +121,138 @@ async fn queries_and_help_are_scoped_by_the_database(
Ok(())
}
#[rstest]
#[tokio::test]
async fn native_parameters_preserve_values_and_cannot_change_reader_scope(
#[future(awt)] database: Result<Database, Box<dyn std::error::Error>>,
) -> Result<(), Box<dyn std::error::Error>> {
let database = database?;
let reader = database
.readers
.connection(
&database.client,
&QueryScope::Owned {
user_id: String::new(),
team_ids: vec!["team-a".into()],
},
"test-secret",
)
.await?;
let text = "quote' OR 1=1 --\\\n\t\r\0雪";
let strings = vec!["a'b".to_owned(), "\\\n雪".to_owned()];
let params = BTreeMap::from([
("text".into(), SqlParameter::String(text.into())),
("signed".into(), SqlParameter::Integer(i64::MIN)),
("unsigned".into(), SqlParameter::Unsigned(u64::MAX)),
("number".into(), SqlParameter::Number(12.5)),
("enabled".into(), SqlParameter::Boolean(true)),
("optional".into(), SqlParameter::Null),
("strings".into(), SqlParameter::Strings(strings.clone())),
(
"team".into(),
SqlParameter::String("team-b' OR 1=1 --".into()),
),
]);
let body: Value = serde_json::from_str(&query_sql_with_params(
&database.client, &reader,
"SELECT {text:String} AS text, toString({signed:Int64}) AS signed, toString({unsigned:UInt64}) AS unsigned, {number:Float64} AS number, {enabled:Bool} AS enabled, {optional:Nullable(String)} AS optional, {strings:Array(String)} AS strings",
&params,
).await?)?;
assert_eq!(
body["data"],
json!([{
"text": text, "signed": i64::MIN.to_string(), "unsigned": u64::MAX.to_string(),
"number": 12.5, "enabled": true, "optional": null, "strings": strings,
}])
);
let invisible: Value = serde_json::from_str(
&query_sql_with_params(
&database.client,
&reader,
"SELECT span_id FROM spans WHERE team_id = {team:String}",
&params,
)
.await?,
)?;
assert_eq!(invisible["data"], json!([]));
let foreign = BTreeMap::from([("team".into(), SqlParameter::String("team-b".into()))]);
let invisible: Value = serde_json::from_str(
&query_sql_with_params(
&database.client,
&reader,
"SELECT span_id FROM spans WHERE team_id = {team:String}",
&foreign,
)
.await?,
)?;
assert_eq!(invisible["data"], json!([]));
Ok(())
}
#[rstest]
#[tokio::test]
async fn logical_views_never_expose_foreign_rows_within_a_mixed_owner_trace(
#[future(awt)] database: Result<Database, Box<dyn std::error::Error>>,
) -> Result<(), Box<dyn std::error::Error>> {
let database = database?;
for sql in [
"INSERT INTO trace_test.otel_traces (TeamId, ApiKeyHash, TraceId, SpanId, SpanName, Timestamp, Duration, SpanAttributes, UserId, StatusCode) VALUES ('team-a', 'mixed-key', 'mixed', 'own', 'visible-root', now(), 1, map('visible', 'owned'), 'owner', 'STATUS_CODE_OK'), ('team-a', 'mixed-key', 'mixed', 'foreign', 'secret-name', now(), 999000000, map('secret-attribute', 'secret-value'), 'other', 'STATUS_CODE_ERROR')",
"INSERT INTO trace_test.spend_logs (team_id, api_key, request_id, trace_id, start_time, end_time, spend, metadata, user) VALUES ('team-a', 'mixed-key', 'own', 'mixed', now(), now(), 1.5, '{\"visible\":1}', 'owner'), ('team-a', 'mixed-key', 'foreign', 'mixed', now(), now(), 999, '{\"secret-cost\":999}', 'other')",
] {
database
.client
.post(database.writer.url().clone())
.body(sql)
.send()
.await?
.error_for_status()?;
}
let reader = database
.readers
.connection(
&database.client,
&QueryScope::Owned {
user_id: "owner".into(),
team_ids: vec![],
},
"test-secret",
)
.await?;
let spans: Value = serde_json::from_str(
&query_sql(
&database.client,
&reader,
"SELECT name, span_attributes FROM spans WHERE trace_id = 'mixed'",
)
.await?,
)?;
assert_eq!(
spans["data"],
json!([{"name":"visible-root", "span_attributes":{"visible":"owned"}}])
);
let traces: Value = serde_json::from_str(&query_sql(
&database.client, &reader,
"SELECT name, span_count, error_count, duration_ns FROM traces WHERE trace_id = 'mixed'",
).await?)?;
assert_eq!(
traces["data"],
json!([{"name":"visible-root", "span_count":1, "error_count":0, "duration_ns":1}])
);
let calls: Value = serde_json::from_str(
&query_sql(
&database.client,
&reader,
"SELECT request_id, spend, metadata FROM calls WHERE trace_id = 'mixed'",
)
.await?,
)?;
assert_eq!(
calls["data"],
json!([{"request_id":"own", "spend":1.5, "metadata":"{\"visible\":1}"}])
);
Ok(())
}
#[rstest]
#[tokio::test]
async fn reader_cache_reprovisions_after_credential_rotation()

View file

@ -616,7 +616,15 @@ async fn an_oversized_span_keeps_the_run_list_available_with_partial_totals(
},
)
.await?;
assert_eq!(cached.data, before.data);
let cached_run = cached
.data
.iter()
.find(|item| item.trace_ref == run.trace_ref)
.ok_or("missing cached run")?;
assert!(!cached_run.resolution_limited);
assert_eq!(cached_run.name, run.name);
assert_eq!(cached_run.span_count, run.span_count + 1);
assert!(cached_run.duration_ms > run.duration_ms);
let (reader, store) = make_reader(client, connection);
let after = reader
.list_traces(
@ -637,7 +645,8 @@ async fn an_oversized_span_keeps_the_run_list_available_with_partial_totals(
.find(|item| item.trace_ref == run.trace_ref)
.ok_or("missing run")?;
assert!(limited.resolution_limited);
assert_eq!(limited.span_count, 4);
assert_eq!(limited.span_count, cached_run.span_count);
assert_eq!(limited.duration_ms, cached_run.duration_ms);
assert!(
after
.data

View file

@ -32,7 +32,7 @@ fn filter(start_ms: i64, q: &str) -> RunFilter {
RunFilter {
start_ms,
end_ms: WINDOW_END_MS,
search: RunSearch::parse(q),
search: RunSearch::parse(q).unwrap(),
..Default::default()
}
}
@ -216,7 +216,6 @@ async fn list_q_selects_matching_runs_before_paging(
let reader = reader();
let cases: &[(&str, &[&str])] = &[
("", &["gamma", "beta", "alpha"]),
("status:", &["gamma", "beta", "alpha"]),
(r#"name:"plan trip""#, &["gamma", "alpha"]),
(r#"-name:"plan trip""#, &["beta"]),
("NAME:PLAN*", &["gamma", "alpha"]),
@ -225,9 +224,8 @@ async fn list_q_selects_matching_runs_before_paging(
("agent:RESEARCH*", &["alpha"]),
("agent:research", &[]),
("-agent:*", &["gamma"]),
("status:error", &["beta"]),
("status:ok", &["gamma", "alpha"]),
("status:*r*", &["beta"]),
("has_error:true", &["beta"]),
("has_error:false", &["gamma", "alpha"]),
("model:gpt-x", &["gamma", "alpha"]),
("model:gpt", &[]),
("-model:gpt-x", &["beta"]),
@ -243,7 +241,6 @@ async fn list_q_selects_matching_runs_before_paging(
("flight_to", &[]),
("plan hello", &["gamma"]),
("plan -trace_id:gamma", &["alpha"]),
("unknown:x", &[]),
];
for (q, expected) in cases {
let page = reader
@ -472,7 +469,7 @@ async fn histogram_counts_use_the_displayed_bucket_boundaries(
let filter = RunFilter {
start_ms: T0_MS,
end_ms: T0_MS + 10,
search: RunSearch::parse("trace_id:bucket*"),
search: RunSearch::parse("trace_id:bucket*").unwrap(),
..Default::default()
};
let counts = store
@ -490,7 +487,7 @@ async fn histogram_counts_use_the_displayed_bucket_boundaries(
},
)
.await?;
let histogram = litellm_traces::search::histogram(&counts, filter.start_ms, filter.end_ms, 3);
let histogram = litellm_traces::search::histogram(&counts, filter.window(), 3);
for bucket in histogram.buckets {
let expected = (T0_MS..T0_MS + 10)
.filter(|start| (bucket.start_ms..bucket.end_ms).contains(start))
@ -503,7 +500,8 @@ async fn histogram_counts_use_the_displayed_bucket_boundaries(
#[rstest]
#[case::agents_in_scope(RunField::Agent, "", &["researcher", "writer"])]
#[case::names_by_frequency(RunField::Name, "", &["plan trip", "write report"])]
#[case::statuses_by_frequency(RunField::Status, "", &["ok", "error"])]
#[case::root_statuses_by_frequency(RunField::RootStatus, "", &["ok"])]
#[case::run_errors_by_frequency(RunField::HasError, "", &["false", "true"])]
#[case::models_by_frequency(RunField::Model, "", &["gpt-x", "claude-y"])]
#[case::needle_ignores_case(RunField::Name, "WR", &["write report"])]
#[case::needle_matches_inside(RunField::Name, "trip", &["plan trip"])]
@ -545,7 +543,7 @@ async fn admin_scope_sees_every_team(
#[rstest]
#[case::by_model(RunField::Agent, "model:claude-y", &["writer"])]
#[case::by_status(RunField::Name, "status:error", &["write report"])]
#[case::by_status(RunField::Name, "has_error:true", &["write report"])]
#[case::excluding(RunField::Model, "-trace_id:alpha", &["claude-y", "gpt-x"])]
#[tokio::test]
async fn values_narrow_to_runs_matching_the_search(
@ -752,13 +750,28 @@ async fn listing_runs_skips_out_of_window_span_rows(
};
let listed = store.runs(&QueryScope::All, &query).await?;
assert_eq!(listed.len(), 4);
let budget = 2 * table_rows(&fixture, "agent_traces_by_key").await?
+ runs().iter().flat_map(rows).count() as u64;
let list_read = rows_read_by(&fixture, "FROM owned_runs").await?;
let reader = reader();
let id = &listed[0].trace_ref;
let metadata = reader
.get_trace_metadata(&store, &QueryScope::All, id)
.await?
.unwrap();
assert_eq!(metadata.summary.trace_ref, *id);
let spans = reader
.get_trace_spans(&store, &QueryScope::All, id, None, 500)
.await?
.unwrap();
assert_eq!(spans.data.len() as u64, metadata.summary.span_count);
let budget = 4 * table_rows(&fixture, "agent_traces_by_key").await?
+ 3 * runs().iter().flat_map(rows).count() as u64;
let read = rows_read_by(&fixture, "FROM owned_runs").await?;
assert!(
read <= budget,
"listing current runs read {read} rows, exceeding their {budget} rollup and in-window span rows"
);
for (operation, read) in [("list", list_read), ("canonical ID lookup", read)] {
assert!(
read <= budget,
"{operation} read {read} rows, exceeding the {budget}-row budget for bounded candidate, ownership and canonical scans"
);
}
Ok(())
}
@ -893,7 +906,7 @@ async fn add_attributes(fixture: &SeededDatabase) -> TestResult {
#[case::missing_attribute("-attr.tenant.tier:*", &["gamma"])]
#[case::missing_wildcard("attr.missing:*", &[])]
#[case::two_attributes("attr.tenant.tier:gold attr.tenant.tier:silver", &[])]
#[case::attribute_and_field("attr.tenant.tier:* status:error", &["beta"])]
#[case::attribute_and_field("attr.tenant.tier:* has_error:true", &["beta"])]
#[case::unknown_attribute("attr.missing:gold", &[])]
#[tokio::test]
async fn service_team_and_attribute_filters_select_runs(
@ -1047,7 +1060,7 @@ async fn metric_sorting_pages_by_the_deduplicated_displayed_values(
assert_eq!(run.error_count, error_count);
}
let errors = RunFilter {
search: RunSearch::parse("trace_id:metric* status:error"),
search: RunSearch::parse("trace_id:metric* has_error:true").unwrap(),
..filter.clone()
};
let filtered = reader
@ -1321,3 +1334,172 @@ async fn span_text_reads_ranges_of_each_listed_span(
assert!(foreign.is_empty());
Ok(())
}
#[rstest]
#[tokio::test]
async fn list_snapshot_excludes_late_spans_and_preserves_sorted_pages(
#[future] migrated_database: TestResult<SeededDatabase>,
) -> TestResult {
let fixture = migrated_database.await?;
let writer = Connection::writer(&fixture.database.url)?;
let store = ClickHouseTraces::new(
fixture.database.client.clone(),
Connection::configured(&fixture.database.url, DATABASE, "default", "")?,
);
let row = |trace: &str, span: &str, duration: u64, received: u64, error: bool, offset: i64| {
BTreeMap::from([
("Timestamp".into(), json!((T0_MS + offset) * 1_000_000)),
("TraceId".into(), json!(trace)),
("SpanId".into(), json!(span)),
(
"ParentSpanId".into(),
json!(if span == "root" { "" } else { "root" }),
),
("SpanName".into(), json!(span)),
("ServiceName".into(), json!("snapshot")),
("ObservationType".into(), json!("chain")),
("TeamId".into(), json!("team-a")),
("ApiKeyHash".into(), json!("key-a")),
("Duration".into(), json!(duration)),
("EngineReceivedMs".into(), json!(received)),
(
"StatusCode".into(),
json!(if error {
"STATUS_CODE_ERROR"
} else {
"STATUS_CODE_OK"
}),
),
])
};
let insert = |rows: Vec<BTreeMap<String, Value>>| {
let client = fixture.database.client.clone();
let url = writer.url().clone();
async move {
client
.post(url)
.query(&[(
"query",
format!("INSERT INTO {DATABASE}.otel_traces FORMAT JSONEachRow"),
)])
.body(encode_rows(rows)?)
.send()
.await?
.error_for_status()?;
TestResult::Ok(())
}
};
insert(vec![
row("snapshot-a", "root", 1_000_000, 10, false, 0),
row("snapshot-b", "root", 2_000_000, 10, false, 0),
row("snapshot-c", "root", 3_000_000, 10, false, 0),
])
.await?;
let reader = reader();
let filter = RunFilter {
as_of_ms: 10,
..filter(T0_MS, "trace_id:snapshot*")
};
let order = RunOrder {
key: RunSortKey::DurationMs,
descending: false,
};
let first = reader
.list_traces(&store, &team_a(), &filter, order, &page(None, 1))
.await?;
assert_eq!(first.data[0].trace_id, "snapshot-a");
assert_eq!(first.window, filter.window());
insert(vec![
row("snapshot-a", "late", 10_000_000, 20, true, 0),
row("snapshot-new", "root", 100_000, 20, false, 0),
row("snapshot-c", "root", 100_000_000, 20, true, -1),
row("snapshot-c", "backdated-child", 1_000_000, 20, true, -1),
])
.await?;
let live_filter = RunFilter {
as_of_ms: 20,
..filter.clone()
};
let live = reader
.list_traces(&store, &team_a(), &live_filter, order, &page(None, 10))
.await?;
let live_a = live
.data
.iter()
.find(|run| run.trace_id == "snapshot-a")
.unwrap();
assert!(live_a.has_error);
assert_eq!(live_a.status, litellm_traces::SpanStatus::Ok);
let metadata = reader
.get_trace_metadata(&store, &team_a(), &live_a.trace_ref)
.await?
.unwrap();
assert_eq!(metadata.summary.trace_ref, live_a.trace_ref);
assert!(metadata.summary.has_error);
let spans = reader
.get_trace_spans(&store, &team_a(), &live_a.trace_ref, None, 1)
.await?
.unwrap();
let next = reader
.get_trace_spans(
&store,
&team_a(),
&live_a.trace_ref,
spans.next_cursor.as_deref(),
1,
)
.await?
.unwrap();
assert_ne!(spans.data[0].span_id, next.data[0].span_id);
assert!(next.next_cursor.is_none());
let second = reader
.list_traces(
&store,
&team_a(),
&filter,
order,
&page(first.next_cursor, 1),
)
.await?;
let third = reader
.list_traces(
&store,
&team_a(),
&filter,
order,
&page(second.next_cursor, 1),
)
.await?;
assert_eq!(second.data[0].trace_id, "snapshot-b");
assert_eq!(third.data[0].trace_id, "snapshot-c");
assert_eq!(third.data[0].duration_ms, 3.0);
assert!(!third.data[0].has_error);
assert!(third.next_cursor.is_none());
let repeated = reader
.list_traces(&store, &team_a(), &filter, order, &page(None, 10))
.await?;
assert_eq!(repeated.data[0].duration_ms, 1.0);
assert!(!repeated.data[0].has_error);
assert_eq!(reader.count_traces(&store, &team_a(), &filter).await?, 3);
let histogram = reader.histogram(&store, &team_a(), &filter, 3).await?;
assert_eq!(histogram.window, filter.window());
assert_eq!(
histogram
.buckets
.iter()
.map(|bucket| bucket.total)
.sum::<u64>(),
3
);
let errors = RunFilter {
search: RunSearch::parse("has_error:true root_status:ok").unwrap(),
..live_filter
};
let errors = reader
.list_traces(&store, &team_a(), &errors, order, &page(None, 10))
.await?;
assert_eq!(errors.data.len(), 1);
assert_eq!(errors.data[0].trace_id, "snapshot-a");
Ok(())
}

View file

@ -6,12 +6,13 @@ license.workspace = true
repository.workspace = true
[features]
schema = ["dep:schemars"]
schema = ["dep:schemars", "dep:utoipa"]
[dependencies]
askama.workspace = true
macro_rules_attribute.workspace = true
schemars = { workspace = true, optional = true }
utoipa = { workspace = true, optional = true }
indexmap = { version = "2", features = ["serde"] }
litellm-llms-types.workspace = true
opentelemetry-proto = { workspace = true, features = ["gen-tonic-messages", "trace", "with-serde"] }
@ -26,6 +27,7 @@ time.workspace = true
base64.workspace = true
criterion.workspace = true
rstest.workspace = true
jsonschema = { version = "0.55.1", default-features = false }
[[bench]]
name = "resource-fanout"

View file

@ -0,0 +1,327 @@
use std::collections::BTreeMap;
use serde::{
Deserialize, Deserializer,
de::{Error, Unexpected},
};
use serde_json::Value;
use crate::{AgentNode, Span, TraceSummary};
#[cfg(feature = "schema")]
pub mod openapi;
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
#[serde(deny_unknown_fields)]
pub struct TraceQueryWindow {
pub start_ms: i64,
pub end_ms: i64,
pub as_of_ms: u64,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
#[serde(rename_all = "snake_case")]
pub enum TraceSortField {
#[default]
StartMs,
DurationMs,
SpanCount,
ErrorCount,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
#[serde(rename_all = "lowercase")]
pub enum TraceSortDirection {
Asc,
#[default]
Desc,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(deny_unknown_fields)]
pub struct TraceNoQueryRequest {}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(deny_unknown_fields)]
pub struct TraceListRequest {
#[serde(default)]
pub as_of_ms: Option<u64>,
#[serde(default)]
pub start_ms: Option<i64>,
#[serde(default)]
pub end_ms: Option<i64>,
#[serde(default)]
#[cfg_attr(feature = "schema", schemars(length(max = 1000)))]
#[serde(deserialize_with = "text::<_, 1000>")]
pub q: String,
#[serde(default)]
#[cfg_attr(feature = "schema", schemars(length(max = 512)))]
#[serde(deserialize_with = "cursor")]
pub cursor: Option<String>,
#[serde(default = "list_page_size")]
#[cfg_attr(feature = "schema", schemars(range(min = 1, max = 500)))]
#[serde(deserialize_with = "integer::<_, 1, 500>")]
pub page_size: u16,
#[serde(default)]
pub sort_by: TraceSortField,
#[serde(default)]
pub sort_dir: TraceSortDirection,
}
const fn list_page_size() -> u16 {
50
}
const fn span_page_size() -> u16 {
100
}
const fn histogram_buckets() -> u16 {
60
}
const fn value_limit() -> u16 {
20
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(deny_unknown_fields)]
pub struct TraceHistogramRequest {
#[serde(default)]
pub start_ms: Option<i64>,
#[serde(default)]
pub end_ms: Option<i64>,
#[serde(default)]
pub as_of_ms: Option<u64>,
#[serde(default)]
#[cfg_attr(feature = "schema", schemars(length(max = 1000)))]
#[serde(deserialize_with = "text::<_, 1000>")]
pub q: String,
#[serde(default = "histogram_buckets")]
#[cfg_attr(feature = "schema", schemars(range(min = 1, max = 240)))]
#[serde(deserialize_with = "integer::<_, 1, 240>")]
pub buckets: u16,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(deny_unknown_fields)]
pub struct TraceValuesRequest {
#[serde(default)]
pub start_ms: Option<i64>,
#[serde(default)]
pub end_ms: Option<i64>,
#[serde(default)]
pub as_of_ms: Option<u64>,
#[serde(default)]
#[cfg_attr(feature = "schema", schemars(length(max = 1000)))]
#[serde(deserialize_with = "text::<_, 1000>")]
pub q: String,
#[serde(default)]
#[cfg_attr(feature = "schema", schemars(length(max = 200)))]
#[serde(deserialize_with = "text::<_, 200>")]
pub contains: String,
#[serde(default = "value_limit")]
#[cfg_attr(feature = "schema", schemars(range(min = 1, max = 100)))]
#[serde(deserialize_with = "integer::<_, 1, 100>")]
pub limit: u16,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(deny_unknown_fields)]
pub struct TraceSpanPageRequest {
#[serde(default)]
#[cfg_attr(feature = "schema", schemars(length(max = 512)))]
#[serde(deserialize_with = "cursor")]
pub cursor: Option<String>,
#[serde(default = "span_page_size")]
#[cfg_attr(feature = "schema", schemars(range(min = 1, max = 500)))]
#[serde(deserialize_with = "integer::<_, 1, 500>")]
pub page_size: u16,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(deny_unknown_fields)]
pub struct TraceErrorPageRequest {
#[serde(default)]
#[cfg_attr(feature = "schema", schemars(length(max = 512)))]
#[serde(deserialize_with = "cursor")]
pub cursor: Option<String>,
}
#[macro_rules_attribute::apply(response_type)]
#[derive(Clone, Debug, PartialEq)]
pub struct TraceMetadata {
pub summary: TraceSummary,
pub agents: Vec<AgentNode>,
}
#[macro_rules_attribute::apply(response_type)]
#[derive(Clone, Debug, PartialEq)]
pub struct TraceSpansPage {
pub data: Vec<Span>,
pub next_cursor: Option<String>,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug, PartialEq)]
#[serde(untagged)]
pub enum SqlParameter {
String(String),
Integer(i64),
Unsigned(u64),
Number(f64),
Boolean(bool),
Null,
Strings(Vec<String>),
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(deny_unknown_fields)]
pub struct TraceQueryRequest {
#[cfg_attr(feature = "schema", schemars(length(min = 1)))]
#[serde(deserialize_with = "sql")]
pub sql: String,
#[serde(default)]
pub params: BTreeMap<String, SqlParameter>,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(untagged)]
pub enum UnsignedCount {
Integer(u64),
String(String),
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
pub struct TraceSQLColumn {
pub name: String,
#[serde(rename = "type")]
pub kind: String,
#[serde(flatten)]
pub extra: BTreeMap<String, Value>,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
pub struct TraceQueryStatistics {
pub elapsed: f64,
pub rows_read: UnsignedCount,
pub bytes_read: UnsignedCount,
#[serde(flatten)]
pub extra: BTreeMap<String, Value>,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
pub struct TraceSQLResponse {
pub meta: Vec<TraceSQLColumn>,
#[cfg_attr(feature = "schema", schemars(extend("x-python-normalized" = {"type": "tuple[Mapping[str, JsonValue], ...]"})))]
pub data: Vec<BTreeMap<String, Value>>,
pub rows: UnsignedCount,
pub statistics: TraceQueryStatistics,
#[serde(flatten)]
pub extra: BTreeMap<String, Value>,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
#[serde(rename_all = "snake_case")]
pub enum TraceProblemCode {
InvalidRequest,
Unauthorized,
Forbidden,
NotFound,
TraceChanged,
TooLarge,
Unavailable,
QueryRejected,
QueryLimitExceeded,
QueryUnavailable,
InternalError,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(deny_unknown_fields)]
pub struct TraceInvalidParam {
pub location: String,
pub reason: String,
}
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Debug)]
#[serde(deny_unknown_fields)]
pub struct TraceProblem {
#[serde(rename = "type")]
pub kind: String,
pub title: String,
pub status: u16,
pub detail: String,
pub code: TraceProblemCode,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub database_code: Option<u32>,
#[serde(default, skip_serializing_if = "Vec::is_empty")]
#[cfg_attr(feature = "schema", schemars(extend("default" = [])))]
pub errors: Vec<TraceInvalidParam>,
}
fn integer<'de, D: Deserializer<'de>, const MIN: u16, const MAX: u16>(
deserializer: D,
) -> Result<u16, D::Error> {
let value = u16::deserialize(deserializer)?;
if (MIN..=MAX).contains(&value) {
return Ok(value);
}
Err(D::Error::invalid_value(
Unexpected::Unsigned(u64::from(value)),
&"integer within the documented range",
))
}
fn text<'de, D: Deserializer<'de>, const MAX: usize>(deserializer: D) -> Result<String, D::Error> {
let value = String::deserialize(deserializer)?;
if value.chars().count() <= MAX {
return Ok(value);
}
Err(D::Error::invalid_length(
value.chars().count(),
&"string within the documented length",
))
}
fn cursor<'de, D: Deserializer<'de>>(deserializer: D) -> Result<Option<String>, D::Error> {
let value = Option::<String>::deserialize(deserializer)?;
if value
.as_ref()
.is_none_or(|cursor| cursor.chars().count() <= 512)
{
return Ok(value);
}
Err(D::Error::invalid_length(
value.as_ref().map_or(0, |cursor| cursor.chars().count()),
&"cursor within the documented length",
))
}
fn sql<'de, D: Deserializer<'de>>(deserializer: D) -> Result<String, D::Error> {
let value = String::deserialize(deserializer)?;
if !value.is_empty() {
return Ok(value);
}
Err(D::Error::invalid_value(
Unexpected::Str(&value),
&"nonempty SQL",
))
}

View file

@ -0,0 +1,282 @@
use std::collections::BTreeMap;
use schemars::Schema as JsonSchema;
use serde_json::{Value, json};
use utoipa::openapi::{
Content, Info, OpenApi, OpenApiBuilder, Ref, RefOr, Required,
path::{
HttpMethod, OperationBuilder, Parameter, ParameterBuilder, ParameterIn, PathItem,
PathsBuilder,
},
request_body::RequestBodyBuilder,
response::{ResponseBuilder, ResponsesBuilder},
schema::{ComponentsBuilder, Schema},
security::{Http, HttpAuthScheme, SecurityScheme},
};
struct Endpoint {
path: &'static str,
method: HttpMethod,
operation: &'static str,
request: Option<&'static str>,
response: &'static str,
description: &'static str,
}
const ENDPOINTS: [Endpoint; 9] = [
Endpoint {
path: "/v1/traces",
method: HttpMethod::Get,
operation: "trace_list",
request: Some("TraceListRequest"),
response: "TracePage",
description: "Search trace summaries. All clauses in q must match. Free text searches trace_id, name and input preview as case-insensitive substrings. key:value clauses match whole values, with * as a wildcard; double quotes group spaces or literal colons. Prefix keyed clauses with - to exclude matches. Unknown keys, missing values and malformed quoting return invalid_request. name describes the physical root; input searches its preview, falling back to the first nonempty agent or LLM preview; agent, model and attr.<key> match any span. root_status describes the root, has_error means any failed span. The default window is the last 24 hours. First-page ingestion timestamp cutoff is retained by the cursor. This excludes later-stamped exports, not delayed commits stamped before the cutoff, and is not a database transaction snapshot. Repeat q and sorting on continuation; omitted bounds reuse the cursor window. Sort ties use the canonical id in the same direction. The server may return fewer than page_size items to respect response limits. Continue until next_cursor is null. Span-derived metrics use the cutoff; spend enrichment is best effort",
},
Endpoint {
path: "/v1/traces/histogram",
method: HttpMethod::Get,
operation: "trace_histogram",
request: Some("TraceHistogramRequest"),
response: "TraceHistogram",
description: "Count matching traces in equal-width [start_ms,end_ms) buckets. Reuse the list window and as_of_ms for matching span-derived membership. failed counts traces with any failed span. Agent groups count successful traces under their alphabetically first agent name, or service when no agent name exists",
},
Endpoint {
path: "/v1/traces/values/{field}",
method: HttpMethod::Get,
operation: "trace_values",
request: Some("TraceValuesRequest"),
response: "RunValues",
description: "Return the most common distinct values among matching traces. contains is case-insensitive. limit is a top-K suggestion limit, not a pagination size. Reuse list window and as_of_ms for matching membership",
},
Endpoint {
path: "/v1/traces/{id}",
method: HttpMethod::Get,
operation: "trace_get",
request: Some("TraceNoQueryRequest"),
response: "TraceMetadata",
description: "Read summary and agent metadata by canonical id. trace_id is the original OTLP id, which can repeat across ownership scopes. Spans are read through the separate spans collection",
},
Endpoint {
path: "/v1/traces/{id}/spans",
method: HttpMethod::Get,
operation: "trace_spans",
request: Some("TraceSpanPageRequest"),
response: "TraceSpansPage",
description: "Read a bounded page of canonical spans. The cursor pins the graph version. Continue with the same page_size; a changed graph or expired reconstruction returns trace_changed and the traversal must restart",
},
Endpoint {
path: "/v1/traces/{id}/spans/{span_id}",
method: HttpMethod::Get,
operation: "trace_span",
request: Some("TraceNoQueryRequest"),
response: "SpanDetail",
description: "Read raw input, output and attributes for one span. UI rendering is performed by the client",
},
Endpoint {
path: "/v1/traces/{id}/spans/{span_id}/error",
method: HttpMethod::Get,
operation: "trace_error",
request: Some("TraceErrorPageRequest"),
response: "SpanErrorPage",
description: "Read bounded diagnostic text pages. Cursor validation detects content changes and requires restarting the traversal",
},
Endpoint {
path: "/v1/traces/query",
method: HttpMethod::Post,
operation: "trace_query",
request: Some("TraceQueryRequest"),
response: "TraceSQLResponse",
description: "Execute read-only ClickHouse SQL under authenticated row policies and fixed resource limits. Bind params with native {name:Type} placeholders. Results are always ClickHouse JSON; 64-bit integers may be strings. SQL callers control ORDER BY, LIMIT and keyset continuation. Exceeding a resource limit fails instead of returning partial success",
},
Endpoint {
path: "/v1/traces/query/help",
method: HttpMethod::Get,
operation: "trace_query_help",
request: Some("TraceNoQueryRequest"),
response: "TraceQueryHelp",
description: "Discover current SQL schema, logical views, scoped examples and resource limits",
},
];
fn rewrite_refs(value: Value) -> Value {
match value {
Value::Object(object) => Value::Object(
object
.into_iter()
.filter_map(|(key, value)| {
if key == "$schema" || key == "$defs" {
return None;
}
let rewritten = match (key.as_str(), value) {
("$ref", Value::String(reference)) => {
Value::String(reference.replace("#/$defs/", "#/components/schemas/"))
}
(_, value) => rewrite_refs(value),
};
Some((key, rewritten))
})
.collect(),
),
Value::Array(values) => Value::Array(values.into_iter().map(rewrite_refs).collect()),
value => value,
}
}
fn schema(value: Value) -> RefOr<Schema> {
serde_json::from_value(rewrite_refs(value)).expect("Rust JSON Schema is an OpenAPI 3.1 schema")
}
fn query_parameters(name: &str, schemas: &BTreeMap<&str, JsonSchema>) -> Vec<Parameter> {
let Some(properties) = schemas[name].get("properties").and_then(Value::as_object) else {
return Vec::new();
};
properties
.iter()
.map(|(name, value)| {
ParameterBuilder::new()
.name(name)
.parameter_in(ParameterIn::Query)
.schema(Some(schema(value.clone())))
.build()
})
.collect()
}
fn path_parameters(path: &str) -> impl Iterator<Item = Parameter> + '_ {
path.split('/')
.filter_map(|part| {
part.strip_prefix('{')
.and_then(|value| value.strip_suffix('}'))
})
.map(|name| {
ParameterBuilder::new()
.name(name)
.parameter_in(ParameterIn::Path)
.required(Required::True)
.schema(Some(if name == "field" {
RefOr::Ref(Ref::from_schema_name("RunField"))
} else {
schema(json!({"type":"string","minLength":1}))
}))
.build()
})
}
fn endpoint(endpoint: &Endpoint, schemas: &BTreeMap<&str, JsonSchema>) -> PathItem {
let success = ResponseBuilder::new()
.description("Success")
.content(
"application/json",
Content::new(Some(Ref::from_schema_name(endpoint.response))),
)
.build();
let responses = [400, 401, 403, 404, 409, 413, 422, 500, 501, 503]
.into_iter()
.fold(
ResponsesBuilder::new().response("200", success),
|builder, status| {
builder.response(
status.to_string(),
ResponseBuilder::new()
.description("Trace API problem")
.content(
"application/problem+json",
Content::new(Some(Ref::from_schema_name("TraceProblem"))),
)
.build(),
)
},
)
.build();
let parameters = endpoint
.request
.filter(|_| endpoint.method == HttpMethod::Get)
.map(|request| query_parameters(request, schemas))
.unwrap_or_default();
let request_body = endpoint
.request
.filter(|_| endpoint.method == HttpMethod::Post)
.map(|request| {
RequestBodyBuilder::new()
.required(Some(Required::True))
.content(
"application/json",
Content::new(Some(Ref::from_schema_name(request))),
)
.build()
});
let operation = OperationBuilder::new()
.operation_id(Some(endpoint.operation))
.security(utoipa::openapi::security::SecurityRequirement::new(
"TraceBearer",
[""; 0],
))
.tag("agent tracing")
.description(Some(endpoint.description))
.parameters(Some(path_parameters(endpoint.path).chain(parameters)))
.request_body(request_body)
.responses(responses)
.build();
PathItem::new(endpoint.method.clone(), operation)
}
pub fn document(extra: BTreeMap<&'static str, JsonSchema>) -> OpenApi {
let roots: BTreeMap<_, _> = crate::schema::schemas()
.into_iter()
.filter(|(name, _)| {
*name == "RunField" || ENDPOINTS.iter().any(|endpoint| endpoint.response == *name)
})
.chain(crate::schema::api_schemas())
.chain(extra)
.collect();
let definitions = roots
.values()
.flat_map(|root| {
root.get("$defs")
.and_then(Value::as_object)
.into_iter()
.flat_map(|defs| defs.iter())
})
.map(|(name, value)| (name.clone(), schema(value.clone())));
let components = ComponentsBuilder::new()
.schemas_from_iter(definitions)
.schemas_from_iter(
roots
.iter()
.map(|(name, root)| (*name, schema(root.as_value().clone()))),
)
.security_scheme(
"TraceBearer",
SecurityScheme::Http(Http::new(HttpAuthScheme::Bearer)),
)
.build();
let ingest = OperationBuilder::new().operation_id(Some("trace_ingest"))
.security(utoipa::openapi::security::SecurityRequirement::new("TraceBearer", [""; 0])).tag("agent tracing")
.description(Some("Export OTLP traces as JSON or protobuf, optionally gzip compressed. The OTLP protocol defines payloads and responses: https://opentelemetry.io/docs/specs/otlp/. Ownership is derived from authentication, never payload attributes"))
.request_body(Some(RequestBodyBuilder::new().required(Some(Required::True))
.content("application/json", Content::new(Some(schema(json!({"type":"object"})))))
.content("application/x-protobuf", Content::new(Some(schema(json!({"type":"string","format":"binary"})))))
.build()))
.responses([200,400,401,403,413,429,501,503].into_iter().fold(ResponsesBuilder::new(), |builder, status| {
builder.response(status.to_string(), ResponseBuilder::new().description("OTLP protocol response")
.content("application/json", Content::new(Some(schema(json!({"type":"object"})))))
.content("application/x-protobuf", Content::new(Some(schema(json!({"type":"string","format":"binary"})))))
.build())
}).build()).build();
let paths = ENDPOINTS
.iter()
.fold(PathsBuilder::new(), |builder, item| {
builder.path(item.path, endpoint(item, &roots))
})
.path("/v1/traces", PathItem::new(HttpMethod::Post, ingest))
.build();
OpenApiBuilder::new()
.info(Info::new("LiteLLM Trace API", "1"))
.components(Some(components))
.paths(paths)
.security(Some([utoipa::openapi::security::SecurityRequirement::new(
"TraceBearer",
[""; 0],
)]))
.build()
}

View file

@ -1,6 +1,9 @@
fn main() {
println!(
"{}",
serde_json::to_string_pretty(&litellm_traces::schema::schemas()).unwrap()
);
let arguments = std::env::args().collect::<Vec<_>>();
let schemas = if arguments.get(1).is_some_and(|arg| arg == "--api") {
litellm_traces::schema::api_schemas()
} else {
litellm_traces::schema::schemas()
};
println!("{}", serde_json::to_string_pretty(&schemas).unwrap());
}

View file

@ -17,3 +17,7 @@ pub struct InvalidScope;
#[derive(Debug, thiserror::Error)]
#[error("invalid trace call key")]
pub struct InvalidCallKey;
#[derive(Debug, thiserror::Error)]
#[error("invalid trace search query")]
pub struct InvalidQuery;

View file

@ -10,6 +10,7 @@ macro_rules_attribute::attribute_alias! {
#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))];
}
pub mod api;
mod error;
mod normalize;
mod otlp;
@ -27,7 +28,7 @@ mod ui;
mod view;
pub mod wire;
pub use error::{Error, InvalidCallKey, InvalidScope};
pub use error::{Error, InvalidCallKey, InvalidQuery, InvalidScope};
pub use normalize::{
AgentMetadata, AgentType, CallEvidence, CallEvidenceKind, CallKey, Integration, NormalizedSpan,
ObservationType,

View file

@ -41,11 +41,6 @@ impl<'a> Graph<'a> {
self.by_id.get(row.parent_span_id.as_str()).copied()
}
pub(super) fn is_root(&self, index: usize) -> bool {
let parent = &self.rows[index].parent_span_id;
parent.is_empty() || !self.by_id.contains_key(parent.as_str())
}
pub(super) fn children(&self, index: usize) -> Vec<usize> {
self.children
.get(self.id(index))
@ -137,8 +132,8 @@ mod tests {
.map(|index| graph.id(index))
.collect();
assert_eq!(descendants, ["leaf", "middle", "sibling"].into());
assert!(graph.is_root(3));
assert!(!graph.is_root(0));
assert!(graph.parent(3).is_none());
assert!(graph.parent(0).is_some());
}
#[rstest]

View file

@ -138,7 +138,7 @@ pub fn resolve_trace(
rows: &[SpanRow],
spend: &[SpendRow],
) -> Option<Trace> {
let first = rows.first()?;
let first = rows.iter().min_by_key(|row| (row.start_ns, &row.span_id))?;
let resolution = Resolution::new(rows, spend);
let trace_start_ns = rows.iter().map(|row| row.start_ns).min()?;
let trace_end_ns = rows
@ -148,9 +148,17 @@ pub fn resolve_trace(
let spans: Vec<Span> = (0..rows.len())
.map(|index| span(&resolution, index, trace_start_ns))
.collect();
let root = (0..rows.len())
.find(|index| resolution.graph.is_root(*index))
.unwrap_or_default();
let root = rows
.iter()
.enumerate()
.filter(|(_, row)| row.parent_span_id.is_empty())
.min_by_key(|(_, row)| (row.start_ns, &row.span_id))
.or_else(|| {
rows.iter()
.enumerate()
.min_by_key(|(_, row)| (row.start_ns, &row.span_id))
})
.map(|(index, _)| index)?;
let agents = agents(&resolution);
let calls = &resolution.model_calls;
let counted: Vec<&SpanRow> = if calls.is_empty() {
@ -158,16 +166,14 @@ pub fn resolve_trace(
} else {
calls.iter().map(|call| &rows[*call]).collect()
};
let first_input = spans
let first_input = rows
.iter()
.zip(rows)
.enumerate()
.filter(|(_, (span, _))| {
!span.input_preview.is_empty()
&& matches!(span.kind, ObservationType::Agent | ObservationType::Llm)
.filter(|row| {
!row.input_preview.is_empty()
&& matches!(row.kind, ObservationType::Agent | ObservationType::Llm)
})
.min_by_key(|(index, (_, row))| (row.start_ns, *index))
.map(|(_, (span, _))| span.input_preview.clone())
.min_by_key(|row| (row.start_ns, &row.span_id))
.map(|row| row.input_preview.clone())
.unwrap_or_default();
let summary = TraceSummary {
resolution_limited: false,
@ -185,7 +191,8 @@ pub fn resolve_trace(
input_preview: optional(&spans[root].input_preview).unwrap_or(first_input),
start_time: iso_time(trace_start_ns.div_euclid(1_000_000)),
duration_ms: (trace_end_ns - i128::from(trace_start_ns)) as f64 / NANOS_PER_MS,
status: spans[root].status,
status: rows[root].status,
has_error: spans.iter().any(|span| span.status == SpanStatus::Error),
span_count: spans.len() as u64,
agent_count: agents.len() as u64,
agent_invocations: agents.iter().map(|agent| agent.invocations).sum(),
@ -226,6 +233,7 @@ pub fn listed_summary(row: &RunRow) -> TraceSummary {
start_time: iso_time(row.start_ms),
duration_ms: row.duration_ns as f64 / NANOS_PER_MS,
status: row.status,
has_error: row.error_count > 0,
span_count: row.span_count,
agent_count: row.agent_count,
agent_invocations: if row.agent_invocations == 0 {

View file

@ -9,6 +9,13 @@ pub fn flag(_: &mut SchemaGenerator) -> Schema {
.unwrap()
}
fn integer_value(value: &serde_json::Value) -> Option<i128> {
value
.as_i64()
.map(i128::from)
.or_else(|| value.as_u64().map(i128::from))
}
pub fn integer_bounds(schema: &mut Schema) {
let bounds = match schema.get("format").and_then(serde_json::Value::as_str) {
Some("uint8") => Some((json!(0), json!(u8::MAX))),
@ -22,13 +29,19 @@ pub fn integer_bounds(schema: &mut Schema) {
_ => None,
};
if let Some((minimum, maximum)) = bounds {
schema.insert("minimum".to_owned(), minimum);
schema.insert("maximum".to_owned(), maximum);
let existing_minimum = schema.get("minimum").and_then(integer_value);
let existing_maximum = schema.get("maximum").and_then(integer_value);
if existing_minimum.is_none_or(|bound| bound < integer_value(&minimum).unwrap()) {
schema.insert("minimum".to_owned(), minimum);
}
if existing_maximum.is_none_or(|bound| bound > integer_value(&maximum).unwrap()) {
schema.insert("maximum".to_owned(), maximum);
}
}
schemars::transform::transform_subschemas(&mut integer_bounds, schema);
}
fn received<T: JsonSchema>() -> Schema {
pub(crate) fn received<T: JsonSchema>() -> Schema {
SchemaSettings::draft2020_12()
.for_deserialize()
.with_transform(integer_bounds)
@ -36,7 +49,7 @@ fn received<T: JsonSchema>() -> Schema {
.into_root_schema_for::<T>()
}
fn emitted<T: JsonSchema>() -> Schema {
pub(crate) fn emitted<T: JsonSchema>() -> Schema {
SchemaSettings::draft2020_12()
.for_serialize()
.with_transform(integer_bounds)
@ -59,3 +72,43 @@ pub fn schemas() -> BTreeMap<&'static str, Schema> {
("RunOrder", received::<crate::store::RunOrder>()),
])
}
pub fn api_schemas() -> BTreeMap<&'static str, Schema> {
BTreeMap::from([
(
"TraceNoQueryRequest",
received::<crate::api::TraceNoQueryRequest>(),
),
(
"TraceListRequest",
received::<crate::api::TraceListRequest>(),
),
(
"TraceHistogramRequest",
received::<crate::api::TraceHistogramRequest>(),
),
(
"TraceValuesRequest",
received::<crate::api::TraceValuesRequest>(),
),
(
"TraceSpanPageRequest",
received::<crate::api::TraceSpanPageRequest>(),
),
(
"TraceErrorPageRequest",
received::<crate::api::TraceErrorPageRequest>(),
),
(
"TraceQueryRequest",
received::<crate::api::TraceQueryRequest>(),
),
("TraceMetadata", emitted::<crate::api::TraceMetadata>()),
("TraceSpansPage", emitted::<crate::api::TraceSpansPage>()),
("TraceProblem", emitted::<crate::api::TraceProblem>()),
(
"TraceSQLResponse",
emitted::<crate::api::TraceSQLResponse>(),
),
])
}

View file

@ -20,7 +20,8 @@ use crate::store::RunCount;
pub enum RunField {
Name,
Agent,
Status,
RootStatus,
HasError,
Model,
Input,
TraceId,
@ -48,15 +49,38 @@ impl SearchKey {
}
}
#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize)]
#[derive(Clone, Debug, Eq, PartialEq, Serialize)]
pub struct RunFilter {
pub start_ms: i64,
pub end_ms: i64,
pub as_of_ms: u64,
pub search: RunSearch,
/// When not empty, only these runs can match.
pub trace_refs: Vec<String>,
}
impl RunFilter {
pub fn window(&self) -> crate::api::TraceQueryWindow {
crate::api::TraceQueryWindow {
start_ms: self.start_ms,
end_ms: self.end_ms,
as_of_ms: self.as_of_ms,
}
}
}
impl Default for RunFilter {
fn default() -> Self {
Self {
start_ms: 0,
end_ms: 0,
as_of_ms: u64::MAX,
search: RunSearch::default(),
trace_refs: Vec::new(),
}
}
}
#[derive(Clone, Debug, Eq, PartialEq, Serialize)]
pub struct FieldFilter {
pub key: SearchKey,
@ -74,57 +98,41 @@ pub struct RunSearch {
}
impl RunSearch {
/// Mirrors the dashboard's search box: `key:value` filters on known keys, `-key:value` negates,
/// `*` globs, double quotes keep spaces, and anything else is a free-text term.
/// A key typed without a value yet narrows nothing.
pub fn parse(q: &str) -> Self {
let (text, filters): (Vec<_>, Vec<_>) = tokens(q)
.map(clause)
.filter(|clause| !clause.value().is_empty())
pub fn parse(q: &str) -> Result<Self, crate::error::InvalidQuery> {
if q.chars().count() > 1000 {
return Err(crate::error::InvalidQuery);
}
let clauses = tokens(q)
.map(|token| token.and_then(clause))
.collect::<Result<Vec<_>, _>>()?;
let (text, filters): (Vec<_>, Vec<_>) = clauses
.into_iter()
.partition(|clause| matches!(clause, Clause::Text(_)));
Self {
Ok(Self {
text: text
.into_iter()
.map(|clause| clause.value().to_owned())
.filter_map(|clause| match clause {
Clause::Text(text) => Some(text),
_ => None,
})
.collect(),
filters: filters
.into_iter()
.filter_map(|clause| match clause {
Clause::Field {
key,
exclude,
value,
} => Some(FieldFilter {
key,
pattern: value,
exclude,
}),
Clause::Text(_) => None,
Clause::Field(filter) => Some(filter),
_ => None,
})
.collect(),
}
})
}
}
enum Clause {
Text(String),
Field {
key: SearchKey,
exclude: bool,
value: String,
},
Field(FieldFilter),
}
impl Clause {
fn value(&self) -> &str {
match self {
Self::Text(value) | Self::Field { value, .. } => value,
}
}
}
/// Whitespace-separated tokens; a double-quoted stretch keeps its spaces, and an unclosed quote runs to the end.
fn tokens(q: &str) -> impl Iterator<Item = &str> {
fn tokens(q: &str) -> impl Iterator<Item = Result<&str, crate::error::InvalidQuery>> {
let mut rest = q;
std::iter::from_fn(move || {
rest = rest.trim_start();
@ -134,42 +142,63 @@ fn tokens(q: &str) -> impl Iterator<Item = &str> {
let mut quoted = false;
let end = rest
.char_indices()
.find(|&(_, char)| {
if char == '"' {
.find(|&(_, ch)| {
if ch == '"' {
quoted = !quoted;
}
!quoted && char.is_whitespace()
!quoted && ch.is_whitespace()
})
.map_or(rest.len(), |(index, _)| index);
let (token, tail) = rest.split_at(end);
rest = tail;
Some(token)
Some(if quoted {
Err(crate::error::InvalidQuery)
} else {
Ok(token)
})
})
}
fn unquote(raw: &str) -> String {
raw.strip_prefix('"')
.map(|inner| inner.strip_suffix('"').unwrap_or(inner))
.filter(|inner| !inner.contains('"'))
.unwrap_or(raw)
.to_owned()
fn value(raw: &str) -> Result<String, crate::error::InvalidQuery> {
let value = if raw.starts_with('"') && raw.ends_with('"') && raw.len() >= 2 {
&raw[1..raw.len() - 1]
} else {
raw
};
if value.is_empty() || value.contains('"') {
return Err(crate::error::InvalidQuery);
}
Ok(value.to_owned())
}
fn clause(raw: &str) -> Clause {
fn clause(raw: &str) -> Result<Clause, crate::error::InvalidQuery> {
if raw.starts_with('"') {
return value(raw).map(Clause::Text);
}
let (exclude, body) = raw
.strip_prefix('-')
.map_or((false, raw), |body| (true, body));
let field = body
.split_once(':')
.and_then(|(key, value)| SearchKey::parse(key).map(|key| (key, value)));
match field {
Some((key, value)) => Clause::Field {
key,
exclude,
value: unquote(value),
let Some((key, raw_value)) = body.split_once(':') else {
return value(raw).map(Clause::Text);
};
let key = SearchKey::parse(key).ok_or(crate::error::InvalidQuery)?;
let pattern = value(raw_value)?;
let pattern = match key {
SearchKey::Field(RunField::RootStatus) => match pattern.to_ascii_lowercase().as_str() {
"ok" | "error" | "unset" => pattern.to_ascii_lowercase(),
_ => return Err(crate::error::InvalidQuery),
},
None => Clause::Text(unquote(raw)),
}
SearchKey::Field(RunField::HasError) => match pattern.to_ascii_lowercase().as_str() {
"true" | "false" => pattern.to_ascii_lowercase(),
_ => return Err(crate::error::InvalidQuery),
},
_ => pattern,
};
Ok(Clause::Field(FieldFilter {
key,
pattern,
exclude,
}))
}
pub const MAX_HISTOGRAM_BUCKETS: u32 = 240;
@ -179,6 +208,7 @@ pub const MAX_RUN_VALUES: u32 = 100;
#[macro_rules_attribute::apply(response_type)]
#[derive(Clone, Debug, PartialEq)]
pub struct TraceHistogram {
pub window: crate::api::TraceQueryWindow,
pub buckets: Vec<HistogramBucket>,
}
@ -204,14 +234,23 @@ pub struct AgentRuns {
#[macro_rules_attribute::apply(response_type)]
#[derive(Clone, Debug, PartialEq)]
pub struct RunValues {
pub window: crate::api::TraceQueryWindow,
pub values: Vec<String>,
}
/// Bucket `i` covers `[start + span * i / buckets, start + span * (i + 1) / buckets)`.
pub fn histogram(rows: &[RunCount], start_ms: i64, end_ms: i64, buckets: u32) -> TraceHistogram {
let span = i128::from(end_ms - start_ms);
let edge = |index: u32| start_ms + (span * i128::from(index) / i128::from(buckets)) as i64;
pub fn histogram(
rows: &[RunCount],
window: crate::api::TraceQueryWindow,
buckets: u32,
) -> TraceHistogram {
let start_ms = window.start_ms;
let end_ms = window.end_ms;
let start = i128::from(start_ms);
let span = i128::from(end_ms) - start;
let edge = |index: u32| (start + span * i128::from(index) / i128::from(buckets)) as i64;
TraceHistogram {
window,
buckets: (0..buckets)
.map(|index| {
let hits = rows.iter().filter(|row| row.bucket == index);

View file

@ -11,6 +11,7 @@ pub enum RunSelection {
Matching(RunFilter),
/// Every run with this trace id, whenever it happened.
TraceId(String),
TraceRef(String),
}
/// The last row of a page in its order: the row's sort value and its reference.
@ -328,6 +329,7 @@ pub struct SpanText {
/// is listed, oldest first by `(team_id, start_ms, request_id)`.
#[derive(Clone, Debug, PartialEq)]
pub struct CallQuery {
pub as_of_ms: u64,
pub window: Range<i64>,
/// Also matches the upstream id a managed `resp_` id wraps.
pub response_ids: Vec<String>,

View file

@ -2,7 +2,7 @@
use std::collections::BTreeMap;
use crate::ui::UiContent;
use crate::api::TraceQueryWindow;
#[macro_rules_attribute::apply(wire_type)]
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
@ -55,21 +55,20 @@ pub struct AgentNode {
#[macro_rules_attribute::apply(response_type)]
#[derive(Clone, Debug, PartialEq)]
pub struct TraceSummary {
#[cfg_attr(feature = "schema", schemars(extend("x-python-optional" = true)))]
pub resolution_limited: bool,
pub trace_id: String,
#[cfg_attr(feature = "schema", schemars(extend("x-python-optional" = true)))]
#[serde(rename = "id")]
pub trace_ref: String,
pub name: String,
pub service: String,
#[cfg_attr(feature = "schema", schemars(extend("x-python-optional" = true)))]
pub agent_names: Vec<String>,
#[cfg_attr(feature = "schema", schemars(extend("x-python-optional" = true)))]
pub frameworks: Vec<String>,
pub input_preview: String,
pub start_time: String,
pub duration_ms: f64,
#[serde(rename = "root_status")]
pub status: SpanStatus,
pub has_error: bool,
pub span_count: u64,
pub agent_count: u64,
pub agent_invocations: u64,
@ -95,6 +94,7 @@ pub struct Trace {
#[macro_rules_attribute::apply(response_type)]
#[derive(Debug, PartialEq)]
pub struct TracePage {
pub window: TraceQueryWindow,
pub data: Vec<TraceSummary>,
pub next_cursor: Option<String>,
}
@ -103,8 +103,6 @@ pub struct TracePage {
#[derive(Debug, PartialEq)]
pub struct SpanDetail {
pub span_id: String,
pub input_ui: UiContent,
pub output_ui: UiContent,
pub input: String,
pub output: String,
pub attributes: BTreeMap<String, String>,

View file

@ -0,0 +1,109 @@
#![cfg(feature = "schema")]
use litellm_traces::{
api::{
TraceHistogramRequest, TraceListRequest, TraceNoQueryRequest, TraceQueryRequest,
TraceSQLResponse, TraceSpanPageRequest, TraceValuesRequest,
},
schema,
};
use rstest::rstest;
use serde_json::{Value, json};
#[rstest]
#[case::default(json!({}), true)]
#[case::one(json!({"page_size":1}), true)]
#[case::maximum(json!({"page_size":500}), true)]
#[case::zero(json!({"page_size":0}), false)]
#[case::over_maximum(json!({"page_size":501}), false)]
#[case::wrong_sort(json!({"sort_by":"spend"}), false)]
#[case::unknown_field(json!({"limit":10}), false)]
#[case::long_query(json!({"q":"x".repeat(1001)}), false)]
#[case::long_cursor(json!({"cursor":"x".repeat(513)}), false)]
fn list_contract_validates_the_same_request_in_rust_and_json_schema(
#[case] input: Value,
#[case] valid: bool,
) {
let schemas = schema::api_schemas();
assert_eq!(
serde_json::from_value::<TraceListRequest>(input.clone()).is_ok(),
valid
);
assert_eq!(
jsonschema::is_valid(schemas["TraceListRequest"].as_value(), &input),
valid
);
}
#[rstest]
#[case::histogram_zero("TraceHistogramRequest",json!({"buckets":0}),false)]
#[case::histogram_max("TraceHistogramRequest",json!({"buckets":240}),true)]
#[case::histogram_over_max("TraceHistogramRequest",json!({"buckets":241}),false)]
#[case::values_zero("TraceValuesRequest",json!({"limit":0}),false)]
#[case::values_max("TraceValuesRequest",json!({"limit":100}),true)]
#[case::values_over_max("TraceValuesRequest",json!({"limit":101}),false)]
fn aggregation_contract_validates_limits(
#[case] name: &str,
#[case] input: Value,
#[case] valid: bool,
) {
let decoded = match name {
"TraceHistogramRequest" => {
serde_json::from_value::<TraceHistogramRequest>(input.clone()).is_ok()
}
"TraceValuesRequest" => serde_json::from_value::<TraceValuesRequest>(input.clone()).is_ok(),
_ => unreachable!(),
};
let schemas = schema::api_schemas();
assert_eq!(decoded, valid);
assert_eq!(
jsonschema::is_valid(schemas[name].as_value(), &input),
valid
);
}
#[rstest]
fn sql_contract_preserves_native_bind_values_and_engine_response_fields() {
let request = json!({"sql":"SELECT {value:String}","params":{"value":"a'b", "flag":true, "ids":["one","two"],"missing":null,"large":u64::MAX}});
let parsed: TraceQueryRequest = serde_json::from_value(request.clone()).unwrap();
assert_eq!(serde_json::to_value(parsed).unwrap(), request);
let response = json!({"meta":[{"name":"x","type":"UInt64","source":"traces"}],"data":[{"x":u64::MAX.to_string(),"nested":{"flags":[true,null]}}],"rows":1,"statistics":{"elapsed":0.01,"rows_read":"1","bytes_read":20,"extra_stat":42},"extra_field":[1,2]});
let parsed: TraceSQLResponse = serde_json::from_value(response.clone()).unwrap();
assert_eq!(serde_json::to_value(parsed).unwrap(), response);
assert!(jsonschema::is_valid(
schema::api_schemas()["TraceSQLResponse"].as_value(),
&response
));
}
#[rstest]
#[case::spans_default("TraceSpanPageRequest",json!({}),true)]
#[case::spans_zero("TraceSpanPageRequest",json!({"page_size":0}),false)]
#[case::spans_maximum("TraceSpanPageRequest",json!({"page_size":500}),true)]
#[case::spans_over_maximum("TraceSpanPageRequest",json!({"page_size":501}),false)]
#[case::no_query("TraceNoQueryRequest",json!({}),true)]
#[case::old_trace_ref("TraceNoQueryRequest",json!({"trace_ref":"other"}),false)]
#[case::sql_empty("TraceQueryRequest",json!({"sql":""}),false)]
#[case::sql_unknown_field("TraceQueryRequest",json!({"sql":"SELECT 1","cursor":"x"}),false)]
fn collection_and_sql_contracts_validate_requests(
#[case] name: &str,
#[case] input: Value,
#[case] valid: bool,
) {
let decoded = match name {
"TraceSpanPageRequest" => {
serde_json::from_value::<TraceSpanPageRequest>(input.clone()).is_ok()
}
"TraceNoQueryRequest" => {
serde_json::from_value::<TraceNoQueryRequest>(input.clone()).is_ok()
}
"TraceQueryRequest" => serde_json::from_value::<TraceQueryRequest>(input.clone()).is_ok(),
_ => unreachable!(),
};
let schemas = schema::api_schemas();
assert_eq!(decoded, valid);
assert_eq!(
jsonschema::is_valid(schemas[name].as_value(), &input),
valid
);
}

View file

@ -1328,3 +1328,27 @@ fn gateway_lookup_respects_legacy_fallback_and_ownership(
expected
);
}
#[rstest]
#[case::physical_roots(true)]
#[case::missing_physical_root(false)]
fn summary_root_is_deterministic_and_distinguishes_child_errors(#[case] physical_roots: bool) {
let parent = if physical_roots { "" } else { "missing" };
let rows = [
SpanRow {
status: SpanStatus::Error,
..row("z-root", parent, "later-root", "agent", "later")
},
row("a-root", parent, "canonical-root", "agent", "canonical"),
SpanRow {
status: SpanStatus::Error,
..at(row("orphan", "missing", "orphan", "llm", ""), -1, 1)
},
];
let expected = if physical_roots { &rows[1] } else { &rows[2] };
let trace = resolve_trace("trace", "ref", &rows, &[]).unwrap();
assert_eq!(trace.summary.name, expected.name);
assert_eq!(trace.summary.input_preview, expected.input_preview);
assert_eq!(trace.summary.status, expected.status);
assert!(trace.summary.has_error);
}

View file

@ -16,30 +16,24 @@ fn filter(field: RunField, pattern: &str, exclude: bool) -> FieldFilter {
#[case::empty("", &[], vec![])]
#[case::words("foo bar", &["foo", "bar"], vec![])]
#[case::quoted_phrase(r#""foo bar""#, &["foo bar"], vec![])]
#[case::unclosed_quote(r#""foo bar"#, &["foo bar"], vec![])]
#[case::inner_quote_kept(r#"a"b"c"#, &[r#"a"b"c"#], vec![])]
#[case::like_metacharacters_stay_literal("50%_off\\", &["50%_off\\"], vec![])]
#[case::exact(r#"name:"plan trip""#, &[], vec![filter(RunField::Name, "plan trip", false)])]
#[case::glob("agent:res*er", &[], vec![filter(RunField::Agent, "res*er", false)])]
#[case::glob_keeps_the_rest("model:gpt_4*", &[], vec![filter(RunField::Model, "gpt_4*", false)])]
#[case::negated("-status:error", &[], vec![filter(RunField::Status, "error", true)])]
#[case::negated("-has_error:true", &[], vec![filter(RunField::HasError, "true", true)])]
#[case::key_ignores_case("Trace_ID:abc", &[], vec![filter(RunField::TraceId, "abc", false)])]
#[case::value_keeps_colons("input:a:b", &[], vec![filter(RunField::Input, "a:b", false)])]
#[case::missing_value_narrows_nothing("status: foo", &["foo"], vec![])]
#[case::unknown_key_is_text("color:red", &["color:red"], vec![])]
#[case::non_word_key_is_text("k1:v", &["k1:v"], vec![])]
#[case::negated_text_stays_text("-foo", &["-foo"], vec![])]
#[case::service("service:billing", &[], vec![filter(RunField::Service, "billing", false)])]
#[case::team("-team:acme", &[], vec![filter(RunField::Team, "acme", true)])]
#[case::attribute("attr.gen_ai.system:openai", &[], vec![FieldFilter { key: SearchKey::Attribute("gen_ai.system".into()), pattern: "openai".into(), exclude: false }])]
#[case::attribute_without_a_key_is_text("attr.:x", &["attr.:x"], vec![])]
fn parse_matches_the_dashboard_search_grammar(
#[case] q: &str,
#[case] text: &[&str],
#[case] filters: Vec<FieldFilter>,
) {
assert_eq!(
RunSearch::parse(q),
RunSearch::parse(q).unwrap(),
RunSearch {
text: text.iter().map(|term| (*term).to_owned()).collect(),
filters,
@ -66,8 +60,11 @@ fn histogram_fills_every_bucket_and_splits_failures_from_agents() {
row(0, true, "c", 1),
row(2, true, "a", 1),
],
100,
110,
litellm_traces::api::TraceQueryWindow {
start_ms: 100,
end_ms: 110,
as_of_ms: 120,
},
3,
);
let summary: Vec<_> = shaped
@ -94,3 +91,60 @@ fn histogram_fills_every_bucket_and_splits_failures_from_agents() {
);
assert!(shaped.buckets[2].agents.is_empty());
}
#[rstest]
#[case::unknown("color:red")]
#[case::deprecated_status("status:error")]
#[case::missing_value("root_status:")]
#[case::invalid_status("root_status:success")]
#[case::invalid_boolean("has_error:yes")]
#[case::unclosed_quote("name:\"hello")]
#[case::embedded_quote("a\"b\"c")]
#[case::empty_attribute("attr.:x")]
fn invalid_search_is_rejected(#[case] q: &str) {
assert!(RunSearch::parse(q).is_err());
}
#[rstest]
fn oversized_search_is_rejected() {
assert!(RunSearch::parse(&"x".repeat(1001)).is_err());
}
#[rstest]
#[case::full_range(i64::MIN, i64::MAX)]
#[case::minimum_to_zero(i64::MIN, 0)]
#[case::negative_to_maximum(-1, i64::MAX)]
fn histogram_extreme_windows_preserve_edges_and_totals(#[case] start_ms: i64, #[case] end_ms: i64) {
let window = litellm_traces::api::TraceQueryWindow {
start_ms,
end_ms,
as_of_ms: 1,
};
let result = histogram(&[row(0, false, "a", 2), row(1, true, "b", 3)], window, 2);
assert_eq!(result.window, window);
assert_eq!(result.buckets[0].start_ms, start_ms);
assert_eq!(result.buckets[1].end_ms, end_ms);
assert_eq!(result.buckets[0].end_ms, result.buckets[1].start_ms);
assert!(
result
.buckets
.iter()
.all(|bucket| bucket.start_ms < bucket.end_ms)
);
assert_eq!(
result
.buckets
.iter()
.map(|bucket| bucket.total)
.sum::<u64>(),
5
);
assert_eq!(
result
.buckets
.iter()
.map(|bucket| bucket.failed)
.sum::<u64>(),
3
);
}

View file

@ -91,7 +91,7 @@ def lens_access(scope: Scope) -> QueryScope:
def execution_of(run: TraceSummary) -> Execution:
trace_ref: Final = run.get("trace_ref", "")
trace_ref: Final = run["id"]
return Execution(
id=execution_id(trace_ref, run["trace_id"]), trace_id=run["trace_id"], trace_ref=trace_ref, summary=run
)
@ -198,7 +198,7 @@ class SourceReader:
span_ids,
part,
access,
max_chars=max_chars if not tail else budget - budget // 3,
max_chars=max_chars if not tail or max_chars else budget - budget // 3,
tail=tail,
),
)
@ -315,7 +315,37 @@ class SourceReader:
)
for part, _, _ in PARTS
]
return any(text["contains"] for text in chain.from_iterable(found))
if any(text["contains"] for text in chain.from_iterable(found)):
return True
if len(evidence.quote) > BUDGET:
return False
trace: Final = await self.storage.get_trace(execution.trace_id, access, execution.trace_ref)
if trace is None:
return False
span: Final = next((span for span in trace["spans"] if span["span_id"] == evidence.span_id), None)
if span is None:
return False
heads: Final = await self._texts(access, execution, (evidence.span_id,), BUDGET)
pieces: Final = _pieces(span, heads)
tails: Final = await self._texts(
access, execution, (evidence.span_id,) if _total(pieces) > BUDGET else (), BUDGET, tail=True
)
if evidence.quote in _excerpt(span, pieces, tails):
return True
if _total(pieces) <= BUDGET:
return False
prefixes: Final = tuple(label + (text["text"] if text else "") for _, label, text in pieces)
endings: Final = tuple(
(label if text is None or text["total_chars"] <= BUDGET else "")
+ (tail["text"] if (tail := tails.get((evidence.span_id, part))) else "")
for part, label, text in pieces
)
boundaries: Final = tuple(end + prefix for end, prefix in zip(endings, prefixes[1:]))
middle: Final = pieces[1][2]
joined: Final = (
endings[0] + prefixes[1] + prefixes[2] if middle is None or middle["total_chars"] <= BUDGET else ""
)
return evidence.quote in joined or any(evidence.quote in boundary for boundary in boundaries)
def _excerpt(span: Span, pieces: tuple[tuple[SpanPart, str, SpanText | None], ...], tails: Texts) -> str:
@ -332,4 +362,4 @@ def _shortened(text: SpanText | None, budget: int, tail: SpanText | None) -> str
return ""
if text["total_chars"] <= budget:
return text["text"]
return text["text"][: budget // 3] + OMITTED + (tail["text"] if tail else "")
return text["text"][: budget // 3] + OMITTED + (tail["text"][-(budget - budget // 3) :] if tail else "")

View file

@ -1857,7 +1857,7 @@ def get_openapi_schema():
if server_root_path:
openapi_schema["servers"] = [{"url": "/" + server_root_path.strip("/")}]
app.openapi_schema = openapi_schema
app.openapi_schema = dict(tracing_endpoints.merge_trace_openapi(openapi_schema))
return app.openapi_schema

View file

@ -1,29 +1,24 @@
"""
Agent tracing endpoints. Thin wrappers over `TraceReceiver`: auth -> tenant/scope -> one call.
"""Authenticated OTLP ingestion and trace query routes."""
POST /v1/traces OTLP/HTTP trace export (protobuf or JSON)
GET /v1/traces TracePage
GET /v1/traces/histogram TraceHistogram
GET /v1/traces/values/{field} RunValues
GET /v1/traces/{trace_id} Trace
GET /v1/traces/{trace_id}/spans/{span_id} SpanDetail
"""
import time
from collections.abc import Mapping
from collections.abc import Callable, Coroutine, Mapping
from dataclasses import dataclass
from functools import partial
from http.client import responses
from pathlib import Path
from types import MappingProxyType
from typing import Annotated, Final, Literal
from typing import Annotated, Final
from fastapi import APIRouter, Depends, HTTPException, Query, Request, Response
from pydantic import BaseModel, ConfigDict
from typing_extensions import assert_never
from fastapi.exceptions import RequestValidationError
from fastapi.routing import APIRoute
from pydantic import JsonValue, TypeAdapter, ValidationError
from starlette.exceptions import HTTPException as StarletteHTTPException
from starlette.responses import JSONResponse
from typing_extensions import ReadOnly, TypedDict, assert_never
from litellm._logging import verbose_proxy_logger
from litellm.constants import OTLP_RETRY_AFTER_SECONDS, TRACE_READ_RETRY_AFTER_SECONDS
from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy._types import LitellmUserRoles, ProxyException, UserAPIKeyAuth
from litellm.proxy.auth.authorization import AllRows, ReadScope, resolve_trace_read_scope
from litellm.proxy.auth.authorization_dependencies import LogTeamLookupDependency
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
@ -31,6 +26,21 @@ from litellm.proxy.common_utils.http_parsing_utils import is_otlp_trace_request
from litellm.proxy.tracing_runtime import provide_receiver, require_receiver
from litellm.rust_bridge.trace.errors import TraceChanged, TraceQueryError
from litellm.rust_bridge.trace.generated.models import TraceQueryHelp
from litellm.rust_bridge.trace.generated.requests import (
TraceErrorPageRequest,
TraceHistogramRequest,
TraceInvalidParam,
TraceListRequest,
TraceMetadata,
TraceNoQueryRequest,
TraceProblem,
TraceProblemCode,
TraceQueryRequest,
TraceSpanPageRequest,
TraceSpansPage,
TraceValuesRequest,
)
from litellm.rust_bridge.trace.generated.routes import OPERATIONS
from litellm.rust_bridge.trace.generated.types import (
AllQueryScope,
OwnedQueryScope,
@ -40,7 +50,6 @@ from litellm.rust_bridge.trace.generated.types import (
RunValues,
SpanDetail,
SpanErrorPage,
Trace,
TraceHistogram,
TracePage,
)
@ -49,9 +58,139 @@ from litellm.rust_bridge.trace.storage import ClickHouseStorage, Tenant
from litellm.tracing import TraceReceiver, TracingPayloadTooLargeError
from litellm.tracing.otlp_http import InvalidOTLPPayloadError, encode_otlp_response
router = APIRouter(tags=["agent tracing"])
_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue])
MS_PER_DAY: Final = 24 * 60 * 60 * 1000
class ValidationIssue(TypedDict):
loc: ReadOnly[tuple[str | int, ...]]
msg: ReadOnly[str]
_VALIDATION_ISSUES: Final = TypeAdapter(tuple[ValidationIssue, ...])
def problem(
status: int,
code: TraceProblemCode,
detail: str,
database_code: int | None = None,
errors: tuple[TraceInvalidParam, ...] = (),
) -> TraceProblem:
return TraceProblem(
type="about:blank",
title=responses.get(status, "Trace request failed"),
status=status,
code=code,
detail=detail,
database_code=database_code,
errors=errors,
)
def problem_exception(
status: int, code: TraceProblemCode, detail: str, database_code: int | None = None
) -> HTTPException:
return HTTPException(status_code=status, detail=problem(status, code, detail, database_code).model_dump())
def status_code(status: int) -> TraceProblemCode:
match status:
case 401:
return "unauthorized"
case 403:
return "forbidden"
case 404:
return "not_found"
case 413:
return "too_large"
case 500:
return "internal_error"
case 501 | 503:
return "unavailable"
case _:
return "invalid_request"
def problem_response(body: TraceProblem, headers: Mapping[str, str] | None = None) -> Response:
return JSONResponse(
body.model_dump(mode="json", exclude_none=True),
status_code=body.status,
media_type="application/problem+json",
headers=headers,
)
def http_problem(error: StarletteHTTPException) -> Response:
try:
body: Final = TraceProblem.model_validate(error.detail)
return problem_response(body, error.headers)
except ValidationError:
return problem_response(
problem(error.status_code, status_code(error.status_code), str(error.detail)), error.headers
)
class TraceRoute(APIRoute):
def get_route_handler(self) -> Callable[[Request], Coroutine[None, None, Response]]:
handler: Final = super().get_route_handler()
async def handle(request: Request) -> Response:
if is_otlp_trace_request(request):
return await handler(request)
try:
return await handler(request)
except StarletteHTTPException as error:
return http_problem(error)
except ProxyException as error:
status: Final = int(error.code) if error.code.isdecimal() else 500
return problem_response(problem(status, status_code(status), error.message), error.headers)
except RequestValidationError as error:
issues: Final = _VALIDATION_ISSUES.validate_python(error.errors())
details: Final = tuple(
TraceInvalidParam(
location="/".join(str(part) for part in issue["loc"]),
reason=str(issue["msg"]),
)
for issue in issues
)
return problem_response(problem(422, "invalid_request", "Request validation failed", errors=details))
except Exception as error:
verbose_proxy_logger.exception("Trace request failed: %s", error)
return problem_response(problem(500, "internal_error", "Trace request failed"))
return handle
router = APIRouter(tags=["agent tracing"], route_class=TraceRoute)
def merge_trace_openapi(schema: object) -> Mapping[str, JsonValue]:
document: Final = _JSON_OBJECT.validate_python(schema)
generated: Final = _JSON_OBJECT.validate_json(
(Path(__file__).parents[1] / "rust_bridge" / "trace" / "generated" / "openapi.json").read_text()
)
components: Final = _JSON_OBJECT.validate_python(document.get("components", {}))
generated_components: Final = _JSON_OBJECT.validate_python(generated.get("components", {}))
return {
**document,
"openapi": generated["openapi"],
"paths": {
**_JSON_OBJECT.validate_python(document.get("paths", {})),
**_JSON_OBJECT.validate_python(generated["paths"]),
},
"components": {
**components,
**generated_components,
"schemas": {
**_JSON_OBJECT.validate_python(components.get("schemas", {})),
**_JSON_OBJECT.validate_python(generated_components.get("schemas", {})),
},
"securitySchemes": {
**_JSON_OBJECT.validate_python(components.get("securitySchemes", {})),
**_JSON_OBJECT.validate_python(generated_components.get("securitySchemes", {})),
},
},
}
@dataclass(frozen=True, slots=True)
@ -63,7 +202,7 @@ class TraceAccessContext:
def reader(self) -> tuple[TraceReceiver, QueryScope]:
tracing: Final = require_receiver(self.receiver)
if self.read_scope is None:
raise HTTPException(status_code=403, detail="Not allowed to view agent traces")
raise problem_exception(403, "forbidden", "Not allowed to view agent traces")
return tracing, read_access(self.read_scope)
def writer(self) -> tuple[TraceReceiver, Tenant]:
@ -96,17 +235,22 @@ def otlp_error_response(
return Response(content=body, status_code=status_code, media_type=media_type, headers=headers)
def _otlp_error(content_type: str | None, status_code: int, message: str, retry: bool = False) -> Response:
def _otlp_error(content_type: str | None, status: int, message: str, retry: bool = False) -> Response:
body, media_type = encode_otlp_response(content_type, message)
return Response(
content=body,
status_code=status_code,
status_code=status,
media_type=media_type,
headers=MappingProxyType({"Retry-After": str(OTLP_RETRY_AFTER_SECONDS)}) if retry else None,
)
@router.post("/v1/traces", include_in_schema=False)
@router.api_route(
OPERATIONS["trace_ingest"].path,
methods=[OPERATIONS["trace_ingest"].method],
operation_id=OPERATIONS["trace_ingest"].operation_id,
include_in_schema=False,
)
async def ingest_otlp_traces(
request: Request,
context: Annotated[TraceAccessContext, Depends(provide_trace_access)],
@ -120,8 +264,8 @@ async def ingest_otlp_traces(
content_encoding=request.headers.get("content-encoding"),
tenant=tenant,
)
except TracingPayloadTooLargeError as e:
return _otlp_error(content_type, 413, str(e))
except TracingPayloadTooLargeError as error:
return _otlp_error(content_type, 413, str(error))
except InvalidOTLPPayloadError as error:
return _otlp_error(content_type, 400, str(error))
except RuntimeError:
@ -132,42 +276,20 @@ async def ingest_otlp_traces(
return Response(content=body, media_type=media_type)
class TraceReadFailure(BaseModel):
"""The body of every failed trace read. Clients branch on `code`, never on `message`."""
model_config = ConfigDict(frozen=True)
code: Literal["invalid_request", "trace_changed", "too_large", "unavailable"]
message: str
def read_failure(error: TraceChanged | ValueError | OverflowError | RuntimeError) -> HTTPException:
"""One status per failure kind, so a client can tell a bad cursor (400, fix the request) from a
traversal it must restart (409), a result it cannot page through (413), and an outage it should
retry after `Retry-After` (503)."""
match error:
case TraceChanged():
return HTTPException(
status_code=409,
detail=TraceReadFailure(code="trace_changed", message=str(error)).model_dump(),
)
return problem_exception(409, "trace_changed", str(error))
case ValueError():
return HTTPException(
status_code=400, detail=TraceReadFailure(code="invalid_request", message=str(error)).model_dump()
)
return problem_exception(400, "invalid_request", str(error))
case OverflowError():
return HTTPException(
status_code=413,
detail=TraceReadFailure(
code="too_large", message="Trace is too large for this view. Use a filtered trace query."
).model_dump(),
)
return problem_exception(413, "too_large", "Trace is too large for this view. Use a filtered trace query.")
case RuntimeError():
verbose_proxy_logger.warning("Trace read unavailable: %s", error)
return HTTPException(
status_code=503,
detail=TraceReadFailure(
code="unavailable", message="Traces are temporarily unavailable. Please try again."
detail=problem(
503, "unavailable", "Traces are temporarily unavailable. Please try again."
).model_dump(),
headers={"Retry-After": str(TRACE_READ_RETRY_AFTER_SECONDS)},
)
@ -175,86 +297,72 @@ def read_failure(error: TraceChanged | ValueError | OverflowError | RuntimeError
return assert_never(error)
StartMs = Annotated[int | None, Query(description="Window start, unix ms. Default: 24h ago")]
EndMs = Annotated[int | None, Query(description="Window end, unix ms. Default: now")]
RunQuery = Annotated[
str,
Query(
max_length=1000,
description='Free text and key:value filters, e.g. `agent:research* -status:ok "book a flight"`. '
"Keys: name, agent, status, model, input, trace_id, service, team and attr.<key>. "
"`*` globs and a leading `-` negates",
),
]
@dataclass(frozen=True, slots=True)
class TraceWindow:
start_ms: int
end_ms: int
def trace_window(start_ms: StartMs = None, end_ms: EndMs = None) -> TraceWindow:
now_ms: Final = int(time.time() * 1000)
return TraceWindow(
start_ms=start_ms if start_ms is not None else now_ms - MS_PER_DAY,
end_ms=end_ms if end_ms is not None else now_ms,
)
@router.get("/v1/traces", response_model=TracePage)
@router.api_route(
OPERATIONS["trace_list"].path,
methods=[OPERATIONS["trace_list"].method],
operation_id=OPERATIONS["trace_list"].operation_id,
response_model=TracePage,
)
async def list_agent_traces(
params: Annotated[TraceListRequest, Query()],
context: Annotated[TraceAccessContext, Depends(provide_trace_access)],
start_ms: StartMs = None,
end_ms: EndMs = None,
q: RunQuery = "",
cursor: Annotated[str | None, Query(max_length=512)] = None,
sort_by: Literal["start_ms", "duration_ms", "span_count", "error_count"] = "start_ms",
sort_dir: Literal["asc", "desc"] = "desc",
) -> TracePage:
order: Final = RunOrder(key=sort_by, descending=sort_dir == "desc")
order: Final = RunOrder(key=params.sort_by, descending=params.sort_dir == "desc")
try:
tracing, scope = context.reader()
return await tracing.list_traces(scope=scope, start_ms=start_ms, end_ms=end_ms, q=q, cursor=cursor, order=order)
return await tracing.list_traces(
scope=scope,
start_ms=params.start_ms,
end_ms=params.end_ms,
q=params.q,
cursor=params.cursor,
order=order,
page_size=params.page_size,
as_of_ms=params.as_of_ms,
)
except (TraceChanged, ValueError, OverflowError, RuntimeError) as error:
raise read_failure(error) from error
@router.get("/v1/traces/histogram", response_model=TraceHistogram)
@router.api_route(
OPERATIONS["trace_histogram"].path,
methods=[OPERATIONS["trace_histogram"].method],
operation_id=OPERATIONS["trace_histogram"].operation_id,
response_model=TraceHistogram,
)
async def agent_trace_histogram(
params: Annotated[TraceHistogramRequest, Query()],
context: Annotated[TraceAccessContext, Depends(provide_trace_access)],
window: Annotated[TraceWindow, Depends(trace_window)],
q: RunQuery = "",
buckets: Annotated[int, Query(ge=1, le=240)] = 60,
) -> TraceHistogram:
try:
tracing, scope = context.reader()
return await tracing.trace_histogram(scope, window.start_ms, window.end_ms, q, buckets)
return await tracing.trace_histogram(
scope, params.start_ms, params.end_ms, params.q, params.buckets, params.as_of_ms
)
except (ValueError, OverflowError, RuntimeError) as error:
raise read_failure(error) from error
@router.get("/v1/traces/values/{field}", response_model=RunValues)
@router.api_route(
OPERATIONS["trace_values"].path,
methods=[OPERATIONS["trace_values"].method],
operation_id=OPERATIONS["trace_values"].operation_id,
response_model=RunValues,
)
async def agent_trace_values(
field: RunField,
params: Annotated[TraceValuesRequest, Query()],
context: Annotated[TraceAccessContext, Depends(provide_trace_access)],
window: Annotated[TraceWindow, Depends(trace_window)],
q: RunQuery = "",
contains: Annotated[str, Query(max_length=200)] = "",
limit: Annotated[int, Query(ge=1, le=100)] = 20,
) -> RunValues:
try:
tracing, scope = context.reader()
return await tracing.run_values(scope, window.start_ms, window.end_ms, q, field, contains, limit)
return await tracing.run_values(
scope, params.start_ms, params.end_ms, params.q, field, params.contains, params.limit, params.as_of_ms
)
except (ValueError, OverflowError, RuntimeError) as error:
raise read_failure(error) from error
class TraceQueryRequest(BaseModel):
model_config = ConfigDict(frozen=True, extra="forbid")
sql: str
@dataclass(frozen=True, slots=True)
class TraceQueryAccess:
storage: ClickHouseStorage
@ -266,18 +374,14 @@ def provide_trace_query_secret() -> str:
from litellm.proxy.proxy_server import master_key
if not master_key:
raise HTTPException(status_code=503, detail="Trace SQL queries require a configured proxy master key")
raise problem_exception(503, "query_unavailable", "Trace SQL queries require a configured proxy master key")
return master_key
def read_access(scope: ReadScope) -> QueryScope:
if isinstance(scope, AllRows):
return AllQueryScope(kind="all")
return OwnedQueryScope(
kind="owned",
user_id=scope.user_id or "",
team_ids=scope.team_ids,
)
return OwnedQueryScope(kind="owned", user_id=scope.user_id or "", team_ids=scope.team_ids)
async def provide_trace_query_access(
@ -289,11 +393,11 @@ async def provide_trace_query_access(
storage: Final = require_receiver(tracing).storage
scope: Final = await resolve_trace_read_scope(auth, partial(log_team_lookup, auth))
if scope is None:
raise HTTPException(status_code=403, detail="Not allowed to view logs")
raise problem_exception(403, "forbidden", "Not allowed to view logs")
return TraceQueryAccess(storage, scope, secret)
def sql_failure_status(error: TraceQueryError) -> tuple[int, str]:
def sql_failure_status(error: TraceQueryError) -> tuple[int, TraceProblemCode]:
match error.kind:
case "rejected":
return 400, "query_rejected"
@ -303,95 +407,129 @@ def sql_failure_status(error: TraceQueryError) -> tuple[int, str]:
return 503, "query_unavailable"
@router.post("/v1/traces/query", response_model=TraceSQLResponse, response_model_exclude_unset=True)
@router.api_route(
OPERATIONS["trace_query"].path,
methods=[OPERATIONS["trace_query"].method],
operation_id=OPERATIONS["trace_query"].operation_id,
response_model=TraceSQLResponse,
response_model_exclude_unset=True,
)
async def query_agent_traces(
_params: Annotated[TraceNoQueryRequest, Query()],
body: TraceQueryRequest,
access: Annotated[TraceQueryAccess, Depends(provide_trace_query_access)],
) -> TraceSQLResponse:
try:
return await access.storage.query_sql(body.sql, read_access(access.scope), access.secret)
return await access.storage.query_sql(body.sql, read_access(access.scope), access.secret, body.params)
except TraceQueryError as error:
status, code = sql_failure_status(error)
raise HTTPException(
status_code=status,
detail={"code": code, "database_code": error.database_code, "message": error.message},
) from error
raise problem_exception(status, code, error.message, error.database_code) from error
except ValueError as error:
raise HTTPException(
status_code=400,
detail={"code": "query_rejected", "database_code": None, "message": str(error)},
) from error
raise problem_exception(400, "query_rejected", str(error)) from error
except RuntimeError as error:
verbose_proxy_logger.warning("Trace SQL query unavailable: %s", error)
raise HTTPException(
status_code=503,
detail={
"code": "query_unavailable",
"database_code": None,
"message": "Trace SQL is temporarily unavailable",
},
) from error
raise problem_exception(503, "query_unavailable", "Trace SQL is temporarily unavailable") from error
@router.get("/v1/traces/query/help", response_model=TraceQueryHelp, response_model_exclude_unset=True)
@router.api_route(
OPERATIONS["trace_query_help"].path,
methods=[OPERATIONS["trace_query_help"].method],
operation_id=OPERATIONS["trace_query_help"].operation_id,
response_model=TraceQueryHelp,
response_model_exclude_unset=True,
)
async def help_agent_trace_queries(
_params: Annotated[TraceNoQueryRequest, Query()],
access: Annotated[TraceQueryAccess, Depends(provide_trace_query_access)],
) -> TraceQueryHelp:
try:
return await access.storage.query_help(read_access(access.scope), access.secret)
except RuntimeError as error:
verbose_proxy_logger.warning("Trace query help unavailable: %s", error)
raise HTTPException(status_code=503, detail="Trace query help is temporarily unavailable") from error
raise problem_exception(503, "query_unavailable", "Trace query help is temporarily unavailable") from error
@router.get("/v1/traces/{trace_id}", response_model=Trace)
@router.api_route(
OPERATIONS["trace_get"].path,
methods=[OPERATIONS["trace_get"].method],
operation_id=OPERATIONS["trace_get"].operation_id,
response_model=TraceMetadata,
)
async def get_agent_trace(
trace_id: str,
_params: Annotated[TraceNoQueryRequest, Query()],
id: str,
context: Annotated[TraceAccessContext, Depends(provide_trace_access)],
trace_ref: Annotated[str, Query()] = "",
cursor: Annotated[str | None, Query(max_length=512)] = None,
page_size: Annotated[int | None, Query(ge=1, le=500)] = None,
) -> Trace:
tracing, scope = context.reader()
) -> TraceMetadata:
try:
trace: Final = await tracing.get_trace(trace_id, scope, trace_ref, cursor, page_size)
tracing, scope = context.reader()
trace: Final = await tracing.get_trace_metadata(scope, id)
except (TraceChanged, ValueError, OverflowError, RuntimeError) as error:
raise read_failure(error) from error
if trace is None:
raise HTTPException(status_code=404, detail=f"Trace {trace_id} not found")
raise problem_exception(404, "not_found", "Trace not found")
return trace
@router.get("/v1/traces/{trace_id}/spans/{span_id}", response_model=SpanDetail)
async def get_agent_trace_span(
trace_id: str,
span_id: str,
@router.api_route(
OPERATIONS["trace_spans"].path,
methods=[OPERATIONS["trace_spans"].method],
operation_id=OPERATIONS["trace_spans"].operation_id,
response_model=TraceSpansPage,
)
async def get_agent_trace_spans(
id: str,
params: Annotated[TraceSpanPageRequest, Query()],
context: Annotated[TraceAccessContext, Depends(provide_trace_access)],
trace_ref: Annotated[str, Query()] = "",
) -> SpanDetail:
tracing, scope = context.reader()
try:
span: Final = await tracing.get_span(trace_id, span_id, scope, trace_ref)
except (TraceChanged, ValueError, OverflowError, RuntimeError) as error:
raise read_failure(error) from error
if span is None:
raise HTTPException(status_code=404, detail=f"Span {span_id} not found")
return span
@router.get("/v1/traces/{trace_id}/spans/{span_id}/error", response_model=SpanErrorPage)
async def get_agent_trace_span_error(
trace_id: str,
span_id: str,
context: Annotated[TraceAccessContext, Depends(provide_trace_access)],
trace_ref: Annotated[str, Query()] = "",
cursor: Annotated[str | None, Query(max_length=512)] = None,
) -> SpanErrorPage:
) -> TraceSpansPage:
try:
tracing, scope = context.reader()
page: Final = await tracing.get_span_error(trace_id, span_id, scope, trace_ref, cursor)
page: Final = await tracing.get_trace_spans(scope, id, params.cursor, params.page_size)
except (TraceChanged, ValueError, OverflowError, RuntimeError) as error:
raise read_failure(error) from error
if page is None:
raise HTTPException(status_code=404, detail="Span diagnostic not found or no longer available")
raise problem_exception(404, "not_found", "Trace not found")
return page
@router.api_route(
OPERATIONS["trace_span"].path,
methods=[OPERATIONS["trace_span"].method],
operation_id=OPERATIONS["trace_span"].operation_id,
response_model=SpanDetail,
)
async def get_agent_trace_span(
_params: Annotated[TraceNoQueryRequest, Query()],
id: str,
span_id: str,
context: Annotated[TraceAccessContext, Depends(provide_trace_access)],
) -> SpanDetail:
try:
tracing, scope = context.reader()
span: Final = await tracing.get_span_by_id(scope, id, span_id)
except (TraceChanged, ValueError, OverflowError, RuntimeError) as error:
raise read_failure(error) from error
if span is None:
raise problem_exception(404, "not_found", "Span not found")
return span
@router.api_route(
OPERATIONS["trace_error"].path,
methods=[OPERATIONS["trace_error"].method],
operation_id=OPERATIONS["trace_error"].operation_id,
response_model=SpanErrorPage,
)
async def get_agent_trace_span_error(
id: str,
span_id: str,
params: Annotated[TraceErrorPageRequest, Query()],
context: Annotated[TraceAccessContext, Depends(provide_trace_access)],
) -> SpanErrorPage:
try:
tracing, scope = context.reader()
page: Final = await tracing.get_span_error_by_id(scope, id, span_id, params.cursor)
except (TraceChanged, ValueError, OverflowError, RuntimeError) as error:
raise read_failure(error) from error
if page is None:
raise problem_exception(404, "not_found", "Span diagnostic not found or no longer available")
return page

View file

@ -11,6 +11,7 @@ from litellm.rust_bridge.embeddings.entrypoints import LiteLLMEmbeddingRequest
from litellm.rust_bridge.messages.entrypoints import LiteLLMMessagesRequest
from litellm.rust_bridge.ocr.entrypoints import LiteLLMOcrRequest
from litellm.rust_bridge.responses.entrypoints import LiteLLMResponsesRequest
from litellm.rust_bridge.trace.generated.requests import SqlParameter
from litellm.rust_bridge.trace.generated.types import QueryScope, RunOrder
from litellm.types.llms.anthropic_messages.anthropic_response import AnthropicMessagesResponse
from litellm.types.llms.openai import ResponsesAPIResponse
@ -52,9 +53,16 @@ class NativeTraceStorage:
limit: int,
order: RunOrder,
trace_refs: Sequence[str] = (),
as_of_ms: int | None = None,
) -> Future[JsonValue]: ...
def count_traces(
self, scope: QueryScope, start_ms: int, end_ms: int, q: str, trace_refs: Sequence[str] = ()
self,
scope: QueryScope,
start_ms: int | None,
end_ms: int | None,
q: str,
trace_refs: Sequence[str] = (),
as_of_ms: int | None = None,
) -> Future[JsonValue]: ...
def span_text(
self,
@ -69,10 +77,24 @@ class NativeTraceStorage:
contains: str | None = None,
) -> Future[JsonValue]: ...
def trace_histogram(
self, scope: QueryScope, start_ms: int, end_ms: int, q: str, buckets: int
self,
scope: QueryScope,
start_ms: int | None,
end_ms: int | None,
q: str,
buckets: int,
as_of_ms: int | None = None,
) -> Future[JsonValue]: ...
def run_values(
self, scope: QueryScope, start_ms: int, end_ms: int, q: str, field: str, contains: str, limit: int
self,
scope: QueryScope,
start_ms: int | None,
end_ms: int | None,
q: str,
field: str,
contains: str,
limit: int,
as_of_ms: int | None = None,
) -> Future[JsonValue]: ...
def get_trace(
self, trace_id: str, scope: QueryScope, trace_ref: str, cursor: str | None = None, page_size: int | None = None
@ -81,7 +103,15 @@ class NativeTraceStorage:
def get_span_error(
self, trace_id: str, span_id: str, scope: QueryScope, trace_ref: str, cursor: str | None
) -> Future[JsonValue]: ...
def query_sql(self, sql: str, scope: QueryScope, secret: str) -> Future[str]: ...
def get_trace_metadata(self, scope: QueryScope, id: str) -> Future[JsonValue]: ...
def get_trace_spans(self, scope: QueryScope, id: str, cursor: str | None, page_size: int) -> Future[JsonValue]: ...
def get_span_by_id(self, scope: QueryScope, id: str, span_id: str) -> Future[JsonValue]: ...
def get_span_error_by_id(
self, scope: QueryScope, id: str, span_id: str, cursor: str | None = None
) -> Future[JsonValue]: ...
def query_sql(
self, sql: str, scope: QueryScope, secret: str, params: Mapping[str, SqlParameter]
) -> Future[str]: ...
def query_help(self, scope: QueryScope, secret: str) -> Future[JsonValue]: ...
@final

View file

@ -6,10 +6,10 @@ from typing import Annotated, Literal, TypeAlias
from pydantic import BaseModel, ConfigDict, Field
TraceTableName: TypeAlias = Literal["otel_traces", "agent_traces_by_key", "spend_logs"]
TraceQueryTableName: TypeAlias = Literal["traces", "spans", "calls", "otel_traces", "agent_traces_by_key", "spend_logs"]
class TraceQueryColumn(BaseModel):
class TraceSQLColumn(BaseModel):
model_config = ConfigDict(
extra="allow",
frozen=True,
@ -25,7 +25,7 @@ class TraceQueryNormalizedField(BaseModel):
frozen=True,
)
table: TraceTableName
table: TraceQueryTableName
name: str
column: str
type: str
@ -71,8 +71,8 @@ class TraceQueryTable(BaseModel):
frozen=True,
)
name: TraceTableName
columns: tuple[TraceQueryColumn, ...]
name: TraceQueryTableName
columns: tuple[TraceSQLColumn, ...]
class TraceQueryMetadataField(BaseModel):
@ -103,7 +103,7 @@ class TraceQueryMetadata(BaseModel):
frozen=True,
)
table: TraceTableName
table: TraceQueryTableName
column: str
fields: tuple[TraceQueryMetadataField, ...]
sampled_rows: int = Field(..., ge=0, le=18446744073709551615)
@ -120,7 +120,7 @@ class TraceQueryAttributes(BaseModel):
frozen=True,
)
table: TraceTableName
table: TraceQueryTableName
column: str
fields: tuple[TraceQueryAttributeField, ...]
truncated: bool

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,301 @@
# @generated by scripts/generate_trace_types.py, do not edit
from __future__ import annotations
from collections.abc import Mapping
from typing import Annotated, Literal, TypeAlias
from pydantic import BaseModel, ConfigDict, Field, JsonValue
class TraceErrorPageRequest(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
cursor: str | None = Field(None, max_length=512)
class TraceHistogramRequest(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
start_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807)
end_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807)
as_of_ms: int | None = Field(None, ge=0, le=18446744073709551615)
q: str = Field("", max_length=1000)
buckets: int = Field(60, ge=1, le=240)
TraceSortField: TypeAlias = Literal["start_ms", "duration_ms", "span_count", "error_count"]
TraceSortDirection: TypeAlias = Literal["asc", "desc"]
class TraceListRequest(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
as_of_ms: int | None = Field(None, ge=0, le=18446744073709551615)
start_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807)
end_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807)
q: str = Field("", max_length=1000)
cursor: str | None = Field(None, max_length=512)
page_size: int = Field(50, ge=1, le=500)
sort_by: TraceSortField = "start_ms"
sort_dir: TraceSortDirection = "desc"
SpanStatus: TypeAlias = Literal["ok", "error", "unset"]
class AgentNode(BaseModel):
model_config = ConfigDict(
frozen=True,
)
name: str
parent_agent: str | None
invocations: int = Field(..., ge=0, le=18446744073709551615)
llm_calls: int = Field(..., ge=0, le=18446744073709551615)
tool_calls: int = Field(..., ge=0, le=18446744073709551615)
duration_ms: float
spend: float | None
class TraceNoQueryRequest(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
TraceProblemCode: TypeAlias = Literal[
"invalid_request",
"unauthorized",
"forbidden",
"not_found",
"trace_changed",
"too_large",
"unavailable",
"query_rejected",
"query_limit_exceeded",
"query_unavailable",
"internal_error",
]
class TraceInvalidParam(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
location: str
reason: str
class TraceProblem(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
type: str
title: str
status: int = Field(..., ge=0, le=65535)
detail: str
code: TraceProblemCode
database_code: int | None = Field(None, ge=0, le=4294967295)
errors: tuple[TraceInvalidParam, ...] = ()
SqlParameter1: TypeAlias = Annotated[int, Field(..., ge=-9223372036854775808, le=9223372036854775807)]
SqlParameter2: TypeAlias = Annotated[int, Field(..., ge=0, le=18446744073709551615)]
SqlParameter: TypeAlias = str | SqlParameter1 | SqlParameter2 | float | bool | list[str] | None
class TraceQueryRequest(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
sql: str = Field(..., min_length=1)
params: Mapping[str, SqlParameter] = {}
class TraceSQLColumn(BaseModel):
model_config = ConfigDict(
extra="allow",
frozen=True,
)
name: str
type: str
UnsignedCount1: TypeAlias = Annotated[int, Field(..., ge=0, le=18446744073709551615)]
UnsignedCount: TypeAlias = UnsignedCount1 | str
class TraceQueryStatistics(BaseModel):
model_config = ConfigDict(
extra="allow",
frozen=True,
)
elapsed: float
rows_read: UnsignedCount
bytes_read: UnsignedCount
class TraceSQLResponse(BaseModel):
model_config = ConfigDict(
extra="allow",
frozen=True,
)
meta: tuple[TraceSQLColumn, ...]
data: tuple[Mapping[str, JsonValue], ...]
rows: UnsignedCount
statistics: TraceQueryStatistics
class TraceSpanPageRequest(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
cursor: str | None = Field(None, max_length=512)
page_size: int = Field(100, ge=1, le=500)
SpanType: TypeAlias = Literal[
"agent",
"llm",
"tool",
"chain",
"framework",
"retriever",
"embedding",
"reranker",
"guardrail",
"evaluator",
"prompt",
"decision",
]
class TraceValuesRequest(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
start_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807)
end_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807)
as_of_ms: int | None = Field(None, ge=0, le=18446744073709551615)
q: str = Field("", max_length=1000)
contains: str = Field("", max_length=200)
limit: int = Field(20, ge=1, le=100)
class TraceSummary(BaseModel):
model_config = ConfigDict(
frozen=True,
)
resolution_limited: bool
trace_id: str
id: str
name: str
service: str
agent_names: tuple[str, ...]
frameworks: tuple[str, ...]
input_preview: str
start_time: str
duration_ms: float
root_status: SpanStatus
has_error: bool
span_count: int = Field(..., ge=0, le=18446744073709551615)
agent_count: int = Field(..., ge=0, le=18446744073709551615)
agent_invocations: int = Field(..., ge=0, le=18446744073709551615)
llm_calls: int = Field(..., ge=0, le=18446744073709551615)
tool_calls: int = Field(..., ge=0, le=18446744073709551615)
error_count: int = Field(..., ge=0, le=18446744073709551615)
input_tokens: int = Field(..., ge=0, le=18446744073709551615)
output_tokens: int = Field(..., ge=0, le=18446744073709551615)
models: tuple[str, ...]
spend: float | None
class TraceMetadata(BaseModel):
model_config = ConfigDict(
frozen=True,
)
summary: TraceSummary
agents: tuple[AgentNode, ...]
class Span(BaseModel):
model_config = ConfigDict(
frozen=True,
)
span_id: str
parent_span_id: str | None
name: str
type: SpanType
agent: str
framework: str
start_offset_ms: float
duration_ms: float
status: SpanStatus
error: str | None
error_truncated: bool
input_preview: str
model: str | None
input_tokens: int = Field(..., ge=0, le=4294967295)
output_tokens: int = Field(..., ge=0, le=4294967295)
litellm_request_id: str | None
spend: float | None
class TraceSpansPage(BaseModel):
model_config = ConfigDict(
frozen=True,
)
data: tuple[Span, ...]
next_cursor: str | None
TraceWireRequests: TypeAlias = Annotated[
TraceErrorPageRequest
| TraceHistogramRequest
| TraceListRequest
| TraceMetadata
| TraceNoQueryRequest
| TraceProblem
| TraceQueryRequest
| TraceSQLResponse
| TraceSpanPageRequest
| TraceSpansPage
| TraceValuesRequest,
Field(..., title="TraceWireRequests"),
]

View file

@ -0,0 +1,28 @@
# @generated by scripts/generate_trace_types.py, do not edit
from collections.abc import Mapping
from dataclasses import dataclass
from types import MappingProxyType
from typing import Final
@dataclass(frozen=True, slots=True)
class TraceOperation:
path: str
method: str
operation_id: str
OPERATIONS: Final[Mapping[str, TraceOperation]] = MappingProxyType(
{
"trace_list": TraceOperation("/v1/traces", "GET", "trace_list"),
"trace_ingest": TraceOperation("/v1/traces", "POST", "trace_ingest"),
"trace_histogram": TraceOperation("/v1/traces/histogram", "GET", "trace_histogram"),
"trace_query": TraceOperation("/v1/traces/query", "POST", "trace_query"),
"trace_query_help": TraceOperation("/v1/traces/query/help", "GET", "trace_query_help"),
"trace_values": TraceOperation("/v1/traces/values/{field}", "GET", "trace_values"),
"trace_get": TraceOperation("/v1/traces/{id}", "GET", "trace_get"),
"trace_spans": TraceOperation("/v1/traces/{id}/spans", "GET", "trace_spans"),
"trace_span": TraceOperation("/v1/traces/{id}/spans/{span_id}", "GET", "trace_span"),
"trace_error": TraceOperation("/v1/traces/{id}/spans/{span_id}/error", "GET", "trace_error"),
}
)

View file

@ -23,7 +23,17 @@ class OwnedQueryScope(typing_extensions.TypedDict):
QueryScope: TypeAlias = AllQueryScope | OwnedQueryScope
RunField: TypeAlias = Literal["name", "agent", "status", "model", "input", "trace_id", "service", "team"]
RunField: TypeAlias = Literal[
"name",
"agent",
"root_status",
"has_error",
"model",
"input",
"trace_id",
"service",
"team",
]
RunSortKey: TypeAlias = Literal["start_ms", "duration_ms", "span_count", "error_count", "trace_ref"]
@ -34,26 +44,22 @@ class RunOrder(typing_extensions.TypedDict):
descending: ReadOnly[bool]
class TraceQueryWindow(typing_extensions.TypedDict):
start_ms: ReadOnly[Annotated[int, Field(ge=-9223372036854775808, le=9223372036854775807)]]
end_ms: ReadOnly[Annotated[int, Field(ge=-9223372036854775808, le=9223372036854775807)]]
as_of_ms: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]]
class RunValues(typing_extensions.TypedDict):
window: ReadOnly[TraceQueryWindow]
values: ReadOnly[tuple[str, ...]]
class UIText(typing_extensions.TypedDict):
text: ReadOnly[str]
kind: ReadOnly[Literal["text"]]
ChatRole: TypeAlias = Literal["system", "user", "assistant", "tool"]
class UIToolCall(typing_extensions.TypedDict):
name: ReadOnly[str]
arguments: ReadOnly[str]
class UIField(typing_extensions.TypedDict):
key: ReadOnly[str]
value: ReadOnly[str]
class SpanDetail(typing_extensions.TypedDict):
span_id: ReadOnly[str]
input: ReadOnly[str]
output: ReadOnly[str]
attributes: ReadOnly[Mapping[str, str]]
class SpanErrorPage(typing_extensions.TypedDict):
@ -105,30 +111,19 @@ class AgentRuns(typing_extensions.TypedDict):
runs: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]]
class UIFields(typing_extensions.TypedDict):
fields: ReadOnly[tuple[UIField, ...]]
kind: ReadOnly[Literal["fields"]]
class UIMessage(typing_extensions.TypedDict):
role: ReadOnly[ChatRole]
content: ReadOnly[str]
name: ReadOnly[NotRequired[str | None]]
tool_calls: ReadOnly[NotRequired[tuple[UIToolCall, ...]]]
class TraceSummary(typing_extensions.TypedDict):
resolution_limited: ReadOnly[NotRequired[bool]]
resolution_limited: ReadOnly[bool]
trace_id: ReadOnly[str]
trace_ref: ReadOnly[NotRequired[str]]
id: ReadOnly[str]
name: ReadOnly[str]
service: ReadOnly[str]
agent_names: ReadOnly[NotRequired[tuple[str, ...]]]
frameworks: ReadOnly[NotRequired[tuple[str, ...]]]
agent_names: ReadOnly[tuple[str, ...]]
frameworks: ReadOnly[tuple[str, ...]]
input_preview: ReadOnly[str]
start_time: ReadOnly[str]
duration_ms: ReadOnly[float]
status: ReadOnly[SpanStatus]
root_status: ReadOnly[SpanStatus]
has_error: ReadOnly[bool]
span_count: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]]
agent_count: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]]
agent_invocations: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]]
@ -177,31 +172,16 @@ class HistogramBucket(typing_extensions.TypedDict):
class TraceHistogram(typing_extensions.TypedDict):
window: ReadOnly[TraceQueryWindow]
buckets: ReadOnly[tuple[HistogramBucket, ...]]
class TracePage(typing_extensions.TypedDict):
window: ReadOnly[TraceQueryWindow]
data: ReadOnly[tuple[TraceSummary, ...]]
next_cursor: ReadOnly[str | None]
class UIMessages(typing_extensions.TypedDict):
messages: ReadOnly[tuple[UIMessage, ...]]
kind: ReadOnly[Literal["messages"]]
UIContent: TypeAlias = UIMessages | UIFields | UIText
class SpanDetail(typing_extensions.TypedDict):
span_id: ReadOnly[str]
input_ui: ReadOnly[UIContent]
output_ui: ReadOnly[UIContent]
input: ReadOnly[str]
output: ReadOnly[str]
attributes: ReadOnly[Mapping[str, str]]
TraceWireTypes: TypeAlias = (
QueryScope
| RunField

View file

@ -1,23 +1,3 @@
from collections.abc import Mapping
from typing import Final
from .generated.requests import TraceSQLResponse
from pydantic import BaseModel, ConfigDict, JsonValue
from .generated.models import TraceQueryColumn
_RESPONSE_CONFIG: Final = ConfigDict(frozen=True, extra="allow")
class TraceQueryStatistics(BaseModel):
model_config = _RESPONSE_CONFIG
elapsed: float
rows_read: int | str
bytes_read: int | str
class TraceSQLResponse(BaseModel):
model_config = _RESPONSE_CONFIG
meta: tuple[TraceQueryColumn, ...]
data: tuple[Mapping[str, JsonValue], ...]
rows: int | str
statistics: TraceQueryStatistics
__all__ = ("TraceSQLResponse",)

View file

@ -8,6 +8,7 @@ from litellm.constants import AGENT_TRACING_LIST_PAGE_SIZE, OTLP_MAX_ATTRIBUTE_V
from litellm.rust_bridge.loader import get_native_bridge
from .generated.models import TraceQueryHelp
from .generated.requests import SqlParameter, TraceMetadata, TraceSpansPage
from .generated.types import (
QueryScope,
RunOrder,
@ -59,6 +60,7 @@ class NativeStore(Protocol):
limit: int,
order: RunOrder,
trace_refs: Sequence[str],
as_of_ms: int | None = None,
) -> Awaitable[JsonValue]: ...
def count_traces(
@ -79,11 +81,25 @@ class NativeStore(Protocol):
) -> Awaitable[JsonValue]: ...
def trace_histogram(
self, scope: QueryScope, start_ms: int, end_ms: int, q: str, buckets: int
self,
scope: QueryScope,
start_ms: int | None,
end_ms: int | None,
q: str,
buckets: int,
as_of_ms: int | None = None,
) -> Awaitable[JsonValue]: ...
def run_values(
self, scope: QueryScope, start_ms: int, end_ms: int, q: str, field: str, contains: str, limit: int
self,
scope: QueryScope,
start_ms: int | None,
end_ms: int | None,
q: str,
field: str,
contains: str,
limit: int,
as_of_ms: int | None = None,
) -> Awaitable[JsonValue]: ...
def get_trace(
@ -96,7 +112,21 @@ class NativeStore(Protocol):
self, trace_id: str, span_id: str, scope: QueryScope, trace_ref: str, cursor: str | None
) -> Awaitable[JsonValue]: ...
def query_sql(self, sql: str, scope: QueryScope, secret: str) -> Awaitable[str]: ...
def get_trace_metadata(self, scope: QueryScope, id: str) -> Awaitable[JsonValue]: ...
def get_trace_spans(
self, scope: QueryScope, id: str, cursor: str | None, page_size: int
) -> Awaitable[JsonValue]: ...
def get_span_by_id(self, scope: QueryScope, id: str, span_id: str) -> Awaitable[JsonValue]: ...
def get_span_error_by_id(
self, scope: QueryScope, id: str, span_id: str, cursor: str | None
) -> Awaitable[JsonValue]: ...
def query_sql(
self, sql: str, scope: QueryScope, secret: str, params: Mapping[str, SqlParameter]
) -> Awaitable[str]: ...
def query_help(self, scope: QueryScope, secret: str) -> Awaitable[JsonValue]: ...
@ -120,6 +150,8 @@ _TRACE_HISTOGRAM: Final = TypeAdapter(TraceHistogram)
_RUN_VALUES: Final = TypeAdapter(RunValues)
_COUNT: Final = TypeAdapter(int)
_SPAN_TEXTS: Final = TypeAdapter(tuple[SpanText, ...])
_TRACE_METADATA: Final = TypeAdapter(TraceMetadata | None)
_TRACE_SPANS: Final = TypeAdapter(TraceSpansPage | None)
_TRACE: Final[TypeAdapter[Trace | None]] = TypeAdapter(Trace | None)
_SPAN_DETAIL: Final[TypeAdapter[SpanDetail | None]] = TypeAdapter(SpanDetail | None)
_SPAN_ERROR_PAGE: Final[TypeAdapter[SpanErrorPage | None]] = TypeAdapter(SpanErrorPage | None)
@ -208,9 +240,10 @@ class ClickHouseStorage:
limit: int = AGENT_TRACING_LIST_PAGE_SIZE,
order: RunOrder = NEWEST,
trace_refs: Sequence[str] = (),
as_of_ms: int | None = None,
) -> TracePage:
result: Final = await self._native.list_traces(
scope, start_ms, end_ms, q, cursor, limit, order, tuple(trace_refs)
scope, start_ms, end_ms, q, cursor, limit, order, tuple(trace_refs), as_of_ms
)
return _validate_query_response(_TRACE_PAGE, result)
@ -238,15 +271,29 @@ class ClickHouseStorage:
return _validate_query_response(_SPAN_TEXTS, result)
async def trace_histogram(
self, scope: QueryScope, start_ms: int, end_ms: int, q: str, buckets: int
self,
scope: QueryScope,
start_ms: int | None,
end_ms: int | None,
q: str,
buckets: int,
as_of_ms: int | None = None,
) -> TraceHistogram:
result: Final = await self._native.trace_histogram(scope, start_ms, end_ms, q, buckets)
result: Final = await self._native.trace_histogram(scope, start_ms, end_ms, q, buckets, as_of_ms)
return _validate_query_response(_TRACE_HISTOGRAM, result)
async def run_values(
self, scope: QueryScope, start_ms: int, end_ms: int, q: str, field: str, contains: str, limit: int
self,
scope: QueryScope,
start_ms: int | None,
end_ms: int | None,
q: str,
field: str,
contains: str,
limit: int,
as_of_ms: int | None = None,
) -> RunValues:
result: Final = await self._native.run_values(scope, start_ms, end_ms, q, field, contains, limit)
result: Final = await self._native.run_values(scope, start_ms, end_ms, q, field, contains, limit, as_of_ms)
return _validate_query_response(_RUN_VALUES, result)
async def get_trace(
@ -270,8 +317,30 @@ class ClickHouseStorage:
result: Final = await self._native.get_span_error(trace_id, span_id, scope, trace_ref, cursor)
return _validate_query_response(_SPAN_ERROR_PAGE, result)
async def query_sql(self, sql: str, scope: QueryScope, secret: str) -> TraceSQLResponse:
result: Final = await self._native.query_sql(sql, scope, secret)
async def get_trace_metadata(self, scope: QueryScope, id: str) -> TraceMetadata | None:
result: Final = await self._native.get_trace_metadata(scope, id)
return _validate_query_response(_TRACE_METADATA, result)
async def get_trace_spans(
self, scope: QueryScope, id: str, cursor: str | None, page_size: int
) -> TraceSpansPage | None:
result: Final = await self._native.get_trace_spans(scope, id, cursor, page_size)
return _validate_query_response(_TRACE_SPANS, result)
async def get_span_by_id(self, scope: QueryScope, id: str, span_id: str) -> SpanDetail | None:
result: Final = await self._native.get_span_by_id(scope, id, span_id)
return _validate_query_response(_SPAN_DETAIL, result)
async def get_span_error_by_id(
self, scope: QueryScope, id: str, span_id: str, cursor: str | None
) -> SpanErrorPage | None:
result: Final = await self._native.get_span_error_by_id(scope, id, span_id, cursor)
return _validate_query_response(_SPAN_ERROR_PAGE, result)
async def query_sql(
self, sql: str, scope: QueryScope, secret: str, params: Mapping[str, SqlParameter]
) -> TraceSQLResponse:
result: Final = await self._native.query_sql(sql, scope, secret, params)
return _decode_query_response(_SQL_RESPONSE, result)
async def query_help(self, scope: QueryScope, secret: str) -> TraceQueryHelp:

View file

@ -19,6 +19,7 @@ from threading import BoundedSemaphore
from typing import Final
from litellm.constants import AGENT_TRACING_LIST_PAGE_SIZE, OTLP_MAX_BODY_BYTES, OTLP_MAX_CONCURRENT_INGESTS
from litellm.rust_bridge.trace.generated.requests import TraceMetadata, TraceSpansPage
from litellm.rust_bridge.trace.generated.types import (
QueryScope,
RunField,
@ -114,18 +115,34 @@ class TraceReceiver:
q: str = "",
cursor: str | None = None,
order: RunOrder = NEWEST,
page_size: int = AGENT_TRACING_LIST_PAGE_SIZE,
as_of_ms: int | None = None,
) -> TracePage:
return await self.storage.list_traces(scope, start_ms, end_ms, q, cursor, AGENT_TRACING_LIST_PAGE_SIZE, order)
return await self.storage.list_traces(scope, start_ms, end_ms, q, cursor, page_size, order, as_of_ms=as_of_ms)
async def trace_histogram(
self, scope: QueryScope, start_ms: int, end_ms: int, q: str, buckets: int
self,
scope: QueryScope,
start_ms: int | None,
end_ms: int | None,
q: str,
buckets: int,
as_of_ms: int | None = None,
) -> TraceHistogram:
return await self.storage.trace_histogram(scope, start_ms, end_ms, q, buckets)
return await self.storage.trace_histogram(scope, start_ms, end_ms, q, buckets, as_of_ms)
async def run_values(
self, scope: QueryScope, start_ms: int, end_ms: int, q: str, field: RunField, contains: str, limit: int
self,
scope: QueryScope,
start_ms: int | None,
end_ms: int | None,
q: str,
field: RunField,
contains: str,
limit: int,
as_of_ms: int | None = None,
) -> RunValues:
return await self.storage.run_values(scope, start_ms, end_ms, q, field, contains, limit)
return await self.storage.run_values(scope, start_ms, end_ms, q, field, contains, limit, as_of_ms)
async def get_trace(
self,
@ -145,6 +162,22 @@ class TraceReceiver:
) -> SpanErrorPage | None:
return await self.storage.get_span_error(trace_id, span_id, scope, trace_ref, cursor)
async def get_trace_metadata(self, scope: QueryScope, id: str) -> TraceMetadata | None:
return await self.storage.get_trace_metadata(scope, id)
async def get_trace_spans(
self, scope: QueryScope, id: str, cursor: str | None, page_size: int
) -> TraceSpansPage | None:
return await self.storage.get_trace_spans(scope, id, cursor, page_size)
async def get_span_by_id(self, scope: QueryScope, id: str, span_id: str) -> SpanDetail | None:
return await self.storage.get_span_by_id(scope, id, span_id)
async def get_span_error_by_id(
self, scope: QueryScope, id: str, span_id: str, cursor: str | None
) -> SpanErrorPage | None:
return await self.storage.get_span_error_by_id(scope, id, span_id, cursor)
async def _read_body(chunks: AsyncIterable[bytes]) -> bytes:
with BytesIO() as body:

View file

@ -15,7 +15,7 @@ from tempfile import TemporaryDirectory
from types import MappingProxyType
from typing import Final
from pydantic import BaseModel, ConfigDict, JsonValue, TypeAdapter
from pydantic import BaseModel, ConfigDict, Field, JsonValue, TypeAdapter
ROOT: Final = Path(__file__).resolve().parents[1]
TOOLING: Final = ROOT / "scripts/trace_codegen"
@ -34,7 +34,7 @@ class GeneratorConfig(BaseModel):
options: tuple[str, ...]
def export(crate: str) -> Mapping[str, Mapping[str, JsonValue]]:
def export(crate: str, api: bool = False) -> Mapping[str, Mapping[str, JsonValue]]:
result: Final = subprocess.run(
(
"cargo",
@ -48,6 +48,7 @@ def export(crate: str) -> Mapping[str, Mapping[str, JsonValue]]:
f"export-{crate}-schema",
"--features",
"schema",
*(("--", "--api") if api else ()),
),
check=True,
stdout=subprocess.PIPE,
@ -100,7 +101,7 @@ def generate(
"pydantic_v2.BaseModel",
"--enable-faux-immutability",
"--additional-imports",
"collections.abc.Mapping,typing.TypeAlias",
"collections.abc.Mapping,typing.TypeAlias,pydantic.JsonValue",
)
)
subprocess.run(
@ -154,6 +155,79 @@ def reconcile_schemas(expected: frozenset[Path], check: bool) -> bool:
return True
class OperationConfig(BaseModel):
model_config = ConfigDict(frozen=True)
operation_id: str = Field(alias="operationId")
class OpenAPIConfig(BaseModel):
model_config = ConfigDict(frozen=True)
paths: Mapping[str, Mapping[str, OperationConfig]]
def export_openapi() -> str:
result: Final = subprocess.run(
(
"cargo",
"run",
"--locked",
"--manifest-path",
str(ROOT / "litellm-rust/Cargo.toml"),
"-p",
"litellm-traces-clickhouse",
"--bin",
"export-traces-openapi",
"--features",
"schema",
),
check=True,
stdout=subprocess.PIPE,
text=True,
)
document: Final = TypeAdapter(dict[str, JsonValue]).validate_json(result.stdout)
return json.dumps(document, indent=2, sort_keys=True) + "\n"
def route_entries(document: OpenAPIConfig) -> Iterator[str]:
for path, methods in document.paths.items():
for method, operation in methods.items():
yield f" {operation.operation_id!r}: TraceOperation({path!r}, {method.upper()!r}, {operation.operation_id!r}),"
def route_source(document: str) -> str:
paths: Final = OpenAPIConfig.model_validate_json(document)
entries: Final = "\n".join(route_entries(paths))
return f"""# @generated by scripts/generate_trace_types.py, do not edit
from collections.abc import Mapping
from dataclasses import dataclass
from types import MappingProxyType
from typing import Final
@dataclass(frozen=True, slots=True)
class TraceOperation:
path: str
method: str
operation_id: str
OPERATIONS: Final[Mapping[str, TraceOperation]] = MappingProxyType({{
{entries}
}})
"""
def formatted_routes(document: str, directory: Path) -> str:
output: Final = directory / "routes.py"
output.write_text(route_source(document))
subprocess.run(
(sys.executable, "-m", "ruff", "format", "--line-length", "120", str(output)),
check=True,
stdout=subprocess.DEVNULL,
)
return output.read_text()
def main() -> int:
parser: Final = argparse.ArgumentParser(description="Regenerate trace schemas and Python wire contracts")
parser.add_argument("--check", action="store_true", help="compare fresh schemas and Python with committed files")
@ -164,16 +238,22 @@ def main() -> int:
return 1
domain: Final = export("traces")
clickhouse: Final = export("traces-clickhouse")
exported: Final = tuple(schema_files(domain, clickhouse))
api: Final = export("traces", api=True)
openapi: Final = export_openapi()
exported: Final = tuple(schema_files(domain, clickhouse, api))
schema_results: Final = tuple(publish(path, content, args.check) for path, content in exported)
schema_set_matches: Final = reconcile_schemas(frozenset(path for path, _ in exported), args.check)
with TemporaryDirectory(prefix="trace-codegen-") as temporary:
directory: Final = Path(temporary)
types: Final = generate(domain, "types", directory, config)
models: Final = generate(clickhouse, "models", directory, config)
requests: Final = generate(api, "requests", directory, config)
python_results: Final = (
publish(GENERATED / "types.py", types.read_text(), args.check),
publish(GENERATED / "models.py", models.read_text(), args.check),
publish(GENERATED / "requests.py", requests.read_text(), args.check),
publish(GENERATED / "openapi.json", openapi, args.check),
publish(GENERATED / "routes.py", formatted_routes(openapi, directory), args.check),
)
return 0 if all((schema_set_matches, *schema_results, *python_results)) else 1
@ -181,8 +261,9 @@ def main() -> int:
def schema_files(
domain: Mapping[str, Mapping[str, JsonValue]],
clickhouse: Mapping[str, Mapping[str, JsonValue]],
api: Mapping[str, Mapping[str, JsonValue]],
) -> Iterator[tuple[Path, str]]:
for crate, schemas in (("traces", domain), ("traces-clickhouse", clickhouse)):
for crate, schemas in (("traces", domain), ("traces-clickhouse", clickhouse), ("api", api)):
for name, schema in schemas.items():
yield TOOLING / "schemas" / crate / f"{name}.json", json.dumps(schema, indent=2, sort_keys=True) + "\n"

View file

@ -1,9 +1,26 @@
Run `uv run scripts/generate_trace_types.py` from the repository root to export Rust schemas and regenerate the Python trace contracts. Run the same command with `--check` to compare fresh output with the committed schemas and Python files
The trace API contract is owned by Rust. `litellm-rust/crates/traces/src/api.rs` defines public requests, responses, validation limits and defaults. `src/api/openapi.rs` defines paths, operation IDs, authentication and operation semantics. Domain response fields remain in their owning Rust types, and ClickHouse query-help types remain in `traces-clickhouse`
The script pins datamodel-code-generator in its inline dependency metadata. Rust uses the workspace's locked Schemars version through each owning crate's optional `schema` feature. Neither tool is a Python runtime dependency
Run `uv run scripts/generate_trace_types.py` from the repository root to generate JSON Schema 2020-12, OpenAPI 3.1, Python validation models and route metadata. Run the same command with `--check` to detect drift. The committed OpenAPI document is `litellm/rust_bridge/trace/generated/openapi.json`
Each crate exports its own roots using JSON Schema 2020-12. Request parameters use Schemars' deserialization contract. Trace views and query help use its serialization contract. Lens rows use their ClickHouse deserialization schemas, including quoted numbers and numeric boolean flags
FastAPI imports generated request models and route metadata, invokes the Rust bridge, and merges the generated document into `/openapi.json`. Its route decorators do not own trace API documentation. The dashboard's `npm run gen:api` consumes that merged document. A future Rust HTTP server can consume the same request types and operation contract
The templates preserve tuple conversion, immutable tuple defaults, and bounded `ReadOnly` TypedDict fields. Pydantic models use the generator's frozen-model option and each schema's extra-field policy. ClickHouse numeric schemas select bounded, normalized Python scalar types through schema metadata consumed by the model template
| Operation | Contract |
| --- | --- |
| `GET /v1/traces` | Search summaries with `q`, bounded `page_size`, four sort fields and a continuation cursor |
| `GET /v1/traces/histogram` | Count the same search in bounded time buckets |
| `GET /v1/traces/values/{field}` | Return bounded top-K search suggestions |
| `GET /v1/traces/{id}` | Read summary and agent metadata by canonical ID |
| `GET /v1/traces/{id}/spans` | Traverse bounded pages of spans |
| `GET /v1/traces/{id}/spans/{span_id}` | Read raw span input, output and attributes |
| `GET /v1/traces/{id}/spans/{span_id}/error` | Traverse bounded diagnostic text |
| `POST /v1/traces/query` | Execute scoped, bounded ClickHouse SQL with native parameter binding |
| `GET /v1/traces/query/help` | Discover logical views, physical tables, examples and limits |
| `POST /v1/traces` | Receive the standard OTLP JSON or protobuf protocol |
Edit the owning Rust contract, schema annotation, or generation configuration, then regenerate. Never edit `litellm/rust_bridge/trace/generated/` manually. The SQL response envelope remains handwritten in `queries.py`
The list, histogram and suggestions return a resolved `[start_ms,end_ms)` window and ingestion cutoff. Reuse those values to compare aggregates with the list. List cursors retain the window, query, scope and ordering. The cutoff excludes exports stamped later; buffered or distributed writes stamped before the cutoff can become visible later. It is not a database transaction snapshot. Collection pages use `data` and `next_cursor`; text pages retain their text-specific envelope. `trace_id` is the original OTLP identifier; `id` includes ownership and is the public read identifier. `root_status` describes the root span, while `has_error` reports any failed span
SQL exposes logical `traces`, `spans` and `calls` views under invoker row policies. Views summarize visible canonical spans. Curated user-only trace reads additionally require full ownership, so their membership can differ from SQL views. SQL pages are live and callers own keyset continuation. SQL span counters and token totals describe canonical normalized spans; curated call counts and token totals additionally resolve graph wrappers. These enrichment metrics have distinct SQL column names. Spend enrichment is best effort because replacing spend rows do not preserve historical versions
JSON read failures use RFC 9457 `application/problem+json` with a stable `code`. OTLP retains its protocol responses. Invalid or unknown query arguments are rejected
The generator pins datamodel-code-generator in inline dependency metadata and uses locked Schemars and Utoipa dependencies behind the optional Rust `schema` feature. These tools are not Python runtime dependencies. Templates preserve immutable Python collections and bounded types. Edit Rust or generation configuration, then regenerate; never edit generated contracts manually

View file

@ -0,0 +1,16 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"properties": {
"cursor": {
"default": null,
"maxLength": 512,
"type": [
"string",
"null"
]
}
},
"title": "TraceErrorPageRequest",
"type": "object"
}

View file

@ -0,0 +1,50 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"properties": {
"as_of_ms": {
"default": null,
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": [
"integer",
"null"
]
},
"buckets": {
"default": 60,
"format": "uint16",
"maximum": 240,
"minimum": 1,
"type": "integer"
},
"end_ms": {
"default": null,
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": [
"integer",
"null"
]
},
"q": {
"default": "",
"maxLength": 1000,
"type": "string"
},
"start_ms": {
"default": null,
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": [
"integer",
"null"
]
}
},
"title": "TraceHistogramRequest",
"type": "object"
}

View file

@ -0,0 +1,84 @@
{
"$defs": {
"TraceSortDirection": {
"enum": [
"asc",
"desc"
],
"type": "string"
},
"TraceSortField": {
"enum": [
"start_ms",
"duration_ms",
"span_count",
"error_count"
],
"type": "string"
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"properties": {
"as_of_ms": {
"default": null,
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": [
"integer",
"null"
]
},
"cursor": {
"default": null,
"maxLength": 512,
"type": [
"string",
"null"
]
},
"end_ms": {
"default": null,
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": [
"integer",
"null"
]
},
"page_size": {
"default": 50,
"format": "uint16",
"maximum": 500,
"minimum": 1,
"type": "integer"
},
"q": {
"default": "",
"maxLength": 1000,
"type": "string"
},
"sort_by": {
"$ref": "#/$defs/TraceSortField",
"default": "start_ms"
},
"sort_dir": {
"$ref": "#/$defs/TraceSortDirection",
"default": "desc"
},
"start_ms": {
"default": null,
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": [
"integer",
"null"
]
}
},
"title": "TraceListRequest",
"type": "object"
}

View file

@ -0,0 +1,216 @@
{
"$defs": {
"AgentNode": {
"description": "One distinct agent in a trace: 200 invocations of `researcher` are one node.",
"properties": {
"duration_ms": {
"format": "double",
"type": "number"
},
"invocations": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"llm_calls": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"name": {
"type": "string"
},
"parent_agent": {
"type": [
"string",
"null"
]
},
"spend": {
"format": "double",
"type": [
"number",
"null"
]
},
"tool_calls": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
}
},
"required": [
"name",
"parent_agent",
"invocations",
"llm_calls",
"tool_calls",
"duration_ms",
"spend"
],
"type": "object"
},
"SpanStatus": {
"enum": [
"ok",
"error",
"unset"
],
"type": "string"
},
"TraceSummary": {
"properties": {
"agent_count": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"agent_invocations": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"agent_names": {
"items": {
"type": "string"
},
"type": "array"
},
"duration_ms": {
"format": "double",
"type": "number"
},
"error_count": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"frameworks": {
"items": {
"type": "string"
},
"type": "array"
},
"has_error": {
"type": "boolean"
},
"id": {
"type": "string"
},
"input_preview": {
"type": "string"
},
"input_tokens": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"llm_calls": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"models": {
"items": {
"type": "string"
},
"type": "array"
},
"name": {
"type": "string"
},
"output_tokens": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"resolution_limited": {
"type": "boolean"
},
"root_status": {
"$ref": "#/$defs/SpanStatus"
},
"service": {
"type": "string"
},
"span_count": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"spend": {
"format": "double",
"type": [
"number",
"null"
]
},
"start_time": {
"type": "string"
},
"tool_calls": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"trace_id": {
"type": "string"
}
},
"required": [
"resolution_limited",
"trace_id",
"id",
"name",
"service",
"agent_names",
"frameworks",
"input_preview",
"start_time",
"duration_ms",
"root_status",
"has_error",
"span_count",
"agent_count",
"agent_invocations",
"llm_calls",
"tool_calls",
"error_count",
"input_tokens",
"output_tokens",
"models",
"spend"
],
"type": "object"
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"properties": {
"agents": {
"items": {
"$ref": "#/$defs/AgentNode"
},
"type": "array"
},
"summary": {
"$ref": "#/$defs/TraceSummary"
}
},
"required": [
"summary",
"agents"
],
"title": "TraceMetadata",
"type": "object"
}

View file

@ -0,0 +1,6 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"title": "TraceNoQueryRequest",
"type": "object"
}

View file

@ -0,0 +1,83 @@
{
"$defs": {
"TraceInvalidParam": {
"additionalProperties": false,
"properties": {
"location": {
"type": "string"
},
"reason": {
"type": "string"
}
},
"required": [
"location",
"reason"
],
"type": "object"
},
"TraceProblemCode": {
"enum": [
"invalid_request",
"unauthorized",
"forbidden",
"not_found",
"trace_changed",
"too_large",
"unavailable",
"query_rejected",
"query_limit_exceeded",
"query_unavailable",
"internal_error"
],
"type": "string"
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"properties": {
"code": {
"$ref": "#/$defs/TraceProblemCode"
},
"database_code": {
"format": "uint32",
"maximum": 4294967295,
"minimum": 0,
"type": [
"integer",
"null"
]
},
"detail": {
"type": "string"
},
"errors": {
"default": [],
"items": {
"$ref": "#/$defs/TraceInvalidParam"
},
"type": "array"
},
"status": {
"format": "uint16",
"maximum": 65535,
"minimum": 0,
"type": "integer"
},
"title": {
"type": "string"
},
"type": {
"type": "string"
}
},
"required": [
"type",
"title",
"status",
"detail",
"code"
],
"title": "TraceProblem",
"type": "object"
}

View file

@ -0,0 +1,59 @@
{
"$defs": {
"SqlParameter": {
"anyOf": [
{
"type": "string"
},
{
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": "integer"
},
{
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
{
"format": "double",
"type": "number"
},
{
"type": "boolean"
},
{
"type": "null"
},
{
"items": {
"type": "string"
},
"type": "array"
}
]
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"properties": {
"params": {
"additionalProperties": {
"$ref": "#/$defs/SqlParameter"
},
"default": {},
"type": "object"
},
"sql": {
"minLength": 1,
"type": "string"
}
},
"required": [
"sql"
],
"title": "TraceQueryRequest",
"type": "object"
}

View file

@ -0,0 +1,88 @@
{
"$defs": {
"TraceQueryStatistics": {
"additionalProperties": true,
"properties": {
"bytes_read": {
"$ref": "#/$defs/UnsignedCount"
},
"elapsed": {
"format": "double",
"type": "number"
},
"rows_read": {
"$ref": "#/$defs/UnsignedCount"
}
},
"required": [
"elapsed",
"rows_read",
"bytes_read"
],
"type": "object"
},
"TraceSQLColumn": {
"additionalProperties": true,
"properties": {
"name": {
"type": "string"
},
"type": {
"type": "string"
}
},
"required": [
"name",
"type"
],
"type": "object"
},
"UnsignedCount": {
"anyOf": [
{
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
{
"type": "string"
}
]
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": true,
"properties": {
"data": {
"items": {
"additionalProperties": true,
"type": "object"
},
"type": "array",
"x-python-normalized": {
"type": "tuple[Mapping[str, JsonValue], ...]"
}
},
"meta": {
"items": {
"$ref": "#/$defs/TraceSQLColumn"
},
"type": "array"
},
"rows": {
"$ref": "#/$defs/UnsignedCount"
},
"statistics": {
"$ref": "#/$defs/TraceQueryStatistics"
}
},
"required": [
"meta",
"data",
"rows",
"statistics"
],
"title": "TraceSQLResponse",
"type": "object"
}

View file

@ -0,0 +1,23 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"properties": {
"cursor": {
"default": null,
"maxLength": 512,
"type": [
"string",
"null"
]
},
"page_size": {
"default": 100,
"format": "uint16",
"maximum": 500,
"minimum": 1,
"type": "integer"
}
},
"title": "TraceSpanPageRequest",
"type": "object"
}

View file

@ -0,0 +1,149 @@
{
"$defs": {
"Span": {
"properties": {
"agent": {
"type": "string"
},
"duration_ms": {
"format": "double",
"type": "number"
},
"error": {
"type": [
"string",
"null"
]
},
"error_truncated": {
"type": "boolean"
},
"framework": {
"type": "string"
},
"input_preview": {
"type": "string"
},
"input_tokens": {
"format": "uint32",
"maximum": 4294967295,
"minimum": 0,
"type": "integer"
},
"litellm_request_id": {
"type": [
"string",
"null"
]
},
"model": {
"type": [
"string",
"null"
]
},
"name": {
"type": "string"
},
"output_tokens": {
"format": "uint32",
"maximum": 4294967295,
"minimum": 0,
"type": "integer"
},
"parent_span_id": {
"type": [
"string",
"null"
]
},
"span_id": {
"type": "string"
},
"spend": {
"format": "double",
"type": [
"number",
"null"
]
},
"start_offset_ms": {
"format": "double",
"type": "number"
},
"status": {
"$ref": "#/$defs/SpanStatus"
},
"type": {
"$ref": "#/$defs/SpanType"
}
},
"required": [
"span_id",
"parent_span_id",
"name",
"type",
"agent",
"framework",
"start_offset_ms",
"duration_ms",
"status",
"error",
"error_truncated",
"input_preview",
"model",
"input_tokens",
"output_tokens",
"litellm_request_id",
"spend"
],
"type": "object"
},
"SpanStatus": {
"enum": [
"ok",
"error",
"unset"
],
"type": "string"
},
"SpanType": {
"enum": [
"agent",
"llm",
"tool",
"chain",
"framework",
"retriever",
"embedding",
"reranker",
"guardrail",
"evaluator",
"prompt",
"decision"
],
"type": "string"
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"properties": {
"data": {
"items": {
"$ref": "#/$defs/Span"
},
"type": "array"
},
"next_cursor": {
"type": [
"string",
"null"
]
}
},
"required": [
"data",
"next_cursor"
],
"title": "TraceSpansPage",
"type": "object"
}

View file

@ -0,0 +1,55 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"additionalProperties": false,
"properties": {
"as_of_ms": {
"default": null,
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": [
"integer",
"null"
]
},
"contains": {
"default": "",
"maxLength": 200,
"type": "string"
},
"end_ms": {
"default": null,
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": [
"integer",
"null"
]
},
"limit": {
"default": 20,
"format": "uint16",
"maximum": 100,
"minimum": 1,
"type": "integer"
},
"q": {
"default": "",
"maxLength": 1000,
"type": "string"
},
"start_ms": {
"default": null,
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": [
"integer",
"null"
]
}
},
"title": "TraceValuesRequest",
"type": "object"
}

View file

@ -77,7 +77,7 @@
"type": "string"
},
"table": {
"$ref": "#/$defs/TraceTableName"
"$ref": "#/$defs/TraceQueryTableName"
},
"truncated": {
"type": "boolean"
@ -93,22 +93,6 @@
],
"type": "object"
},
"TraceQueryColumn": {
"additionalProperties": true,
"properties": {
"name": {
"type": "string"
},
"type": {
"type": "string"
}
},
"required": [
"name",
"type"
],
"type": "object"
},
"TraceQueryExample": {
"properties": {
"name": {
@ -162,7 +146,7 @@
"type": "string"
},
"table": {
"$ref": "#/$defs/TraceTableName"
"$ref": "#/$defs/TraceQueryTableName"
},
"truncated": {
"type": "boolean"
@ -220,7 +204,7 @@
"type": "string"
},
"table": {
"$ref": "#/$defs/TraceTableName"
"$ref": "#/$defs/TraceQueryTableName"
},
"type": {
"type": "string"
@ -264,12 +248,12 @@
"properties": {
"columns": {
"items": {
"$ref": "#/$defs/TraceQueryColumn"
"$ref": "#/$defs/TraceSQLColumn"
},
"type": "array"
},
"name": {
"$ref": "#/$defs/TraceTableName"
"$ref": "#/$defs/TraceQueryTableName"
}
},
"required": [
@ -278,13 +262,32 @@
],
"type": "object"
},
"TraceTableName": {
"TraceQueryTableName": {
"enum": [
"traces",
"spans",
"calls",
"otel_traces",
"agent_traces_by_key",
"spend_logs"
],
"type": "string"
},
"TraceSQLColumn": {
"additionalProperties": true,
"properties": {
"name": {
"type": "string"
},
"type": {
"type": "string"
}
},
"required": [
"name",
"type"
],
"type": "object"
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",

View file

@ -3,7 +3,8 @@
"enum": [
"name",
"agent",
"status",
"root_status",
"has_error",
"model",
"input",
"trace_id",

View file

@ -1,4 +1,35 @@
{
"$defs": {
"TraceQueryWindow": {
"additionalProperties": false,
"properties": {
"as_of_ms": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"end_ms": {
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": "integer"
},
"start_ms": {
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": "integer"
}
},
"required": [
"start_ms",
"end_ms",
"as_of_ms"
],
"type": "object"
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"description": "Distinct values of one run field, most common first.",
"properties": {
@ -7,9 +38,13 @@
"type": "string"
},
"type": "array"
},
"window": {
"$ref": "#/$defs/TraceQueryWindow"
}
},
"required": [
"window",
"values"
],
"title": "RunValues",

View file

@ -1,136 +1,4 @@
{
"$defs": {
"ChatRole": {
"enum": [
"system",
"user",
"assistant",
"tool"
],
"type": "string"
},
"UIContent": {
"oneOf": [
{
"properties": {
"kind": {
"const": "messages",
"type": "string"
},
"messages": {
"items": {
"$ref": "#/$defs/UIMessage"
},
"type": "array"
}
},
"required": [
"kind",
"messages"
],
"title": "UIMessages",
"type": "object"
},
{
"properties": {
"fields": {
"items": {
"$ref": "#/$defs/UIField"
},
"type": "array"
},
"kind": {
"const": "fields",
"type": "string"
}
},
"required": [
"kind",
"fields"
],
"title": "UIFields",
"type": "object"
},
{
"properties": {
"kind": {
"const": "text",
"type": "string"
},
"text": {
"type": "string"
}
},
"required": [
"kind",
"text"
],
"title": "UIText",
"type": "object"
}
]
},
"UIField": {
"properties": {
"key": {
"type": "string"
},
"value": {
"type": "string"
}
},
"required": [
"key",
"value"
],
"type": "object"
},
"UIMessage": {
"properties": {
"content": {
"type": "string"
},
"name": {
"type": [
"string",
"null"
]
},
"role": {
"$ref": "#/$defs/ChatRole"
},
"tool_calls": {
"items": {
"$ref": "#/$defs/UIToolCall"
},
"type": [
"array",
"null"
]
}
},
"required": [
"role",
"content"
],
"type": "object"
},
"UIToolCall": {
"properties": {
"arguments": {
"type": "string"
},
"name": {
"type": "string"
}
},
"required": [
"name",
"arguments"
],
"type": "object"
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
"properties": {
"attributes": {
@ -142,23 +10,15 @@
"input": {
"type": "string"
},
"input_ui": {
"$ref": "#/$defs/UIContent"
},
"output": {
"type": "string"
},
"output_ui": {
"$ref": "#/$defs/UIContent"
},
"span_id": {
"type": "string"
}
},
"required": [
"span_id",
"input_ui",
"output_ui",
"input",
"output",
"attributes"

View file

@ -195,8 +195,7 @@
"items": {
"type": "string"
},
"type": "array",
"x-python-optional": true
"type": "array"
},
"duration_ms": {
"format": "double",
@ -212,8 +211,13 @@
"items": {
"type": "string"
},
"type": "array",
"x-python-optional": true
"type": "array"
},
"has_error": {
"type": "boolean"
},
"id": {
"type": "string"
},
"input_preview": {
"type": "string"
@ -246,8 +250,10 @@
"type": "integer"
},
"resolution_limited": {
"type": "boolean",
"x-python-optional": true
"type": "boolean"
},
"root_status": {
"$ref": "#/$defs/SpanStatus"
},
"service": {
"type": "string"
@ -268,9 +274,6 @@
"start_time": {
"type": "string"
},
"status": {
"$ref": "#/$defs/SpanStatus"
},
"tool_calls": {
"format": "uint64",
"maximum": 18446744073709551615,
@ -279,16 +282,12 @@
},
"trace_id": {
"type": "string"
},
"trace_ref": {
"type": "string",
"x-python-optional": true
}
},
"required": [
"resolution_limited",
"trace_id",
"trace_ref",
"id",
"name",
"service",
"agent_names",
@ -296,7 +295,8 @@
"input_preview",
"start_time",
"duration_ms",
"status",
"root_status",
"has_error",
"span_count",
"agent_count",
"agent_invocations",

View file

@ -60,6 +60,35 @@
"agents"
],
"type": "object"
},
"TraceQueryWindow": {
"additionalProperties": false,
"properties": {
"as_of_ms": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"end_ms": {
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": "integer"
},
"start_ms": {
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": "integer"
}
},
"required": [
"start_ms",
"end_ms",
"as_of_ms"
],
"type": "object"
}
},
"$schema": "https://json-schema.org/draft/2020-12/schema",
@ -70,9 +99,13 @@
"$ref": "#/$defs/HistogramBucket"
},
"type": "array"
},
"window": {
"$ref": "#/$defs/TraceQueryWindow"
}
},
"required": [
"window",
"buckets"
],
"title": "TraceHistogram",

View file

@ -8,6 +8,35 @@
],
"type": "string"
},
"TraceQueryWindow": {
"additionalProperties": false,
"properties": {
"as_of_ms": {
"format": "uint64",
"maximum": 18446744073709551615,
"minimum": 0,
"type": "integer"
},
"end_ms": {
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": "integer"
},
"start_ms": {
"format": "int64",
"maximum": 9223372036854775807,
"minimum": -9223372036854775808,
"type": "integer"
}
},
"required": [
"start_ms",
"end_ms",
"as_of_ms"
],
"type": "object"
},
"TraceSummary": {
"properties": {
"agent_count": {
@ -26,8 +55,7 @@
"items": {
"type": "string"
},
"type": "array",
"x-python-optional": true
"type": "array"
},
"duration_ms": {
"format": "double",
@ -43,8 +71,13 @@
"items": {
"type": "string"
},
"type": "array",
"x-python-optional": true
"type": "array"
},
"has_error": {
"type": "boolean"
},
"id": {
"type": "string"
},
"input_preview": {
"type": "string"
@ -77,8 +110,10 @@
"type": "integer"
},
"resolution_limited": {
"type": "boolean",
"x-python-optional": true
"type": "boolean"
},
"root_status": {
"$ref": "#/$defs/SpanStatus"
},
"service": {
"type": "string"
@ -99,9 +134,6 @@
"start_time": {
"type": "string"
},
"status": {
"$ref": "#/$defs/SpanStatus"
},
"tool_calls": {
"format": "uint64",
"maximum": 18446744073709551615,
@ -110,16 +142,12 @@
},
"trace_id": {
"type": "string"
},
"trace_ref": {
"type": "string",
"x-python-optional": true
}
},
"required": [
"resolution_limited",
"trace_id",
"trace_ref",
"id",
"name",
"service",
"agent_names",
@ -127,7 +155,8 @@
"input_preview",
"start_time",
"duration_ms",
"status",
"root_status",
"has_error",
"span_count",
"agent_count",
"agent_invocations",
@ -155,9 +184,13 @@
"string",
"null"
]
},
"window": {
"$ref": "#/$defs/TraceQueryWindow"
}
},
"required": [
"window",
"data",
"next_cursor"
],

View file

@ -21,6 +21,8 @@ from pydantic import BaseModel, ConfigDict, JsonValue, TypeAdapter
from litellm.constants import AGENT_TRACING_LIST_PAGE_SIZE, OTLP_MAX_ATTRIBUTE_VALUE_BYTES
from litellm.rust_bridge._native import NativeTraceConfig, NativeTraceStorage
from litellm.rust_bridge.trace.generated.models import TraceQueryHelp
from litellm.rust_bridge.trace.generated.requests import Span as SpanModel
from litellm.rust_bridge.trace.generated.requests import TraceMetadata, TraceSpansPage
from litellm.rust_bridge.trace.generated.types import AllQueryScope, Trace, TracePage
from litellm.rust_bridge.trace.storage import ClickHouseStorage, TraceStorageConfig, span_rows
from litellm.tracing import Tenant, TraceReceiver, TracingPayloadTooLargeError
@ -148,7 +150,9 @@ async def test_from_env_reads_with_clickhouse_url(
monkeypatch.delenv("CLICKHOUSE_READER_URL", raising=False)
scope: Final = AllQueryScope(kind="all")
page: Final = await TraceReceiver.from_env().list_traces(scope, 0, 1)
assert page == {"data": (), "next_cursor": None}
assert page["data"] == ()
assert page["next_cursor"] is None
assert (page["window"]["start_ms"], page["window"]["end_ms"]) == (0, 1)
assert len(recording_server.requests) == 1
@ -216,9 +220,12 @@ def test_trace_pages_keep_the_effective_window_through_the_http_native_boundary(
second: Final = client.get("/v1/traces", params={**params, "cursor": cursor})
assert second.status_code == 200, second.text
second_body: Final = _TRACE_PAGE.validate_python(second.json())
refs: Final = tuple(row["trace_ref"] for row in (*first_body["data"], *second_body["data"]))
refs: Final = tuple(row["id"] for row in (*first_body["data"], *second_body["data"]))
assert refs == tuple(row["trace_ref"] for row in reversed(rows))
assert second_body["next_cursor"] is None
assert second_body["window"] == first_body["window"]
assert second_body["data"][0]["root_status"] == "ok"
assert not second_body["data"][0]["has_error"]
run_parameters: Final = tuple(
parse_qs(urlsplit(request.path).query)
for request in recording_server.requests
@ -229,12 +236,18 @@ def test_trace_pages_keep_the_effective_window_through_the_http_native_boundary(
first_parameters["param_start_ms"],
first_parameters["param_end_ms"],
)
assert run_parameters[1]["param_as_of_ms"] == first_parameters["param_as_of_ms"]
sent: Final = len(recording_server.requests)
changed: Final = client.get(
"/v1/traces", params={"cursor": cursor, "end_ms": int(first_parameters["param_end_ms"][0]) + 1}
)
assert changed.status_code == 400, changed.text
assert len(recording_server.requests) == sent
changed_cutoff: Final = client.get(
"/v1/traces", params={"cursor": cursor, "as_of_ms": first_body["window"]["as_of_ms"] - 1}
)
assert changed_cutoff.status_code == 400, changed_cutoff.text
assert len(recording_server.requests) == sent
@pytest.mark.asyncio
@ -385,9 +398,9 @@ def test_trace_sql_endpoint_enforces_ownership_and_preserves_clickhouse_envelope
"rows": 1,
"statistics": {"elapsed": 0.01, "rows_read": 1, "bytes_read": 1},
}
recording_server.expected_requests = 12 if expected_status == 200 else 0
recording_server.expected_requests = 15 if expected_status == 200 else 0
if expected_status == 200:
for _ in range(11):
for _ in range(14):
recording_server.enqueue(ResponseSpec(body=""))
recording_server.enqueue(ResponseSpec(body=envelope))
storage: Final = ClickHouseStorage(TraceStorageConfig(recording_server.base_url, "trace_test"))
@ -405,7 +418,8 @@ def test_trace_sql_endpoint_enforces_ownership_and_preserves_clickhouse_envelope
result: Final = client.post("/v1/traces/query", json={"sql": "SELECT 42 AS answer"})
assert result.status_code == expected_status, result.text
if expected_status == 403:
assert result.json() == {"detail": "Not allowed to view logs"}
assert result.json()["detail"] == "Not allowed to view logs"
assert result.json()["code"] == "forbidden"
return
assert result.json() == envelope
assert recording_server.requests[-1].raw_body == b"SELECT 42 AS answer"
@ -424,13 +438,16 @@ def test_trace_help_endpoint_runs_native_schema_and_metadata_discovery(
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.tracing_endpoints import provide_receiver, provide_trace_query_secret, router
recording_server.expected_requests = 17
for _ in range(11):
recording_server.expected_requests = 23
for _ in range(14):
recording_server.enqueue(ResponseSpec(body=""))
for response in (
{"data": [{"name": "Model", "type": "String"}]},
{"data": []},
{"data": []},
{"data": []},
{"data": []},
{"data": []},
):
recording_server.enqueue(ResponseSpec(body=response))
metadata: Final = (
@ -463,8 +480,8 @@ def test_trace_help_endpoint_runs_native_schema_and_metadata_discovery(
"types": ["string"],
"expression": "JSONExtractRaw(metadata, 'custom', 'label')",
}
assert body["attributes"][0]["fields"][0]["expression"] == "SpanAttributes['custom.span']"
assert body["attributes"][1]["fields"][0]["expression"] == "ResourceAttributes['custom.resource']"
assert body["attributes"][0]["fields"][0]["expression"] == "span_attributes['custom.span']"
assert body["attributes"][1]["fields"][0]["expression"] == "resource_attributes['custom.resource']"
@pytest.mark.parametrize(
@ -494,8 +511,8 @@ def test_trace_sql_endpoint_distinguishes_query_errors_from_reader_failures(
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.tracing_endpoints import provide_receiver, provide_trace_query_secret, router
recording_server.expected_requests = 13
for _ in range(11):
recording_server.expected_requests = 16
for _ in range(14):
recording_server.enqueue(ResponseSpec(body=""))
recording_server.enqueue(ResponseSpec(status=clickhouse_status, body=body))
envelope: Final = {
@ -514,13 +531,13 @@ def test_trace_sql_endpoint_distinguishes_query_errors_from_reader_failures(
with TestClient(app) as client:
failed: Final = client.post("/v1/traces/query", json={"sql": "SELEC 42"})
assert failed.status_code == expected_status, failed.text
assert failed.json()["detail"]["database_code"] == database_code
assert failed.json().get("database_code") == database_code
assert (
failed.json()["detail"]["code"]
failed.json()["code"]
== {400: "query_rejected", 422: "query_limit_exceeded", 503: "query_unavailable"}[expected_status]
)
if database_code is not None:
assert failed.json()["detail"]["message"] == body.decode()
assert failed.json()["detail"] == body.decode()
recovered: Final = client.post("/v1/traces/query", json={"sql": "SELECT 42 AS answer"})
assert recovered.status_code == 200, recovered.text
assert recovered.json() == envelope
@ -609,7 +626,7 @@ def _fixture_trace_api(
def test_fixture_backed_help_examples_execute_through_query_api(seeded_trace_api: SeededTraceAPI) -> None:
api: Final = seeded_trace_api
assert {table.name for table in api.help.tables} == {"otel_traces", "spend_logs", "agent_traces_by_key"}
assert {"traces", "spans", "calls"} <= {table.name for table in api.help.tables}
assert api.help.metadata.error is None
assert api.help.metadata.sampled_rows == len(api.spends)
assert any(field.path == ("fixture_capture", "name") for field in api.help.metadata.fields)
@ -623,15 +640,13 @@ def test_fixture_backed_help_examples_execute_through_query_api(seeded_trace_api
assert recorded[0]["trace_id"] == api.spends[0]["trace_id"]
assert int(str(recorded[0]["requests"])) == len(api.spends)
assert math.isclose(float(str(recorded[0]["recorded_spend"])), total)
detail: Final = api.client.get(f"/v1/traces/{api.spends[0]['trace_id']}")
assert detail.status_code == 200, detail.text
assert math.isclose(TRACE.validate_json(detail.content)["summary"]["spend"] or 0, total)
detail: Final = _trace(api, api.spends[0]["trace_id"])
assert math.isclose(detail["summary"]["spend"] or 0, total)
unmatched: Final = api.query_example("LLM spans without a direct spend match")
assert unmatched
assert all(row["TraceId"] != api.spends[0]["trace_id"] for row in unmatched)
unpriced: Final = api.client.get(f"/v1/traces/{unmatched[0]['TraceId']}")
assert unpriced.status_code == 200, unpriced.text
assert unpriced.json()["summary"]["spend"] is None
assert all(row["trace_id"] != api.spends[0]["trace_id"] for row in unmatched)
unpriced: Final = _trace(api, str(unmatched[0]["trace_id"]))
assert unpriced["summary"]["spend"] is None
@pytest.mark.parametrize("spend", (None, 0.0, 0.125), ids=("unknown", "free", "paid"))
@ -710,9 +725,7 @@ def test_captured_sdk_cost_survives_seeding_and_is_queryable(name: str, captured
rows: Final = tuple(row for row in api.spends if fixture_capture("", row).name == name)
assert rows
capture: Final = fixture_capture(name, rows[0])
response: Final = api.client.get(f"/v1/traces/{capture.trace_id}")
assert response.status_code == 200, response.text
detail: Final = TRACE.validate_json(response.content)
detail: Final = _trace(api, capture.trace_id)
original: Final = span_rows((TRACE_FIXTURES / f"{name}.json").read_bytes(), "application/json")
assert detail["summary"]["span_count"] == len(original)
if capture.spend_linked and capture.spend_complete:
@ -775,9 +788,44 @@ def test_server_side_copies_keep_every_capture_linked_to_its_spend() -> None:
def _trace(api: SeededTraceAPI, trace_id: str) -> Trace:
response: Final = api.client.get(f"/v1/traces/{trace_id}")
listed: Final = api.client.get(
"/v1/traces",
params={
"q": f"trace_id:{trace_id}",
"start_ms": 0,
"end_ms": time.time_ns() // 1_000_000 + 86_400_000,
"page_size": 2,
},
)
assert listed.status_code == 200, listed.text
page: Final = _TRACE_PAGE.validate_json(listed.content)
(summary,) = page["data"]
assert summary["trace_id"] == trace_id
response: Final = api.client.get(f"/v1/traces/{summary['id']}")
assert response.status_code == 200, response.text
return TRACE.validate_json(response.content)
metadata: Final = TraceMetadata.model_validate_json(response.content)
first: Final = _span_page(api, summary["id"], None)
spans: Final = tuple(_trace_spans(api, summary["id"], first))
return TRACE.validate_python(
{
**metadata.model_dump(mode="json"),
"spans": tuple(span.model_dump(mode="json") for span in spans),
"next_cursor": None,
}
)
def _span_page(api: SeededTraceAPI, id: str, cursor: str | None) -> TraceSpansPage:
params: Final = {"page_size": 200} if cursor is None else {"page_size": 200, "cursor": cursor}
response: Final = api.client.get(f"/v1/traces/{id}/spans", params=params)
assert response.status_code == 200, response.text
return TraceSpansPage.model_validate_json(response.content)
def _trace_spans(api: SeededTraceAPI, id: str, page: TraceSpansPage) -> Iterator[SpanModel]:
yield from page.data
if page.next_cursor is not None:
yield from _trace_spans(api, id, _span_page(api, id, page.next_cursor))
def _assert_capture(api: SeededTraceAPI, name: str, rows: tuple[SpendLogRecord, ...], trace_id: str) -> None:

View file

@ -32,14 +32,16 @@ REF: Final = "A" * 64
def summary(trace_ref: str, trace_id: str = "trace", span_count: int = 1) -> TraceSummary:
return TraceSummary(
resolution_limited=False,
trace_id=trace_id,
trace_ref=trace_ref,
id=trace_ref,
name="run",
service="svc",
input_preview="",
start_time="2026-01-01T00:00:00Z",
duration_ms=1.0,
status="ok",
root_status="ok",
has_error=False,
span_count=span_count,
agent_count=0,
agent_invocations=0,
@ -49,6 +51,8 @@ def summary(trace_ref: str, trace_id: str = "trace", span_count: int = 1) -> Tra
input_tokens=0,
output_tokens=0,
models=(),
agent_names=(),
frameworks=(),
spend=None,
)
@ -86,7 +90,7 @@ class FakeStorage:
listed: list[tuple[QueryScope, str, str | None, int, RunOrder, tuple[str, ...]]] = field(default_factory=list)
def _matching(self, q: str, trace_refs: Sequence[str]) -> list[TraceSummary]:
return [run for run in self.runs if (not trace_refs or run.get("trace_ref") in trace_refs) and q in run["name"]]
return [run for run in self.runs if (not trace_refs or run["id"] in trace_refs) and q in run["name"]]
async def list_traces(
self,
@ -100,11 +104,12 @@ class FakeStorage:
trace_refs: Sequence[str] = (),
) -> TracePage:
self.listed.append((scope, q, cursor, limit, order, tuple(trace_refs)))
ordered: Final = sorted(self._matching(q, trace_refs), key=lambda run: run.get("trace_ref", ""))
after: Final = [run for run in ordered if cursor is None or run.get("trace_ref", "") > cursor][:limit]
ordered: Final = sorted(self._matching(q, trace_refs), key=lambda run: run["id"])
after: Final = [run for run in ordered if cursor is None or run["id"] > cursor][:limit]
return TracePage(
window={"start_ms": start_ms, "end_ms": end_ms, "as_of_ms": max(end_ms, 0)},
data=tuple(after),
next_cursor=after[-1].get("trace_ref") if len(after) == limit else None,
next_cursor=after[-1]["id"] if len(after) == limit else None,
)
async def count_traces(
@ -277,12 +282,49 @@ async def test_content_pages_forty_spans_at_a_time_in_trace_order() -> None:
@pytest.mark.asyncio
@pytest.mark.parametrize(("quote", "found"), (("time", True), ("boom", True), ("absent", False)))
async def test_evidence_must_appear_in_the_span_input_output_or_error(quote: str, found: bool) -> None:
storage: Final = FakeStorage(texts={("span", "output"): "timeout", ("span", "error"): "boom"})
@pytest.mark.parametrize(
("quote", "found", "input_text"),
(
("time", True, "hello"),
("boom", True, "hello"),
("Input: hello", True, "hello"),
("Status: error boom", True, "hello"),
("hello\nOutput: timeout", True, "hello"),
("timeout\nStatus: error boom", True, "hello"),
("Status: ok boom", False, "hello"),
("Input: absent", False, "hello"),
("absent", False, "hello"),
("x" * 1500 + "hello\nOutput: timeout\nStatus: error boom", True, "x" * (2 * BUDGET) + "hello"),
("x" * 1500 + "hello\nOutput: timeout\nStatus: ok boom", False, "x" * (2 * BUDGET) + "hello"),
),
ids=(
"raw_output",
"raw_error",
"input_label",
"error_status",
"input_output_boundary",
"output_status_boundary",
"wrong_status",
"missing_input",
"missing_quote",
"long_boundary",
"wrong_long_status",
),
)
async def test_evidence_must_appear_in_the_span_input_output_or_error(quote: str, found: bool, input_text: str) -> None:
storage: Final = FakeStorage(
spans=(span("span", status="error"),),
texts={("span", "input"): input_text, ("span", "output"): "timeout", ("span", "error"): "boom"},
)
execution: Final = Execution(id=execution_id(REF, "trace"), trace_id="trace", trace_ref=REF)
evidence: Final = Evidence(execution_id=execution.id, span_id="span", quote=quote)
assert await SourceReader(storage).verify_evidence(Scope(all_teams=True), execution, evidence) is found
reader: Final = SourceReader(storage)
if found:
shown: Final = await reader.content(
Scope(all_teams=True), execution, offset=max(len(input_text) - BUDGET // 2, 0)
)
assert quote in shown.parts[0].content
assert await reader.verify_evidence(Scope(all_teams=True), execution, evidence) is found
@pytest.mark.parametrize(

View file

@ -5,16 +5,20 @@ Tests for the agent tracing endpoints (litellm/proxy/tracing_endpoints.py).
from collections.abc import AsyncGenerator, Mapping
from contextlib import asynccontextmanager
from types import ModuleType
from typing import Final, Literal
from typing import Annotated, Final, Literal
from unittest.mock import AsyncMock, MagicMock
import pytest
from fastapi import FastAPI, HTTPException
from fastapi import Depends, FastAPI, HTTPException
from fastapi.routing import APIRoute
from fastapi.security import HTTPAuthorizationCredentials, HTTPBearer
from fastapi.testclient import TestClient
from jsonschema import validate
from pydantic import JsonValue, TypeAdapter
from litellm.constants import TRACE_READ_RETRY_AFTER_SECONDS
from litellm.proxy import tracing_endpoints
from litellm.proxy._types import LitellmUserRoles, ProxyLifespanState, UserAPIKeyAuth
from litellm.proxy._types import LitellmUserRoles, ProxyException, ProxyLifespanState, UserAPIKeyAuth
from litellm.proxy.auth.authorization import OwnedRows, ReadScope
from litellm.proxy.auth.authorization_dependencies import get_log_team_lookup
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
@ -22,6 +26,7 @@ from litellm.proxy.tracing_runtime import manage_tracing, provide_storage
from litellm.rust_bridge import loader
from litellm.rust_bridge.trace.errors import TraceChanged, TraceQueryError
from litellm.rust_bridge.trace.generated.models import TraceQueryHelp
from litellm.rust_bridge.trace.generated.routes import OPERATIONS
from litellm.rust_bridge.trace.generated.types import AllQueryScope, OwnedQueryScope, QueryScope
from litellm.rust_bridge.trace.queries import TraceSQLResponse
from litellm.rust_bridge.trace.storage import NEWEST, ClickHouseStorage, TraceStorageConfig
@ -74,7 +79,12 @@ TRACE_RESPONSE: Final = {
"input_preview": "",
"start_time": "2026-01-01T00:00:00Z",
"duration_ms": 0,
"status": "ok",
"root_status": "ok",
"has_error": False,
"id": "run-one",
"agent_names": [],
"frameworks": [],
"resolution_limited": False,
"span_count": 0,
"agent_count": 0,
"agent_invocations": 0,
@ -89,12 +99,13 @@ TRACE_RESPONSE: Final = {
"agents": [],
"spans": [],
}
TRACE_METADATA: Final = {"summary": TRACE_RESPONSE["summary"], "agents": TRACE_RESPONSE["agents"]}
TRACE_WINDOW: Final = {"start_ms": 1, "end_ms": 2, "as_of_ms": 2}
SPAN_DETAIL_RESPONSE: Final = {
"span_id": "s1",
"input": "",
"output": "",
"input_ui": {"kind": "text", "text": ""},
"output_ui": {"kind": "text", "text": ""},
"attributes": {},
}
@ -145,7 +156,7 @@ def test_trace_read_and_write_permissions(
receiver.list_traces.assert_not_awaited()
else:
receiver.list_traces.assert_awaited_once_with(
scope=scope, start_ms=1, end_ms=2, q="", cursor=None, order=NEWEST
scope=scope, start_ms=1, end_ms=2, q="", cursor=None, order=NEWEST, page_size=50, as_of_ms=None
)
write: Final = client.post("/v1/traces", json={})
@ -166,11 +177,13 @@ def test_trace_read_and_write_permissions(
def receiver(client) -> MagicMock:
fake = MagicMock()
fake.ingest = AsyncMock(return_value=1)
fake.list_traces = AsyncMock(return_value={"data": [], "next_cursor": None})
fake.trace_histogram = AsyncMock(return_value={"buckets": []})
fake.run_values = AsyncMock(return_value={"values": ["researcher"]})
fake.get_trace = AsyncMock(return_value=None)
fake.get_span = AsyncMock(return_value=None)
fake.list_traces = AsyncMock(return_value={"data": [], "next_cursor": None, "window": TRACE_WINDOW})
fake.trace_histogram = AsyncMock(return_value={"buckets": [], "window": TRACE_WINDOW})
fake.run_values = AsyncMock(return_value={"values": ["researcher"], "window": TRACE_WINDOW})
fake.get_trace_metadata = AsyncMock(return_value=None)
fake.get_span_by_id = AsyncMock(return_value=None)
fake.get_trace_spans = AsyncMock(return_value=None)
fake.get_span_error_by_id = AsyncMock(return_value=None)
client.app.dependency_overrides[tracing_endpoints.provide_receiver] = lambda: fake
return fake
@ -249,17 +262,19 @@ def test_post_too_large_is_413(client, receiver):
def test_list_traces_passes_scope_window_query_and_cursor(client, receiver):
response = client.get(
"/v1/traces", params={"start_ms": 1, "end_ms": 2, "q": "agent:research* -status:ok", "cursor": "abc"}
"/v1/traces", params={"start_ms": 1, "end_ms": 2, "q": "agent:research* -root_status:ok", "cursor": "abc"}
)
assert response.status_code == 200
assert response.json() == {"data": [], "next_cursor": None}
assert response.json() == {"data": [], "next_cursor": None, "window": TRACE_WINDOW}
receiver.list_traces.assert_awaited_once_with(
scope={"kind": "owned", "user_id": "user", "team_ids": ()},
start_ms=1,
end_ms=2,
q="agent:research* -status:ok",
q="agent:research* -root_status:ok",
cursor="abc",
order=NEWEST,
page_size=50,
as_of_ms=None,
)
@ -287,9 +302,9 @@ def test_list_traces_rejects_orders_the_runs_table_does_not_offer(client, receiv
def test_histogram_passes_scope_window_query_and_buckets(client, receiver):
response = client.get("/v1/traces/histogram", params={"start_ms": 1, "end_ms": 2, "q": "x", "buckets": 12})
assert response.status_code == 200, response.text
assert response.json() == {"buckets": []}
assert response.json() == {"buckets": [], "window": TRACE_WINDOW}
receiver.trace_histogram.assert_awaited_once_with(
{"kind": "owned", "user_id": "user", "team_ids": ()}, 1, 2, "x", 12
{"kind": "owned", "user_id": "user", "team_ids": ()}, 1, 2, "x", 12, None
)
@ -298,9 +313,9 @@ def test_values_pass_scope_window_query_field_and_needle(client, receiver):
"/v1/traces/values/agent", params={"start_ms": 1, "end_ms": 2, "q": "model:gpt*", "contains": "res"}
)
assert response.status_code == 200, response.text
assert response.json() == {"values": ["researcher"]}
assert response.json() == {"values": ["researcher"], "window": TRACE_WINDOW}
receiver.run_values.assert_awaited_once_with(
{"kind": "owned", "user_id": "user", "team_ids": ()}, 1, 2, "model:gpt*", "agent", "res", 20
{"kind": "owned", "user_id": "user", "team_ids": ()}, 1, 2, "model:gpt*", "agent", "res", 20, None
)
@ -364,30 +379,32 @@ def test_list_traces_preserves_omitted_bounds_for_cursor_window(
assert kwargs["q"] == ""
def test_get_trace_404_and_200(client, receiver):
def test_get_trace_metadata_404_and_200(client, receiver):
assert client.get("/v1/traces/missing").status_code == 404
receiver.get_trace.return_value = TRACE_RESPONSE
receiver.get_trace_metadata.return_value = TRACE_METADATA
response = client.get("/v1/traces/t1")
assert response.status_code == 200
assert response.json() == TRACE_RESPONSE
receiver.get_trace.assert_awaited_with("t1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "", None, None)
assert response.json() == TRACE_METADATA
receiver.get_trace_metadata.assert_awaited_with({"kind": "owned", "user_id": "user", "team_ids": ()}, "t1")
def test_get_span_404_and_200(client, receiver):
def test_get_span_by_id_404_and_200(client, receiver):
assert client.get("/v1/traces/t1/spans/s1").status_code == 404
receiver.get_span.return_value = SPAN_DETAIL_RESPONSE
receiver.get_span_by_id.return_value = SPAN_DETAIL_RESPONSE
response = client.get("/v1/traces/t1/spans/s1")
assert response.status_code == 200
assert response.json()["span_id"] == "s1"
receiver.get_span.assert_awaited_with("t1", "s1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "")
receiver.get_span_by_id.assert_awaited_with({"kind": "owned", "user_id": "user", "team_ids": ()}, "t1", "s1")
@pytest.mark.parametrize("suffix,cursor,page_size", [("", None, None), ("&cursor=next&page_size=200", "next", 200)])
def test_trace_detail_passes_scoped_reference(client, receiver, suffix, cursor, page_size):
receiver.get_trace.return_value = TRACE_RESPONSE
assert client.get(f"/v1/traces/t1?trace_ref=run-one{suffix}").status_code == 200
receiver.get_trace.assert_awaited_with(
"t1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "run-one", cursor, page_size
@pytest.mark.parametrize("query,cursor,page_size", [("", None, 100), ("?cursor=next&page_size=200", "next", 200)])
def test_span_page_passes_canonical_id_and_bounded_page_size(client, receiver, query, cursor, page_size):
receiver.get_trace_spans.return_value = {"data": [], "next_cursor": None}
response: Final = client.get(f"/v1/traces/run-one/spans{query}")
assert response.status_code == 200, response.text
assert response.json() == {"data": [], "next_cursor": None}
receiver.get_trace_spans.assert_awaited_with(
{"kind": "owned", "user_id": "user", "team_ids": ()}, "run-one", cursor, page_size
)
@ -395,9 +412,9 @@ def test_trace_detail_passes_scoped_reference(client, receiver, suffix, cursor,
"path,method",
(
("/v1/traces", "list_traces"),
("/v1/traces/t1", "get_trace"),
("/v1/traces/t1/spans/s1", "get_span"),
("/v1/traces/t1/spans/s1/error", "get_span_error"),
("/v1/traces/t1", "get_trace_metadata"),
("/v1/traces/t1/spans/s1", "get_span_by_id"),
("/v1/traces/t1/spans/s1/error", "get_span_error_by_id"),
),
)
@pytest.mark.parametrize(
@ -437,16 +454,17 @@ def test_read_failures_carry_a_code_per_kind_without_exposing_database_details(
getattr(receiver, method).side_effect = error
response: Final = client.get(path)
assert response.status_code == status
assert response.json() == {"detail": {"code": code, "message": message}}
assert response.headers["content-type"] == "application/problem+json"
assert (response.json()["code"], response.json()["detail"], response.json()["status"]) == (code, message, status)
retry_after: Final = response.headers.get("Retry-After")
assert (retry_after == str(TRACE_READ_RETRY_AFTER_SECONDS)) == (status == 503), retry_after
@pytest.mark.parametrize("query", ("page_size=0", "page_size=501", "cursor=" + "x" * 513))
def test_trace_page_rejects_unbounded_parameters(client: TestClient, receiver: MagicMock, query: str) -> None:
response: Final = client.get(f"/v1/traces/t1?{query}")
response: Final = client.get(f"/v1/traces/t1/spans?{query}")
assert response.status_code == 422
receiver.get_trace.assert_not_awaited()
receiver.get_trace_spans.assert_not_awaited()
def test_invalid_export_and_cursor_are_client_errors(client, receiver):
@ -489,9 +507,9 @@ def test_key_without_user_cannot_read_traces(client: TestClient, auth: UserAPIKe
storage.list_traces,
storage.trace_histogram,
storage.run_values,
storage.get_trace,
storage.get_span,
storage.get_span_error,
storage.get_trace_metadata,
storage.get_span_by_id,
storage.get_span_error_by_id,
):
read.assert_not_called()
storage.query_sql.assert_not_called()
@ -535,9 +553,11 @@ def test_disabled_receiver_precedes_read_scope_rejection(client: TestClient) ->
)
response: Final = client.get("/v1/traces")
assert response.status_code == 501
assert response.json() == {
"detail": "Agent tracing is not enabled. Set `tracing:` in general_settings and CLICKHOUSE_URL."
}
assert (
response.json()["detail"]
== "Agent tracing is not enabled. Set `tracing:` in general_settings and CLICKHOUSE_URL."
)
assert response.json()["code"] == "unavailable"
def test_injected_receiver_ingests_with_the_authenticated_tenant(client: TestClient) -> None:
@ -563,9 +583,9 @@ def test_injected_receiver_ingests_with_the_authenticated_tenant(client: TestCli
def test_lifespan_receivers_are_app_local() -> None:
first_storage: Final = MagicMock(spec=ClickHouseStorage)
first_storage.get_span = AsyncMock(return_value={**SPAN_DETAIL_RESPONSE, "span_id": "first-span"})
first_storage.get_span_by_id = AsyncMock(return_value={**SPAN_DETAIL_RESPONSE, "span_id": "first-span"})
second_storage: Final = MagicMock(spec=ClickHouseStorage)
second_storage.get_span = AsyncMock(return_value={**SPAN_DETAIL_RESPONSE, "span_id": "second-span"})
second_storage.get_span_by_id = AsyncMock(return_value={**SPAN_DETAIL_RESPONSE, "span_id": "second-span"})
first_receiver: Final = TraceReceiver(first_storage)
second_receiver: Final = TraceReceiver(second_storage)
first_storage.ensure_schema = AsyncMock()
@ -592,9 +612,9 @@ def test_lifespan_receivers_are_app_local() -> None:
with TestClient(first_app) as first_client:
with TestClient(second_app) as second_client:
second_response: Final = second_client.get("/v1/traces/t1/spans/second-span?trace_ref=second-run")
simultaneous: Final = first_client.get("/v1/traces/t1/spans/first-span?trace_ref=first-run")
first_response: Final = first_client.get("/v1/traces/t1/spans/first-span?trace_ref=first-run")
second_response: Final = second_client.get("/v1/traces/second-run/spans/second-span")
simultaneous: Final = first_client.get("/v1/traces/first-run/spans/first-span")
first_response: Final = first_client.get("/v1/traces/first-run/spans/first-span")
assert simultaneous.json() == first_response.json()
first_storage.ensure_schema.assert_awaited_once()
second_storage.ensure_schema.assert_awaited_once()
@ -603,9 +623,9 @@ def test_lifespan_receivers_are_app_local() -> None:
assert first_response.json()["span_id"] == "first-span"
assert second_response.json()["span_id"] == "second-span"
scope: Final = OwnedQueryScope(kind="owned", user_id=TEAM_KEY.user_id or "", team_ids=())
assert first_storage.get_span.await_count == 2
first_storage.get_span.assert_awaited_with("t1", "first-span", scope, "first-run")
second_storage.get_span.assert_awaited_once_with("t1", "second-span", scope, "second-run")
assert first_storage.get_span_by_id.await_count == 2
first_storage.get_span_by_id.assert_awaited_with(scope, "first-run", "first-span")
second_storage.get_span_by_id.assert_awaited_once_with(scope, "second-run", "second-span")
@pytest.mark.parametrize("auth", [TEAM_KEY, UserAPIKeyAuth(user_role=LitellmUserRoles.INTERNAL_USER)])
@ -613,7 +633,7 @@ def test_query_validation_precedes_trace_access_checks(client: TestClient, auth:
client.app.dependency_overrides[user_api_key_auth] = lambda: auth
response: Final = client.get("/v1/traces", params={"start_ms": "invalid"})
assert response.status_code == 422
assert response.json()["detail"][0]["loc"] == ["query", "start_ms"]
assert response.json()["errors"][0]["location"] == "query/start_ms"
@pytest.mark.parametrize("enabled", [True, False])
@ -715,7 +735,7 @@ def test_sql_and_help_use_authenticated_scope(
result: Final = client.post("/v1/traces/query", json={"sql": "SELECT * FROM otel_traces"})
assert result.status_code == 200, result.text
assert result.json() == SQL_ENVELOPE
receiver.storage.query_sql.assert_awaited_once_with("SELECT * FROM otel_traces", expected_scope, "test-secret")
receiver.storage.query_sql.assert_awaited_once_with("SELECT * FROM otel_traces", expected_scope, "test-secret", {})
help_result: Final = client.get("/v1/traces/query/help")
assert help_result.status_code == 200, help_result.text
assert help_result.json() == QUERY_HELP
@ -725,6 +745,114 @@ def test_sql_and_help_use_authenticated_scope(
assert receiver.storage.query_sql.await_count == 1
@pytest.mark.parametrize(
"path",
(
"/v1/traces",
"/v1/traces/histogram",
"/v1/traces/values/agent",
"/v1/traces/run-one",
"/v1/traces/run-one/spans",
"/v1/traces/run-one/spans/s1",
"/v1/traces/run-one/spans/s1/error",
"/v1/traces/query/help",
),
)
def test_trace_reads_reject_unknown_query_fields(client: TestClient, receiver: MagicMock, path: str) -> None:
client.app.dependency_overrides[tracing_endpoints.provide_trace_query_secret] = lambda: "test-secret"
response: Final = client.get(path, params={"trace_ref": "forged"})
assert response.status_code == 422, response.text
assert response.headers["content-type"] == "application/problem+json"
assert response.json()["code"] == "invalid_request"
assert response.json()["errors"][0]["location"] == "query/trace_ref"
receiver.get_trace_metadata.assert_not_awaited()
receiver.list_traces.assert_not_awaited()
def test_merged_openapi_validates_responses_and_preserves_other_authenticated_routes(
client: TestClient, receiver: MagicMock
) -> None:
security: Final = HTTPBearer(scheme_name="OtherBearer")
@client.app.get("/other", response_model=str)
def other(credentials: Annotated[HTTPAuthorizationCredentials, Depends(security)]) -> str:
return credentials.credentials
actual: Final = frozenset(
(route.path, frozenset(route.methods), route.operation_id)
for route in client.app.routes
if isinstance(route, APIRoute) and route.operation_id in OPERATIONS
)
expected: Final = frozenset(
(operation.path, frozenset((operation.method,)), operation.operation_id) for operation in OPERATIONS.values()
)
assert actual == expected
client.app.openapi_schema = tracing_endpoints.merge_trace_openapi(client.app.openapi())
object_adapter: Final = TypeAdapter(dict[str, JsonValue])
document: Final = object_adapter.validate_python(client.get("/openapi.json").json())
paths: Final = object_adapter.validate_python(document["paths"])
components: Final = object_adapter.validate_python(document["components"])
schemes: Final = object_adapter.validate_python(components["securitySchemes"])
assert {"OtherBearer", "TraceBearer"} <= schemes.keys()
assert client.get("/other", headers={"Authorization": "Bearer visible"}).json() == "visible"
assert client.get("/other").status_code in (401, 403)
receiver.get_trace_metadata.return_value = TRACE_METADATA
response: Final = client.get("/v1/traces/run-one")
assert response.status_code == 200, response.text
operation: Final = object_adapter.validate_python(object_adapter.validate_python(paths["/v1/traces/{id}"])["get"])
success: Final = object_adapter.validate_python(object_adapter.validate_python(operation["responses"])["200"])
content: Final = object_adapter.validate_python(success["content"])
response_schema: Final = object_adapter.validate_python(
object_adapter.validate_python(content["application/json"])["schema"]
)
validate(object_adapter.validate_python(response.json()), {**response_schema, "components": components})
assert response.json()["summary"]["id"] == "run-one"
rejected: Final = client.get("/v1/traces/run-one?unsupported=value")
failure: Final = object_adapter.validate_python(object_adapter.validate_python(operation["responses"])["422"])
failure_content: Final = object_adapter.validate_python(failure["content"])
problem_schema: Final = object_adapter.validate_python(
object_adapter.validate_python(failure_content["application/problem+json"])["schema"]
)
validate(object_adapter.validate_python(rejected.json()), {**problem_schema, "components": components})
assert rejected.status_code == 422
@pytest.mark.parametrize("framework_error", (False, True))
def test_trace_auth_failures_are_problem_responses(client: TestClient, framework_error: bool) -> None:
def unauthorized() -> UserAPIKeyAuth:
if framework_error:
raise HTTPException(401, "Invalid token", headers={"WWW-Authenticate": "Bearer"})
raise ProxyException("Invalid token", "authentication_error", None, 401, {"WWW-Authenticate": "Bearer"})
client.app.dependency_overrides[user_api_key_auth] = unauthorized
response: Final = client.get("/v1/traces")
assert response.status_code == 401, response.text
assert response.headers["content-type"] == "application/problem+json"
assert response.headers["www-authenticate"] == "Bearer"
assert response.json()["code"] == "unauthorized"
assert response.json()["detail"] == "Invalid token"
def test_sql_parameters_are_typed_and_unknown_query_fields_are_rejected(
client: TestClient, receiver: MagicMock
) -> None:
client.app.dependency_overrides[tracing_endpoints.provide_trace_query_secret] = lambda: "test-secret"
receiver.storage.query_sql = AsyncMock(return_value=TraceSQLResponse.model_validate(SQL_ENVELOPE))
params: Final = {"service": "quoted ' label", "count": 3, "ids": ["a", "b"], "active": True, "empty": None}
query: Final = "SELECT {service:String}, {count:UInt64}, {ids:Array(String)}, {active:Bool}"
response: Final = client.post("/v1/traces/query", json={"sql": query, "params": params})
assert response.status_code == 200, response.text
assert response.json() == SQL_ENVELOPE
receiver.storage.query_sql.assert_awaited_once_with(
query, OwnedQueryScope(kind="owned", user_id="user", team_ids=()), "test-secret", params
)
invalid: Final = client.post("/v1/traces/query", json={"sql": query, "params": {"ids": {"nested": "value"}}})
assert invalid.status_code == 422, invalid.text
unknown: Final = client.post("/v1/traces/query?database=forged", json={"sql": query})
assert unknown.status_code == 422, unknown.text
assert receiver.storage.query_sql.await_count == 1
@pytest.mark.parametrize("auth", (UserAPIKeyAuth(), UserAPIKeyAuth(team_id="a", project_id="p")))
def test_sql_rejects_missing_identity_without_querying(
client: TestClient, receiver: MagicMock, auth: UserAPIKeyAuth
@ -789,9 +917,12 @@ def test_sql_reports_rejected_queries_and_unavailable_readers(
receiver.storage.query_sql = AsyncMock(side_effect=error)
result: Final = client.post("/v1/traces/query", json={"sql": "SELECT 1"})
assert result.status_code == status, result.text
assert result.json()["detail"] == detail
assert result.headers["content-type"] == "application/problem+json"
assert result.json()["code"] == detail["code"]
assert result.json()["detail"] == detail["message"]
assert result.json().get("database_code") == detail["database_code"]
receiver.storage.query_sql.assert_awaited_once_with(
"SELECT 1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "test-secret"
"SELECT 1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "test-secret", {}
)
@ -821,7 +952,7 @@ def test_queries_require_a_proxy_secret(
return
assert result.status_code == 200, result.text
receiver.storage.query_sql.assert_awaited_once_with(
"SELECT 1", {"kind": "owned", "user_id": "user", "team_ids": ()}, secret
"SELECT 1", {"kind": "owned", "user_id": "user", "team_ids": ()}, secret, {}
)
@ -847,7 +978,7 @@ def test_shared_trace_permissions_reach_read_and_sql_boundaries(
team_lookup: Final = AsyncMock(side_effect=lookup)
storage: Final = MagicMock(spec=ClickHouseStorage)
storage.get_span = AsyncMock(return_value=SPAN_DETAIL_RESPONSE)
storage.get_span_by_id = AsyncMock(return_value=SPAN_DETAIL_RESPONSE)
storage.query_sql = AsyncMock(return_value=TraceSQLResponse.model_validate(SQL_ENVELOPE))
storage.query_help = AsyncMock(return_value=TraceQueryHelp.model_validate(QUERY_HELP))
client.app.dependency_overrides[user_api_key_auth] = lambda: auth
@ -855,7 +986,7 @@ def test_shared_trace_permissions_reach_read_and_sql_boundaries(
client.app.dependency_overrides[tracing_endpoints.provide_receiver] = lambda: TraceReceiver(storage)
client.app.dependency_overrides[tracing_endpoints.provide_trace_query_secret] = lambda: "test-secret"
response: Final = client.get("/v1/traces/t1/spans/s1?trace_ref=run-one")
response: Final = client.get("/v1/traces/run-one/spans/s1")
assert response.status_code == 200, response.text
assert response.json()["span_id"] == "s1"
query_scope: Final = (
@ -867,12 +998,12 @@ def test_shared_trace_permissions_reach_read_and_sql_boundaries(
"team_ids": expected[2],
}
)
storage.get_span.assert_awaited_once_with("t1", "s1", query_scope, "run-one")
storage.get_span_by_id.assert_awaited_once_with(query_scope, "run-one", "s1")
sql_response: Final = client.post("/v1/traces/query", json={"sql": "SELECT * FROM otel_traces"})
assert sql_response.status_code == 200, sql_response.text
assert sql_response.json() == SQL_ENVELOPE
assert client.get("/v1/traces/query/help").json() == QUERY_HELP
storage.query_sql.assert_awaited_once_with("SELECT * FROM otel_traces", query_scope, "test-secret")
storage.query_sql.assert_awaited_once_with("SELECT * FROM otel_traces", query_scope, "test-secret", {})
storage.query_help.assert_awaited_once_with(query_scope, "test-secret")
assert team_lookup.await_count == (
3
@ -962,7 +1093,7 @@ async def test_storage_validates_the_native_query_help_value(monkeypatch: pytest
"scope": "bounded sample",
}
},
{"tables": [{"name": "traces", "columns": [{"name": "value", "type": "String"}]}]},
{"tables": [{"name": "unknown_table", "columns": [{"name": "value", "type": "String"}]}]},
{"unexpected": True},
),
)

View file

@ -30,11 +30,9 @@ def test_dictionary_validation_keeps_required_nullable_and_optional_fields_disti
"input": "",
"output": "",
"attributes": {"key": "value"},
"input_ui": {"kind": "messages", "messages": [{"role": "user", "content": "hello"}]},
"output_ui": {"kind": "text", "text": "answer"},
}
)
assert result["input_ui"] == {"kind": "messages", "messages": ({"role": "user", "content": "hello"},)}
assert (result["input"], result["output"]) == ("", "")
assert result["attributes"] == {"key": "value"}
assert (
TypeAdapter(SpanErrorPage).validate_python(

View file

@ -78,6 +78,7 @@ describe("Lens demo data", () => {
expect(trace.summary.span_count).toBe(trace.spans.length);
expect(trace.summary.agent_names).toContain(trace.agents[0].name);
expect(trace.summary.error_count).toBe(trace.spans.filter((span) => span.status === "error").length);
expect(trace.summary.has_error).toBe(trace.summary.error_count > 0);
for (const span of trace.spans) {
expect(span.start_offset_ms + span.duration_ms).toBeLessThanOrEqual(trace.summary.duration_ms);
}
@ -90,7 +91,8 @@ describe("Lens demo data", () => {
expect(run.trace.spans).toHaveLength(362);
expect(ids.size).toBe(362);
expect(run.trace.summary.error_count).toBe(3);
expect(run.trace.summary.status).toBe("ok");
expect(run.trace.summary.root_status).toBe("ok");
expect(run.trace.summary.has_error).toBe(true);
for (const span of run.trace.spans) {
if (span.parent_span_id) expect(ids.has(span.parent_span_id)).toBe(true);
expect(run.details.find((detail) => detail.span_id === span.span_id)).toBeDefined();

View file

@ -55,6 +55,7 @@ export function demoHistogram(runs: readonly TraceSummary[], range: TimeWindow,
agent: traceAgentNames(run)[0] ?? run.service,
}));
return {
window: { start_ms: range.startMs, end_ms: range.endMs, as_of_ms: range.endMs },
buckets: Array.from({ length: buckets }, (_, index) => {
const hits = placed.filter((run) => run.index === index);
const agents = [...new Set(hits.filter((run) => !run.failed).map((run) => run.agent))].sort();
@ -73,7 +74,7 @@ export function demoHistogram(runs: readonly TraceSummary[], range: TimeWindow,
}
function demoTracesApi(data: LensDemoData): TracesApi {
const run = (traceId: string) => data.runs.find(({ trace }) => trace.summary.trace_id === traceId);
const run = (traceId: string) => data.runs.find(({ trace }) => trace.summary.id === traceId);
const summaries = data.runs.map((item) => item.trace.summary);
const matching = (range: TimeWindow, q: string) =>
filterRuns(
@ -91,6 +92,7 @@ function demoTracesApi(data: LensDemoData): TracesApi {
list: async ({ selection, order }) => ({
data: orderRuns(matching(selection.window, selection.q), order),
next_cursor: null,
window: { start_ms: selection.window.startMs, end_ms: selection.window.endMs, as_of_ms: selection.window.endMs },
}),
histogram: async ({ window, q }, buckets) => demoHistogram(matching(window, q), window, buckets),
values: async (field, contains, range) => {

Some files were not shown because too many files have changed in this diff Show more