diff --git a/litellm-rust/Cargo.lock b/litellm-rust/Cargo.lock index f7b667c8ab2..96089010a49 100644 --- a/litellm-rust/Cargo.lock +++ b/litellm-rust/Cargo.lock @@ -4458,6 +4458,7 @@ dependencies = [ "base64 0.22.1", "criterion", "indexmap 2.14.0", + "jsonschema", "litellm-llms-types", "macro_rules_attribute", "opentelemetry-proto", @@ -4469,6 +4470,7 @@ dependencies = [ "strum", "thiserror 2.0.19", "time", + "utoipa", ] [[package]] @@ -7977,6 +7979,29 @@ version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6c140620e7ffbb22c2dee59cafe6084a59b5ffc27a8859a5f0d494b5d52b6be" +[[package]] +name = "utoipa" +version = "5.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8bde15df68e80b16c7d16b9616e80770ad158988daa56a27dccd1e55558b0160" +dependencies = [ + "indexmap 2.14.0", + "serde", + "serde_json", + "utoipa-gen", +] + +[[package]] +name = "utoipa-gen" +version = "5.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ba0b99ee52df3028635d93840c797102da61f8a7bb3cf751032455895b52ef8" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "uuid" version = "1.24.0" diff --git a/litellm-rust/Cargo.toml b/litellm-rust/Cargo.toml index b0766f11e87..7aaa2febeeb 100644 --- a/litellm-rust/Cargo.toml +++ b/litellm-rust/Cargo.toml @@ -83,6 +83,7 @@ pyo3-async-runtimes = { version = "0.29.0", features = ["tokio-runtime"] } rand = "0.8" macro_rules_attribute = "0.2.3" schemars = "1" +utoipa = { version = "5.5.0", default-features = false, features = ["macros"] } reqwest = { version = "0.12", default-features = false, features = ["json", "multipart", "rustls-tls", "http2", "stream"] } qdrant-client = { version = "1.19.0", default-features = false } uuid = { version = "1", features = ["v4"] } diff --git a/litellm-rust/crates/python-bridge/src/routes/traces.rs b/litellm-rust/crates/python-bridge/src/routes/traces.rs index 7816fd4ab59..04fa846ab18 100644 --- a/litellm-rust/crates/python-bridge/src/routes/traces.rs +++ b/litellm-rust/crates/python-bridge/src/routes/traces.rs @@ -7,6 +7,7 @@ use std::{ use litellm_http::ClientVariant; use litellm_traces::{ QueryScope, Tenant, + api::TraceQueryWindow, search::{RunField, RunFilter, RunSearch}, store::{RunOrder, SpanPart, TextRange}, }; @@ -98,13 +99,38 @@ fn map_read_error(error: ReadError) -> PyErr { } } -fn run_filter(start_ms: i64, end_ms: i64, q: &str, trace_refs: Vec) -> RunFilter { - RunFilter { +fn run_window( + start_ms: Option, + end_ms: Option, + as_of_ms: Option, + cursor: Option<&str>, +) -> PyResult { + let now_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_millis() as u64; + resolve_run_window( start_ms, end_ms, - search: RunSearch::parse(q), + as_of_ms, + cursor, + TraceQueryWindow { + start_ms: now_ms as i64 - 86_400_000, + end_ms: now_ms as i64, + as_of_ms: now_ms.saturating_sub(1), + }, + ) + .map_err(map_read_error) +} + +fn run_filter(window: TraceQueryWindow, q: &str, trace_refs: Vec) -> PyResult { + Ok(RunFilter { + start_ms: window.start_ms, + end_ms: window.end_ms, + as_of_ms: window.as_of_ms, + search: RunSearch::parse(q).map_err(|error| PyValueError::new_err(error.to_string()))?, trace_refs, - } + }) } fn parsed(kind: &str, value: &str) -> PyResult { @@ -278,7 +304,7 @@ impl NativeTraceStorage { ) } - #[pyo3(signature = (scope, start_ms, end_ms, q, cursor, limit, order, trace_refs=Vec::new()))] + #[pyo3(signature = (scope, start_ms, end_ms, q, cursor, limit, order, trace_refs=Vec::new(), as_of_ms=None))] #[expect( clippy::too_many_arguments, reason = "one parameter per Python argument" @@ -294,19 +320,10 @@ impl NativeTraceStorage { limit: u32, #[pyo3(from_py_with = litellm_host_python::from_py_argument)] order: RunOrder, trace_refs: Vec, + as_of_ms: Option, ) -> PyResult> { - let now_ms = SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap_or_default() - .as_millis() as i64; - let window = resolve_run_window( - start_ms, - end_ms, - cursor.as_deref(), - (now_ms - 86_400_000, now_ms), - ) - .map_err(map_read_error)?; - let filter = run_filter(window.0, window.1, q, trace_refs); + let window = run_window(start_ms, end_ms, as_of_ms, cursor.as_deref())?; + let filter = run_filter(window, q, trace_refs)?; let page = PageRequest { cursor, limit }; let client = crate::http::host_client(py, ClientVariant::NoRedirect)?; let connection = self.config.storage().reader().clone(); @@ -323,17 +340,23 @@ impl NativeTraceStorage { ) } - #[pyo3(signature = (scope, start_ms, end_ms, q, trace_refs=Vec::new()))] + #[pyo3(signature = (scope, start_ms, end_ms, q, trace_refs=Vec::new(), as_of_ms=None))] + #[expect( + clippy::too_many_arguments, + reason = "one parameter per Python argument" + )] fn count_traces<'py>( &self, py: Python<'py>, #[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope, - start_ms: i64, - end_ms: i64, + start_ms: Option, + end_ms: Option, q: &str, trace_refs: Vec, + as_of_ms: Option, ) -> PyResult> { - let filter = run_filter(start_ms, end_ms, q, trace_refs); + let window = run_window(start_ms, end_ms, as_of_ms, None)?; + let filter = run_filter(window, q, trace_refs)?; let client = crate::http::host_client(py, ClientVariant::NoRedirect)?; let connection = self.config.storage().reader().clone(); let reader = Arc::clone(&self.reader); @@ -389,16 +412,23 @@ impl NativeTraceStorage { ) } + #[pyo3(signature = (scope, start_ms, end_ms, q, buckets, as_of_ms=None))] + #[expect( + clippy::too_many_arguments, + reason = "one parameter per Python argument" + )] fn trace_histogram<'py>( &self, py: Python<'py>, #[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope, - start_ms: i64, - end_ms: i64, + start_ms: Option, + end_ms: Option, q: &str, buckets: u32, + as_of_ms: Option, ) -> PyResult> { - let filter = run_filter(start_ms, end_ms, q, Vec::new()); + let window = run_window(start_ms, end_ms, as_of_ms, None)?; + let filter = run_filter(window, q, Vec::new())?; let client = crate::http::host_client(py, ClientVariant::NoRedirect)?; let connection = self.config.storage().reader().clone(); let reader = Arc::clone(&self.reader); @@ -416,21 +446,24 @@ impl NativeTraceStorage { clippy::too_many_arguments, reason = "one parameter per Python argument" )] + #[pyo3(signature = (scope, start_ms, end_ms, q, field, contains, limit, as_of_ms=None))] fn run_values<'py>( &self, py: Python<'py>, #[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope, - start_ms: i64, - end_ms: i64, + start_ms: Option, + end_ms: Option, q: &str, field: &str, contains: &str, limit: u32, + as_of_ms: Option, ) -> PyResult> { let field = field .parse::() .map_err(|_| PyValueError::new_err(format!("unknown run field {field}")))?; - let filter = run_filter(start_ms, end_ms, q, Vec::new()); + let window = run_window(start_ms, end_ms, as_of_ms, None)?; + let filter = run_filter(window, q, Vec::new())?; let contains = contains.to_owned(); let client = crate::http::host_client(py, ClientVariant::NoRedirect)?; let connection = self.config.storage().reader().clone(); @@ -542,12 +575,103 @@ impl NativeTraceStorage { ) } + fn get_trace_metadata<'py>( + &self, + py: Python<'py>, + #[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope, + id: String, + ) -> PyResult> { + let client = crate::http::host_client(py, ClientVariant::NoRedirect)?; + let connection = self.config.storage().reader().clone(); + let reader = Arc::clone(&self.reader); + crate::execution::run_async( + py, + async move { + let store = ClickHouseTraces::new(client, connection); + reader.get_trace_metadata(&store, &scope, &id).await + }, + map_read_error, + ) + } + + #[pyo3(signature = (scope, id, cursor, page_size))] + fn get_trace_spans<'py>( + &self, + py: Python<'py>, + #[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope, + id: String, + cursor: Option, + page_size: u32, + ) -> PyResult> { + let client = crate::http::host_client(py, ClientVariant::NoRedirect)?; + let connection = self.config.storage().reader().clone(); + let reader = Arc::clone(&self.reader); + crate::execution::run_async( + py, + async move { + let store = ClickHouseTraces::new(client, connection); + reader + .get_trace_spans(&store, &scope, &id, cursor.as_deref(), page_size) + .await + }, + map_read_error, + ) + } + + fn get_span_by_id<'py>( + &self, + py: Python<'py>, + #[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope, + id: String, + span_id: String, + ) -> PyResult> { + let client = crate::http::host_client(py, ClientVariant::NoRedirect)?; + let connection = self.config.storage().reader().clone(); + let reader = Arc::clone(&self.reader); + crate::execution::run_async( + py, + async move { + let store = ClickHouseTraces::new(client, connection); + reader.get_span_by_id(&store, &scope, &id, &span_id).await + }, + map_read_error, + ) + } + + #[pyo3(signature = (scope, id, span_id, cursor=None))] + fn get_span_error_by_id<'py>( + &self, + py: Python<'py>, + #[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope, + id: String, + span_id: String, + cursor: Option, + ) -> PyResult> { + let client = crate::http::host_client(py, ClientVariant::NoRedirect)?; + let connection = self.config.storage().reader().clone(); + let reader = Arc::clone(&self.reader); + crate::execution::run_async( + py, + async move { + let store = ClickHouseTraces::new(client, connection); + reader + .get_span_error_by_id(&store, &scope, &id, &span_id, cursor.as_deref()) + .await + }, + map_read_error, + ) + } + fn query_sql<'py>( &self, py: Python<'py>, sql: String, #[pyo3(from_py_with = litellm_host_python::from_py_argument)] scope: QueryScope, secret: String, + #[pyo3(from_py_with = litellm_host_python::from_py_argument)] params: BTreeMap< + String, + litellm_traces::api::SqlParameter, + >, ) -> PyResult> { if sql.trim().is_empty() { return Err(map_error( @@ -561,7 +685,13 @@ impl NativeTraceStorage { async move { let _permit = readers.acquire()?; let connection = readers.connection(&client, &scope, &secret).await?; - litellm_traces_clickhouse::query_sql(&client, &connection, &sql).await + litellm_traces_clickhouse::query_sql_with_params( + &client, + &connection, + &sql, + ¶ms, + ) + .await }, map_sql_error, ) diff --git a/litellm-rust/crates/storage-clickhouse/src/read.rs b/litellm-rust/crates/storage-clickhouse/src/read.rs index 85b5c758e94..0153fbce9a6 100644 --- a/litellm-rust/crates/storage-clickhouse/src/read.rs +++ b/litellm-rust/crates/storage-clickhouse/src/read.rs @@ -25,6 +25,8 @@ pub enum Parameter { Integer(i64), Unsigned(u64), Float(f64), + Boolean(bool), + Null, Strings(Vec), } @@ -35,6 +37,8 @@ impl Parameter { Self::Integer(value) => value.to_string(), Self::Unsigned(value) => value.to_string(), Self::Float(value) => value.to_string(), + Self::Boolean(value) => u8::from(*value).to_string(), + Self::Null => "\\N".to_owned(), Self::Strings(values) => format!( "[{}]", values diff --git a/litellm-rust/crates/storage-clickhouse/tests/transport.rs b/litellm-rust/crates/storage-clickhouse/tests/transport.rs index b6a1dff4e11..be5f09ae300 100644 --- a/litellm-rust/crates/storage-clickhouse/tests/transport.rs +++ b/litellm-rust/crates/storage-clickhouse/tests/transport.rs @@ -38,6 +38,8 @@ struct QueryParams { signed: i64, unsigned: u64, float: f64, + boolean: bool, + nullable: Option, text: String, strings: Vec, } @@ -79,6 +81,8 @@ async fn typed_fetch_encodes_parameters_and_validates_rows( .and(query_param("param_signed", i64::MIN.to_string())) .and(query_param("param_unsigned", u64::MAX.to_string())) .and(query_param("param_float", "12.5")) + .and(query_param("param_boolean", "1")) + .and(query_param("param_nullable", "\\N")) .and(query_param("param_text", "line\\nbreak")) .and(query_param("param_strings", "['a\\'b','雪']")) .and(query_param("readonly", "1")) @@ -93,6 +97,8 @@ async fn typed_fetch_encodes_parameters_and_validates_rows( signed: i64::MIN, unsigned: u64::MAX, float: 12.5, + boolean: true, + nullable: None, text: "line\nbreak".into(), strings: vec!["a'b".into(), "雪".into()], }; diff --git a/litellm-rust/crates/traces-cache/src/cache.rs b/litellm-rust/crates/traces-cache/src/cache.rs index 681e04bf18e..617b5c5d3cd 100644 --- a/litellm-rust/crates/traces-cache/src/cache.rs +++ b/litellm-rust/crates/traces-cache/src/cache.rs @@ -45,7 +45,7 @@ impl SnapshotKey { pub(crate) fn run( source: &str, access: &QueryScope, - run: (&str, &str, &str, &str), + run: (&str, &str, &str, &str, u64), ) -> Result { Self::digest(&("run", source, access, run)) } @@ -55,7 +55,7 @@ impl SnapshotKey { access: &QueryScope, filter: &RunFilter, ) -> Result { - Self::digest(&("run_page_v1", source, access, filter)).map(|key| key.0) + Self::digest(&("run_page_v2", source, access, filter)).map(|key| key.0) } pub(crate) fn scope(source: &str, access: &QueryScope) -> Result { diff --git a/litellm-rust/crates/traces-cache/src/cursor.rs b/litellm-rust/crates/traces-cache/src/cursor.rs index d5e5acdc7a1..14a8f0bdc03 100644 --- a/litellm-rust/crates/traces-cache/src/cursor.rs +++ b/litellm-rust/crates/traces-cache/src/cursor.rs @@ -1,4 +1,5 @@ use base64::{Engine, engine::general_purpose::URL_SAFE}; +use litellm_traces::api::TraceQueryWindow; use litellm_traces::store::{RunCursor, RunOrder, RunRow, SpanPart}; use serde::{Deserialize, Serialize}; @@ -45,7 +46,7 @@ impl Cursor { pub(super) struct RunPosition { order: RunOrder, query_scope: String, - window: (i64, i64), + window: TraceQueryWindow, value: i64, trace_ref: String, } @@ -55,7 +56,7 @@ impl RunPosition { order: RunOrder, row: &RunRow, query_scope: &str, - window: (i64, i64), + window: TraceQueryWindow, ) -> Self { let RunCursor { value, trace_ref } = order.cursor(row); Self { @@ -73,6 +74,7 @@ impl RunPosition { pub(super) struct SpanPosition { pub(super) trace_ref: String, pub(super) snapshot_ms: u64, + pub(super) page_size: u32, pub(super) offset: usize, pub(super) version: String, } @@ -89,7 +91,7 @@ pub(super) fn run_position( cursor: Option<&str>, order: RunOrder, query_scope: &str, - window: (i64, i64), + window: TraceQueryWindow, ) -> Result, ReadError> { let Some(cursor) = cursor.filter(|cursor| !cursor.is_empty()) else { return Ok(None); @@ -113,15 +115,17 @@ pub(super) fn run_position( pub fn resolve_run_window( start_ms: Option, end_ms: Option, + as_of_ms: Option, cursor: Option<&str>, - default_window: (i64, i64), -) -> Result<(i64, i64), ReadError> { + default: TraceQueryWindow, +) -> Result> { let Some(cursor) = cursor.filter(|cursor| !cursor.is_empty()) else { - let window = ( - start_ms.unwrap_or(default_window.0), - end_ms.unwrap_or(default_window.1), - ); - return if window.0 < window.1 { + let window = TraceQueryWindow { + start_ms: start_ms.unwrap_or(default.start_ms), + end_ms: end_ms.unwrap_or(default.end_ms), + as_of_ms: as_of_ms.unwrap_or(default.as_of_ms), + }; + return if window.start_ms < window.end_ms && window.as_of_ms <= default.as_of_ms { Ok(window) } else { Err(ReadError::InvalidParameters) @@ -130,13 +134,16 @@ pub fn resolve_run_window( let Cursor::Run(position) = Cursor::decode(cursor, "trace")? else { return Err(ReadError::InvalidCursor("trace")); }; - if position.window.0 >= position.window.1 - || start_ms.is_some_and(|start| start != position.window.0) - || end_ms.is_some_and(|end| end != position.window.1) + let window = position.window; + if window.start_ms >= window.end_ms + || window.as_of_ms > default.as_of_ms + || start_ms.is_some_and(|start| start != window.start_ms) + || end_ms.is_some_and(|end| end != window.end_ms) + || as_of_ms.is_some_and(|cutoff| cutoff != window.as_of_ms) { return Err(ReadError::InvalidCursor("trace")); } - Ok(position.window) + Ok(window) } pub(super) fn span_position(cursor: &str) -> Result> { @@ -174,7 +181,14 @@ mod tests { use super::*; - const WINDOW: (i64, i64) = (10, 100); + const WINDOW: TraceQueryWindow = window(10, 100, 150); + const fn window(start_ms: i64, end_ms: i64, as_of_ms: u64) -> TraceQueryWindow { + TraceQueryWindow { + start_ms, + end_ms, + as_of_ms, + } + } fn run(order: RunOrder, value: i64, trace_ref: &str) -> String { Cursor::Run(RunPosition { @@ -200,6 +214,7 @@ mod tests { Cursor::Span(SpanPosition { trace_ref: "ref".into(), snapshot_ms: 1, + page_size: 2, offset: 2, version: "A".repeat(64), }) @@ -237,11 +252,12 @@ mod tests { #[case::order(BY_ERRORS, "query", WINDOW)] #[case::direction(RunOrder { descending: false, ..RunOrder::NEWEST }, "query", WINDOW)] #[case::scope(RunOrder::NEWEST, "other-query", WINDOW)] - #[case::window(RunOrder::NEWEST, "query", (11, 100))] + #[case::window(RunOrder::NEWEST, "query", window(11, 100, 150))] + #[case::cutoff(RunOrder::NEWEST, "query", window(10, 100, 151))] fn run_cursor_rejects_a_changed_query( #[case] order: RunOrder, #[case] scope: &str, - #[case] window: (i64, i64), + #[case] window: TraceQueryWindow, ) { assert!(matches!( run_position::( @@ -290,41 +306,94 @@ mod tests { #[case] expected: (i64, i64), ) { assert_eq!( - resolve_run_window::(start, end, None, (0, 50)).unwrap(), + { + let resolved = resolve_run_window::( + start, + end, + None, + None, + window(0, 50, 150), + ) + .unwrap(); + assert_eq!(resolved.as_of_ms, 150); + (resolved.start_ms, resolved.end_ms) + }, expected ); } #[rstest] #[case::omitted(None, None)] - #[case::start(Some(WINDOW.0), None)] - #[case::end(None, Some(WINDOW.1))] - #[case::explicit(Some(WINDOW.0), Some(WINDOW.1))] + #[case::start(Some(WINDOW.start_ms), None)] + #[case::end(None, Some(WINDOW.end_ms))] + #[case::explicit(Some(WINDOW.start_ms), Some(WINDOW.end_ms))] fn cursor_keeps_its_window_when_the_default_clock_advances( #[case] start: Option, #[case] end: Option, ) { let cursor = run(RunOrder::NEWEST, 1, "ref"); assert_eq!( - resolve_run_window::(start, end, Some(&cursor), (200, 300)).unwrap(), + resolve_run_window::( + start, + end, + None, + Some(&cursor), + window(200, 300, 350) + ) + .unwrap(), WINDOW ); } #[rstest] - #[case::start(Some(WINDOW.0 + 1), None)] - #[case::end(None, Some(WINDOW.1 + 1))] + #[case::start(Some(WINDOW.start_ms + 1), None)] + #[case::end(None, Some(WINDOW.end_ms + 1))] fn cursor_rejects_explicit_window_changes( #[case] start: Option, #[case] end: Option, ) { let cursor = run(RunOrder::NEWEST, 1, "ref"); assert!(matches!( - resolve_run_window::(start, end, Some(&cursor), (200, 300)), + resolve_run_window::( + start, + end, + None, + Some(&cursor), + window(200, 300, 350) + ), Err(ReadError::InvalidCursor("trace")) )); } + #[rstest] + fn cursor_rejects_a_changed_ingestion_cutoff() { + let cursor = run(RunOrder::NEWEST, 1, "ref"); + assert!(matches!( + resolve_run_window::( + None, + None, + Some(WINDOW.as_of_ms + 1), + Some(&cursor), + window(200, 300, 350) + ), + Err(ReadError::InvalidCursor("trace")) + )); + } + + #[rstest] + fn first_page_rejects_a_future_ingestion_cutoff() { + assert!(matches!( + resolve_run_window::( + None, + None, + Some(WINDOW.as_of_ms + 1), + None, + WINDOW + ), + Err(ReadError::InvalidParameters) + )); + } + #[rstest] fn span_cursor_round_trips() { let position = span_position::(&span()).unwrap(); diff --git a/litellm-rust/crates/traces-cache/src/list.rs b/litellm-rust/crates/traces-cache/src/list.rs index 6d5eddf88d4..48cdb99780f 100644 --- a/litellm-rust/crates/traces-cache/src/list.rs +++ b/litellm-rust/crates/traces-cache/src/list.rs @@ -8,7 +8,7 @@ use litellm_traces::{ use crate::{ ReadError, SnapshotKey, TraceReader, TraceStore, cache::{Freshness, ListedRun}, - reader::{map_store_error, now_ms, spans}, + reader::{map_store_error, now_ms, settle, spans}, spend::{spend, spend_window, spend_within}, store::StoreError, }; @@ -27,6 +27,7 @@ fn cache_key( source: &str, access: &QueryScope, row: &RunRow, + as_of_ms: u64, ) -> Result> { Ok(SnapshotKey::run( source, @@ -36,6 +37,7 @@ fn cache_key( &row.api_key_hash, &row.trace_id, &row.trace_ref, + as_of_ms, ), )?) } @@ -50,6 +52,8 @@ fn summary(row: &RunRow, listed: Option<&ListedRun>) -> TraceSummary { duration_ms: row.duration_ns as f64 / 1_000_000.0, span_count: row.span_count, error_count: row.error_count, + has_error: row.error_count > 0, + status: row.status, ..summary } } @@ -61,11 +65,12 @@ pub(super) async fn list_summaries( store: &S, access: &QueryScope, runs: &[RunRow], + as_of_ms: u64, ) -> Result, ReadError> { let mut keys = Vec::with_capacity(runs.len()); let mut listed = Vec::with_capacity(runs.len()); for row in runs { - let key = cache_key(store.source(), access, row)?; + let key = cache_key(store.source(), access, row, as_of_ms)?; listed.push(reader.lists.runs.get(&key).await); keys.push(key); } @@ -74,7 +79,7 @@ pub(super) async fn list_summaries( .zip(&listed) .filter_map(|(row, listed)| listed.is_none().then_some(row)) .collect(); - let mut resolved = resolve_runs(reader, store, access, &misses) + let mut resolved = resolve_runs(reader, store, access, &misses, as_of_ms) .await? .into_iter(); let mut summaries = Vec::with_capacity(runs.len()); @@ -101,6 +106,7 @@ async fn resolve_runs( store: &S, access: &QueryScope, runs: &[&RunRow], + snapshot_ms: u64, ) -> Result>, ReadError> { let (Some(start_ms), Some(end_ms)) = ( runs.iter().map(|row| row.start_ms).min(), @@ -119,24 +125,23 @@ async fn resolve_runs( trace_refs: runs.iter().map(|row| row.trace_ref.clone()).collect(), window: start_ms..end_ms.saturating_add(1), }; - let snapshot_ms = now_ms(); let spans = match spans(store, access, selection, snapshot_ms).await { Ok(spans) => spans, Err(StoreError::TooLarge) => { let mut resolved = Vec::with_capacity(runs.len()); for row in runs { - resolved.push(resolve_run(reader, store, access, row).await?); + resolved.push(resolve_run(reader, store, access, row, snapshot_ms).await?); } return Ok(resolved); } Err(error) => return Err(map_store_error(error)), }; - let Some(spend_rows) = spend(store, access, &spans).await else { + let Some(spend_rows) = spend(store, access, &spans, snapshot_ms).await else { // The batch's combined spend read failed; a run's own narrower window may still // resolve, so fall back per run instead of leaving every run in the batch costless. let mut resolved = Vec::with_capacity(runs.len()); for row in runs { - resolved.push(resolve_run(reader, store, access, row).await?); + resolved.push(resolve_run(reader, store, access, row, snapshot_ms).await?); } return Ok(resolved); }; @@ -148,7 +153,7 @@ async fn resolve_runs( &right.api_key_hash, &right.trace_id, )) - .then(left.start_ns.cmp(&right.start_ns)) + .then((left.start_ns, &left.span_id).cmp(&(right.start_ns, &right.span_id))) }); let by_run: HashMap<_, &[SpanRow]> = spans .chunk_by(|left, right| { @@ -174,7 +179,7 @@ async fn resolve_runs( resolve_trace(&row.trace_id, &row.trace_ref, spans, spend).map(|trace| { ListedRun::Resolved( Box::new(trace.summary), - Freshness::of(spans, true, snapshot_ms), + Freshness::of(spans, true, now_ms()), ) }) }) @@ -186,11 +191,13 @@ async fn resolve_run( store: &S, access: &QueryScope, row: &RunRow, + as_of_ms: u64, ) -> Result, ReadError> { - match reader - .current(store, access, &row.trace_id, &row.trace_ref) - .await - { + match settle( + reader + .pinned(store, access, &row.trace_id, &row.trace_ref, as_of_ms) + .await, + ) { Ok(snapshot) => Ok(snapshot.map(|snapshot| { ListedRun::Resolved( Box::new(snapshot.trace().summary.clone()), diff --git a/litellm-rust/crates/traces-cache/src/reader.rs b/litellm-rust/crates/traces-cache/src/reader.rs index 0b679c5cbdc..60313af6b8d 100644 --- a/litellm-rust/crates/traces-cache/src/reader.rs +++ b/litellm-rust/crates/traces-cache/src/reader.rs @@ -10,7 +10,6 @@ use litellm_traces::{ CountBy, CountValue, RunCountQuery, RunOrder, RunQuery, RunSelection, SpanPart, SpanQuery, SpanRow, SpanSelection, SpanText, SpanTextQuery, TextRange, }, - to_ui_content, }; use crate::{ @@ -50,7 +49,7 @@ impl From for Miss { } } -fn settle(result: Result>>) -> Result, ReadError> { +pub(super) fn settle(result: Result>>) -> Result, ReadError> { match result { Ok(value) => Ok(Some(value)), Err(miss) => match &*miss { @@ -87,12 +86,7 @@ impl TraceReader { return Err(ReadError::InvalidParameters); } let query_scope = SnapshotKey::run_page_scope(store.source(), access, filter)?; - let after = run_position( - page.cursor.as_deref(), - order, - &query_scope, - (filter.start_ms, filter.end_ms), - )?; + let after = run_position(page.cursor.as_deref(), order, &query_scope, filter.window())?; let scope = SnapshotKey::scope(store.source(), access)?; let accepted = self.lists.limits.get(&scope).await.unwrap_or(u32::MAX); let mut page_size = page.limit.min(500).min(accepted); @@ -119,18 +113,23 @@ impl TraceReader { order, last, &query_scope, - (filter.start_ms, filter.end_ms), + filter.window(), )) .encode() }); let data = { let mut summaries = Vec::with_capacity(rows.len()); for batch in run_batches(&rows) { - summaries.extend(list_summaries(self, store, access, batch).await?); + summaries + .extend(list_summaries(self, store, access, batch, filter.as_of_ms).await?); } summaries }; - Ok(TracePage { data, next_cursor }) + Ok(TracePage { + data, + next_cursor, + window: filter.window(), + }) } pub async fn histogram( @@ -157,7 +156,7 @@ impl TraceReader { .run_counts(access, &query) .await .map_err(map_store_error)?; - Ok(histogram(&rows, filter.start_ms, filter.end_ms, buckets)) + Ok(histogram(&rows, filter.window(), buckets)) } pub async fn values( @@ -186,6 +185,7 @@ impl TraceReader { .await .map_err(map_store_error)?; Ok(RunValues { + window: filter.window(), values: rows.into_iter().map(|row| row.value).collect(), }) } @@ -245,6 +245,76 @@ impl TraceReader { .map_err(map_store_error) } + pub async fn get_trace_metadata( + &self, + store: &S, + access: &QueryScope, + id: &str, + ) -> Result, ReadError> { + let Some(row) = run_by_id(store, access, id).await? else { + return Ok(None); + }; + Ok(self + .current(store, access, &row.trace_id, id) + .await? + .map(|snapshot| { + let trace = snapshot.trace(); + litellm_traces::api::TraceMetadata { + summary: trace.summary.clone(), + agents: trace.agents.clone(), + } + })) + } + + pub async fn get_trace_spans( + &self, + store: &S, + access: &QueryScope, + id: &str, + cursor: Option<&str>, + page_size: u32, + ) -> Result, ReadError> { + let Some(row) = run_by_id(store, access, id).await? else { + return Ok(None); + }; + Ok(self + .get_trace_page(store, access, &row.trace_id, id, cursor, page_size) + .await? + .map(|trace| litellm_traces::api::TraceSpansPage { + data: trace.spans, + next_cursor: trace.next_cursor, + })) + } + + pub async fn get_span_by_id( + &self, + store: &S, + access: &QueryScope, + id: &str, + span_id: &str, + ) -> Result, ReadError> { + let Some(row) = run_by_id(store, access, id).await? else { + return Ok(None); + }; + self.get_span(store, access, &row.trace_id, span_id, id) + .await + } + + pub async fn get_span_error_by_id( + &self, + store: &S, + access: &QueryScope, + id: &str, + span_id: &str, + cursor: Option<&str>, + ) -> Result, ReadError> { + let Some(row) = run_by_id(store, access, id).await? else { + return Ok(None); + }; + self.get_span_error(store, access, &row.trace_id, span_id, id, cursor) + .await + } + pub async fn get_trace( &self, store: &S, @@ -283,13 +353,17 @@ impl TraceReader { let position = SpanPosition { trace_ref, snapshot_ms: snapshot.snapshot_ms(), + page_size, offset: 0, version: snapshot.version().to_owned(), }; return page(&snapshot, &position, page_size, self.response_bytes).map(Some); }; let position = span_position(cursor)?; - if position.trace_ref != trace_ref || position.snapshot_ms == 0 { + if position.trace_ref != trace_ref + || position.snapshot_ms == 0 + || position.page_size != page_size + { return Err(ReadError::InvalidCursor("span")); } let Some(snapshot) = settle( @@ -325,7 +399,7 @@ impl TraceReader { ) } - async fn pinned( + pub(super) async fn pinned( &self, store: &S, access: &QueryScope, @@ -344,7 +418,7 @@ impl TraceReader { let rows = spans(store, access, selection, snapshot_ms) .await .map_err(|error| Miss::Read(map_store_error(error)))?; - let spend_rows = spend(store, access, &rows).await; + let spend_rows = spend(store, access, &rows, snapshot_ms).await; let freshness = Freshness::of(&rows, spend_rows.is_some(), snapshot_ms); resolve_trace( trace_id, @@ -392,8 +466,6 @@ impl TraceReader { output }; Ok(Some(SpanDetail { - input_ui: to_ui_content(&input.text), - output_ui: to_ui_content(&output), span_id: span_id.to_owned(), input: input.text, output, @@ -552,10 +624,38 @@ pub(super) async fn spans( async move { store.spans(access, &query).await } }) .await?; - rows.sort_by_key(|row| row.start_ns); + rows.sort_by(|left, right| { + (left.start_ns, &left.span_id).cmp(&(right.start_ns, &right.span_id)) + }); Ok(rows) } +async fn run_by_id( + store: &S, + access: &QueryScope, + id: &str, +) -> Result, ReadError> { + if id.len() != 64 + || !id + .bytes() + .all(|ch| ch.is_ascii_digit() || (b'A'..=b'F').contains(&ch)) + { + return Err(ReadError::InvalidParameters); + } + let query = RunQuery { + selection: RunSelection::TraceRef(id.to_owned()), + order: RunOrder::BY_REFERENCE, + after: None, + limit: 1, + }; + Ok(store + .runs(access, &query) + .await + .map_err(map_store_error)? + .into_iter() + .find(|row| row.trace_ref == id)) +} + async fn reference( store: &S, access: &QueryScope, @@ -595,6 +695,7 @@ fn page( Cursor::Span(SpanPosition { trace_ref: position.trace_ref.clone(), snapshot_ms: position.snapshot_ms, + page_size: position.page_size, offset: end, version: snapshot.version().to_owned(), }) diff --git a/litellm-rust/crates/traces-cache/src/spend.rs b/litellm-rust/crates/traces-cache/src/spend.rs index ffcac16f34e..333b1a0b8d3 100644 --- a/litellm-rust/crates/traces-cache/src/spend.rs +++ b/litellm-rust/crates/traces-cache/src/spend.rs @@ -33,6 +33,7 @@ pub(super) async fn spend( store: &S, access: &QueryScope, rows: &[SpanRow], + as_of_ms: u64, ) -> Option> { let lookup = SpendLookup::new(rows); let Some(window) = spend_window(rows) else { @@ -43,6 +44,7 @@ pub(super) async fn spend( } let calls = read_all(|after, limit| { let query = CallQuery { + as_of_ms, window: window.clone(), response_ids: lookup.response_ids.clone(), request_ids: lookup.request_ids.clone(), diff --git a/litellm-rust/crates/traces-cache/tests/read.rs b/litellm-rust/crates/traces-cache/tests/read.rs index f32d5dd025e..d5138f50961 100644 --- a/litellm-rust/crates/traces-cache/tests/read.rs +++ b/litellm-rust/crates/traces-cache/tests/read.rs @@ -342,7 +342,7 @@ fn window(start_ms: i64, end_ms: i64, q: &str) -> RunFilter { RunFilter { start_ms, end_ms, - search: RunSearch::parse(q), + search: RunSearch::parse(q).unwrap(), ..Default::default() } } @@ -462,6 +462,33 @@ async fn pages_reuse_one_trace_snapshot_and_concatenate_in_order() { assert_eq!(store.calls(Operation::TraceSpans), 1); } +#[rstest] +#[case::smaller(1)] +#[case::larger(3)] +#[tokio::test] +async fn span_cursors_reject_changed_page_sizes(#[case] page_size: u32) { + let store = FakeStore::with_spans("ref", (0..5).map(span).collect()); + let reader = TraceReader::new(usize::MAX); + let access = access(); + let first = reader + .get_trace_page(&store, &access, "trace", "ref", None, 2) + .await + .unwrap() + .unwrap(); + let result = reader + .get_trace_page( + &store, + &access, + "trace", + "ref", + first.next_cursor.as_deref(), + page_size, + ) + .await; + assert!(matches!(result, Err(ReadError::InvalidCursor("span")))); + assert_eq!(store.calls(Operation::TraceSpans), 1); +} + #[rstest] #[tokio::test] async fn snapshot_versions_are_stable_across_readers_and_detect_changes() { @@ -1068,7 +1095,7 @@ async fn values_narrow_by_the_search_and_the_needle() { runs: 1, }, ]; - let filter = window(0, 10, "-status:error"); + let filter = window(0, 10, "-has_error:true"); let values = TraceReader::new(usize::MAX) .values(&store, &access(), &filter, RunField::Agent, "re_", 20) .await @@ -1197,6 +1224,8 @@ async fn cached_run_keeps_all_canonical_metrics_current_for_every_sort(#[case] k duration_ms: changed.duration_ns as f64 / 1_000_000.0, span_count: changed.span_count, error_count: changed.error_count, + has_error: changed.error_count > 0, + status: changed.status, ..first.data[0].clone() }; assert_eq!(expected.spend, Some(1.5)); diff --git a/litellm-rust/crates/traces-clickhouse/Cargo.toml b/litellm-rust/crates/traces-clickhouse/Cargo.toml index b90ad8a7bf8..9c072c390f8 100644 --- a/litellm-rust/crates/traces-clickhouse/Cargo.toml +++ b/litellm-rust/crates/traces-clickhouse/Cargo.toml @@ -41,3 +41,8 @@ wiremock.workspace = true name = "export-traces-clickhouse-schema" path = "src/bin/export_schema.rs" required-features = ["schema"] + +[[bin]] +name = "export-traces-openapi" +path = "src/bin/export_openapi.rs" +required-features = ["schema"] diff --git a/litellm-rust/crates/traces-clickhouse/migrations/0016_spans.sql b/litellm-rust/crates/traces-clickhouse/migrations/0016_spans.sql new file mode 100644 index 00000000000..69cb326d605 --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/migrations/0016_spans.sql @@ -0,0 +1,42 @@ +CREATE VIEW IF NOT EXISTS {database}.spans +SQL SECURITY INVOKER +AS SELECT + TeamId AS team_id, + ApiKeyHash AS api_key_hash, + UserId AS user_id, + TraceId AS trace_id, + hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) AS trace_ref, + SpanId AS span_id, + ParentSpanId AS parent_span_id, + SpanName AS name, + SpanKind AS span_kind, + ServiceName AS service, + Timestamp AS start_time, + Duration AS duration_ns, + Duration / 1000000.0 AS duration_ms, + multiIf(StatusCode = 'STATUS_CODE_ERROR', 'error', StatusCode = 'STATUS_CODE_OK', 'ok', 'unset') AS status, + StatusCode AS status_code, + StatusMessage AS status_message, + ObservationType AS observation_type, + AgentName AS agent_name, + Framework AS framework, + Model AS model, + InputTokens AS input_tokens, + OutputTokens AS output_tokens, + ResourceAttributes AS resource_attributes, + SpanAttributes AS span_attributes, + AgentMetadata AS agent_metadata, + Input AS input, + Output AS output, + InputPreview AS input_preview, + WrapperCandidate AS wrapper_candidate, + LiteLLMRequestId AS request_id, + CallKeys AS call_keys, + CallEvidence AS call_evidence, + ToolCallId AS tool_call_id, + EngineReceivedMs AS ingested_at_ms +FROM ( + SELECT * FROM {database}.otel_traces + ORDER BY Timestamp, EngineReceivedMs, StatusMessage, Duration, StatusCode + LIMIT 1 BY TeamId, ApiKeyHash, TraceId, SpanId +) diff --git a/litellm-rust/crates/traces-clickhouse/migrations/0017_calls.sql b/litellm-rust/crates/traces-clickhouse/migrations/0017_calls.sql new file mode 100644 index 00000000000..a519f7f7d65 --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/migrations/0017_calls.sql @@ -0,0 +1,35 @@ +CREATE VIEW IF NOT EXISTS {database}.calls +SQL SECURITY INVOKER +AS SELECT + team_id, + api_key AS api_key_hash, + user AS user_id, + request_id, + response_id, + litellm_call_id, + trace_id, + span_id, + call_type, + model, + model_group, + custom_llm_provider AS provider, + spend, + prompt_tokens AS input_tokens, + completion_tokens AS output_tokens, + total_tokens, + cache_read_tokens, + cache_write_tokens, + start_time, + end_time, + completion_start_time, + dateDiff('millisecond', start_time, end_time) AS duration_ms, + status, + error_str AS error, + cache_hit, + session_id, + request_tags, + metadata, + messages, + response, + EngineReceivedMs AS ingested_at_ms +FROM {database}.spend_logs FINAL diff --git a/litellm-rust/crates/traces-clickhouse/migrations/0018_traces.sql b/litellm-rust/crates/traces-clickhouse/migrations/0018_traces.sql new file mode 100644 index 00000000000..8ccde9ff7de --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/migrations/0018_traces.sql @@ -0,0 +1,38 @@ +CREATE VIEW IF NOT EXISTS {database}.traces +SQL SECURITY INVOKER +AS SELECT + team_id, + api_key_hash, + trace_id, + any(trace_ref) AS id, + if(uniqExact(user_id) = 1, any(user_id), '') AS user_id, + if(countIf(parent_span_id = '') > 0, + argMinIf(s.name, tuple(s.start_time, span_id), parent_span_id = ''), + argMin(s.name, tuple(s.start_time, span_id))) AS name, + argMin(s.service, tuple(s.start_time, span_id)) AS service, + coalesce(nullIf(if(countIf(parent_span_id = '') > 0, + argMinIf(s.input_preview, tuple(s.start_time, span_id), parent_span_id = ''), + argMin(s.input_preview, tuple(s.start_time, span_id))), ''), + argMinIf(s.input_preview, tuple(s.start_time, span_id), + s.input_preview != '' AND observation_type IN ('agent', 'llm'))) AS input_preview, + if(countIf(parent_span_id = '') > 0, + argMinIf(status, tuple(s.start_time, span_id), parent_span_id = ''), + argMin(status, tuple(s.start_time, span_id))) AS root_status, + min(s.start_time) AS start_time, + max(s.start_time + toIntervalNanosecond(s.duration_ns)) AS end_time, + toUInt64(greatest(toUnixTimestamp64Nano(end_time) - toUnixTimestamp64Nano(start_time), 0)) AS duration_ns, + duration_ns / 1000000.0 AS duration_ms, + count() AS span_count, + countIf(status = 'error') AS error_count, + CAST(error_count > 0, 'Bool') AS has_error, + countIf(observation_type = 'agent') AS agent_span_count, + arraySort(groupUniqArrayIf(if(agent_name = '', s.name, agent_name), agent_name != '' OR observation_type = 'agent')) AS agent_names, + length(agent_names) AS agent_label_count, + arraySort(groupUniqArrayIf(framework, framework != '')) AS frameworks, + countIf(observation_type = 'llm') AS llm_span_count, + countIf(observation_type = 'tool') AS tool_span_count, + sum(s.input_tokens) AS span_input_tokens, + sum(s.output_tokens) AS span_output_tokens, + arraySort(groupUniqArrayIf(model, model != '')) AS models +FROM {database}.spans AS s +GROUP BY team_id, api_key_hash, trace_id diff --git a/litellm-rust/crates/traces-clickhouse/query/calls.sql b/litellm-rust/crates/traces-clickhouse/query/calls.sql index 3ec431c9ad6..99a4eb439bc 100644 --- a/litellm-rust/crates/traces-clickhouse/query/calls.sql +++ b/litellm-rust/crates/traces-clickhouse/query/calls.sql @@ -9,7 +9,8 @@ FROM ( extract(tryBase64Decode(substring(response_id, 6)), 'response_id:([^;]+)'), '') AS upstream_response_id FROM owned_calls - WHERE start_time >= fromUnixTimestamp64Milli({start_ms:Int64}) + WHERE EngineReceivedMs <= {as_of_ms:UInt64} + AND start_time >= fromUnixTimestamp64Milli({start_ms:Int64}) AND start_time < fromUnixTimestamp64Milli({end_ms:Int64}) ) WHERE response_id IN {response_ids:Array(String)} diff --git a/litellm-rust/crates/traces-clickhouse/query/canonical_spans.sql b/litellm-rust/crates/traces-clickhouse/query/canonical_spans.sql new file mode 100644 index 00000000000..fa5dcb229e1 --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/query/canonical_spans.sql @@ -0,0 +1,24 @@ +candidate_runs AS ( + SELECT TeamId, ApiKeyHash, TraceId + FROM owned_runs + WHERE ({trace_id:String} != '' AND TraceId = {trace_id:String}) + OR ({trace_ref:String} != '' AND hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) = {trace_ref:String}) + GROUP BY TeamId, ApiKeyHash, TraceId + UNION ALL + SELECT TeamId, ApiKeyHash, TraceId + FROM owned_spans + WHERE {trace_id:String} = '' AND {trace_ref:String} = '' + AND EngineReceivedMs <= {as_of_ms:UInt64} + AND Timestamp >= fromUnixTimestamp64Milli({start_ms:Int64}) + AND Timestamp < fromUnixTimestamp64Milli({end_ms:Int64}) + GROUP BY TeamId, ApiKeyHash, TraceId +), +canonical_spans AS ( + SELECT * FROM owned_spans + WHERE EngineReceivedMs <= {as_of_ms:UInt64} + AND Timestamp >= (SELECT min(StartTs) FROM owned_runs WHERE (TeamId, ApiKeyHash, TraceId) IN (SELECT TeamId, ApiKeyHash, TraceId FROM candidate_runs)) + AND (TeamId, ApiKeyHash, TraceId) IN (SELECT TeamId, ApiKeyHash, TraceId FROM candidate_runs) + AND (TeamId, ApiKeyHash, TraceId) IN (SELECT TeamId, ApiKeyHash, TraceId FROM owned_runs GROUP BY TeamId, ApiKeyHash, TraceId) + ORDER BY Timestamp, EngineReceivedMs, StatusMessage, Duration, StatusCode + LIMIT 1 BY TeamId, ApiKeyHash, TraceId, SpanId +) diff --git a/litellm-rust/crates/traces-clickhouse/query/help/correlated_calls.sql b/litellm-rust/crates/traces-clickhouse/query/help/correlated_calls.sql index c4a6932c39b..fc6e28d7239 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/correlated_calls.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/correlated_calls.sql @@ -1,14 +1,15 @@ -SELECT t.TraceId, t.SpanId, s.request_id, s.spend, s.metadata -FROM otel_traces AS t +SELECT t.trace_id AS trace_id, t.span_id AS span_id, + s.request_id AS request_id, s.spend AS spend, s.metadata AS metadata +FROM spans AS t INNER JOIN ( SELECT * - FROM spend_logs FINAL + FROM calls WHERE start_time >= now() - INTERVAL 1 DAY ) AS s - ON t.LiteLLMRequestId = s.response_id - AND t.TeamId = s.team_id - AND ((t.UserId != '' AND t.UserId = s.user) - OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key)) -WHERE t.Timestamp >= now() - INTERVAL 1 DAY - AND t.LiteLLMRequestId != '' + ON t.request_id = s.response_id + AND t.team_id = s.team_id + AND ((t.user_id != '' AND t.user_id = s.user_id) + OR (t.api_key_hash != '' AND t.api_key_hash = s.api_key_hash)) +WHERE t.start_time >= now() - INTERVAL 1 DAY + AND t.request_id != '' LIMIT 100 diff --git a/litellm-rust/crates/traces-clickhouse/query/help/custom_metadata.sql b/litellm-rust/crates/traces-clickhouse/query/help/custom_metadata.sql index 6b6ff531349..4c10687a7cb 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/custom_metadata.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/custom_metadata.sql @@ -1,6 +1,6 @@ SELECT request_id, response_id, model, spend, JSONExtractString(metadata, 'project') AS project -FROM spend_logs FINAL +FROM calls WHERE start_time >= now() - INTERVAL 1 DAY AND JSONHas(metadata, 'project') AND JSONExtractString(metadata, 'project') = 'example' diff --git a/litellm-rust/crates/traces-clickhouse/query/help/discover_keys.sql b/litellm-rust/crates/traces-clickhouse/query/help/discover_keys.sql index 4e09539adb5..a944736396b 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/discover_keys.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/discover_keys.sql @@ -1,6 +1,6 @@ SELECT DISTINCT arrayJoin(JSONExtractKeys(metadata)) AS key -FROM spend_logs FINAL +FROM calls WHERE start_time >= now() - INTERVAL 30 DAY ORDER BY key LIMIT 200 diff --git a/litellm-rust/crates/traces-clickhouse/query/help/failed_spans.sql b/litellm-rust/crates/traces-clickhouse/query/help/failed_spans.sql index 4d989acbece..9fd28792405 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/failed_spans.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/failed_spans.sql @@ -1,12 +1,6 @@ -SELECT TeamId AS team, ApiKeyHash AS api_key, TraceId AS trace_id, - SpanId AS span_id, StatusMessage AS message -FROM ( - SELECT * - FROM otel_traces - WHERE Timestamp >= now() - INTERVAL 1 DAY - ORDER BY Timestamp, EngineReceivedMs, StatusMessage, Duration, StatusCode - LIMIT 1 BY TeamId, ApiKeyHash, TraceId, SpanId -) -WHERE StatusCode = 'STATUS_CODE_ERROR' -ORDER BY Timestamp DESC, team, api_key, trace_id, span_id +SELECT team_id AS team, api_key_hash AS api_key, trace_id, + span_id, status_message AS message +FROM spans +WHERE start_time >= now() - INTERVAL 1 DAY AND status = 'error' +ORDER BY start_time DESC, team, api_key, trace_id, span_id LIMIT 100 diff --git a/litellm-rust/crates/traces-clickhouse/query/help/metadata_filter.sql b/litellm-rust/crates/traces-clickhouse/query/help/metadata_filter.sql index d4106586a6a..406face3de5 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/metadata_filter.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/metadata_filter.sql @@ -1,6 +1,6 @@ -SELECT team_id AS team, api_key, request_id, spend, +SELECT team_id AS team, api_key_hash AS api_key, request_id, spend, JSONExtractString(metadata, 'labels', 'priority') AS priority -FROM spend_logs FINAL +FROM calls WHERE start_time >= now() - INTERVAL 1 DAY AND JSONHas(metadata, 'labels', 'priority') AND JSONExtractString(metadata, 'labels', 'priority') = 'high' diff --git a/litellm-rust/crates/traces-clickhouse/query/help/model_spend.sql b/litellm-rust/crates/traces-clickhouse/query/help/model_spend.sql index e8911480b22..477d3c2f5a8 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/model_spend.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/model_spend.sql @@ -7,9 +7,9 @@ FROM ( team_id, model, count() AS requests, countIf(isNull(spend) OR NOT isFinite(spend)) AS unknown_cost_requests, sum(spend) AS recorded_spend, - sum(prompt_tokens) AS input_tokens, - sum(completion_tokens) AS output_tokens - FROM spend_logs FINAL + sum(c.input_tokens) AS input_tokens, + sum(c.output_tokens) AS output_tokens + FROM calls AS c WHERE start_time >= now() - INTERVAL 1 DAY GROUP BY team_id, model ) diff --git a/litellm-rust/crates/traces-clickhouse/query/help/nested_metadata.sql b/litellm-rust/crates/traces-clickhouse/query/help/nested_metadata.sql index cccec2177a0..d582c2aa189 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/nested_metadata.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/nested_metadata.sql @@ -2,7 +2,7 @@ SELECT request_id, JSONType(metadata, 'labels', 'priority') AS type, JSONExtractRaw(metadata, 'labels', 'priority') AS value -FROM spend_logs FINAL +FROM calls WHERE start_time >= now() - INTERVAL 1 DAY AND JSONHas(metadata, 'labels', 'priority') LIMIT 100 diff --git a/litellm-rust/crates/traces-clickhouse/query/help/recent_spans.sql b/litellm-rust/crates/traces-clickhouse/query/help/recent_spans.sql index c1e88560110..313d18aad59 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/recent_spans.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/recent_spans.sql @@ -1,8 +1,7 @@ SELECT - TraceId, SpanId, Model, InputTokens, OutputTokens, - Duration / 1000000 AS duration_ms -FROM otel_traces -WHERE Timestamp >= now() - INTERVAL 1 DAY - AND ObservationType = 'llm' -ORDER BY Timestamp DESC + trace_id, span_id, model, input_tokens, output_tokens, duration_ms +FROM spans +WHERE start_time >= now() - INTERVAL 1 DAY + AND observation_type = 'llm' +ORDER BY start_time DESC, trace_ref, span_id LIMIT 100 diff --git a/litellm-rust/crates/traces-clickhouse/query/help/recent_spend.sql b/litellm-rust/crates/traces-clickhouse/query/help/recent_spend.sql index 809b1bd44f0..98ce8049762 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/recent_spend.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/recent_spend.sql @@ -1,7 +1,7 @@ SELECT request_id, response_id, trace_id, span_id, model, spend, - prompt_tokens, completion_tokens, status -FROM spend_logs FINAL + input_tokens, output_tokens, status +FROM calls WHERE start_time >= now() - INTERVAL 1 DAY ORDER BY start_time DESC, request_id LIMIT 100 diff --git a/litellm-rust/crates/traces-clickhouse/query/help/trace_spend.sql b/litellm-rust/crates/traces-clickhouse/query/help/trace_spend.sql index 3f0ec16186d..5dca9a66af8 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/trace_spend.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/trace_spend.sql @@ -1,8 +1,8 @@ SELECT - team_id, api_key, trace_id, count() AS requests, + team_id, api_key_hash AS api_key, trace_id, count() AS requests, countIf(isNull(spend) OR NOT isFinite(spend)) AS unknown_cost_requests, if(unknown_cost_requests = 0, sum(spend), NULL) AS recorded_spend -FROM spend_logs FINAL +FROM calls WHERE start_time >= now() - INTERVAL 1 DAY AND trace_id != '' GROUP BY team_id, api_key, trace_id diff --git a/litellm-rust/crates/traces-clickhouse/query/help/trace_summary.sql b/litellm-rust/crates/traces-clickhouse/query/help/trace_summary.sql index 7a5dcaf10ff..211a0c15dc7 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/trace_summary.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/trace_summary.sql @@ -1,12 +1,7 @@ -SELECT TeamId AS team, ApiKeyHash AS api_key, TraceId AS trace_id, - ifNull(any(RootName), '') AS name, - toUInt32(sum(SpanCount)) AS spans, - toUInt32(sum(LlmCount)) AS llm_calls, - toUInt32(sum(ErrorCount)) AS errors, - toUInt32(sum(InputTokens)) AS input_tokens, - toUInt32(sum(OutputTokens)) AS output_tokens -FROM agent_traces_by_key -GROUP BY TeamId, ApiKeyHash, TraceId -HAVING min(StartTs) >= now() - INTERVAL 1 DAY +SELECT id, team_id AS team, api_key_hash AS api_key, trace_id, name, + span_count AS spans, llm_span_count, error_count AS errors, + span_input_tokens, span_output_tokens +FROM traces +WHERE start_time >= now() - INTERVAL 1 DAY ORDER BY team, api_key, trace_id LIMIT 100 diff --git a/litellm-rust/crates/traces-clickhouse/query/help/unmatched_spans.sql b/litellm-rust/crates/traces-clickhouse/query/help/unmatched_spans.sql index d5ad0fbb87e..46d6f2c914a 100644 --- a/litellm-rust/crates/traces-clickhouse/query/help/unmatched_spans.sql +++ b/litellm-rust/crates/traces-clickhouse/query/help/unmatched_spans.sql @@ -1,18 +1,18 @@ SELECT - t.TraceId, t.SpanId, t.Model, t.LiteLLMRequestId, - t.InputTokens, t.OutputTokens -FROM otel_traces AS t + t.trace_id, t.span_id, t.model, t.request_id, + t.input_tokens, t.output_tokens +FROM spans AS t LEFT ANTI JOIN ( SELECT * - FROM spend_logs FINAL + FROM calls WHERE start_time >= now() - INTERVAL 1 DAY ) AS s - ON t.TeamId = s.team_id - AND ((t.UserId != '' AND t.UserId = s.user) - OR (t.ApiKeyHash != '' AND t.ApiKeyHash = s.api_key)) - AND t.LiteLLMRequestId != '' - AND (t.LiteLLMRequestId = s.response_id OR t.LiteLLMRequestId = s.request_id) -WHERE t.Timestamp >= now() - INTERVAL 1 DAY - AND t.ObservationType = 'llm' -ORDER BY t.Timestamp DESC, t.SpanId + ON t.team_id = s.team_id + AND ((t.user_id != '' AND t.user_id = s.user_id) + OR (t.api_key_hash != '' AND t.api_key_hash = s.api_key_hash)) + AND t.request_id != '' + AND (t.request_id = s.response_id OR t.request_id = s.request_id) +WHERE t.start_time >= now() - INTERVAL 1 DAY + AND t.observation_type = 'llm' +ORDER BY t.start_time DESC, t.span_id LIMIT 100 diff --git a/litellm-rust/crates/traces-clickhouse/query/matching_runs.sql b/litellm-rust/crates/traces-clickhouse/query/matching_runs.sql index 6249cfd4111..862c830144c 100644 --- a/litellm-rust/crates/traces-clickhouse/query/matching_runs.sql +++ b/litellm-rust/crates/traces-clickhouse/query/matching_runs.sql @@ -1,69 +1,42 @@ SELECT TraceId AS trace_id, hex(SHA256(concat(TeamId, char(0), ApiKeyHash, char(0), TraceId))) AS trace_ref, - if(length(groupUniqArrayArray(UserIds)) = 1, arrayElement(groupUniqArrayArray(UserIds), 1), '') AS user_id, TeamId AS team_id, ApiKeyHash AS api_key_hash, - ifNull(any(RootName), '') AS name, any(ServiceName) AS service, - ifNull(any(RootInput), '') AS input_preview, ifNull(any(RootStatus), '') AS status, - toUnixTimestamp64Milli(min(StartTs)) AS start_ms, - if(any(metrics.span_count) > 0, - toUInt64(greatest(toUnixTimestamp64Nano(any(metrics.end_ts)) - toUnixTimestamp64Nano(any(metrics.start_ts)), 0)), - toUInt64(greatest(toUnixTimestamp64Nano(max(EndTs)) - toUnixTimestamp64Nano(min(StartTs)), 0))) AS duration_ns, - if(any(metrics.span_count) > 0, any(metrics.span_count), sum(SpanCount)) AS span_count, - sum(AgentCount) AS agent_invocations, - sum(LlmCount) AS llm_calls, sum(ToolCount) AS tool_calls, + if(length(groupUniqArray(UserId)) = 1, arrayElement(groupUniqArray(UserId), 1), '') AS user_id, + TeamId AS team_id, ApiKeyHash AS api_key_hash, + if(countIf(ParentSpanId = '') > 0, argMinIf(SpanName, tuple(Timestamp, SpanId), ParentSpanId = ''), argMin(SpanName, tuple(Timestamp, SpanId))) AS name, + argMin(ServiceName, tuple(Timestamp, SpanId)) AS service, + if(countIf(ParentSpanId = '') > 0, argMinIf(InputPreview, tuple(Timestamp, SpanId), ParentSpanId = ''), argMin(InputPreview, tuple(Timestamp, SpanId))) AS root_input, + if(root_input != '', root_input, argMinIf(InputPreview, tuple(Timestamp, SpanId), InputPreview != '' AND ObservationType IN ('agent', 'llm'))) AS input_preview, + if(countIf(ParentSpanId = '') > 0, argMinIf(StatusCode, tuple(Timestamp, SpanId), ParentSpanId = ''), argMin(StatusCode, tuple(Timestamp, SpanId))) AS status, + toUnixTimestamp64Milli(min(Timestamp)) AS start_ms, + toUInt64(greatest(toUnixTimestamp64Nano(max(Timestamp + toIntervalNanosecond(Duration))) - toUnixTimestamp64Nano(min(Timestamp)), 0)) AS duration_ns, + count() AS span_count, + countIf(ObservationType = 'agent') AS agent_invocations, + countIf(ObservationType = 'llm') AS llm_calls, + countIf(ObservationType = 'tool') AS tool_calls, sum(InputTokens) AS input_tokens, sum(OutputTokens) AS output_tokens, - groupUniqArrayArray(Models) AS models, - if(any(metrics.span_count) > 0, any(metrics.error_count), sum(ErrorCount)) AS error_count, - arraySort(groupUniqArrayArray(AgentNames)) AS search_agents, - if(error_count > 0, 'error', 'ok') AS search_status, - length(groupUniqArrayArray(AgentIdentities)) AS agent_count, - arraySort(groupUniqArrayArray(Frameworks)) AS frameworks -FROM owned_runs -LEFT JOIN ( - SELECT TeamId, ApiKeyHash, TraceId, - min(Timestamp) AS start_ts, - max(Timestamp + toIntervalNanosecond(Duration)) AS end_ts, - count() AS span_count, - countIf(StatusCode = 'STATUS_CODE_ERROR') AS error_count - FROM ( - SELECT TeamId, ApiKeyHash, TraceId, SpanId, Timestamp, Duration, StatusCode - FROM owned_spans - WHERE (TeamId, ApiKeyHash, TraceId) IN ( - SELECT TeamId, ApiKeyHash, TraceId - FROM owned_runs - WHERE {trace_id:String} = '' OR TraceId = {trace_id:String} - GROUP BY TeamId, ApiKeyHash, TraceId - HAVING {trace_id:String} != '' OR ( - min(StartTs) >= fromUnixTimestamp64Milli({start_ms:Int64}) - AND min(StartTs) < fromUnixTimestamp64Milli({end_ms:Int64})) - ) - AND ({trace_id:String} != '' OR Timestamp >= fromUnixTimestamp64Milli({start_ms:Int64})) - ORDER BY Timestamp, EngineReceivedMs, StatusMessage, Duration, StatusCode - LIMIT 1 BY TeamId, ApiKeyHash, TraceId, SpanId - ) - GROUP BY TeamId, ApiKeyHash, TraceId -) AS metrics USING (TeamId, ApiKeyHash, TraceId) -LEFT JOIN ( - SELECT TeamId, ApiKeyHash, TraceId, - groupUniqArrayArray(arrayFilter(i -> (mapContains(ResourceAttributes, {attribute_keys:Array(String)}[i]) - AND ResourceAttributes[{attribute_keys:Array(String)}[i]] ILIKE {attribute_patterns:Array(String)}[i]) - OR (mapContains(SpanAttributes, {attribute_keys:Array(String)}[i]) - AND SpanAttributes[{attribute_keys:Array(String)}[i]] ILIKE {attribute_patterns:Array(String)}[i]), - arrayEnumerate({attribute_keys:Array(String)}))) AS matched_attributes - FROM owned_spans - WHERE notEmpty({attribute_keys:Array(String)}) - AND Timestamp >= fromUnixTimestamp64Milli({start_ms:Int64}) - GROUP BY TeamId, ApiKeyHash, TraceId -) AS attributes USING (TeamId, ApiKeyHash, TraceId) -WHERE {trace_id:String} = '' OR TraceId = {trace_id:String} + groupUniqArrayIf(toString(Model), Model != '') AS models, + countIf(StatusCode = 'STATUS_CODE_ERROR') AS error_count, + arraySort(groupUniqArrayIf(if(AgentName = '', SpanName, AgentName), AgentName != '' OR ObservationType = 'agent')) AS search_agents, + multiIf(status = 'STATUS_CODE_OK', 'ok', status = 'STATUS_CODE_ERROR', 'error', 'unset') AS search_root_status, + if(error_count > 0, 'true', 'false') AS search_has_error, + length(groupUniqArrayIf(if(AgentName = '', SpanName, AgentName), ObservationType = 'agent')) AS agent_count, + arraySort(groupUniqArrayIf(toString(Framework), Framework != '')) AS frameworks, + groupUniqArrayArray(arrayFilter(i -> (mapContains(ResourceAttributes, {attribute_keys:Array(String)}[i]) AND ResourceAttributes[{attribute_keys:Array(String)}[i]] ILIKE {attribute_patterns:Array(String)}[i]) + OR (mapContains(SpanAttributes, {attribute_keys:Array(String)}[i]) AND SpanAttributes[{attribute_keys:Array(String)}[i]] ILIKE {attribute_patterns:Array(String)}[i]), + arrayEnumerate({attribute_keys:Array(String)}))) AS matched_attributes +FROM canonical_spans GROUP BY TeamId, ApiKeyHash, TraceId -HAVING {trace_id:String} != '' - OR (min(StartTs) >= fromUnixTimestamp64Milli({start_ms:Int64}) - AND min(StartTs) < fromUnixTimestamp64Milli({end_ms:Int64}) +HAVING ({trace_id:String} != '' AND TraceId = {trace_id:String}) + OR ({trace_ref:String} != '' AND trace_ref = {trace_ref:String}) + OR ({trace_id:String} = '' AND {trace_ref:String} = '' + AND min(Timestamp) >= fromUnixTimestamp64Milli({start_ms:Int64}) + AND min(Timestamp) < fromUnixTimestamp64Milli({end_ms:Int64}) AND arrayAll(t -> trace_id ILIKE t OR input_preview ILIKE t OR name ILIKE t, {text:Array(String)}) AND arrayAll((f, p, m) -> (m = 'exclude') != multiIf( f = 'name', name ILIKE p, f = 'agent', arrayExists(a -> a ILIKE p, search_agents), - f = 'status', search_status ILIKE p, + f = 'root_status', search_root_status ILIKE p, + f = 'has_error', search_has_error ILIKE p, f = 'model', arrayExists(x -> x ILIKE p, models), f = 'input', input_preview ILIKE p, f = 'trace_id', trace_id ILIKE p, @@ -71,6 +44,6 @@ HAVING {trace_id:String} != '' f = 'team', team_id ILIKE p, false), {filter_fields:Array(String)}, {filter_patterns:Array(String)}, {filter_modes:Array(String)}) - AND arrayAll((i, m) -> (m = 'exclude') != has(any(matched_attributes), i), + AND arrayAll((i, m) -> (m = 'exclude') != has(matched_attributes, i), arrayEnumerate({attribute_keys:Array(String)}), {attribute_modes:Array(String)}) AND (empty({trace_refs:Array(String)}) OR trace_ref IN {trace_refs:Array(String)})) diff --git a/litellm-rust/crates/traces-clickhouse/query/run_attribute_counts.sql b/litellm-rust/crates/traces-clickhouse/query/run_attribute_counts.sql index 32886c877dd..e457ae18194 100644 --- a/litellm-rust/crates/traces-clickhouse/query/run_attribute_counts.sql +++ b/litellm-rust/crates/traces-clickhouse/query/run_attribute_counts.sql @@ -5,7 +5,7 @@ FROM ( toUInt32(intDiv((runs.start_ms - {start_ms:Int64} + 1) * {buckets:UInt32} - 1, {end_ms:Int64} - {start_ms:Int64}))) AS bucket, toUInt8({by_failed:UInt8} = 1 AND runs.error_count > 0) AS failed, value - FROM owned_spans AS spans + FROM canonical_spans AS spans INNER JOIN runs ON spans.TeamId = runs.team_id AND spans.ApiKeyHash = runs.api_key_hash AND spans.TraceId = runs.trace_id ARRAY JOIN if({attribute_key:String} = '', diff --git a/litellm-rust/crates/traces-clickhouse/query/run_counts.sql b/litellm-rust/crates/traces-clickhouse/query/run_counts.sql index 2bf3c8891de..7dadf3211bb 100644 --- a/litellm-rust/crates/traces-clickhouse/query/run_counts.sql +++ b/litellm-rust/crates/traces-clickhouse/query/run_counts.sql @@ -1,5 +1,5 @@ SELECT if({buckets:UInt32} = 0, toUInt32(0), - toUInt32(intDiv((start_ms - {start_ms:Int64} + 1) * {buckets:UInt32} - 1, {end_ms:Int64} - {start_ms:Int64}))) AS bucket, + toUInt32(intDiv((toInt128(start_ms) - toInt128({start_ms:Int64}) + 1) * {buckets:UInt32} - 1, toInt128({end_ms:Int64}) - toInt128({start_ms:Int64})))) AS bucket, toUInt8({by_failed:UInt8} = 1 AND error_count > 0) AS failed, value, count() AS runs @@ -9,7 +9,8 @@ ARRAY JOIN multiIf( {value:String} = 'primary_agent', [ifNull(nullIf(arrayElement(search_agents, 1), ''), service)], {value:String} = 'name', [name], {value:String} = 'agent', search_agents, - {value:String} = 'status', [search_status], + {value:String} = 'root_status', [search_root_status], + {value:String} = 'has_error', [search_has_error], {value:String} = 'model', models, {value:String} = 'input', [input_preview], {value:String} = 'trace_id', [trace_id], diff --git a/litellm-rust/crates/traces-clickhouse/query/runs_page.sql b/litellm-rust/crates/traces-clickhouse/query/runs_page.sql index 135edca3966..16af068401d 100644 --- a/litellm-rust/crates/traces-clickhouse/query/runs_page.sql +++ b/litellm-rust/crates/traces-clickhouse/query/runs_page.sql @@ -1,5 +1,5 @@ page AS ( -SELECT * EXCEPT (search_status), +SELECT * EXCEPT (root_input, search_root_status, search_has_error, matched_attributes), multiIf({sort_key:String} = 'duration_ms', toInt64(least(duration_ns, toUInt64(9223372036854775807))), {sort_key:String} = 'span_count', toInt64(least(span_count, toUInt64(9223372036854775807))), {sort_key:String} = 'error_count', toInt64(least(error_count, toUInt64(9223372036854775807))), diff --git a/litellm-rust/crates/traces-clickhouse/src/bin/export_openapi.rs b/litellm-rust/crates/traces-clickhouse/src/bin/export_openapi.rs new file mode 100644 index 00000000000..1c8e32f1de4 --- /dev/null +++ b/litellm-rust/crates/traces-clickhouse/src/bin/export_openapi.rs @@ -0,0 +1,5 @@ +fn main() { + let document = + litellm_traces::api::openapi::document(litellm_traces_clickhouse::wire_schema::schemas()); + println!("{}", document.to_pretty_json().unwrap()); +} diff --git a/litellm-rust/crates/traces-clickhouse/src/lib.rs b/litellm-rust/crates/traces-clickhouse/src/lib.rs index 1eec998b0a1..7dc332eb09a 100644 --- a/litellm-rust/crates/traces-clickhouse/src/lib.rs +++ b/litellm-rust/crates/traces-clickhouse/src/lib.rs @@ -28,11 +28,11 @@ pub use error::Error; pub use insert::{InsertRow, InsertTable, encode_rows, insert_rows, insert_shared_rows}; pub use litellm_storage_clickhouse::{Connection, Parameter}; pub use litellm_traces::QueryScope; -pub use query::{QueryHelp, execute_read, query_help, query_sql}; +pub use query::{QueryHelp, execute_read, query_help, query_sql, query_sql_with_params}; pub use query_access::QueryReaders; pub use reads::ClickHouseTraces; pub use schema::{ NORMALIZED_FIELD_DEFINITIONS, NormalizedFieldDefinition, ensure_schema, schema_statements, }; pub use span_row::span_rows; -pub use table::TraceTable; +pub use table::{QueryTable, TraceTable}; diff --git a/litellm-rust/crates/traces-clickhouse/src/query.rs b/litellm-rust/crates/traces-clickhouse/src/query.rs index 5c027e8ef6f..3e21a92c28e 100644 --- a/litellm-rust/crates/traces-clickhouse/src/query.rs +++ b/litellm-rust/crates/traces-clickhouse/src/query.rs @@ -5,6 +5,7 @@ use futures_util::{ stream::{self, TryStreamExt}, }; use litellm_http::Client; +use litellm_traces::api::{SqlParameter, TraceSQLColumn}; use litellm_traces::query::guide::{Example, QueryGuide, Section}; use serde::{Deserialize, Serialize, Serializer}; use serde_json::Value; @@ -14,7 +15,7 @@ use super::{ Connection, Error, NORMALIZED_FIELD_DEFINITIONS, NormalizedFieldDefinition, Parameter, query_access::READER_LIMITS, }; -use crate::TraceTable; +use crate::QueryTable; mod guide; pub mod named; @@ -23,7 +24,7 @@ mod number; const SAMPLE_ROWS: usize = 200; const MAX_FIELDS: usize = 200; const MAX_DEPTH: usize = 16; -const METADATA_SQL: &str = "SELECT metadata FROM spend_logs FINAL \ +const METADATA_SQL: &str = "SELECT metadata FROM calls \ WHERE start_time >= now() - INTERVAL 7 DAY AND length(metadata) <= 8192 \ LIMIT 201"; const METADATA_SCOPE: &str = "Up to 200 unordered rows from the last 7 days, excluding metadata larger than 8192 bytes; up to 200 paths and 16 levels. Missing paths may exist outside this sample. Array indexes are 1-based and describe sampled positions, not a fixed schema"; @@ -96,22 +97,12 @@ struct MetadataField { expression: String, } -#[macro_rules_attribute::apply(wire_type)] -#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryColumn"))] -struct ColumnSchema { - name: String, - #[serde(rename = "type")] - kind: String, - #[serde(flatten)] - details: BTreeMap, -} - #[macro_rules_attribute::apply(response_type)] #[cfg_attr(feature = "schema", schemars(deny_unknown_fields))] #[cfg_attr(feature = "schema", schemars(rename = "TraceQueryTable"))] struct TableSchema { - name: TraceTable, - columns: Vec, + name: QueryTable, + columns: Vec, } trait Unobserved { @@ -197,7 +188,7 @@ impl Unobserved for MetadataSample { #[cfg_attr(feature = "schema", schemars(deny_unknown_fields))] #[cfg_attr(feature = "schema", schemars(rename = "TraceQueryMetadata"))] struct MetadataCatalog { - table: TraceTable, + table: QueryTable, column: &'static str, #[serde(flatten)] discovery: Discovery, @@ -234,7 +225,7 @@ impl Unobserved for AttributeSample { #[cfg_attr(feature = "schema", schemars(deny_unknown_fields))] #[cfg_attr(feature = "schema", schemars(rename = "TraceQueryAttributes"))] struct AttributeCatalog { - table: TraceTable, + table: QueryTable, column: &'static str, #[serde(flatten)] discovery: Discovery, @@ -246,7 +237,7 @@ struct AttributeCatalog { #[cfg_attr(feature = "schema", schemars(deny_unknown_fields))] #[cfg_attr(feature = "schema", schemars(rename = "TraceQueryNormalizedField"))] struct NormalizedField { - table: TraceTable, + table: QueryTable, name: &'static str, column: &'static str, #[serde(rename = "type")] @@ -257,7 +248,7 @@ struct NormalizedField { impl From<&NormalizedFieldDefinition> for NormalizedField { fn from(field: &NormalizedFieldDefinition) -> Self { Self { - table: TraceTable::OtelTraces, + table: QueryTable::OtelTraces, name: field.name, column: field.clickhouse_column, kind: field.clickhouse_type, @@ -277,10 +268,10 @@ struct Relationship { } const RELATIONSHIPS: [Relationship; 1] = [Relationship { - left: "otel_traces.LiteLLMRequestId", - right: "spend_logs.response_id", - additional_predicates: "otel_traces.TeamId = spend_logs.team_id AND ((otel_traces.UserId != '' AND otel_traces.UserId = spend_logs.user) OR (otel_traces.ApiKeyHash != '' AND otel_traces.ApiKeyHash = spend_logs.api_key))", - meaning: "LiteLLMRequestId contains the first normalized request or provider response ID. This relationship matches response IDs only; CallKeys retains all typed identifiers. Cached requests can share response_id; joins may return multiple spend rows", + left: "spans.request_id", + right: "calls.response_id", + additional_predicates: "spans.team_id = calls.team_id AND ((spans.user_id != '' AND spans.user_id = calls.user_id) OR (spans.api_key_hash != '' AND spans.api_key_hash = calls.api_key_hash))", + meaning: "request_id contains the first normalized request or provider response ID. This relationship matches response IDs only; call_keys retains all typed identifiers. Cached requests can share response_id; joins may return multiple calls", }]; #[macro_rules_attribute::apply(response_type)] @@ -321,6 +312,30 @@ pub async fn query_sql( execute_read(client, connection, sql, &BTreeMap::new()).await } +pub async fn query_sql_with_params( + client: &Client, + connection: &Connection, + sql: &str, + params: &BTreeMap, +) -> Result { + let parameters = params + .iter() + .map(|(name, value)| { + let parameter = match value { + SqlParameter::String(value) => Parameter::Text(value.clone()), + SqlParameter::Integer(value) => Parameter::Integer(*value), + SqlParameter::Unsigned(value) => Parameter::Unsigned(*value), + SqlParameter::Number(value) => Parameter::Float(*value), + SqlParameter::Boolean(value) => Parameter::Boolean(*value), + SqlParameter::Null => Parameter::Null, + SqlParameter::Strings(value) => Parameter::Strings(value.clone()), + }; + (name.clone(), parameter) + }) + .collect(); + execute_read(client, connection, sql, ¶meters).await +} + async fn rows( client: &Client, connection: &Connection, @@ -415,11 +430,11 @@ fn metadata_sample(sample: &[MetadataRow]) -> MetadataSample { } pub async fn query_help(client: &Client, connection: &Connection) -> Result { - let tables = stream::iter(TraceTable::iter()) + let tables = stream::iter(QueryTable::iter()) .then(|table| async move { Ok::<_, Error>(TableSchema { name: table, - columns: rows::( + columns: rows::( client, connection, &format!("DESCRIBE TABLE {table}"), @@ -430,7 +445,7 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result>() .await?; let metadata = MetadataCatalog { - table: TraceTable::SpendLogs, + table: QueryTable::Calls, column: "metadata", discovery: match rows::(client, connection, METADATA_SQL).await { Ok(sample) => Discovery::Observed(metadata_sample(&sample)), @@ -439,11 +454,11 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result= now() - INTERVAL 7 DAY \ + (SELECT {column} FROM spans WHERE start_time >= now() - INTERVAL 7 DAY \ LIMIT 200) ORDER BY key LIMIT 201" ); let discovery = match rows::(client, connection, &sql).await { @@ -462,7 +477,7 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result Discovery::Unavailable(error.to_string()), }; AttributeCatalog { - table: TraceTable::OtelTraces, + table: QueryTable::Spans, column, discovery, discovery_sql: sql, @@ -484,6 +499,7 @@ pub async fn query_help(client: &Client, connection: &Connection) -> Result Result { } impl QueryGuide<'_> { - pub fn sections(&self) -> Result<[String; 4], Error> { + pub fn sections(&self) -> Result<[String; 5], Error> { Ok([ render(&self.as_live_schema())?, render(&self.as_normalized_fields())?, render(&self.as_metadata())?, render(&self.as_attributes())?, + render(&self.as_parameters())?, ]) } diff --git a/litellm-rust/crates/traces-clickhouse/src/query/named.rs b/litellm-rust/crates/traces-clickhouse/src/query/named.rs index acee1f0a965..9b035e53126 100644 --- a/litellm-rust/crates/traces-clickhouse/src/query/named.rs +++ b/litellm-rust/crates/traces-clickhouse/src/query/named.rs @@ -79,9 +79,11 @@ impl From<&RunSearch> for SearchColumns { } } -#[derive(Debug, Default, Serialize)] +#[derive(Debug, Serialize)] struct RunsFilter { trace_id: String, + trace_ref: String, + as_of_ms: u64, start_ms: i64, end_ms: i64, #[serde(flatten)] @@ -89,10 +91,26 @@ struct RunsFilter { trace_refs: Vec, } +impl Default for RunsFilter { + fn default() -> Self { + Self { + trace_id: String::new(), + trace_ref: String::new(), + as_of_ms: u64::MAX, + start_ms: 0, + end_ms: 0, + search: SearchColumns::default(), + trace_refs: Vec::new(), + } + } +} + impl From<&RunFilter> for RunsFilter { fn from(filter: &RunFilter) -> Self { Self { trace_id: String::new(), + trace_ref: String::new(), + as_of_ms: filter.as_of_ms, start_ms: filter.start_ms, end_ms: filter.end_ms, search: (&filter.search).into(), @@ -105,6 +123,10 @@ impl From<&RunSelection> for RunsFilter { fn from(selection: &RunSelection) -> Self { match selection { RunSelection::Matching(filter) => filter.into(), + RunSelection::TraceRef(trace_ref) => Self { + trace_ref: trace_ref.clone(), + ..Self::default() + }, RunSelection::TraceId(trace_id) => Self { trace_id: trace_id.clone(), ..Self::default() @@ -189,6 +211,8 @@ pub(crate) struct RunRowWire(#[serde(with = "RunRowEncoding")] pub RunRow); macro_rules! over_matching_runs { ($($tail:expr),+ $(,)?) => { owned!( + ",\n", + include_str!("../../query/canonical_spans.sql"), ",\nruns AS (\n", include_str!("../../query/matching_runs.sql"), ")", @@ -521,6 +545,7 @@ impl Query for SpanTexts { #[derive(Debug, Serialize)] pub(crate) struct CallsParams { + as_of_ms: u64, #[serde(flatten)] access: AccessParams, start_ms: i64, @@ -540,6 +565,7 @@ impl CallsParams { let after = query.after.clone().unwrap_or_default(); Self { access: access.into(), + as_of_ms: query.as_of_ms, start_ms: query.window.start, end_ms: query.window.end, response_ids: query.response_ids.clone(), diff --git a/litellm-rust/crates/traces-clickhouse/src/query_access.rs b/litellm-rust/crates/traces-clickhouse/src/query_access.rs index 51436713867..a6553d624da 100644 --- a/litellm-rust/crates/traces-clickhouse/src/query_access.rs +++ b/litellm-rust/crates/traces-clickhouse/src/query_access.rs @@ -9,7 +9,7 @@ use sha2::{Digest, Sha256}; use strum::IntoEnumIterator; use tokio::sync::{OwnedSemaphorePermit, Semaphore}; -use super::{Connection, Error, TraceTable}; +use super::{Connection, Error, QueryTable, TraceTable}; const MIB: u64 = 1024 * 1024; @@ -141,7 +141,7 @@ impl QueryReaders { ) .await?; } - for table in TraceTable::iter() { + for table in QueryTable::iter() { self.execute( client, format!("GRANT SELECT ON `{database}`.{table} TO {user}"), diff --git a/litellm-rust/crates/traces-clickhouse/src/table.rs b/litellm-rust/crates/traces-clickhouse/src/table.rs index c74cf6d4de1..e74518f1a05 100644 --- a/litellm-rust/crates/traces-clickhouse/src/table.rs +++ b/litellm-rust/crates/traces-clickhouse/src/table.rs @@ -10,3 +10,17 @@ pub enum TraceTable { AgentTracesByKey, SpendLogs, } + +#[macro_rules_attribute::apply(response_type)] +#[cfg_attr(feature = "schema", schemars(rename = "TraceQueryTableName"))] +#[derive(Clone, Copy, Debug, strum::Display, strum::EnumIter)] +#[serde(rename_all = "snake_case")] +#[strum(serialize_all = "snake_case")] +pub enum QueryTable { + Traces, + Spans, + Calls, + OtelTraces, + AgentTracesByKey, + SpendLogs, +} diff --git a/litellm-rust/crates/traces-clickhouse/templates/query_help.jinja b/litellm-rust/crates/traces-clickhouse/templates/query_help.jinja index ead9022183b..d369e0513f5 100644 --- a/litellm-rust/crates/traces-clickhouse/templates/query_help.jinja +++ b/litellm-rust/crates/traces-clickhouse/templates/query_help.jinja @@ -1,4 +1,7 @@ {% block live_schema -%} +Use traces for canonical trace metrics, spans for one earliest copy of each span, and calls for replacement-deduplicated spend records +traces.id is the canonical trace ID used by the curated API; join spans with traces.id = spans.trace_ref +otel_traces, agent_traces_by_key and spend_logs are physical diagnostic tables. Repeated exports can inflate raw span and rollup counts {% for table in tables -%} {{ table.name }} {% for column in table.columns -%} @@ -7,6 +10,12 @@ {% endfor -%} {%- endblock %} +{% block parameters -%} +Pass values in the optional params object using ClickHouse typed placeholders. The engine binds values without rewriting SQL +{"sql":"SELECT id, name, duration_ms FROM traces WHERE start_time >= fromUnixTimestamp64Milli({start_ms:Int64}) AND has({teams:Array(String)}, team_id) ORDER BY start_time DESC, id LIMIT {limit:UInt32}","params":{"start_ms":0,"teams":["example-team"],"limit":50}} +Values support strings, integers, numbers, booleans, null and string arrays. Match each placeholder type to its value, for example {enabled:Bool} or {optional:Nullable(String)} +{%- endblock %} + {% block normalized_fields -%} {% for field in normalized_fields -%} {{ field.name }}: otel_traces.{{ field.clickhouse_column }} ({{ field.clickhouse_type }}) @@ -152,8 +161,9 @@ Filter calls by nested metadata {%- endblock %} {% block time_window -%} -Always bound Timestamp or start_time and use LIMIT; add TeamId/ApiKeyHash or team_id/api_key filters when investigating one tenant +Always bound start_time and use LIMIT; add team_id/api_key_hash filters when investigating one tenant Raw SQL returns one bounded response without a cursor. Callers own ORDER BY, LIMIT and keyset predicates for pagination +SQL views are live relations; separate queries do not freeze membership or span versions {%- endblock %} {% block reader_limits -%} @@ -161,7 +171,7 @@ The reader enforces {{ limits.result_rows }} result rows, {{ limits.result_mib() {%- endblock %} {% block reader_profile -%} -LiteLLM provisions SELECT-only readers from the configured ClickHouse connection and enforces request-log visibility through row policies. Callers see their own user rows and permitted teams. Provisioning requires CREATE USER, ALTER USER, CREATE ROW POLICY, and GRANT SELECT permissions +LiteLLM provisions SELECT-only readers and INVOKER views using ClickHouse row policies. Callers see their own user rows and permitted teams. SQL traces summarize visible spans, while curated user-only reads require full trace ownership. Provisioning requires CREATE USER, ALTER USER, CREATE ROW POLICY, and GRANT SELECT permissions {%- endblock %} {% block output_format -%} @@ -173,7 +183,7 @@ metadata is a JSON-encoded String; use JSONHas before typed extraction to distin {%- endblock %} {% block map_values -%} -SpanAttributes and ResourceAttributes are Map(String, String); missing map keys return an empty string, so use mapContains for existence checks +spans.span_attributes and spans.resource_attributes are Map(String, String); missing map keys return an empty string, so use mapContains for existence checks {%- endblock %} {% block literal_keys -%} @@ -181,7 +191,7 @@ Use the discovered path components as separate JSONExtract arguments; a dot insi {%- endblock %} {% block time_units -%} -Duration is nanoseconds; Timestamp has nanosecond precision, spend start_time has millisecond precision +spans.duration_ns and traces.duration_ns are nanoseconds; duration_ms preserves fractional milliseconds. Span and trace start_time has nanosecond precision, call start_time has millisecond precision {%- endblock %} {% block missing_spend -%} @@ -193,13 +203,14 @@ Recorded spend by trace totals only requests whose spend_logs.trace_id is popula {%- endblock %} {% block spend_totals -%} -Use spend_logs FINAL to collapse replacement rows before totals. Shared response IDs and multiple spans can multiply costs in joins; require one spend match per response ID and ownership before aggregating. Missing IDs or costs leave totals unknown +calls collapses spend replacement rows before totals. Shared response IDs and multiple spans can multiply costs in joins; require one call match per response ID and ownership before aggregating. Missing IDs or costs leave totals unknown {%- endblock %} {% block trace_rollups -%} -agent_traces_by_key uses SimpleAggregateFunction columns; group by TeamId, ApiKeyHash and TraceId, using min(StartTs), max(EndTs), sum(SpanCount) and groupUniqArrayArray(Models). Do not use Merge combinators -Rollup counts and raw span reads can include repeated exports. Curated trace metrics select one copy per TeamId, ApiKeyHash, TraceId and SpanId, ordered by Timestamp, EngineReceivedMs, StatusMessage, Duration and StatusCode, before filtering status or aggregating -AgentNames contains searchable agent labels, including SpanName for unnamed agents. AgentIdentities contains distinct labels from agent spans, and Frameworks contains observed framework names +traces and spans select one copy per team_id, api_key_hash, trace_id and span_id, ordered by source Timestamp, EngineReceivedMs, StatusMessage, Duration and StatusCode, before filtering status or aggregating +traces.root_status is the earliest physical root's status, or the earliest span's status when no root exists. has_error reports any canonical span error +agent_span_count, llm_span_count and tool_span_count count normalized span types; agent_label_count counts distinct agent_names. span_input_tokens and span_output_tokens sum canonical span tokens. Curated trace summaries resolve graph wrappers and active calls, so their enriched invocation and token totals can differ +The diagnostic agent_traces_by_key table uses SimpleAggregateFunction columns and counts repeated exports. Group by TeamId, ApiKeyHash and TraceId; do not use Merge combinators {%- endblock %} {% block sampling -%} diff --git a/litellm-rust/crates/traces-clickhouse/tests/migrations.rs b/litellm-rust/crates/traces-clickhouse/tests/migrations.rs index 8d71dc9303d..e9195b91cf8 100644 --- a/litellm-rust/crates/traces-clickhouse/tests/migrations.rs +++ b/litellm-rust/crates/traces-clickhouse/tests/migrations.rs @@ -165,6 +165,7 @@ async fn schema_supports_span_rollups_and_spend_joins( .calls( &team, &CallQuery { + as_of_ms: u64::MAX, window: timestamp / 1_000_000 - 1000..timestamp / 1_000_000 + 1000, response_ids: vec!["response-1".into()], request_ids: Vec::new(), @@ -819,7 +820,8 @@ async fn reused_trace_ids_stay_separate_runs_through_filters_and_span_text( end_ms: window.end, search: litellm_traces::search::RunSearch::parse( "service:review attr.swarm:release", - ), + ) + .unwrap(), ..Default::default() }), after: None, @@ -843,6 +845,26 @@ async fn reused_trace_ids_stay_separate_runs_through_filters_and_span_text( let own = store.runs(&owned("one", &[]), &by_trace_id).await?; assert_eq!(own.len(), 1); let reader = TraceReader::new(usize::MAX); + for run in &filtered { + let metadata = reader + .get_trace_metadata(&store, &QueryScope::All, &run.trace_ref) + .await? + .unwrap(); + assert_eq!(metadata.summary.trace_ref, run.trace_ref); + assert_eq!(metadata.summary.input_preview, run.input_preview); + let spans = reader + .get_trace_spans(&store, &QueryScope::All, &run.trace_ref, None, 10) + .await? + .unwrap(); + assert_eq!(spans.data.len(), 1); + assert_eq!(spans.data[0].input_preview, run.input_preview); + } + assert!( + reader + .get_trace_metadata(&store, &owned("two", &[]), &own[0].trace_ref) + .await? + .is_none() + ); let read = |trace_ref: String, contains: &'static str| { let reader = &reader; let store = &store; @@ -1045,7 +1067,14 @@ async fn query_help_discovers_live_schema_and_runs_its_examples( let writer = Connection::writer(&database.url)?; ensure_schema(&database.client, &writer, "trace_test", 7).await?; execute_write(&database, "CREATE USER help_reader").await?; - for table in ["otel_traces", "agent_traces_by_key", "spend_logs"] { + for table in [ + "traces", + "spans", + "calls", + "otel_traces", + "agent_traces_by_key", + "spend_logs", + ] { execute_write( &database, &format!("GRANT SELECT ON trace_test.{table} TO help_reader"), @@ -1125,7 +1154,14 @@ async fn query_help_discovers_live_schema_and_runs_its_examples( ); let guide = help["guide"].as_str().ok_or("missing rendered guide")?; assert!(guide.starts_with("Trace SQL query guide")); - for table in ["otel_traces", "agent_traces_by_key", "spend_logs"] { + for table in [ + "traces", + "spans", + "calls", + "otel_traces", + "agent_traces_by_key", + "spend_logs", + ] { let described = read_json(&database, &format!("DESCRIBE TABLE {table}")).await?; let schema = help["tables"] .as_array() @@ -1168,8 +1204,13 @@ async fn query_help_discovers_live_schema_and_runs_its_examples( !populated ); let tables = help["tables"].as_array().ok_or("missing tables")?; - assert_eq!(tables.len(), 3); - let columns = tables[0]["columns"].as_array().ok_or("missing columns")?; + assert_eq!(tables.len(), 6); + let columns = tables + .iter() + .find(|table| table["name"] == "otel_traces") + .ok_or("missing raw span table")?["columns"] + .as_array() + .ok_or("missing columns")?; for field in NORMALIZED_FIELD_DEFINITIONS { assert!( columns @@ -1222,8 +1263,8 @@ async fn query_help_discovers_live_schema_and_runs_its_examples( assert!(guide.contains("CustomColumn: String")); assert!(!guide.contains("private-metadata-value")); assert!(guide.contains("JSONExtractRaw(metadata, '&{{key}}', 'nested.key')")); - assert!(guide.contains("SpanAttributes['custom.tag']")); - assert!(guide.contains("ResourceAttributes['custom.resource']")); + assert!(guide.contains("span_attributes['custom.tag']")); + assert!(guide.contains("resource_attributes['custom.resource']")); assert_eq!(help["attributes"][0]["fields"][0]["key"], "custom.tag"); assert_eq!(help["attributes"][1]["fields"][0]["key"], "custom.resource"); for field in fields { @@ -1285,7 +1326,7 @@ async fn query_help_discovers_live_schema_and_runs_its_examples( "{sql}" ); if populated && example["name"] == "Traces correlated with LLM call metadata" { - assert_eq!(values["data"][0]["TraceId"], "trace-1"); + assert_eq!(values["data"][0]["trace_id"], "trace-1"); assert_eq!(values["data"][0]["spend"], 0.25); } } @@ -1310,7 +1351,14 @@ async fn query_help_preserves_schema_and_guide_when_discovery_hits_reader_limits "CREATE USER help_reader SETTINGS max_rows_to_read = 1", ) .await?; - for table in ["otel_traces", "agent_traces_by_key", "spend_logs"] { + for table in [ + "traces", + "spans", + "calls", + "otel_traces", + "agent_traces_by_key", + "spend_logs", + ] { execute_write( &database, &format!("GRANT SELECT ON trace_test.{table} TO help_reader"), @@ -1336,7 +1384,7 @@ async fn query_help_preserves_schema_and_guide_when_discovery_hits_reader_limits let help = serde_json::to_value( litellm_traces_clickhouse::query_help(&database.client, &reader).await?, )?; - assert_eq!(help["tables"].as_array().ok_or("tables")?.len(), 3); + assert_eq!(help["tables"].as_array().ok_or("tables")?.len(), 6); assert!(!help["examples"].as_array().ok_or("examples")?.is_empty()); assert_eq!( help["normalized_fields"] @@ -1495,6 +1543,7 @@ async fn trusted_and_sql_readers_share_request_log_visibility( .calls( &owned(user, &teams), &CallQuery { + as_of_ms: u64::MAX, window: timestamp / 1_000_000 - 1..timestamp / 1_000_000 + 1, response_ids: vec!["shared-response".into()], request_ids: Vec::new(), @@ -1577,7 +1626,7 @@ async fn rollup_cost_completeness_preserves_missing_ids_and_fails_closed_for_his ) .await?; let listed = listed["data"].as_array().ok_or("missing runs")?; - assert_eq!(listed.len(), 4); + assert_eq!(listed.len(), 3); let owner = list_runs(&database, &reader, &owned("owner", &[]), window, None, 10).await?; let owner: std::collections::BTreeSet<_> = owner["data"] .as_array() diff --git a/litellm-rust/crates/traces-clickhouse/tests/queries.rs b/litellm-rust/crates/traces-clickhouse/tests/queries.rs index 758c7d5d54c..24ba86222da 100644 --- a/litellm-rust/crates/traces-clickhouse/tests/queries.rs +++ b/litellm-rust/crates/traces-clickhouse/tests/queries.rs @@ -176,6 +176,155 @@ async fn documented_failed_spans_filters_status_after_selecting_the_canonical_co Ok(()) } +#[rstest] +#[case::admin(QueryScope::All)] +#[case::team(QueryScope::Owned { user_id: String::new(), team_ids: vec!["team-a".into()] })] +#[tokio::test] +async fn logical_trace_core_metrics_agree_with_curated_metrics_after_duplicate_exports( + #[future(awt)] migrated_database: TestResult, + fixture_clock: TestResult, + #[case] scope: QueryScope, +) -> TestResult { + let fixture = migrated_database?; + let start_ns = fixture_clock? * 1_000_000_000 + 900_000; + let rows = [ + ( + "key-a", + "root", + "", + start_ns, + 900_000, + "STATUS_CODE_OK", + 1, + 12, + ), + ( + "key-a", + "root", + "", + start_ns, + 9_000_000, + "STATUS_CODE_ERROR", + 2, + 999, + ), + ( + "key-a", + "child", + "root", + start_ns + 1_200_000, + 700_000, + "STATUS_CODE_ERROR", + 1, + 3, + ), + ( + "key-a", + "child", + "root", + start_ns + 1_200_000, + 100_000, + "STATUS_CODE_OK", + 2, + 999, + ), + ( + "key-alt", + "root", + "", + start_ns, + 7_000_000, + "STATUS_CODE_OK", + 1, + 5, + ), + ] + .map( + |(key, span, parent, timestamp, duration, status, received, tokens)| { + BTreeMap::from([ + ("TeamId".into(), json!("team-a")), + ("ApiKeyHash".into(), json!(key)), + ("TraceId".into(), json!("duplicates")), + ("SpanId".into(), json!(span)), + ("ParentSpanId".into(), json!(parent)), + ("SpanName".into(), json!(span)), + ( + "ObservationType".into(), + json!(if parent.is_empty() { "agent" } else { "llm" }), + ), + ( + "InputPreview".into(), + json!(if parent.is_empty() { "" } else { "child input" }), + ), + ("Timestamp".into(), json!(timestamp)), + ("Duration".into(), json!(duration)), + ("StatusCode".into(), json!(status)), + ("EngineReceivedMs".into(), json!(received)), + ("InputTokens".into(), json!(tokens)), + ]) + }, + ); + let writer = litellm_traces_clickhouse::Connection::writer(&fixture.database.url)?; + let encoded = litellm_traces_clickhouse::encode_rows(rows.to_vec())?; + fixture + .database + .client + .post(writer.url().clone()) + .body(format!( + "INSERT INTO {}.otel_traces FORMAT JSONEachRow\n{encoded}", + fixtures::DATABASE + )) + .send() + .await? + .error_for_status()?; + let connection = fixture + .readers + .connection(&fixture.database.client, &scope, "fixture-secret") + .await?; + let logical: QueryResult = serde_json::from_str(&query_sql( + &fixture.database.client, &connection, + "SELECT id, api_key_hash, span_count, error_count, duration_ns, duration_ms, span_input_tokens, input_preview, root_status, has_error, agent_span_count, agent_label_count, llm_span_count, tool_span_count FROM traces ORDER BY id", + ).await?)?; + assert_eq!(logical.data.len(), 2); + let primary = logical + .data + .iter() + .find(|row| row["api_key_hash"] == "key-a") + .ok_or("missing primary trace")?; + assert_eq!(primary["span_count"], 2); + assert_eq!(primary["error_count"], 1); + assert_eq!(primary["duration_ns"], 1_900_000); + assert_eq!(primary["duration_ms"], 1.9); + assert_eq!(primary["span_input_tokens"], 15); + assert_eq!(primary["agent_span_count"], 1); + assert_eq!(primary["agent_label_count"], 1); + assert_eq!(primary["llm_span_count"], 1); + assert_eq!(primary["tool_span_count"], 0); + assert_eq!(primary["input_preview"], "child input"); + assert_eq!(primary["root_status"], "ok"); + assert_eq!(primary["has_error"], true); + let alternate = logical + .data + .iter() + .find(|row| row["api_key_hash"] == "key-alt") + .ok_or("missing alternate trace")?; + assert_eq!(alternate["span_count"], 1); + let store = ClickHouseTraces::new(fixture.database.client.clone(), connection); + let curated = store.runs(&scope, &newest(10, None)).await?; + for row in curated { + let sql_row = logical + .data + .iter() + .find(|value| value["id"] == row.trace_ref) + .ok_or("missing logical trace")?; + assert_eq!(sql_row["span_count"], row.span_count); + assert_eq!(sql_row["error_count"], row.error_count); + assert_eq!(sql_row["duration_ns"], row.duration_ns); + assert_eq!(sql_row["input_preview"], row.input_preview); + } + Ok(()) +} + #[fixture] fn fixture_clock() -> TestResult { let spans = litellm_traces::decode_otlp( diff --git a/litellm-rust/crates/traces-clickhouse/tests/queries/rollups.expected.json b/litellm-rust/crates/traces-clickhouse/tests/queries/rollups.expected.json index af24543bab3..b962ea17954 100644 --- a/litellm-rust/crates/traces-clickhouse/tests/queries/rollups.expected.json +++ b/litellm-rust/crates/traces-clickhouse/tests/queries/rollups.expected.json @@ -1,87 +1,94 @@ { "admin": [ { + "id": "62AB4F77353031F31CA3F7DBDCDB9384D00A2C2D683F8E318DD52CEDC8F50F12", "team": "team-a", "api_key": "key-a", "trace_id": "01010101010101010101010101010101", "name": "review", "spans": 3, - "llm_calls": 1, + "llm_span_count": 1, "errors": 1, - "input_tokens": 12, - "output_tokens": 6 + "span_input_tokens": 12, + "span_output_tokens": 6 }, { + "id": "191720068A91A381A3C4764EF56D734B47B21ECFF199F0CDECEC265D7FCC0F10", "team": "team-a", "api_key": "key-alt", "trace_id": "01010101010101010101010101010101", "name": "alternate", "spans": 1, - "llm_calls": 0, + "llm_span_count": 0, "errors": 0, - "input_tokens": 0, - "output_tokens": 0 + "span_input_tokens": 0, + "span_output_tokens": 0 }, { + "id": "9C868E6B1426BCC5379F5D413F185B57ADBC324A0C58C93311784730A2D3FE12", "team": "team-b", "api_key": "key-b", "trace_id": "01010101010101010101010101010101", "name": "other-team", "spans": 1, - "llm_calls": 0, + "llm_span_count": 0, "errors": 0, - "input_tokens": 0, - "output_tokens": 0 + "span_input_tokens": 0, + "span_output_tokens": 0 } ], "team": [ { + "id": "62AB4F77353031F31CA3F7DBDCDB9384D00A2C2D683F8E318DD52CEDC8F50F12", "team": "team-a", "api_key": "key-a", "trace_id": "01010101010101010101010101010101", "name": "review", "spans": 3, - "llm_calls": 1, + "llm_span_count": 1, "errors": 1, - "input_tokens": 12, - "output_tokens": 6 + "span_input_tokens": 12, + "span_output_tokens": 6 }, { + "id": "191720068A91A381A3C4764EF56D734B47B21ECFF199F0CDECEC265D7FCC0F10", "team": "team-a", "api_key": "key-alt", "trace_id": "01010101010101010101010101010101", "name": "alternate", "spans": 1, - "llm_calls": 0, + "llm_span_count": 0, "errors": 0, - "input_tokens": 0, - "output_tokens": 0 + "span_input_tokens": 0, + "span_output_tokens": 0 } ], "key": [ { + "id": "62AB4F77353031F31CA3F7DBDCDB9384D00A2C2D683F8E318DD52CEDC8F50F12", "team": "team-a", "api_key": "key-a", "trace_id": "01010101010101010101010101010101", "name": "review", "spans": 3, - "llm_calls": 1, + "llm_span_count": 1, "errors": 1, - "input_tokens": 12, - "output_tokens": 6 + "span_input_tokens": 12, + "span_output_tokens": 6 } ], "other_team": [ { + "id": "9C868E6B1426BCC5379F5D413F185B57ADBC324A0C58C93311784730A2D3FE12", "team": "team-b", "api_key": "key-b", "trace_id": "01010101010101010101010101010101", "name": "other-team", "spans": 1, - "llm_calls": 0, + "llm_span_count": 0, "errors": 0, - "input_tokens": 0, - "output_tokens": 0 + "span_input_tokens": 0, + "span_output_tokens": 0 } ] } diff --git a/litellm-rust/crates/traces-clickhouse/tests/query_access.rs b/litellm-rust/crates/traces-clickhouse/tests/query_access.rs index 362661ec85c..cf3beedc458 100644 --- a/litellm-rust/crates/traces-clickhouse/tests/query_access.rs +++ b/litellm-rust/crates/traces-clickhouse/tests/query_access.rs @@ -1,8 +1,10 @@ use std::collections::BTreeMap; use litellm_http::Client; +use litellm_traces::api::SqlParameter; use litellm_traces_clickhouse::{ Connection, Error, QueryReaders, QueryScope, ensure_schema, query_help, query_sql, + query_sql_with_params, }; use rstest::{fixture, rstest}; use serde_json::{Value, json}; @@ -69,6 +71,13 @@ async fn queries_and_help_are_scoped_by_the_database( "SELECT id FROM (SELECT SpanId AS id FROM otel_traces UNION DISTINCT SELECT SpanId AS id FROM trace_test.otel_traces) ORDER BY id", "SELECT t.SpanId AS id FROM otel_traces t INNER JOIN spend_logs s ON t.SpanId = s.request_id ORDER BY id", "SELECT request_id AS id FROM spend_logs FINAL ORDER BY id", + "SELECT span_id AS id FROM spans ORDER BY id", + "SELECT span_id AS id FROM spans ORDER BY id FORMAT CSV", + "SELECT span_id AS id FROM spans ORDER BY id SETTINGS http_x_clickhouse_format_overrides_output_format = 0 FORMAT CSV", + "SELECT span_id AS id FROM trace_test.spans ORDER BY id", + "WITH visible AS (SELECT * FROM spans) SELECT span_id AS id FROM visible ORDER BY id", + "SELECT t.span_id AS id FROM spans t INNER JOIN calls c ON t.span_id = c.request_id ORDER BY id", + "SELECT request_id AS id FROM calls ORDER BY id", ]; for sql in queries { let body: Value = serde_json::from_str(&query_sql(&database.client, &reader, sql).await?)?; @@ -92,6 +101,15 @@ async fn queries_and_help_are_scoped_by_the_database( .await?, )?; assert_eq!(summary["data"][0]["count"], json!(expected.len())); + let canonical: Value = serde_json::from_str( + &query_sql( + &database.client, + &reader, + "SELECT sum(span_count) AS count FROM traces", + ) + .await?, + )?; + assert_eq!(canonical["data"], summary["data"]); let help = serde_json::to_string(&query_help(&database.client, &reader).await?)?; assert_eq!(help.contains("secret_b"), expected.contains(&"b")); assert_eq!(help.contains("secret-b"), expected.contains(&"b")); @@ -103,6 +121,138 @@ async fn queries_and_help_are_scoped_by_the_database( Ok(()) } +#[rstest] +#[tokio::test] +async fn native_parameters_preserve_values_and_cannot_change_reader_scope( + #[future(awt)] database: Result>, +) -> Result<(), Box> { + let database = database?; + let reader = database + .readers + .connection( + &database.client, + &QueryScope::Owned { + user_id: String::new(), + team_ids: vec!["team-a".into()], + }, + "test-secret", + ) + .await?; + let text = "quote' OR 1=1 --\\\n\t\r\0雪"; + let strings = vec!["a'b".to_owned(), "\\\n雪".to_owned()]; + let params = BTreeMap::from([ + ("text".into(), SqlParameter::String(text.into())), + ("signed".into(), SqlParameter::Integer(i64::MIN)), + ("unsigned".into(), SqlParameter::Unsigned(u64::MAX)), + ("number".into(), SqlParameter::Number(12.5)), + ("enabled".into(), SqlParameter::Boolean(true)), + ("optional".into(), SqlParameter::Null), + ("strings".into(), SqlParameter::Strings(strings.clone())), + ( + "team".into(), + SqlParameter::String("team-b' OR 1=1 --".into()), + ), + ]); + let body: Value = serde_json::from_str(&query_sql_with_params( + &database.client, &reader, + "SELECT {text:String} AS text, toString({signed:Int64}) AS signed, toString({unsigned:UInt64}) AS unsigned, {number:Float64} AS number, {enabled:Bool} AS enabled, {optional:Nullable(String)} AS optional, {strings:Array(String)} AS strings", + ¶ms, + ).await?)?; + assert_eq!( + body["data"], + json!([{ + "text": text, "signed": i64::MIN.to_string(), "unsigned": u64::MAX.to_string(), + "number": 12.5, "enabled": true, "optional": null, "strings": strings, + }]) + ); + let invisible: Value = serde_json::from_str( + &query_sql_with_params( + &database.client, + &reader, + "SELECT span_id FROM spans WHERE team_id = {team:String}", + ¶ms, + ) + .await?, + )?; + assert_eq!(invisible["data"], json!([])); + let foreign = BTreeMap::from([("team".into(), SqlParameter::String("team-b".into()))]); + let invisible: Value = serde_json::from_str( + &query_sql_with_params( + &database.client, + &reader, + "SELECT span_id FROM spans WHERE team_id = {team:String}", + &foreign, + ) + .await?, + )?; + assert_eq!(invisible["data"], json!([])); + Ok(()) +} + +#[rstest] +#[tokio::test] +async fn logical_views_never_expose_foreign_rows_within_a_mixed_owner_trace( + #[future(awt)] database: Result>, +) -> Result<(), Box> { + let database = database?; + for sql in [ + "INSERT INTO trace_test.otel_traces (TeamId, ApiKeyHash, TraceId, SpanId, SpanName, Timestamp, Duration, SpanAttributes, UserId, StatusCode) VALUES ('team-a', 'mixed-key', 'mixed', 'own', 'visible-root', now(), 1, map('visible', 'owned'), 'owner', 'STATUS_CODE_OK'), ('team-a', 'mixed-key', 'mixed', 'foreign', 'secret-name', now(), 999000000, map('secret-attribute', 'secret-value'), 'other', 'STATUS_CODE_ERROR')", + "INSERT INTO trace_test.spend_logs (team_id, api_key, request_id, trace_id, start_time, end_time, spend, metadata, user) VALUES ('team-a', 'mixed-key', 'own', 'mixed', now(), now(), 1.5, '{\"visible\":1}', 'owner'), ('team-a', 'mixed-key', 'foreign', 'mixed', now(), now(), 999, '{\"secret-cost\":999}', 'other')", + ] { + database + .client + .post(database.writer.url().clone()) + .body(sql) + .send() + .await? + .error_for_status()?; + } + let reader = database + .readers + .connection( + &database.client, + &QueryScope::Owned { + user_id: "owner".into(), + team_ids: vec![], + }, + "test-secret", + ) + .await?; + let spans: Value = serde_json::from_str( + &query_sql( + &database.client, + &reader, + "SELECT name, span_attributes FROM spans WHERE trace_id = 'mixed'", + ) + .await?, + )?; + assert_eq!( + spans["data"], + json!([{"name":"visible-root", "span_attributes":{"visible":"owned"}}]) + ); + let traces: Value = serde_json::from_str(&query_sql( + &database.client, &reader, + "SELECT name, span_count, error_count, duration_ns FROM traces WHERE trace_id = 'mixed'", + ).await?)?; + assert_eq!( + traces["data"], + json!([{"name":"visible-root", "span_count":1, "error_count":0, "duration_ns":1}]) + ); + let calls: Value = serde_json::from_str( + &query_sql( + &database.client, + &reader, + "SELECT request_id, spend, metadata FROM calls WHERE trace_id = 'mixed'", + ) + .await?, + )?; + assert_eq!( + calls["data"], + json!([{"request_id":"own", "spend":1.5, "metadata":"{\"visible\":1}"}]) + ); + Ok(()) +} + #[rstest] #[tokio::test] async fn reader_cache_reprovisions_after_credential_rotation() diff --git a/litellm-rust/crates/traces-clickhouse/tests/reads.rs b/litellm-rust/crates/traces-clickhouse/tests/reads.rs index 9866e8bc096..052b8ffe8c3 100644 --- a/litellm-rust/crates/traces-clickhouse/tests/reads.rs +++ b/litellm-rust/crates/traces-clickhouse/tests/reads.rs @@ -616,7 +616,15 @@ async fn an_oversized_span_keeps_the_run_list_available_with_partial_totals( }, ) .await?; - assert_eq!(cached.data, before.data); + let cached_run = cached + .data + .iter() + .find(|item| item.trace_ref == run.trace_ref) + .ok_or("missing cached run")?; + assert!(!cached_run.resolution_limited); + assert_eq!(cached_run.name, run.name); + assert_eq!(cached_run.span_count, run.span_count + 1); + assert!(cached_run.duration_ms > run.duration_ms); let (reader, store) = make_reader(client, connection); let after = reader .list_traces( @@ -637,7 +645,8 @@ async fn an_oversized_span_keeps_the_run_list_available_with_partial_totals( .find(|item| item.trace_ref == run.trace_ref) .ok_or("missing run")?; assert!(limited.resolution_limited); - assert_eq!(limited.span_count, 4); + assert_eq!(limited.span_count, cached_run.span_count); + assert_eq!(limited.duration_ms, cached_run.duration_ms); assert!( after .data diff --git a/litellm-rust/crates/traces-clickhouse/tests/search.rs b/litellm-rust/crates/traces-clickhouse/tests/search.rs index 0d24f2e3534..d99504a4493 100644 --- a/litellm-rust/crates/traces-clickhouse/tests/search.rs +++ b/litellm-rust/crates/traces-clickhouse/tests/search.rs @@ -32,7 +32,7 @@ fn filter(start_ms: i64, q: &str) -> RunFilter { RunFilter { start_ms, end_ms: WINDOW_END_MS, - search: RunSearch::parse(q), + search: RunSearch::parse(q).unwrap(), ..Default::default() } } @@ -216,7 +216,6 @@ async fn list_q_selects_matching_runs_before_paging( let reader = reader(); let cases: &[(&str, &[&str])] = &[ ("", &["gamma", "beta", "alpha"]), - ("status:", &["gamma", "beta", "alpha"]), (r#"name:"plan trip""#, &["gamma", "alpha"]), (r#"-name:"plan trip""#, &["beta"]), ("NAME:PLAN*", &["gamma", "alpha"]), @@ -225,9 +224,8 @@ async fn list_q_selects_matching_runs_before_paging( ("agent:RESEARCH*", &["alpha"]), ("agent:research", &[]), ("-agent:*", &["gamma"]), - ("status:error", &["beta"]), - ("status:ok", &["gamma", "alpha"]), - ("status:*r*", &["beta"]), + ("has_error:true", &["beta"]), + ("has_error:false", &["gamma", "alpha"]), ("model:gpt-x", &["gamma", "alpha"]), ("model:gpt", &[]), ("-model:gpt-x", &["beta"]), @@ -243,7 +241,6 @@ async fn list_q_selects_matching_runs_before_paging( ("flight_to", &[]), ("plan hello", &["gamma"]), ("plan -trace_id:gamma", &["alpha"]), - ("unknown:x", &[]), ]; for (q, expected) in cases { let page = reader @@ -472,7 +469,7 @@ async fn histogram_counts_use_the_displayed_bucket_boundaries( let filter = RunFilter { start_ms: T0_MS, end_ms: T0_MS + 10, - search: RunSearch::parse("trace_id:bucket*"), + search: RunSearch::parse("trace_id:bucket*").unwrap(), ..Default::default() }; let counts = store @@ -490,7 +487,7 @@ async fn histogram_counts_use_the_displayed_bucket_boundaries( }, ) .await?; - let histogram = litellm_traces::search::histogram(&counts, filter.start_ms, filter.end_ms, 3); + let histogram = litellm_traces::search::histogram(&counts, filter.window(), 3); for bucket in histogram.buckets { let expected = (T0_MS..T0_MS + 10) .filter(|start| (bucket.start_ms..bucket.end_ms).contains(start)) @@ -503,7 +500,8 @@ async fn histogram_counts_use_the_displayed_bucket_boundaries( #[rstest] #[case::agents_in_scope(RunField::Agent, "", &["researcher", "writer"])] #[case::names_by_frequency(RunField::Name, "", &["plan trip", "write report"])] -#[case::statuses_by_frequency(RunField::Status, "", &["ok", "error"])] +#[case::root_statuses_by_frequency(RunField::RootStatus, "", &["ok"])] +#[case::run_errors_by_frequency(RunField::HasError, "", &["false", "true"])] #[case::models_by_frequency(RunField::Model, "", &["gpt-x", "claude-y"])] #[case::needle_ignores_case(RunField::Name, "WR", &["write report"])] #[case::needle_matches_inside(RunField::Name, "trip", &["plan trip"])] @@ -545,7 +543,7 @@ async fn admin_scope_sees_every_team( #[rstest] #[case::by_model(RunField::Agent, "model:claude-y", &["writer"])] -#[case::by_status(RunField::Name, "status:error", &["write report"])] +#[case::by_status(RunField::Name, "has_error:true", &["write report"])] #[case::excluding(RunField::Model, "-trace_id:alpha", &["claude-y", "gpt-x"])] #[tokio::test] async fn values_narrow_to_runs_matching_the_search( @@ -752,13 +750,28 @@ async fn listing_runs_skips_out_of_window_span_rows( }; let listed = store.runs(&QueryScope::All, &query).await?; assert_eq!(listed.len(), 4); - let budget = 2 * table_rows(&fixture, "agent_traces_by_key").await? - + runs().iter().flat_map(rows).count() as u64; + let list_read = rows_read_by(&fixture, "FROM owned_runs").await?; + let reader = reader(); + let id = &listed[0].trace_ref; + let metadata = reader + .get_trace_metadata(&store, &QueryScope::All, id) + .await? + .unwrap(); + assert_eq!(metadata.summary.trace_ref, *id); + let spans = reader + .get_trace_spans(&store, &QueryScope::All, id, None, 500) + .await? + .unwrap(); + assert_eq!(spans.data.len() as u64, metadata.summary.span_count); + let budget = 4 * table_rows(&fixture, "agent_traces_by_key").await? + + 3 * runs().iter().flat_map(rows).count() as u64; let read = rows_read_by(&fixture, "FROM owned_runs").await?; - assert!( - read <= budget, - "listing current runs read {read} rows, exceeding their {budget} rollup and in-window span rows" - ); + for (operation, read) in [("list", list_read), ("canonical ID lookup", read)] { + assert!( + read <= budget, + "{operation} read {read} rows, exceeding the {budget}-row budget for bounded candidate, ownership and canonical scans" + ); + } Ok(()) } @@ -893,7 +906,7 @@ async fn add_attributes(fixture: &SeededDatabase) -> TestResult { #[case::missing_attribute("-attr.tenant.tier:*", &["gamma"])] #[case::missing_wildcard("attr.missing:*", &[])] #[case::two_attributes("attr.tenant.tier:gold attr.tenant.tier:silver", &[])] -#[case::attribute_and_field("attr.tenant.tier:* status:error", &["beta"])] +#[case::attribute_and_field("attr.tenant.tier:* has_error:true", &["beta"])] #[case::unknown_attribute("attr.missing:gold", &[])] #[tokio::test] async fn service_team_and_attribute_filters_select_runs( @@ -1047,7 +1060,7 @@ async fn metric_sorting_pages_by_the_deduplicated_displayed_values( assert_eq!(run.error_count, error_count); } let errors = RunFilter { - search: RunSearch::parse("trace_id:metric* status:error"), + search: RunSearch::parse("trace_id:metric* has_error:true").unwrap(), ..filter.clone() }; let filtered = reader @@ -1321,3 +1334,172 @@ async fn span_text_reads_ranges_of_each_listed_span( assert!(foreign.is_empty()); Ok(()) } + +#[rstest] +#[tokio::test] +async fn list_snapshot_excludes_late_spans_and_preserves_sorted_pages( + #[future] migrated_database: TestResult, +) -> TestResult { + let fixture = migrated_database.await?; + let writer = Connection::writer(&fixture.database.url)?; + let store = ClickHouseTraces::new( + fixture.database.client.clone(), + Connection::configured(&fixture.database.url, DATABASE, "default", "")?, + ); + let row = |trace: &str, span: &str, duration: u64, received: u64, error: bool, offset: i64| { + BTreeMap::from([ + ("Timestamp".into(), json!((T0_MS + offset) * 1_000_000)), + ("TraceId".into(), json!(trace)), + ("SpanId".into(), json!(span)), + ( + "ParentSpanId".into(), + json!(if span == "root" { "" } else { "root" }), + ), + ("SpanName".into(), json!(span)), + ("ServiceName".into(), json!("snapshot")), + ("ObservationType".into(), json!("chain")), + ("TeamId".into(), json!("team-a")), + ("ApiKeyHash".into(), json!("key-a")), + ("Duration".into(), json!(duration)), + ("EngineReceivedMs".into(), json!(received)), + ( + "StatusCode".into(), + json!(if error { + "STATUS_CODE_ERROR" + } else { + "STATUS_CODE_OK" + }), + ), + ]) + }; + let insert = |rows: Vec>| { + let client = fixture.database.client.clone(); + let url = writer.url().clone(); + async move { + client + .post(url) + .query(&[( + "query", + format!("INSERT INTO {DATABASE}.otel_traces FORMAT JSONEachRow"), + )]) + .body(encode_rows(rows)?) + .send() + .await? + .error_for_status()?; + TestResult::Ok(()) + } + }; + insert(vec![ + row("snapshot-a", "root", 1_000_000, 10, false, 0), + row("snapshot-b", "root", 2_000_000, 10, false, 0), + row("snapshot-c", "root", 3_000_000, 10, false, 0), + ]) + .await?; + let reader = reader(); + let filter = RunFilter { + as_of_ms: 10, + ..filter(T0_MS, "trace_id:snapshot*") + }; + let order = RunOrder { + key: RunSortKey::DurationMs, + descending: false, + }; + let first = reader + .list_traces(&store, &team_a(), &filter, order, &page(None, 1)) + .await?; + assert_eq!(first.data[0].trace_id, "snapshot-a"); + assert_eq!(first.window, filter.window()); + insert(vec![ + row("snapshot-a", "late", 10_000_000, 20, true, 0), + row("snapshot-new", "root", 100_000, 20, false, 0), + row("snapshot-c", "root", 100_000_000, 20, true, -1), + row("snapshot-c", "backdated-child", 1_000_000, 20, true, -1), + ]) + .await?; + let live_filter = RunFilter { + as_of_ms: 20, + ..filter.clone() + }; + let live = reader + .list_traces(&store, &team_a(), &live_filter, order, &page(None, 10)) + .await?; + let live_a = live + .data + .iter() + .find(|run| run.trace_id == "snapshot-a") + .unwrap(); + assert!(live_a.has_error); + assert_eq!(live_a.status, litellm_traces::SpanStatus::Ok); + let metadata = reader + .get_trace_metadata(&store, &team_a(), &live_a.trace_ref) + .await? + .unwrap(); + assert_eq!(metadata.summary.trace_ref, live_a.trace_ref); + assert!(metadata.summary.has_error); + let spans = reader + .get_trace_spans(&store, &team_a(), &live_a.trace_ref, None, 1) + .await? + .unwrap(); + let next = reader + .get_trace_spans( + &store, + &team_a(), + &live_a.trace_ref, + spans.next_cursor.as_deref(), + 1, + ) + .await? + .unwrap(); + assert_ne!(spans.data[0].span_id, next.data[0].span_id); + assert!(next.next_cursor.is_none()); + + let second = reader + .list_traces( + &store, + &team_a(), + &filter, + order, + &page(first.next_cursor, 1), + ) + .await?; + let third = reader + .list_traces( + &store, + &team_a(), + &filter, + order, + &page(second.next_cursor, 1), + ) + .await?; + assert_eq!(second.data[0].trace_id, "snapshot-b"); + assert_eq!(third.data[0].trace_id, "snapshot-c"); + assert_eq!(third.data[0].duration_ms, 3.0); + assert!(!third.data[0].has_error); + assert!(third.next_cursor.is_none()); + let repeated = reader + .list_traces(&store, &team_a(), &filter, order, &page(None, 10)) + .await?; + assert_eq!(repeated.data[0].duration_ms, 1.0); + assert!(!repeated.data[0].has_error); + assert_eq!(reader.count_traces(&store, &team_a(), &filter).await?, 3); + let histogram = reader.histogram(&store, &team_a(), &filter, 3).await?; + assert_eq!(histogram.window, filter.window()); + assert_eq!( + histogram + .buckets + .iter() + .map(|bucket| bucket.total) + .sum::(), + 3 + ); + let errors = RunFilter { + search: RunSearch::parse("has_error:true root_status:ok").unwrap(), + ..live_filter + }; + let errors = reader + .list_traces(&store, &team_a(), &errors, order, &page(None, 10)) + .await?; + assert_eq!(errors.data.len(), 1); + assert_eq!(errors.data[0].trace_id, "snapshot-a"); + Ok(()) +} diff --git a/litellm-rust/crates/traces/Cargo.toml b/litellm-rust/crates/traces/Cargo.toml index 12bb55551c3..268ce630403 100644 --- a/litellm-rust/crates/traces/Cargo.toml +++ b/litellm-rust/crates/traces/Cargo.toml @@ -6,12 +6,13 @@ license.workspace = true repository.workspace = true [features] -schema = ["dep:schemars"] +schema = ["dep:schemars", "dep:utoipa"] [dependencies] askama.workspace = true macro_rules_attribute.workspace = true schemars = { workspace = true, optional = true } +utoipa = { workspace = true, optional = true } indexmap = { version = "2", features = ["serde"] } litellm-llms-types.workspace = true opentelemetry-proto = { workspace = true, features = ["gen-tonic-messages", "trace", "with-serde"] } @@ -26,6 +27,7 @@ time.workspace = true base64.workspace = true criterion.workspace = true rstest.workspace = true +jsonschema = { version = "0.55.1", default-features = false } [[bench]] name = "resource-fanout" diff --git a/litellm-rust/crates/traces/src/api.rs b/litellm-rust/crates/traces/src/api.rs new file mode 100644 index 00000000000..c9567906d52 --- /dev/null +++ b/litellm-rust/crates/traces/src/api.rs @@ -0,0 +1,327 @@ +use std::collections::BTreeMap; + +use serde::{ + Deserialize, Deserializer, + de::{Error, Unexpected}, +}; +use serde_json::Value; + +use crate::{AgentNode, Span, TraceSummary}; + +#[cfg(feature = "schema")] +pub mod openapi; + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[serde(deny_unknown_fields)] +pub struct TraceQueryWindow { + pub start_ms: i64, + pub end_ms: i64, + pub as_of_ms: u64, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] +#[serde(rename_all = "snake_case")] +pub enum TraceSortField { + #[default] + StartMs, + DurationMs, + SpanCount, + ErrorCount, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] +#[serde(rename_all = "lowercase")] +pub enum TraceSortDirection { + Asc, + #[default] + Desc, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(deny_unknown_fields)] +pub struct TraceNoQueryRequest {} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(deny_unknown_fields)] +pub struct TraceListRequest { + #[serde(default)] + pub as_of_ms: Option, + #[serde(default)] + pub start_ms: Option, + #[serde(default)] + pub end_ms: Option, + #[serde(default)] + #[cfg_attr(feature = "schema", schemars(length(max = 1000)))] + #[serde(deserialize_with = "text::<_, 1000>")] + pub q: String, + #[serde(default)] + #[cfg_attr(feature = "schema", schemars(length(max = 512)))] + #[serde(deserialize_with = "cursor")] + pub cursor: Option, + #[serde(default = "list_page_size")] + #[cfg_attr(feature = "schema", schemars(range(min = 1, max = 500)))] + #[serde(deserialize_with = "integer::<_, 1, 500>")] + pub page_size: u16, + #[serde(default)] + pub sort_by: TraceSortField, + #[serde(default)] + pub sort_dir: TraceSortDirection, +} + +const fn list_page_size() -> u16 { + 50 +} + +const fn span_page_size() -> u16 { + 100 +} + +const fn histogram_buckets() -> u16 { + 60 +} + +const fn value_limit() -> u16 { + 20 +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(deny_unknown_fields)] +pub struct TraceHistogramRequest { + #[serde(default)] + pub start_ms: Option, + #[serde(default)] + pub end_ms: Option, + #[serde(default)] + pub as_of_ms: Option, + #[serde(default)] + #[cfg_attr(feature = "schema", schemars(length(max = 1000)))] + #[serde(deserialize_with = "text::<_, 1000>")] + pub q: String, + #[serde(default = "histogram_buckets")] + #[cfg_attr(feature = "schema", schemars(range(min = 1, max = 240)))] + #[serde(deserialize_with = "integer::<_, 1, 240>")] + pub buckets: u16, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(deny_unknown_fields)] +pub struct TraceValuesRequest { + #[serde(default)] + pub start_ms: Option, + #[serde(default)] + pub end_ms: Option, + #[serde(default)] + pub as_of_ms: Option, + #[serde(default)] + #[cfg_attr(feature = "schema", schemars(length(max = 1000)))] + #[serde(deserialize_with = "text::<_, 1000>")] + pub q: String, + #[serde(default)] + #[cfg_attr(feature = "schema", schemars(length(max = 200)))] + #[serde(deserialize_with = "text::<_, 200>")] + pub contains: String, + #[serde(default = "value_limit")] + #[cfg_attr(feature = "schema", schemars(range(min = 1, max = 100)))] + #[serde(deserialize_with = "integer::<_, 1, 100>")] + pub limit: u16, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(deny_unknown_fields)] +pub struct TraceSpanPageRequest { + #[serde(default)] + #[cfg_attr(feature = "schema", schemars(length(max = 512)))] + #[serde(deserialize_with = "cursor")] + pub cursor: Option, + #[serde(default = "span_page_size")] + #[cfg_attr(feature = "schema", schemars(range(min = 1, max = 500)))] + #[serde(deserialize_with = "integer::<_, 1, 500>")] + pub page_size: u16, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(deny_unknown_fields)] +pub struct TraceErrorPageRequest { + #[serde(default)] + #[cfg_attr(feature = "schema", schemars(length(max = 512)))] + #[serde(deserialize_with = "cursor")] + pub cursor: Option, +} + +#[macro_rules_attribute::apply(response_type)] +#[derive(Clone, Debug, PartialEq)] +pub struct TraceMetadata { + pub summary: TraceSummary, + pub agents: Vec, +} + +#[macro_rules_attribute::apply(response_type)] +#[derive(Clone, Debug, PartialEq)] +pub struct TraceSpansPage { + pub data: Vec, + pub next_cursor: Option, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug, PartialEq)] +#[serde(untagged)] +pub enum SqlParameter { + String(String), + Integer(i64), + Unsigned(u64), + Number(f64), + Boolean(bool), + Null, + Strings(Vec), +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(deny_unknown_fields)] +pub struct TraceQueryRequest { + #[cfg_attr(feature = "schema", schemars(length(min = 1)))] + #[serde(deserialize_with = "sql")] + pub sql: String, + #[serde(default)] + pub params: BTreeMap, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(untagged)] +pub enum UnsignedCount { + Integer(u64), + String(String), +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +pub struct TraceSQLColumn { + pub name: String, + #[serde(rename = "type")] + pub kind: String, + #[serde(flatten)] + pub extra: BTreeMap, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +pub struct TraceQueryStatistics { + pub elapsed: f64, + pub rows_read: UnsignedCount, + pub bytes_read: UnsignedCount, + #[serde(flatten)] + pub extra: BTreeMap, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +pub struct TraceSQLResponse { + pub meta: Vec, + #[cfg_attr(feature = "schema", schemars(extend("x-python-normalized" = {"type": "tuple[Mapping[str, JsonValue], ...]"})))] + pub data: Vec>, + pub rows: UnsignedCount, + pub statistics: TraceQueryStatistics, + #[serde(flatten)] + pub extra: BTreeMap, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[serde(rename_all = "snake_case")] +pub enum TraceProblemCode { + InvalidRequest, + Unauthorized, + Forbidden, + NotFound, + TraceChanged, + TooLarge, + Unavailable, + QueryRejected, + QueryLimitExceeded, + QueryUnavailable, + InternalError, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(deny_unknown_fields)] +pub struct TraceInvalidParam { + pub location: String, + pub reason: String, +} + +#[macro_rules_attribute::apply(wire_type)] +#[derive(Clone, Debug)] +#[serde(deny_unknown_fields)] +pub struct TraceProblem { + #[serde(rename = "type")] + pub kind: String, + pub title: String, + pub status: u16, + pub detail: String, + pub code: TraceProblemCode, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub database_code: Option, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + #[cfg_attr(feature = "schema", schemars(extend("default" = [])))] + pub errors: Vec, +} + +fn integer<'de, D: Deserializer<'de>, const MIN: u16, const MAX: u16>( + deserializer: D, +) -> Result { + let value = u16::deserialize(deserializer)?; + if (MIN..=MAX).contains(&value) { + return Ok(value); + } + Err(D::Error::invalid_value( + Unexpected::Unsigned(u64::from(value)), + &"integer within the documented range", + )) +} + +fn text<'de, D: Deserializer<'de>, const MAX: usize>(deserializer: D) -> Result { + let value = String::deserialize(deserializer)?; + if value.chars().count() <= MAX { + return Ok(value); + } + Err(D::Error::invalid_length( + value.chars().count(), + &"string within the documented length", + )) +} + +fn cursor<'de, D: Deserializer<'de>>(deserializer: D) -> Result, D::Error> { + let value = Option::::deserialize(deserializer)?; + if value + .as_ref() + .is_none_or(|cursor| cursor.chars().count() <= 512) + { + return Ok(value); + } + Err(D::Error::invalid_length( + value.as_ref().map_or(0, |cursor| cursor.chars().count()), + &"cursor within the documented length", + )) +} + +fn sql<'de, D: Deserializer<'de>>(deserializer: D) -> Result { + let value = String::deserialize(deserializer)?; + if !value.is_empty() { + return Ok(value); + } + Err(D::Error::invalid_value( + Unexpected::Str(&value), + &"nonempty SQL", + )) +} diff --git a/litellm-rust/crates/traces/src/api/openapi.rs b/litellm-rust/crates/traces/src/api/openapi.rs new file mode 100644 index 00000000000..0eb0d8998d2 --- /dev/null +++ b/litellm-rust/crates/traces/src/api/openapi.rs @@ -0,0 +1,282 @@ +use std::collections::BTreeMap; + +use schemars::Schema as JsonSchema; +use serde_json::{Value, json}; +use utoipa::openapi::{ + Content, Info, OpenApi, OpenApiBuilder, Ref, RefOr, Required, + path::{ + HttpMethod, OperationBuilder, Parameter, ParameterBuilder, ParameterIn, PathItem, + PathsBuilder, + }, + request_body::RequestBodyBuilder, + response::{ResponseBuilder, ResponsesBuilder}, + schema::{ComponentsBuilder, Schema}, + security::{Http, HttpAuthScheme, SecurityScheme}, +}; + +struct Endpoint { + path: &'static str, + method: HttpMethod, + operation: &'static str, + request: Option<&'static str>, + response: &'static str, + description: &'static str, +} + +const ENDPOINTS: [Endpoint; 9] = [ + Endpoint { + path: "/v1/traces", + method: HttpMethod::Get, + operation: "trace_list", + request: Some("TraceListRequest"), + response: "TracePage", + description: "Search trace summaries. All clauses in q must match. Free text searches trace_id, name and input preview as case-insensitive substrings. key:value clauses match whole values, with * as a wildcard; double quotes group spaces or literal colons. Prefix keyed clauses with - to exclude matches. Unknown keys, missing values and malformed quoting return invalid_request. name describes the physical root; input searches its preview, falling back to the first nonempty agent or LLM preview; agent, model and attr. match any span. root_status describes the root, has_error means any failed span. The default window is the last 24 hours. First-page ingestion timestamp cutoff is retained by the cursor. This excludes later-stamped exports, not delayed commits stamped before the cutoff, and is not a database transaction snapshot. Repeat q and sorting on continuation; omitted bounds reuse the cursor window. Sort ties use the canonical id in the same direction. The server may return fewer than page_size items to respect response limits. Continue until next_cursor is null. Span-derived metrics use the cutoff; spend enrichment is best effort", + }, + Endpoint { + path: "/v1/traces/histogram", + method: HttpMethod::Get, + operation: "trace_histogram", + request: Some("TraceHistogramRequest"), + response: "TraceHistogram", + description: "Count matching traces in equal-width [start_ms,end_ms) buckets. Reuse the list window and as_of_ms for matching span-derived membership. failed counts traces with any failed span. Agent groups count successful traces under their alphabetically first agent name, or service when no agent name exists", + }, + Endpoint { + path: "/v1/traces/values/{field}", + method: HttpMethod::Get, + operation: "trace_values", + request: Some("TraceValuesRequest"), + response: "RunValues", + description: "Return the most common distinct values among matching traces. contains is case-insensitive. limit is a top-K suggestion limit, not a pagination size. Reuse list window and as_of_ms for matching membership", + }, + Endpoint { + path: "/v1/traces/{id}", + method: HttpMethod::Get, + operation: "trace_get", + request: Some("TraceNoQueryRequest"), + response: "TraceMetadata", + description: "Read summary and agent metadata by canonical id. trace_id is the original OTLP id, which can repeat across ownership scopes. Spans are read through the separate spans collection", + }, + Endpoint { + path: "/v1/traces/{id}/spans", + method: HttpMethod::Get, + operation: "trace_spans", + request: Some("TraceSpanPageRequest"), + response: "TraceSpansPage", + description: "Read a bounded page of canonical spans. The cursor pins the graph version. Continue with the same page_size; a changed graph or expired reconstruction returns trace_changed and the traversal must restart", + }, + Endpoint { + path: "/v1/traces/{id}/spans/{span_id}", + method: HttpMethod::Get, + operation: "trace_span", + request: Some("TraceNoQueryRequest"), + response: "SpanDetail", + description: "Read raw input, output and attributes for one span. UI rendering is performed by the client", + }, + Endpoint { + path: "/v1/traces/{id}/spans/{span_id}/error", + method: HttpMethod::Get, + operation: "trace_error", + request: Some("TraceErrorPageRequest"), + response: "SpanErrorPage", + description: "Read bounded diagnostic text pages. Cursor validation detects content changes and requires restarting the traversal", + }, + Endpoint { + path: "/v1/traces/query", + method: HttpMethod::Post, + operation: "trace_query", + request: Some("TraceQueryRequest"), + response: "TraceSQLResponse", + description: "Execute read-only ClickHouse SQL under authenticated row policies and fixed resource limits. Bind params with native {name:Type} placeholders. Results are always ClickHouse JSON; 64-bit integers may be strings. SQL callers control ORDER BY, LIMIT and keyset continuation. Exceeding a resource limit fails instead of returning partial success", + }, + Endpoint { + path: "/v1/traces/query/help", + method: HttpMethod::Get, + operation: "trace_query_help", + request: Some("TraceNoQueryRequest"), + response: "TraceQueryHelp", + description: "Discover current SQL schema, logical views, scoped examples and resource limits", + }, +]; + +fn rewrite_refs(value: Value) -> Value { + match value { + Value::Object(object) => Value::Object( + object + .into_iter() + .filter_map(|(key, value)| { + if key == "$schema" || key == "$defs" { + return None; + } + let rewritten = match (key.as_str(), value) { + ("$ref", Value::String(reference)) => { + Value::String(reference.replace("#/$defs/", "#/components/schemas/")) + } + (_, value) => rewrite_refs(value), + }; + Some((key, rewritten)) + }) + .collect(), + ), + Value::Array(values) => Value::Array(values.into_iter().map(rewrite_refs).collect()), + value => value, + } +} + +fn schema(value: Value) -> RefOr { + serde_json::from_value(rewrite_refs(value)).expect("Rust JSON Schema is an OpenAPI 3.1 schema") +} + +fn query_parameters(name: &str, schemas: &BTreeMap<&str, JsonSchema>) -> Vec { + let Some(properties) = schemas[name].get("properties").and_then(Value::as_object) else { + return Vec::new(); + }; + properties + .iter() + .map(|(name, value)| { + ParameterBuilder::new() + .name(name) + .parameter_in(ParameterIn::Query) + .schema(Some(schema(value.clone()))) + .build() + }) + .collect() +} + +fn path_parameters(path: &str) -> impl Iterator + '_ { + path.split('/') + .filter_map(|part| { + part.strip_prefix('{') + .and_then(|value| value.strip_suffix('}')) + }) + .map(|name| { + ParameterBuilder::new() + .name(name) + .parameter_in(ParameterIn::Path) + .required(Required::True) + .schema(Some(if name == "field" { + RefOr::Ref(Ref::from_schema_name("RunField")) + } else { + schema(json!({"type":"string","minLength":1})) + })) + .build() + }) +} + +fn endpoint(endpoint: &Endpoint, schemas: &BTreeMap<&str, JsonSchema>) -> PathItem { + let success = ResponseBuilder::new() + .description("Success") + .content( + "application/json", + Content::new(Some(Ref::from_schema_name(endpoint.response))), + ) + .build(); + let responses = [400, 401, 403, 404, 409, 413, 422, 500, 501, 503] + .into_iter() + .fold( + ResponsesBuilder::new().response("200", success), + |builder, status| { + builder.response( + status.to_string(), + ResponseBuilder::new() + .description("Trace API problem") + .content( + "application/problem+json", + Content::new(Some(Ref::from_schema_name("TraceProblem"))), + ) + .build(), + ) + }, + ) + .build(); + let parameters = endpoint + .request + .filter(|_| endpoint.method == HttpMethod::Get) + .map(|request| query_parameters(request, schemas)) + .unwrap_or_default(); + let request_body = endpoint + .request + .filter(|_| endpoint.method == HttpMethod::Post) + .map(|request| { + RequestBodyBuilder::new() + .required(Some(Required::True)) + .content( + "application/json", + Content::new(Some(Ref::from_schema_name(request))), + ) + .build() + }); + let operation = OperationBuilder::new() + .operation_id(Some(endpoint.operation)) + .security(utoipa::openapi::security::SecurityRequirement::new( + "TraceBearer", + [""; 0], + )) + .tag("agent tracing") + .description(Some(endpoint.description)) + .parameters(Some(path_parameters(endpoint.path).chain(parameters))) + .request_body(request_body) + .responses(responses) + .build(); + PathItem::new(endpoint.method.clone(), operation) +} + +pub fn document(extra: BTreeMap<&'static str, JsonSchema>) -> OpenApi { + let roots: BTreeMap<_, _> = crate::schema::schemas() + .into_iter() + .filter(|(name, _)| { + *name == "RunField" || ENDPOINTS.iter().any(|endpoint| endpoint.response == *name) + }) + .chain(crate::schema::api_schemas()) + .chain(extra) + .collect(); + let definitions = roots + .values() + .flat_map(|root| { + root.get("$defs") + .and_then(Value::as_object) + .into_iter() + .flat_map(|defs| defs.iter()) + }) + .map(|(name, value)| (name.clone(), schema(value.clone()))); + let components = ComponentsBuilder::new() + .schemas_from_iter(definitions) + .schemas_from_iter( + roots + .iter() + .map(|(name, root)| (*name, schema(root.as_value().clone()))), + ) + .security_scheme( + "TraceBearer", + SecurityScheme::Http(Http::new(HttpAuthScheme::Bearer)), + ) + .build(); + let ingest = OperationBuilder::new().operation_id(Some("trace_ingest")) + .security(utoipa::openapi::security::SecurityRequirement::new("TraceBearer", [""; 0])).tag("agent tracing") + .description(Some("Export OTLP traces as JSON or protobuf, optionally gzip compressed. The OTLP protocol defines payloads and responses: https://opentelemetry.io/docs/specs/otlp/. Ownership is derived from authentication, never payload attributes")) + .request_body(Some(RequestBodyBuilder::new().required(Some(Required::True)) + .content("application/json", Content::new(Some(schema(json!({"type":"object"}))))) + .content("application/x-protobuf", Content::new(Some(schema(json!({"type":"string","format":"binary"}))))) + .build())) + .responses([200,400,401,403,413,429,501,503].into_iter().fold(ResponsesBuilder::new(), |builder, status| { + builder.response(status.to_string(), ResponseBuilder::new().description("OTLP protocol response") + .content("application/json", Content::new(Some(schema(json!({"type":"object"}))))) + .content("application/x-protobuf", Content::new(Some(schema(json!({"type":"string","format":"binary"}))))) + .build()) + }).build()).build(); + let paths = ENDPOINTS + .iter() + .fold(PathsBuilder::new(), |builder, item| { + builder.path(item.path, endpoint(item, &roots)) + }) + .path("/v1/traces", PathItem::new(HttpMethod::Post, ingest)) + .build(); + OpenApiBuilder::new() + .info(Info::new("LiteLLM Trace API", "1")) + .components(Some(components)) + .paths(paths) + .security(Some([utoipa::openapi::security::SecurityRequirement::new( + "TraceBearer", + [""; 0], + )])) + .build() +} diff --git a/litellm-rust/crates/traces/src/bin/export_schema.rs b/litellm-rust/crates/traces/src/bin/export_schema.rs index 25d1250ef12..35b8b2bf619 100644 --- a/litellm-rust/crates/traces/src/bin/export_schema.rs +++ b/litellm-rust/crates/traces/src/bin/export_schema.rs @@ -1,6 +1,9 @@ fn main() { - println!( - "{}", - serde_json::to_string_pretty(&litellm_traces::schema::schemas()).unwrap() - ); + let arguments = std::env::args().collect::>(); + let schemas = if arguments.get(1).is_some_and(|arg| arg == "--api") { + litellm_traces::schema::api_schemas() + } else { + litellm_traces::schema::schemas() + }; + println!("{}", serde_json::to_string_pretty(&schemas).unwrap()); } diff --git a/litellm-rust/crates/traces/src/error.rs b/litellm-rust/crates/traces/src/error.rs index e01b0541216..cb89d73cfc6 100644 --- a/litellm-rust/crates/traces/src/error.rs +++ b/litellm-rust/crates/traces/src/error.rs @@ -17,3 +17,7 @@ pub struct InvalidScope; #[derive(Debug, thiserror::Error)] #[error("invalid trace call key")] pub struct InvalidCallKey; + +#[derive(Debug, thiserror::Error)] +#[error("invalid trace search query")] +pub struct InvalidQuery; diff --git a/litellm-rust/crates/traces/src/lib.rs b/litellm-rust/crates/traces/src/lib.rs index 4a7afa5694b..f1837a07866 100644 --- a/litellm-rust/crates/traces/src/lib.rs +++ b/litellm-rust/crates/traces/src/lib.rs @@ -10,6 +10,7 @@ macro_rules_attribute::attribute_alias! { #[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]; } +pub mod api; mod error; mod normalize; mod otlp; @@ -27,7 +28,7 @@ mod ui; mod view; pub mod wire; -pub use error::{Error, InvalidCallKey, InvalidScope}; +pub use error::{Error, InvalidCallKey, InvalidQuery, InvalidScope}; pub use normalize::{ AgentMetadata, AgentType, CallEvidence, CallEvidenceKind, CallKey, Integration, NormalizedSpan, ObservationType, diff --git a/litellm-rust/crates/traces/src/resolve/graph.rs b/litellm-rust/crates/traces/src/resolve/graph.rs index b891d1ee1bb..4a7f8e54d5f 100644 --- a/litellm-rust/crates/traces/src/resolve/graph.rs +++ b/litellm-rust/crates/traces/src/resolve/graph.rs @@ -41,11 +41,6 @@ impl<'a> Graph<'a> { self.by_id.get(row.parent_span_id.as_str()).copied() } - pub(super) fn is_root(&self, index: usize) -> bool { - let parent = &self.rows[index].parent_span_id; - parent.is_empty() || !self.by_id.contains_key(parent.as_str()) - } - pub(super) fn children(&self, index: usize) -> Vec { self.children .get(self.id(index)) @@ -137,8 +132,8 @@ mod tests { .map(|index| graph.id(index)) .collect(); assert_eq!(descendants, ["leaf", "middle", "sibling"].into()); - assert!(graph.is_root(3)); - assert!(!graph.is_root(0)); + assert!(graph.parent(3).is_none()); + assert!(graph.parent(0).is_some()); } #[rstest] diff --git a/litellm-rust/crates/traces/src/resolve/view.rs b/litellm-rust/crates/traces/src/resolve/view.rs index 07f97b60c49..6ef2289c96f 100644 --- a/litellm-rust/crates/traces/src/resolve/view.rs +++ b/litellm-rust/crates/traces/src/resolve/view.rs @@ -138,7 +138,7 @@ pub fn resolve_trace( rows: &[SpanRow], spend: &[SpendRow], ) -> Option { - let first = rows.first()?; + let first = rows.iter().min_by_key(|row| (row.start_ns, &row.span_id))?; let resolution = Resolution::new(rows, spend); let trace_start_ns = rows.iter().map(|row| row.start_ns).min()?; let trace_end_ns = rows @@ -148,9 +148,17 @@ pub fn resolve_trace( let spans: Vec = (0..rows.len()) .map(|index| span(&resolution, index, trace_start_ns)) .collect(); - let root = (0..rows.len()) - .find(|index| resolution.graph.is_root(*index)) - .unwrap_or_default(); + let root = rows + .iter() + .enumerate() + .filter(|(_, row)| row.parent_span_id.is_empty()) + .min_by_key(|(_, row)| (row.start_ns, &row.span_id)) + .or_else(|| { + rows.iter() + .enumerate() + .min_by_key(|(_, row)| (row.start_ns, &row.span_id)) + }) + .map(|(index, _)| index)?; let agents = agents(&resolution); let calls = &resolution.model_calls; let counted: Vec<&SpanRow> = if calls.is_empty() { @@ -158,16 +166,14 @@ pub fn resolve_trace( } else { calls.iter().map(|call| &rows[*call]).collect() }; - let first_input = spans + let first_input = rows .iter() - .zip(rows) - .enumerate() - .filter(|(_, (span, _))| { - !span.input_preview.is_empty() - && matches!(span.kind, ObservationType::Agent | ObservationType::Llm) + .filter(|row| { + !row.input_preview.is_empty() + && matches!(row.kind, ObservationType::Agent | ObservationType::Llm) }) - .min_by_key(|(index, (_, row))| (row.start_ns, *index)) - .map(|(_, (span, _))| span.input_preview.clone()) + .min_by_key(|row| (row.start_ns, &row.span_id)) + .map(|row| row.input_preview.clone()) .unwrap_or_default(); let summary = TraceSummary { resolution_limited: false, @@ -185,7 +191,8 @@ pub fn resolve_trace( input_preview: optional(&spans[root].input_preview).unwrap_or(first_input), start_time: iso_time(trace_start_ns.div_euclid(1_000_000)), duration_ms: (trace_end_ns - i128::from(trace_start_ns)) as f64 / NANOS_PER_MS, - status: spans[root].status, + status: rows[root].status, + has_error: spans.iter().any(|span| span.status == SpanStatus::Error), span_count: spans.len() as u64, agent_count: agents.len() as u64, agent_invocations: agents.iter().map(|agent| agent.invocations).sum(), @@ -226,6 +233,7 @@ pub fn listed_summary(row: &RunRow) -> TraceSummary { start_time: iso_time(row.start_ms), duration_ms: row.duration_ns as f64 / NANOS_PER_MS, status: row.status, + has_error: row.error_count > 0, span_count: row.span_count, agent_count: row.agent_count, agent_invocations: if row.agent_invocations == 0 { diff --git a/litellm-rust/crates/traces/src/schema.rs b/litellm-rust/crates/traces/src/schema.rs index 9ef05f1ce74..16422b6ba23 100644 --- a/litellm-rust/crates/traces/src/schema.rs +++ b/litellm-rust/crates/traces/src/schema.rs @@ -9,6 +9,13 @@ pub fn flag(_: &mut SchemaGenerator) -> Schema { .unwrap() } +fn integer_value(value: &serde_json::Value) -> Option { + value + .as_i64() + .map(i128::from) + .or_else(|| value.as_u64().map(i128::from)) +} + pub fn integer_bounds(schema: &mut Schema) { let bounds = match schema.get("format").and_then(serde_json::Value::as_str) { Some("uint8") => Some((json!(0), json!(u8::MAX))), @@ -22,13 +29,19 @@ pub fn integer_bounds(schema: &mut Schema) { _ => None, }; if let Some((minimum, maximum)) = bounds { - schema.insert("minimum".to_owned(), minimum); - schema.insert("maximum".to_owned(), maximum); + let existing_minimum = schema.get("minimum").and_then(integer_value); + let existing_maximum = schema.get("maximum").and_then(integer_value); + if existing_minimum.is_none_or(|bound| bound < integer_value(&minimum).unwrap()) { + schema.insert("minimum".to_owned(), minimum); + } + if existing_maximum.is_none_or(|bound| bound > integer_value(&maximum).unwrap()) { + schema.insert("maximum".to_owned(), maximum); + } } schemars::transform::transform_subschemas(&mut integer_bounds, schema); } -fn received() -> Schema { +pub(crate) fn received() -> Schema { SchemaSettings::draft2020_12() .for_deserialize() .with_transform(integer_bounds) @@ -36,7 +49,7 @@ fn received() -> Schema { .into_root_schema_for::() } -fn emitted() -> Schema { +pub(crate) fn emitted() -> Schema { SchemaSettings::draft2020_12() .for_serialize() .with_transform(integer_bounds) @@ -59,3 +72,43 @@ pub fn schemas() -> BTreeMap<&'static str, Schema> { ("RunOrder", received::()), ]) } + +pub fn api_schemas() -> BTreeMap<&'static str, Schema> { + BTreeMap::from([ + ( + "TraceNoQueryRequest", + received::(), + ), + ( + "TraceListRequest", + received::(), + ), + ( + "TraceHistogramRequest", + received::(), + ), + ( + "TraceValuesRequest", + received::(), + ), + ( + "TraceSpanPageRequest", + received::(), + ), + ( + "TraceErrorPageRequest", + received::(), + ), + ( + "TraceQueryRequest", + received::(), + ), + ("TraceMetadata", emitted::()), + ("TraceSpansPage", emitted::()), + ("TraceProblem", emitted::()), + ( + "TraceSQLResponse", + emitted::(), + ), + ]) +} diff --git a/litellm-rust/crates/traces/src/search.rs b/litellm-rust/crates/traces/src/search.rs index b6e437e7289..bb2b903449a 100644 --- a/litellm-rust/crates/traces/src/search.rs +++ b/litellm-rust/crates/traces/src/search.rs @@ -20,7 +20,8 @@ use crate::store::RunCount; pub enum RunField { Name, Agent, - Status, + RootStatus, + HasError, Model, Input, TraceId, @@ -48,15 +49,38 @@ impl SearchKey { } } -#[derive(Clone, Debug, Default, Eq, PartialEq, Serialize)] +#[derive(Clone, Debug, Eq, PartialEq, Serialize)] pub struct RunFilter { pub start_ms: i64, pub end_ms: i64, + pub as_of_ms: u64, pub search: RunSearch, /// When not empty, only these runs can match. pub trace_refs: Vec, } +impl RunFilter { + pub fn window(&self) -> crate::api::TraceQueryWindow { + crate::api::TraceQueryWindow { + start_ms: self.start_ms, + end_ms: self.end_ms, + as_of_ms: self.as_of_ms, + } + } +} + +impl Default for RunFilter { + fn default() -> Self { + Self { + start_ms: 0, + end_ms: 0, + as_of_ms: u64::MAX, + search: RunSearch::default(), + trace_refs: Vec::new(), + } + } +} + #[derive(Clone, Debug, Eq, PartialEq, Serialize)] pub struct FieldFilter { pub key: SearchKey, @@ -74,57 +98,41 @@ pub struct RunSearch { } impl RunSearch { - /// Mirrors the dashboard's search box: `key:value` filters on known keys, `-key:value` negates, - /// `*` globs, double quotes keep spaces, and anything else is a free-text term. - /// A key typed without a value yet narrows nothing. - pub fn parse(q: &str) -> Self { - let (text, filters): (Vec<_>, Vec<_>) = tokens(q) - .map(clause) - .filter(|clause| !clause.value().is_empty()) + pub fn parse(q: &str) -> Result { + if q.chars().count() > 1000 { + return Err(crate::error::InvalidQuery); + } + let clauses = tokens(q) + .map(|token| token.and_then(clause)) + .collect::, _>>()?; + let (text, filters): (Vec<_>, Vec<_>) = clauses + .into_iter() .partition(|clause| matches!(clause, Clause::Text(_))); - Self { + Ok(Self { text: text .into_iter() - .map(|clause| clause.value().to_owned()) + .filter_map(|clause| match clause { + Clause::Text(text) => Some(text), + _ => None, + }) .collect(), filters: filters .into_iter() .filter_map(|clause| match clause { - Clause::Field { - key, - exclude, - value, - } => Some(FieldFilter { - key, - pattern: value, - exclude, - }), - Clause::Text(_) => None, + Clause::Field(filter) => Some(filter), + _ => None, }) .collect(), - } + }) } } enum Clause { Text(String), - Field { - key: SearchKey, - exclude: bool, - value: String, - }, + Field(FieldFilter), } -impl Clause { - fn value(&self) -> &str { - match self { - Self::Text(value) | Self::Field { value, .. } => value, - } - } -} - -/// Whitespace-separated tokens; a double-quoted stretch keeps its spaces, and an unclosed quote runs to the end. -fn tokens(q: &str) -> impl Iterator { +fn tokens(q: &str) -> impl Iterator> { let mut rest = q; std::iter::from_fn(move || { rest = rest.trim_start(); @@ -134,42 +142,63 @@ fn tokens(q: &str) -> impl Iterator { let mut quoted = false; let end = rest .char_indices() - .find(|&(_, char)| { - if char == '"' { + .find(|&(_, ch)| { + if ch == '"' { quoted = !quoted; } - !quoted && char.is_whitespace() + !quoted && ch.is_whitespace() }) .map_or(rest.len(), |(index, _)| index); let (token, tail) = rest.split_at(end); rest = tail; - Some(token) + Some(if quoted { + Err(crate::error::InvalidQuery) + } else { + Ok(token) + }) }) } -fn unquote(raw: &str) -> String { - raw.strip_prefix('"') - .map(|inner| inner.strip_suffix('"').unwrap_or(inner)) - .filter(|inner| !inner.contains('"')) - .unwrap_or(raw) - .to_owned() +fn value(raw: &str) -> Result { + let value = if raw.starts_with('"') && raw.ends_with('"') && raw.len() >= 2 { + &raw[1..raw.len() - 1] + } else { + raw + }; + if value.is_empty() || value.contains('"') { + return Err(crate::error::InvalidQuery); + } + Ok(value.to_owned()) } -fn clause(raw: &str) -> Clause { +fn clause(raw: &str) -> Result { + if raw.starts_with('"') { + return value(raw).map(Clause::Text); + } let (exclude, body) = raw .strip_prefix('-') .map_or((false, raw), |body| (true, body)); - let field = body - .split_once(':') - .and_then(|(key, value)| SearchKey::parse(key).map(|key| (key, value))); - match field { - Some((key, value)) => Clause::Field { - key, - exclude, - value: unquote(value), + let Some((key, raw_value)) = body.split_once(':') else { + return value(raw).map(Clause::Text); + }; + let key = SearchKey::parse(key).ok_or(crate::error::InvalidQuery)?; + let pattern = value(raw_value)?; + let pattern = match key { + SearchKey::Field(RunField::RootStatus) => match pattern.to_ascii_lowercase().as_str() { + "ok" | "error" | "unset" => pattern.to_ascii_lowercase(), + _ => return Err(crate::error::InvalidQuery), }, - None => Clause::Text(unquote(raw)), - } + SearchKey::Field(RunField::HasError) => match pattern.to_ascii_lowercase().as_str() { + "true" | "false" => pattern.to_ascii_lowercase(), + _ => return Err(crate::error::InvalidQuery), + }, + _ => pattern, + }; + Ok(Clause::Field(FieldFilter { + key, + pattern, + exclude, + })) } pub const MAX_HISTOGRAM_BUCKETS: u32 = 240; @@ -179,6 +208,7 @@ pub const MAX_RUN_VALUES: u32 = 100; #[macro_rules_attribute::apply(response_type)] #[derive(Clone, Debug, PartialEq)] pub struct TraceHistogram { + pub window: crate::api::TraceQueryWindow, pub buckets: Vec, } @@ -204,14 +234,23 @@ pub struct AgentRuns { #[macro_rules_attribute::apply(response_type)] #[derive(Clone, Debug, PartialEq)] pub struct RunValues { + pub window: crate::api::TraceQueryWindow, pub values: Vec, } /// Bucket `i` covers `[start + span * i / buckets, start + span * (i + 1) / buckets)`. -pub fn histogram(rows: &[RunCount], start_ms: i64, end_ms: i64, buckets: u32) -> TraceHistogram { - let span = i128::from(end_ms - start_ms); - let edge = |index: u32| start_ms + (span * i128::from(index) / i128::from(buckets)) as i64; +pub fn histogram( + rows: &[RunCount], + window: crate::api::TraceQueryWindow, + buckets: u32, +) -> TraceHistogram { + let start_ms = window.start_ms; + let end_ms = window.end_ms; + let start = i128::from(start_ms); + let span = i128::from(end_ms) - start; + let edge = |index: u32| (start + span * i128::from(index) / i128::from(buckets)) as i64; TraceHistogram { + window, buckets: (0..buckets) .map(|index| { let hits = rows.iter().filter(|row| row.bucket == index); diff --git a/litellm-rust/crates/traces/src/store.rs b/litellm-rust/crates/traces/src/store.rs index b26eb57edb8..4539687b1a7 100644 --- a/litellm-rust/crates/traces/src/store.rs +++ b/litellm-rust/crates/traces/src/store.rs @@ -11,6 +11,7 @@ pub enum RunSelection { Matching(RunFilter), /// Every run with this trace id, whenever it happened. TraceId(String), + TraceRef(String), } /// The last row of a page in its order: the row's sort value and its reference. @@ -328,6 +329,7 @@ pub struct SpanText { /// is listed, oldest first by `(team_id, start_ms, request_id)`. #[derive(Clone, Debug, PartialEq)] pub struct CallQuery { + pub as_of_ms: u64, pub window: Range, /// Also matches the upstream id a managed `resp_` id wraps. pub response_ids: Vec, diff --git a/litellm-rust/crates/traces/src/view.rs b/litellm-rust/crates/traces/src/view.rs index a864c740526..e1ef8438984 100644 --- a/litellm-rust/crates/traces/src/view.rs +++ b/litellm-rust/crates/traces/src/view.rs @@ -2,7 +2,7 @@ use std::collections::BTreeMap; -use crate::ui::UiContent; +use crate::api::TraceQueryWindow; #[macro_rules_attribute::apply(wire_type)] #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -55,21 +55,20 @@ pub struct AgentNode { #[macro_rules_attribute::apply(response_type)] #[derive(Clone, Debug, PartialEq)] pub struct TraceSummary { - #[cfg_attr(feature = "schema", schemars(extend("x-python-optional" = true)))] pub resolution_limited: bool, pub trace_id: String, - #[cfg_attr(feature = "schema", schemars(extend("x-python-optional" = true)))] + #[serde(rename = "id")] pub trace_ref: String, pub name: String, pub service: String, - #[cfg_attr(feature = "schema", schemars(extend("x-python-optional" = true)))] pub agent_names: Vec, - #[cfg_attr(feature = "schema", schemars(extend("x-python-optional" = true)))] pub frameworks: Vec, pub input_preview: String, pub start_time: String, pub duration_ms: f64, + #[serde(rename = "root_status")] pub status: SpanStatus, + pub has_error: bool, pub span_count: u64, pub agent_count: u64, pub agent_invocations: u64, @@ -95,6 +94,7 @@ pub struct Trace { #[macro_rules_attribute::apply(response_type)] #[derive(Debug, PartialEq)] pub struct TracePage { + pub window: TraceQueryWindow, pub data: Vec, pub next_cursor: Option, } @@ -103,8 +103,6 @@ pub struct TracePage { #[derive(Debug, PartialEq)] pub struct SpanDetail { pub span_id: String, - pub input_ui: UiContent, - pub output_ui: UiContent, pub input: String, pub output: String, pub attributes: BTreeMap, diff --git a/litellm-rust/crates/traces/tests/api.rs b/litellm-rust/crates/traces/tests/api.rs new file mode 100644 index 00000000000..cbab96dee9c --- /dev/null +++ b/litellm-rust/crates/traces/tests/api.rs @@ -0,0 +1,109 @@ +#![cfg(feature = "schema")] + +use litellm_traces::{ + api::{ + TraceHistogramRequest, TraceListRequest, TraceNoQueryRequest, TraceQueryRequest, + TraceSQLResponse, TraceSpanPageRequest, TraceValuesRequest, + }, + schema, +}; +use rstest::rstest; +use serde_json::{Value, json}; + +#[rstest] +#[case::default(json!({}), true)] +#[case::one(json!({"page_size":1}), true)] +#[case::maximum(json!({"page_size":500}), true)] +#[case::zero(json!({"page_size":0}), false)] +#[case::over_maximum(json!({"page_size":501}), false)] +#[case::wrong_sort(json!({"sort_by":"spend"}), false)] +#[case::unknown_field(json!({"limit":10}), false)] +#[case::long_query(json!({"q":"x".repeat(1001)}), false)] +#[case::long_cursor(json!({"cursor":"x".repeat(513)}), false)] +fn list_contract_validates_the_same_request_in_rust_and_json_schema( + #[case] input: Value, + #[case] valid: bool, +) { + let schemas = schema::api_schemas(); + assert_eq!( + serde_json::from_value::(input.clone()).is_ok(), + valid + ); + assert_eq!( + jsonschema::is_valid(schemas["TraceListRequest"].as_value(), &input), + valid + ); +} + +#[rstest] +#[case::histogram_zero("TraceHistogramRequest",json!({"buckets":0}),false)] +#[case::histogram_max("TraceHistogramRequest",json!({"buckets":240}),true)] +#[case::histogram_over_max("TraceHistogramRequest",json!({"buckets":241}),false)] +#[case::values_zero("TraceValuesRequest",json!({"limit":0}),false)] +#[case::values_max("TraceValuesRequest",json!({"limit":100}),true)] +#[case::values_over_max("TraceValuesRequest",json!({"limit":101}),false)] +fn aggregation_contract_validates_limits( + #[case] name: &str, + #[case] input: Value, + #[case] valid: bool, +) { + let decoded = match name { + "TraceHistogramRequest" => { + serde_json::from_value::(input.clone()).is_ok() + } + "TraceValuesRequest" => serde_json::from_value::(input.clone()).is_ok(), + _ => unreachable!(), + }; + let schemas = schema::api_schemas(); + assert_eq!(decoded, valid); + assert_eq!( + jsonschema::is_valid(schemas[name].as_value(), &input), + valid + ); +} + +#[rstest] +fn sql_contract_preserves_native_bind_values_and_engine_response_fields() { + let request = json!({"sql":"SELECT {value:String}","params":{"value":"a'b", "flag":true, "ids":["one","two"],"missing":null,"large":u64::MAX}}); + let parsed: TraceQueryRequest = serde_json::from_value(request.clone()).unwrap(); + assert_eq!(serde_json::to_value(parsed).unwrap(), request); + let response = json!({"meta":[{"name":"x","type":"UInt64","source":"traces"}],"data":[{"x":u64::MAX.to_string(),"nested":{"flags":[true,null]}}],"rows":1,"statistics":{"elapsed":0.01,"rows_read":"1","bytes_read":20,"extra_stat":42},"extra_field":[1,2]}); + let parsed: TraceSQLResponse = serde_json::from_value(response.clone()).unwrap(); + assert_eq!(serde_json::to_value(parsed).unwrap(), response); + assert!(jsonschema::is_valid( + schema::api_schemas()["TraceSQLResponse"].as_value(), + &response + )); +} + +#[rstest] +#[case::spans_default("TraceSpanPageRequest",json!({}),true)] +#[case::spans_zero("TraceSpanPageRequest",json!({"page_size":0}),false)] +#[case::spans_maximum("TraceSpanPageRequest",json!({"page_size":500}),true)] +#[case::spans_over_maximum("TraceSpanPageRequest",json!({"page_size":501}),false)] +#[case::no_query("TraceNoQueryRequest",json!({}),true)] +#[case::old_trace_ref("TraceNoQueryRequest",json!({"trace_ref":"other"}),false)] +#[case::sql_empty("TraceQueryRequest",json!({"sql":""}),false)] +#[case::sql_unknown_field("TraceQueryRequest",json!({"sql":"SELECT 1","cursor":"x"}),false)] +fn collection_and_sql_contracts_validate_requests( + #[case] name: &str, + #[case] input: Value, + #[case] valid: bool, +) { + let decoded = match name { + "TraceSpanPageRequest" => { + serde_json::from_value::(input.clone()).is_ok() + } + "TraceNoQueryRequest" => { + serde_json::from_value::(input.clone()).is_ok() + } + "TraceQueryRequest" => serde_json::from_value::(input.clone()).is_ok(), + _ => unreachable!(), + }; + let schemas = schema::api_schemas(); + assert_eq!(decoded, valid); + assert_eq!( + jsonschema::is_valid(schemas[name].as_value(), &input), + valid + ); +} diff --git a/litellm-rust/crates/traces/tests/resolve.rs b/litellm-rust/crates/traces/tests/resolve.rs index 6e001856c4e..ddea1605d60 100644 --- a/litellm-rust/crates/traces/tests/resolve.rs +++ b/litellm-rust/crates/traces/tests/resolve.rs @@ -1328,3 +1328,27 @@ fn gateway_lookup_respects_legacy_fallback_and_ownership( expected ); } + +#[rstest] +#[case::physical_roots(true)] +#[case::missing_physical_root(false)] +fn summary_root_is_deterministic_and_distinguishes_child_errors(#[case] physical_roots: bool) { + let parent = if physical_roots { "" } else { "missing" }; + let rows = [ + SpanRow { + status: SpanStatus::Error, + ..row("z-root", parent, "later-root", "agent", "later") + }, + row("a-root", parent, "canonical-root", "agent", "canonical"), + SpanRow { + status: SpanStatus::Error, + ..at(row("orphan", "missing", "orphan", "llm", ""), -1, 1) + }, + ]; + let expected = if physical_roots { &rows[1] } else { &rows[2] }; + let trace = resolve_trace("trace", "ref", &rows, &[]).unwrap(); + assert_eq!(trace.summary.name, expected.name); + assert_eq!(trace.summary.input_preview, expected.input_preview); + assert_eq!(trace.summary.status, expected.status); + assert!(trace.summary.has_error); +} diff --git a/litellm-rust/crates/traces/tests/search.rs b/litellm-rust/crates/traces/tests/search.rs index 6644021d042..4a6f9448f3d 100644 --- a/litellm-rust/crates/traces/tests/search.rs +++ b/litellm-rust/crates/traces/tests/search.rs @@ -16,30 +16,24 @@ fn filter(field: RunField, pattern: &str, exclude: bool) -> FieldFilter { #[case::empty("", &[], vec![])] #[case::words("foo bar", &["foo", "bar"], vec![])] #[case::quoted_phrase(r#""foo bar""#, &["foo bar"], vec![])] -#[case::unclosed_quote(r#""foo bar"#, &["foo bar"], vec![])] -#[case::inner_quote_kept(r#"a"b"c"#, &[r#"a"b"c"#], vec![])] #[case::like_metacharacters_stay_literal("50%_off\\", &["50%_off\\"], vec![])] #[case::exact(r#"name:"plan trip""#, &[], vec![filter(RunField::Name, "plan trip", false)])] #[case::glob("agent:res*er", &[], vec![filter(RunField::Agent, "res*er", false)])] #[case::glob_keeps_the_rest("model:gpt_4*", &[], vec![filter(RunField::Model, "gpt_4*", false)])] -#[case::negated("-status:error", &[], vec![filter(RunField::Status, "error", true)])] +#[case::negated("-has_error:true", &[], vec![filter(RunField::HasError, "true", true)])] #[case::key_ignores_case("Trace_ID:abc", &[], vec![filter(RunField::TraceId, "abc", false)])] #[case::value_keeps_colons("input:a:b", &[], vec![filter(RunField::Input, "a:b", false)])] -#[case::missing_value_narrows_nothing("status: foo", &["foo"], vec![])] -#[case::unknown_key_is_text("color:red", &["color:red"], vec![])] -#[case::non_word_key_is_text("k1:v", &["k1:v"], vec![])] #[case::negated_text_stays_text("-foo", &["-foo"], vec![])] #[case::service("service:billing", &[], vec![filter(RunField::Service, "billing", false)])] #[case::team("-team:acme", &[], vec![filter(RunField::Team, "acme", true)])] #[case::attribute("attr.gen_ai.system:openai", &[], vec![FieldFilter { key: SearchKey::Attribute("gen_ai.system".into()), pattern: "openai".into(), exclude: false }])] -#[case::attribute_without_a_key_is_text("attr.:x", &["attr.:x"], vec![])] fn parse_matches_the_dashboard_search_grammar( #[case] q: &str, #[case] text: &[&str], #[case] filters: Vec, ) { assert_eq!( - RunSearch::parse(q), + RunSearch::parse(q).unwrap(), RunSearch { text: text.iter().map(|term| (*term).to_owned()).collect(), filters, @@ -66,8 +60,11 @@ fn histogram_fills_every_bucket_and_splits_failures_from_agents() { row(0, true, "c", 1), row(2, true, "a", 1), ], - 100, - 110, + litellm_traces::api::TraceQueryWindow { + start_ms: 100, + end_ms: 110, + as_of_ms: 120, + }, 3, ); let summary: Vec<_> = shaped @@ -94,3 +91,60 @@ fn histogram_fills_every_bucket_and_splits_failures_from_agents() { ); assert!(shaped.buckets[2].agents.is_empty()); } + +#[rstest] +#[case::unknown("color:red")] +#[case::deprecated_status("status:error")] +#[case::missing_value("root_status:")] +#[case::invalid_status("root_status:success")] +#[case::invalid_boolean("has_error:yes")] +#[case::unclosed_quote("name:\"hello")] +#[case::embedded_quote("a\"b\"c")] +#[case::empty_attribute("attr.:x")] +fn invalid_search_is_rejected(#[case] q: &str) { + assert!(RunSearch::parse(q).is_err()); +} + +#[rstest] +fn oversized_search_is_rejected() { + assert!(RunSearch::parse(&"x".repeat(1001)).is_err()); +} + +#[rstest] +#[case::full_range(i64::MIN, i64::MAX)] +#[case::minimum_to_zero(i64::MIN, 0)] +#[case::negative_to_maximum(-1, i64::MAX)] +fn histogram_extreme_windows_preserve_edges_and_totals(#[case] start_ms: i64, #[case] end_ms: i64) { + let window = litellm_traces::api::TraceQueryWindow { + start_ms, + end_ms, + as_of_ms: 1, + }; + let result = histogram(&[row(0, false, "a", 2), row(1, true, "b", 3)], window, 2); + assert_eq!(result.window, window); + assert_eq!(result.buckets[0].start_ms, start_ms); + assert_eq!(result.buckets[1].end_ms, end_ms); + assert_eq!(result.buckets[0].end_ms, result.buckets[1].start_ms); + assert!( + result + .buckets + .iter() + .all(|bucket| bucket.start_ms < bucket.end_ms) + ); + assert_eq!( + result + .buckets + .iter() + .map(|bucket| bucket.total) + .sum::(), + 5 + ); + assert_eq!( + result + .buckets + .iter() + .map(|bucket| bucket.failed) + .sum::(), + 3 + ); +} diff --git a/litellm/proxy/lens/sources.py b/litellm/proxy/lens/sources.py index 11afe38cee7..6d877045d66 100644 --- a/litellm/proxy/lens/sources.py +++ b/litellm/proxy/lens/sources.py @@ -91,7 +91,7 @@ def lens_access(scope: Scope) -> QueryScope: def execution_of(run: TraceSummary) -> Execution: - trace_ref: Final = run.get("trace_ref", "") + trace_ref: Final = run["id"] return Execution( id=execution_id(trace_ref, run["trace_id"]), trace_id=run["trace_id"], trace_ref=trace_ref, summary=run ) @@ -198,7 +198,7 @@ class SourceReader: span_ids, part, access, - max_chars=max_chars if not tail else budget - budget // 3, + max_chars=max_chars if not tail or max_chars else budget - budget // 3, tail=tail, ), ) @@ -315,7 +315,37 @@ class SourceReader: ) for part, _, _ in PARTS ] - return any(text["contains"] for text in chain.from_iterable(found)) + if any(text["contains"] for text in chain.from_iterable(found)): + return True + if len(evidence.quote) > BUDGET: + return False + trace: Final = await self.storage.get_trace(execution.trace_id, access, execution.trace_ref) + if trace is None: + return False + span: Final = next((span for span in trace["spans"] if span["span_id"] == evidence.span_id), None) + if span is None: + return False + heads: Final = await self._texts(access, execution, (evidence.span_id,), BUDGET) + pieces: Final = _pieces(span, heads) + tails: Final = await self._texts( + access, execution, (evidence.span_id,) if _total(pieces) > BUDGET else (), BUDGET, tail=True + ) + if evidence.quote in _excerpt(span, pieces, tails): + return True + if _total(pieces) <= BUDGET: + return False + prefixes: Final = tuple(label + (text["text"] if text else "") for _, label, text in pieces) + endings: Final = tuple( + (label if text is None or text["total_chars"] <= BUDGET else "") + + (tail["text"] if (tail := tails.get((evidence.span_id, part))) else "") + for part, label, text in pieces + ) + boundaries: Final = tuple(end + prefix for end, prefix in zip(endings, prefixes[1:])) + middle: Final = pieces[1][2] + joined: Final = ( + endings[0] + prefixes[1] + prefixes[2] if middle is None or middle["total_chars"] <= BUDGET else "" + ) + return evidence.quote in joined or any(evidence.quote in boundary for boundary in boundaries) def _excerpt(span: Span, pieces: tuple[tuple[SpanPart, str, SpanText | None], ...], tails: Texts) -> str: @@ -332,4 +362,4 @@ def _shortened(text: SpanText | None, budget: int, tail: SpanText | None) -> str return "" if text["total_chars"] <= budget: return text["text"] - return text["text"][: budget // 3] + OMITTED + (tail["text"] if tail else "") + return text["text"][: budget // 3] + OMITTED + (tail["text"][-(budget - budget // 3) :] if tail else "") diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index f40e541dd86..eda176913eb 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -1857,7 +1857,7 @@ def get_openapi_schema(): if server_root_path: openapi_schema["servers"] = [{"url": "/" + server_root_path.strip("/")}] - app.openapi_schema = openapi_schema + app.openapi_schema = dict(tracing_endpoints.merge_trace_openapi(openapi_schema)) return app.openapi_schema diff --git a/litellm/proxy/tracing_endpoints.py b/litellm/proxy/tracing_endpoints.py index 2dc768f4d2f..a09873765bb 100644 --- a/litellm/proxy/tracing_endpoints.py +++ b/litellm/proxy/tracing_endpoints.py @@ -1,29 +1,24 @@ -""" -Agent tracing endpoints. Thin wrappers over `TraceReceiver`: auth -> tenant/scope -> one call. +"""Authenticated OTLP ingestion and trace query routes.""" -POST /v1/traces OTLP/HTTP trace export (protobuf or JSON) -GET /v1/traces TracePage -GET /v1/traces/histogram TraceHistogram -GET /v1/traces/values/{field} RunValues -GET /v1/traces/{trace_id} Trace -GET /v1/traces/{trace_id}/spans/{span_id} SpanDetail -""" - -import time -from collections.abc import Mapping +from collections.abc import Callable, Coroutine, Mapping from dataclasses import dataclass from functools import partial from http.client import responses +from pathlib import Path from types import MappingProxyType -from typing import Annotated, Final, Literal +from typing import Annotated, Final from fastapi import APIRouter, Depends, HTTPException, Query, Request, Response -from pydantic import BaseModel, ConfigDict -from typing_extensions import assert_never +from fastapi.exceptions import RequestValidationError +from fastapi.routing import APIRoute +from pydantic import JsonValue, TypeAdapter, ValidationError +from starlette.exceptions import HTTPException as StarletteHTTPException +from starlette.responses import JSONResponse +from typing_extensions import ReadOnly, TypedDict, assert_never from litellm._logging import verbose_proxy_logger from litellm.constants import OTLP_RETRY_AFTER_SECONDS, TRACE_READ_RETRY_AFTER_SECONDS -from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth +from litellm.proxy._types import LitellmUserRoles, ProxyException, UserAPIKeyAuth from litellm.proxy.auth.authorization import AllRows, ReadScope, resolve_trace_read_scope from litellm.proxy.auth.authorization_dependencies import LogTeamLookupDependency from litellm.proxy.auth.user_api_key_auth import user_api_key_auth @@ -31,6 +26,21 @@ from litellm.proxy.common_utils.http_parsing_utils import is_otlp_trace_request from litellm.proxy.tracing_runtime import provide_receiver, require_receiver from litellm.rust_bridge.trace.errors import TraceChanged, TraceQueryError from litellm.rust_bridge.trace.generated.models import TraceQueryHelp +from litellm.rust_bridge.trace.generated.requests import ( + TraceErrorPageRequest, + TraceHistogramRequest, + TraceInvalidParam, + TraceListRequest, + TraceMetadata, + TraceNoQueryRequest, + TraceProblem, + TraceProblemCode, + TraceQueryRequest, + TraceSpanPageRequest, + TraceSpansPage, + TraceValuesRequest, +) +from litellm.rust_bridge.trace.generated.routes import OPERATIONS from litellm.rust_bridge.trace.generated.types import ( AllQueryScope, OwnedQueryScope, @@ -40,7 +50,6 @@ from litellm.rust_bridge.trace.generated.types import ( RunValues, SpanDetail, SpanErrorPage, - Trace, TraceHistogram, TracePage, ) @@ -49,9 +58,139 @@ from litellm.rust_bridge.trace.storage import ClickHouseStorage, Tenant from litellm.tracing import TraceReceiver, TracingPayloadTooLargeError from litellm.tracing.otlp_http import InvalidOTLPPayloadError, encode_otlp_response -router = APIRouter(tags=["agent tracing"]) +_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) -MS_PER_DAY: Final = 24 * 60 * 60 * 1000 + +class ValidationIssue(TypedDict): + loc: ReadOnly[tuple[str | int, ...]] + msg: ReadOnly[str] + + +_VALIDATION_ISSUES: Final = TypeAdapter(tuple[ValidationIssue, ...]) + + +def problem( + status: int, + code: TraceProblemCode, + detail: str, + database_code: int | None = None, + errors: tuple[TraceInvalidParam, ...] = (), +) -> TraceProblem: + return TraceProblem( + type="about:blank", + title=responses.get(status, "Trace request failed"), + status=status, + code=code, + detail=detail, + database_code=database_code, + errors=errors, + ) + + +def problem_exception( + status: int, code: TraceProblemCode, detail: str, database_code: int | None = None +) -> HTTPException: + return HTTPException(status_code=status, detail=problem(status, code, detail, database_code).model_dump()) + + +def status_code(status: int) -> TraceProblemCode: + match status: + case 401: + return "unauthorized" + case 403: + return "forbidden" + case 404: + return "not_found" + case 413: + return "too_large" + case 500: + return "internal_error" + case 501 | 503: + return "unavailable" + case _: + return "invalid_request" + + +def problem_response(body: TraceProblem, headers: Mapping[str, str] | None = None) -> Response: + return JSONResponse( + body.model_dump(mode="json", exclude_none=True), + status_code=body.status, + media_type="application/problem+json", + headers=headers, + ) + + +def http_problem(error: StarletteHTTPException) -> Response: + try: + body: Final = TraceProblem.model_validate(error.detail) + return problem_response(body, error.headers) + except ValidationError: + return problem_response( + problem(error.status_code, status_code(error.status_code), str(error.detail)), error.headers + ) + + +class TraceRoute(APIRoute): + def get_route_handler(self) -> Callable[[Request], Coroutine[None, None, Response]]: + handler: Final = super().get_route_handler() + + async def handle(request: Request) -> Response: + if is_otlp_trace_request(request): + return await handler(request) + try: + return await handler(request) + except StarletteHTTPException as error: + return http_problem(error) + except ProxyException as error: + status: Final = int(error.code) if error.code.isdecimal() else 500 + return problem_response(problem(status, status_code(status), error.message), error.headers) + except RequestValidationError as error: + issues: Final = _VALIDATION_ISSUES.validate_python(error.errors()) + details: Final = tuple( + TraceInvalidParam( + location="/".join(str(part) for part in issue["loc"]), + reason=str(issue["msg"]), + ) + for issue in issues + ) + return problem_response(problem(422, "invalid_request", "Request validation failed", errors=details)) + except Exception as error: + verbose_proxy_logger.exception("Trace request failed: %s", error) + return problem_response(problem(500, "internal_error", "Trace request failed")) + + return handle + + +router = APIRouter(tags=["agent tracing"], route_class=TraceRoute) + + +def merge_trace_openapi(schema: object) -> Mapping[str, JsonValue]: + document: Final = _JSON_OBJECT.validate_python(schema) + generated: Final = _JSON_OBJECT.validate_json( + (Path(__file__).parents[1] / "rust_bridge" / "trace" / "generated" / "openapi.json").read_text() + ) + components: Final = _JSON_OBJECT.validate_python(document.get("components", {})) + generated_components: Final = _JSON_OBJECT.validate_python(generated.get("components", {})) + return { + **document, + "openapi": generated["openapi"], + "paths": { + **_JSON_OBJECT.validate_python(document.get("paths", {})), + **_JSON_OBJECT.validate_python(generated["paths"]), + }, + "components": { + **components, + **generated_components, + "schemas": { + **_JSON_OBJECT.validate_python(components.get("schemas", {})), + **_JSON_OBJECT.validate_python(generated_components.get("schemas", {})), + }, + "securitySchemes": { + **_JSON_OBJECT.validate_python(components.get("securitySchemes", {})), + **_JSON_OBJECT.validate_python(generated_components.get("securitySchemes", {})), + }, + }, + } @dataclass(frozen=True, slots=True) @@ -63,7 +202,7 @@ class TraceAccessContext: def reader(self) -> tuple[TraceReceiver, QueryScope]: tracing: Final = require_receiver(self.receiver) if self.read_scope is None: - raise HTTPException(status_code=403, detail="Not allowed to view agent traces") + raise problem_exception(403, "forbidden", "Not allowed to view agent traces") return tracing, read_access(self.read_scope) def writer(self) -> tuple[TraceReceiver, Tenant]: @@ -96,17 +235,22 @@ def otlp_error_response( return Response(content=body, status_code=status_code, media_type=media_type, headers=headers) -def _otlp_error(content_type: str | None, status_code: int, message: str, retry: bool = False) -> Response: +def _otlp_error(content_type: str | None, status: int, message: str, retry: bool = False) -> Response: body, media_type = encode_otlp_response(content_type, message) return Response( content=body, - status_code=status_code, + status_code=status, media_type=media_type, headers=MappingProxyType({"Retry-After": str(OTLP_RETRY_AFTER_SECONDS)}) if retry else None, ) -@router.post("/v1/traces", include_in_schema=False) +@router.api_route( + OPERATIONS["trace_ingest"].path, + methods=[OPERATIONS["trace_ingest"].method], + operation_id=OPERATIONS["trace_ingest"].operation_id, + include_in_schema=False, +) async def ingest_otlp_traces( request: Request, context: Annotated[TraceAccessContext, Depends(provide_trace_access)], @@ -120,8 +264,8 @@ async def ingest_otlp_traces( content_encoding=request.headers.get("content-encoding"), tenant=tenant, ) - except TracingPayloadTooLargeError as e: - return _otlp_error(content_type, 413, str(e)) + except TracingPayloadTooLargeError as error: + return _otlp_error(content_type, 413, str(error)) except InvalidOTLPPayloadError as error: return _otlp_error(content_type, 400, str(error)) except RuntimeError: @@ -132,42 +276,20 @@ async def ingest_otlp_traces( return Response(content=body, media_type=media_type) -class TraceReadFailure(BaseModel): - """The body of every failed trace read. Clients branch on `code`, never on `message`.""" - - model_config = ConfigDict(frozen=True) - - code: Literal["invalid_request", "trace_changed", "too_large", "unavailable"] - message: str - - def read_failure(error: TraceChanged | ValueError | OverflowError | RuntimeError) -> HTTPException: - """One status per failure kind, so a client can tell a bad cursor (400, fix the request) from a - traversal it must restart (409), a result it cannot page through (413), and an outage it should - retry after `Retry-After` (503).""" match error: case TraceChanged(): - return HTTPException( - status_code=409, - detail=TraceReadFailure(code="trace_changed", message=str(error)).model_dump(), - ) + return problem_exception(409, "trace_changed", str(error)) case ValueError(): - return HTTPException( - status_code=400, detail=TraceReadFailure(code="invalid_request", message=str(error)).model_dump() - ) + return problem_exception(400, "invalid_request", str(error)) case OverflowError(): - return HTTPException( - status_code=413, - detail=TraceReadFailure( - code="too_large", message="Trace is too large for this view. Use a filtered trace query." - ).model_dump(), - ) + return problem_exception(413, "too_large", "Trace is too large for this view. Use a filtered trace query.") case RuntimeError(): verbose_proxy_logger.warning("Trace read unavailable: %s", error) return HTTPException( status_code=503, - detail=TraceReadFailure( - code="unavailable", message="Traces are temporarily unavailable. Please try again." + detail=problem( + 503, "unavailable", "Traces are temporarily unavailable. Please try again." ).model_dump(), headers={"Retry-After": str(TRACE_READ_RETRY_AFTER_SECONDS)}, ) @@ -175,86 +297,72 @@ def read_failure(error: TraceChanged | ValueError | OverflowError | RuntimeError return assert_never(error) -StartMs = Annotated[int | None, Query(description="Window start, unix ms. Default: 24h ago")] -EndMs = Annotated[int | None, Query(description="Window end, unix ms. Default: now")] -RunQuery = Annotated[ - str, - Query( - max_length=1000, - description='Free text and key:value filters, e.g. `agent:research* -status:ok "book a flight"`. ' - "Keys: name, agent, status, model, input, trace_id, service, team and attr.. " - "`*` globs and a leading `-` negates", - ), -] - - -@dataclass(frozen=True, slots=True) -class TraceWindow: - start_ms: int - end_ms: int - - -def trace_window(start_ms: StartMs = None, end_ms: EndMs = None) -> TraceWindow: - now_ms: Final = int(time.time() * 1000) - return TraceWindow( - start_ms=start_ms if start_ms is not None else now_ms - MS_PER_DAY, - end_ms=end_ms if end_ms is not None else now_ms, - ) - - -@router.get("/v1/traces", response_model=TracePage) +@router.api_route( + OPERATIONS["trace_list"].path, + methods=[OPERATIONS["trace_list"].method], + operation_id=OPERATIONS["trace_list"].operation_id, + response_model=TracePage, +) async def list_agent_traces( + params: Annotated[TraceListRequest, Query()], context: Annotated[TraceAccessContext, Depends(provide_trace_access)], - start_ms: StartMs = None, - end_ms: EndMs = None, - q: RunQuery = "", - cursor: Annotated[str | None, Query(max_length=512)] = None, - sort_by: Literal["start_ms", "duration_ms", "span_count", "error_count"] = "start_ms", - sort_dir: Literal["asc", "desc"] = "desc", ) -> TracePage: - order: Final = RunOrder(key=sort_by, descending=sort_dir == "desc") + order: Final = RunOrder(key=params.sort_by, descending=params.sort_dir == "desc") try: tracing, scope = context.reader() - return await tracing.list_traces(scope=scope, start_ms=start_ms, end_ms=end_ms, q=q, cursor=cursor, order=order) + return await tracing.list_traces( + scope=scope, + start_ms=params.start_ms, + end_ms=params.end_ms, + q=params.q, + cursor=params.cursor, + order=order, + page_size=params.page_size, + as_of_ms=params.as_of_ms, + ) except (TraceChanged, ValueError, OverflowError, RuntimeError) as error: raise read_failure(error) from error -@router.get("/v1/traces/histogram", response_model=TraceHistogram) +@router.api_route( + OPERATIONS["trace_histogram"].path, + methods=[OPERATIONS["trace_histogram"].method], + operation_id=OPERATIONS["trace_histogram"].operation_id, + response_model=TraceHistogram, +) async def agent_trace_histogram( + params: Annotated[TraceHistogramRequest, Query()], context: Annotated[TraceAccessContext, Depends(provide_trace_access)], - window: Annotated[TraceWindow, Depends(trace_window)], - q: RunQuery = "", - buckets: Annotated[int, Query(ge=1, le=240)] = 60, ) -> TraceHistogram: try: tracing, scope = context.reader() - return await tracing.trace_histogram(scope, window.start_ms, window.end_ms, q, buckets) + return await tracing.trace_histogram( + scope, params.start_ms, params.end_ms, params.q, params.buckets, params.as_of_ms + ) except (ValueError, OverflowError, RuntimeError) as error: raise read_failure(error) from error -@router.get("/v1/traces/values/{field}", response_model=RunValues) +@router.api_route( + OPERATIONS["trace_values"].path, + methods=[OPERATIONS["trace_values"].method], + operation_id=OPERATIONS["trace_values"].operation_id, + response_model=RunValues, +) async def agent_trace_values( field: RunField, + params: Annotated[TraceValuesRequest, Query()], context: Annotated[TraceAccessContext, Depends(provide_trace_access)], - window: Annotated[TraceWindow, Depends(trace_window)], - q: RunQuery = "", - contains: Annotated[str, Query(max_length=200)] = "", - limit: Annotated[int, Query(ge=1, le=100)] = 20, ) -> RunValues: try: tracing, scope = context.reader() - return await tracing.run_values(scope, window.start_ms, window.end_ms, q, field, contains, limit) + return await tracing.run_values( + scope, params.start_ms, params.end_ms, params.q, field, params.contains, params.limit, params.as_of_ms + ) except (ValueError, OverflowError, RuntimeError) as error: raise read_failure(error) from error -class TraceQueryRequest(BaseModel): - model_config = ConfigDict(frozen=True, extra="forbid") - sql: str - - @dataclass(frozen=True, slots=True) class TraceQueryAccess: storage: ClickHouseStorage @@ -266,18 +374,14 @@ def provide_trace_query_secret() -> str: from litellm.proxy.proxy_server import master_key if not master_key: - raise HTTPException(status_code=503, detail="Trace SQL queries require a configured proxy master key") + raise problem_exception(503, "query_unavailable", "Trace SQL queries require a configured proxy master key") return master_key def read_access(scope: ReadScope) -> QueryScope: if isinstance(scope, AllRows): return AllQueryScope(kind="all") - return OwnedQueryScope( - kind="owned", - user_id=scope.user_id or "", - team_ids=scope.team_ids, - ) + return OwnedQueryScope(kind="owned", user_id=scope.user_id or "", team_ids=scope.team_ids) async def provide_trace_query_access( @@ -289,11 +393,11 @@ async def provide_trace_query_access( storage: Final = require_receiver(tracing).storage scope: Final = await resolve_trace_read_scope(auth, partial(log_team_lookup, auth)) if scope is None: - raise HTTPException(status_code=403, detail="Not allowed to view logs") + raise problem_exception(403, "forbidden", "Not allowed to view logs") return TraceQueryAccess(storage, scope, secret) -def sql_failure_status(error: TraceQueryError) -> tuple[int, str]: +def sql_failure_status(error: TraceQueryError) -> tuple[int, TraceProblemCode]: match error.kind: case "rejected": return 400, "query_rejected" @@ -303,95 +407,129 @@ def sql_failure_status(error: TraceQueryError) -> tuple[int, str]: return 503, "query_unavailable" -@router.post("/v1/traces/query", response_model=TraceSQLResponse, response_model_exclude_unset=True) +@router.api_route( + OPERATIONS["trace_query"].path, + methods=[OPERATIONS["trace_query"].method], + operation_id=OPERATIONS["trace_query"].operation_id, + response_model=TraceSQLResponse, + response_model_exclude_unset=True, +) async def query_agent_traces( + _params: Annotated[TraceNoQueryRequest, Query()], body: TraceQueryRequest, access: Annotated[TraceQueryAccess, Depends(provide_trace_query_access)], ) -> TraceSQLResponse: try: - return await access.storage.query_sql(body.sql, read_access(access.scope), access.secret) + return await access.storage.query_sql(body.sql, read_access(access.scope), access.secret, body.params) except TraceQueryError as error: status, code = sql_failure_status(error) - raise HTTPException( - status_code=status, - detail={"code": code, "database_code": error.database_code, "message": error.message}, - ) from error + raise problem_exception(status, code, error.message, error.database_code) from error except ValueError as error: - raise HTTPException( - status_code=400, - detail={"code": "query_rejected", "database_code": None, "message": str(error)}, - ) from error + raise problem_exception(400, "query_rejected", str(error)) from error except RuntimeError as error: verbose_proxy_logger.warning("Trace SQL query unavailable: %s", error) - raise HTTPException( - status_code=503, - detail={ - "code": "query_unavailable", - "database_code": None, - "message": "Trace SQL is temporarily unavailable", - }, - ) from error + raise problem_exception(503, "query_unavailable", "Trace SQL is temporarily unavailable") from error -@router.get("/v1/traces/query/help", response_model=TraceQueryHelp, response_model_exclude_unset=True) +@router.api_route( + OPERATIONS["trace_query_help"].path, + methods=[OPERATIONS["trace_query_help"].method], + operation_id=OPERATIONS["trace_query_help"].operation_id, + response_model=TraceQueryHelp, + response_model_exclude_unset=True, +) async def help_agent_trace_queries( + _params: Annotated[TraceNoQueryRequest, Query()], access: Annotated[TraceQueryAccess, Depends(provide_trace_query_access)], ) -> TraceQueryHelp: try: return await access.storage.query_help(read_access(access.scope), access.secret) except RuntimeError as error: verbose_proxy_logger.warning("Trace query help unavailable: %s", error) - raise HTTPException(status_code=503, detail="Trace query help is temporarily unavailable") from error + raise problem_exception(503, "query_unavailable", "Trace query help is temporarily unavailable") from error -@router.get("/v1/traces/{trace_id}", response_model=Trace) +@router.api_route( + OPERATIONS["trace_get"].path, + methods=[OPERATIONS["trace_get"].method], + operation_id=OPERATIONS["trace_get"].operation_id, + response_model=TraceMetadata, +) async def get_agent_trace( - trace_id: str, + _params: Annotated[TraceNoQueryRequest, Query()], + id: str, context: Annotated[TraceAccessContext, Depends(provide_trace_access)], - trace_ref: Annotated[str, Query()] = "", - cursor: Annotated[str | None, Query(max_length=512)] = None, - page_size: Annotated[int | None, Query(ge=1, le=500)] = None, -) -> Trace: - tracing, scope = context.reader() +) -> TraceMetadata: try: - trace: Final = await tracing.get_trace(trace_id, scope, trace_ref, cursor, page_size) + tracing, scope = context.reader() + trace: Final = await tracing.get_trace_metadata(scope, id) except (TraceChanged, ValueError, OverflowError, RuntimeError) as error: raise read_failure(error) from error if trace is None: - raise HTTPException(status_code=404, detail=f"Trace {trace_id} not found") + raise problem_exception(404, "not_found", "Trace not found") return trace -@router.get("/v1/traces/{trace_id}/spans/{span_id}", response_model=SpanDetail) -async def get_agent_trace_span( - trace_id: str, - span_id: str, +@router.api_route( + OPERATIONS["trace_spans"].path, + methods=[OPERATIONS["trace_spans"].method], + operation_id=OPERATIONS["trace_spans"].operation_id, + response_model=TraceSpansPage, +) +async def get_agent_trace_spans( + id: str, + params: Annotated[TraceSpanPageRequest, Query()], context: Annotated[TraceAccessContext, Depends(provide_trace_access)], - trace_ref: Annotated[str, Query()] = "", -) -> SpanDetail: - tracing, scope = context.reader() - try: - span: Final = await tracing.get_span(trace_id, span_id, scope, trace_ref) - except (TraceChanged, ValueError, OverflowError, RuntimeError) as error: - raise read_failure(error) from error - if span is None: - raise HTTPException(status_code=404, detail=f"Span {span_id} not found") - return span - - -@router.get("/v1/traces/{trace_id}/spans/{span_id}/error", response_model=SpanErrorPage) -async def get_agent_trace_span_error( - trace_id: str, - span_id: str, - context: Annotated[TraceAccessContext, Depends(provide_trace_access)], - trace_ref: Annotated[str, Query()] = "", - cursor: Annotated[str | None, Query(max_length=512)] = None, -) -> SpanErrorPage: +) -> TraceSpansPage: try: tracing, scope = context.reader() - page: Final = await tracing.get_span_error(trace_id, span_id, scope, trace_ref, cursor) + page: Final = await tracing.get_trace_spans(scope, id, params.cursor, params.page_size) except (TraceChanged, ValueError, OverflowError, RuntimeError) as error: raise read_failure(error) from error if page is None: - raise HTTPException(status_code=404, detail="Span diagnostic not found or no longer available") + raise problem_exception(404, "not_found", "Trace not found") + return page + + +@router.api_route( + OPERATIONS["trace_span"].path, + methods=[OPERATIONS["trace_span"].method], + operation_id=OPERATIONS["trace_span"].operation_id, + response_model=SpanDetail, +) +async def get_agent_trace_span( + _params: Annotated[TraceNoQueryRequest, Query()], + id: str, + span_id: str, + context: Annotated[TraceAccessContext, Depends(provide_trace_access)], +) -> SpanDetail: + try: + tracing, scope = context.reader() + span: Final = await tracing.get_span_by_id(scope, id, span_id) + except (TraceChanged, ValueError, OverflowError, RuntimeError) as error: + raise read_failure(error) from error + if span is None: + raise problem_exception(404, "not_found", "Span not found") + return span + + +@router.api_route( + OPERATIONS["trace_error"].path, + methods=[OPERATIONS["trace_error"].method], + operation_id=OPERATIONS["trace_error"].operation_id, + response_model=SpanErrorPage, +) +async def get_agent_trace_span_error( + id: str, + span_id: str, + params: Annotated[TraceErrorPageRequest, Query()], + context: Annotated[TraceAccessContext, Depends(provide_trace_access)], +) -> SpanErrorPage: + try: + tracing, scope = context.reader() + page: Final = await tracing.get_span_error_by_id(scope, id, span_id, params.cursor) + except (TraceChanged, ValueError, OverflowError, RuntimeError) as error: + raise read_failure(error) from error + if page is None: + raise problem_exception(404, "not_found", "Span diagnostic not found or no longer available") return page diff --git a/litellm/rust_bridge/_native.pyi b/litellm/rust_bridge/_native.pyi index e74f9fcb25d..50e016652d2 100644 --- a/litellm/rust_bridge/_native.pyi +++ b/litellm/rust_bridge/_native.pyi @@ -11,6 +11,7 @@ from litellm.rust_bridge.embeddings.entrypoints import LiteLLMEmbeddingRequest from litellm.rust_bridge.messages.entrypoints import LiteLLMMessagesRequest from litellm.rust_bridge.ocr.entrypoints import LiteLLMOcrRequest from litellm.rust_bridge.responses.entrypoints import LiteLLMResponsesRequest +from litellm.rust_bridge.trace.generated.requests import SqlParameter from litellm.rust_bridge.trace.generated.types import QueryScope, RunOrder from litellm.types.llms.anthropic_messages.anthropic_response import AnthropicMessagesResponse from litellm.types.llms.openai import ResponsesAPIResponse @@ -52,9 +53,16 @@ class NativeTraceStorage: limit: int, order: RunOrder, trace_refs: Sequence[str] = (), + as_of_ms: int | None = None, ) -> Future[JsonValue]: ... def count_traces( - self, scope: QueryScope, start_ms: int, end_ms: int, q: str, trace_refs: Sequence[str] = () + self, + scope: QueryScope, + start_ms: int | None, + end_ms: int | None, + q: str, + trace_refs: Sequence[str] = (), + as_of_ms: int | None = None, ) -> Future[JsonValue]: ... def span_text( self, @@ -69,10 +77,24 @@ class NativeTraceStorage: contains: str | None = None, ) -> Future[JsonValue]: ... def trace_histogram( - self, scope: QueryScope, start_ms: int, end_ms: int, q: str, buckets: int + self, + scope: QueryScope, + start_ms: int | None, + end_ms: int | None, + q: str, + buckets: int, + as_of_ms: int | None = None, ) -> Future[JsonValue]: ... def run_values( - self, scope: QueryScope, start_ms: int, end_ms: int, q: str, field: str, contains: str, limit: int + self, + scope: QueryScope, + start_ms: int | None, + end_ms: int | None, + q: str, + field: str, + contains: str, + limit: int, + as_of_ms: int | None = None, ) -> Future[JsonValue]: ... def get_trace( self, trace_id: str, scope: QueryScope, trace_ref: str, cursor: str | None = None, page_size: int | None = None @@ -81,7 +103,15 @@ class NativeTraceStorage: def get_span_error( self, trace_id: str, span_id: str, scope: QueryScope, trace_ref: str, cursor: str | None ) -> Future[JsonValue]: ... - def query_sql(self, sql: str, scope: QueryScope, secret: str) -> Future[str]: ... + def get_trace_metadata(self, scope: QueryScope, id: str) -> Future[JsonValue]: ... + def get_trace_spans(self, scope: QueryScope, id: str, cursor: str | None, page_size: int) -> Future[JsonValue]: ... + def get_span_by_id(self, scope: QueryScope, id: str, span_id: str) -> Future[JsonValue]: ... + def get_span_error_by_id( + self, scope: QueryScope, id: str, span_id: str, cursor: str | None = None + ) -> Future[JsonValue]: ... + def query_sql( + self, sql: str, scope: QueryScope, secret: str, params: Mapping[str, SqlParameter] + ) -> Future[str]: ... def query_help(self, scope: QueryScope, secret: str) -> Future[JsonValue]: ... @final diff --git a/litellm/rust_bridge/trace/generated/models.py b/litellm/rust_bridge/trace/generated/models.py index 65aa0882962..e09d77e89ac 100644 --- a/litellm/rust_bridge/trace/generated/models.py +++ b/litellm/rust_bridge/trace/generated/models.py @@ -6,10 +6,10 @@ from typing import Annotated, Literal, TypeAlias from pydantic import BaseModel, ConfigDict, Field -TraceTableName: TypeAlias = Literal["otel_traces", "agent_traces_by_key", "spend_logs"] +TraceQueryTableName: TypeAlias = Literal["traces", "spans", "calls", "otel_traces", "agent_traces_by_key", "spend_logs"] -class TraceQueryColumn(BaseModel): +class TraceSQLColumn(BaseModel): model_config = ConfigDict( extra="allow", frozen=True, @@ -25,7 +25,7 @@ class TraceQueryNormalizedField(BaseModel): frozen=True, ) - table: TraceTableName + table: TraceQueryTableName name: str column: str type: str @@ -71,8 +71,8 @@ class TraceQueryTable(BaseModel): frozen=True, ) - name: TraceTableName - columns: tuple[TraceQueryColumn, ...] + name: TraceQueryTableName + columns: tuple[TraceSQLColumn, ...] class TraceQueryMetadataField(BaseModel): @@ -103,7 +103,7 @@ class TraceQueryMetadata(BaseModel): frozen=True, ) - table: TraceTableName + table: TraceQueryTableName column: str fields: tuple[TraceQueryMetadataField, ...] sampled_rows: int = Field(..., ge=0, le=18446744073709551615) @@ -120,7 +120,7 @@ class TraceQueryAttributes(BaseModel): frozen=True, ) - table: TraceTableName + table: TraceQueryTableName column: str fields: tuple[TraceQueryAttributeField, ...] truncated: bool diff --git a/litellm/rust_bridge/trace/generated/openapi.json b/litellm/rust_bridge/trace/generated/openapi.json new file mode 100644 index 00000000000..1a9b59ffd85 --- /dev/null +++ b/litellm/rust_bridge/trace/generated/openapi.json @@ -0,0 +1,3024 @@ +{ + "components": { + "schemas": { + "AgentNode": { + "description": "One distinct agent in a trace: 200 invocations of `researcher` are one node.", + "properties": { + "duration_ms": { + "format": "double", + "type": "number" + }, + "invocations": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "llm_calls": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "name": { + "type": "string" + }, + "parent_agent": { + "type": [ + "string", + "null" + ] + }, + "spend": { + "format": "double", + "type": [ + "number", + "null" + ] + }, + "tool_calls": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + } + }, + "required": [ + "name", + "parent_agent", + "invocations", + "llm_calls", + "tool_calls", + "duration_ms", + "spend" + ], + "type": "object" + }, + "AgentRuns": { + "properties": { + "agent": { + "type": "string" + }, + "runs": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + } + }, + "required": [ + "agent", + "runs" + ], + "type": "object" + }, + "HistogramBucket": { + "properties": { + "agents": { + "description": "Runs that did not fail, by their alphabetically first agent label, or service when unlabelled.", + "items": { + "$ref": "#/components/schemas/AgentRuns" + }, + "type": "array" + }, + "end_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + }, + "failed": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "start_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + }, + "total": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + } + }, + "required": [ + "start_ms", + "end_ms", + "total", + "failed", + "agents" + ], + "type": "object" + }, + "MapValueType": { + "enum": [ + "String" + ], + "type": "string" + }, + "MetadataValueType": { + "enum": [ + "array", + "boolean", + "integer", + "null", + "number", + "object", + "string" + ], + "type": "string" + }, + "PathPart": { + "anyOf": [ + { + "type": "string" + }, + { + "format": "uint", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + } + ] + }, + "RunField": { + "enum": [ + "name", + "agent", + "root_status", + "has_error", + "model", + "input", + "trace_id", + "service", + "team" + ], + "title": "RunField", + "type": "string" + }, + "RunValues": { + "description": "Distinct values of one run field, most common first.", + "properties": { + "values": { + "items": { + "type": "string" + }, + "type": "array" + }, + "window": { + "$ref": "#/components/schemas/TraceQueryWindow" + } + }, + "required": [ + "window", + "values" + ], + "title": "RunValues", + "type": "object" + }, + "Span": { + "properties": { + "agent": { + "type": "string" + }, + "duration_ms": { + "format": "double", + "type": "number" + }, + "error": { + "type": [ + "string", + "null" + ] + }, + "error_truncated": { + "type": "boolean" + }, + "framework": { + "type": "string" + }, + "input_preview": { + "type": "string" + }, + "input_tokens": { + "format": "uint32", + "maximum": 4294967295, + "minimum": 0, + "type": "integer" + }, + "litellm_request_id": { + "type": [ + "string", + "null" + ] + }, + "model": { + "type": [ + "string", + "null" + ] + }, + "name": { + "type": "string" + }, + "output_tokens": { + "format": "uint32", + "maximum": 4294967295, + "minimum": 0, + "type": "integer" + }, + "parent_span_id": { + "type": [ + "string", + "null" + ] + }, + "span_id": { + "type": "string" + }, + "spend": { + "format": "double", + "type": [ + "number", + "null" + ] + }, + "start_offset_ms": { + "format": "double", + "type": "number" + }, + "status": { + "$ref": "#/components/schemas/SpanStatus" + }, + "type": { + "$ref": "#/components/schemas/SpanType" + } + }, + "required": [ + "span_id", + "parent_span_id", + "name", + "type", + "agent", + "framework", + "start_offset_ms", + "duration_ms", + "status", + "error", + "error_truncated", + "input_preview", + "model", + "input_tokens", + "output_tokens", + "litellm_request_id", + "spend" + ], + "type": "object" + }, + "SpanDetail": { + "properties": { + "attributes": { + "additionalProperties": { + "type": "string" + }, + "type": "object" + }, + "input": { + "type": "string" + }, + "output": { + "type": "string" + }, + "span_id": { + "type": "string" + } + }, + "required": [ + "span_id", + "input", + "output", + "attributes" + ], + "title": "SpanDetail", + "type": "object" + }, + "SpanErrorPage": { + "properties": { + "message": { + "type": "string" + }, + "next_cursor": { + "type": [ + "string", + "null" + ] + }, + "span_id": { + "type": "string" + }, + "total_chars": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + } + }, + "required": [ + "span_id", + "message", + "total_chars", + "next_cursor" + ], + "title": "SpanErrorPage", + "type": "object" + }, + "SpanStatus": { + "enum": [ + "ok", + "error", + "unset" + ], + "type": "string" + }, + "SpanType": { + "enum": [ + "agent", + "llm", + "tool", + "chain", + "framework", + "retriever", + "embedding", + "reranker", + "guardrail", + "evaluator", + "prompt", + "decision" + ], + "type": "string" + }, + "SqlParameter": { + "anyOf": [ + { + "type": "string" + }, + { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + }, + { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + { + "format": "double", + "type": "number" + }, + { + "type": "boolean" + }, + { + "type": "null" + }, + { + "items": { + "type": "string" + }, + "type": "array" + } + ] + }, + "TraceErrorPageRequest": { + "additionalProperties": false, + "properties": { + "cursor": { + "maxLength": 512, + "type": [ + "string", + "null" + ] + } + }, + "title": "TraceErrorPageRequest", + "type": "object" + }, + "TraceHistogram": { + "description": "Matching runs per equal-width slice of the window.", + "properties": { + "buckets": { + "items": { + "$ref": "#/components/schemas/HistogramBucket" + }, + "type": "array" + }, + "window": { + "$ref": "#/components/schemas/TraceQueryWindow" + } + }, + "required": [ + "window", + "buckets" + ], + "title": "TraceHistogram", + "type": "object" + }, + "TraceHistogramRequest": { + "additionalProperties": false, + "properties": { + "as_of_ms": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": [ + "integer", + "null" + ] + }, + "buckets": { + "default": 60, + "format": "uint16", + "maximum": 240, + "minimum": 1, + "type": "integer" + }, + "end_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + }, + "q": { + "default": "", + "maxLength": 1000, + "type": "string" + }, + "start_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + "title": "TraceHistogramRequest", + "type": "object" + }, + "TraceInvalidParam": { + "additionalProperties": false, + "properties": { + "location": { + "type": "string" + }, + "reason": { + "type": "string" + } + }, + "required": [ + "location", + "reason" + ], + "type": "object" + }, + "TraceListRequest": { + "additionalProperties": false, + "properties": { + "as_of_ms": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": [ + "integer", + "null" + ] + }, + "cursor": { + "maxLength": 512, + "type": [ + "string", + "null" + ] + }, + "end_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + }, + "page_size": { + "default": 50, + "format": "uint16", + "maximum": 500, + "minimum": 1, + "type": "integer" + }, + "q": { + "default": "", + "maxLength": 1000, + "type": "string" + }, + "sort_by": { + "$ref": "#/components/schemas/TraceSortField" + }, + "sort_dir": { + "$ref": "#/components/schemas/TraceSortDirection" + }, + "start_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + "title": "TraceListRequest", + "type": "object" + }, + "TraceMetadata": { + "properties": { + "agents": { + "items": { + "$ref": "#/components/schemas/AgentNode" + }, + "type": "array" + }, + "summary": { + "$ref": "#/components/schemas/TraceSummary" + } + }, + "required": [ + "summary", + "agents" + ], + "title": "TraceMetadata", + "type": "object" + }, + "TraceNoQueryRequest": { + "additionalProperties": false, + "title": "TraceNoQueryRequest", + "type": "object" + }, + "TracePage": { + "properties": { + "data": { + "items": { + "$ref": "#/components/schemas/TraceSummary" + }, + "type": "array" + }, + "next_cursor": { + "type": [ + "string", + "null" + ] + }, + "window": { + "$ref": "#/components/schemas/TraceQueryWindow" + } + }, + "required": [ + "window", + "data", + "next_cursor" + ], + "title": "TracePage", + "type": "object" + }, + "TraceProblem": { + "additionalProperties": false, + "properties": { + "code": { + "$ref": "#/components/schemas/TraceProblemCode" + }, + "database_code": { + "format": "uint32", + "maximum": 4294967295, + "minimum": 0, + "type": [ + "integer", + "null" + ] + }, + "detail": { + "type": "string" + }, + "errors": { + "default": [], + "items": { + "$ref": "#/components/schemas/TraceInvalidParam" + }, + "type": "array" + }, + "status": { + "format": "uint16", + "maximum": 65535, + "minimum": 0, + "type": "integer" + }, + "title": { + "type": "string" + }, + "type": { + "type": "string" + } + }, + "required": [ + "type", + "title", + "status", + "detail", + "code" + ], + "title": "TraceProblem", + "type": "object" + }, + "TraceProblemCode": { + "enum": [ + "invalid_request", + "unauthorized", + "forbidden", + "not_found", + "trace_changed", + "too_large", + "unavailable", + "query_rejected", + "query_limit_exceeded", + "query_unavailable", + "internal_error" + ], + "type": "string" + }, + "TraceQueryAttributeField": { + "additionalProperties": false, + "properties": { + "expression": { + "type": "string" + }, + "key": { + "type": "string" + }, + "type": { + "$ref": "#/components/schemas/MapValueType" + } + }, + "required": [ + "key", + "type", + "expression" + ], + "type": "object" + }, + "TraceQueryAttributes": { + "additionalProperties": false, + "properties": { + "column": { + "type": "string" + }, + "discovery_sql": { + "type": "string" + }, + "error": { + "type": [ + "string", + "null" + ] + }, + "fields": { + "items": { + "$ref": "#/components/schemas/TraceQueryAttributeField" + }, + "type": "array" + }, + "scope": { + "type": "string" + }, + "table": { + "$ref": "#/components/schemas/TraceQueryTableName" + }, + "truncated": { + "type": "boolean" + } + }, + "required": [ + "table", + "column", + "fields", + "truncated", + "discovery_sql", + "scope" + ], + "type": "object" + }, + "TraceQueryExample": { + "properties": { + "name": { + "type": "string" + }, + "sql": { + "type": "string" + } + }, + "required": [ + "name", + "sql" + ], + "type": "object" + }, + "TraceQueryHelp": { + "additionalProperties": false, + "properties": { + "access": { + "type": "string" + }, + "attributes": { + "items": { + "$ref": "#/components/schemas/TraceQueryAttributes" + }, + "type": "array" + }, + "dialect": { + "type": "string" + }, + "examples": { + "items": { + "$ref": "#/components/schemas/TraceQueryExample" + }, + "type": "array" + }, + "gotchas": { + "items": { + "type": "string" + }, + "type": "array" + }, + "guide": { + "type": "string" + }, + "metadata": { + "$ref": "#/components/schemas/TraceQueryMetadata" + }, + "normalized_fields": { + "items": { + "$ref": "#/components/schemas/TraceQueryNormalizedField" + }, + "type": "array" + }, + "relationships": { + "items": { + "$ref": "#/components/schemas/TraceQueryRelationship" + }, + "type": "array" + }, + "response": { + "type": "string" + }, + "tables": { + "items": { + "$ref": "#/components/schemas/TraceQueryTable" + }, + "type": "array" + } + }, + "required": [ + "dialect", + "access", + "response", + "tables", + "normalized_fields", + "metadata", + "attributes", + "relationships", + "examples", + "gotchas", + "guide" + ], + "title": "TraceQueryHelp", + "type": "object" + }, + "TraceQueryMetadata": { + "additionalProperties": false, + "properties": { + "column": { + "type": "string" + }, + "error": { + "type": [ + "string", + "null" + ] + }, + "fields": { + "items": { + "$ref": "#/components/schemas/TraceQueryMetadataField" + }, + "type": "array" + }, + "invalid_json_rows": { + "format": "uint", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "sample_sql": { + "type": "string" + }, + "sampled_rows": { + "format": "uint", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "scope": { + "type": "string" + }, + "table": { + "$ref": "#/components/schemas/TraceQueryTableName" + }, + "truncated": { + "type": "boolean" + } + }, + "required": [ + "table", + "column", + "fields", + "sampled_rows", + "invalid_json_rows", + "truncated", + "sample_sql", + "scope" + ], + "type": "object" + }, + "TraceQueryMetadataField": { + "additionalProperties": false, + "properties": { + "expression": { + "type": "string" + }, + "path": { + "items": { + "$ref": "#/components/schemas/PathPart" + }, + "type": "array" + }, + "types": { + "items": { + "$ref": "#/components/schemas/MetadataValueType" + }, + "type": "array", + "uniqueItems": true + } + }, + "required": [ + "path", + "types", + "expression" + ], + "type": "object" + }, + "TraceQueryNormalizedField": { + "additionalProperties": false, + "properties": { + "column": { + "type": "string" + }, + "meaning": { + "type": "string" + }, + "name": { + "type": "string" + }, + "table": { + "$ref": "#/components/schemas/TraceQueryTableName" + }, + "type": { + "type": "string" + } + }, + "required": [ + "table", + "name", + "column", + "type", + "meaning" + ], + "type": "object" + }, + "TraceQueryRelationship": { + "additionalProperties": false, + "properties": { + "additional_predicates": { + "type": "string" + }, + "left": { + "type": "string" + }, + "meaning": { + "type": "string" + }, + "right": { + "type": "string" + } + }, + "required": [ + "left", + "right", + "additional_predicates", + "meaning" + ], + "type": "object" + }, + "TraceQueryRequest": { + "additionalProperties": false, + "properties": { + "params": { + "additionalProperties": { + "$ref": "#/components/schemas/SqlParameter" + }, + "default": {}, + "type": "object" + }, + "sql": { + "minLength": 1, + "type": "string" + } + }, + "required": [ + "sql" + ], + "title": "TraceQueryRequest", + "type": "object" + }, + "TraceQueryStatistics": { + "additionalProperties": true, + "properties": { + "bytes_read": { + "$ref": "#/components/schemas/UnsignedCount" + }, + "elapsed": { + "format": "double", + "type": "number" + }, + "rows_read": { + "$ref": "#/components/schemas/UnsignedCount" + } + }, + "required": [ + "elapsed", + "rows_read", + "bytes_read" + ], + "type": "object" + }, + "TraceQueryTable": { + "additionalProperties": false, + "properties": { + "columns": { + "items": { + "$ref": "#/components/schemas/TraceSQLColumn" + }, + "type": "array" + }, + "name": { + "$ref": "#/components/schemas/TraceQueryTableName" + } + }, + "required": [ + "name", + "columns" + ], + "type": "object" + }, + "TraceQueryTableName": { + "enum": [ + "traces", + "spans", + "calls", + "otel_traces", + "agent_traces_by_key", + "spend_logs" + ], + "type": "string" + }, + "TraceQueryWindow": { + "additionalProperties": false, + "properties": { + "as_of_ms": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "end_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + }, + "start_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + } + }, + "required": [ + "start_ms", + "end_ms", + "as_of_ms" + ], + "type": "object" + }, + "TraceSQLColumn": { + "additionalProperties": true, + "properties": { + "name": { + "type": "string" + }, + "type": { + "type": "string" + } + }, + "required": [ + "name", + "type" + ], + "type": "object" + }, + "TraceSQLResponse": { + "additionalProperties": true, + "properties": { + "data": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": "array", + "x-python-normalized": { + "type": "tuple[Mapping[str, JsonValue], ...]" + } + }, + "meta": { + "items": { + "$ref": "#/components/schemas/TraceSQLColumn" + }, + "type": "array" + }, + "rows": { + "$ref": "#/components/schemas/UnsignedCount" + }, + "statistics": { + "$ref": "#/components/schemas/TraceQueryStatistics" + } + }, + "required": [ + "meta", + "data", + "rows", + "statistics" + ], + "title": "TraceSQLResponse", + "type": "object" + }, + "TraceSortDirection": { + "enum": [ + "asc", + "desc" + ], + "type": "string" + }, + "TraceSortField": { + "enum": [ + "start_ms", + "duration_ms", + "span_count", + "error_count" + ], + "type": "string" + }, + "TraceSpanPageRequest": { + "additionalProperties": false, + "properties": { + "cursor": { + "maxLength": 512, + "type": [ + "string", + "null" + ] + }, + "page_size": { + "default": 100, + "format": "uint16", + "maximum": 500, + "minimum": 1, + "type": "integer" + } + }, + "title": "TraceSpanPageRequest", + "type": "object" + }, + "TraceSpansPage": { + "properties": { + "data": { + "items": { + "$ref": "#/components/schemas/Span" + }, + "type": "array" + }, + "next_cursor": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "data", + "next_cursor" + ], + "title": "TraceSpansPage", + "type": "object" + }, + "TraceSummary": { + "properties": { + "agent_count": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "agent_invocations": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "agent_names": { + "items": { + "type": "string" + }, + "type": "array" + }, + "duration_ms": { + "format": "double", + "type": "number" + }, + "error_count": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "frameworks": { + "items": { + "type": "string" + }, + "type": "array" + }, + "has_error": { + "type": "boolean" + }, + "id": { + "type": "string" + }, + "input_preview": { + "type": "string" + }, + "input_tokens": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "llm_calls": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "models": { + "items": { + "type": "string" + }, + "type": "array" + }, + "name": { + "type": "string" + }, + "output_tokens": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "resolution_limited": { + "type": "boolean" + }, + "root_status": { + "$ref": "#/components/schemas/SpanStatus" + }, + "service": { + "type": "string" + }, + "span_count": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "spend": { + "format": "double", + "type": [ + "number", + "null" + ] + }, + "start_time": { + "type": "string" + }, + "tool_calls": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "trace_id": { + "type": "string" + } + }, + "required": [ + "resolution_limited", + "trace_id", + "id", + "name", + "service", + "agent_names", + "frameworks", + "input_preview", + "start_time", + "duration_ms", + "root_status", + "has_error", + "span_count", + "agent_count", + "agent_invocations", + "llm_calls", + "tool_calls", + "error_count", + "input_tokens", + "output_tokens", + "models", + "spend" + ], + "type": "object" + }, + "TraceValuesRequest": { + "additionalProperties": false, + "properties": { + "as_of_ms": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": [ + "integer", + "null" + ] + }, + "contains": { + "default": "", + "maxLength": 200, + "type": "string" + }, + "end_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + }, + "limit": { + "default": 20, + "format": "uint16", + "maximum": 100, + "minimum": 1, + "type": "integer" + }, + "q": { + "default": "", + "maxLength": 1000, + "type": "string" + }, + "start_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + "title": "TraceValuesRequest", + "type": "object" + }, + "UnsignedCount": { + "anyOf": [ + { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + { + "type": "string" + } + ] + } + }, + "securitySchemes": { + "TraceBearer": { + "scheme": "bearer", + "type": "http" + } + } + }, + "info": { + "title": "LiteLLM Trace API", + "version": "1" + }, + "openapi": "3.1.0", + "paths": { + "/v1/traces": { + "get": { + "description": "Search trace summaries. All clauses in q must match. Free text searches trace_id, name and input preview as case-insensitive substrings. key:value clauses match whole values, with * as a wildcard; double quotes group spaces or literal colons. Prefix keyed clauses with - to exclude matches. Unknown keys, missing values and malformed quoting return invalid_request. name describes the physical root; input searches its preview, falling back to the first nonempty agent or LLM preview; agent, model and attr. match any span. root_status describes the root, has_error means any failed span. The default window is the last 24 hours. First-page ingestion timestamp cutoff is retained by the cursor. This excludes later-stamped exports, not delayed commits stamped before the cutoff, and is not a database transaction snapshot. Repeat q and sorting on continuation; omitted bounds reuse the cursor window. Sort ties use the canonical id in the same direction. The server may return fewer than page_size items to respect response limits. Continue until next_cursor is null. Span-derived metrics use the cutoff; spend enrichment is best effort", + "operationId": "trace_list", + "parameters": [ + { + "in": "query", + "name": "as_of_ms", + "required": false, + "schema": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": [ + "integer", + "null" + ] + } + }, + { + "in": "query", + "name": "start_ms", + "required": false, + "schema": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + { + "in": "query", + "name": "end_ms", + "required": false, + "schema": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + { + "in": "query", + "name": "q", + "required": false, + "schema": { + "default": "", + "maxLength": 1000, + "type": "string" + } + }, + { + "in": "query", + "name": "cursor", + "required": false, + "schema": { + "maxLength": 512, + "type": [ + "string", + "null" + ] + } + }, + { + "in": "query", + "name": "page_size", + "required": false, + "schema": { + "default": 50, + "format": "uint16", + "maximum": 500, + "minimum": 1, + "type": "integer" + } + }, + { + "in": "query", + "name": "sort_by", + "required": false, + "schema": { + "$ref": "#/components/schemas/TraceSortField" + } + }, + { + "in": "query", + "name": "sort_dir", + "required": false, + "schema": { + "$ref": "#/components/schemas/TraceSortDirection" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/TracePage" + } + } + }, + "description": "Success" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "409": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "501": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + }, + "post": { + "description": "Export OTLP traces as JSON or protobuf, optionally gzip compressed. The OTLP protocol defines payloads and responses: https://opentelemetry.io/docs/specs/otlp/. Ownership is derived from authentication, never payload attributes", + "operationId": "trace_ingest", + "requestBody": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + }, + "application/x-protobuf": { + "schema": { + "format": "binary", + "type": "string" + } + } + }, + "required": true + }, + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + }, + "application/x-protobuf": { + "schema": { + "format": "binary", + "type": "string" + } + } + }, + "description": "OTLP protocol response" + }, + "400": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + }, + "application/x-protobuf": { + "schema": { + "format": "binary", + "type": "string" + } + } + }, + "description": "OTLP protocol response" + }, + "401": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + }, + "application/x-protobuf": { + "schema": { + "format": "binary", + "type": "string" + } + } + }, + "description": "OTLP protocol response" + }, + "403": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + }, + "application/x-protobuf": { + "schema": { + "format": "binary", + "type": "string" + } + } + }, + "description": "OTLP protocol response" + }, + "413": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + }, + "application/x-protobuf": { + "schema": { + "format": "binary", + "type": "string" + } + } + }, + "description": "OTLP protocol response" + }, + "429": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + }, + "application/x-protobuf": { + "schema": { + "format": "binary", + "type": "string" + } + } + }, + "description": "OTLP protocol response" + }, + "501": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + }, + "application/x-protobuf": { + "schema": { + "format": "binary", + "type": "string" + } + } + }, + "description": "OTLP protocol response" + }, + "503": { + "content": { + "application/json": { + "schema": { + "type": "object" + } + }, + "application/x-protobuf": { + "schema": { + "format": "binary", + "type": "string" + } + } + }, + "description": "OTLP protocol response" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + } + }, + "/v1/traces/histogram": { + "get": { + "description": "Count matching traces in equal-width [start_ms,end_ms) buckets. Reuse the list window and as_of_ms for matching span-derived membership. failed counts traces with any failed span. Agent groups count successful traces under their alphabetically first agent name, or service when no agent name exists", + "operationId": "trace_histogram", + "parameters": [ + { + "in": "query", + "name": "start_ms", + "required": false, + "schema": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + { + "in": "query", + "name": "end_ms", + "required": false, + "schema": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + { + "in": "query", + "name": "as_of_ms", + "required": false, + "schema": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": [ + "integer", + "null" + ] + } + }, + { + "in": "query", + "name": "q", + "required": false, + "schema": { + "default": "", + "maxLength": 1000, + "type": "string" + } + }, + { + "in": "query", + "name": "buckets", + "required": false, + "schema": { + "default": 60, + "format": "uint16", + "maximum": 240, + "minimum": 1, + "type": "integer" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/TraceHistogram" + } + } + }, + "description": "Success" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "409": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "501": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + } + }, + "/v1/traces/query": { + "post": { + "description": "Execute read-only ClickHouse SQL under authenticated row policies and fixed resource limits. Bind params with native {name:Type} placeholders. Results are always ClickHouse JSON; 64-bit integers may be strings. SQL callers control ORDER BY, LIMIT and keyset continuation. Exceeding a resource limit fails instead of returning partial success", + "operationId": "trace_query", + "parameters": [], + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/TraceQueryRequest" + } + } + }, + "required": true + }, + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/TraceSQLResponse" + } + } + }, + "description": "Success" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "409": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "501": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + } + }, + "/v1/traces/query/help": { + "get": { + "description": "Discover current SQL schema, logical views, scoped examples and resource limits", + "operationId": "trace_query_help", + "parameters": [], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/TraceQueryHelp" + } + } + }, + "description": "Success" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "409": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "501": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + } + }, + "/v1/traces/values/{field}": { + "get": { + "description": "Return the most common distinct values among matching traces. contains is case-insensitive. limit is a top-K suggestion limit, not a pagination size. Reuse list window and as_of_ms for matching membership", + "operationId": "trace_values", + "parameters": [ + { + "in": "path", + "name": "field", + "required": true, + "schema": { + "$ref": "#/components/schemas/RunField" + } + }, + { + "in": "query", + "name": "start_ms", + "required": false, + "schema": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + { + "in": "query", + "name": "end_ms", + "required": false, + "schema": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + { + "in": "query", + "name": "as_of_ms", + "required": false, + "schema": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": [ + "integer", + "null" + ] + } + }, + { + "in": "query", + "name": "q", + "required": false, + "schema": { + "default": "", + "maxLength": 1000, + "type": "string" + } + }, + { + "in": "query", + "name": "contains", + "required": false, + "schema": { + "default": "", + "maxLength": 200, + "type": "string" + } + }, + { + "in": "query", + "name": "limit", + "required": false, + "schema": { + "default": 20, + "format": "uint16", + "maximum": 100, + "minimum": 1, + "type": "integer" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/RunValues" + } + } + }, + "description": "Success" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "409": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "501": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + } + }, + "/v1/traces/{id}": { + "get": { + "description": "Read summary and agent metadata by canonical id. trace_id is the original OTLP id, which can repeat across ownership scopes. Spans are read through the separate spans collection", + "operationId": "trace_get", + "parameters": [ + { + "in": "path", + "name": "id", + "required": true, + "schema": { + "minLength": 1, + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/TraceMetadata" + } + } + }, + "description": "Success" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "409": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "501": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + } + }, + "/v1/traces/{id}/spans": { + "get": { + "description": "Read a bounded page of canonical spans. The cursor pins the graph version. Continue with the same page_size; a changed graph or expired reconstruction returns trace_changed and the traversal must restart", + "operationId": "trace_spans", + "parameters": [ + { + "in": "path", + "name": "id", + "required": true, + "schema": { + "minLength": 1, + "type": "string" + } + }, + { + "in": "query", + "name": "cursor", + "required": false, + "schema": { + "maxLength": 512, + "type": [ + "string", + "null" + ] + } + }, + { + "in": "query", + "name": "page_size", + "required": false, + "schema": { + "default": 100, + "format": "uint16", + "maximum": 500, + "minimum": 1, + "type": "integer" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/TraceSpansPage" + } + } + }, + "description": "Success" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "409": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "501": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + } + }, + "/v1/traces/{id}/spans/{span_id}": { + "get": { + "description": "Read raw input, output and attributes for one span. UI rendering is performed by the client", + "operationId": "trace_span", + "parameters": [ + { + "in": "path", + "name": "id", + "required": true, + "schema": { + "minLength": 1, + "type": "string" + } + }, + { + "in": "path", + "name": "span_id", + "required": true, + "schema": { + "minLength": 1, + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/SpanDetail" + } + } + }, + "description": "Success" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "409": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "501": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + } + }, + "/v1/traces/{id}/spans/{span_id}/error": { + "get": { + "description": "Read bounded diagnostic text pages. Cursor validation detects content changes and requires restarting the traversal", + "operationId": "trace_error", + "parameters": [ + { + "in": "path", + "name": "id", + "required": true, + "schema": { + "minLength": 1, + "type": "string" + } + }, + { + "in": "path", + "name": "span_id", + "required": true, + "schema": { + "minLength": 1, + "type": "string" + } + }, + { + "in": "query", + "name": "cursor", + "required": false, + "schema": { + "maxLength": 512, + "type": [ + "string", + "null" + ] + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/SpanErrorPage" + } + } + }, + "description": "Success" + }, + "400": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "401": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "403": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "404": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "409": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "413": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "422": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "500": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "501": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + }, + "503": { + "content": { + "application/problem+json": { + "schema": { + "$ref": "#/components/schemas/TraceProblem" + } + } + }, + "description": "Trace API problem" + } + }, + "security": [ + { + "TraceBearer": [] + } + ], + "tags": [ + "agent tracing" + ] + } + } + }, + "security": [ + { + "TraceBearer": [] + } + ] +} diff --git a/litellm/rust_bridge/trace/generated/requests.py b/litellm/rust_bridge/trace/generated/requests.py new file mode 100644 index 00000000000..9636170e565 --- /dev/null +++ b/litellm/rust_bridge/trace/generated/requests.py @@ -0,0 +1,301 @@ +# @generated by scripts/generate_trace_types.py, do not edit + +from __future__ import annotations + +from collections.abc import Mapping +from typing import Annotated, Literal, TypeAlias + +from pydantic import BaseModel, ConfigDict, Field, JsonValue + + +class TraceErrorPageRequest(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + cursor: str | None = Field(None, max_length=512) + + +class TraceHistogramRequest(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + start_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807) + end_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807) + as_of_ms: int | None = Field(None, ge=0, le=18446744073709551615) + q: str = Field("", max_length=1000) + buckets: int = Field(60, ge=1, le=240) + + +TraceSortField: TypeAlias = Literal["start_ms", "duration_ms", "span_count", "error_count"] + + +TraceSortDirection: TypeAlias = Literal["asc", "desc"] + + +class TraceListRequest(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + as_of_ms: int | None = Field(None, ge=0, le=18446744073709551615) + start_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807) + end_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807) + q: str = Field("", max_length=1000) + cursor: str | None = Field(None, max_length=512) + page_size: int = Field(50, ge=1, le=500) + sort_by: TraceSortField = "start_ms" + sort_dir: TraceSortDirection = "desc" + + +SpanStatus: TypeAlias = Literal["ok", "error", "unset"] + + +class AgentNode(BaseModel): + model_config = ConfigDict( + frozen=True, + ) + + name: str + parent_agent: str | None + invocations: int = Field(..., ge=0, le=18446744073709551615) + llm_calls: int = Field(..., ge=0, le=18446744073709551615) + tool_calls: int = Field(..., ge=0, le=18446744073709551615) + duration_ms: float + spend: float | None + + +class TraceNoQueryRequest(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + +TraceProblemCode: TypeAlias = Literal[ + "invalid_request", + "unauthorized", + "forbidden", + "not_found", + "trace_changed", + "too_large", + "unavailable", + "query_rejected", + "query_limit_exceeded", + "query_unavailable", + "internal_error", +] + + +class TraceInvalidParam(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + location: str + reason: str + + +class TraceProblem(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + type: str + title: str + status: int = Field(..., ge=0, le=65535) + detail: str + code: TraceProblemCode + database_code: int | None = Field(None, ge=0, le=4294967295) + errors: tuple[TraceInvalidParam, ...] = () + + +SqlParameter1: TypeAlias = Annotated[int, Field(..., ge=-9223372036854775808, le=9223372036854775807)] + + +SqlParameter2: TypeAlias = Annotated[int, Field(..., ge=0, le=18446744073709551615)] + + +SqlParameter: TypeAlias = str | SqlParameter1 | SqlParameter2 | float | bool | list[str] | None + + +class TraceQueryRequest(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + sql: str = Field(..., min_length=1) + params: Mapping[str, SqlParameter] = {} + + +class TraceSQLColumn(BaseModel): + model_config = ConfigDict( + extra="allow", + frozen=True, + ) + + name: str + type: str + + +UnsignedCount1: TypeAlias = Annotated[int, Field(..., ge=0, le=18446744073709551615)] + + +UnsignedCount: TypeAlias = UnsignedCount1 | str + + +class TraceQueryStatistics(BaseModel): + model_config = ConfigDict( + extra="allow", + frozen=True, + ) + + elapsed: float + rows_read: UnsignedCount + bytes_read: UnsignedCount + + +class TraceSQLResponse(BaseModel): + model_config = ConfigDict( + extra="allow", + frozen=True, + ) + + meta: tuple[TraceSQLColumn, ...] + data: tuple[Mapping[str, JsonValue], ...] + rows: UnsignedCount + statistics: TraceQueryStatistics + + +class TraceSpanPageRequest(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + cursor: str | None = Field(None, max_length=512) + page_size: int = Field(100, ge=1, le=500) + + +SpanType: TypeAlias = Literal[ + "agent", + "llm", + "tool", + "chain", + "framework", + "retriever", + "embedding", + "reranker", + "guardrail", + "evaluator", + "prompt", + "decision", +] + + +class TraceValuesRequest(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + start_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807) + end_ms: int | None = Field(None, ge=-9223372036854775808, le=9223372036854775807) + as_of_ms: int | None = Field(None, ge=0, le=18446744073709551615) + q: str = Field("", max_length=1000) + contains: str = Field("", max_length=200) + limit: int = Field(20, ge=1, le=100) + + +class TraceSummary(BaseModel): + model_config = ConfigDict( + frozen=True, + ) + + resolution_limited: bool + trace_id: str + id: str + name: str + service: str + agent_names: tuple[str, ...] + frameworks: tuple[str, ...] + input_preview: str + start_time: str + duration_ms: float + root_status: SpanStatus + has_error: bool + span_count: int = Field(..., ge=0, le=18446744073709551615) + agent_count: int = Field(..., ge=0, le=18446744073709551615) + agent_invocations: int = Field(..., ge=0, le=18446744073709551615) + llm_calls: int = Field(..., ge=0, le=18446744073709551615) + tool_calls: int = Field(..., ge=0, le=18446744073709551615) + error_count: int = Field(..., ge=0, le=18446744073709551615) + input_tokens: int = Field(..., ge=0, le=18446744073709551615) + output_tokens: int = Field(..., ge=0, le=18446744073709551615) + models: tuple[str, ...] + spend: float | None + + +class TraceMetadata(BaseModel): + model_config = ConfigDict( + frozen=True, + ) + + summary: TraceSummary + agents: tuple[AgentNode, ...] + + +class Span(BaseModel): + model_config = ConfigDict( + frozen=True, + ) + + span_id: str + parent_span_id: str | None + name: str + type: SpanType + agent: str + framework: str + start_offset_ms: float + duration_ms: float + status: SpanStatus + error: str | None + error_truncated: bool + input_preview: str + model: str | None + input_tokens: int = Field(..., ge=0, le=4294967295) + output_tokens: int = Field(..., ge=0, le=4294967295) + litellm_request_id: str | None + spend: float | None + + +class TraceSpansPage(BaseModel): + model_config = ConfigDict( + frozen=True, + ) + + data: tuple[Span, ...] + next_cursor: str | None + + +TraceWireRequests: TypeAlias = Annotated[ + TraceErrorPageRequest + | TraceHistogramRequest + | TraceListRequest + | TraceMetadata + | TraceNoQueryRequest + | TraceProblem + | TraceQueryRequest + | TraceSQLResponse + | TraceSpanPageRequest + | TraceSpansPage + | TraceValuesRequest, + Field(..., title="TraceWireRequests"), +] diff --git a/litellm/rust_bridge/trace/generated/routes.py b/litellm/rust_bridge/trace/generated/routes.py new file mode 100644 index 00000000000..916b33e674c --- /dev/null +++ b/litellm/rust_bridge/trace/generated/routes.py @@ -0,0 +1,28 @@ +# @generated by scripts/generate_trace_types.py, do not edit +from collections.abc import Mapping +from dataclasses import dataclass +from types import MappingProxyType +from typing import Final + + +@dataclass(frozen=True, slots=True) +class TraceOperation: + path: str + method: str + operation_id: str + + +OPERATIONS: Final[Mapping[str, TraceOperation]] = MappingProxyType( + { + "trace_list": TraceOperation("/v1/traces", "GET", "trace_list"), + "trace_ingest": TraceOperation("/v1/traces", "POST", "trace_ingest"), + "trace_histogram": TraceOperation("/v1/traces/histogram", "GET", "trace_histogram"), + "trace_query": TraceOperation("/v1/traces/query", "POST", "trace_query"), + "trace_query_help": TraceOperation("/v1/traces/query/help", "GET", "trace_query_help"), + "trace_values": TraceOperation("/v1/traces/values/{field}", "GET", "trace_values"), + "trace_get": TraceOperation("/v1/traces/{id}", "GET", "trace_get"), + "trace_spans": TraceOperation("/v1/traces/{id}/spans", "GET", "trace_spans"), + "trace_span": TraceOperation("/v1/traces/{id}/spans/{span_id}", "GET", "trace_span"), + "trace_error": TraceOperation("/v1/traces/{id}/spans/{span_id}/error", "GET", "trace_error"), + } +) diff --git a/litellm/rust_bridge/trace/generated/types.py b/litellm/rust_bridge/trace/generated/types.py index ff85d0e7fad..03c37cf30ea 100644 --- a/litellm/rust_bridge/trace/generated/types.py +++ b/litellm/rust_bridge/trace/generated/types.py @@ -23,7 +23,17 @@ class OwnedQueryScope(typing_extensions.TypedDict): QueryScope: TypeAlias = AllQueryScope | OwnedQueryScope -RunField: TypeAlias = Literal["name", "agent", "status", "model", "input", "trace_id", "service", "team"] +RunField: TypeAlias = Literal[ + "name", + "agent", + "root_status", + "has_error", + "model", + "input", + "trace_id", + "service", + "team", +] RunSortKey: TypeAlias = Literal["start_ms", "duration_ms", "span_count", "error_count", "trace_ref"] @@ -34,26 +44,22 @@ class RunOrder(typing_extensions.TypedDict): descending: ReadOnly[bool] +class TraceQueryWindow(typing_extensions.TypedDict): + start_ms: ReadOnly[Annotated[int, Field(ge=-9223372036854775808, le=9223372036854775807)]] + end_ms: ReadOnly[Annotated[int, Field(ge=-9223372036854775808, le=9223372036854775807)]] + as_of_ms: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]] + + class RunValues(typing_extensions.TypedDict): + window: ReadOnly[TraceQueryWindow] values: ReadOnly[tuple[str, ...]] -class UIText(typing_extensions.TypedDict): - text: ReadOnly[str] - kind: ReadOnly[Literal["text"]] - - -ChatRole: TypeAlias = Literal["system", "user", "assistant", "tool"] - - -class UIToolCall(typing_extensions.TypedDict): - name: ReadOnly[str] - arguments: ReadOnly[str] - - -class UIField(typing_extensions.TypedDict): - key: ReadOnly[str] - value: ReadOnly[str] +class SpanDetail(typing_extensions.TypedDict): + span_id: ReadOnly[str] + input: ReadOnly[str] + output: ReadOnly[str] + attributes: ReadOnly[Mapping[str, str]] class SpanErrorPage(typing_extensions.TypedDict): @@ -105,30 +111,19 @@ class AgentRuns(typing_extensions.TypedDict): runs: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]] -class UIFields(typing_extensions.TypedDict): - fields: ReadOnly[tuple[UIField, ...]] - kind: ReadOnly[Literal["fields"]] - - -class UIMessage(typing_extensions.TypedDict): - role: ReadOnly[ChatRole] - content: ReadOnly[str] - name: ReadOnly[NotRequired[str | None]] - tool_calls: ReadOnly[NotRequired[tuple[UIToolCall, ...]]] - - class TraceSummary(typing_extensions.TypedDict): - resolution_limited: ReadOnly[NotRequired[bool]] + resolution_limited: ReadOnly[bool] trace_id: ReadOnly[str] - trace_ref: ReadOnly[NotRequired[str]] + id: ReadOnly[str] name: ReadOnly[str] service: ReadOnly[str] - agent_names: ReadOnly[NotRequired[tuple[str, ...]]] - frameworks: ReadOnly[NotRequired[tuple[str, ...]]] + agent_names: ReadOnly[tuple[str, ...]] + frameworks: ReadOnly[tuple[str, ...]] input_preview: ReadOnly[str] start_time: ReadOnly[str] duration_ms: ReadOnly[float] - status: ReadOnly[SpanStatus] + root_status: ReadOnly[SpanStatus] + has_error: ReadOnly[bool] span_count: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]] agent_count: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]] agent_invocations: ReadOnly[Annotated[int, Field(ge=0, le=18446744073709551615)]] @@ -177,31 +172,16 @@ class HistogramBucket(typing_extensions.TypedDict): class TraceHistogram(typing_extensions.TypedDict): + window: ReadOnly[TraceQueryWindow] buckets: ReadOnly[tuple[HistogramBucket, ...]] class TracePage(typing_extensions.TypedDict): + window: ReadOnly[TraceQueryWindow] data: ReadOnly[tuple[TraceSummary, ...]] next_cursor: ReadOnly[str | None] -class UIMessages(typing_extensions.TypedDict): - messages: ReadOnly[tuple[UIMessage, ...]] - kind: ReadOnly[Literal["messages"]] - - -UIContent: TypeAlias = UIMessages | UIFields | UIText - - -class SpanDetail(typing_extensions.TypedDict): - span_id: ReadOnly[str] - input_ui: ReadOnly[UIContent] - output_ui: ReadOnly[UIContent] - input: ReadOnly[str] - output: ReadOnly[str] - attributes: ReadOnly[Mapping[str, str]] - - TraceWireTypes: TypeAlias = ( QueryScope | RunField diff --git a/litellm/rust_bridge/trace/queries.py b/litellm/rust_bridge/trace/queries.py index 3be7a5fa038..afcd7778986 100644 --- a/litellm/rust_bridge/trace/queries.py +++ b/litellm/rust_bridge/trace/queries.py @@ -1,23 +1,3 @@ -from collections.abc import Mapping -from typing import Final +from .generated.requests import TraceSQLResponse -from pydantic import BaseModel, ConfigDict, JsonValue - -from .generated.models import TraceQueryColumn - -_RESPONSE_CONFIG: Final = ConfigDict(frozen=True, extra="allow") - - -class TraceQueryStatistics(BaseModel): - model_config = _RESPONSE_CONFIG - elapsed: float - rows_read: int | str - bytes_read: int | str - - -class TraceSQLResponse(BaseModel): - model_config = _RESPONSE_CONFIG - meta: tuple[TraceQueryColumn, ...] - data: tuple[Mapping[str, JsonValue], ...] - rows: int | str - statistics: TraceQueryStatistics +__all__ = ("TraceSQLResponse",) diff --git a/litellm/rust_bridge/trace/storage.py b/litellm/rust_bridge/trace/storage.py index 8071d2399b4..648b578bb00 100644 --- a/litellm/rust_bridge/trace/storage.py +++ b/litellm/rust_bridge/trace/storage.py @@ -8,6 +8,7 @@ from litellm.constants import AGENT_TRACING_LIST_PAGE_SIZE, OTLP_MAX_ATTRIBUTE_V from litellm.rust_bridge.loader import get_native_bridge from .generated.models import TraceQueryHelp +from .generated.requests import SqlParameter, TraceMetadata, TraceSpansPage from .generated.types import ( QueryScope, RunOrder, @@ -59,6 +60,7 @@ class NativeStore(Protocol): limit: int, order: RunOrder, trace_refs: Sequence[str], + as_of_ms: int | None = None, ) -> Awaitable[JsonValue]: ... def count_traces( @@ -79,11 +81,25 @@ class NativeStore(Protocol): ) -> Awaitable[JsonValue]: ... def trace_histogram( - self, scope: QueryScope, start_ms: int, end_ms: int, q: str, buckets: int + self, + scope: QueryScope, + start_ms: int | None, + end_ms: int | None, + q: str, + buckets: int, + as_of_ms: int | None = None, ) -> Awaitable[JsonValue]: ... def run_values( - self, scope: QueryScope, start_ms: int, end_ms: int, q: str, field: str, contains: str, limit: int + self, + scope: QueryScope, + start_ms: int | None, + end_ms: int | None, + q: str, + field: str, + contains: str, + limit: int, + as_of_ms: int | None = None, ) -> Awaitable[JsonValue]: ... def get_trace( @@ -96,7 +112,21 @@ class NativeStore(Protocol): self, trace_id: str, span_id: str, scope: QueryScope, trace_ref: str, cursor: str | None ) -> Awaitable[JsonValue]: ... - def query_sql(self, sql: str, scope: QueryScope, secret: str) -> Awaitable[str]: ... + def get_trace_metadata(self, scope: QueryScope, id: str) -> Awaitable[JsonValue]: ... + + def get_trace_spans( + self, scope: QueryScope, id: str, cursor: str | None, page_size: int + ) -> Awaitable[JsonValue]: ... + + def get_span_by_id(self, scope: QueryScope, id: str, span_id: str) -> Awaitable[JsonValue]: ... + + def get_span_error_by_id( + self, scope: QueryScope, id: str, span_id: str, cursor: str | None + ) -> Awaitable[JsonValue]: ... + + def query_sql( + self, sql: str, scope: QueryScope, secret: str, params: Mapping[str, SqlParameter] + ) -> Awaitable[str]: ... def query_help(self, scope: QueryScope, secret: str) -> Awaitable[JsonValue]: ... @@ -120,6 +150,8 @@ _TRACE_HISTOGRAM: Final = TypeAdapter(TraceHistogram) _RUN_VALUES: Final = TypeAdapter(RunValues) _COUNT: Final = TypeAdapter(int) _SPAN_TEXTS: Final = TypeAdapter(tuple[SpanText, ...]) +_TRACE_METADATA: Final = TypeAdapter(TraceMetadata | None) +_TRACE_SPANS: Final = TypeAdapter(TraceSpansPage | None) _TRACE: Final[TypeAdapter[Trace | None]] = TypeAdapter(Trace | None) _SPAN_DETAIL: Final[TypeAdapter[SpanDetail | None]] = TypeAdapter(SpanDetail | None) _SPAN_ERROR_PAGE: Final[TypeAdapter[SpanErrorPage | None]] = TypeAdapter(SpanErrorPage | None) @@ -208,9 +240,10 @@ class ClickHouseStorage: limit: int = AGENT_TRACING_LIST_PAGE_SIZE, order: RunOrder = NEWEST, trace_refs: Sequence[str] = (), + as_of_ms: int | None = None, ) -> TracePage: result: Final = await self._native.list_traces( - scope, start_ms, end_ms, q, cursor, limit, order, tuple(trace_refs) + scope, start_ms, end_ms, q, cursor, limit, order, tuple(trace_refs), as_of_ms ) return _validate_query_response(_TRACE_PAGE, result) @@ -238,15 +271,29 @@ class ClickHouseStorage: return _validate_query_response(_SPAN_TEXTS, result) async def trace_histogram( - self, scope: QueryScope, start_ms: int, end_ms: int, q: str, buckets: int + self, + scope: QueryScope, + start_ms: int | None, + end_ms: int | None, + q: str, + buckets: int, + as_of_ms: int | None = None, ) -> TraceHistogram: - result: Final = await self._native.trace_histogram(scope, start_ms, end_ms, q, buckets) + result: Final = await self._native.trace_histogram(scope, start_ms, end_ms, q, buckets, as_of_ms) return _validate_query_response(_TRACE_HISTOGRAM, result) async def run_values( - self, scope: QueryScope, start_ms: int, end_ms: int, q: str, field: str, contains: str, limit: int + self, + scope: QueryScope, + start_ms: int | None, + end_ms: int | None, + q: str, + field: str, + contains: str, + limit: int, + as_of_ms: int | None = None, ) -> RunValues: - result: Final = await self._native.run_values(scope, start_ms, end_ms, q, field, contains, limit) + result: Final = await self._native.run_values(scope, start_ms, end_ms, q, field, contains, limit, as_of_ms) return _validate_query_response(_RUN_VALUES, result) async def get_trace( @@ -270,8 +317,30 @@ class ClickHouseStorage: result: Final = await self._native.get_span_error(trace_id, span_id, scope, trace_ref, cursor) return _validate_query_response(_SPAN_ERROR_PAGE, result) - async def query_sql(self, sql: str, scope: QueryScope, secret: str) -> TraceSQLResponse: - result: Final = await self._native.query_sql(sql, scope, secret) + async def get_trace_metadata(self, scope: QueryScope, id: str) -> TraceMetadata | None: + result: Final = await self._native.get_trace_metadata(scope, id) + return _validate_query_response(_TRACE_METADATA, result) + + async def get_trace_spans( + self, scope: QueryScope, id: str, cursor: str | None, page_size: int + ) -> TraceSpansPage | None: + result: Final = await self._native.get_trace_spans(scope, id, cursor, page_size) + return _validate_query_response(_TRACE_SPANS, result) + + async def get_span_by_id(self, scope: QueryScope, id: str, span_id: str) -> SpanDetail | None: + result: Final = await self._native.get_span_by_id(scope, id, span_id) + return _validate_query_response(_SPAN_DETAIL, result) + + async def get_span_error_by_id( + self, scope: QueryScope, id: str, span_id: str, cursor: str | None + ) -> SpanErrorPage | None: + result: Final = await self._native.get_span_error_by_id(scope, id, span_id, cursor) + return _validate_query_response(_SPAN_ERROR_PAGE, result) + + async def query_sql( + self, sql: str, scope: QueryScope, secret: str, params: Mapping[str, SqlParameter] + ) -> TraceSQLResponse: + result: Final = await self._native.query_sql(sql, scope, secret, params) return _decode_query_response(_SQL_RESPONSE, result) async def query_help(self, scope: QueryScope, secret: str) -> TraceQueryHelp: diff --git a/litellm/tracing/receiver.py b/litellm/tracing/receiver.py index 99c9e84670e..a247b34da38 100644 --- a/litellm/tracing/receiver.py +++ b/litellm/tracing/receiver.py @@ -19,6 +19,7 @@ from threading import BoundedSemaphore from typing import Final from litellm.constants import AGENT_TRACING_LIST_PAGE_SIZE, OTLP_MAX_BODY_BYTES, OTLP_MAX_CONCURRENT_INGESTS +from litellm.rust_bridge.trace.generated.requests import TraceMetadata, TraceSpansPage from litellm.rust_bridge.trace.generated.types import ( QueryScope, RunField, @@ -114,18 +115,34 @@ class TraceReceiver: q: str = "", cursor: str | None = None, order: RunOrder = NEWEST, + page_size: int = AGENT_TRACING_LIST_PAGE_SIZE, + as_of_ms: int | None = None, ) -> TracePage: - return await self.storage.list_traces(scope, start_ms, end_ms, q, cursor, AGENT_TRACING_LIST_PAGE_SIZE, order) + return await self.storage.list_traces(scope, start_ms, end_ms, q, cursor, page_size, order, as_of_ms=as_of_ms) async def trace_histogram( - self, scope: QueryScope, start_ms: int, end_ms: int, q: str, buckets: int + self, + scope: QueryScope, + start_ms: int | None, + end_ms: int | None, + q: str, + buckets: int, + as_of_ms: int | None = None, ) -> TraceHistogram: - return await self.storage.trace_histogram(scope, start_ms, end_ms, q, buckets) + return await self.storage.trace_histogram(scope, start_ms, end_ms, q, buckets, as_of_ms) async def run_values( - self, scope: QueryScope, start_ms: int, end_ms: int, q: str, field: RunField, contains: str, limit: int + self, + scope: QueryScope, + start_ms: int | None, + end_ms: int | None, + q: str, + field: RunField, + contains: str, + limit: int, + as_of_ms: int | None = None, ) -> RunValues: - return await self.storage.run_values(scope, start_ms, end_ms, q, field, contains, limit) + return await self.storage.run_values(scope, start_ms, end_ms, q, field, contains, limit, as_of_ms) async def get_trace( self, @@ -145,6 +162,22 @@ class TraceReceiver: ) -> SpanErrorPage | None: return await self.storage.get_span_error(trace_id, span_id, scope, trace_ref, cursor) + async def get_trace_metadata(self, scope: QueryScope, id: str) -> TraceMetadata | None: + return await self.storage.get_trace_metadata(scope, id) + + async def get_trace_spans( + self, scope: QueryScope, id: str, cursor: str | None, page_size: int + ) -> TraceSpansPage | None: + return await self.storage.get_trace_spans(scope, id, cursor, page_size) + + async def get_span_by_id(self, scope: QueryScope, id: str, span_id: str) -> SpanDetail | None: + return await self.storage.get_span_by_id(scope, id, span_id) + + async def get_span_error_by_id( + self, scope: QueryScope, id: str, span_id: str, cursor: str | None + ) -> SpanErrorPage | None: + return await self.storage.get_span_error_by_id(scope, id, span_id, cursor) + async def _read_body(chunks: AsyncIterable[bytes]) -> bytes: with BytesIO() as body: diff --git a/scripts/generate_trace_types.py b/scripts/generate_trace_types.py index 5d751e674c1..2cde8dc3313 100644 --- a/scripts/generate_trace_types.py +++ b/scripts/generate_trace_types.py @@ -15,7 +15,7 @@ from tempfile import TemporaryDirectory from types import MappingProxyType from typing import Final -from pydantic import BaseModel, ConfigDict, JsonValue, TypeAdapter +from pydantic import BaseModel, ConfigDict, Field, JsonValue, TypeAdapter ROOT: Final = Path(__file__).resolve().parents[1] TOOLING: Final = ROOT / "scripts/trace_codegen" @@ -34,7 +34,7 @@ class GeneratorConfig(BaseModel): options: tuple[str, ...] -def export(crate: str) -> Mapping[str, Mapping[str, JsonValue]]: +def export(crate: str, api: bool = False) -> Mapping[str, Mapping[str, JsonValue]]: result: Final = subprocess.run( ( "cargo", @@ -48,6 +48,7 @@ def export(crate: str) -> Mapping[str, Mapping[str, JsonValue]]: f"export-{crate}-schema", "--features", "schema", + *(("--", "--api") if api else ()), ), check=True, stdout=subprocess.PIPE, @@ -100,7 +101,7 @@ def generate( "pydantic_v2.BaseModel", "--enable-faux-immutability", "--additional-imports", - "collections.abc.Mapping,typing.TypeAlias", + "collections.abc.Mapping,typing.TypeAlias,pydantic.JsonValue", ) ) subprocess.run( @@ -154,6 +155,79 @@ def reconcile_schemas(expected: frozenset[Path], check: bool) -> bool: return True +class OperationConfig(BaseModel): + model_config = ConfigDict(frozen=True) + operation_id: str = Field(alias="operationId") + + +class OpenAPIConfig(BaseModel): + model_config = ConfigDict(frozen=True) + paths: Mapping[str, Mapping[str, OperationConfig]] + + +def export_openapi() -> str: + result: Final = subprocess.run( + ( + "cargo", + "run", + "--locked", + "--manifest-path", + str(ROOT / "litellm-rust/Cargo.toml"), + "-p", + "litellm-traces-clickhouse", + "--bin", + "export-traces-openapi", + "--features", + "schema", + ), + check=True, + stdout=subprocess.PIPE, + text=True, + ) + document: Final = TypeAdapter(dict[str, JsonValue]).validate_json(result.stdout) + return json.dumps(document, indent=2, sort_keys=True) + "\n" + + +def route_entries(document: OpenAPIConfig) -> Iterator[str]: + for path, methods in document.paths.items(): + for method, operation in methods.items(): + yield f" {operation.operation_id!r}: TraceOperation({path!r}, {method.upper()!r}, {operation.operation_id!r})," + + +def route_source(document: str) -> str: + paths: Final = OpenAPIConfig.model_validate_json(document) + entries: Final = "\n".join(route_entries(paths)) + return f"""# @generated by scripts/generate_trace_types.py, do not edit +from collections.abc import Mapping +from dataclasses import dataclass +from types import MappingProxyType +from typing import Final + + +@dataclass(frozen=True, slots=True) +class TraceOperation: + path: str + method: str + operation_id: str + + +OPERATIONS: Final[Mapping[str, TraceOperation]] = MappingProxyType({{ +{entries} +}}) +""" + + +def formatted_routes(document: str, directory: Path) -> str: + output: Final = directory / "routes.py" + output.write_text(route_source(document)) + subprocess.run( + (sys.executable, "-m", "ruff", "format", "--line-length", "120", str(output)), + check=True, + stdout=subprocess.DEVNULL, + ) + return output.read_text() + + def main() -> int: parser: Final = argparse.ArgumentParser(description="Regenerate trace schemas and Python wire contracts") parser.add_argument("--check", action="store_true", help="compare fresh schemas and Python with committed files") @@ -164,16 +238,22 @@ def main() -> int: return 1 domain: Final = export("traces") clickhouse: Final = export("traces-clickhouse") - exported: Final = tuple(schema_files(domain, clickhouse)) + api: Final = export("traces", api=True) + openapi: Final = export_openapi() + exported: Final = tuple(schema_files(domain, clickhouse, api)) schema_results: Final = tuple(publish(path, content, args.check) for path, content in exported) schema_set_matches: Final = reconcile_schemas(frozenset(path for path, _ in exported), args.check) with TemporaryDirectory(prefix="trace-codegen-") as temporary: directory: Final = Path(temporary) types: Final = generate(domain, "types", directory, config) models: Final = generate(clickhouse, "models", directory, config) + requests: Final = generate(api, "requests", directory, config) python_results: Final = ( publish(GENERATED / "types.py", types.read_text(), args.check), publish(GENERATED / "models.py", models.read_text(), args.check), + publish(GENERATED / "requests.py", requests.read_text(), args.check), + publish(GENERATED / "openapi.json", openapi, args.check), + publish(GENERATED / "routes.py", formatted_routes(openapi, directory), args.check), ) return 0 if all((schema_set_matches, *schema_results, *python_results)) else 1 @@ -181,8 +261,9 @@ def main() -> int: def schema_files( domain: Mapping[str, Mapping[str, JsonValue]], clickhouse: Mapping[str, Mapping[str, JsonValue]], + api: Mapping[str, Mapping[str, JsonValue]], ) -> Iterator[tuple[Path, str]]: - for crate, schemas in (("traces", domain), ("traces-clickhouse", clickhouse)): + for crate, schemas in (("traces", domain), ("traces-clickhouse", clickhouse), ("api", api)): for name, schema in schemas.items(): yield TOOLING / "schemas" / crate / f"{name}.json", json.dumps(schema, indent=2, sort_keys=True) + "\n" diff --git a/scripts/trace_codegen/README.md b/scripts/trace_codegen/README.md index f9071001c78..e68281649ff 100644 --- a/scripts/trace_codegen/README.md +++ b/scripts/trace_codegen/README.md @@ -1,9 +1,26 @@ -Run `uv run scripts/generate_trace_types.py` from the repository root to export Rust schemas and regenerate the Python trace contracts. Run the same command with `--check` to compare fresh output with the committed schemas and Python files +The trace API contract is owned by Rust. `litellm-rust/crates/traces/src/api.rs` defines public requests, responses, validation limits and defaults. `src/api/openapi.rs` defines paths, operation IDs, authentication and operation semantics. Domain response fields remain in their owning Rust types, and ClickHouse query-help types remain in `traces-clickhouse` -The script pins datamodel-code-generator in its inline dependency metadata. Rust uses the workspace's locked Schemars version through each owning crate's optional `schema` feature. Neither tool is a Python runtime dependency +Run `uv run scripts/generate_trace_types.py` from the repository root to generate JSON Schema 2020-12, OpenAPI 3.1, Python validation models and route metadata. Run the same command with `--check` to detect drift. The committed OpenAPI document is `litellm/rust_bridge/trace/generated/openapi.json` -Each crate exports its own roots using JSON Schema 2020-12. Request parameters use Schemars' deserialization contract. Trace views and query help use its serialization contract. Lens rows use their ClickHouse deserialization schemas, including quoted numbers and numeric boolean flags +FastAPI imports generated request models and route metadata, invokes the Rust bridge, and merges the generated document into `/openapi.json`. Its route decorators do not own trace API documentation. The dashboard's `npm run gen:api` consumes that merged document. A future Rust HTTP server can consume the same request types and operation contract -The templates preserve tuple conversion, immutable tuple defaults, and bounded `ReadOnly` TypedDict fields. Pydantic models use the generator's frozen-model option and each schema's extra-field policy. ClickHouse numeric schemas select bounded, normalized Python scalar types through schema metadata consumed by the model template +| Operation | Contract | +| --- | --- | +| `GET /v1/traces` | Search summaries with `q`, bounded `page_size`, four sort fields and a continuation cursor | +| `GET /v1/traces/histogram` | Count the same search in bounded time buckets | +| `GET /v1/traces/values/{field}` | Return bounded top-K search suggestions | +| `GET /v1/traces/{id}` | Read summary and agent metadata by canonical ID | +| `GET /v1/traces/{id}/spans` | Traverse bounded pages of spans | +| `GET /v1/traces/{id}/spans/{span_id}` | Read raw span input, output and attributes | +| `GET /v1/traces/{id}/spans/{span_id}/error` | Traverse bounded diagnostic text | +| `POST /v1/traces/query` | Execute scoped, bounded ClickHouse SQL with native parameter binding | +| `GET /v1/traces/query/help` | Discover logical views, physical tables, examples and limits | +| `POST /v1/traces` | Receive the standard OTLP JSON or protobuf protocol | -Edit the owning Rust contract, schema annotation, or generation configuration, then regenerate. Never edit `litellm/rust_bridge/trace/generated/` manually. The SQL response envelope remains handwritten in `queries.py` +The list, histogram and suggestions return a resolved `[start_ms,end_ms)` window and ingestion cutoff. Reuse those values to compare aggregates with the list. List cursors retain the window, query, scope and ordering. The cutoff excludes exports stamped later; buffered or distributed writes stamped before the cutoff can become visible later. It is not a database transaction snapshot. Collection pages use `data` and `next_cursor`; text pages retain their text-specific envelope. `trace_id` is the original OTLP identifier; `id` includes ownership and is the public read identifier. `root_status` describes the root span, while `has_error` reports any failed span + +SQL exposes logical `traces`, `spans` and `calls` views under invoker row policies. Views summarize visible canonical spans. Curated user-only trace reads additionally require full ownership, so their membership can differ from SQL views. SQL pages are live and callers own keyset continuation. SQL span counters and token totals describe canonical normalized spans; curated call counts and token totals additionally resolve graph wrappers. These enrichment metrics have distinct SQL column names. Spend enrichment is best effort because replacing spend rows do not preserve historical versions + +JSON read failures use RFC 9457 `application/problem+json` with a stable `code`. OTLP retains its protocol responses. Invalid or unknown query arguments are rejected + +The generator pins datamodel-code-generator in inline dependency metadata and uses locked Schemars and Utoipa dependencies behind the optional Rust `schema` feature. These tools are not Python runtime dependencies. Templates preserve immutable Python collections and bounded types. Edit Rust or generation configuration, then regenerate; never edit generated contracts manually diff --git a/scripts/trace_codegen/schemas/api/TraceErrorPageRequest.json b/scripts/trace_codegen/schemas/api/TraceErrorPageRequest.json new file mode 100644 index 00000000000..b3ebc514f55 --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceErrorPageRequest.json @@ -0,0 +1,16 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "cursor": { + "default": null, + "maxLength": 512, + "type": [ + "string", + "null" + ] + } + }, + "title": "TraceErrorPageRequest", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceHistogramRequest.json b/scripts/trace_codegen/schemas/api/TraceHistogramRequest.json new file mode 100644 index 00000000000..f3128e10bbf --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceHistogramRequest.json @@ -0,0 +1,50 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "as_of_ms": { + "default": null, + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": [ + "integer", + "null" + ] + }, + "buckets": { + "default": 60, + "format": "uint16", + "maximum": 240, + "minimum": 1, + "type": "integer" + }, + "end_ms": { + "default": null, + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + }, + "q": { + "default": "", + "maxLength": 1000, + "type": "string" + }, + "start_ms": { + "default": null, + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + "title": "TraceHistogramRequest", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceListRequest.json b/scripts/trace_codegen/schemas/api/TraceListRequest.json new file mode 100644 index 00000000000..57fe53c992e --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceListRequest.json @@ -0,0 +1,84 @@ +{ + "$defs": { + "TraceSortDirection": { + "enum": [ + "asc", + "desc" + ], + "type": "string" + }, + "TraceSortField": { + "enum": [ + "start_ms", + "duration_ms", + "span_count", + "error_count" + ], + "type": "string" + } + }, + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "as_of_ms": { + "default": null, + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": [ + "integer", + "null" + ] + }, + "cursor": { + "default": null, + "maxLength": 512, + "type": [ + "string", + "null" + ] + }, + "end_ms": { + "default": null, + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + }, + "page_size": { + "default": 50, + "format": "uint16", + "maximum": 500, + "minimum": 1, + "type": "integer" + }, + "q": { + "default": "", + "maxLength": 1000, + "type": "string" + }, + "sort_by": { + "$ref": "#/$defs/TraceSortField", + "default": "start_ms" + }, + "sort_dir": { + "$ref": "#/$defs/TraceSortDirection", + "default": "desc" + }, + "start_ms": { + "default": null, + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + "title": "TraceListRequest", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceMetadata.json b/scripts/trace_codegen/schemas/api/TraceMetadata.json new file mode 100644 index 00000000000..d9c95d59445 --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceMetadata.json @@ -0,0 +1,216 @@ +{ + "$defs": { + "AgentNode": { + "description": "One distinct agent in a trace: 200 invocations of `researcher` are one node.", + "properties": { + "duration_ms": { + "format": "double", + "type": "number" + }, + "invocations": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "llm_calls": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "name": { + "type": "string" + }, + "parent_agent": { + "type": [ + "string", + "null" + ] + }, + "spend": { + "format": "double", + "type": [ + "number", + "null" + ] + }, + "tool_calls": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + } + }, + "required": [ + "name", + "parent_agent", + "invocations", + "llm_calls", + "tool_calls", + "duration_ms", + "spend" + ], + "type": "object" + }, + "SpanStatus": { + "enum": [ + "ok", + "error", + "unset" + ], + "type": "string" + }, + "TraceSummary": { + "properties": { + "agent_count": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "agent_invocations": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "agent_names": { + "items": { + "type": "string" + }, + "type": "array" + }, + "duration_ms": { + "format": "double", + "type": "number" + }, + "error_count": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "frameworks": { + "items": { + "type": "string" + }, + "type": "array" + }, + "has_error": { + "type": "boolean" + }, + "id": { + "type": "string" + }, + "input_preview": { + "type": "string" + }, + "input_tokens": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "llm_calls": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "models": { + "items": { + "type": "string" + }, + "type": "array" + }, + "name": { + "type": "string" + }, + "output_tokens": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "resolution_limited": { + "type": "boolean" + }, + "root_status": { + "$ref": "#/$defs/SpanStatus" + }, + "service": { + "type": "string" + }, + "span_count": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "spend": { + "format": "double", + "type": [ + "number", + "null" + ] + }, + "start_time": { + "type": "string" + }, + "tool_calls": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "trace_id": { + "type": "string" + } + }, + "required": [ + "resolution_limited", + "trace_id", + "id", + "name", + "service", + "agent_names", + "frameworks", + "input_preview", + "start_time", + "duration_ms", + "root_status", + "has_error", + "span_count", + "agent_count", + "agent_invocations", + "llm_calls", + "tool_calls", + "error_count", + "input_tokens", + "output_tokens", + "models", + "spend" + ], + "type": "object" + } + }, + "$schema": "https://json-schema.org/draft/2020-12/schema", + "properties": { + "agents": { + "items": { + "$ref": "#/$defs/AgentNode" + }, + "type": "array" + }, + "summary": { + "$ref": "#/$defs/TraceSummary" + } + }, + "required": [ + "summary", + "agents" + ], + "title": "TraceMetadata", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceNoQueryRequest.json b/scripts/trace_codegen/schemas/api/TraceNoQueryRequest.json new file mode 100644 index 00000000000..2c9295c6325 --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceNoQueryRequest.json @@ -0,0 +1,6 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "title": "TraceNoQueryRequest", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceProblem.json b/scripts/trace_codegen/schemas/api/TraceProblem.json new file mode 100644 index 00000000000..567fa8aae39 --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceProblem.json @@ -0,0 +1,83 @@ +{ + "$defs": { + "TraceInvalidParam": { + "additionalProperties": false, + "properties": { + "location": { + "type": "string" + }, + "reason": { + "type": "string" + } + }, + "required": [ + "location", + "reason" + ], + "type": "object" + }, + "TraceProblemCode": { + "enum": [ + "invalid_request", + "unauthorized", + "forbidden", + "not_found", + "trace_changed", + "too_large", + "unavailable", + "query_rejected", + "query_limit_exceeded", + "query_unavailable", + "internal_error" + ], + "type": "string" + } + }, + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "code": { + "$ref": "#/$defs/TraceProblemCode" + }, + "database_code": { + "format": "uint32", + "maximum": 4294967295, + "minimum": 0, + "type": [ + "integer", + "null" + ] + }, + "detail": { + "type": "string" + }, + "errors": { + "default": [], + "items": { + "$ref": "#/$defs/TraceInvalidParam" + }, + "type": "array" + }, + "status": { + "format": "uint16", + "maximum": 65535, + "minimum": 0, + "type": "integer" + }, + "title": { + "type": "string" + }, + "type": { + "type": "string" + } + }, + "required": [ + "type", + "title", + "status", + "detail", + "code" + ], + "title": "TraceProblem", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceQueryRequest.json b/scripts/trace_codegen/schemas/api/TraceQueryRequest.json new file mode 100644 index 00000000000..098965a3234 --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceQueryRequest.json @@ -0,0 +1,59 @@ +{ + "$defs": { + "SqlParameter": { + "anyOf": [ + { + "type": "string" + }, + { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + }, + { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + { + "format": "double", + "type": "number" + }, + { + "type": "boolean" + }, + { + "type": "null" + }, + { + "items": { + "type": "string" + }, + "type": "array" + } + ] + } + }, + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "params": { + "additionalProperties": { + "$ref": "#/$defs/SqlParameter" + }, + "default": {}, + "type": "object" + }, + "sql": { + "minLength": 1, + "type": "string" + } + }, + "required": [ + "sql" + ], + "title": "TraceQueryRequest", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceSQLResponse.json b/scripts/trace_codegen/schemas/api/TraceSQLResponse.json new file mode 100644 index 00000000000..f6a6ee395d1 --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceSQLResponse.json @@ -0,0 +1,88 @@ +{ + "$defs": { + "TraceQueryStatistics": { + "additionalProperties": true, + "properties": { + "bytes_read": { + "$ref": "#/$defs/UnsignedCount" + }, + "elapsed": { + "format": "double", + "type": "number" + }, + "rows_read": { + "$ref": "#/$defs/UnsignedCount" + } + }, + "required": [ + "elapsed", + "rows_read", + "bytes_read" + ], + "type": "object" + }, + "TraceSQLColumn": { + "additionalProperties": true, + "properties": { + "name": { + "type": "string" + }, + "type": { + "type": "string" + } + }, + "required": [ + "name", + "type" + ], + "type": "object" + }, + "UnsignedCount": { + "anyOf": [ + { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + { + "type": "string" + } + ] + } + }, + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": true, + "properties": { + "data": { + "items": { + "additionalProperties": true, + "type": "object" + }, + "type": "array", + "x-python-normalized": { + "type": "tuple[Mapping[str, JsonValue], ...]" + } + }, + "meta": { + "items": { + "$ref": "#/$defs/TraceSQLColumn" + }, + "type": "array" + }, + "rows": { + "$ref": "#/$defs/UnsignedCount" + }, + "statistics": { + "$ref": "#/$defs/TraceQueryStatistics" + } + }, + "required": [ + "meta", + "data", + "rows", + "statistics" + ], + "title": "TraceSQLResponse", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceSpanPageRequest.json b/scripts/trace_codegen/schemas/api/TraceSpanPageRequest.json new file mode 100644 index 00000000000..64a3ed35c6a --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceSpanPageRequest.json @@ -0,0 +1,23 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "cursor": { + "default": null, + "maxLength": 512, + "type": [ + "string", + "null" + ] + }, + "page_size": { + "default": 100, + "format": "uint16", + "maximum": 500, + "minimum": 1, + "type": "integer" + } + }, + "title": "TraceSpanPageRequest", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceSpansPage.json b/scripts/trace_codegen/schemas/api/TraceSpansPage.json new file mode 100644 index 00000000000..88a3294f598 --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceSpansPage.json @@ -0,0 +1,149 @@ +{ + "$defs": { + "Span": { + "properties": { + "agent": { + "type": "string" + }, + "duration_ms": { + "format": "double", + "type": "number" + }, + "error": { + "type": [ + "string", + "null" + ] + }, + "error_truncated": { + "type": "boolean" + }, + "framework": { + "type": "string" + }, + "input_preview": { + "type": "string" + }, + "input_tokens": { + "format": "uint32", + "maximum": 4294967295, + "minimum": 0, + "type": "integer" + }, + "litellm_request_id": { + "type": [ + "string", + "null" + ] + }, + "model": { + "type": [ + "string", + "null" + ] + }, + "name": { + "type": "string" + }, + "output_tokens": { + "format": "uint32", + "maximum": 4294967295, + "minimum": 0, + "type": "integer" + }, + "parent_span_id": { + "type": [ + "string", + "null" + ] + }, + "span_id": { + "type": "string" + }, + "spend": { + "format": "double", + "type": [ + "number", + "null" + ] + }, + "start_offset_ms": { + "format": "double", + "type": "number" + }, + "status": { + "$ref": "#/$defs/SpanStatus" + }, + "type": { + "$ref": "#/$defs/SpanType" + } + }, + "required": [ + "span_id", + "parent_span_id", + "name", + "type", + "agent", + "framework", + "start_offset_ms", + "duration_ms", + "status", + "error", + "error_truncated", + "input_preview", + "model", + "input_tokens", + "output_tokens", + "litellm_request_id", + "spend" + ], + "type": "object" + }, + "SpanStatus": { + "enum": [ + "ok", + "error", + "unset" + ], + "type": "string" + }, + "SpanType": { + "enum": [ + "agent", + "llm", + "tool", + "chain", + "framework", + "retriever", + "embedding", + "reranker", + "guardrail", + "evaluator", + "prompt", + "decision" + ], + "type": "string" + } + }, + "$schema": "https://json-schema.org/draft/2020-12/schema", + "properties": { + "data": { + "items": { + "$ref": "#/$defs/Span" + }, + "type": "array" + }, + "next_cursor": { + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "data", + "next_cursor" + ], + "title": "TraceSpansPage", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/api/TraceValuesRequest.json b/scripts/trace_codegen/schemas/api/TraceValuesRequest.json new file mode 100644 index 00000000000..100a720352d --- /dev/null +++ b/scripts/trace_codegen/schemas/api/TraceValuesRequest.json @@ -0,0 +1,55 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "as_of_ms": { + "default": null, + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": [ + "integer", + "null" + ] + }, + "contains": { + "default": "", + "maxLength": 200, + "type": "string" + }, + "end_ms": { + "default": null, + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + }, + "limit": { + "default": 20, + "format": "uint16", + "maximum": 100, + "minimum": 1, + "type": "integer" + }, + "q": { + "default": "", + "maxLength": 1000, + "type": "string" + }, + "start_ms": { + "default": null, + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": [ + "integer", + "null" + ] + } + }, + "title": "TraceValuesRequest", + "type": "object" +} diff --git a/scripts/trace_codegen/schemas/traces-clickhouse/TraceQueryHelp.json b/scripts/trace_codegen/schemas/traces-clickhouse/TraceQueryHelp.json index b09dec6ac77..3f3b7334e03 100644 --- a/scripts/trace_codegen/schemas/traces-clickhouse/TraceQueryHelp.json +++ b/scripts/trace_codegen/schemas/traces-clickhouse/TraceQueryHelp.json @@ -77,7 +77,7 @@ "type": "string" }, "table": { - "$ref": "#/$defs/TraceTableName" + "$ref": "#/$defs/TraceQueryTableName" }, "truncated": { "type": "boolean" @@ -93,22 +93,6 @@ ], "type": "object" }, - "TraceQueryColumn": { - "additionalProperties": true, - "properties": { - "name": { - "type": "string" - }, - "type": { - "type": "string" - } - }, - "required": [ - "name", - "type" - ], - "type": "object" - }, "TraceQueryExample": { "properties": { "name": { @@ -162,7 +146,7 @@ "type": "string" }, "table": { - "$ref": "#/$defs/TraceTableName" + "$ref": "#/$defs/TraceQueryTableName" }, "truncated": { "type": "boolean" @@ -220,7 +204,7 @@ "type": "string" }, "table": { - "$ref": "#/$defs/TraceTableName" + "$ref": "#/$defs/TraceQueryTableName" }, "type": { "type": "string" @@ -264,12 +248,12 @@ "properties": { "columns": { "items": { - "$ref": "#/$defs/TraceQueryColumn" + "$ref": "#/$defs/TraceSQLColumn" }, "type": "array" }, "name": { - "$ref": "#/$defs/TraceTableName" + "$ref": "#/$defs/TraceQueryTableName" } }, "required": [ @@ -278,13 +262,32 @@ ], "type": "object" }, - "TraceTableName": { + "TraceQueryTableName": { "enum": [ + "traces", + "spans", + "calls", "otel_traces", "agent_traces_by_key", "spend_logs" ], "type": "string" + }, + "TraceSQLColumn": { + "additionalProperties": true, + "properties": { + "name": { + "type": "string" + }, + "type": { + "type": "string" + } + }, + "required": [ + "name", + "type" + ], + "type": "object" } }, "$schema": "https://json-schema.org/draft/2020-12/schema", diff --git a/scripts/trace_codegen/schemas/traces/RunField.json b/scripts/trace_codegen/schemas/traces/RunField.json index ab22c4c67cd..6ffc8bd4a0f 100644 --- a/scripts/trace_codegen/schemas/traces/RunField.json +++ b/scripts/trace_codegen/schemas/traces/RunField.json @@ -3,7 +3,8 @@ "enum": [ "name", "agent", - "status", + "root_status", + "has_error", "model", "input", "trace_id", diff --git a/scripts/trace_codegen/schemas/traces/RunValues.json b/scripts/trace_codegen/schemas/traces/RunValues.json index 9105ffb28c1..a80467c92b5 100644 --- a/scripts/trace_codegen/schemas/traces/RunValues.json +++ b/scripts/trace_codegen/schemas/traces/RunValues.json @@ -1,4 +1,35 @@ { + "$defs": { + "TraceQueryWindow": { + "additionalProperties": false, + "properties": { + "as_of_ms": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "end_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + }, + "start_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + } + }, + "required": [ + "start_ms", + "end_ms", + "as_of_ms" + ], + "type": "object" + } + }, "$schema": "https://json-schema.org/draft/2020-12/schema", "description": "Distinct values of one run field, most common first.", "properties": { @@ -7,9 +38,13 @@ "type": "string" }, "type": "array" + }, + "window": { + "$ref": "#/$defs/TraceQueryWindow" } }, "required": [ + "window", "values" ], "title": "RunValues", diff --git a/scripts/trace_codegen/schemas/traces/SpanDetail.json b/scripts/trace_codegen/schemas/traces/SpanDetail.json index d86dd37cd80..5252bb94ccf 100644 --- a/scripts/trace_codegen/schemas/traces/SpanDetail.json +++ b/scripts/trace_codegen/schemas/traces/SpanDetail.json @@ -1,136 +1,4 @@ { - "$defs": { - "ChatRole": { - "enum": [ - "system", - "user", - "assistant", - "tool" - ], - "type": "string" - }, - "UIContent": { - "oneOf": [ - { - "properties": { - "kind": { - "const": "messages", - "type": "string" - }, - "messages": { - "items": { - "$ref": "#/$defs/UIMessage" - }, - "type": "array" - } - }, - "required": [ - "kind", - "messages" - ], - "title": "UIMessages", - "type": "object" - }, - { - "properties": { - "fields": { - "items": { - "$ref": "#/$defs/UIField" - }, - "type": "array" - }, - "kind": { - "const": "fields", - "type": "string" - } - }, - "required": [ - "kind", - "fields" - ], - "title": "UIFields", - "type": "object" - }, - { - "properties": { - "kind": { - "const": "text", - "type": "string" - }, - "text": { - "type": "string" - } - }, - "required": [ - "kind", - "text" - ], - "title": "UIText", - "type": "object" - } - ] - }, - "UIField": { - "properties": { - "key": { - "type": "string" - }, - "value": { - "type": "string" - } - }, - "required": [ - "key", - "value" - ], - "type": "object" - }, - "UIMessage": { - "properties": { - "content": { - "type": "string" - }, - "name": { - "type": [ - "string", - "null" - ] - }, - "role": { - "$ref": "#/$defs/ChatRole" - }, - "tool_calls": { - "items": { - "$ref": "#/$defs/UIToolCall" - }, - "type": [ - "array", - "null" - ] - } - }, - "required": [ - "role", - "content" - ], - "type": "object" - }, - "UIToolCall": { - "properties": { - "arguments": { - "type": "string" - }, - "name": { - "type": "string" - } - }, - "required": [ - "name", - "arguments" - ], - "type": "object" - } - }, "$schema": "https://json-schema.org/draft/2020-12/schema", "properties": { "attributes": { @@ -142,23 +10,15 @@ "input": { "type": "string" }, - "input_ui": { - "$ref": "#/$defs/UIContent" - }, "output": { "type": "string" }, - "output_ui": { - "$ref": "#/$defs/UIContent" - }, "span_id": { "type": "string" } }, "required": [ "span_id", - "input_ui", - "output_ui", "input", "output", "attributes" diff --git a/scripts/trace_codegen/schemas/traces/Trace.json b/scripts/trace_codegen/schemas/traces/Trace.json index dc66f77412c..a42c519b683 100644 --- a/scripts/trace_codegen/schemas/traces/Trace.json +++ b/scripts/trace_codegen/schemas/traces/Trace.json @@ -195,8 +195,7 @@ "items": { "type": "string" }, - "type": "array", - "x-python-optional": true + "type": "array" }, "duration_ms": { "format": "double", @@ -212,8 +211,13 @@ "items": { "type": "string" }, - "type": "array", - "x-python-optional": true + "type": "array" + }, + "has_error": { + "type": "boolean" + }, + "id": { + "type": "string" }, "input_preview": { "type": "string" @@ -246,8 +250,10 @@ "type": "integer" }, "resolution_limited": { - "type": "boolean", - "x-python-optional": true + "type": "boolean" + }, + "root_status": { + "$ref": "#/$defs/SpanStatus" }, "service": { "type": "string" @@ -268,9 +274,6 @@ "start_time": { "type": "string" }, - "status": { - "$ref": "#/$defs/SpanStatus" - }, "tool_calls": { "format": "uint64", "maximum": 18446744073709551615, @@ -279,16 +282,12 @@ }, "trace_id": { "type": "string" - }, - "trace_ref": { - "type": "string", - "x-python-optional": true } }, "required": [ "resolution_limited", "trace_id", - "trace_ref", + "id", "name", "service", "agent_names", @@ -296,7 +295,8 @@ "input_preview", "start_time", "duration_ms", - "status", + "root_status", + "has_error", "span_count", "agent_count", "agent_invocations", diff --git a/scripts/trace_codegen/schemas/traces/TraceHistogram.json b/scripts/trace_codegen/schemas/traces/TraceHistogram.json index 9871d6d5683..a8a5d11f3b6 100644 --- a/scripts/trace_codegen/schemas/traces/TraceHistogram.json +++ b/scripts/trace_codegen/schemas/traces/TraceHistogram.json @@ -60,6 +60,35 @@ "agents" ], "type": "object" + }, + "TraceQueryWindow": { + "additionalProperties": false, + "properties": { + "as_of_ms": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "end_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + }, + "start_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + } + }, + "required": [ + "start_ms", + "end_ms", + "as_of_ms" + ], + "type": "object" } }, "$schema": "https://json-schema.org/draft/2020-12/schema", @@ -70,9 +99,13 @@ "$ref": "#/$defs/HistogramBucket" }, "type": "array" + }, + "window": { + "$ref": "#/$defs/TraceQueryWindow" } }, "required": [ + "window", "buckets" ], "title": "TraceHistogram", diff --git a/scripts/trace_codegen/schemas/traces/TracePage.json b/scripts/trace_codegen/schemas/traces/TracePage.json index dfa02ba0d39..777af9481e0 100644 --- a/scripts/trace_codegen/schemas/traces/TracePage.json +++ b/scripts/trace_codegen/schemas/traces/TracePage.json @@ -8,6 +8,35 @@ ], "type": "string" }, + "TraceQueryWindow": { + "additionalProperties": false, + "properties": { + "as_of_ms": { + "format": "uint64", + "maximum": 18446744073709551615, + "minimum": 0, + "type": "integer" + }, + "end_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + }, + "start_ms": { + "format": "int64", + "maximum": 9223372036854775807, + "minimum": -9223372036854775808, + "type": "integer" + } + }, + "required": [ + "start_ms", + "end_ms", + "as_of_ms" + ], + "type": "object" + }, "TraceSummary": { "properties": { "agent_count": { @@ -26,8 +55,7 @@ "items": { "type": "string" }, - "type": "array", - "x-python-optional": true + "type": "array" }, "duration_ms": { "format": "double", @@ -43,8 +71,13 @@ "items": { "type": "string" }, - "type": "array", - "x-python-optional": true + "type": "array" + }, + "has_error": { + "type": "boolean" + }, + "id": { + "type": "string" }, "input_preview": { "type": "string" @@ -77,8 +110,10 @@ "type": "integer" }, "resolution_limited": { - "type": "boolean", - "x-python-optional": true + "type": "boolean" + }, + "root_status": { + "$ref": "#/$defs/SpanStatus" }, "service": { "type": "string" @@ -99,9 +134,6 @@ "start_time": { "type": "string" }, - "status": { - "$ref": "#/$defs/SpanStatus" - }, "tool_calls": { "format": "uint64", "maximum": 18446744073709551615, @@ -110,16 +142,12 @@ }, "trace_id": { "type": "string" - }, - "trace_ref": { - "type": "string", - "x-python-optional": true } }, "required": [ "resolution_limited", "trace_id", - "trace_ref", + "id", "name", "service", "agent_names", @@ -127,7 +155,8 @@ "input_preview", "start_time", "duration_ms", - "status", + "root_status", + "has_error", "span_count", "agent_count", "agent_invocations", @@ -155,9 +184,13 @@ "string", "null" ] + }, + "window": { + "$ref": "#/$defs/TraceQueryWindow" } }, "required": [ + "window", "data", "next_cursor" ], diff --git a/tests/test_litellm_rust/test_traces.py b/tests/test_litellm_rust/test_traces.py index 822f11a893d..1b07dbae038 100644 --- a/tests/test_litellm_rust/test_traces.py +++ b/tests/test_litellm_rust/test_traces.py @@ -21,6 +21,8 @@ from pydantic import BaseModel, ConfigDict, JsonValue, TypeAdapter from litellm.constants import AGENT_TRACING_LIST_PAGE_SIZE, OTLP_MAX_ATTRIBUTE_VALUE_BYTES from litellm.rust_bridge._native import NativeTraceConfig, NativeTraceStorage from litellm.rust_bridge.trace.generated.models import TraceQueryHelp +from litellm.rust_bridge.trace.generated.requests import Span as SpanModel +from litellm.rust_bridge.trace.generated.requests import TraceMetadata, TraceSpansPage from litellm.rust_bridge.trace.generated.types import AllQueryScope, Trace, TracePage from litellm.rust_bridge.trace.storage import ClickHouseStorage, TraceStorageConfig, span_rows from litellm.tracing import Tenant, TraceReceiver, TracingPayloadTooLargeError @@ -148,7 +150,9 @@ async def test_from_env_reads_with_clickhouse_url( monkeypatch.delenv("CLICKHOUSE_READER_URL", raising=False) scope: Final = AllQueryScope(kind="all") page: Final = await TraceReceiver.from_env().list_traces(scope, 0, 1) - assert page == {"data": (), "next_cursor": None} + assert page["data"] == () + assert page["next_cursor"] is None + assert (page["window"]["start_ms"], page["window"]["end_ms"]) == (0, 1) assert len(recording_server.requests) == 1 @@ -216,9 +220,12 @@ def test_trace_pages_keep_the_effective_window_through_the_http_native_boundary( second: Final = client.get("/v1/traces", params={**params, "cursor": cursor}) assert second.status_code == 200, second.text second_body: Final = _TRACE_PAGE.validate_python(second.json()) - refs: Final = tuple(row["trace_ref"] for row in (*first_body["data"], *second_body["data"])) + refs: Final = tuple(row["id"] for row in (*first_body["data"], *second_body["data"])) assert refs == tuple(row["trace_ref"] for row in reversed(rows)) assert second_body["next_cursor"] is None + assert second_body["window"] == first_body["window"] + assert second_body["data"][0]["root_status"] == "ok" + assert not second_body["data"][0]["has_error"] run_parameters: Final = tuple( parse_qs(urlsplit(request.path).query) for request in recording_server.requests @@ -229,12 +236,18 @@ def test_trace_pages_keep_the_effective_window_through_the_http_native_boundary( first_parameters["param_start_ms"], first_parameters["param_end_ms"], ) + assert run_parameters[1]["param_as_of_ms"] == first_parameters["param_as_of_ms"] sent: Final = len(recording_server.requests) changed: Final = client.get( "/v1/traces", params={"cursor": cursor, "end_ms": int(first_parameters["param_end_ms"][0]) + 1} ) assert changed.status_code == 400, changed.text assert len(recording_server.requests) == sent + changed_cutoff: Final = client.get( + "/v1/traces", params={"cursor": cursor, "as_of_ms": first_body["window"]["as_of_ms"] - 1} + ) + assert changed_cutoff.status_code == 400, changed_cutoff.text + assert len(recording_server.requests) == sent @pytest.mark.asyncio @@ -385,9 +398,9 @@ def test_trace_sql_endpoint_enforces_ownership_and_preserves_clickhouse_envelope "rows": 1, "statistics": {"elapsed": 0.01, "rows_read": 1, "bytes_read": 1}, } - recording_server.expected_requests = 12 if expected_status == 200 else 0 + recording_server.expected_requests = 15 if expected_status == 200 else 0 if expected_status == 200: - for _ in range(11): + for _ in range(14): recording_server.enqueue(ResponseSpec(body="")) recording_server.enqueue(ResponseSpec(body=envelope)) storage: Final = ClickHouseStorage(TraceStorageConfig(recording_server.base_url, "trace_test")) @@ -405,7 +418,8 @@ def test_trace_sql_endpoint_enforces_ownership_and_preserves_clickhouse_envelope result: Final = client.post("/v1/traces/query", json={"sql": "SELECT 42 AS answer"}) assert result.status_code == expected_status, result.text if expected_status == 403: - assert result.json() == {"detail": "Not allowed to view logs"} + assert result.json()["detail"] == "Not allowed to view logs" + assert result.json()["code"] == "forbidden" return assert result.json() == envelope assert recording_server.requests[-1].raw_body == b"SELECT 42 AS answer" @@ -424,13 +438,16 @@ def test_trace_help_endpoint_runs_native_schema_and_metadata_discovery( from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.proxy.tracing_endpoints import provide_receiver, provide_trace_query_secret, router - recording_server.expected_requests = 17 - for _ in range(11): + recording_server.expected_requests = 23 + for _ in range(14): recording_server.enqueue(ResponseSpec(body="")) for response in ( {"data": [{"name": "Model", "type": "String"}]}, {"data": []}, {"data": []}, + {"data": []}, + {"data": []}, + {"data": []}, ): recording_server.enqueue(ResponseSpec(body=response)) metadata: Final = ( @@ -463,8 +480,8 @@ def test_trace_help_endpoint_runs_native_schema_and_metadata_discovery( "types": ["string"], "expression": "JSONExtractRaw(metadata, 'custom', 'label')", } - assert body["attributes"][0]["fields"][0]["expression"] == "SpanAttributes['custom.span']" - assert body["attributes"][1]["fields"][0]["expression"] == "ResourceAttributes['custom.resource']" + assert body["attributes"][0]["fields"][0]["expression"] == "span_attributes['custom.span']" + assert body["attributes"][1]["fields"][0]["expression"] == "resource_attributes['custom.resource']" @pytest.mark.parametrize( @@ -494,8 +511,8 @@ def test_trace_sql_endpoint_distinguishes_query_errors_from_reader_failures( from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.proxy.tracing_endpoints import provide_receiver, provide_trace_query_secret, router - recording_server.expected_requests = 13 - for _ in range(11): + recording_server.expected_requests = 16 + for _ in range(14): recording_server.enqueue(ResponseSpec(body="")) recording_server.enqueue(ResponseSpec(status=clickhouse_status, body=body)) envelope: Final = { @@ -514,13 +531,13 @@ def test_trace_sql_endpoint_distinguishes_query_errors_from_reader_failures( with TestClient(app) as client: failed: Final = client.post("/v1/traces/query", json={"sql": "SELEC 42"}) assert failed.status_code == expected_status, failed.text - assert failed.json()["detail"]["database_code"] == database_code + assert failed.json().get("database_code") == database_code assert ( - failed.json()["detail"]["code"] + failed.json()["code"] == {400: "query_rejected", 422: "query_limit_exceeded", 503: "query_unavailable"}[expected_status] ) if database_code is not None: - assert failed.json()["detail"]["message"] == body.decode() + assert failed.json()["detail"] == body.decode() recovered: Final = client.post("/v1/traces/query", json={"sql": "SELECT 42 AS answer"}) assert recovered.status_code == 200, recovered.text assert recovered.json() == envelope @@ -609,7 +626,7 @@ def _fixture_trace_api( def test_fixture_backed_help_examples_execute_through_query_api(seeded_trace_api: SeededTraceAPI) -> None: api: Final = seeded_trace_api - assert {table.name for table in api.help.tables} == {"otel_traces", "spend_logs", "agent_traces_by_key"} + assert {"traces", "spans", "calls"} <= {table.name for table in api.help.tables} assert api.help.metadata.error is None assert api.help.metadata.sampled_rows == len(api.spends) assert any(field.path == ("fixture_capture", "name") for field in api.help.metadata.fields) @@ -623,15 +640,13 @@ def test_fixture_backed_help_examples_execute_through_query_api(seeded_trace_api assert recorded[0]["trace_id"] == api.spends[0]["trace_id"] assert int(str(recorded[0]["requests"])) == len(api.spends) assert math.isclose(float(str(recorded[0]["recorded_spend"])), total) - detail: Final = api.client.get(f"/v1/traces/{api.spends[0]['trace_id']}") - assert detail.status_code == 200, detail.text - assert math.isclose(TRACE.validate_json(detail.content)["summary"]["spend"] or 0, total) + detail: Final = _trace(api, api.spends[0]["trace_id"]) + assert math.isclose(detail["summary"]["spend"] or 0, total) unmatched: Final = api.query_example("LLM spans without a direct spend match") assert unmatched - assert all(row["TraceId"] != api.spends[0]["trace_id"] for row in unmatched) - unpriced: Final = api.client.get(f"/v1/traces/{unmatched[0]['TraceId']}") - assert unpriced.status_code == 200, unpriced.text - assert unpriced.json()["summary"]["spend"] is None + assert all(row["trace_id"] != api.spends[0]["trace_id"] for row in unmatched) + unpriced: Final = _trace(api, str(unmatched[0]["trace_id"])) + assert unpriced["summary"]["spend"] is None @pytest.mark.parametrize("spend", (None, 0.0, 0.125), ids=("unknown", "free", "paid")) @@ -710,9 +725,7 @@ def test_captured_sdk_cost_survives_seeding_and_is_queryable(name: str, captured rows: Final = tuple(row for row in api.spends if fixture_capture("", row).name == name) assert rows capture: Final = fixture_capture(name, rows[0]) - response: Final = api.client.get(f"/v1/traces/{capture.trace_id}") - assert response.status_code == 200, response.text - detail: Final = TRACE.validate_json(response.content) + detail: Final = _trace(api, capture.trace_id) original: Final = span_rows((TRACE_FIXTURES / f"{name}.json").read_bytes(), "application/json") assert detail["summary"]["span_count"] == len(original) if capture.spend_linked and capture.spend_complete: @@ -775,9 +788,44 @@ def test_server_side_copies_keep_every_capture_linked_to_its_spend() -> None: def _trace(api: SeededTraceAPI, trace_id: str) -> Trace: - response: Final = api.client.get(f"/v1/traces/{trace_id}") + listed: Final = api.client.get( + "/v1/traces", + params={ + "q": f"trace_id:{trace_id}", + "start_ms": 0, + "end_ms": time.time_ns() // 1_000_000 + 86_400_000, + "page_size": 2, + }, + ) + assert listed.status_code == 200, listed.text + page: Final = _TRACE_PAGE.validate_json(listed.content) + (summary,) = page["data"] + assert summary["trace_id"] == trace_id + response: Final = api.client.get(f"/v1/traces/{summary['id']}") assert response.status_code == 200, response.text - return TRACE.validate_json(response.content) + metadata: Final = TraceMetadata.model_validate_json(response.content) + first: Final = _span_page(api, summary["id"], None) + spans: Final = tuple(_trace_spans(api, summary["id"], first)) + return TRACE.validate_python( + { + **metadata.model_dump(mode="json"), + "spans": tuple(span.model_dump(mode="json") for span in spans), + "next_cursor": None, + } + ) + + +def _span_page(api: SeededTraceAPI, id: str, cursor: str | None) -> TraceSpansPage: + params: Final = {"page_size": 200} if cursor is None else {"page_size": 200, "cursor": cursor} + response: Final = api.client.get(f"/v1/traces/{id}/spans", params=params) + assert response.status_code == 200, response.text + return TraceSpansPage.model_validate_json(response.content) + + +def _trace_spans(api: SeededTraceAPI, id: str, page: TraceSpansPage) -> Iterator[SpanModel]: + yield from page.data + if page.next_cursor is not None: + yield from _trace_spans(api, id, _span_page(api, id, page.next_cursor)) def _assert_capture(api: SeededTraceAPI, name: str, rows: tuple[SpendLogRecord, ...], trace_id: str) -> None: diff --git a/tests/unit/proxy/lens/test_sources.py b/tests/unit/proxy/lens/test_sources.py index 833a477924e..3259efddc5b 100644 --- a/tests/unit/proxy/lens/test_sources.py +++ b/tests/unit/proxy/lens/test_sources.py @@ -32,14 +32,16 @@ REF: Final = "A" * 64 def summary(trace_ref: str, trace_id: str = "trace", span_count: int = 1) -> TraceSummary: return TraceSummary( + resolution_limited=False, trace_id=trace_id, - trace_ref=trace_ref, + id=trace_ref, name="run", service="svc", input_preview="", start_time="2026-01-01T00:00:00Z", duration_ms=1.0, - status="ok", + root_status="ok", + has_error=False, span_count=span_count, agent_count=0, agent_invocations=0, @@ -49,6 +51,8 @@ def summary(trace_ref: str, trace_id: str = "trace", span_count: int = 1) -> Tra input_tokens=0, output_tokens=0, models=(), + agent_names=(), + frameworks=(), spend=None, ) @@ -86,7 +90,7 @@ class FakeStorage: listed: list[tuple[QueryScope, str, str | None, int, RunOrder, tuple[str, ...]]] = field(default_factory=list) def _matching(self, q: str, trace_refs: Sequence[str]) -> list[TraceSummary]: - return [run for run in self.runs if (not trace_refs or run.get("trace_ref") in trace_refs) and q in run["name"]] + return [run for run in self.runs if (not trace_refs or run["id"] in trace_refs) and q in run["name"]] async def list_traces( self, @@ -100,11 +104,12 @@ class FakeStorage: trace_refs: Sequence[str] = (), ) -> TracePage: self.listed.append((scope, q, cursor, limit, order, tuple(trace_refs))) - ordered: Final = sorted(self._matching(q, trace_refs), key=lambda run: run.get("trace_ref", "")) - after: Final = [run for run in ordered if cursor is None or run.get("trace_ref", "") > cursor][:limit] + ordered: Final = sorted(self._matching(q, trace_refs), key=lambda run: run["id"]) + after: Final = [run for run in ordered if cursor is None or run["id"] > cursor][:limit] return TracePage( + window={"start_ms": start_ms, "end_ms": end_ms, "as_of_ms": max(end_ms, 0)}, data=tuple(after), - next_cursor=after[-1].get("trace_ref") if len(after) == limit else None, + next_cursor=after[-1]["id"] if len(after) == limit else None, ) async def count_traces( @@ -277,12 +282,49 @@ async def test_content_pages_forty_spans_at_a_time_in_trace_order() -> None: @pytest.mark.asyncio -@pytest.mark.parametrize(("quote", "found"), (("time", True), ("boom", True), ("absent", False))) -async def test_evidence_must_appear_in_the_span_input_output_or_error(quote: str, found: bool) -> None: - storage: Final = FakeStorage(texts={("span", "output"): "timeout", ("span", "error"): "boom"}) +@pytest.mark.parametrize( + ("quote", "found", "input_text"), + ( + ("time", True, "hello"), + ("boom", True, "hello"), + ("Input: hello", True, "hello"), + ("Status: error boom", True, "hello"), + ("hello\nOutput: timeout", True, "hello"), + ("timeout\nStatus: error boom", True, "hello"), + ("Status: ok boom", False, "hello"), + ("Input: absent", False, "hello"), + ("absent", False, "hello"), + ("x" * 1500 + "hello\nOutput: timeout\nStatus: error boom", True, "x" * (2 * BUDGET) + "hello"), + ("x" * 1500 + "hello\nOutput: timeout\nStatus: ok boom", False, "x" * (2 * BUDGET) + "hello"), + ), + ids=( + "raw_output", + "raw_error", + "input_label", + "error_status", + "input_output_boundary", + "output_status_boundary", + "wrong_status", + "missing_input", + "missing_quote", + "long_boundary", + "wrong_long_status", + ), +) +async def test_evidence_must_appear_in_the_span_input_output_or_error(quote: str, found: bool, input_text: str) -> None: + storage: Final = FakeStorage( + spans=(span("span", status="error"),), + texts={("span", "input"): input_text, ("span", "output"): "timeout", ("span", "error"): "boom"}, + ) execution: Final = Execution(id=execution_id(REF, "trace"), trace_id="trace", trace_ref=REF) evidence: Final = Evidence(execution_id=execution.id, span_id="span", quote=quote) - assert await SourceReader(storage).verify_evidence(Scope(all_teams=True), execution, evidence) is found + reader: Final = SourceReader(storage) + if found: + shown: Final = await reader.content( + Scope(all_teams=True), execution, offset=max(len(input_text) - BUDGET // 2, 0) + ) + assert quote in shown.parts[0].content + assert await reader.verify_evidence(Scope(all_teams=True), execution, evidence) is found @pytest.mark.parametrize( diff --git a/tests/unit/proxy/test_tracing_endpoints.py b/tests/unit/proxy/test_tracing_endpoints.py index 8bff4c38400..d9c5e3f2b3a 100644 --- a/tests/unit/proxy/test_tracing_endpoints.py +++ b/tests/unit/proxy/test_tracing_endpoints.py @@ -5,16 +5,20 @@ Tests for the agent tracing endpoints (litellm/proxy/tracing_endpoints.py). from collections.abc import AsyncGenerator, Mapping from contextlib import asynccontextmanager from types import ModuleType -from typing import Final, Literal +from typing import Annotated, Final, Literal from unittest.mock import AsyncMock, MagicMock import pytest -from fastapi import FastAPI, HTTPException +from fastapi import Depends, FastAPI, HTTPException +from fastapi.routing import APIRoute +from fastapi.security import HTTPAuthorizationCredentials, HTTPBearer from fastapi.testclient import TestClient +from jsonschema import validate +from pydantic import JsonValue, TypeAdapter from litellm.constants import TRACE_READ_RETRY_AFTER_SECONDS from litellm.proxy import tracing_endpoints -from litellm.proxy._types import LitellmUserRoles, ProxyLifespanState, UserAPIKeyAuth +from litellm.proxy._types import LitellmUserRoles, ProxyException, ProxyLifespanState, UserAPIKeyAuth from litellm.proxy.auth.authorization import OwnedRows, ReadScope from litellm.proxy.auth.authorization_dependencies import get_log_team_lookup from litellm.proxy.auth.user_api_key_auth import user_api_key_auth @@ -22,6 +26,7 @@ from litellm.proxy.tracing_runtime import manage_tracing, provide_storage from litellm.rust_bridge import loader from litellm.rust_bridge.trace.errors import TraceChanged, TraceQueryError from litellm.rust_bridge.trace.generated.models import TraceQueryHelp +from litellm.rust_bridge.trace.generated.routes import OPERATIONS from litellm.rust_bridge.trace.generated.types import AllQueryScope, OwnedQueryScope, QueryScope from litellm.rust_bridge.trace.queries import TraceSQLResponse from litellm.rust_bridge.trace.storage import NEWEST, ClickHouseStorage, TraceStorageConfig @@ -74,7 +79,12 @@ TRACE_RESPONSE: Final = { "input_preview": "", "start_time": "2026-01-01T00:00:00Z", "duration_ms": 0, - "status": "ok", + "root_status": "ok", + "has_error": False, + "id": "run-one", + "agent_names": [], + "frameworks": [], + "resolution_limited": False, "span_count": 0, "agent_count": 0, "agent_invocations": 0, @@ -89,12 +99,13 @@ TRACE_RESPONSE: Final = { "agents": [], "spans": [], } +TRACE_METADATA: Final = {"summary": TRACE_RESPONSE["summary"], "agents": TRACE_RESPONSE["agents"]} +TRACE_WINDOW: Final = {"start_ms": 1, "end_ms": 2, "as_of_ms": 2} + SPAN_DETAIL_RESPONSE: Final = { "span_id": "s1", "input": "", "output": "", - "input_ui": {"kind": "text", "text": ""}, - "output_ui": {"kind": "text", "text": ""}, "attributes": {}, } @@ -145,7 +156,7 @@ def test_trace_read_and_write_permissions( receiver.list_traces.assert_not_awaited() else: receiver.list_traces.assert_awaited_once_with( - scope=scope, start_ms=1, end_ms=2, q="", cursor=None, order=NEWEST + scope=scope, start_ms=1, end_ms=2, q="", cursor=None, order=NEWEST, page_size=50, as_of_ms=None ) write: Final = client.post("/v1/traces", json={}) @@ -166,11 +177,13 @@ def test_trace_read_and_write_permissions( def receiver(client) -> MagicMock: fake = MagicMock() fake.ingest = AsyncMock(return_value=1) - fake.list_traces = AsyncMock(return_value={"data": [], "next_cursor": None}) - fake.trace_histogram = AsyncMock(return_value={"buckets": []}) - fake.run_values = AsyncMock(return_value={"values": ["researcher"]}) - fake.get_trace = AsyncMock(return_value=None) - fake.get_span = AsyncMock(return_value=None) + fake.list_traces = AsyncMock(return_value={"data": [], "next_cursor": None, "window": TRACE_WINDOW}) + fake.trace_histogram = AsyncMock(return_value={"buckets": [], "window": TRACE_WINDOW}) + fake.run_values = AsyncMock(return_value={"values": ["researcher"], "window": TRACE_WINDOW}) + fake.get_trace_metadata = AsyncMock(return_value=None) + fake.get_span_by_id = AsyncMock(return_value=None) + fake.get_trace_spans = AsyncMock(return_value=None) + fake.get_span_error_by_id = AsyncMock(return_value=None) client.app.dependency_overrides[tracing_endpoints.provide_receiver] = lambda: fake return fake @@ -249,17 +262,19 @@ def test_post_too_large_is_413(client, receiver): def test_list_traces_passes_scope_window_query_and_cursor(client, receiver): response = client.get( - "/v1/traces", params={"start_ms": 1, "end_ms": 2, "q": "agent:research* -status:ok", "cursor": "abc"} + "/v1/traces", params={"start_ms": 1, "end_ms": 2, "q": "agent:research* -root_status:ok", "cursor": "abc"} ) assert response.status_code == 200 - assert response.json() == {"data": [], "next_cursor": None} + assert response.json() == {"data": [], "next_cursor": None, "window": TRACE_WINDOW} receiver.list_traces.assert_awaited_once_with( scope={"kind": "owned", "user_id": "user", "team_ids": ()}, start_ms=1, end_ms=2, - q="agent:research* -status:ok", + q="agent:research* -root_status:ok", cursor="abc", order=NEWEST, + page_size=50, + as_of_ms=None, ) @@ -287,9 +302,9 @@ def test_list_traces_rejects_orders_the_runs_table_does_not_offer(client, receiv def test_histogram_passes_scope_window_query_and_buckets(client, receiver): response = client.get("/v1/traces/histogram", params={"start_ms": 1, "end_ms": 2, "q": "x", "buckets": 12}) assert response.status_code == 200, response.text - assert response.json() == {"buckets": []} + assert response.json() == {"buckets": [], "window": TRACE_WINDOW} receiver.trace_histogram.assert_awaited_once_with( - {"kind": "owned", "user_id": "user", "team_ids": ()}, 1, 2, "x", 12 + {"kind": "owned", "user_id": "user", "team_ids": ()}, 1, 2, "x", 12, None ) @@ -298,9 +313,9 @@ def test_values_pass_scope_window_query_field_and_needle(client, receiver): "/v1/traces/values/agent", params={"start_ms": 1, "end_ms": 2, "q": "model:gpt*", "contains": "res"} ) assert response.status_code == 200, response.text - assert response.json() == {"values": ["researcher"]} + assert response.json() == {"values": ["researcher"], "window": TRACE_WINDOW} receiver.run_values.assert_awaited_once_with( - {"kind": "owned", "user_id": "user", "team_ids": ()}, 1, 2, "model:gpt*", "agent", "res", 20 + {"kind": "owned", "user_id": "user", "team_ids": ()}, 1, 2, "model:gpt*", "agent", "res", 20, None ) @@ -364,30 +379,32 @@ def test_list_traces_preserves_omitted_bounds_for_cursor_window( assert kwargs["q"] == "" -def test_get_trace_404_and_200(client, receiver): +def test_get_trace_metadata_404_and_200(client, receiver): assert client.get("/v1/traces/missing").status_code == 404 - receiver.get_trace.return_value = TRACE_RESPONSE + receiver.get_trace_metadata.return_value = TRACE_METADATA response = client.get("/v1/traces/t1") assert response.status_code == 200 - assert response.json() == TRACE_RESPONSE - receiver.get_trace.assert_awaited_with("t1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "", None, None) + assert response.json() == TRACE_METADATA + receiver.get_trace_metadata.assert_awaited_with({"kind": "owned", "user_id": "user", "team_ids": ()}, "t1") -def test_get_span_404_and_200(client, receiver): +def test_get_span_by_id_404_and_200(client, receiver): assert client.get("/v1/traces/t1/spans/s1").status_code == 404 - receiver.get_span.return_value = SPAN_DETAIL_RESPONSE + receiver.get_span_by_id.return_value = SPAN_DETAIL_RESPONSE response = client.get("/v1/traces/t1/spans/s1") assert response.status_code == 200 assert response.json()["span_id"] == "s1" - receiver.get_span.assert_awaited_with("t1", "s1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "") + receiver.get_span_by_id.assert_awaited_with({"kind": "owned", "user_id": "user", "team_ids": ()}, "t1", "s1") -@pytest.mark.parametrize("suffix,cursor,page_size", [("", None, None), ("&cursor=next&page_size=200", "next", 200)]) -def test_trace_detail_passes_scoped_reference(client, receiver, suffix, cursor, page_size): - receiver.get_trace.return_value = TRACE_RESPONSE - assert client.get(f"/v1/traces/t1?trace_ref=run-one{suffix}").status_code == 200 - receiver.get_trace.assert_awaited_with( - "t1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "run-one", cursor, page_size +@pytest.mark.parametrize("query,cursor,page_size", [("", None, 100), ("?cursor=next&page_size=200", "next", 200)]) +def test_span_page_passes_canonical_id_and_bounded_page_size(client, receiver, query, cursor, page_size): + receiver.get_trace_spans.return_value = {"data": [], "next_cursor": None} + response: Final = client.get(f"/v1/traces/run-one/spans{query}") + assert response.status_code == 200, response.text + assert response.json() == {"data": [], "next_cursor": None} + receiver.get_trace_spans.assert_awaited_with( + {"kind": "owned", "user_id": "user", "team_ids": ()}, "run-one", cursor, page_size ) @@ -395,9 +412,9 @@ def test_trace_detail_passes_scoped_reference(client, receiver, suffix, cursor, "path,method", ( ("/v1/traces", "list_traces"), - ("/v1/traces/t1", "get_trace"), - ("/v1/traces/t1/spans/s1", "get_span"), - ("/v1/traces/t1/spans/s1/error", "get_span_error"), + ("/v1/traces/t1", "get_trace_metadata"), + ("/v1/traces/t1/spans/s1", "get_span_by_id"), + ("/v1/traces/t1/spans/s1/error", "get_span_error_by_id"), ), ) @pytest.mark.parametrize( @@ -437,16 +454,17 @@ def test_read_failures_carry_a_code_per_kind_without_exposing_database_details( getattr(receiver, method).side_effect = error response: Final = client.get(path) assert response.status_code == status - assert response.json() == {"detail": {"code": code, "message": message}} + assert response.headers["content-type"] == "application/problem+json" + assert (response.json()["code"], response.json()["detail"], response.json()["status"]) == (code, message, status) retry_after: Final = response.headers.get("Retry-After") assert (retry_after == str(TRACE_READ_RETRY_AFTER_SECONDS)) == (status == 503), retry_after @pytest.mark.parametrize("query", ("page_size=0", "page_size=501", "cursor=" + "x" * 513)) def test_trace_page_rejects_unbounded_parameters(client: TestClient, receiver: MagicMock, query: str) -> None: - response: Final = client.get(f"/v1/traces/t1?{query}") + response: Final = client.get(f"/v1/traces/t1/spans?{query}") assert response.status_code == 422 - receiver.get_trace.assert_not_awaited() + receiver.get_trace_spans.assert_not_awaited() def test_invalid_export_and_cursor_are_client_errors(client, receiver): @@ -489,9 +507,9 @@ def test_key_without_user_cannot_read_traces(client: TestClient, auth: UserAPIKe storage.list_traces, storage.trace_histogram, storage.run_values, - storage.get_trace, - storage.get_span, - storage.get_span_error, + storage.get_trace_metadata, + storage.get_span_by_id, + storage.get_span_error_by_id, ): read.assert_not_called() storage.query_sql.assert_not_called() @@ -535,9 +553,11 @@ def test_disabled_receiver_precedes_read_scope_rejection(client: TestClient) -> ) response: Final = client.get("/v1/traces") assert response.status_code == 501 - assert response.json() == { - "detail": "Agent tracing is not enabled. Set `tracing:` in general_settings and CLICKHOUSE_URL." - } + assert ( + response.json()["detail"] + == "Agent tracing is not enabled. Set `tracing:` in general_settings and CLICKHOUSE_URL." + ) + assert response.json()["code"] == "unavailable" def test_injected_receiver_ingests_with_the_authenticated_tenant(client: TestClient) -> None: @@ -563,9 +583,9 @@ def test_injected_receiver_ingests_with_the_authenticated_tenant(client: TestCli def test_lifespan_receivers_are_app_local() -> None: first_storage: Final = MagicMock(spec=ClickHouseStorage) - first_storage.get_span = AsyncMock(return_value={**SPAN_DETAIL_RESPONSE, "span_id": "first-span"}) + first_storage.get_span_by_id = AsyncMock(return_value={**SPAN_DETAIL_RESPONSE, "span_id": "first-span"}) second_storage: Final = MagicMock(spec=ClickHouseStorage) - second_storage.get_span = AsyncMock(return_value={**SPAN_DETAIL_RESPONSE, "span_id": "second-span"}) + second_storage.get_span_by_id = AsyncMock(return_value={**SPAN_DETAIL_RESPONSE, "span_id": "second-span"}) first_receiver: Final = TraceReceiver(first_storage) second_receiver: Final = TraceReceiver(second_storage) first_storage.ensure_schema = AsyncMock() @@ -592,9 +612,9 @@ def test_lifespan_receivers_are_app_local() -> None: with TestClient(first_app) as first_client: with TestClient(second_app) as second_client: - second_response: Final = second_client.get("/v1/traces/t1/spans/second-span?trace_ref=second-run") - simultaneous: Final = first_client.get("/v1/traces/t1/spans/first-span?trace_ref=first-run") - first_response: Final = first_client.get("/v1/traces/t1/spans/first-span?trace_ref=first-run") + second_response: Final = second_client.get("/v1/traces/second-run/spans/second-span") + simultaneous: Final = first_client.get("/v1/traces/first-run/spans/first-span") + first_response: Final = first_client.get("/v1/traces/first-run/spans/first-span") assert simultaneous.json() == first_response.json() first_storage.ensure_schema.assert_awaited_once() second_storage.ensure_schema.assert_awaited_once() @@ -603,9 +623,9 @@ def test_lifespan_receivers_are_app_local() -> None: assert first_response.json()["span_id"] == "first-span" assert second_response.json()["span_id"] == "second-span" scope: Final = OwnedQueryScope(kind="owned", user_id=TEAM_KEY.user_id or "", team_ids=()) - assert first_storage.get_span.await_count == 2 - first_storage.get_span.assert_awaited_with("t1", "first-span", scope, "first-run") - second_storage.get_span.assert_awaited_once_with("t1", "second-span", scope, "second-run") + assert first_storage.get_span_by_id.await_count == 2 + first_storage.get_span_by_id.assert_awaited_with(scope, "first-run", "first-span") + second_storage.get_span_by_id.assert_awaited_once_with(scope, "second-run", "second-span") @pytest.mark.parametrize("auth", [TEAM_KEY, UserAPIKeyAuth(user_role=LitellmUserRoles.INTERNAL_USER)]) @@ -613,7 +633,7 @@ def test_query_validation_precedes_trace_access_checks(client: TestClient, auth: client.app.dependency_overrides[user_api_key_auth] = lambda: auth response: Final = client.get("/v1/traces", params={"start_ms": "invalid"}) assert response.status_code == 422 - assert response.json()["detail"][0]["loc"] == ["query", "start_ms"] + assert response.json()["errors"][0]["location"] == "query/start_ms" @pytest.mark.parametrize("enabled", [True, False]) @@ -715,7 +735,7 @@ def test_sql_and_help_use_authenticated_scope( result: Final = client.post("/v1/traces/query", json={"sql": "SELECT * FROM otel_traces"}) assert result.status_code == 200, result.text assert result.json() == SQL_ENVELOPE - receiver.storage.query_sql.assert_awaited_once_with("SELECT * FROM otel_traces", expected_scope, "test-secret") + receiver.storage.query_sql.assert_awaited_once_with("SELECT * FROM otel_traces", expected_scope, "test-secret", {}) help_result: Final = client.get("/v1/traces/query/help") assert help_result.status_code == 200, help_result.text assert help_result.json() == QUERY_HELP @@ -725,6 +745,114 @@ def test_sql_and_help_use_authenticated_scope( assert receiver.storage.query_sql.await_count == 1 +@pytest.mark.parametrize( + "path", + ( + "/v1/traces", + "/v1/traces/histogram", + "/v1/traces/values/agent", + "/v1/traces/run-one", + "/v1/traces/run-one/spans", + "/v1/traces/run-one/spans/s1", + "/v1/traces/run-one/spans/s1/error", + "/v1/traces/query/help", + ), +) +def test_trace_reads_reject_unknown_query_fields(client: TestClient, receiver: MagicMock, path: str) -> None: + client.app.dependency_overrides[tracing_endpoints.provide_trace_query_secret] = lambda: "test-secret" + response: Final = client.get(path, params={"trace_ref": "forged"}) + assert response.status_code == 422, response.text + assert response.headers["content-type"] == "application/problem+json" + assert response.json()["code"] == "invalid_request" + assert response.json()["errors"][0]["location"] == "query/trace_ref" + receiver.get_trace_metadata.assert_not_awaited() + receiver.list_traces.assert_not_awaited() + + +def test_merged_openapi_validates_responses_and_preserves_other_authenticated_routes( + client: TestClient, receiver: MagicMock +) -> None: + security: Final = HTTPBearer(scheme_name="OtherBearer") + + @client.app.get("/other", response_model=str) + def other(credentials: Annotated[HTTPAuthorizationCredentials, Depends(security)]) -> str: + return credentials.credentials + + actual: Final = frozenset( + (route.path, frozenset(route.methods), route.operation_id) + for route in client.app.routes + if isinstance(route, APIRoute) and route.operation_id in OPERATIONS + ) + expected: Final = frozenset( + (operation.path, frozenset((operation.method,)), operation.operation_id) for operation in OPERATIONS.values() + ) + assert actual == expected + client.app.openapi_schema = tracing_endpoints.merge_trace_openapi(client.app.openapi()) + object_adapter: Final = TypeAdapter(dict[str, JsonValue]) + document: Final = object_adapter.validate_python(client.get("/openapi.json").json()) + paths: Final = object_adapter.validate_python(document["paths"]) + components: Final = object_adapter.validate_python(document["components"]) + schemes: Final = object_adapter.validate_python(components["securitySchemes"]) + assert {"OtherBearer", "TraceBearer"} <= schemes.keys() + assert client.get("/other", headers={"Authorization": "Bearer visible"}).json() == "visible" + assert client.get("/other").status_code in (401, 403) + receiver.get_trace_metadata.return_value = TRACE_METADATA + response: Final = client.get("/v1/traces/run-one") + assert response.status_code == 200, response.text + operation: Final = object_adapter.validate_python(object_adapter.validate_python(paths["/v1/traces/{id}"])["get"]) + success: Final = object_adapter.validate_python(object_adapter.validate_python(operation["responses"])["200"]) + content: Final = object_adapter.validate_python(success["content"]) + response_schema: Final = object_adapter.validate_python( + object_adapter.validate_python(content["application/json"])["schema"] + ) + validate(object_adapter.validate_python(response.json()), {**response_schema, "components": components}) + assert response.json()["summary"]["id"] == "run-one" + rejected: Final = client.get("/v1/traces/run-one?unsupported=value") + failure: Final = object_adapter.validate_python(object_adapter.validate_python(operation["responses"])["422"]) + failure_content: Final = object_adapter.validate_python(failure["content"]) + problem_schema: Final = object_adapter.validate_python( + object_adapter.validate_python(failure_content["application/problem+json"])["schema"] + ) + validate(object_adapter.validate_python(rejected.json()), {**problem_schema, "components": components}) + assert rejected.status_code == 422 + + +@pytest.mark.parametrize("framework_error", (False, True)) +def test_trace_auth_failures_are_problem_responses(client: TestClient, framework_error: bool) -> None: + def unauthorized() -> UserAPIKeyAuth: + if framework_error: + raise HTTPException(401, "Invalid token", headers={"WWW-Authenticate": "Bearer"}) + raise ProxyException("Invalid token", "authentication_error", None, 401, {"WWW-Authenticate": "Bearer"}) + + client.app.dependency_overrides[user_api_key_auth] = unauthorized + response: Final = client.get("/v1/traces") + assert response.status_code == 401, response.text + assert response.headers["content-type"] == "application/problem+json" + assert response.headers["www-authenticate"] == "Bearer" + assert response.json()["code"] == "unauthorized" + assert response.json()["detail"] == "Invalid token" + + +def test_sql_parameters_are_typed_and_unknown_query_fields_are_rejected( + client: TestClient, receiver: MagicMock +) -> None: + client.app.dependency_overrides[tracing_endpoints.provide_trace_query_secret] = lambda: "test-secret" + receiver.storage.query_sql = AsyncMock(return_value=TraceSQLResponse.model_validate(SQL_ENVELOPE)) + params: Final = {"service": "quoted ' label", "count": 3, "ids": ["a", "b"], "active": True, "empty": None} + query: Final = "SELECT {service:String}, {count:UInt64}, {ids:Array(String)}, {active:Bool}" + response: Final = client.post("/v1/traces/query", json={"sql": query, "params": params}) + assert response.status_code == 200, response.text + assert response.json() == SQL_ENVELOPE + receiver.storage.query_sql.assert_awaited_once_with( + query, OwnedQueryScope(kind="owned", user_id="user", team_ids=()), "test-secret", params + ) + invalid: Final = client.post("/v1/traces/query", json={"sql": query, "params": {"ids": {"nested": "value"}}}) + assert invalid.status_code == 422, invalid.text + unknown: Final = client.post("/v1/traces/query?database=forged", json={"sql": query}) + assert unknown.status_code == 422, unknown.text + assert receiver.storage.query_sql.await_count == 1 + + @pytest.mark.parametrize("auth", (UserAPIKeyAuth(), UserAPIKeyAuth(team_id="a", project_id="p"))) def test_sql_rejects_missing_identity_without_querying( client: TestClient, receiver: MagicMock, auth: UserAPIKeyAuth @@ -789,9 +917,12 @@ def test_sql_reports_rejected_queries_and_unavailable_readers( receiver.storage.query_sql = AsyncMock(side_effect=error) result: Final = client.post("/v1/traces/query", json={"sql": "SELECT 1"}) assert result.status_code == status, result.text - assert result.json()["detail"] == detail + assert result.headers["content-type"] == "application/problem+json" + assert result.json()["code"] == detail["code"] + assert result.json()["detail"] == detail["message"] + assert result.json().get("database_code") == detail["database_code"] receiver.storage.query_sql.assert_awaited_once_with( - "SELECT 1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "test-secret" + "SELECT 1", {"kind": "owned", "user_id": "user", "team_ids": ()}, "test-secret", {} ) @@ -821,7 +952,7 @@ def test_queries_require_a_proxy_secret( return assert result.status_code == 200, result.text receiver.storage.query_sql.assert_awaited_once_with( - "SELECT 1", {"kind": "owned", "user_id": "user", "team_ids": ()}, secret + "SELECT 1", {"kind": "owned", "user_id": "user", "team_ids": ()}, secret, {} ) @@ -847,7 +978,7 @@ def test_shared_trace_permissions_reach_read_and_sql_boundaries( team_lookup: Final = AsyncMock(side_effect=lookup) storage: Final = MagicMock(spec=ClickHouseStorage) - storage.get_span = AsyncMock(return_value=SPAN_DETAIL_RESPONSE) + storage.get_span_by_id = AsyncMock(return_value=SPAN_DETAIL_RESPONSE) storage.query_sql = AsyncMock(return_value=TraceSQLResponse.model_validate(SQL_ENVELOPE)) storage.query_help = AsyncMock(return_value=TraceQueryHelp.model_validate(QUERY_HELP)) client.app.dependency_overrides[user_api_key_auth] = lambda: auth @@ -855,7 +986,7 @@ def test_shared_trace_permissions_reach_read_and_sql_boundaries( client.app.dependency_overrides[tracing_endpoints.provide_receiver] = lambda: TraceReceiver(storage) client.app.dependency_overrides[tracing_endpoints.provide_trace_query_secret] = lambda: "test-secret" - response: Final = client.get("/v1/traces/t1/spans/s1?trace_ref=run-one") + response: Final = client.get("/v1/traces/run-one/spans/s1") assert response.status_code == 200, response.text assert response.json()["span_id"] == "s1" query_scope: Final = ( @@ -867,12 +998,12 @@ def test_shared_trace_permissions_reach_read_and_sql_boundaries( "team_ids": expected[2], } ) - storage.get_span.assert_awaited_once_with("t1", "s1", query_scope, "run-one") + storage.get_span_by_id.assert_awaited_once_with(query_scope, "run-one", "s1") sql_response: Final = client.post("/v1/traces/query", json={"sql": "SELECT * FROM otel_traces"}) assert sql_response.status_code == 200, sql_response.text assert sql_response.json() == SQL_ENVELOPE assert client.get("/v1/traces/query/help").json() == QUERY_HELP - storage.query_sql.assert_awaited_once_with("SELECT * FROM otel_traces", query_scope, "test-secret") + storage.query_sql.assert_awaited_once_with("SELECT * FROM otel_traces", query_scope, "test-secret", {}) storage.query_help.assert_awaited_once_with(query_scope, "test-secret") assert team_lookup.await_count == ( 3 @@ -962,7 +1093,7 @@ async def test_storage_validates_the_native_query_help_value(monkeypatch: pytest "scope": "bounded sample", } }, - {"tables": [{"name": "traces", "columns": [{"name": "value", "type": "String"}]}]}, + {"tables": [{"name": "unknown_table", "columns": [{"name": "value", "type": "String"}]}]}, {"unexpected": True}, ), ) diff --git a/tests/unit/rust_bridge/trace/test_queries.py b/tests/unit/rust_bridge/trace/test_queries.py index fc36f3a7ddb..157cf93d9ad 100644 --- a/tests/unit/rust_bridge/trace/test_queries.py +++ b/tests/unit/rust_bridge/trace/test_queries.py @@ -30,11 +30,9 @@ def test_dictionary_validation_keeps_required_nullable_and_optional_fields_disti "input": "", "output": "", "attributes": {"key": "value"}, - "input_ui": {"kind": "messages", "messages": [{"role": "user", "content": "hello"}]}, - "output_ui": {"kind": "text", "text": "answer"}, } ) - assert result["input_ui"] == {"kind": "messages", "messages": ({"role": "user", "content": "hello"},)} + assert (result["input"], result["output"]) == ("", "") assert result["attributes"] == {"key": "value"} assert ( TypeAdapter(SpanErrorPage).validate_python( diff --git a/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.test.ts b/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.test.ts index ecc378eb4bf..369748c0811 100644 --- a/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.test.ts +++ b/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.test.ts @@ -78,6 +78,7 @@ describe("Lens demo data", () => { expect(trace.summary.span_count).toBe(trace.spans.length); expect(trace.summary.agent_names).toContain(trace.agents[0].name); expect(trace.summary.error_count).toBe(trace.spans.filter((span) => span.status === "error").length); + expect(trace.summary.has_error).toBe(trace.summary.error_count > 0); for (const span of trace.spans) { expect(span.start_offset_ms + span.duration_ms).toBeLessThanOrEqual(trace.summary.duration_ms); } @@ -90,7 +91,8 @@ describe("Lens demo data", () => { expect(run.trace.spans).toHaveLength(362); expect(ids.size).toBe(362); expect(run.trace.summary.error_count).toBe(3); - expect(run.trace.summary.status).toBe("ok"); + expect(run.trace.summary.root_status).toBe("ok"); + expect(run.trace.summary.has_error).toBe(true); for (const span of run.trace.spans) { if (span.parent_span_id) expect(ids.has(span.parent_span_id)).toBe(true); expect(run.details.find((detail) => detail.span_id === span.span_id)).toBeDefined(); diff --git a/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.ts b/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.ts index a73e72eb94a..8f5a8f7749e 100644 --- a/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.ts +++ b/ui/litellm-dashboard/src/components/lens/data/demo/createLensDemo.ts @@ -55,6 +55,7 @@ export function demoHistogram(runs: readonly TraceSummary[], range: TimeWindow, agent: traceAgentNames(run)[0] ?? run.service, })); return { + window: { start_ms: range.startMs, end_ms: range.endMs, as_of_ms: range.endMs }, buckets: Array.from({ length: buckets }, (_, index) => { const hits = placed.filter((run) => run.index === index); const agents = [...new Set(hits.filter((run) => !run.failed).map((run) => run.agent))].sort(); @@ -73,7 +74,7 @@ export function demoHistogram(runs: readonly TraceSummary[], range: TimeWindow, } function demoTracesApi(data: LensDemoData): TracesApi { - const run = (traceId: string) => data.runs.find(({ trace }) => trace.summary.trace_id === traceId); + const run = (traceId: string) => data.runs.find(({ trace }) => trace.summary.id === traceId); const summaries = data.runs.map((item) => item.trace.summary); const matching = (range: TimeWindow, q: string) => filterRuns( @@ -91,6 +92,7 @@ function demoTracesApi(data: LensDemoData): TracesApi { list: async ({ selection, order }) => ({ data: orderRuns(matching(selection.window, selection.q), order), next_cursor: null, + window: { start_ms: selection.window.startMs, end_ms: selection.window.endMs, as_of_ms: selection.window.endMs }, }), histogram: async ({ window, q }, buckets) => demoHistogram(matching(window, q), window, buckets), values: async (field, contains, range) => { diff --git a/ui/litellm-dashboard/src/components/lens/data/demo/fixtures.ts b/ui/litellm-dashboard/src/components/lens/data/demo/fixtures.ts index 666ad56a95f..5d573d9eefe 100644 --- a/ui/litellm-dashboard/src/components/lens/data/demo/fixtures.ts +++ b/ui/litellm-dashboard/src/components/lens/data/demo/fixtures.ts @@ -73,7 +73,10 @@ function makeTrace(scene: Scenario, index: number, now: number) { span_count: toolCount + 2, spend: model.spend, start_time: iso(now - (index + 1) * 35 * 60_000), - status: scene.failed ? "error" : "ok", + root_status: base.status, + has_error: Boolean(scene.failed), + id: demoRef(traceId), + resolution_limited: false, tool_calls: toolCount, trace_id: traceId, }, @@ -251,7 +254,7 @@ export function createLensDemoData(now = Date.now()) { id: executionId(trace.summary.trace_id), trace_id: trace.summary.trace_id, trace_ref: demoRef(trace.summary.trace_id), - summary: { ...trace.summary, trace_ref: demoRef(trace.summary.trace_id) }, + summary: trace.summary, })); const relevant = findings.filter((f) => f.check_id === definition.check); const jobs: Job[] = [0, 1].map((day) => { diff --git a/ui/litellm-dashboard/src/components/lens/data/demo/lensDemoLongTrace.ts b/ui/litellm-dashboard/src/components/lens/data/demo/lensDemoLongTrace.ts index 866f1042b73..601cf759f11 100644 --- a/ui/litellm-dashboard/src/components/lens/data/demo/lensDemoLongTrace.ts +++ b/ui/litellm-dashboard/src/components/lens/data/demo/lensDemoLongTrace.ts @@ -82,6 +82,7 @@ export function withReleaseCases(run: { trace: Trace; details: SpanDetail[] }) { llm_calls: models.length, tool_calls: caseCount, error_count: failedCases.size, + has_error: failedCases.size > 0, input_tokens: models.reduce((sum, span) => sum + span.input_tokens, 0), output_tokens: models.reduce((sum, span) => sum + span.output_tokens, 0), spend: models.reduce((sum, span) => sum + (span.spend ?? 0), 0), diff --git a/ui/litellm-dashboard/src/components/lens/setup/InvestigationSetup.tsx b/ui/litellm-dashboard/src/components/lens/setup/InvestigationSetup.tsx index 475f8cce1bc..09306fb54c7 100644 --- a/ui/litellm-dashboard/src/components/lens/setup/InvestigationSetup.tsx +++ b/ui/litellm-dashboard/src/components/lens/setup/InvestigationSetup.tsx @@ -230,12 +230,7 @@ function SetupEditor({ {(run: TraceRef) => ( - setTrace(null)} - /> + setTrace(null)} /> )} diff --git a/ui/litellm-dashboard/src/components/lens/setup/MatchingActivityPreview.tsx b/ui/litellm-dashboard/src/components/lens/setup/MatchingActivityPreview.tsx index 280b1ec0381..add116d40be 100644 --- a/ui/litellm-dashboard/src/components/lens/setup/MatchingActivityPreview.tsx +++ b/ui/litellm-dashboard/src/components/lens/setup/MatchingActivityPreview.tsx @@ -8,7 +8,7 @@ import { AgentTracesTable, type RunPicks } from "../traces/list/AgentTracesTable import type { TraceSummary } from "../traces/types"; import type { MatchingPreview, PreviewSelection } from "./useMatchingActivity"; -const executionOf = (run: TraceSummary) => `${run.trace_ref}:${run.trace_id}`; +const executionOf = (run: TraceSummary) => `${run.id}:${run.trace_id}`; const runPicks = (ids: PreviewSelection["ids"], toggle: PreviewSelection["toggle"]): RunPicks => ({ isPicked: (run) => ids.includes(executionOf(run)), diff --git a/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/deep_agent_trace.json b/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/deep_agent_trace.json index 810bbb024a0..0f91dc42a18 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/deep_agent_trace.json +++ b/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/deep_agent_trace.json @@ -1,16 +1,22 @@ { "summary": { "trace_id": "4bad42b84e9de3ba46fc870185f8f023", + "id": "4bad42b84e9de3ba46fc870185f8f023", + "agent_names": [], + "frameworks": [], + "resolution_limited": false, + "spend": null, "name": "deep_research_agent", "service": "agent-demo", "input_preview": "Should we store OTEL agent spans in ClickHouse or Postgres at 50k spans/sec?", "start_time": "2026-09-30T04:36:29.377138Z", "duration_ms": 51385.449, - "status": "ok", + "root_status": "ok", "span_count": 126, "agent_count": 2, "llm_calls": 7, "tool_calls": 26, + "has_error": false, "error_count": 0, "input_tokens": 30175, "output_tokens": 2620, diff --git a/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/research_trace.json b/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/research_trace.json index 021755dcc70..6f6eaf79554 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/research_trace.json +++ b/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/research_trace.json @@ -1,16 +1,22 @@ { "summary": { "trace_id": "e309a123963901e74c29cd2d3c86ff9e", + "id": "e309a123963901e74c29cd2d3c86ff9e", + "agent_names": [], + "frameworks": [], + "resolution_limited": false, + "spend": null, "name": "research_lead", "service": "research-agent", "input_preview": "[{\"role\": \"user\", \"content\": \"Should we store OTEL agent spans in ClickHouse or Postgres at 50k spans/sec?\"}]", "start_time": "2026-09-30T06:43:54.291000+00:00", "duration_ms": 40198.10688, - "status": "ok", + "root_status": "ok", "span_count": 216, "agent_count": 3, "llm_calls": 21, "tool_calls": 25, + "has_error": false, "error_count": 0, "input_tokens": 69506, "output_tokens": 2960, diff --git a/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/swarm_trace.json b/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/swarm_trace.json index 83cb6c4ba79..e03b99e887c 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/swarm_trace.json +++ b/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/swarm_trace.json @@ -1,16 +1,22 @@ { "summary": { "trace_id": "0602f23f8fa5d3a2c1a12521739ca866", + "id": "0602f23f8fa5d3a2c1a12521739ca866", + "agent_names": [], + "frameworks": [], + "resolution_limited": false, + "spend": null, "name": "orchestrator", "service": "swarm-orchestrator", "input_preview": "", "start_time": "2026-09-30T06:48:02.550000+00:00", "duration_ms": 17145.690112, - "status": "ok", + "root_status": "ok", "span_count": 207, "agent_count": 4, "llm_calls": 53, "tool_calls": 45, + "has_error": true, "error_count": 23, "input_tokens": 42351, "output_tokens": 4356, diff --git a/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/trace_list.json b/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/trace_list.json index ce0c22434c1..482dbfaf818 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/trace_list.json +++ b/ui/litellm-dashboard/src/components/lens/traces/__fixtures__/trace_list.json @@ -1,17 +1,24 @@ { + "window": { "start_ms": 1790750627373, "end_ms": 1790750634292, "as_of_ms": 1790750634292 }, "data": [ { "trace_id": "e309a123963901e74c29cd2d3c86ff9e", + "id": "e309a123963901e74c29cd2d3c86ff9e", + "agent_names": [], + "frameworks": [], + "resolution_limited": false, + "spend": null, "name": "research_lead", "service": "research-agent", "input_preview": "[{\"role\": \"user\", \"content\": \"Should we store OTEL agent spans in ClickHouse or Postgres at 50k spans/sec?\"}]", "start_time": "2026-09-30T06:43:54.291000+00:00", "duration_ms": 40198.0, - "status": "ok", + "root_status": "ok", "span_count": 216, "agent_count": 6, "llm_calls": 21, "tool_calls": 25, + "has_error": false, "error_count": 0, "input_tokens": 69506, "output_tokens": 2960, @@ -20,16 +27,22 @@ }, { "trace_id": "f78f6df35480060fafadac887e234241", + "id": "f78f6df35480060fafadac887e234241", + "agent_names": [], + "frameworks": [], + "resolution_limited": false, + "spend": null, "name": "support_triage_agent", "service": "research-agent", "input_preview": "[{\"role\": \"user\", \"content\": \"Customer acme-404 says billing is wrong. What plan are they on?\"}]", "start_time": "2026-09-30T06:43:52.928000+00:00", "duration_ms": 1315.0, - "status": "ok", + "root_status": "ok", "span_count": 5, "agent_count": 1, "llm_calls": 1, "tool_calls": 1, + "has_error": true, "error_count": 2, "input_tokens": 659, "output_tokens": 60, @@ -38,16 +51,22 @@ }, { "trace_id": "71498cec128bbea430f01de04b972f36", + "id": "71498cec128bbea430f01de04b972f36", + "agent_names": [], + "frameworks": [], + "resolution_limited": false, + "spend": null, "name": "support_triage_agent", "service": "research-agent", "input_preview": "[{\"role\": \"user\", \"content\": \"Customer acme-42 gets 429s after upgrading. What plan are they on and what should they check?\"}]", "start_time": "2026-09-30T06:43:47.373000+00:00", "duration_ms": 5551.0, - "status": "ok", + "root_status": "ok", "span_count": 9, "agent_count": 1, "llm_calls": 2, "tool_calls": 2, + "has_error": false, "error_count": 0, "input_tokens": 1552, "output_tokens": 202, diff --git a/ui/litellm-dashboard/src/components/lens/traces/api.ts b/ui/litellm-dashboard/src/components/lens/traces/api.ts index d00a27f50c1..7927b437838 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/api.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/api.ts @@ -11,12 +11,13 @@ import { } from "../../networking"; import type { TimeWindow } from "@/components/shared/timeRange/timeRange"; import type { RunOrder } from "./list/runOrder"; -import type { RunField, SpanDetail, SpanErrorPage, Trace, TraceHistogram, TracePage } from "./types"; +import type { RunField, RunValues, SpanDetail, SpanErrorPage, Trace, TraceHistogram, TracePage } from "./types"; /** Which runs: the window and the `q` search the server applies before counting or paging. */ export interface RunSelection { readonly window: TimeWindow; readonly q: string; + readonly asOfMs?: number; } /** Where in the ordered sequence a page starts; `null` is the first page. */ @@ -41,43 +42,39 @@ export interface TracesApi { readonly scope: string; /** False for a fixed snapshot: nothing new arrives, so live tail and tracing setup don't apply. */ readonly live: boolean; - handoff(traceId: string, spanId?: string | null, traceRef?: string): TraceHandoff; + handoff(id: string, spanId?: string | null): TraceHandoff; list(request: RunListRequest): Promise; histogram(selection: RunSelection, buckets: number): Promise; values(field: RunField, contains: string, range: TimeWindow): Promise; anyRecorded(): Promise; - trace(traceId: string, traceRef?: string, cursor?: string | null): Promise; - span(traceId: string, spanId: string, traceRef?: string): Promise; - spanError( - traceId: string, - spanId: string, - options: { readonly traceRef?: string; readonly cursor?: string | null }, - ): Promise; + trace(id: string, cursor?: string | null): Promise; + span(id: string, spanId: string): Promise; + spanError(traceId: string, spanId: string, options: { readonly cursor?: string | null }): Promise; } -/** A one-liner Claude Code / Codex can run to read the trace. */ -export const agentHandoffText = (traceId: string, spanId?: string | null, traceRef?: string): string => { - const url = `${getProxyBaseUrl().replace(/\/$/, "")}/v1/traces/${traceId}?format=md${spanId ? `&span_id=${spanId}` : ""}${traceRef ? `&trace_ref=${traceRef}` : ""}`; - const what = spanId ? "this step of a LiteLLM agent trace" : "this LiteLLM agent trace"; - return `Read ${what} and explain what happened and why it failed:\ncurl -s -H "Authorization: Bearer $LITELLM_API_KEY" "${url}"`; +export const agentHandoffText = (id: string, spanId?: string | null): string => { + const base = `${getProxyBaseUrl().replace(/\/$/, "")}/v1/traces/${encodeURIComponent(id)}`; + const urls = spanId ? [`${base}/spans/${encodeURIComponent(spanId)}`] : [base, `${base}/spans`]; + const commands = urls.map((url) => `curl -s -H "Authorization: Bearer $LITELLM_API_KEY" "${url}"`).join("\n"); + return `Read this LiteLLM agent trace and explain what happened. Follow next_cursor on the spans endpoint to load remaining steps:\n${commands}`; }; export function liveTracesApi(accessToken: string): TracesApi { return { scope: accessToken, live: true, - handoff: (traceId, spanId, traceRef) => ({ - text: agentHandoffText(traceId, spanId, traceRef), + handoff: (traceId, spanId) => ({ + text: agentHandoffText(traceId, spanId), copied: "Command copied", }), list: (request) => agentTraceListCall(accessToken, request), - histogram: ({ window, q }, buckets) => + histogram: ({ window, q, asOfMs }, buckets) => apiClient.get("/v1/traces/histogram", { accessToken, - query: { start_ms: window.startMs, end_ms: window.endMs, q: q || undefined, buckets }, + query: { start_ms: window.startMs, end_ms: window.endMs, as_of_ms: asOfMs, q: q || undefined, buckets }, }), values: async (field, contains, range) => { - const found = await apiClient.get<{ values: string[] }>(`/v1/traces/values/${field}`, { + const found = await apiClient.get(`/v1/traces/values/${field}`, { accessToken, query: { start_ms: range.startMs, end_ms: range.endMs, contains: contains || undefined }, }); @@ -87,8 +84,8 @@ export function liveTracesApi(accessToken: string): TracesApi { const page = await apiClient.get("/v1/traces", { accessToken, query: { start_ms: 0 } }); return page.data.length > 0; }, - trace: (traceId, traceRef, cursor) => agentTraceCall(accessToken, traceId, traceRef, cursor), - span: (traceId, spanId, traceRef) => agentTraceSpanCall(accessToken, traceId, spanId, traceRef), + trace: (id, cursor) => agentTraceCall(accessToken, id, cursor), + span: (id, spanId) => agentTraceSpanCall(accessToken, id, spanId), spanError: (traceId, spanId, options) => agentTraceSpanErrorCall(accessToken, traceId, spanId, options), }; } diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/content/ContentTab.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/content/ContentTab.tsx index 16cebc1c712..d55b3eed6c6 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/content/ContentTab.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/content/ContentTab.tsx @@ -3,7 +3,7 @@ import { useQuery, type UseQueryOptions } from "@tanstack/react-query"; import { useTracesApi } from "../../api"; -import type { Span, SpanDetail, UIContent } from "../../types"; +import type { Span, SpanDetail } from "../../types"; import { parseMessages, prettyPayload } from "../../utils"; import { Payload, TextBody } from "./PayloadBody"; import { payloadView } from "./payload"; @@ -18,31 +18,28 @@ export function useSpanDetail(accessToken: string, traceId: string, spanId: stri const traces = useTracesApi(accessToken); const queryOptions: UseQueryOptions = { queryKey: ["agentTraceSpan", traceId, traceRef, spanId, accessToken], - queryFn: () => traces.span(traceId, spanId as string, traceRef), + queryFn: () => traces.span(traceRef ?? traceId, spanId as string), enabled: spanId !== null, staleTime: Infinity, }; return useQuery(queryOptions); } -const messageCount = (raw: string, content: UIContent | undefined): number | undefined => - content?.kind === "messages" ? content.messages.length : parseMessages(raw)?.length; +const messageCount = (raw: string): number | undefined => parseMessages(raw)?.length; function PayloadSection({ title, raw, - content, span, role, }: { title: string; raw: string; - content: UIContent | undefined; span: Span; role: "input" | "output"; }) { - const view = payloadView(raw, content, span.type === "tool" && role === "output"); - const count = role === "input" ? messageCount(raw, content) : undefined; + const view = payloadView(raw, span.type === "tool" && role === "output"); + const count = role === "input" ? messageCount(raw) : undefined; return (
Loading span…} {detailQuery.isError &&
Could not load span: {detailQuery.error.message}
} - {detail?.input ? ( - - ) : null} - {detail?.output ? ( - - ) : null} + {detail?.input ? : null} + {detail?.output ? : null} {empty && span.status !== "error" && (
No content recorded for this span.
)} diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/content/SpanError.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/content/SpanError.tsx index e327971d74a..a9b7dd6130b 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/content/SpanError.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/content/SpanError.tsx @@ -59,7 +59,7 @@ export function StoredDiagnostic({ accessToken, traceId, traceRef, span }: Diagn const [cursor, setCursor] = useState(null); const queryOptions: UseQueryOptions = { queryKey: ["agentTraceSpanError", traceId, traceRef, span.span_id, accessToken, cursor], - queryFn: () => traces.spanError(traceId, span.span_id, { traceRef, cursor }), + queryFn: () => traces.spanError(traceRef ?? traceId, span.span_id, { cursor }), enabled: opened, staleTime: Infinity, gcTime: 0, diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/content/payload.test.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/content/payload.test.ts index cee2742d1f6..afe7cdd23f9 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/content/payload.test.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/content/payload.test.ts @@ -59,19 +59,9 @@ describe("fieldNode", () => { }); describe("payloadView", () => { - it("renders standard fields as a tree with JSON values unfolded", () => { - const view = payloadView( - "raw", - { - kind: "fields", - fields: [ - { key: "start_event", value: "AgentWorkflowStartEvent()" }, - { key: "tags", value: '{"run_id": "r1"}' }, - ], - }, - false, - ); - expect(view).toEqual({ + it("renders JSON fields as a tree with nested values unfolded", () => { + const raw = JSON.stringify({ start_event: "AgentWorkflowStartEvent()", tags: { run_id: "r1" } }); + expect(payloadView(raw, false)).toEqual({ kind: "fields", entries: [ ["start_event", { kind: "text", text: "AgentWorkflowStartEvent()", format: "code" }], @@ -80,57 +70,42 @@ describe("payloadView", () => { }); }); - it("reads standard text by its format and falls back to the raw text for empty fields", () => { - expect(payloadView("", { kind: "text", text: "An **answer**" }, false)).toEqual({ - kind: "text", - text: "An **answer**", - format: "markdown", - }); - expect(payloadView("raw text", { kind: "fields", fields: [] }, false)).toEqual({ - kind: "text", - text: "raw text", - format: "plain", - }); + it("reads raw text by its format", () => { + expect(payloadView("An **answer**", false)).toEqual({ kind: "text", text: "An **answer**", format: "markdown" }); + expect(payloadView("raw text", false)).toEqual({ kind: "text", text: "raw text", format: "plain" }); }); - it("shows a tool's output as its result whatever shape the store sent", () => { - expect(payloadView("raw", { kind: "messages", messages: [{ role: "tool", content: "denied" }] }, true)).toEqual({ + it("shows a tool's output as its result", () => { + expect(payloadView(JSON.stringify([{ role: "tool", content: "denied" }]), true)).toEqual({ kind: "tool-result", text: "denied", }); - expect(payloadView("raw", { kind: "text", text: "42" }, true)).toEqual({ kind: "tool-result", text: "42" }); - expect(payloadView('{"a": 1}', { kind: "fields", fields: [{ key: "a", value: "1" }] }, true)).toEqual({ - kind: "tool-result", - text: '{"a": 1}', - }); - expect(payloadView("plain output", undefined, true)).toEqual({ kind: "tool-result", text: "plain output" }); + expect(payloadView("42", true)).toEqual({ kind: "tool-result", text: "42" }); + expect(payloadView('{"a":1}', true)).toEqual({ kind: "tool-result", text: '{"a":1}' }); }); it("keeps a tool's message with calls as a conversation", () => { - const view = payloadView( - "raw", - { - kind: "messages", - messages: [{ role: "assistant", content: "", tool_calls: [{ name: "f", arguments: '{"x": 1}' }] }], - }, - true, - ); - expect(view).toEqual({ + const raw = JSON.stringify({ + role: "assistant", + content: "", + tool_calls: [{ function: { name: "f", arguments: '{"x":1}' } }], + }); + expect(payloadView(raw, true)).toEqual({ kind: "messages", messages: [{ role: "assistant", content: "", tool_calls: [{ name: "f", args: { x: 1 } }] }], }); }); - it("classifies raw payloads: messages, objects, lists and text", () => { - expect(payloadView('[{"role":"user","content":"hi"}]', undefined, false)).toEqual({ + it("classifies raw messages, objects, lists and code", () => { + expect(payloadView('[{"role":"user","content":"hi"}]', false)).toEqual({ kind: "messages", messages: [{ role: "user", content: "hi" }], }); - expect(payloadView('{"file": "/tmp/x"}', undefined, false)).toEqual({ + expect(payloadView('{"file":"/tmp/x"}', false)).toEqual({ kind: "fields", entries: [["file", { kind: "text", text: "/tmp/x", format: "plain" }]], }); - expect(payloadView("[1, 2]", undefined, false)).toEqual({ + expect(payloadView("[1,2]", false)).toEqual({ kind: "fields", entries: [ [ @@ -145,8 +120,8 @@ describe("payloadView", () => { ], ], }); - expect(payloadView("{}", undefined, false)).toEqual({ kind: "text", text: "{}", format: "plain" }); - expect(payloadView("StopEvent(result=1)", undefined, false)).toEqual({ + expect(payloadView("{}", false)).toEqual({ kind: "text", text: "{}", format: "plain" }); + expect(payloadView("StopEvent(result=1)", false)).toEqual({ kind: "text", text: "StopEvent(result=1)", format: "code", diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/content/payload.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/content/payload.ts index 0cfda77dba1..107962b3260 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/content/payload.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/content/payload.ts @@ -1,4 +1,4 @@ -import type { TraceMessage, UIContent, UIMessage } from "../../types"; +import type { TraceMessage } from "../../types"; import { parseJson, parseMessages } from "../../utils"; export type TextFormat = "markdown" | "code" | "plain"; @@ -51,40 +51,18 @@ export function fieldNode(value: unknown): FieldNode { export const fieldEntries = (pairs: readonly (readonly [string, string])[]): readonly FieldEntry[] => pairs.map(([key, value]): FieldEntry => [key, fieldNode(value)]); -export const toTraceMessage = (message: UIMessage): TraceMessage => ({ - ...message, - tool_calls: message.tool_calls?.map((call) => ({ - name: call.name, - args: parseJson(call.arguments) ?? call.arguments, - })), -}); - -const singleText = (messages: readonly UIMessage[]): string | null => - messages.length === 1 && !messages[0].tool_calls?.length ? messages[0].content : null; - function textView(text: string): PayloadView { return { kind: "text", text, format: textFormat(text) }; } -function standardView(content: UIContent, raw: string, toolOutput: boolean): PayloadView { - switch (content.kind) { - case "messages": { - const toolText = toolOutput ? singleText(content.messages) : null; - if (toolText !== null) return { kind: "tool-result", text: toolText }; - return { kind: "messages", messages: content.messages.map(toTraceMessage) }; - } - case "fields": - if (toolOutput) return { kind: "tool-result", text: raw }; - if (content.fields.length === 0) return textView(raw); - return { kind: "fields", entries: fieldEntries(content.fields.map((field) => [field.key, field.value])) }; - case "text": - return toolOutput ? { kind: "tool-result", text: content.text } : textView(content.text); - } -} - function rawView(raw: string, toolOutput: boolean): PayloadView { const messages = parseMessages(raw); - if (messages) return { kind: "messages", messages }; + if (messages) { + if (toolOutput && messages.length === 1 && !messages[0].tool_calls?.length) { + return { kind: "tool-result", text: messages[0].content }; + } + return { kind: "messages", messages }; + } if (toolOutput) return { kind: "tool-result", text: raw }; const node = fieldNode(raw); if (node.kind === "object" && node.entries.length > 0) return { kind: "fields", entries: node.entries }; @@ -92,7 +70,6 @@ function rawView(raw: string, toolOutput: boolean): PayloadView { return textView(raw); } -/** How a span's input or output reads best: the standard UI shape when the store sent one, else the raw payload. */ -export function payloadView(raw: string, content: UIContent | undefined, toolOutput: boolean): PayloadView { - return content ? standardView(content, raw, toolOutput) : rawView(raw, toolOutput); +export function payloadView(raw: string, toolOutput: boolean): PayloadView { + return rawView(raw, toolOutput); } diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.tsx index 860cb091d18..ff6c102ae99 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/TraceConversation.tsx @@ -27,11 +27,11 @@ export function TraceConversation({ const [limit, setLimit] = useState(CONVERSATION_PAGE_SIZE); const steps = conversationSteps(trace.spans); const visible = steps.slice(0, limit); - const { trace_id: traceId, trace_ref: traceRef } = trace.summary; + const { id: traceId } = trace.summary; const queries = useQueries({ queries: visible.map((span) => ({ - queryKey: ["agentTraceSpan", traceId, traceRef, span.span_id, accessToken], - queryFn: (): Promise => traces.span(traceId, span.span_id, traceRef), + queryKey: ["agentTraceSpan", traceId, span.span_id, accessToken], + queryFn: (): Promise => traces.span(traceId, span.span_id), staleTime: Infinity, retry: false, })), diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.test.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.test.ts index 2a42089a739..16ab62f8a5f 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.test.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.test.ts @@ -45,27 +45,23 @@ describe("trace conversation", () => { }); }); - it("uses normalized message content and structured tool arguments from the gateway", () => { + it("renders raw message content and structured tool arguments", () => { const model = { ...root, span_id: "model", parent_span_id: "root", type: "llm" } as Span; - const normalized: SpanDetail = { + const raw: SpanDetail = { span_id: "model", - input: "unparsed input", - output: "unparsed output", + input: JSON.stringify([user]), + output: JSON.stringify({ + role: "assistant", + content: "Checking", + tool_calls: [{ function: { name: "lookup", arguments: '{"order":42}' } }], + }), attributes: {}, - input_ui: { kind: "messages", messages: [{ role: "user", content: user.content }] }, - output_ui: { - kind: "messages", - messages: [ - { role: "assistant", content: "Checking", tool_calls: [{ name: "lookup", arguments: '{"order":42}' }] }, - ], - }, }; const details = new Map([ ["root", detail("root", [], [])], - ["model", normalized], + ["model", raw], ]); - const items = buildConversation([root, model], details, true); - expect(items[0].messages).toEqual([ + expect(buildConversation([root, model], details, true)[0].messages).toEqual([ user, { role: "assistant", content: "Checking", tool_calls: [{ name: "lookup", args: { order: 42 } }] }, ]); @@ -105,16 +101,7 @@ describe("trace conversation", () => { const tool = { ...model, span_id: "tool", name: "lookup", type: "tool", start_offset_ms: 2 } as Span; const args = { order: 42, active: true, filters: { tags: ["paid"] }, empty: null, label: "42", text: "true" }; const typedCall = { ...call, tool_calls: [{ name: "lookup", args }] }; - const toolDetail: SpanDetail = { - ...detail("tool", args, "Shipped"), - input_ui: { - kind: "fields", - fields: Object.entries(args).map(([key, value]) => ({ - key, - value: typeof value === "string" ? value : JSON.stringify(value), - })), - }, - }; + const toolDetail = detail("tool", args, "Shipped"); const details = new Map([ ["root", detail("root", [user], [])], ["model", detail("model", [user], [typedCall])], diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.ts b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.ts index 1adbee35570..5adb275df5f 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/conversation/conversation.ts @@ -1,6 +1,5 @@ -import type { Span, SpanDetail, TraceMessage, TraceToolCall, UIContent } from "../../types"; +import type { Span, SpanDetail, TraceMessage, TraceToolCall } from "../../types"; import { isFrameworkSpan, parseJson, parseMessages, prettyPayload } from "../../utils"; -import { toTraceMessage } from "../content/payload"; export const CONVERSATION_PAGE_SIZE = 20; @@ -14,19 +13,10 @@ export function conversationSteps(spans: readonly Span[]): Span[] { .sort((a, b) => a.start_offset_ms - b.start_offset_ms); } -function contentText(value: string, content?: UIContent): string { - if (content?.kind === "text") return content.text; - if (content?.kind === "fields") - return JSON.stringify(Object.fromEntries(content.fields.map((field) => [field.key, field.value])), null, 2); - if (content?.kind === "messages") return content.messages.map((message) => message.content).join("\n"); - return prettyPayload(value); -} - -function messages(value: string, content: UIContent | undefined, role: string): TraceMessage[] { - if (content?.kind === "messages") return content.messages.map(toTraceMessage); - const parsed = !content ? parseMessages(value) : null; +function messages(value: string, role: string): TraceMessage[] { + const parsed = parseMessages(value); if (parsed) return parsed; - const text = contentText(value, content); + const text = prettyPayload(value); return text ? [{ role, content: text }] : []; } @@ -78,11 +68,7 @@ function toolItem( pending: TraceToolCall[], items: ConversationItem[], ): ConversationItem { - const args = - parseJson(detail.input) ?? - (detail.input_ui?.kind === "fields" - ? Object.fromEntries(detail.input_ui.fields.map((field) => [field.key, field.value])) - : detail.input); + const args = parseJson(detail.input) ?? detail.input; const call = { name: span.name, args }; const match = pending.findIndex( (candidate) => @@ -97,7 +83,7 @@ function toolItem( message.tool_calls = message.tool_calls.filter((call) => call !== matched); } } - const result = contentText(detail.output, detail.output_ui); + const result = prettyPayload(detail.output); return { id: span.span_id, span, messages: [], toolCall: call, toolResult: result }; } @@ -196,7 +182,7 @@ export function buildConversation( const key = branch(span); const history = histories.get(key) ?? []; if (event.output) { - const output = messages(detail.output, detail.output_ui, "assistant"); + const output = messages(detail.output, "assistant"); const fresh = withoutForwardedAnswers( span.span_id, newConversationMessages(history, output), @@ -215,8 +201,8 @@ export function buildConversation( histories.set(key, [...history, { role: "tool", name: span.name, content: item.toolResult ?? "" }]); continue; } - const input = messages(detail.input, detail.input_ui, "user"); - const output = messages(detail.output, detail.output_ui, "assistant"); + const input = messages(detail.input, "user"); + const output = messages(detail.output, "assistant"); const fresh = newConversationMessages(history, input); if (span.type === "agent" || span.parent_span_id === null) { histories.set(key, input); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunHeader.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunHeader.tsx index 4eac48e9d36..228edd20a99 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunHeader.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunHeader.tsx @@ -80,7 +80,7 @@ interface RunHeaderProps { /** Run identity, view switch and totals in two tight rows. */ export function RunHeader({ trace, handoff, onBack, embedded }: RunHeaderProps) { const { summary } = trace; - const failed = summary.status === "error"; + const failed = summary.root_status === "error"; return (
diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.test.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.test.tsx index d26d2048a58..77d8e555106 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.test.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.test.tsx @@ -257,7 +257,7 @@ describe("RunView", () => { expect(screen.getAllByRole("treeitem")).toHaveLength(2); expect(screen.getByRole("banner")).toHaveTextContent(before ?? ""); expect(screen.queryByRole("button", { name: "Load more steps" })).not.toBeInTheDocument(); - expect(vi.mocked(agentTraceCall).mock.calls.map((call) => call[3])).toEqual([null, "next-page", "next-page"]); + expect(vi.mocked(agentTraceCall).mock.calls.map((call) => call[2])).toEqual([null, "next-page", "next-page"]); }); it("refreshes a failed later page from one new snapshot", async () => { @@ -303,7 +303,7 @@ describe("RunView", () => { await user.click(screen.getByRole("button", { name: "Load more steps" })); expect(await screen.findByText("fresh-tool")).toBeVisible(); expect(screen.getAllByRole("treeitem")).toHaveLength(2); - expect(vi.mocked(agentTraceCall).mock.calls.map((call) => call[3])).toEqual([ + expect(vi.mocked(agentTraceCall).mock.calls.map((call) => call[2])).toEqual([ null, "old-second", "old-third", @@ -338,7 +338,7 @@ describe("RunView", () => { expect(await screen.findByText("later-page-tool")).toBeVisible(); expect(screen.getAllByRole("treeitem")).toHaveLength(2); expect(screen.queryByRole("button", { name: "Refresh trace" })).not.toBeInTheDocument(); - expect(vi.mocked(agentTraceCall).mock.calls.map((call) => call[3])).toEqual([null, "next-page", "next-page"]); + expect(vi.mocked(agentTraceCall).mock.calls.map((call) => call[2])).toEqual([null, "next-page", "next-page"]); }); it("offers no retry for a page that is too large, only a refresh", async () => { @@ -455,7 +455,7 @@ describe("RunView", () => { }); it("distinguishes a completed run with recovered step errors from a failed run", async () => { - renderRun({ ...research, summary: { ...research.summary, status: "ok", error_count: 2 } }); + renderRun({ ...research, summary: { ...research.summary, root_status: "ok", has_error: true, error_count: 2 } }); const header = await screen.findByRole("banner"); expect(header).toHaveTextContent("Completed"); expect(header).toHaveTextContent("Step errors 2"); @@ -491,9 +491,9 @@ describe("RunView", () => { renderRun(research); await user.click(await screen.findByRole("button", { name: /copy for agent/i })); - expect(copyToClipboard).toHaveBeenCalledWith(agentHandoffText(research.summary.trace_id), "Command copied"); - expect(agentHandoffText("t1")).toContain('"http://proxy.test/v1/traces/t1?format=md"'); - expect(agentHandoffText("t1", "s1")).toContain("&span_id=s1"); + expect(copyToClipboard).toHaveBeenCalledWith(agentHandoffText(research.summary.id), "Command copied"); + expect(agentHandoffText("t1")).toContain('"http://proxy.test/v1/traces/t1"'); + expect(agentHandoffText("t1", "s1")).toContain("/t1/spans/s1"); }); }); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.tsx index 78fb40d2f06..ac093b2387b 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/run/RunView.tsx @@ -60,7 +60,7 @@ export function RunView(props: RunViewProps) { const traceId = useDeferredValue(props.traceId); const traceRef = useDeferredValue(props.traceRef); const switching = traceId !== props.traceId || traceRef !== props.traceRef; - const shownKey = traceKey({ traceId, traceRef }); + const shownKey = traceKey({ traceId: traceRef ?? traceId }); return ( {({ reset }) => ( @@ -94,7 +94,7 @@ function LoadedRun({ const queryKey = ["agentTrace", traceId, traceRef, accessToken]; const traceQueryOptions = { queryKey, - queryFn: ({ pageParam }: { pageParam: string | null }) => traces.trace(traceId, traceRef, pageParam), + queryFn: ({ pageParam }: { pageParam: string | null }) => traces.trace(traceRef ?? traceId, pageParam), initialPageParam: null as string | null, getNextPageParam: (lastPage: Trace) => lastPage.next_cursor ?? undefined, staleTime: 30_000, @@ -131,12 +131,7 @@ function LoadedRun({ inert={switching} data-testid="run-view" > - + {(traceQuery.hasNextPage || failure) && ( = { const standardDetail: SpanDetail = { span_id: "llm1", - input: "raw input left unparsed", - output: "raw output left unparsed", - input_ui: { kind: "fields", fields: [{ key: "ticket_id", value: "T-981" }] }, - output_ui: { - kind: "messages", - messages: [ - { - role: "assistant", - content: "Refund approved for T-981.", - name: null, - tool_calls: [{ name: "issue_refund", arguments: '{"amount_usd": 40}' }], - }, - ], - }, + input: JSON.stringify({ ticket_id: "T-981" }), + output: JSON.stringify({ + role: "assistant", + content: "Refund approved for T-981.", + tool_calls: [{ function: { name: "issue_refund", arguments: '{"amount_usd": 40}' } }], + }), attributes: {}, }; const textDetail: SpanDetail = { span_id: "llm1", input: "", - output: '{"answer": "all done"}', - output_ui: { kind: "text", text: "all done" }, + output: "all done", attributes: {}, }; const workflowDetail: SpanDetail = { span_id: "root", - input: '{"init_state": {"config": {"timeout": null}}}', - input_ui: { - kind: "fields", - fields: [ - { key: "init_state", value: '{"config": {"timeout": null, "workers": 4}}' }, - { key: "start_event", value: "AgentWorkflowStartEvent()" }, - ], - }, - output: "StopEvent(result='done')", - output_ui: { kind: "text", text: "An **agent trace** records each step" }, + input: JSON.stringify({ + init_state: { config: { timeout: null, workers: 4 } }, + start_event: "AgentWorkflowStartEvent()", + }), + output: "An **agent trace** records each step", attributes: {}, }; const langchainDetail: SpanDetail = { span_id: "root", input: "", - output: '{"messages": []}', - output_ui: { - kind: "fields", - fields: [ - { - key: "messages", - value: JSON.stringify([ - { type: "human", data: { content: "What is an agent trace?" } }, - { type: "ai", data: { content: "A record of every step an agent took." } }, - ]), - }, + output: JSON.stringify({ + messages: [ + { type: "human", data: { content: "What is an agent trace?" } }, + { type: "ai", data: { content: "A record of every step an agent took." } }, ], - }, + }), attributes: {}, }; const failedToolMessageDetail: SpanDetail = { span_id: "tool1", input: '{"customer_id":"acme-404"}', - output: "raw tool output", - output_ui: { kind: "messages", messages: [{ role: "tool", content: "permission denied: /etc/shadow" }] }, + output: JSON.stringify([{ role: "tool", content: "permission denied: /etc/shadow" }]), attributes: {}, }; @@ -210,7 +194,7 @@ describe("DetailPane", () => { expect(screen.getByRole("tab", { name: "Attributes" })).toBeInTheDocument(); expect(await screen.findByText("Customer acme-404 says billing is wrong.")).toBeVisible(); expect(screen.getAllByText("get_customer_plan").length).toBeGreaterThan(0); - expect(vi.mocked(agentTraceSpanCall)).toHaveBeenCalledWith("sk-test", "t1", "llm1", undefined); + expect(vi.mocked(agentTraceSpanCall)).toHaveBeenCalledWith("sk-test", "t1", "llm1"); }); it("keeps the output visible when a step contains a long input conversation", async () => { @@ -222,7 +206,6 @@ describe("DetailPane", () => { vi.mocked(agentTraceSpanCall).mockResolvedValue({ ...details.root, input: JSON.stringify(messages), - input_ui: { kind: "messages", messages }, }); renderPane(spanRow(root)); const input = await screen.findByRole("button", { name: "Input 120 messages" }); @@ -285,14 +268,14 @@ describe("DetailPane", () => { expect(pane).toHaveTextContent("TimeoutError('slow')"); }); - it("'Copy step' copies a curl for just this span as Markdown", async () => { + it("'Copy step' copies a curl for just this span", async () => { const user = userEvent.setup(); const writeText = vi.fn().mockResolvedValue(undefined); Object.defineProperty(navigator, "clipboard", { value: { writeText }, configurable: true }); renderPane(spanRow(llm)); await user.click(screen.getByRole("button", { name: "Copy step" })); await waitFor(() => expect(writeText).toHaveBeenCalled()); - expect(writeText.mock.calls[0][0]).toContain("http://proxy.test/v1/traces/t1?format=md&span_id=llm1"); + expect(writeText.mock.calls[0][0]).toContain("http://proxy.test/v1/traces/t1/spans/llm1"); }); it("renders the assistant tool call as a card and expands a long argument on click", async () => { @@ -326,19 +309,17 @@ describe("DetailPane", () => { expect(within(output).getAllByText("get_customer_plan", { ignore: "[inert] *" })).not.toHaveLength(0); }); - it("renders the standard input_ui / output_ui instead of re-parsing the raw payload", async () => { + it("renders fields and assistant tool calls from the raw payload", async () => { vi.mocked(agentTraceSpanCall).mockResolvedValue(standardDetail); renderPane(spanRow(llm)); const input = await screen.findByRole("region", { name: /^Input/ }); expect(input).toHaveTextContent("ticket_id"); expect(input).toHaveTextContent("T-981"); - expect(input).not.toHaveTextContent("raw input left unparsed"); const output = screen.getByRole("region", { name: "Output" }); expect(output).toHaveTextContent("Assistant"); expect(output).toHaveTextContent("Refund approved for T-981."); expect(output).toHaveTextContent("issue_refund"); expect(output).toHaveTextContent("amount_usd"); - expect(output).not.toHaveTextContent("raw output left unparsed"); }); it("keeps the failed-tool styling when a tool's output arrives as a single message", async () => { @@ -350,7 +331,7 @@ describe("DetailPane", () => { expect(output).not.toHaveTextContent("Assistant"); }); - it("shows a text output_ui as its plain text", async () => { + it("shows plain text output", async () => { vi.mocked(agentTraceSpanCall).mockResolvedValue(textDetail); renderPane(spanRow(llm)); const output = await screen.findByRole("region", { name: "Output" }); @@ -358,7 +339,7 @@ describe("DetailPane", () => { expect(output).not.toHaveTextContent("answer"); }); - it("unfolds JSON-encoded field values into a nested tree and shows the raw payload on request", async () => { + it("unfolds JSON fields into a nested tree and shows the raw payload on request", async () => { const user = userEvent.setup(); vi.mocked(agentTraceSpanCall).mockResolvedValue(workflowDetail); renderPane(spanRow(root)); @@ -373,7 +354,7 @@ describe("DetailPane", () => { const output = screen.getByRole("region", { name: "Output" }); expect(within(output).getByText("agent trace", { selector: "strong" })).toBeVisible(); await user.click(within(output).getByRole("radio", { name: "Raw" })); - expect(within(output).getByText("StopEvent(result='done')", { selector: "pre" })).toBeVisible(); + expect(within(output).getByText("An **agent trace** records each step", { selector: "pre" })).toBeVisible(); expect(within(output).queryByText("agent trace", { selector: "strong" })).not.toBeInTheDocument(); expect(within(input).getByText("AgentWorkflowStartEvent()")).toBeVisible(); }); diff --git a/ui/litellm-dashboard/src/components/lens/traces/detail/span/GroupPane.tsx b/ui/litellm-dashboard/src/components/lens/traces/detail/span/GroupPane.tsx index caa1c2d76ec..358e0e69565 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/detail/span/GroupPane.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/detail/span/GroupPane.tsx @@ -33,7 +33,7 @@ export function GroupPane({ const tokens = row.members.reduce((sum, m) => sum + m.input_tokens + m.output_tokens, 0); const firstFailure = row.members.find((m) => m.status === "error" && m.error); const sample = (firstFailure ?? row.members[0]).span_id; - const handoff = useTracesApi(accessToken).handoff(trace.summary.trace_id, sample, trace.summary.trace_ref); + const handoff = useTracesApi(accessToken).handoff(trace.summary.id, sample); return (
- + diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.integration.test.tsx b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.integration.test.tsx index 8c3e0ea557e..44422d36c31 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.integration.test.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.integration.test.tsx @@ -56,6 +56,7 @@ const serve = (data: readonly TraceSummary[]) => { vi.mocked(agentTraceListCall).mockImplementation(async (_token, { selection, order }) => ({ data: orderRuns(matching(selection.window.startMs, selection.window.endMs, selection.q), order), next_cursor: null, + window: { start_ms: selection.window.startMs, end_ms: selection.window.endMs, as_of_ms: selection.window.endMs }, })); vi.mocked(apiClient.get).mockImplementation(async (path: string, options?: { query?: Record }) => { const query = options?.query ?? {}; @@ -112,6 +113,7 @@ describe("AgentTracesSection", () => { setupIntersectionMocking(vi.fn); testQueryClient.clear(); vi.mocked(agentTraceListCall).mockReset(); + vi.mocked(apiClient.get).mockReset(); vi.mocked(apiClient.get).mockResolvedValue({ data: [] }); }); @@ -130,16 +132,16 @@ describe("AgentTracesSection", () => { renderWithProviders( , - { searchParams: "?q=status:error" }, + { searchParams: "?q=has_error:true" }, ); await user.click(await screen.findByRole("button", { name: "Investigate these runs" })); - expect(onInvestigate).toHaveBeenCalledWith({ q: "status:error", lookbackHours: 168 }); + expect(onInvestigate).toHaveBeenCalledWith({ q: "has_error:true", lookbackHours: 168 }); }); it("loads the next page only once the list scrolls near its end, then stops at the last page", async () => { vi.mocked(agentTraceListCall) - .mockResolvedValueOnce({ data: runs.slice(0, 1), next_cursor: "next" }) - .mockResolvedValueOnce({ data: runs.slice(1, 2), next_cursor: null }); + .mockResolvedValueOnce({ data: runs.slice(0, 1), next_cursor: "next", window: (traceList as TracePage).window }) + .mockResolvedValueOnce({ data: runs.slice(1, 2), next_cursor: null, window: (traceList as TracePage).window }); renderSection(); expect(await screen.findByTestId("agent-trace-row")).toBeVisible(); expect(screen.queryByRole("button", { name: "Load more" })).not.toBeInTheDocument(); @@ -155,7 +157,11 @@ describe("AgentTracesSection", () => { it("keeps the original time window and loaded rows when another page fails", async () => { const user = userEvent.setup(); const now = vi.spyOn(Date, "now").mockReturnValue(Date.parse("2026-10-01T00:00Z")); - vi.mocked(agentTraceListCall).mockResolvedValueOnce({ data: runs.slice(0, 1), next_cursor: "next" }); + vi.mocked(agentTraceListCall).mockResolvedValueOnce({ + data: runs.slice(0, 1), + next_cursor: "next", + window: (traceList as TracePage).window, + }); renderSection(); expect(await screen.findByTestId("agent-trace-row")).toBeVisible(); const first = listRequest(0); @@ -166,8 +172,22 @@ describe("AgentTracesSection", () => { expect(screen.getAllByTestId("agent-trace-row")).toHaveLength(1); expect(screen.queryByTestId("runs-placeholder")).not.toBeInTheDocument(); expect(agentTraceListCall).toHaveBeenCalledTimes(2); - expect(listRequest(1)).toEqual({ ...first, page: { cursor: "next" } }); - vi.mocked(agentTraceListCall).mockResolvedValueOnce({ data: runs.slice(1, 2), next_cursor: null }); + const window = (traceList as TracePage).window; + const continued = { + ...first, + selection: { + window: { startMs: window.start_ms, endMs: window.end_ms }, + q: first!.selection.q, + asOfMs: window.as_of_ms, + }, + page: { cursor: "next" }, + }; + expect(listRequest(1)).toEqual(continued); + vi.mocked(agentTraceListCall).mockResolvedValueOnce({ + data: runs.slice(1, 2), + next_cursor: null, + window: (traceList as TracePage).window, + }); await user.click(screen.getByRole("button", { name: "Retry" })); await waitFor(() => expect(screen.getAllByTestId("agent-trace-row")).toHaveLength(2)); expect(screen.queryByRole("alert")).not.toBeInTheDocument(); @@ -230,6 +250,31 @@ describe("AgentTracesSection", () => { expect(screen.getByRole("combobox", { name: "Search runs" })).toBeInTheDocument(); }); + it("uses the list cutoff for the histogram when another run arrives between reads", async () => { + pinNowToFixtures(); + const row = runs[0]; + const late = { ...row, id: "late-arrival", trace_id: "late-arrival" }; + const cutoff = PINNED_DAY.anchorMs + 100; + const window = { + start_ms: PINNED_DAY.anchorMs - 86_400_000 + 1, + end_ms: PINNED_DAY.anchorMs - 1, + as_of_ms: cutoff, + }; + vi.mocked(agentTraceListCall).mockResolvedValue({ data: [row], next_cursor: null, window }); + vi.mocked(apiClient.get).mockImplementation(async (path: string, options?: { query?: Record }) => { + if (path !== "/v1/traces/histogram") return { data: [] }; + const params = options?.query ?? {}; + const visible = Number(params.as_of_ms) === cutoff ? [row] : [row, late]; + const range = { startMs: Number(params.start_ms), endMs: Number(params.end_ms) }; + return { ...demoHistogram(visible, range, Number(params.buckets)), window }; + }); + renderWindowed(); + await screen.findByTestId("agent-trace-row"); + await waitFor(() => expect(bucketRunCounts().reduce((sum, count) => sum + count, 0)).toBe(1)); + const request = vi.mocked(apiClient.get).mock.calls.find(([path]) => path === "/v1/traces/histogram"); + expect(request?.[1]?.query).toMatchObject({ start_ms: window.start_ms, end_ms: window.end_ms, as_of_ms: cutoff }); + }); + it("keeps the trace list available when traces exist outside the current time window", async () => { vi.mocked(agentTraceListCall).mockResolvedValue({ ...(traceList as TracePage), data: [] }); vi.mocked(apiClient.get).mockResolvedValue(traceList); @@ -340,7 +385,7 @@ describe("AgentTracesSection", () => { const failed = rows.find((row) => row.textContent?.includes("acme-404")) as HTMLElement; expect(within(failed).getByLabelText("2 errors")).toBeInTheDocument(); expect(screen.getByRole("columnheader", { name: "Cost" })).toBeInTheDocument(); - expect(within(failed).getByText("—")).toBeInTheDocument(); + expect(within(failed).getAllByRole("cell")[6]).toHaveTextContent("—"); }); it("shows the spend returned for a run", async () => { @@ -421,7 +466,7 @@ describe("AgentTracesSection", () => { expect(agentCell(cliRun)).toHaveTextContent(/^Claude Code$/); expect(within(plainRun).queryByRole("img", { hidden: true })).not.toBeInTheDocument(); expect(within(plainRun).getByTestId("span-icon")).toBeInTheDocument(); - expect(agentCell(plainRun)).toHaveTextContent((runs[2].agent_names ?? [runs[2].service]).join(", ")); + expect(agentCell(plainRun)).toHaveTextContent(runs[2].agent_names.join(", ") || "—"); }); it("opens a run in a side drawer over the list and swaps runs without closing it", async () => { @@ -505,7 +550,7 @@ describe("AgentTracesSection", () => { vi.mocked(agentTraceListCall).mockResolvedValue(traceList as TracePage); const onUrlUpdate = vi.fn(); renderWithProviders(, { - searchParams: "?trace=older-than-the-list&trace_ref=ref-9", + searchParams: "?trace=older-than-the-list", onUrlUpdate, }); const drawer = await screen.findByRole("complementary", { name: "Trace details" }); @@ -515,8 +560,7 @@ describe("AgentTracesSection", () => { expect(rows.every((row) => row.getAttribute("aria-selected") === "false")).toBe(true); fireEvent.click(rows[1]); - await waitFor(() => expect(lastUrl(onUrlUpdate).get("trace")).toBe(runs[1].trace_id)); - expect(lastUrl(onUrlUpdate).get("trace_ref")).toBe(runs[1].trace_ref ?? null); + await waitFor(() => expect(lastUrl(onUrlUpdate).get("trace")).toBe(runs[1].id)); expect(onUrlUpdate.mock.lastCall?.[0].options.history).toBe("push"); fireEvent.keyDown(document.body, { key: "Escape" }); @@ -529,21 +573,21 @@ describe("AgentTracesSection", () => { pinNowToFixtures(); serve(runs); const onUrlUpdate = vi.fn(); - const failed = filterRuns(runs, "status:error"); + const failed = filterRuns(runs, "has_error:true"); renderWithProviders(, { - searchParams: "?q=status:error", + searchParams: "?q=has_error:true", onUrlUpdate, }); expect(await screen.findAllByTestId("agent-trace-row")).toHaveLength(failed.length); expect(failed.length).toBeLessThan(runs.length); const search = screen.getByRole("combobox", { name: "Search runs" }); - expect(search).toHaveTextContent("status:error"); + expect(search).toHaveTextContent("has_error:true"); await user.clear(search); await waitFor(() => expect(screen.getAllByTestId("agent-trace-row")).toHaveLength(runs.length)); - await user.type(search, "-status:error"); + await user.type(search, "-has_error:true"); await waitFor(() => expect(screen.getAllByTestId("agent-trace-row")).toHaveLength(runs.length - failed.length)); - await waitFor(() => expect(lastUrl(onUrlUpdate).get("q")).toBe("-status:error")); + await waitFor(() => expect(lastUrl(onUrlUpdate).get("q")).toBe("-has_error:true")); }); it("plots every loaded run on the timeline", async () => { @@ -570,6 +614,7 @@ describe("AgentTracesSection", () => { fireEvent.pointerUp(area, { clientX: x(to), pointerId: 1 }); }; const rowCount = () => screen.queryAllByTestId("agent-trace-row").length; + await waitFor(() => expect(screen.getByTestId("timeline")).toHaveAttribute("aria-busy", "false")); const withRuns = bucketRunCounts().flatMap((count, i) => (count > 0 ? [i] : [])); const first = withRuns[0]; // The pan below moves a [0, first] bracket to the far right; it must end up clear of every run. @@ -688,6 +733,11 @@ describe("AgentTracesPage", () => { resolve({ data: orderRuns(filterRuns(runs, selection.q), order), next_cursor: null, + window: { + start_ms: selection.window.startMs, + end_ms: selection.window.endMs, + as_of_ms: selection.window.endMs, + }, }); }), ); diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.tsx b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.tsx index ed1eccfa57b..f0b5d016a03 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/list/AgentTracesSection.tsx @@ -1,7 +1,6 @@ "use client"; import { ArrowLeft, ScanSearch } from "lucide-react"; -import moment from "moment"; import { type ComponentProps, useMemo, useState } from "react"; import { RunsToolbar } from "./runSearch/RunsToolbar"; @@ -106,11 +105,16 @@ export function AgentTracesSection({ if (setup.disabledDetail == null) void history.refetch(); }; - // A live range ends "now" (the list query uses Date.now() too); round to the minute so the histogram is stable. - const minuteEndMs = moment().endOf("minute").valueOf(); - const window = useMemo(() => timeWindow(range, minuteEndMs), [range, minuteEndMs]); - const histogram = useTraceHistogram(accessToken, { window, q: query }, isActive); - const shownRange = zoom ?? window; + const [initialTime] = useState(Date.now); + const cutoff = traces.resolvedWindow?.as_of_ms; + const overviewWindow = useMemo(() => timeWindow(range, cutoff ?? initialTime), [range, cutoff, initialTime]); + const resolvedWindow = traces.resolvedWindow + ? { startMs: traces.resolvedWindow.start_ms, endMs: traces.resolvedWindow.end_ms } + : null; + const window = zoom ? overviewWindow : resolvedWindow ?? overviewWindow; + const histogramSelection = { window, q: query, asOfMs: cutoff }; + const histogram = useTraceHistogram(accessToken, histogramSelection, isActive && traces.resolvedWindow !== null); + const shownRange = resolvedWindow ?? zoom ?? window; const runs = traces.traces; const runRefs = useMemo(() => runs.map(traceRefOf), [runs]); @@ -158,7 +162,6 @@ export function AgentTracesSection({ {(shown: TraceRef) => ( openRun(null)} @@ -189,7 +192,7 @@ export function AgentTracesSection({ { }; const firstLine = (text: string): string => text.split("\n")[0] ?? text; -const runKey = (run: TraceSummary): string => run.trace_ref || run.trace_id; +const runKey = (run: TraceSummary): string => run.id; const PREFETCH_MARGIN = "0px 0px 480px 0px"; const PLACEHOLDER_ROWS = [0, 1, 2]; diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/runOrder.test.ts b/ui/litellm-dashboard/src/components/lens/traces/list/runOrder.test.ts index 1ad9b8da139..e47c1ea4c5a 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/runOrder.test.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/runOrder.test.ts @@ -10,7 +10,7 @@ const run = (overrides: Partial): TraceSummary => ({ ...template, const RUN_OVERRIDES: readonly Partial[] = [ { trace_id: "a", - trace_ref: "ref-a", + id: "ref-a", start_time: "2026-09-30T06:00:00Z", duration_ms: 500, span_count: 3, @@ -18,7 +18,7 @@ const RUN_OVERRIDES: readonly Partial[] = [ }, { trace_id: "b", - trace_ref: "ref-b", + id: "ref-b", start_time: "2026-09-30T07:00:00Z", duration_ms: 500, span_count: 9, @@ -26,7 +26,7 @@ const RUN_OVERRIDES: readonly Partial[] = [ }, { trace_id: "c", - trace_ref: "ref-c", + id: "ref-c", start_time: "2026-09-30T05:00:00Z", duration_ms: 50, span_count: 1, diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/runOrder.ts b/ui/litellm-dashboard/src/components/lens/traces/list/runOrder.ts index 5772ad3828f..7230db40c08 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/runOrder.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/runOrder.ts @@ -30,7 +30,7 @@ const compareText = (left: string, right: string): number => { return left < right ? -1 : 1; }; -const reference = (run: TraceSummary): string => run.trace_ref || run.trace_id; +const reference = (run: TraceSummary): string => run.id; /** Runs in `order`, as the server pages them: by the key, then by trace reference the same way. */ export function orderRuns(runs: readonly TraceSummary[], order: RunOrder): TraceSummary[] { diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/RunSearch.test.tsx b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/RunSearch.test.tsx index f4da4b49cdc..2a377b81d70 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/RunSearch.test.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/RunSearch.test.tsx @@ -12,7 +12,7 @@ describe("RunSearch", () => { it("offers the run fields, then the agents seen in the loaded runs", async () => { const user = userEvent.setup(); render(); - expect(screen.getByText("Search runs, or filter like agent:researcher status:error")).toBeVisible(); + expect(screen.getByText("Search runs, or filter like agent:researcher has_error:true")).toBeVisible(); await user.click(box()); expect( within(listbox()) @@ -24,11 +24,19 @@ describe("RunSearch", () => { within(listbox()) .getAllByRole("option") .map((o) => o.textContent), - ).toEqual(["billing-agent", "cron", "researcher", "triage"]); + ).toEqual(["billing-agent", "researcher", "triage"]); await user.click(within(listbox()).getByRole("option", { name: "researcher" })); expect(box()).toHaveTextContent(/^agent:researcher $/, { normalizeWhitespace: false }); }); + it("explains that attribute filters cannot be copied as SQL", async () => { + const user = userEvent.setup(); + render(); + await user.click(box()); + expect(screen.getByRole("note")).toHaveTextContent("Use the traces API for attribute or unsupported field filters"); + expect(screen.queryByRole("button", { name: "Copy as curl" })).not.toBeInTheDocument(); + }); + it("copies the filtered list as a trace query bounded to the shown range", async () => { const user = userEvent.setup(); const range = { startMs: 1_700_000_000_000, endMs: 1_700_003_600_000 }; @@ -40,7 +48,7 @@ describe("RunSearch", () => { expect(command).toContain('/v1/traces/query"'); expect(command).toContain("fromUnixTimestamp64Milli(1700000000000)"); expect(command).toContain("fromUnixTimestamp64Milli(1700003600000)"); - expect(command).toContain("arrayExists(x -> x ILIKE 'res', agents)"); + expect(command).toContain("arrayExists(x -> x ILIKE 'res', agent_names)"); expect(screen.getByRole("button", { name: "Copied" })).toBeVisible(); expect(box()).toHaveAttribute("aria-expanded", "true"); }); diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/RunSearch.tsx b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/RunSearch.tsx index 7730c1871d8..cb4cd3da665 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/RunSearch.tsx +++ b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/RunSearch.tsx @@ -6,12 +6,15 @@ import type { TraceSummary } from "../../types"; import type { TimeWindow } from "@/components/shared/timeRange/timeRange"; import { SearchBox } from "@/components/shared/search/SearchBox"; +import { parseQuery } from "@/components/shared/search/language"; +import { toSearchQuery } from "@/components/shared/search/searchQuery"; import { itemValues } from "@/components/shared/search/valueSource"; import { NEWEST, type RunOrder } from "../runOrder"; import { RUN_INDEX, RUN_QUERY } from "./runQuery"; -import { runQueryCommand } from "./runSql"; +import { runQueryCommand, runQuerySql } from "./runSql"; -const COPY_HINT = "Copy this list as a curl call to the trace query API, which takes full SQL over the same rows."; +const COPY_HINT = + "Query current visible traces with SQL. User scopes can include partial traces excluded by the curated list."; interface RunSearchProps { value: string; @@ -26,6 +29,7 @@ interface RunSearchProps { /** The runs list query box: free text plus `key:value` filters over run fields, copyable as a trace query. */ export function RunSearch({ value, onChange, runs, range, order = NEWEST, busy = false }: RunSearchProps) { + const result = runQuerySql(toSearchQuery(parseQuery(RUN_QUERY, value)), range, order); const command = useMemo(() => runQueryCommand(range, order), [range, order]); return ( - - - + {result.kind === "unsupported" ? ( +
+ {result.reason} +
+ ) : ( + + + + )}
); } diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/__fixtures__/runs.ts b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/__fixtures__/runs.ts index cd886dee349..7fcebfac459 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/__fixtures__/runs.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/__fixtures__/runs.ts @@ -15,7 +15,12 @@ export const run = (overrides: Partial): TraceSummary => ({ span_count: 1, spend: null, start_time: "2026-10-01T00:00:00Z", - status: "ok", + root_status: "ok", + has_error: (overrides.error_count ?? 0) > 0, + id: overrides.trace_id ?? "trace", + agent_names: [], + frameworks: [], + resolution_limited: false, tool_calls: 0, trace_id: "trace", ...overrides, diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runQuery.test.ts b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runQuery.test.ts index d30b8afe382..d028a1d2de6 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runQuery.test.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runQuery.test.ts @@ -15,15 +15,18 @@ describe("filterRuns", () => { expect(ids("gpt-5")).toEqual([]); }); - it("splits runs by status, judged by recorded errors", () => { - expect(ids("status:error")).toEqual(["bbb222"]); - expect(ids("-status:error")).toEqual(["aaa111", "ccc333"]); - expect(ids("status:OK")).toEqual(["aaa111", "ccc333"]); + it("filters recorded errors separately from root status", () => { + expect(ids("has_error:true")).toEqual(["bbb222"]); + expect(ids("-has_error:true")).toEqual(["aaa111", "ccc333"]); + expect(ids("has_error:FALSE")).toEqual(["aaa111", "ccc333"]); + expect(ids("root_status:ok has_error:true")).toEqual(["bbb222"]); + expect(ids("root_status:error")).toEqual([]); }); - it("reads agents from the trace, falling back to the service, and models from the run", () => { + it("reads agent labels, service and models independently", () => { expect(ids("agent:triage")).toEqual(["aaa111"]); - expect(ids("agent:cron")).toEqual(["ccc333"]); + expect(ids("agent:cron")).toEqual([]); + expect(ids("service:cron")).toEqual(["ccc333"]); expect(ids("model:gpt-5")).toEqual(["aaa111"]); expect(ids("name:support")).toEqual(["aaa111"]); }); @@ -34,14 +37,14 @@ describe("filterRuns", () => { }); it("combines field clauses with free text", () => { - expect(ids("agent:*e* -status:error refund")).toEqual(["aaa111"]); + expect(ids("agent:*e* -has_error:true refund")).toEqual(["aaa111"]); }); }); describe("RUN_INDEX values", () => { it("lists the loaded agents and statuses for autocomplete", () => { - expect(fieldValues(RUN_INDEX, runs, "agent")).toEqual(["billing-agent", "cron", "researcher", "triage"]); - expect(fieldValues(RUN_INDEX, runs, "status")).toEqual(["error", "ok"]); + expect(fieldValues(RUN_INDEX, runs, "agent")).toEqual(["billing-agent", "researcher", "triage"]); + expect(fieldValues(RUN_INDEX, runs, "root_status")).toEqual(["ok"]); expect(fieldValues(RUN_INDEX, [run({ models: [] })], "model")).toEqual([]); }); }); diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runQuery.ts b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runQuery.ts index 8d0a074e27d..7f383110125 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runQuery.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runQuery.ts @@ -1,7 +1,7 @@ import { Bot, Box, Braces, CircleDashed, Hash, SquareChevronRight } from "lucide-react"; -import type { TraceSummary } from "../../types"; -import { previewText, traceAgentNames } from "../../utils"; +import type { RunField as TraceRunField, TraceSummary } from "../../types"; +import { traceAgentNames } from "../../utils"; import { type ClientIndex, filterItems } from "@/components/shared/search/evaluate"; import { ALL_OPERATORS, type FieldSpec, type QueryLanguage } from "@/components/shared/search/language"; @@ -9,11 +9,13 @@ import { ALL_OPERATORS, type FieldSpec, type QueryLanguage } from "@/components/ const RUN_FIELDS = { name: { group: "Run attributes", icon: SquareChevronRight, suggestValues: true }, agent: { group: "Run attributes", icon: Bot, suggestValues: true }, - status: { group: "Run attributes", icon: CircleDashed, suggestValues: true }, + root_status: { group: "Run attributes", icon: CircleDashed, suggestValues: true }, + has_error: { group: "Run attributes", icon: CircleDashed, suggestValues: true }, + service: { group: "Run attributes", icon: Box, suggestValues: true }, model: { group: "Run attributes", icon: Box, suggestValues: true }, input: { group: "Content", icon: Braces, suggestValues: false }, trace_id: { group: "Identity", icon: Hash, suggestValues: false }, -} as const satisfies Record; +} as const satisfies Partial>; export type RunField = keyof typeof RUN_FIELDS; @@ -24,12 +26,14 @@ export const RUN_INDEX: ClientIndex = { read: { name: (run) => [run.name], agent: traceAgentNames, - status: (run) => [run.error_count > 0 ? "error" : "ok"], + root_status: (run) => [run.root_status], + has_error: (run) => [String(run.has_error)], + service: (run) => [run.service], model: (run) => run.models, - input: (run) => [previewText(run.input_preview)], + input: (run) => [run.input_preview], trace_id: (run) => [run.trace_id], }, - freeText: (run) => [run.trace_id, previewText(run.input_preview), run.name], + freeText: (run) => [run.trace_id, run.input_preview, run.name], }; export const filterRuns = (runs: TraceSummary[], query: string): TraceSummary[] => diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runSql.test.ts b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runSql.test.ts index 2babd8302e6..7492bba2434 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runSql.test.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runSql.test.ts @@ -8,13 +8,18 @@ import type { RunOrder } from "../runOrder"; const query = (text: string) => toSearchQuery(parseQuery(RUN_QUERY, text)); const predicates = (text: string) => runPredicates(query(text)); +const sqlFor = (text: string, range = RANGE, order?: RunOrder): string => { + const result = runQuerySql(query(text), range, order); + expect(result.kind).toBe("sql"); + return result.kind === "sql" ? result.sql : result.reason; +}; const RANGE = { startMs: 1_700_000_000_000, endMs: 1_700_003_600_000 }; describe("runPredicates", () => { - it("turns each run field into a predicate over the per-trace rollup columns", () => { - expect(predicates("agent:researcher status:error model:gpt-5 name:support trace_id:aaa111")).toEqual([ - "arrayExists(x -> x ILIKE 'researcher', agents)", - "errors > 0", + it("turns each run field into a predicate over the logical trace columns", () => { + expect(predicates("agent:researcher has_error:true model:gpt-5 name:support trace_id:aaa111")).toEqual([ + "arrayExists(x -> x ILIKE 'researcher', agent_names)", + "has_error = true", "arrayExists(x -> x ILIKE 'gpt-5', models)", "name ILIKE 'support'", "trace_id ILIKE 'aaa111'", @@ -24,59 +29,71 @@ describe("runPredicates", () => { it("negates and globs like the list filter", () => { expect(predicates("-model:gpt* input:*vector*")).toEqual([ "NOT (arrayExists(x -> x ILIKE 'gpt%', models))", - "input ILIKE '%vector%'", + "input_preview ILIKE '%vector%'", ]); }); it("keeps LIKE metacharacters and quotes literal", () => { expect(predicates(String.raw`name:"50%_off's"`)).toEqual([String.raw`name ILIKE '50\\%\\_off\'s'`]); - expect(predicates(String.raw`input:back\slash`)).toEqual([String.raw`input ILIKE 'back\\\\slash'`]); + expect(predicates(String.raw`input:back\slash`)).toEqual([String.raw`input_preview ILIKE 'back\\\\slash'`]); }); - it("resolves a status glob statically, since status has two values", () => { - expect(predicates("status:ok")).toEqual(["errors = 0"]); - expect(predicates("status:*")).toEqual(["true"]); - expect(predicates("-status:pending")).toEqual(["NOT (false)"]); + it("distinguishes root status from the presence of any error", () => { + expect(predicates("root_status:ok has_error:true")).toEqual(["root_status ILIKE 'ok'", "has_error = true"]); + expect(predicates("-has_error:true root_status:unset")).toEqual([ + "NOT (has_error = true)", + "root_status ILIKE 'unset'", + ]); }); it("searches free text across trace id, input and name, and skips a key without a value", () => { expect(predicates("agent: refund")).toEqual([ - "(trace_id ILIKE '%refund%' OR input ILIKE '%refund%' OR name ILIKE '%refund%')", + "(trace_id ILIKE '%refund%' OR input_preview ILIKE '%refund%' OR name ILIKE '%refund%')", + ]); + expect(predicates("a*b")).toEqual([ + "(trace_id ILIKE '%a*b%' OR input_preview ILIKE '%a*b%' OR name ILIKE '%a*b%')", ]); - expect(predicates("a*b")).toEqual(["(trace_id ILIKE '%a*b%' OR input ILIKE '%a*b%' OR name ILIKE '%a*b%')"]); }); }); describe("runQuerySql", () => { - it("groups the rollup by trace, bounds the window the list shows and applies the filters", () => { - const sql = runQuerySql(query("agent:researcher status:error"), RANGE); + it("queries canonical logical traces within the displayed time window", () => { + const sql = sqlFor("agent:researcher has_error:true"); expect(sql).toBe( [ - "SELECT TraceId AS trace_id, any(RootName) AS name, any(RootInput) AS input, sum(ErrorCount) AS errors,", - " groupUniqArrayArray(AgentNames) AS agents, groupUniqArrayArray(Models) AS models,", - " sum(SpanCount) AS steps, dateDiff('millisecond', min(StartTs), max(EndTs)) AS duration_ms", - "FROM agent_traces_by_key", - "GROUP BY TraceId", - `HAVING min(StartTs) >= fromUnixTimestamp64Milli(${RANGE.startMs}) AND min(StartTs) < fromUnixTimestamp64Milli(${RANGE.endMs})`, - " AND arrayExists(x -> x ILIKE 'researcher', agents)", - " AND errors > 0", - "ORDER BY min(StartTs) DESC, trace_id DESC", + "SELECT id, trace_id, name, input_preview, root_status, has_error,", + " agent_names, models, span_count, error_count, start_time, duration_ms", + "FROM traces", + `WHERE start_time >= fromUnixTimestamp64Milli(${RANGE.startMs}) AND start_time < fromUnixTimestamp64Milli(${RANGE.endMs})`, + " AND arrayExists(x -> x ILIKE 'researcher', agent_names)", + " AND has_error = true", + "ORDER BY start_time DESC, id DESC", "LIMIT 100", ].join("\n"), ); }); + it.each(["attr.tenant:demo", "unknown:value", "root_status:pending", "root_status:*", "has_error:error"])( + "does not copy unsupported query %s as SQL", + (text) => { + const result = runQuerySql(query(text), RANGE); + expect(result.kind).toBe("unsupported"); + expect(result.kind === "unsupported" && result.reason).toBeTruthy(); + }, + ); + it("falls back to the last day without a range", () => { - expect(runQuerySql(query(""))).toContain("HAVING min(StartTs) >= now() - INTERVAL 1 DAY\nORDER BY"); + const result = runQuerySql(query("")); + expect(result.kind === "sql" && result.sql).toContain("WHERE start_time >= now() - INTERVAL 1 DAY\nORDER BY"); }); it.each<{ order: RunOrder; clause: string }>([ - { order: { key: "duration_ms", descending: false }, clause: "ORDER BY duration_ms ASC, trace_id ASC" }, - { order: { key: "span_count", descending: true }, clause: "ORDER BY steps DESC, trace_id DESC" }, - { order: { key: "error_count", descending: true }, clause: "ORDER BY errors DESC, trace_id DESC" }, - { order: { key: "start_ms", descending: false }, clause: "ORDER BY min(StartTs) ASC, trace_id ASC" }, + { order: { key: "duration_ms", descending: false }, clause: "ORDER BY duration_ms ASC, id ASC" }, + { order: { key: "span_count", descending: true }, clause: "ORDER BY span_count DESC, id DESC" }, + { order: { key: "error_count", descending: true }, clause: "ORDER BY error_count DESC, id DESC" }, + { order: { key: "start_ms", descending: false }, clause: "ORDER BY start_time ASC, id ASC" }, ])("orders the copied rows like the list, $clause", ({ order, clause }) => { - expect(runQuerySql(query(""), RANGE, order)).toContain(`\n${clause}\nLIMIT 100`); + expect(sqlFor("", RANGE, order)).toContain(`\n${clause}\nLIMIT 100`); expect(runQueryCommand(RANGE, order)(query(""))).toContain(clause); }); }); @@ -96,9 +113,9 @@ describe("traceQueryCommand", () => { describe("runQueryCommand", () => { it("wraps the bounded, filtered query in the trace query call", () => { - const command = runQueryCommand(RANGE)(query("agent:researcher status:error")); - expect(command).toBe(traceQueryCommand(runQuerySql(query("agent:researcher status:error"), RANGE))); - expect(command).toContain("arrayExists(x -> x ILIKE 'researcher', agents)"); - expect(command).toContain("errors > 0"); + const command = runQueryCommand(RANGE)(query("agent:researcher has_error:true")); + expect(command).toBe(traceQueryCommand(sqlFor("agent:researcher has_error:true"))); + expect(command).toContain("arrayExists(x -> x ILIKE 'researcher', agent_names)"); + expect(command).toContain("has_error = true"); }); }); diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runSql.ts b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runSql.ts index f4daa04a2b6..9dd90c7c391 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runSql.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/runSearch/runSql.ts @@ -1,28 +1,26 @@ import { getProxyBaseUrl } from "@/components/networking"; import type { TimeWindow } from "@/components/shared/timeRange/timeRange"; -import { isNegatedOp, valueMatcher } from "@/components/shared/search/language"; +import { isNegatedOp } from "@/components/shared/search/language"; import type { SearchFilter, SearchQuery } from "@/components/shared/search/searchQuery"; import { NEWEST, type RunOrder, type RunSortKey } from "../runOrder"; import type { RunField } from "./runQuery"; -const RUN_ROWS = `SELECT TraceId AS trace_id, any(RootName) AS name, any(RootInput) AS input, sum(ErrorCount) AS errors, - groupUniqArrayArray(AgentNames) AS agents, groupUniqArrayArray(Models) AS models, - sum(SpanCount) AS steps, dateDiff('millisecond', min(StartTs), max(EndTs)) AS duration_ms -FROM agent_traces_by_key -GROUP BY TraceId`; +const RUN_ROWS = `SELECT id, trace_id, name, input_preview, root_status, has_error, + agent_names, models, span_count, error_count, start_time, duration_ms +FROM traces`; const ORDER_COLUMNS: Record = { - start_ms: "min(StartTs)", + start_ms: "start_time", duration_ms: "duration_ms", - span_count: "steps", - error_count: "errors", + span_count: "span_count", + error_count: "error_count", }; /** The list's order with the same tie-break the server pages by. */ const orderBy = (order: RunOrder): string => { const direction = order.descending ? "DESC" : "ASC"; - return `ORDER BY ${ORDER_COLUMNS[order.key]} ${direction}, trace_id ${direction}`; + return `ORDER BY ${ORDER_COLUMNS[order.key]} ${direction}, id ${direction}`; }; const sqlString = (value: string): string => `'${value.replaceAll("\\", "\\\\").replaceAll("'", "\\'")}'`; @@ -35,25 +33,18 @@ const likePattern = (value: string): string => likeLiteral(value).replaceAll("*" const matches = (column: string, pattern: string): string => `${column} ILIKE ${sqlString(pattern)}`; const anyMatches = (column: string, pattern: string): string => `arrayExists(x -> ${matches("x", pattern)}, ${column})`; -const STATUS_PREDICATES = { error: "errors > 0", ok: "errors = 0" } as const; - -/** Status has two values, so a glob over it resolves statically to one, both or neither predicate. */ -function statusPredicate(value: string): string { - const hits = (["error", "ok"] as const).filter(valueMatcher(value)).map((status) => STATUS_PREDICATES[status]); - if (hits.length === 2) return "true"; - return hits[0] ?? "false"; -} - const FIELD_PREDICATES: Record string> = { name: (value) => matches("name", likePattern(value)), - agent: (value) => anyMatches("agents", likePattern(value)), - status: statusPredicate, + agent: (value) => anyMatches("agent_names", likePattern(value)), + root_status: (value) => matches("root_status", likePattern(value)), + has_error: (value) => `has_error = ${value.toLowerCase()}`, + service: (value) => matches("service", likePattern(value)), model: (value) => anyMatches("models", likePattern(value)), - input: (value) => matches("input", likePattern(value)), + input: (value) => matches("input_preview", likePattern(value)), trace_id: (value) => matches("trace_id", likePattern(value)), }; -const FREE_TEXT_COLUMNS = ["trace_id", "input", "name"] as const; +const FREE_TEXT_COLUMNS = ["trace_id", "input_preview", "name"] as const; const textPredicate = (term: string): string => { const contains = `%${likeLiteral(term)}%`; @@ -67,18 +58,41 @@ function filterPredicate(filter: SearchFilter): string { const timeBound = (range: TimeWindow | undefined): string => range - ? `min(StartTs) >= fromUnixTimestamp64Milli(${range.startMs}) AND min(StartTs) < fromUnixTimestamp64Milli(${range.endMs})` - : "min(StartTs) >= now() - INTERVAL 1 DAY"; + ? `start_time >= fromUnixTimestamp64Milli(${range.startMs}) AND start_time < fromUnixTimestamp64Milli(${range.endMs})` + : "start_time >= now() - INTERVAL 1 DAY"; export const runPredicates = (query: SearchQuery): string[] => [ ...query.text.map(textPredicate), ...query.filters.map(filterPredicate), ]; -/** The runs list as a trace query: one row per trace from the per-key rollup, filtered and ordered like the list. */ -export function runQuerySql(query: SearchQuery, range?: TimeWindow, order: RunOrder = NEWEST): string { - const having = [timeBound(range), ...runPredicates(query)].join("\n AND "); - return `${RUN_ROWS}\nHAVING ${having}\n${orderBy(order)}\nLIMIT 100`; +export type RunSQL = + | { readonly kind: "sql"; readonly sql: string } + | { readonly kind: "unsupported"; readonly reason: string }; + +const unsupported = (query: SearchQuery): string | undefined => { + if (query.text.some((term) => term.includes(":"))) + return "Use the traces API for attribute or unsupported field filters"; + if ( + query.filters.some( + (filter) => filter.field === "root_status" && !["ok", "error", "unset"].includes(filter.value.toLowerCase()), + ) + ) + return "Choose root_status:ok, root_status:error, or root_status:unset"; + if ( + query.filters.some( + (filter) => filter.field === "has_error" && !["true", "false"].includes(filter.value.toLowerCase()), + ) + ) + return "Choose has_error:true or has_error:false"; + return undefined; +}; + +export function runQuerySql(query: SearchQuery, range?: TimeWindow, order: RunOrder = NEWEST): RunSQL { + const reason = unsupported(query); + if (reason) return { kind: "unsupported", reason }; + const where = [timeBound(range), ...runPredicates(query)].join("\n AND "); + return { kind: "sql", sql: `${RUN_ROWS}\nWHERE ${where}\n${orderBy(order)}\nLIMIT 100` }; } /** Runs `sql` through the trace query API; the quoted heredoc keeps the SQL's own quotes intact. */ @@ -92,5 +106,7 @@ export const traceQueryCommand = (sql: string): string => export const runQueryCommand = (range?: TimeWindow, order: RunOrder = NEWEST) => - (query: SearchQuery): string => - traceQueryCommand(runQuerySql(query, range, order)); + (query: SearchQuery): string => { + const result = runQuerySql(query, range, order); + return result.kind === "sql" ? traceQueryCommand(result.sql) : result.reason; + }; diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/traceReadFailure.test.ts b/ui/litellm-dashboard/src/components/lens/traces/list/traceReadFailure.test.ts index c57325af563..e4f07397f32 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/traceReadFailure.test.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/traceReadFailure.test.ts @@ -10,8 +10,10 @@ import { traceReadRetryDelay, } from "./traceReadFailure"; -const failure = (status: number, code?: string, retryAfterMs: number | null = null) => - new ApiError("boom", status, { detail: code ? { code, message: "boom" } : "boom" }, retryAfterMs); +const failure = (status: number, code?: string, retryAfterMs: number | null = null) => { + const body = { type: "about:blank", title: "Request failed", status, detail: "boom", ...(code ? { code } : {}) }; + return new ApiError("boom", status, body, retryAfterMs); +}; describe("classifyTraceReadFailure", () => { it.each([ diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/traceReadFailure.ts b/ui/litellm-dashboard/src/components/lens/traces/list/traceReadFailure.ts index b4b180e9caa..d1f21b3b2f2 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/traceReadFailure.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/traceReadFailure.ts @@ -29,10 +29,8 @@ const KIND_BY_STATUS: Readonly> = { }; const failureCode = (body: unknown): string | undefined => { - if (typeof body !== "object" || body === null || !("detail" in body)) return undefined; - const detail = body.detail; - if (typeof detail !== "object" || detail === null || !("code" in detail)) return undefined; - return typeof detail.code === "string" ? detail.code : undefined; + if (typeof body !== "object" || body === null || !("code" in body)) return undefined; + return typeof body.code === "string" ? body.code : undefined; }; export function classifyTraceReadFailure(error: unknown): TraceReadFailure { diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/useAgentTraces.ts b/ui/litellm-dashboard/src/components/lens/traces/list/useAgentTraces.ts index 4670130c2ac..65ff14304e9 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/useAgentTraces.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/useAgentTraces.ts @@ -1,4 +1,4 @@ -import { useTracesApi } from "../api"; +import { type RunSelection, useTracesApi } from "../api"; import { keepPreviousData, useInfiniteQuery, useQuery, type UseQueryOptions } from "@tanstack/react-query"; import { useMemo } from "react"; @@ -14,12 +14,8 @@ import { import type { TracePage, TraceSummary } from "../types"; import type { RunOrder } from "./runOrder"; -interface LoadedTracePage extends TracePage { - window: TimeWindow; -} - interface PagePosition { - readonly window: TimeWindow; + readonly window: TracePage["window"]; readonly cursor: string; } @@ -56,6 +52,7 @@ interface UseAgentTracesOptions { export interface AgentTracesResult { traces: TraceSummary[]; + resolvedWindow: TracePage["window"] | null; isLoading: boolean; isFetching: boolean; /** Rows belong to the previous order, search or window while this one loads. */ @@ -82,14 +79,16 @@ export function useAgentTraces({ }: UseAgentTracesOptions): AgentTracesResult { const traces = useTracesApi(accessToken); const isLiveTail = isLive(range); - const fetchPage = async (pageParam: unknown): Promise => { + const fetchPage = async (pageParam: unknown): Promise => { const position = pageParam as PagePosition | null; - const window = position?.window ?? zoom ?? timeWindow(range, Date.now()); - const selection = { window, q }; + const window = position + ? { startMs: position.window.start_ms, endMs: position.window.end_ms } + : zoom ?? timeWindow(range, Date.now()); + const selection: RunSelection = { window, q, ...(position ? { asOfMs: position.window.as_of_ms } : {}) }; const page = { cursor: position?.cursor ?? null }; - return { ...(await traces.list({ selection, order, page })), window }; + return traces.list({ selection, order, page }); }; - const queryOptions: Parameters>[0] = { + const queryOptions: Parameters>[0] = { queryKey: ["agentTraces", traces.scope, range.hours, range.anchorMs, q, zoom, order.key, order.descending], placeholderData: keepPreviousData, queryFn: ({ pageParam }) => fetchPage(pageParam), @@ -104,13 +103,14 @@ export function useAgentTraces({ refetchOnReconnect: (q) => !requiresUserAction(q.state.error), refetchIntervalInBackground: false, }; - const query = useInfiniteQuery(queryOptions); + const query = useInfiniteQuery(queryOptions); const loaded = useMemo(() => query.data?.pages.flatMap((page) => page.data) ?? [], [query.data]); const notEnabled = isTracingNotEnabled(query.error); return { traces: loaded, + resolvedWindow: query.isPlaceholderData ? null : query.data?.pages[0]?.window ?? null, isLoading: query.isLoading, isFetching: query.isFetching, isPlaceholder: query.isPlaceholderData, diff --git a/ui/litellm-dashboard/src/components/lens/traces/list/useTraceHistogram.ts b/ui/litellm-dashboard/src/components/lens/traces/list/useTraceHistogram.ts index 87d5dee0eae..abf384f3d76 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/list/useTraceHistogram.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/list/useTraceHistogram.ts @@ -39,9 +39,9 @@ export function useTraceHistogram( enabled: boolean, ): TraceHistogramResult { const traces = useTracesApi(accessToken); - const { window, q } = selection; + const { window, q, asOfMs } = selection; const histogramOptions = { - queryKey: ["agentTraceHistogram", traces.scope, window.startMs, window.endMs, q], + queryKey: ["agentTraceHistogram", traces.scope, window.startMs, window.endMs, q, asOfMs], queryFn: () => traces.histogram(selection, BUCKETS), enabled, placeholderData: keepPreviousData, diff --git a/ui/litellm-dashboard/src/components/lens/traces/routing.ts b/ui/litellm-dashboard/src/components/lens/traces/routing.ts index de75a226f51..459167925ae 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/routing.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/routing.ts @@ -13,11 +13,10 @@ export type SpanTab = (typeof SPAN_TABS)[number]; export interface TraceRef { traceId: string; - traceRef?: string; } -export const traceRefOf = (run: TraceSummary): TraceRef => ({ traceId: run.trace_id, traceRef: run.trace_ref }); -export const traceKey = (ref: TraceRef): string => ref.traceRef || ref.traceId; +export const traceRefOf = (run: TraceSummary): TraceRef => ({ traceId: run.id }); +export const traceKey = (ref: TraceRef): string => ref.traceId; /** Which step, view and detail section of an open run are showing. Owned by the URL in the drawer, locally in sheets. */ export interface RunSelection { @@ -35,7 +34,6 @@ export interface RunSelection { export const OPEN_TRACE_PARSERS = { trace: parseAsString, - trace_ref: parseAsString, span: parseAsString, view: parseAsStringLiteral(TRACE_VIEWS).withDefault("steps"), span_tab: parseAsStringLiteral(SPAN_TABS).withDefault("content"), @@ -76,7 +74,7 @@ export function useOpenTraceRouting(): OpenTraceRouting { const [params, setParams] = useQueryStates(OPEN_TRACE_PARSERS, { history: "push" }); const openTrace = useCallback( (ref: TraceRef | null) => { - const run = { trace: ref?.traceId ?? null, trace_ref: ref?.traceRef || null }; + const run = { trace: ref?.traceId ?? null }; void setParams(ref === null ? { ...FRESH_RUN, ...run, fullscreen: null } : { ...FRESH_RUN, ...run }); }, [setParams], @@ -103,7 +101,7 @@ export function useOpenTraceRouting(): OpenTraceRouting { [setParams], ); return { - trace: params.trace === null ? null : { traceId: params.trace, traceRef: params.trace_ref ?? undefined }, + trace: params.trace === null ? null : { traceId: params.trace }, openTrace, selection: { spanId: params.span, diff --git a/ui/litellm-dashboard/src/components/lens/traces/types.ts b/ui/litellm-dashboard/src/components/lens/traces/types.ts index 4b0e738f537..fe4e93c1d83 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/types.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/types.ts @@ -1,31 +1,34 @@ import type { components, paths } from "@/lib/http/schema"; -export type Trace = paths["/v1/traces/{trace_id}"]["get"]["responses"][200]["content"]["application/json"]; +export type TraceMetadata = paths["/v1/traces/{id}"]["get"]["responses"][200]["content"]["application/json"]; +export type TraceSpansPage = paths["/v1/traces/{id}/spans"]["get"]["responses"][200]["content"]["application/json"]; export type TracePage = paths["/v1/traces"]["get"]["responses"][200]["content"]["application/json"]; export type TraceHistogram = paths["/v1/traces/histogram"]["get"]["responses"][200]["content"]["application/json"]; +export type RunValues = paths["/v1/traces/values/{field}"]["get"]["responses"][200]["content"]["application/json"]; export type RunField = paths["/v1/traces/values/{field}"]["get"]["parameters"]["path"]["field"]; export type SpanErrorPage = - paths["/v1/traces/{trace_id}/spans/{span_id}/error"]["get"]["responses"][200]["content"]["application/json"]; -type ApiSpanDetail = - paths["/v1/traces/{trace_id}/spans/{span_id}"]["get"]["responses"][200]["content"]["application/json"]; -export type Span = Trace["spans"][number]; + paths["/v1/traces/{id}/spans/{span_id}/error"]["get"]["responses"][200]["content"]["application/json"]; +export type SpanDetail = + paths["/v1/traces/{id}/spans/{span_id}"]["get"]["responses"][200]["content"]["application/json"]; +export type Span = components["schemas"]["Span"]; export type SpanType = Span["type"]; export type SpanStatus = Span["status"]; -export type AgentNode = Trace["agents"][number]; -export type TraceSummary = Trace["summary"]; -export type SpanDetail = Omit & - Partial>; -export type UIToolCall = components["schemas"]["UIToolCall"]; -export type UIMessage = components["schemas"]["UIMessage"]; -export type UIField = components["schemas"]["UIField"]; -export type UIContent = ApiSpanDetail["input_ui"]; +export type AgentNode = TraceMetadata["agents"][number]; +export type TraceSummary = TraceMetadata["summary"]; +export type Trace = TraceMetadata & { + spans: Span[]; + next_cursor?: string | null; + spans_complete?: boolean; +}; export interface TraceToolCall { name: string; args: unknown; } -export type TraceMessage = Omit & { +export interface TraceMessage { role: string; + content: string; + name?: string; tool_calls?: TraceToolCall[]; -}; +} diff --git a/ui/litellm-dashboard/src/components/lens/traces/utils.test.ts b/ui/litellm-dashboard/src/components/lens/traces/utils.test.ts index eb7cca4bcac..0e5d15cdfd8 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/utils.test.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/utils.test.ts @@ -304,6 +304,17 @@ describe("payload helpers", () => { expect(parseMessages('[{"role":"user","content":42}]')).toBeNull(); }); + it("renders raw OpenAI tool calls with structured arguments and absent text", () => { + const raw = { + role: "assistant", + content: null, + tool_calls: [{ type: "function", function: { name: "lookup", arguments: '{"order":42}' } }], + }; + expect(parseMessages(JSON.stringify(raw))).toEqual([ + { role: "assistant", content: "", tool_calls: [{ name: "lookup", args: { order: 42 } }] }, + ]); + }); + it("reads LangChain's serialized messages with their roles, names and tool calls", () => { const dumped = [ { type: "human", data: { content: "What is an agent trace?", name: null } }, diff --git a/ui/litellm-dashboard/src/components/lens/traces/utils.ts b/ui/litellm-dashboard/src/components/lens/traces/utils.ts index 99a2a1194ec..d1881ad36fb 100644 --- a/ui/litellm-dashboard/src/components/lens/traces/utils.ts +++ b/ui/litellm-dashboard/src/components/lens/traces/utils.ts @@ -9,8 +9,7 @@ import type { Span, TraceMessage, TraceSummary, TraceToolCall } from "./types"; /* Formatting */ /* ------------------------------------------------------------------ */ -export const traceAgentNames = (trace: TraceSummary): readonly string[] => - trace.agent_names ?? (trace.service ? [trace.service] : []); +export const traceAgentNames = (trace: TraceSummary): readonly string[] => trace.agent_names; export const fmtMs = (ms: number): string => { if (ms >= 60_000) return `${(ms / 60_000).toFixed(1)}m`; @@ -348,13 +347,33 @@ const parseLangchainMessage = (value: object): TraceMessage | null => { }; }; +const parseToolCall = (value: unknown): TraceToolCall | null => { + if (typeof value !== "object" || value === null) return null; + const nested: unknown = Reflect.get(value, "function"); + const call = typeof nested === "object" && nested !== null ? nested : value; + const name: unknown = Reflect.get(call, "name"); + if (typeof name !== "string") return null; + const args: unknown = Reflect.get(call, "args") ?? Reflect.get(call, "arguments"); + return { name, args: typeof args === "string" ? parseJson(args) ?? args : args ?? {} }; +}; + const parseMessage = (value: unknown): TraceMessage | null => { if (typeof value !== "object" || value === null) return null; if (!("role" in value) && "data" in value) return parseLangchainMessage(value); const role: unknown = Reflect.get(value, "role"); - const content: unknown = Reflect.get(value, "content") ?? Reflect.get(value, "parts"); + const rawCalls: unknown = Reflect.get(value, "tool_calls"); + const calls = Array.isArray(rawCalls) + ? rawCalls.map(parseToolCall).filter((call): call is TraceToolCall => call !== null) + : []; + const rawContent: unknown = Reflect.get(value, "content") ?? Reflect.get(value, "parts"); + const content = rawContent == null && calls.length > 0 ? "" : rawContent; if (typeof role !== "string" || (typeof content !== "string" && !Array.isArray(content))) return null; - return { ...value, role, content: messageText(typeof content === "string" ? content : JSON.stringify(content)) }; + return { + ...value, + role, + content: messageText(typeof content === "string" ? content : JSON.stringify(content)), + ...(calls.length > 0 ? { tool_calls: calls } : {}), + }; }; /** An llm span's input (array of messages) or output (one message); null when it isn't one. */ diff --git a/ui/litellm-dashboard/src/components/networking.test.ts b/ui/litellm-dashboard/src/components/networking.test.ts index 62670467827..ba93dcf31b5 100644 --- a/ui/litellm-dashboard/src/components/networking.test.ts +++ b/ui/litellm-dashboard/src/components/networking.test.ts @@ -2,6 +2,8 @@ import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"; import { clearTokenCookies } from "@/utils/cookieUtils"; import * as Networking from "./networking"; import { uiHref } from "@/utils/uiHref"; +import researchJson from "./lens/traces/__fixtures__/research_trace.json"; +import type { Trace, TraceMetadata, TraceSpansPage } from "./lens/traces/types"; vi.mock("@/utils/cookieUtils", () => ({ clearTokenCookies: vi.fn(), @@ -871,3 +873,40 @@ describe("schema-bound dashboard responses", () => { expect(result.users[0]).toEqual(user); }); }); + +describe("trace metadata and span pages", () => { + afterEach(() => vi.unstubAllGlobals()); + + it("reads canonical metadata separately and follows the span cursor with the same page size", async () => { + const trace = researchJson as Trace; + const id = "run/one"; + const metadata: TraceMetadata = { summary: { ...trace.summary, id }, agents: trace.agents }; + const fetchSpy = vi.fn(async (input) => { + const url = new URL(input instanceof Request ? input.url : String(input), "http://proxy.test"); + if (!url.pathname.endsWith("/spans")) return new Response(JSON.stringify(metadata)); + const page: TraceSpansPage = url.searchParams.get("cursor") + ? { data: trace.spans.slice(1), next_cursor: null } + : { data: trace.spans.slice(0, 1), next_cursor: "next-page" }; + return new Response(JSON.stringify(page)); + }); + vi.stubGlobal("fetch", fetchSpy); + const first = await Networking.agentTraceCall("token", id); + const second = await Networking.agentTraceCall("token", id, first.next_cursor); + expect(first.summary.id).toBe(id); + expect([...first.spans, ...second.spans]).toEqual(trace.spans); + expect(first.spans_complete).toBe(false); + expect(second.spans_complete).toBe(true); + const urls = fetchSpy.mock.calls.map( + ([input]) => new URL(input instanceof Request ? input.url : String(input), "http://proxy.test"), + ); + expect(urls.map((url) => url.pathname)).toEqual([ + expect.stringMatching(/\/v1\/traces\/run%2Fone$/), + expect.stringMatching(/\/v1\/traces\/run%2Fone\/spans$/), + expect.stringMatching(/\/v1\/traces\/run%2Fone$/), + expect.stringMatching(/\/v1\/traces\/run%2Fone\/spans$/), + ]); + const pages = urls.filter((url) => url.pathname.endsWith("/spans")); + expect(pages.map((url) => url.searchParams.get("cursor"))).toEqual([null, "next-page"]); + expect(pages.map((url) => url.searchParams.get("page_size"))).toEqual(["200", "200"]); + }); +}); diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index 441e1d55d61..e94f5bc32cd 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -118,7 +118,7 @@ import type { AutoRouterPresetsResponse } from "@/lib/autorouter_presets"; import type { VectorStoreIndex } from "@/app/(dashboard)/vector-stores/_components/IndexesTab"; import type { RoutingDecision } from "./logs/detail/RoutingDecisionCard"; import type { RunListRequest } from "./lens/traces/api"; -import type { SpanDetail, SpanErrorPage, Trace, TracePage } from "./lens/traces/types"; +import type { SpanDetail, SpanErrorPage, Trace, TraceMetadata, TracePage, TraceSpansPage } from "./lens/traces/types"; import { createApiClient, deriveErrorMessage, @@ -1965,6 +1965,7 @@ export const agentTraceListCall = async ( const query = { start_ms: selection.window.startMs, end_ms: selection.window.endMs, + as_of_ms: selection.asOfMs, q: selection.q || undefined, sort_by: order.key, sort_dir: order.descending ? "desc" : "asc", @@ -1976,38 +1977,34 @@ export const agentTraceListCall = async ( export const sendOtlpTraceCall = async (accessToken: string, exportRequest: object): Promise => apiClient.post(`/v1/traces`, { accessToken, body: exportRequest }); -export const agentTraceCall = async ( - accessToken: string, - traceId: string, - traceRef?: string, - cursor?: string | null, -): Promise => - apiClient.get(`/v1/traces/${encodeURIComponent(traceId)}`, { - accessToken, - query: { trace_ref: traceRef || undefined, cursor: cursor ?? undefined, page_size: 200 }, - }); +export const agentTraceCall = async (accessToken: string, id: string, cursor?: string | null): Promise => { + const path = `/v1/traces/${encodeURIComponent(id)}`; + const options = { accessToken }; + const pageOptions = { accessToken, query: { cursor: cursor ?? undefined, page_size: 200 } }; + const [metadata, page] = await Promise.all([ + apiClient.get(path, options), + apiClient.get(`${path}/spans`, pageOptions), + ]); + return { ...metadata, spans: page.data, next_cursor: page.next_cursor, spans_complete: page.next_cursor === null }; +}; -export const agentTraceSpanCall = async ( - accessToken: string, - traceId: string, - spanId: string, - traceRef?: string, -): Promise => - apiClient.get(`/v1/traces/${encodeURIComponent(traceId)}/spans/${encodeURIComponent(spanId)}`, { +export const agentTraceSpanCall = async (accessToken: string, id: string, spanId: string): Promise => + apiClient.get(`/v1/traces/${encodeURIComponent(id)}/spans/${encodeURIComponent(spanId)}`, { accessToken, - query: { trace_ref: traceRef || undefined }, }); export const agentTraceSpanErrorCall = async ( accessToken: string, - traceId: string, + id: string, spanId: string, - options: { traceRef?: string; cursor?: string | null }, -): Promise => - apiClient.get(`/v1/traces/${encodeURIComponent(traceId)}/spans/${encodeURIComponent(spanId)}/error`, { - accessToken, - query: { trace_ref: options.traceRef || undefined, cursor: options.cursor || undefined }, - }); + options: { cursor?: string | null }, +): Promise => { + const requestOptions = { accessToken, query: { cursor: options.cursor || undefined } }; + return apiClient.get( + `/v1/traces/${encodeURIComponent(id)}/spans/${encodeURIComponent(spanId)}/error`, + requestOptions, + ); +}; export const adminSpendLogsCall = async (accessToken: string) => { try { diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 7e541e11798..a9bc9f687cd 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -22605,11 +22605,11 @@ export interface paths { path?: never; cookie?: never; }; - /** List Agent Traces */ - get: operations["list_agent_traces_v1_traces_get"]; + /** @description Search trace summaries. All clauses in q must match. Free text searches trace_id, name and input preview as case-insensitive substrings. key:value clauses match whole values, with * as a wildcard; double quotes group spaces or literal colons. Prefix keyed clauses with - to exclude matches. Unknown keys, missing values and malformed quoting return invalid_request. name describes the physical root; input searches its preview, falling back to the first nonempty agent or LLM preview; agent, model and attr. match any span. root_status describes the root, has_error means any failed span. The default window is the last 24 hours. First-page ingestion timestamp cutoff is retained by the cursor. This excludes later-stamped exports, not delayed commits stamped before the cutoff, and is not a database transaction snapshot. Repeat q and sorting on continuation; omitted bounds reuse the cursor window. Sort ties use the canonical id in the same direction. The server may return fewer than page_size items to respect response limits. Continue until next_cursor is null. Span-derived metrics use the cutoff; spend enrichment is best effort */ + get: operations["trace_list"]; put?: never; - /** Ingest Otlp Traces */ - post: operations["ingest_otlp_traces_v1_traces_post"]; + /** @description Export OTLP traces as JSON or protobuf, optionally gzip compressed. The OTLP protocol defines payloads and responses: https://opentelemetry.io/docs/specs/otlp/. Ownership is derived from authentication, never payload attributes */ + post: operations["trace_ingest"]; delete?: never; options?: never; head?: never; @@ -22623,8 +22623,8 @@ export interface paths { path?: never; cookie?: never; }; - /** Agent Trace Histogram */ - get: operations["agent_trace_histogram_v1_traces_histogram_get"]; + /** @description Count matching traces in equal-width [start_ms,end_ms) buckets. Reuse the list window and as_of_ms for matching span-derived membership. failed counts traces with any failed span. Agent groups count successful traces under their alphabetically first agent name, or service when no agent name exists */ + get: operations["trace_histogram"]; put?: never; post?: never; delete?: never; @@ -22642,8 +22642,8 @@ export interface paths { }; get?: never; put?: never; - /** Query Agent Traces */ - post: operations["query_agent_traces_v1_traces_query_post"]; + /** @description Execute read-only ClickHouse SQL under authenticated row policies and fixed resource limits. Bind params with native {name:Type} placeholders. Results are always ClickHouse JSON; 64-bit integers may be strings. SQL callers control ORDER BY, LIMIT and keyset continuation. Exceeding a resource limit fails instead of returning partial success */ + post: operations["trace_query"]; delete?: never; options?: never; head?: never; @@ -22657,8 +22657,8 @@ export interface paths { path?: never; cookie?: never; }; - /** Help Agent Trace Queries */ - get: operations["help_agent_trace_queries_v1_traces_query_help_get"]; + /** @description Discover current SQL schema, logical views, scoped examples and resource limits */ + get: operations["trace_query_help"]; put?: never; post?: never; delete?: never; @@ -22674,8 +22674,8 @@ export interface paths { path?: never; cookie?: never; }; - /** Agent Trace Values */ - get: operations["agent_trace_values_v1_traces_values__field__get"]; + /** @description Return the most common distinct values among matching traces. contains is case-insensitive. limit is a top-K suggestion limit, not a pagination size. Reuse list window and as_of_ms for matching membership */ + get: operations["trace_values"]; put?: never; post?: never; delete?: never; @@ -22684,15 +22684,15 @@ export interface paths { patch?: never; trace?: never; }; - "/v1/traces/{trace_id}": { + "/v1/traces/{id}": { parameters: { query?: never; header?: never; path?: never; cookie?: never; }; - /** Get Agent Trace */ - get: operations["get_agent_trace_v1_traces__trace_id__get"]; + /** @description Read summary and agent metadata by canonical id. trace_id is the original OTLP id, which can repeat across ownership scopes. Spans are read through the separate spans collection */ + get: operations["trace_get"]; put?: never; post?: never; delete?: never; @@ -22701,15 +22701,15 @@ export interface paths { patch?: never; trace?: never; }; - "/v1/traces/{trace_id}/spans/{span_id}": { + "/v1/traces/{id}/spans": { parameters: { query?: never; header?: never; path?: never; cookie?: never; }; - /** Get Agent Trace Span */ - get: operations["get_agent_trace_span_v1_traces__trace_id__spans__span_id__get"]; + /** @description Read a bounded page of canonical spans. The cursor pins the graph version. Continue with the same page_size; a changed graph or expired reconstruction returns trace_changed and the traversal must restart */ + get: operations["trace_spans"]; put?: never; post?: never; delete?: never; @@ -22718,15 +22718,32 @@ export interface paths { patch?: never; trace?: never; }; - "/v1/traces/{trace_id}/spans/{span_id}/error": { + "/v1/traces/{id}/spans/{span_id}": { parameters: { query?: never; header?: never; path?: never; cookie?: never; }; - /** Get Agent Trace Span Error */ - get: operations["get_agent_trace_span_error_v1_traces__trace_id__spans__span_id__error_get"]; + /** @description Read raw input, output and attributes for one span. UI rendering is performed by the client */ + get: operations["trace_span"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/v1/traces/{id}/spans/{span_id}/error": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** @description Read bounded diagnostic text pages. Cursor validation detects content changes and requires restarting the traversal */ + get: operations["trace_error"]; put?: never; post?: never; delete?: never; @@ -25945,21 +25962,19 @@ export interface components { /** Updated By */ updated_by: string; }; - /** AgentNode */ + /** @description One distinct agent in a trace: 200 invocations of `researcher` are one node. */ AgentNode: { - /** Duration Ms */ + /** Format: double */ duration_ms: number; - /** Invocations */ + /** Format: uint64 */ invocations: number; - /** Llm Calls */ + /** Format: uint64 */ llm_calls: number; - /** Name */ name: string; - /** Parent Agent */ parent_agent: string | null; - /** Spend */ + /** Format: double */ spend: number | null; - /** Tool Calls */ + /** Format: uint64 */ tool_calls: number; }; /** AgentObjectPermission */ @@ -26061,11 +26076,9 @@ export interface components { /** Updated By */ updated_by?: string | null; }; - /** AgentRuns */ AgentRuns: { - /** Agent */ agent: string; - /** Runs */ + /** Format: uint64 */ runs: number; }; /** @@ -33125,17 +33138,16 @@ export interface components { */ vault_token?: string | null; }; - /** HistogramBucket */ HistogramBucket: { - /** Agents */ + /** @description Runs that did not fail, by their alphabetically first agent label, or service when unlabelled. */ agents: components["schemas"]["AgentRuns"][]; - /** End Ms */ + /** Format: int64 */ end_ms: number; - /** Failed */ + /** Format: uint64 */ failed: number; - /** Start Ms */ + /** Format: int64 */ start_ms: number; - /** Total */ + /** Format: uint64 */ total: number; }; /** Hyperparameters */ @@ -37787,6 +37799,8 @@ export interface components { /** Last Authenticated At */ last_authenticated_at?: string | null; }; + /** @enum {string} */ + MapValueType: "String"; /** * Mcp * @description Give the model access to additional tools via remote Model Context Protocol @@ -38081,6 +38095,8 @@ export interface components { } & { [key: string]: unknown; }; + /** @enum {string} */ + MetadataValueType: "array" | "boolean" | "integer" | "null" | "number" | "object" | "string"; /** MetricWithMetadata */ MetricWithMetadata: { /** Api Key Breakdown */ @@ -40547,6 +40563,7 @@ export interface components { /** Tpm Limit */ tpm_limit?: number | null; }; + PathPart: string | number; /** * PendingSafetyCheck * @description A pending safety check for the computer call. @@ -44869,6 +44886,11 @@ export interface components { /** Run Id */ run_id: string; }; + /** + * RunField + * @enum {string} + */ + RunField: "name" | "agent" | "root_status" | "has_error" | "model" | "input" | "trace_id" | "service" | "team"; /** RunRequest */ RunRequest: { /** Agent Name */ @@ -44881,10 +44903,13 @@ export interface components { /** Start */ start?: string | null; }; - /** RunValues */ + /** + * RunValues + * @description Distinct values of one run field, most common first. + */ RunValues: { - /** Values */ values: string[]; + window: components["schemas"]["TraceQueryWindow"]; }; /** SCIMEnterpriseUser */ SCIMEnterpriseUser: { @@ -45811,77 +45836,51 @@ export interface components { /** Version */ version?: string; }; - /** Span */ Span: { - /** Agent */ agent: string; - /** Duration Ms */ + /** Format: double */ duration_ms: number; - /** Error */ error: string | null; - /** Error Truncated */ error_truncated: boolean; - /** Framework */ framework: string; - /** Input Preview */ input_preview: string; - /** Input Tokens */ + /** Format: uint32 */ input_tokens: number; - /** Litellm Request Id */ litellm_request_id: string | null; - /** Model */ model: string | null; - /** Name */ name: string; - /** Output Tokens */ + /** Format: uint32 */ output_tokens: number; - /** Parent Span Id */ parent_span_id: string | null; - /** Span Id */ span_id: string; - /** Spend */ + /** Format: double */ spend: number | null; - /** Start Offset Ms */ + /** Format: double */ start_offset_ms: number; - /** - * Status - * @enum {string} - */ - status: "ok" | "error" | "unset"; - /** - * Type - * @enum {string} - */ - type: "agent" | "llm" | "tool" | "chain" | "framework" | "retriever" | "embedding" | "reranker" | "guardrail" | "evaluator" | "prompt" | "decision"; + status: components["schemas"]["SpanStatus"]; + type: components["schemas"]["SpanType"]; }; /** SpanDetail */ SpanDetail: { - /** Attributes */ attributes: { [key: string]: string; }; - /** Input */ input: string; - /** Input Ui */ - input_ui: components["schemas"]["UIMessages"] | components["schemas"]["UIFields"] | components["schemas"]["UIText"]; - /** Output */ output: string; - /** Output Ui */ - output_ui: components["schemas"]["UIMessages"] | components["schemas"]["UIFields"] | components["schemas"]["UIText"]; - /** Span Id */ span_id: string; }; /** SpanErrorPage */ SpanErrorPage: { - /** Message */ message: string; - /** Next Cursor */ next_cursor: string | null; - /** Span Id */ span_id: string; - /** Total Chars */ + /** Format: uint64 */ total_chars: number; }; + /** @enum {string} */ + SpanStatus: "ok" | "error" | "unset"; + /** @enum {string} */ + SpanType: "agent" | "llm" | "tool" | "chain" | "framework" | "retriever" | "embedding" | "reranker" | "guardrail" | "evaluator" | "prompt" | "decision"; /** SpendAnalyticsPaginatedResponse */ SpendAnalyticsPaginatedResponse: { metadata?: components["schemas"]["DailySpendMetadata"]; @@ -46016,6 +46015,7 @@ export interface components { */ total_tokens: number; }; + SqlParameter: string | number | boolean | null | string[]; /** StandardLoggingHeuristicV2Forecast */ StandardLoggingHeuristicV2Forecast: { /** Predicted Tier */ @@ -47723,27 +47723,69 @@ export interface components { } & { [key: string]: unknown; }; - /** Trace */ - Trace: { - /** Agents */ + /** TraceErrorPageRequest */ + TraceErrorPageRequest: { + cursor?: string | null; + }; + /** + * TraceHistogram + * @description Matching runs per equal-width slice of the window. + */ + TraceHistogram: { + buckets: components["schemas"]["HistogramBucket"][]; + window: components["schemas"]["TraceQueryWindow"]; + }; + /** TraceHistogramRequest */ + TraceHistogramRequest: { + /** Format: uint64 */ + as_of_ms?: number | null; + /** + * Format: uint16 + * @default 60 + */ + buckets: number; + /** Format: int64 */ + end_ms?: number | null; + /** @default */ + q: string; + /** Format: int64 */ + start_ms?: number | null; + }; + TraceInvalidParam: { + location: string; + reason: string; + }; + /** TraceListRequest */ + TraceListRequest: { + /** Format: uint64 */ + as_of_ms?: number | null; + cursor?: string | null; + /** Format: int64 */ + end_ms?: number | null; + /** + * Format: uint16 + * @default 50 + */ + page_size: number; + /** @default */ + q: string; + sort_by?: components["schemas"]["TraceSortField"]; + sort_dir?: components["schemas"]["TraceSortDirection"]; + /** Format: int64 */ + start_ms?: number | null; + }; + /** TraceMetadata */ + TraceMetadata: { agents: components["schemas"]["AgentNode"][]; - /** Next Cursor */ - next_cursor?: string | null; - /** Spans */ - spans: components["schemas"]["Span"][]; summary: components["schemas"]["TraceSummary"]; }; - /** TraceHistogram */ - TraceHistogram: { - /** Buckets */ - buckets: components["schemas"]["HistogramBucket"][]; - }; + /** TraceNoQueryRequest */ + TraceNoQueryRequest: Record; /** TracePage */ TracePage: { - /** Data */ data: components["schemas"]["TraceSummary"][]; - /** Next Cursor */ next_cursor: string | null; + window: components["schemas"]["TraceQueryWindow"]; }; /** TracePart */ TracePart: { @@ -47768,225 +47810,200 @@ export interface components { */ truncated: boolean; }; - /** TraceQueryAttributeField */ - TraceQueryAttributeField: { - /** Expression */ - expression: string; - /** Key */ - key: string; - /** - * Type - * @constant - */ - type: "String"; + /** TraceProblem */ + TraceProblem: { + code: components["schemas"]["TraceProblemCode"]; + /** Format: uint32 */ + database_code?: number | null; + detail: string; + /** @default [] */ + errors: components["schemas"]["TraceInvalidParam"][]; + /** Format: uint16 */ + status: number; + title: string; + type: string; + }; + /** @enum {string} */ + TraceProblemCode: "invalid_request" | "unauthorized" | "forbidden" | "not_found" | "trace_changed" | "too_large" | "unavailable" | "query_rejected" | "query_limit_exceeded" | "query_unavailable" | "internal_error"; + TraceQueryAttributeField: { + expression: string; + key: string; + type: components["schemas"]["MapValueType"]; }; - /** TraceQueryAttributes */ TraceQueryAttributes: { - /** Column */ column: string; - /** Discovery Sql */ discovery_sql: string; - /** Error */ error?: string | null; - /** Fields */ fields: components["schemas"]["TraceQueryAttributeField"][]; - /** Scope */ scope: string; - /** - * Table - * @enum {string} - */ - table: "otel_traces" | "agent_traces_by_key" | "spend_logs"; - /** Truncated */ + table: components["schemas"]["TraceQueryTableName"]; truncated: boolean; }; - /** TraceQueryColumn */ - TraceQueryColumn: { - /** Name */ - name: string; - /** Type */ - type: string; - } & { - [key: string]: unknown; - }; - /** TraceQueryExample */ TraceQueryExample: { - /** Name */ name: string; - /** Sql */ sql: string; }; /** TraceQueryHelp */ TraceQueryHelp: { - /** Access */ access: string; - /** Attributes */ attributes: components["schemas"]["TraceQueryAttributes"][]; - /** Dialect */ dialect: string; - /** Examples */ examples: components["schemas"]["TraceQueryExample"][]; - /** Gotchas */ gotchas: string[]; - /** Guide */ guide: string; metadata: components["schemas"]["TraceQueryMetadata"]; - /** Normalized Fields */ normalized_fields: components["schemas"]["TraceQueryNormalizedField"][]; - /** Relationships */ relationships: components["schemas"]["TraceQueryRelationship"][]; - /** Response */ response: string; - /** Tables */ tables: components["schemas"]["TraceQueryTable"][]; }; - /** TraceQueryMetadata */ TraceQueryMetadata: { - /** Column */ column: string; - /** Error */ error?: string | null; - /** Fields */ fields: components["schemas"]["TraceQueryMetadataField"][]; - /** Invalid Json Rows */ + /** Format: uint */ invalid_json_rows: number; - /** Sample Sql */ sample_sql: string; - /** Sampled Rows */ + /** Format: uint */ sampled_rows: number; - /** Scope */ scope: string; - /** - * Table - * @enum {string} - */ - table: "otel_traces" | "agent_traces_by_key" | "spend_logs"; - /** Truncated */ + table: components["schemas"]["TraceQueryTableName"]; truncated: boolean; }; - /** TraceQueryMetadataField */ TraceQueryMetadataField: { - /** Expression */ expression: string; - /** Path */ - path: (string | number)[]; - /** Types */ - types: ("array" | "boolean" | "integer" | "null" | "number" | "object" | "string")[]; + path: components["schemas"]["PathPart"][]; + types: components["schemas"]["MetadataValueType"][]; }; - /** TraceQueryNormalizedField */ TraceQueryNormalizedField: { - /** Column */ column: string; - /** Meaning */ meaning: string; - /** Name */ name: string; - /** - * Table - * @enum {string} - */ - table: "otel_traces" | "agent_traces_by_key" | "spend_logs"; - /** Type */ + table: components["schemas"]["TraceQueryTableName"]; type: string; }; - /** TraceQueryRelationship */ TraceQueryRelationship: { - /** Additional Predicates */ additional_predicates: string; - /** Left */ left: string; - /** Meaning */ meaning: string; - /** Right */ right: string; }; /** TraceQueryRequest */ TraceQueryRequest: { - /** Sql */ + /** @default {} */ + params: { + [key: string]: components["schemas"]["SqlParameter"]; + }; sql: string; }; - /** TraceQueryStatistics */ TraceQueryStatistics: { - /** Bytes Read */ - bytes_read: number | string; - /** Elapsed */ + bytes_read: components["schemas"]["UnsignedCount"]; + /** Format: double */ elapsed: number; - /** Rows Read */ - rows_read: number | string; + rows_read: components["schemas"]["UnsignedCount"]; } & { [key: string]: unknown; }; - /** TraceQueryTable */ TraceQueryTable: { - /** Columns */ - columns: components["schemas"]["TraceQueryColumn"][]; - /** - * Name - * @enum {string} - */ - name: "otel_traces" | "agent_traces_by_key" | "spend_logs"; + columns: components["schemas"]["TraceSQLColumn"][]; + name: components["schemas"]["TraceQueryTableName"]; + }; + /** @enum {string} */ + TraceQueryTableName: "traces" | "spans" | "calls" | "otel_traces" | "agent_traces_by_key" | "spend_logs"; + TraceQueryWindow: { + /** Format: uint64 */ + as_of_ms: number; + /** Format: int64 */ + end_ms: number; + /** Format: int64 */ + start_ms: number; + }; + TraceSQLColumn: { + name: string; + type: string; + } & { + [key: string]: unknown; }; /** TraceSQLResponse */ TraceSQLResponse: { - /** Data */ data: { - [key: string]: components["schemas"]["JsonValue"]; + [key: string]: unknown; }[]; - /** Meta */ - meta: components["schemas"]["TraceQueryColumn"][]; - /** Rows */ - rows: number | string; + meta: components["schemas"]["TraceSQLColumn"][]; + rows: components["schemas"]["UnsignedCount"]; statistics: components["schemas"]["TraceQueryStatistics"]; } & { [key: string]: unknown; }; - /** TraceSummary */ - TraceSummary: { - /** Agent Count */ - agent_count: number; - /** Agent Invocations */ - agent_invocations: number; - /** Agent Names */ - agent_names?: string[]; - /** Duration Ms */ - duration_ms: number; - /** Error Count */ - error_count: number; - /** Frameworks */ - frameworks?: string[]; - /** Input Preview */ - input_preview: string; - /** Input Tokens */ - input_tokens: number; - /** Llm Calls */ - llm_calls: number; - /** Models */ - models: string[]; - /** Name */ - name: string; - /** Output Tokens */ - output_tokens: number; - /** Resolution Limited */ - resolution_limited?: boolean; - /** Service */ - service: string; - /** Span Count */ - span_count: number; - /** Spend */ - spend: number | null; - /** Start Time */ - start_time: string; + /** @enum {string} */ + TraceSortDirection: "asc" | "desc"; + /** @enum {string} */ + TraceSortField: "start_ms" | "duration_ms" | "span_count" | "error_count"; + /** TraceSpanPageRequest */ + TraceSpanPageRequest: { + cursor?: string | null; /** - * Status - * @enum {string} + * Format: uint16 + * @default 100 */ - status: "ok" | "error" | "unset"; - /** Tool Calls */ + page_size: number; + }; + /** TraceSpansPage */ + TraceSpansPage: { + data: components["schemas"]["Span"][]; + next_cursor: string | null; + }; + TraceSummary: { + /** Format: uint64 */ + agent_count: number; + /** Format: uint64 */ + agent_invocations: number; + agent_names: string[]; + /** Format: double */ + duration_ms: number; + /** Format: uint64 */ + error_count: number; + frameworks: string[]; + has_error: boolean; + id: string; + input_preview: string; + /** Format: uint64 */ + input_tokens: number; + /** Format: uint64 */ + llm_calls: number; + models: string[]; + name: string; + /** Format: uint64 */ + output_tokens: number; + resolution_limited: boolean; + root_status: components["schemas"]["SpanStatus"]; + service: string; + /** Format: uint64 */ + span_count: number; + /** Format: double */ + spend: number | null; + start_time: string; + /** Format: uint64 */ tool_calls: number; - /** Trace Id */ trace_id: string; - /** Trace Ref */ - trace_ref?: string; + }; + /** TraceValuesRequest */ + TraceValuesRequest: { + /** Format: uint64 */ + as_of_ms?: number | null; + /** @default */ + contains: string; + /** Format: int64 */ + end_ms?: number | null; + /** + * Format: uint16 + * @default 20 + */ + limit: number; + /** @default */ + q: string; + /** Format: int64 */ + start_ms?: number | null; }; /** TrainedTierArtifact */ TrainedTierArtifact: { @@ -48062,47 +48079,6 @@ export interface components { } & { [key: string]: unknown; }; - /** UIField */ - UIField: { - /** Key */ - key: string; - /** Value */ - value: string; - }; - /** UIFields */ - UIFields: { - /** Fields */ - fields: components["schemas"]["UIField"][]; - /** - * Kind - * @constant - */ - kind: "fields"; - }; - /** UIMessage */ - UIMessage: { - /** Content */ - content: string; - /** Name */ - name?: string | null; - /** - * Role - * @enum {string} - */ - role: "system" | "user" | "assistant" | "tool"; - /** Tool Calls */ - tool_calls?: components["schemas"]["UIToolCall"][]; - }; - /** UIMessages */ - UIMessages: { - /** - * Kind - * @constant - */ - kind: "messages"; - /** Messages */ - messages: components["schemas"]["UIMessage"][]; - }; /** * UISettingsResponse * @description Response model for UI settings @@ -48121,16 +48097,6 @@ export interface components { [key: string]: unknown; }; }; - /** UIText */ - UIText: { - /** - * Kind - * @constant - */ - kind: "text"; - /** Text */ - text: string; - }; /** * UIThemeConfig * @description Configuration for UI theme customization @@ -48166,13 +48132,6 @@ export interface components { [key: string]: unknown; }; }; - /** UIToolCall */ - UIToolCall: { - /** Arguments */ - arguments: string; - /** Name */ - name: string; - }; /** UiDiscoveryEndpoints */ UiDiscoveryEndpoints: { /** Admin Ui Disabled */ @@ -48214,6 +48173,7 @@ export interface components { */ blocked_users: string[]; }; + UnsignedCount: number | string; /** UpdateCredentialItem */ UpdateCredentialItem: { /** Credential Info */ @@ -81302,18 +81262,17 @@ export interface operations { }; }; }; - list_agent_traces_v1_traces_get: { + trace_list: { parameters: { query?: { - /** @description Window start, unix ms. Default: 24h ago */ + as_of_ms?: number | null; start_ms?: number | null; - /** @description Window end, unix ms. Default: now */ end_ms?: number | null; - /** @description Free text and key:value filters, e.g. `agent:research* -status:ok "book a flight"`. Keys: name, agent, status, model, input, trace_id, service, team and attr.. `*` globs and a leading `-` negates */ q?: string; cursor?: string | null; - sort_by?: "start_ms" | "duration_ms" | "span_count" | "error_count"; - sort_dir?: "asc" | "desc"; + page_size?: number; + sort_by?: components["schemas"]["TraceSortField"]; + sort_dir?: components["schemas"]["TraceSortDirection"]; }; header?: never; path?: never; @@ -81321,7 +81280,7 @@ export interface operations { }; requestBody?: never; responses: { - /** @description Successful Response */ + /** @description Success */ 200: { headers: { [name: string]: unknown; @@ -81330,47 +81289,202 @@ export interface operations { "application/json": components["schemas"]["TracePage"]; }; }; - /** @description Validation Error */ + /** @description Trace API problem */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ 422: { headers: { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HTTPValidationError"]; + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; }; }; }; }; - ingest_otlp_traces_v1_traces_post: { + trace_ingest: { parameters: { query?: never; header?: never; path?: never; cookie?: never; }; - requestBody?: never; + requestBody: { + content: { + "application/json": Record; + "application/x-protobuf": string; + }; + }; responses: { - /** @description Successful Response */ + /** @description OTLP protocol response */ 200: { headers: { [name: string]: unknown; }; content: { - "application/json": unknown; + "application/json": Record; + "application/x-protobuf": string; + }; + }; + /** @description OTLP protocol response */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": Record; + "application/x-protobuf": string; + }; + }; + /** @description OTLP protocol response */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": Record; + "application/x-protobuf": string; + }; + }; + /** @description OTLP protocol response */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": Record; + "application/x-protobuf": string; + }; + }; + /** @description OTLP protocol response */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": Record; + "application/x-protobuf": string; + }; + }; + /** @description OTLP protocol response */ + 429: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": Record; + "application/x-protobuf": string; + }; + }; + /** @description OTLP protocol response */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": Record; + "application/x-protobuf": string; + }; + }; + /** @description OTLP protocol response */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": Record; + "application/x-protobuf": string; }; }; }; }; - agent_trace_histogram_v1_traces_histogram_get: { + trace_histogram: { parameters: { query?: { - /** @description Free text and key:value filters, e.g. `agent:research* -status:ok "book a flight"`. Keys: name, agent, status, model, input, trace_id, service, team and attr.. `*` globs and a leading `-` negates */ + start_ms?: number | null; + end_ms?: number | null; + as_of_ms?: number | null; q?: string; buckets?: number; - /** @description Window start, unix ms. Default: 24h ago */ - start_ms?: number | null; - /** @description Window end, unix ms. Default: now */ - end_ms?: number | null; }; header?: never; path?: never; @@ -81378,7 +81492,7 @@ export interface operations { }; requestBody?: never; responses: { - /** @description Successful Response */ + /** @description Success */ 200: { headers: { [name: string]: unknown; @@ -81387,18 +81501,99 @@ export interface operations { "application/json": components["schemas"]["TraceHistogram"]; }; }; - /** @description Validation Error */ + /** @description Trace API problem */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ 422: { headers: { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HTTPValidationError"]; + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; }; }; }; }; - query_agent_traces_v1_traces_query_post: { + trace_query: { parameters: { query?: never; header?: never; @@ -81411,7 +81606,7 @@ export interface operations { }; }; responses: { - /** @description Successful Response */ + /** @description Success */ 200: { headers: { [name: string]: unknown; @@ -81420,18 +81615,99 @@ export interface operations { "application/json": components["schemas"]["TraceSQLResponse"]; }; }; - /** @description Validation Error */ + /** @description Trace API problem */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ 422: { headers: { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HTTPValidationError"]; + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; }; }; }; }; - help_agent_trace_queries_v1_traces_query_help_get: { + trace_query_help: { parameters: { query?: never; header?: never; @@ -81440,7 +81716,7 @@ export interface operations { }; requestBody?: never; responses: { - /** @description Successful Response */ + /** @description Success */ 200: { headers: { [name: string]: unknown; @@ -81449,29 +81725,117 @@ export interface operations { "application/json": components["schemas"]["TraceQueryHelp"]; }; }; + /** @description Trace API problem */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; }; }; - agent_trace_values_v1_traces_values__field__get: { + trace_values: { parameters: { query?: { - /** @description Free text and key:value filters, e.g. `agent:research* -status:ok "book a flight"`. Keys: name, agent, status, model, input, trace_id, service, team and attr.. `*` globs and a leading `-` negates */ + start_ms?: number | null; + end_ms?: number | null; + as_of_ms?: number | null; q?: string; contains?: string; limit?: number; - /** @description Window start, unix ms. Default: 24h ago */ - start_ms?: number | null; - /** @description Window end, unix ms. Default: now */ - end_ms?: number | null; }; header?: never; path: { - field: "name" | "agent" | "status" | "model" | "input" | "trace_id" | "service" | "team"; + field: components["schemas"]["RunField"]; }; cookie?: never; }; requestBody?: never; responses: { - /** @description Successful Response */ + /** @description Success */ 200: { headers: { [name: string]: unknown; @@ -81480,67 +81844,338 @@ export interface operations { "application/json": components["schemas"]["RunValues"]; }; }; - /** @description Validation Error */ + /** @description Trace API problem */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ 422: { headers: { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HTTPValidationError"]; + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; }; }; }; }; - get_agent_trace_v1_traces__trace_id__get: { + trace_get: { parameters: { - query?: { - trace_ref?: string; - cursor?: string | null; - page_size?: number | null; - }; + query?: never; header?: never; path: { - trace_id: string; + id: string; }; cookie?: never; }; requestBody?: never; responses: { - /** @description Successful Response */ + /** @description Success */ 200: { headers: { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["Trace"]; + "application/json": components["schemas"]["TraceMetadata"]; }; }; - /** @description Validation Error */ + /** @description Trace API problem */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ 422: { headers: { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HTTPValidationError"]; + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; }; }; }; }; - get_agent_trace_span_v1_traces__trace_id__spans__span_id__get: { + trace_spans: { parameters: { query?: { - trace_ref?: string; + cursor?: string | null; + page_size?: number; }; header?: never; path: { - trace_id: string; + id: string; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Success */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["TraceSpansPage"]; + }; + }; + /** @description Trace API problem */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + }; + }; + trace_span: { + parameters: { + query?: never; + header?: never; + path: { + id: string; span_id: string; }; cookie?: never; }; requestBody?: never; responses: { - /** @description Successful Response */ + /** @description Success */ 200: { headers: { [name: string]: unknown; @@ -81549,33 +82184,113 @@ export interface operations { "application/json": components["schemas"]["SpanDetail"]; }; }; - /** @description Validation Error */ + /** @description Trace API problem */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ 422: { headers: { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HTTPValidationError"]; + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; }; }; }; }; - get_agent_trace_span_error_v1_traces__trace_id__spans__span_id__error_get: { + trace_error: { parameters: { query?: { - trace_ref?: string; cursor?: string | null; }; header?: never; path: { - trace_id: string; + id: string; span_id: string; }; cookie?: never; }; requestBody?: never; responses: { - /** @description Successful Response */ + /** @description Success */ 200: { headers: { [name: string]: unknown; @@ -81584,13 +82299,94 @@ export interface operations { "application/json": components["schemas"]["SpanErrorPage"]; }; }; - /** @description Validation Error */ + /** @description Trace API problem */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 401: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 403: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 404: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 413: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ 422: { headers: { [name: string]: unknown; }; content: { - "application/json": components["schemas"]["HTTPValidationError"]; + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 500: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 501: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; + }; + }; + /** @description Trace API problem */ + 503: { + headers: { + [name: string]: unknown; + }; + content: { + "application/problem+json": components["schemas"]["TraceProblem"]; }; }; };