mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-14 23:21:35 +00:00
* feat(proxy): add logging_endpoints package init
* feat(proxy): add POST /v1/callbacks/logs to replay logging payloads through the success/failure callback fan-out
* feat(proxy): register callback_logs_router
* test(proxy): add logging_endpoints test package init
* test(proxy): cover /v1/callbacks/logs replay, admin guard, and partial-failure handling
* refactor(proxy): move callback-logs request/response models to litellm/types/proxy
* refactor(proxy): wrap callback-logs replay in CallbackLogsReplayer class with payload logging
* test(proxy): update callback-logs tests for class-based replayer and separated types
* fix(proxy): cover /v1/callbacks/ in backend component allowlist
The new /v1/callbacks/logs route was dropped by both component
allowlists, failing test_gateway_plus_backend_covers_full_app. It's an
admin-only spend-logging route, so it belongs on the backend (control
plane) alongside the existing /callbacks family.
* refactor(proxy): use builtin dict/list generics in callback-logs endpoint
Switch Dict/List from typing to builtin dict/list to satisfy the ruff
strict-rule budget (UP006).
* refactor(proxy): use builtin dict/list generics in callback-logs types
UP006: builtin generics over typing.Dict/List.
* chore(ui): regenerate schema.d.ts for /v1/callbacks/logs
Run npm run gen:api to add the CallbackLogRecord/CallbackLogsRequest/
CallbackLogsResponse types and the /v1/callbacks/logs path, keeping the
dashboard types in sync with the proxy OpenAPI spec.
* fix(proxy): force stream=False when replaying callback logs
A replayed StandardLoggingPayload is a terminal, fully-aggregated event —
the producer (e.g. the rust realtime gateway) already collected the whole
session before POSTing. Marking the rebuilt Logging object as streaming made
async_success_handler wait for a complete_streaming_response that never
arrives, so the spend log was never written. Realtime sessions now land in
LiteLLM_SpendLogs.
* feat(litellm-rust): CustomLogger callback layer posting to /v1/callbacks/logs
integrations/ mirrors litellm/integrations/: a sync, typed CustomLogger trait
(base contract), a typed StandardLoggingPayload, and LiteLLMPythonProxyAPILogger
— the first concrete logger, owning a bounded channel + background worker that
batches and POSTs to the Python proxy's /v1/callbacks/logs.
* feat(litellm-rust): RealTimeStreaming per-session log collector
1:1 with Python's RealTimeStreaming: observe() accumulates O(1) usage/model/id
per event (never buffers frames); log_messages() builds one StandardLoggingPayload
on session close and fans out to the CustomLogger callbacks. request_id == the
OpenAI realtime session id (sess_…), with the gateway id as fallback.
* feat(litellm-rust): wire realtime logging into the splice (lock-free observe)
The collector is owned on the splice task and observed via a synchronous &mut
callback threaded through providers::realtime::realtime() — no Arc/Mutex/atomic
on the per-frame hot path. On session close the bridge flushes one payload.
AppState carries the registered loggers; main spawns the proxy logger.
* docs(litellm-rust): ai-gateway realtime logging architecture
* docs(litellm-rust): document request-log egress to the LiteLLM control plane
Add a 'Request logging' guide to the ai-gateway README: how to point the gateway
at a LiteLLM proxy via LITELLM_PROXY_BASE_URL (+ LITELLM_MASTER_KEY for the
admin-only /v1/callbacks/logs POST), and the non-blocking / one-payload-per-session
behavior.
* feat(litellm-rust): make log-egress tunables env-overridable
Channel capacity, batch size, and flush interval now read from
LITELLM_LOG_CHANNEL_CAPACITY / LITELLM_LOG_BATCH_SIZE / LITELLM_LOG_FLUSH_INTERVAL_MS,
falling back to the DEFAULT_* consts on missing/invalid/non-positive values.
Grouped behind an EgressTunables::from_env() read once at logger construction.
* docs(litellm-rust): document log-egress tuning env vars
* docs(litellm-rust): require constants in a crate-level constants.rs
Mirror of Python's litellm/constants.py rule — magic numbers and fixed strings
go in src/constants.rs, not inline in feature modules; env-overridable tunables
keep their DEFAULT_* value there.
* refactor(litellm-rust): move ai-gateway constants into constants.rs
Per the new rule: the log-egress defaults (proxy base, ingest path, channel
capacity, batch size, flush interval) and the realtime provider default move to
crates/ai-gateway/src/constants.rs; modules import from it.
* ci: run logging_endpoints tests in the proxy-infra coverage shard
tests/test_litellm/proxy/logging_endpoints wasn't in any coverage-uploading
job, so callback_logs_endpoints.py showed only import-level coverage (~35%) on
codecov/patch despite being ~98% covered locally. Add it to proxy-infra's
test-path so the test is exercised under --cov.
* fix(litellm-rust): hash the master key before logging — never send the raw credential
Greptile/Veria P1: user_api_key_hash was the plaintext LITELLM_MASTER_KEY, which
fans out to spend logs and every callback (Langfuse/Datadog) and could be
recovered from logs. SHA-256 it (auth::hash_token, matching the proxy's
hash_token); the field is named *_hash and the proxy stores it verbatim when it
isn't sk-prefixed, so the DB value is identical with zero plaintext exposure.
* fix(litellm-rust): observe realtime logging on upstream events only
Greptile P1: observe ran on the client->upstream arm too, so an authenticated
client could send a fabricated response.done and inflate its own spend log.
session.created/response.done are server->client events; observe the upstream
arm only.
* feat(proxy): bound callback-logs batch + return per-record failures
Greptile P2: cap /v1/callbacks/logs at MAX_CALLBACK_LOG_RECORDS (default 1000,
env-overridable) so one POST can't trigger an unbounded callback/DB fan-out; and
return per-record {index, error} failures so a caller (the rust gateway) can
distinguish a transient callback error from a structurally bad payload.
* chore(ui): regenerate schema.d.ts for CallbackLogFailure / failures field
* fix(constants): make MAX_CALLBACK_LOG_RECORDS a plain constant
It doesn't need to be env-configurable (only the rust egress tunables are). As an
os.getenv var it tripped tests/documentation_tests/test_env_keys.py, which requires
every env key to be documented in the (separate-repo) config_settings.md. Plain
constant → not scanned → code-quality + documentation checks pass.
* docs(litellm-rust): trim ai-gateway ARCHITECTURE.md to one diagram + notes
* docs(litellm-rust): tighten the README request-logging section
* docs(litellm-rust): ARCHITECTURE.md is just the diagram (gateway = inference, spend = callback)
* docs(litellm-rust): drop em-dashes from the request-logging section
---------
Co-authored-by: Ishaan Jaffer <ishaanjaffer0324@gmail.com>
146 lines
3.3 KiB
Python
146 lines
3.3 KiB
Python
"""Path allowlist for the UI backend (control plane) component.
|
|
|
|
The backend exposes management/admin endpoints consumed by the UI: keys, users,
|
|
teams, orgs, customers, budgets, tags, workflows, model management, spend &
|
|
analytics, settings (router/cache/cost-tracking/fallbacks), SSO/onboarding,
|
|
audit logs, debug, enterprise admin, and UI bootstrap helpers (logo, favicon,
|
|
.well-known config).
|
|
|
|
Anything LLM data-plane is dropped — those run on the gateway component.
|
|
"""
|
|
|
|
BACKEND_PATH_PREFIXES: tuple[str, ...] = (
|
|
# Identity / access
|
|
"/key/",
|
|
"/v2/key/",
|
|
"/user/",
|
|
"/v2/user/",
|
|
"/team/",
|
|
"/v2/team/",
|
|
"/organization/",
|
|
"/customer/",
|
|
"/end_user/",
|
|
"/sso/",
|
|
"/login",
|
|
"/v2/login",
|
|
"/v3/login",
|
|
"/logout",
|
|
"/token",
|
|
"/onboarding/",
|
|
"/audit",
|
|
"/oauth/",
|
|
"/invitation/",
|
|
"/jwt/",
|
|
# Models & routing config
|
|
"/model/",
|
|
"/v1/model/info",
|
|
"/v2/model/",
|
|
"/model_group",
|
|
"/model_access_group/",
|
|
"/model_hub/",
|
|
"/v1/access_group",
|
|
"/access_group/",
|
|
"/router/",
|
|
"/router_settings",
|
|
"/adaptive_router/",
|
|
"/fallback",
|
|
"/fallbacks",
|
|
"/cache_settings",
|
|
"/cost_tracking",
|
|
"/cost/",
|
|
"/credentials",
|
|
"/credential",
|
|
"/provider/budgets",
|
|
# Tools / agents (registry & policy admin)
|
|
"/v1/tool/",
|
|
"/v1/agents",
|
|
# Guardrails admin
|
|
"/v2/guardrails/",
|
|
# MCP server admin + BYOK OAuth flow (UI-initiated) + dynamic per-server endpoints
|
|
"/v1/mcp/",
|
|
"/test/",
|
|
"/{mcp_server_name}/",
|
|
# Budgets / tags / workflows / memory mgmt
|
|
"/budget/",
|
|
"/tag/",
|
|
"/workflow/",
|
|
"/v1/workflows/",
|
|
"/project/",
|
|
"/memory/",
|
|
"/mcp/",
|
|
# Spend / analytics
|
|
"/spend/",
|
|
"/analytics/",
|
|
"/global/",
|
|
"/user_agent",
|
|
"/usage/",
|
|
"/daily/",
|
|
# CloudZero cost-export admin (init / settings / export / dry-run / delete)
|
|
"/cloudzero/",
|
|
# Caching admin
|
|
"/cache/",
|
|
"/caching/",
|
|
# Callbacks / hooks
|
|
"/active/callbacks",
|
|
"/callbacks",
|
|
"/team_callback",
|
|
# Rust data-plane gateway → proxy control-plane API (logging today, auth later)
|
|
"/v1/rust_control_plane/",
|
|
# Alerting / email / IP allowlist
|
|
"/alerting/",
|
|
"/email/",
|
|
"/add/allowed_ip",
|
|
"/delete/allowed_ip",
|
|
"/get/",
|
|
# Enterprise admin
|
|
"/enterprise/",
|
|
# Debug / config / profiling
|
|
"/debug/",
|
|
"/config/",
|
|
"/memory-usage-in-mem-cache",
|
|
"/otel-spans",
|
|
"/lazy/",
|
|
"/in_product_nudges",
|
|
# Admin reload / schedule
|
|
"/reload/",
|
|
"/schedule/",
|
|
"/settings",
|
|
"/update/",
|
|
"/upload/",
|
|
# Dev / admin utilities
|
|
"/utils/",
|
|
# UI bootstrap helpers (assets the dashboard fetches)
|
|
"/get_logo_url",
|
|
"/get_image",
|
|
"/get_favicon",
|
|
"/.well-known/",
|
|
"/litellm/.well-known/",
|
|
"/ui_discovery/",
|
|
"/ui-config",
|
|
"/sso_settings",
|
|
"/public/",
|
|
"/robots.txt",
|
|
# Health (k8s probes)
|
|
"/health",
|
|
# Plugin system
|
|
"/api/plugins",
|
|
"/plugin-proxy/",
|
|
)
|
|
|
|
BACKEND_EXACT_PATHS: frozenset[str] = frozenset(
|
|
{
|
|
"/",
|
|
"/routes",
|
|
"/openapi.json",
|
|
"/docs",
|
|
"/docs/oauth2-redirect",
|
|
"/redoc",
|
|
"/fallback/login",
|
|
}
|
|
)
|
|
|
|
BACKEND_MOUNT_PATHS: frozenset[str] = frozenset(
|
|
{
|
|
"/swagger", # API documentation static assets belong to the backend
|
|
}
|
|
)
|