mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
feat(integrations): add Tickerr callback for LLM failure reporting
Adds Tickerr as a built-in LiteLLM callback. Tickerr is a crowd-sourced
outage radar for AI agents — when an agent hits a 5xx error, it reports
anonymously to Tickerr and receives live signal from other agents hitting
the same issue (including a fallback model recommendation).
Changes:
- litellm/integrations/tickerr.py — TickerrLogger (stdlib-only, no new deps)
- litellm/__init__.py — adds "tickerr" to _custom_logger_compatible_callbacks_literal
- litellm/litellm_core_utils/custom_logger_registry.py — registers TickerrLogger
- litellm/integrations/callback_configs.json — UI config entry
- docs/my-website/docs/observability/tickerr.md — integration docs
Usage:
litellm.callbacks = ["tickerr"]
No API key required. Anonymous. Non-blocking (daemon thread, 5s timeout).
https://tickerr.ai
This commit is contained in:
parent
f318ef03bd
commit
64aab2de56
5 changed files with 337 additions and 0 deletions
107
docs/my-website/docs/observability/tickerr.md
Normal file
107
docs/my-website/docs/observability/tickerr.md
Normal file
|
|
@ -0,0 +1,107 @@
|
|||
import Tabs from '@theme/Tabs';
|
||||
import TabItem from '@theme/TabItem';
|
||||
|
||||
# Tickerr — Outage Radar for AI Agents
|
||||
|
||||
[Tickerr](https://tickerr.ai) is a crowd-sourced outage detector for LLM APIs. When your agent hits a 5xx error or rate limit, Tickerr tells you:
|
||||
|
||||
- How many other agents are seeing the same issue right now
|
||||
- Current signal state: `quiet` → `detecting` → `confirmed` → `recovering`
|
||||
- Which model to fall back to
|
||||
|
||||
**No API key required. Anonymous. Zero overhead on success paths.**
|
||||
|
||||
## Quick Start
|
||||
|
||||
<Tabs>
|
||||
<TabItem value="sdk" label="Python SDK">
|
||||
|
||||
```python
|
||||
import litellm
|
||||
|
||||
litellm.callbacks = ["tickerr"]
|
||||
|
||||
# Every failed LiteLLM call is now reported to Tickerr automatically.
|
||||
# Tickerr fires in a background thread — your agent is never blocked.
|
||||
response = litellm.completion(
|
||||
model="claude-haiku-4-5",
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
)
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
<TabItem value="proxy" label="LiteLLM Proxy config.yaml">
|
||||
|
||||
```yaml
|
||||
model_list:
|
||||
- model_name: claude-haiku
|
||||
litellm_params:
|
||||
model: anthropic/claude-haiku-4-5
|
||||
api_key: os.environ/ANTHROPIC_API_KEY
|
||||
|
||||
litellm_settings:
|
||||
callbacks: ["tickerr"]
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
## What Gets Reported
|
||||
|
||||
Tickerr receives only:
|
||||
|
||||
| Field | Example |
|
||||
|-------|---------|
|
||||
| Provider | `anthropic` |
|
||||
| Model | `claude-haiku-4-5` |
|
||||
| HTTP status code | `529` |
|
||||
| Error type | `overloaded` |
|
||||
| Latency (ms) | `1240` |
|
||||
|
||||
No prompts, no responses, no personal data.
|
||||
|
||||
## What You Get Back
|
||||
|
||||
Each report updates the live signal at [tickerr.ai/agent-reports](https://tickerr.ai/agent-reports).
|
||||
|
||||
To read the signal from your agent, use the [Tickerr MCP server](https://tickerr.ai/mcp-server) `report_incident` tool. It returns a structured response:
|
||||
|
||||
```
|
||||
CURRENT SIGNAL (anthropic/claude-haiku-4-5)
|
||||
Status: CONFIRMED
|
||||
Agents reporting (last 10 min): 14
|
||||
Total reports (last 10 min): 31
|
||||
|
||||
RECOMMENDATION
|
||||
Action: FALLBACK
|
||||
Switch to: gpt-4o-mini (openai)
|
||||
```
|
||||
|
||||
## Optional Configuration
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
os.environ["TICKERR_CLIENT_TIER"] = "pro" # free | pro | team | enterprise | api_pay_as_you_go
|
||||
os.environ["TICKERR_REGION"] = "us-east-1" # optional, for regional breakdown
|
||||
```
|
||||
|
||||
## Signal States
|
||||
|
||||
| State | Meaning |
|
||||
|-------|---------|
|
||||
| `quiet` | No reports in last 10 min |
|
||||
| `detecting` | 1–2 agents reporting |
|
||||
| `confirmed` | 3+ distinct agents — issue verified |
|
||||
| `recovering` | Reports dropping, recovery signals arriving |
|
||||
|
||||
## Opt Out
|
||||
|
||||
[tickerr.ai/mcp/opt-out](https://tickerr.ai/mcp/opt-out)
|
||||
|
||||
## Links
|
||||
|
||||
- [Tickerr](https://tickerr.ai) — live AI status dashboard (90+ tools)
|
||||
- [Agent reports](https://tickerr.ai/agent-reports) — live feed
|
||||
- [Tickerr MCP server](https://tickerr.ai/mcp-server) — 9-tool MCP for agents
|
||||
- [REST API](https://tickerr.ai/api/v1/report) — report without LiteLLM
|
||||
|
|
@ -149,6 +149,7 @@ _custom_logger_compatible_callbacks_literal = Literal[
|
|||
"posthog",
|
||||
"levo",
|
||||
"compression_interception",
|
||||
"tickerr",
|
||||
]
|
||||
cold_storage_custom_logger: Optional[_custom_logger_compatible_callbacks_literal] = None
|
||||
logged_real_time_event_types: Optional[Union[List[str], Literal["*"]]] = None
|
||||
|
|
|
|||
|
|
@ -1,4 +1,25 @@
|
|||
[
|
||||
{
|
||||
"id": "tickerr",
|
||||
"displayName": "Tickerr",
|
||||
"logo": "tickerr.png",
|
||||
"supports_key_team_logging": false,
|
||||
"dynamic_params": {
|
||||
"TICKERR_CLIENT_TIER": {
|
||||
"type": "text",
|
||||
"ui_name": "Client Tier",
|
||||
"description": "Optional. Your LLM plan tier: free, pro, team, enterprise, or api_pay_as_you_go. Used to correlate signals by tier.",
|
||||
"required": false
|
||||
},
|
||||
"TICKERR_REGION": {
|
||||
"type": "text",
|
||||
"ui_name": "Region",
|
||||
"description": "Optional. Your deployment region, e.g. us-east-1. Used for regional signal breakdown.",
|
||||
"required": false
|
||||
}
|
||||
},
|
||||
"description": "Outage radar for AI agents. Reports LLM API failures anonymously to Tickerr and returns live signal from other agents — how many are hitting the same issue and which model to fall back to. No API key required."
|
||||
},
|
||||
{
|
||||
"id": "arize",
|
||||
"displayName": "Arize",
|
||||
|
|
|
|||
206
litellm/integrations/tickerr.py
Normal file
206
litellm/integrations/tickerr.py
Normal file
|
|
@ -0,0 +1,206 @@
|
|||
"""
|
||||
Tickerr — outage radar for AI agents.
|
||||
|
||||
Reports LLM API failures to https://tickerr.ai so agents can
|
||||
see how many other agents are hitting the same issue and get
|
||||
a live routing recommendation (RETRY / RETRY_WITH_DELAY / FALLBACK).
|
||||
|
||||
Zero dependencies beyond stdlib. Anonymous. Non-blocking.
|
||||
|
||||
Usage:
|
||||
litellm.callbacks = ["tickerr"]
|
||||
|
||||
No API key required.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import threading
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
|
||||
_REPORT_URL = "https://tickerr.ai/api/v1/report"
|
||||
_UA = "litellm-tickerr/1.0"
|
||||
|
||||
# Map litellm custom_llm_provider → Tickerr provider slug
|
||||
_PROVIDER_MAP: Dict[str, str] = {
|
||||
"openai": "openai",
|
||||
"anthropic": "anthropic",
|
||||
"google": "google",
|
||||
"vertex_ai": "google",
|
||||
"gemini": "google",
|
||||
"cohere": "cohere",
|
||||
"mistral": "mistral",
|
||||
"groq": "groq",
|
||||
"together_ai": "together",
|
||||
"huggingface": "huggingface",
|
||||
"replicate": "replicate",
|
||||
"deepinfra": "deepinfra",
|
||||
"perplexity": "perplexity",
|
||||
"fireworks_ai": "fireworks",
|
||||
"openrouter": "openrouter",
|
||||
"azure": "azure",
|
||||
"bedrock": "aws",
|
||||
"ai21": "ai21",
|
||||
"cerebras": "cerebras",
|
||||
"xai": "xai",
|
||||
"deepseek": "deepseek",
|
||||
"ollama": "ollama",
|
||||
"nlp_cloud": "nlp_cloud",
|
||||
}
|
||||
|
||||
_ERROR_TYPE_MAP: Dict[int, str] = {
|
||||
429: "rate_limit",
|
||||
529: "overloaded",
|
||||
503: "overloaded",
|
||||
500: "overloaded",
|
||||
408: "timeout",
|
||||
524: "timeout",
|
||||
401: "auth",
|
||||
403: "auth",
|
||||
}
|
||||
|
||||
|
||||
def _normalize_provider(model: str, kwargs: Dict[str, Any]) -> str:
|
||||
custom = (
|
||||
kwargs.get("litellm_params", {}).get("custom_llm_provider")
|
||||
or kwargs.get("custom_llm_provider")
|
||||
or ""
|
||||
)
|
||||
if custom:
|
||||
return _PROVIDER_MAP.get(custom.lower(), custom.lower())
|
||||
if "/" in model:
|
||||
prefix = model.split("/")[0].lower()
|
||||
return _PROVIDER_MAP.get(prefix, prefix)
|
||||
if re.match(r"^claude", model, re.I):
|
||||
return "anthropic"
|
||||
if re.match(r"^gpt|^o[1-9]", model, re.I):
|
||||
return "openai"
|
||||
if re.match(r"^gemini", model, re.I):
|
||||
return "google"
|
||||
if re.match(r"^mistral|^mixtral", model, re.I):
|
||||
return "mistral"
|
||||
if re.match(r"^llama", model, re.I):
|
||||
return "meta"
|
||||
if re.match(r"^command", model, re.I):
|
||||
return "cohere"
|
||||
if re.match(r"^grok", model, re.I):
|
||||
return "xai"
|
||||
if re.match(r"^deepseek", model, re.I):
|
||||
return "deepseek"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _extract_status_code(exception: Optional[BaseException]) -> Optional[int]:
|
||||
if exception is None:
|
||||
return None
|
||||
code = getattr(exception, "status_code", None)
|
||||
if isinstance(code, int):
|
||||
return code
|
||||
if isinstance(code, str) and code.isdigit():
|
||||
return int(code)
|
||||
return None
|
||||
|
||||
|
||||
def _fire_and_forget(payload: Dict[str, Any]) -> None:
|
||||
"""POST to Tickerr in a daemon thread — never blocks the caller."""
|
||||
|
||||
def _send() -> None:
|
||||
try:
|
||||
import json as _json
|
||||
import urllib.request
|
||||
|
||||
data = _json.dumps(payload).encode()
|
||||
req = urllib.request.Request(
|
||||
_REPORT_URL,
|
||||
data=data,
|
||||
headers={"Content-Type": "application/json", "User-Agent": _UA},
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=5):
|
||||
pass
|
||||
except Exception:
|
||||
pass # never crash the caller
|
||||
|
||||
t = threading.Thread(target=_send, daemon=True)
|
||||
t.start()
|
||||
|
||||
|
||||
class TickerrLogger(CustomLogger):
|
||||
"""
|
||||
LiteLLM built-in callback for Tickerr.
|
||||
|
||||
Activated with:
|
||||
litellm.callbacks = ["tickerr"]
|
||||
|
||||
Optional env vars:
|
||||
TICKERR_CLIENT_TIER — "free" | "pro" | "team" | "enterprise" | "api_pay_as_you_go"
|
||||
TICKERR_REGION — e.g. "us-east-1"
|
||||
"""
|
||||
|
||||
def __init__(self, **kwargs: Any) -> None:
|
||||
super().__init__(**kwargs)
|
||||
self.client_tier: Optional[str] = os.environ.get("TICKERR_CLIENT_TIER")
|
||||
self.region: Optional[str] = os.environ.get("TICKERR_REGION")
|
||||
|
||||
# ── sync ──────────────────────────────────────────────────────────────────
|
||||
|
||||
def log_failure_event(
|
||||
self,
|
||||
kwargs: Dict[str, Any],
|
||||
response_obj: Any,
|
||||
start_time: float,
|
||||
end_time: float,
|
||||
) -> None:
|
||||
self._report(kwargs, start_time, end_time, is_resolution=False)
|
||||
|
||||
# ── async ─────────────────────────────────────────────────────────────────
|
||||
|
||||
async def async_log_failure_event(
|
||||
self,
|
||||
kwargs: Dict[str, Any],
|
||||
response_obj: Any,
|
||||
start_time: float,
|
||||
end_time: float,
|
||||
) -> None:
|
||||
self._report(kwargs, start_time, end_time, is_resolution=False)
|
||||
|
||||
# ── internal ──────────────────────────────────────────────────────────────
|
||||
|
||||
def _report(
|
||||
self,
|
||||
kwargs: Dict[str, Any],
|
||||
start_time: float,
|
||||
end_time: float,
|
||||
is_resolution: bool,
|
||||
) -> None:
|
||||
model: str = kwargs.get("model", "") or ""
|
||||
exception: Optional[BaseException] = kwargs.get("exception")
|
||||
latency_ms = round((end_time - start_time) * 1000)
|
||||
|
||||
provider = _normalize_provider(model, kwargs)
|
||||
status_code = _extract_status_code(exception)
|
||||
error_type = _ERROR_TYPE_MAP.get(status_code, "overloaded") if status_code else None
|
||||
|
||||
# Strip provider prefix: "anthropic/claude-3-5-haiku" → "claude-3-5-haiku"
|
||||
model_clean = model.split("/", 1)[-1] if "/" in model else model
|
||||
|
||||
payload: Dict[str, Any] = {
|
||||
"provider": provider,
|
||||
"model": model_clean or None,
|
||||
"is_resolution": is_resolution,
|
||||
"latency_ms": latency_ms,
|
||||
}
|
||||
if status_code is not None:
|
||||
payload["error_code"] = status_code
|
||||
if error_type:
|
||||
payload["error_type"] = error_type
|
||||
if self.client_tier:
|
||||
payload["client_tier"] = self.client_tier
|
||||
if self.region:
|
||||
payload["region"] = self.region
|
||||
|
||||
_fire_and_forget(payload)
|
||||
|
|
@ -45,6 +45,7 @@ from litellm.integrations.posthog import PostHogLogger
|
|||
from litellm.integrations.prometheus import PrometheusLogger
|
||||
from litellm.integrations.s3_v2 import S3Logger
|
||||
from litellm.integrations.sqs import SQSLogger
|
||||
from litellm.integrations.tickerr import TickerrLogger
|
||||
from litellm.integrations.vector_store_integrations.vector_store_pre_call_hook import (
|
||||
VectorStorePreCallHook,
|
||||
)
|
||||
|
|
@ -102,6 +103,7 @@ class CustomLoggerRegistry:
|
|||
"focus": FocusLogger,
|
||||
"vantage": VantageLogger,
|
||||
"posthog": PostHogLogger,
|
||||
"tickerr": TickerrLogger,
|
||||
}
|
||||
|
||||
try:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue