Merge pull request #20702 from emerzon/fix/issue-20698-stream-chunk-thinking-blocks

fix(streaming): preserve interleaved thinking/redacted_thinking blocks
This commit is contained in:
Sameer Kankute 2026-02-09 16:32:43 +05:30 • committed by GitHub
commit f929461fc6
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
19 changed files with 1589 additions and 260 deletions

View file

@ -309,7 +309,7 @@ Support for more providers. Missing a provider or LLM Platform, raise a [feature
| [Deepgram (`deepgram`)](https://docs.litellm.ai/docs/providers/deepgram) | ✅ | ✅ | ✅ | | | ✅ | | | | |
| [DeepInfra (`deepinfra`)](https://docs.litellm.ai/docs/providers/deepinfra) | ✅ | ✅ | ✅ | | | | | | | |
| [Deepseek (`deepseek`)](https://docs.litellm.ai/docs/providers/deepseek) | ✅ | ✅ | ✅ | | | | | | | |
| [ElevenLabs (`elevenlabs`)](https://docs.litellm.ai/docs/providers/elevenlabs) | ✅ | ✅ | ✅ | | | | ✅ | | | |
| [ElevenLabs (`elevenlabs`)](https://docs.litellm.ai/docs/providers/elevenlabs) | ✅ | ✅ | ✅ | | | ✅ | ✅ | | | |
| [Empower (`empower`)](https://docs.litellm.ai/docs/providers/empower) | ✅ | ✅ | ✅ | | | | | | | |
| [Fal AI (`fal_ai`)](https://docs.litellm.ai/docs/providers/fal_ai) | ✅ | ✅ | ✅ | | ✅ | | | | | |
| [Featherless AI (`featherless_ai`)](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | | | | | | | |

View file

@ -0,0 +1,220 @@
---
slug: fastapi-middleware-performance
title: "Your Middleware Could Be a Bottleneck"
date: 2026-02-07T10:00:00
authors:
- name: Krrish Dholakia
title: "CEO, LiteLLM"
url: https://www.linkedin.com/in/krish-d/
image_url: https://pbs.twimg.com/profile_images/1298587542745358340/DZv3Oj-h_400x400.jpg
- name: Ishaan Jaff
title: "CTO, LiteLLM"
url: https://www.linkedin.com/in/reffajnaahsi/
image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg
- name: Ryan Crabbe
title: "Performance Engineer, LiteLLM"
url: https://www.linkedin.com/in/ryan-crabbe-0b9687214
image_url: https://media.licdn.com/dms/image/v2/D5603AQHt1t9Z4BJ6Gw/profile-displayphoto-shrink_400_400/profile-displayphoto-shrink_400_400/0/1724453682340?e=1772064000&v=beta&t=VXdmr13rsNB05wyA2F1TENOB5UuDHUZ0FCHTolNyR5M
description: "How we improved LiteLLM proxy latency and throughput by replacing a single middleware base class"
tags: [performance, fastapi, middleware]
hide_table_of_contents: false
---
import { BaseHTTPMiddlewareAnimation, PureASGIAnimation, BenchmarkVisualization } from '@site/src/components/MiddlewareDiagrams';
> How we improved LiteLLM proxy latency and throughput by replacing a single, simple middleware base class
---
## Our Setup
The LiteLLM proxy server has two middleware layers. The first is Starlette's `CORSMiddleware` (re-exported by FastAPI), which is a pure ASGI middleware. Then we have a simple BaseHTTPMiddleware called PrometheusAuthMiddleware.
The job of `PrometheusAuthMiddleware` is to authenticate requests to the `/metrics` endpoint. It's not on by default, you enable it with a flag in your proxy config:
<details>
<summary>Proxy config flag</summary>
```yaml
litellm_settings:
require_auth_for_metrics_endpoint: true
```
</details>
The middleware checks two things: is the request hitting `/metrics`, and is auth even enabled? If both checks fail, which they do for the vast majority of requests, it just passes the request through unchanged.
<details>
<summary>PrometheusAuthMiddleware source</summary>
```python
class PrometheusAuthMiddleware(BaseHTTPMiddleware):
async def dispatch(self, request: Request, call_next):
if self._is_prometheus_metrics_endpoint(request):
if self._should_run_auth_on_metrics_endpoint() is True:
try:
await user_api_key_auth(request=request, api_key=...)
except Exception as e:
return JSONResponse(status_code=401, content=...)
response = await call_next(request)
return response
@staticmethod
def _is_prometheus_metrics_endpoint(request: Request):
if "/metrics" in request.url.path:
return True
return False
```
</details>
Looks harmless. Subclass `BaseHTTPMiddleware`, implement `dispatch()`, done. This is what you will see in Starlette's documentation<sup>[1](#footnote-1)</sup>.
{/* truncate */}
---
## What BaseHTTPMiddleware Actually Does
When you write a `dispatch()` method, you'd expect the request to flow straight through your function and out the other side. What actually happens is much more involved.
On every request, even a pure passthrough (meaning nothing happens), `BaseHTTPMiddleware` creates **7 intermediate objects and tasks**:
<BaseHTTPMiddlewareAnimation />
It wraps the request in a new object to track body state, creates a synchronization event, allocates an in-memory channel to pass messages between your middleware and the inner app, sets up a task group to manage the lifecycle, and then runs your actual route handler in a *separate background task* when you call `call_next()`. The response body then flows back through that in-memory channel, gets re-wrapped in a streaming response object, and finally reaches the caller. That's a lot.
For a middleware that for us, does nothing on 99.9% of requests, paying this cost doesn't make sense.
Compare that to a pure ASGI middleware, which we can have just check the request path and continue along.
<PureASGIAnimation />
Our middleware is doing something really simple. For the vast majority of requests it doesn't need to do anything at all but just let the request pass through. It doesn't need task groups, memory streams, or cancel scopes. It needs a function call.
---
## Comparing Both
We replaced the `BaseHTTPMiddleware` subclass with a pure ASGI middleware. To benchmark the difference, we used Apache Bench<sup>[2](#footnote-2)</sup> to compare both configurations of LiteLLM's middleware stack: the old setup (1 pure ASGI + 1 `BaseHTTPMiddleware`) against the new setup (2 pure ASGI).
A minimal FastAPI app serves `GET /health` → `PlainTextResponse("ok")`. The endpoint does zero work to isolate the middleware overhead: any difference between configs is purely the cost of the middleware plumbing itself. Both middlewares are just calling the next layer. Same work, different base class.
Apache Bench (`ab`) fires requests at the server with 1,000 concurrent connections and a single uvicorn worker. One worker means one event loop, so the benchmark directly measures how each middleware design handles concurrent load on a single thread.
<BenchmarkVisualization />
<details>
<summary>Try it yourself</summary>
Save the script below as `benchmark_middleware.py`, then run:
```bash
# Terminal 1 — start the "before" server (1 ASGI + 1 BaseHTTPMiddleware)
python benchmark_middleware.py --middleware mixed
# Terminal 2 — benchmark it
ab -n 50000 -c 1000 http://localhost:8000/health
# Stop the server, then start the "after" server (2x pure ASGI)
python benchmark_middleware.py --middleware asgi
# Terminal 2 — benchmark again
ab -n 50000 -c 1000 http://localhost:8000/health
```
```python
import argparse
import uvicorn
from fastapi import FastAPI
from fastapi.responses import PlainTextResponse
from starlette.middleware.base import BaseHTTPMiddleware
from starlette.requests import Request
from starlette.types import ASGIApp, Receive, Scope, Send
class NoOpBaseHTTPMiddleware(BaseHTTPMiddleware):
async def dispatch(self, request: Request, call_next):
return await call_next(request)
class NoOpPureASGIMiddleware:
def __init__(self, app: ASGIApp) -> None:
self.app = app
async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None:
await self.app(scope, receive, send)
def create_app(middleware_type: str | None = None, layers: int = 2) -> FastAPI:
app = FastAPI()
@app.get("/health")
async def health():
return PlainTextResponse("ok")
if middleware_type == "mixed":
app.add_middleware(NoOpBaseHTTPMiddleware)
app.add_middleware(NoOpPureASGIMiddleware)
elif middleware_type == "asgi":
for _ in range(layers):
app.add_middleware(NoOpPureASGIMiddleware)
return app
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument("--middleware", choices=["asgi", "mixed"], default=None)
parser.add_argument("--layers", type=int, default=2)
parser.add_argument("--port", type=int, default=8000)
args = parser.parse_args()
app = create_app(middleware_type=args.middleware, layers=args.layers)
uvicorn.run(app, host="0.0.0.0", port=args.port, workers=1, log_level="warning")
```
</details>
---
## Our Change
Here's what we replaced it with:
```python
class PrometheusAuthMiddleware:
def __init__(self, app: ASGIApp) -> None:
self.app = app
async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None:
if scope["type"] != "http" or "/metrics" not in scope.get("path", ""):
await self.app(scope, receive, send)
return
if litellm.require_auth_for_metrics_endpoint is True:
request = Request(scope, receive)
api_key = request.headers.get("Authorization") or ""
try:
await user_api_key_auth(request=request, api_key=api_key)
except Exception as e:
# send 401 directly via ASGI protocol
...
return
await self.app(scope, receive, send)
```
For the 99.9% of requests that aren't hitting `/metrics`, the middleware is now one dict lookup, one string check, and one function call. No objects allocated, no tasks spawned.
It's important to evaluate if the tools you're using are the right fit for the job as your software grows and handles more responsiblity. We're now putting in a static analysis check to prevent this from happening again with any newly introduced middlewares. If we find the use case is necessary then that's okay and we'll reevalute but for everything LiteLLM needs to do at the moment it's not.
This middleware change was one part of a broader optimization effort on the LiteLLM proxy. Across all optimizations combined, we've measured about a **30% reduction in proxy overhead** over the past two weeks.
---
<a id="footnote-1"></a>
<sup>1</sup> [Starlette Middleware — BaseHTTPMiddleware](https://starlette.dev/middleware/#basehttpmiddleware)
<a id="footnote-2"></a>
<sup>2</sup> [Apache HTTP server benchmarking tool (`ab`)](https://httpd.apache.org/docs/2.4/programs/ab.html)

View file

@ -0,0 +1,133 @@
import React, { useState, useEffect, useCallback, useRef } from 'react';
import styles from './styles.module.css';
interface Stage {
label: string;
subtitle: string;
code: string;
}
const STAGES: Stage[] = [
{
label: 'Request Wrapping',
subtitle: '_CachedRequest',
code: 'request = _CachedRequest(scope, receive)',
},
{
label: 'Sync Event',
subtitle: 'anyio.Event()',
code: 'response_sent = anyio.Event()',
},
{
label: 'Memory Stream',
subtitle: 'create_memory_object_stream()',
code: 'send_stream, recv_stream = anyio.create_memory_object_stream()',
},
{
label: 'Task Group',
subtitle: 'create_task_group()',
code: 'async with anyio.create_task_group() as task_group:',
},
{
label: 'Background Task',
subtitle: 'task_group.start_soon(coro)',
code: 'task_group.start_soon(coro) # app runs in separate task',
},
{
label: 'Nested Task Group',
subtitle: 'receive_or_disconnect()',
code: 'async with anyio.create_task_group() as task_group: ...',
},
{
label: 'Response Wrapping',
subtitle: '_StreamingResponse',
code: 'response = _StreamingResponse(status_code=..., content=body_stream())',
},
];
const INTERVAL_MS = 1200;
const PAUSE_MS = 600;
export default function BaseHTTPMiddlewareAnimation() {
const [activeStage, setActiveStage] = useState(0);
const [paused, setPaused] = useState(false);
const [expandedStage, setExpandedStage] = useState<number | null>(null);
const timerRef = useRef<ReturnType<typeof setTimeout> | null>(null);
const clearTimer = useCallback(() => {
if (timerRef.current !== null) {
clearTimeout(timerRef.current);
timerRef.current = null;
}
}, []);
useEffect(() => {
if (paused) return;
const advance = () => {
setActiveStage((prev) => {
const next = (prev + 1) % STAGES.length;
// If wrapping around, add extra pause
if (next === 0) {
timerRef.current = setTimeout(() => {
timerRef.current = setTimeout(advance, INTERVAL_MS);
}, PAUSE_MS);
return next;
}
timerRef.current = setTimeout(advance, INTERVAL_MS);
return next;
});
};
timerRef.current = setTimeout(advance, INTERVAL_MS);
return clearTimer;
}, [paused, clearTimer]);
const handleStageClick = (index: number) => {
clearTimer();
setPaused(true);
setActiveStage(index);
if (expandedStage === index) {
// Close panel and resume
setExpandedStage(null);
setPaused(false);
} else {
setExpandedStage(index);
}
};
return (
<div className={styles.pipelineWrapper}>
<div className={styles.pipelineLabel}>7 steps per request</div>
<div className={styles.pipeline}>
{STAGES.map((stage, i) => (
<div className={styles.stageWrapper} key={i}>
<div
className={`${styles.stage} ${activeStage === i ? styles.stageActive : ''}`}
onClick={() => handleStageClick(i)}
role="button"
tabIndex={0}
onKeyDown={(e) => {
if (e.key === 'Enter' || e.key === ' ') handleStageClick(i);
}}
>
<div className={styles.stageNumber}>{i + 1}</div>
<div className={styles.stageLabel}>{stage.label}</div>
<div className={styles.stageSubtitle}>{stage.subtitle}</div>
</div>
</div>
))}
</div>
<div
className={`${styles.codePanel} ${expandedStage !== null ? styles.codePanelOpen : ''}`}
>
{expandedStage !== null && (
<pre className={styles.codePanelCode}>
<code>{STAGES[expandedStage].code}</code>
</pre>
)}
</div>
</div>
);
}

View file

@ -0,0 +1,337 @@
import React, { useState, useEffect, useRef, useCallback } from 'react';
import styles from './styles.module.css';
/* ── Constants ── */
const TOTAL_REQUESTS = 50_000;
const DURATION_AFTER_MS = 8_000; // "After" column finishes in 8s
const DURATION_BEFORE_MS = 13_920; // 74% slower → 8000 * 1.74
const TICK_MS = 50;
const RESET_PAUSE_MS = 2_000;
const MAX_DOTS = 14;
const BEFORE_RPS = 3_785;
const AFTER_RPS = 6_577;
const BEFORE_P50 = 21;
const AFTER_P50 = 13;
const BEFORE_LAYERS = [
{ label: 'ab client', warning: false },
{ label: 'uvicorn \u00B7 1 worker', warning: false },
{ label: 'ASGI Middleware', warning: false },
{ label: 'BaseHTTPMiddleware', warning: true },
{ label: 'GET /health \u2192 "ok"', warning: false },
];
const AFTER_LAYERS = [
{ label: 'ab client', warning: false },
{ label: 'uvicorn \u00B7 1 worker', warning: false },
{ label: 'ASGI Middleware', warning: false },
{ label: 'ASGI Middleware', warning: false },
{ label: 'GET /health \u2192 "ok"', warning: false },
];
const BENCHMARK_RUNS = [
{ config: 'Before (1 ASGI + 1 BaseHTTP)', run: 1, rps: 3596, p50: 21 },
{ config: 'Before (1 ASGI + 1 BaseHTTP)', run: 2, rps: 3599, p50: 21 },
{ config: 'Before (1 ASGI + 1 BaseHTTP)', run: 3, rps: 4161, p50: 21 },
{ config: 'After (2x Pure ASGI)', run: 1, rps: 6504, p50: 13 },
{ config: 'After (2x Pure ASGI)', run: 2, rps: 6631, p50: 13 },
{ config: 'After (2x Pure ASGI)', run: 3, rps: 6595, p50: 13 },
];
/* ── Dot type ── */
interface Dot {
id: number;
progress: number; // 0..1 (top to bottom)
}
/* ── Component ── */
export default function BenchmarkVisualization() {
const [elapsed, setElapsed] = useState(0);
const [running, setRunning] = useState(false);
const [afterDone, setAfterDone] = useState(false);
const [beforeDone, setBeforeDone] = useState(false);
const [tableOpen, setTableOpen] = useState(false);
const [beforeDots, setBeforeDots] = useState<Dot[]>([]);
const [afterDots, setAfterDots] = useState<Dot[]>([]);
const dotIdRef = useRef(0);
const observerRef = useRef<IntersectionObserver | null>(null);
const wrapperRef = useRef<HTMLDivElement | null>(null);
const timerRef = useRef<ReturnType<typeof setInterval> | null>(null);
const hasStartedRef = useRef(false);
const beforeProgress = Math.min(elapsed / DURATION_BEFORE_MS, 1);
const afterProgress = Math.min(elapsed / DURATION_AFTER_MS, 1);
const beforeCompleted = Math.round(beforeProgress * TOTAL_REQUESTS);
const afterCompleted = Math.round(afterProgress * TOTAL_REQUESTS);
const beforeCurrentRPS = running && !beforeDone
? Math.round(BEFORE_RPS * (0.9 + Math.random() * 0.2))
: beforeDone ? 0 : 0;
const afterCurrentRPS = running && !afterDone
? Math.round(AFTER_RPS * (0.9 + Math.random() * 0.2))
: afterDone ? 0 : 0;
const reset = useCallback(() => {
setElapsed(0);
setAfterDone(false);
setBeforeDone(false);
setBeforeDots([]);
setAfterDots([]);
dotIdRef.current = 0;
}, []);
// Start/restart loop
const startSimulation = useCallback(() => {
reset();
setRunning(true);
}, [reset]);
// IntersectionObserver to auto-start on scroll
useEffect(() => {
observerRef.current = new IntersectionObserver(
([entry]) => {
if (entry.isIntersecting && !hasStartedRef.current) {
hasStartedRef.current = true;
startSimulation();
}
},
{ threshold: 0.3 }
);
if (wrapperRef.current) {
observerRef.current.observe(wrapperRef.current);
}
return () => {
observerRef.current?.disconnect();
};
}, [startSimulation]);
// Main tick
useEffect(() => {
if (!running) return;
timerRef.current = setInterval(() => {
setElapsed((prev) => {
const next = prev + TICK_MS;
if (next >= DURATION_AFTER_MS) setAfterDone(true);
if (next >= DURATION_BEFORE_MS) setBeforeDone(true);
// Both done → schedule reset
if (next >= DURATION_BEFORE_MS) {
setTimeout(() => {
startSimulation();
}, RESET_PAUSE_MS);
setRunning(false);
return next;
}
return next;
});
}, TICK_MS);
return () => {
if (timerRef.current) clearInterval(timerRef.current);
};
}, [running, startSimulation]);
// Dot animation
useEffect(() => {
if (!running) return;
const dotInterval = setInterval(() => {
const spawnBefore = !beforeDone && Math.random() < 0.4;
const spawnAfter = !afterDone && Math.random() < 0.65;
if (spawnBefore) {
setBeforeDots((prev) => {
const dots = [...prev, { id: dotIdRef.current++, progress: 0 }];
return dots.slice(-MAX_DOTS);
});
}
if (spawnAfter) {
setAfterDots((prev) => {
const dots = [...prev, { id: dotIdRef.current++, progress: 0 }];
return dots.slice(-MAX_DOTS);
});
}
// Advance existing dots
setBeforeDots((prev) =>
prev
.map((d) => ({ ...d, progress: d.progress + 0.08 }))
.filter((d) => d.progress <= 1)
);
setAfterDots((prev) =>
prev
.map((d) => ({ ...d, progress: d.progress + 0.14 }))
.filter((d) => d.progress <= 1)
);
}, 100);
return () => clearInterval(dotInterval);
}, [running, beforeDone, afterDone]);
const renderFlowStack = (
layers: { label: string; warning: boolean }[],
dots: Dot[],
isBefore: boolean
) => (
<div className={styles.flowStack}>
<div className={styles.dotsCanvas}>
{dots.map((dot) => (
<div
key={dot.id}
className={`${styles.dot} ${isBefore ? styles.dotSlow : styles.dotFast}`}
style={{
top: `${dot.progress * 92}%`,
left: `${48 + Math.sin(dot.id * 1.7) * 12}%`,
opacity: dot.progress > 0.85 ? (1 - dot.progress) * 6 : 0.8,
}}
/>
))}
</div>
{layers.map((layer, i) => (
<React.Fragment key={i}>
{i > 0 && <div className={styles.flowArrow}>&darr;</div>}
<div
className={`${styles.flowLayer} ${layer.warning ? styles.flowLayerWarning : ''}`}
>
{layer.label}
{layer.warning && <span className={styles.overheadTag}>&larr; overhead</span>}
</div>
</React.Fragment>
))}
</div>
);
const formatNum = (n: number) => n.toLocaleString();
return (
<div className={styles.benchmarkWrapper} ref={wrapperRef}>
<div className={styles.benchmarkConfig}>
50,000 requests &middot; 1,000 concurrent &middot; 1 worker
</div>
<div className={styles.benchmarkColumns}>
{/* Before column */}
<div className={styles.benchmarkColumn}>
<div className={`${styles.columnTitle} ${styles.columnTitleBefore}`}>
Before (1 ASGI + 1 BaseHTTP)
{beforeDone && (
<span className={`${styles.doneBadge} ${styles.doneBadgeBefore}`}>done</span>
)}
</div>
{renderFlowStack(BEFORE_LAYERS, beforeDots, true)}
<div className={styles.statsRow}>
<div className={styles.stat}>
<div className={styles.statValue}>{formatNum(beforeCurrentRPS)}</div>
<div className={styles.statLabel}>RPS</div>
</div>
<div className={styles.stat}>
<div className={styles.statValue}>{formatNum(beforeCompleted)}</div>
<div className={styles.statLabel}>Completed</div>
</div>
<div className={styles.stat}>
<div className={styles.statValue}>{BEFORE_P50}ms</div>
<div className={styles.statLabel}>P50</div>
</div>
</div>
<div className={styles.progressBar}>
<div
className={`${styles.progressFill} ${styles.progressFillBefore}`}
style={{ width: `${beforeProgress * 100}%` }}
/>
</div>
</div>
{/* After column */}
<div className={styles.benchmarkColumn}>
<div className={`${styles.columnTitle} ${styles.columnTitleAfter}`}>
After (2x Pure ASGI)
{afterDone && (
<span className={`${styles.doneBadge} ${styles.doneBadgeAfter}`}>done</span>
)}
</div>
{renderFlowStack(AFTER_LAYERS, afterDots, false)}
<div className={styles.statsRow}>
<div className={styles.stat}>
<div className={styles.statValue}>{formatNum(afterCurrentRPS)}</div>
<div className={styles.statLabel}>RPS</div>
</div>
<div className={styles.stat}>
<div className={styles.statValue}>{formatNum(afterCompleted)}</div>
<div className={styles.statLabel}>Completed</div>
</div>
<div className={styles.stat}>
<div className={styles.statValue}>{AFTER_P50}ms</div>
<div className={styles.statLabel}>P50</div>
</div>
</div>
<div className={styles.progressBar}>
<div
className={`${styles.progressFill} ${styles.progressFillAfter}`}
style={{ width: `${afterProgress * 100}%` }}
/>
</div>
</div>
</div>
{/* Summary stats */}
<div className={styles.summaryStats}>
<div className={styles.summaryItem}>
<div className={styles.summaryValue}>+74%</div>
<div className={styles.summaryLabel}>Throughput (RPS)</div>
</div>
<div className={styles.summaryItem}>
<div className={styles.summaryValue}>-38%</div>
<div className={styles.summaryLabel}>Median Latency (P50)</div>
</div>
</div>
{/* Collapsible per-run data */}
<div className={styles.collapsible}>
<button
className={styles.collapsibleToggle}
onClick={() => setTableOpen(!tableOpen)}
>
<span
className={`${styles.collapsibleChevron} ${
tableOpen ? styles.collapsibleChevronOpen : ''
}`}
>
&#9654;
</span>
Per-run data (3 runs each)
</button>
<div
className={`${styles.collapsibleContent} ${
tableOpen ? styles.collapsibleContentOpen : ''
}`}
>
<table className={styles.dataTable}>
<thead>
<tr>
<th>Config</th>
<th>Run</th>
<th>RPS</th>
<th>P50 (ms)</th>
</tr>
</thead>
<tbody>
{BENCHMARK_RUNS.map((row, i) => (
<tr key={i}>
<td>{row.config}</td>
<td>{row.run}</td>
<td>{formatNum(row.rps)}</td>
<td>{row.p50}</td>
</tr>
))}
</tbody>
</table>
</div>
</div>
</div>
);
}

View file

@ -0,0 +1,67 @@
import React, { useState, useEffect, useRef, useCallback } from 'react';
import styles from './styles.module.css';
interface Stage {
label: string;
subtitle: string;
}
const STAGES: Stage[] = [
{ label: 'Scope Check', subtitle: 'scope["type"] != "http"' },
{ label: 'Direct Call', subtitle: 'await self.app(scope, receive, send)' },
];
const INTERVAL_MS = 1200;
const PAUSE_MS = 600;
export default function PureASGIAnimation() {
const [activeStage, setActiveStage] = useState(0);
const timerRef = useRef<ReturnType<typeof setTimeout> | null>(null);
const clearTimer = useCallback(() => {
if (timerRef.current !== null) {
clearTimeout(timerRef.current);
timerRef.current = null;
}
}, []);
useEffect(() => {
const advance = () => {
setActiveStage((prev) => {
const next = (prev + 1) % STAGES.length;
if (next === 0) {
timerRef.current = setTimeout(() => {
timerRef.current = setTimeout(advance, INTERVAL_MS);
}, PAUSE_MS);
return next;
}
timerRef.current = setTimeout(advance, INTERVAL_MS);
return next;
});
};
timerRef.current = setTimeout(advance, INTERVAL_MS);
return clearTimer;
}, [clearTimer]);
return (
<div className={styles.pipelineWrapper}>
<div className={styles.pipelineLabel}>2 steps per request</div>
<div className={`${styles.pipeline} ${styles.pipelineTwoCol}`}>
{STAGES.map((stage, i) => (
<div className={styles.stageWrapper} key={i}>
<div
className={`${styles.stage} ${styles.stageNoClick} ${
activeStage === i ? styles.stageActiveGreen : ''
}`}
>
<div className={styles.stageNumber}>{i + 1}</div>
<div className={styles.stageLabel}>{stage.label}</div>
<div className={styles.stageSubtitle}>{stage.subtitle}</div>
</div>
</div>
))}
</div>
</div>
);
}

View file

@ -0,0 +1,3 @@
export { default as BaseHTTPMiddlewareAnimation } from './BaseHTTPMiddlewareAnimation';
export { default as PureASGIAnimation } from './PureASGIAnimation';
export { default as BenchmarkVisualization } from './BenchmarkVisualization';

View file

@ -0,0 +1,494 @@
/* ── Shared custom properties ── */
:root {
--mw-stage-bg: #f8f9fa;
--mw-stage-border: #dee2e6;
--mw-stage-active-bg: #e8f4fd;
--mw-stage-active-border: #3b82f6;
--mw-stage-green-active-bg: #ecfdf5;
--mw-stage-green-active-border: #10b981;
--mw-dot-color: #3b82f6;
--mw-warning-accent: #ef4444;
--mw-success-accent: #10b981;
--mw-text-primary: #1a1a2e;
--mw-text-secondary: #6b7280;
--mw-code-bg: #f1f5f9;
--mw-panel-bg: #ffffff;
--mw-panel-border: #e5e7eb;
--mw-bar-bg: #e5e7eb;
--mw-arrow-color: #9ca3af;
--mw-column-bg: #fafafa;
--mw-column-border: #e5e7eb;
--mw-layer-bg: #f3f4f6;
--mw-layer-border: #d1d5db;
--mw-layer-warning-bg: #fef2f2;
--mw-layer-warning-border: #fca5a5;
--mw-progress-bg: #e5e7eb;
}
[data-theme='dark'] {
--mw-stage-bg: #1e1e2e;
--mw-stage-border: #374151;
--mw-stage-active-bg: #1e3a5f;
--mw-stage-active-border: #60a5fa;
--mw-stage-green-active-bg: #064e3b;
--mw-stage-green-active-border: #34d399;
--mw-dot-color: #60a5fa;
--mw-warning-accent: #f87171;
--mw-success-accent: #34d399;
--mw-text-primary: #e5e7eb;
--mw-text-secondary: #9ca3af;
--mw-code-bg: #1e293b;
--mw-panel-bg: #111827;
--mw-panel-border: #374151;
--mw-bar-bg: #374151;
--mw-arrow-color: #6b7280;
--mw-column-bg: #111827;
--mw-column-border: #374151;
--mw-layer-bg: #1f2937;
--mw-layer-border: #4b5563;
--mw-layer-warning-bg: #451a1a;
--mw-layer-warning-border: #b91c1c;
--mw-progress-bg: #374151;
}
/* ── Pipeline (shared between BaseHTTP and PureASGI) ── */
.pipelineWrapper {
margin: 1.5rem 0;
}
.pipelineLabel {
text-align: center;
font-size: 0.85rem;
font-weight: 600;
color: var(--mw-text-secondary);
margin-bottom: 0.75rem;
text-transform: uppercase;
letter-spacing: 0.05em;
}
.pipeline {
display: flex;
flex-wrap: wrap;
justify-content: center;
align-items: stretch;
gap: 0.75rem;
padding: 0.5rem 0;
}
.pipelineTwoCol {
max-width: 480px;
margin: 0 auto;
}
.stageWrapper {
display: flex;
align-items: center;
width: 160px;
flex-shrink: 0;
}
.pipelineTwoCol .stageWrapper {
width: 200px;
}
.arrow {
display: none;
}
.stage {
flex: 1;
padding: 0.85rem 0.75rem;
min-height: 100px;
display: flex;
flex-direction: column;
justify-content: center;
background: var(--mw-stage-bg);
border: 2px solid var(--mw-stage-border);
border-radius: 8px;
text-align: center;
cursor: pointer;
transition: background 0.4s ease, border-color 0.4s ease, box-shadow 0.4s ease;
user-select: none;
}
.stage:hover {
border-color: var(--mw-stage-active-border);
}
.stageActive {
background: var(--mw-stage-active-bg);
border-color: var(--mw-stage-active-border);
box-shadow: 0 0 0 3px rgba(59, 130, 246, 0.15);
}
.stageActiveGreen {
background: var(--mw-stage-green-active-bg);
border-color: var(--mw-stage-green-active-border);
box-shadow: 0 0 0 3px rgba(16, 185, 129, 0.15);
}
.stageNoClick {
cursor: default;
}
.stageNumber {
font-size: 0.7rem;
font-weight: 700;
color: var(--mw-text-secondary);
margin-bottom: 0.3rem;
}
.stageLabel {
font-size: 0.85rem;
font-weight: 600;
color: var(--mw-text-primary);
margin-bottom: 0.25rem;
line-height: 1.3;
}
.stageSubtitle {
font-size: 0.72rem;
color: var(--mw-text-secondary);
font-family: 'SFMono-Regular', Consolas, 'Liberation Mono', Menlo, monospace;
word-break: break-word;
line-height: 1.3;
}
/* ── Code panel (accordion) ── */
.codePanel {
max-height: 0;
overflow: hidden;
transition: max-height 0.35s ease, padding 0.35s ease;
background: var(--mw-code-bg);
border-radius: 0 0 8px 8px;
margin-top: 0.5rem;
}
.codePanelOpen {
max-height: 120px;
padding: 0.75rem 1rem;
}
.codePanelCode {
font-family: 'SFMono-Regular', Consolas, 'Liberation Mono', Menlo, monospace;
font-size: 0.8rem;
color: var(--mw-text-primary);
white-space: pre;
margin: 0;
line-height: 1.5;
}
/* ── Benchmark Visualization ── */
.benchmarkWrapper {
margin: 1.5rem 0;
}
.benchmarkConfig {
text-align: center;
font-size: 0.85rem;
color: var(--mw-text-secondary);
margin-bottom: 1rem;
font-weight: 500;
}
.benchmarkColumns {
display: flex;
gap: 1.5rem;
}
.benchmarkColumn {
flex: 1;
background: var(--mw-column-bg);
border: 1px solid var(--mw-column-border);
border-radius: 12px;
padding: 1.25rem;
position: relative;
overflow: hidden;
}
.columnTitle {
font-size: 0.9rem;
font-weight: 700;
color: var(--mw-text-primary);
text-align: center;
margin-bottom: 1rem;
}
.columnTitleBefore {
color: var(--mw-warning-accent);
}
.columnTitleAfter {
color: var(--mw-success-accent);
}
/* ── Request flow stack ── */
.flowStack {
display: flex;
flex-direction: column;
align-items: center;
gap: 0;
position: relative;
min-height: 280px;
}
.flowLayer {
width: 100%;
max-width: 260px;
padding: 0.6rem 0.75rem;
background: var(--mw-layer-bg);
border: 1px solid var(--mw-layer-border);
border-radius: 6px;
text-align: center;
font-size: 0.78rem;
font-weight: 500;
color: var(--mw-text-primary);
position: relative;
z-index: 1;
}
.flowLayerWarning {
background: var(--mw-layer-warning-bg);
border-color: var(--mw-layer-warning-border);
font-weight: 700;
}
.flowArrow {
display: flex;
justify-content: center;
color: var(--mw-arrow-color);
font-size: 0.9rem;
padding: 0.15rem 0;
position: relative;
z-index: 0;
min-height: 20px;
}
.overheadTag {
font-size: 0.65rem;
color: var(--mw-warning-accent);
margin-left: 0.4rem;
}
/* ── Dots layer (canvas for flowing dots) ── */
.dotsCanvas {
position: absolute;
top: 0;
left: 0;
width: 100%;
height: 100%;
pointer-events: none;
z-index: 2;
}
.dot {
position: absolute;
width: 6px;
height: 6px;
border-radius: 50%;
background: var(--mw-dot-color);
opacity: 0.8;
}
.dotSlow {
background: var(--mw-warning-accent);
}
.dotFast {
background: var(--mw-success-accent);
}
/* ── Stats & progress ── */
.statsRow {
display: flex;
justify-content: space-around;
margin-top: 1rem;
padding-top: 0.75rem;
border-top: 1px solid var(--mw-panel-border);
}
.stat {
text-align: center;
}
.statValue {
font-size: 1.1rem;
font-weight: 700;
color: var(--mw-text-primary);
font-variant-numeric: tabular-nums;
}
.statLabel {
font-size: 0.7rem;
color: var(--mw-text-secondary);
text-transform: uppercase;
letter-spacing: 0.04em;
}
.progressBar {
width: 100%;
height: 6px;
background: var(--mw-progress-bg);
border-radius: 3px;
margin-top: 0.75rem;
overflow: hidden;
}
.progressFill {
height: 100%;
border-radius: 3px;
transition: width 0.1s linear;
}
.progressFillBefore {
background: var(--mw-warning-accent);
}
.progressFillAfter {
background: var(--mw-success-accent);
}
/* ── Summary stats below simulation ── */
.summaryStats {
display: flex;
justify-content: center;
gap: 2rem;
margin-top: 1.5rem;
flex-wrap: wrap;
}
.summaryItem {
text-align: center;
padding: 0.75rem 1.25rem;
background: var(--mw-stage-bg);
border-radius: 8px;
border: 1px solid var(--mw-panel-border);
}
.summaryValue {
font-size: 1.5rem;
font-weight: 800;
color: var(--mw-success-accent);
}
.summaryLabel {
font-size: 0.8rem;
color: var(--mw-text-secondary);
margin-top: 0.2rem;
}
/* ── Collapsible table ── */
.collapsible {
margin-top: 1.5rem;
}
.collapsibleToggle {
background: none;
border: 1px solid var(--mw-panel-border);
border-radius: 6px;
padding: 0.5rem 1rem;
cursor: pointer;
font-size: 0.85rem;
color: var(--mw-text-primary);
width: 100%;
text-align: left;
display: flex;
align-items: center;
gap: 0.5rem;
transition: background 0.2s;
}
.collapsibleToggle:hover {
background: var(--mw-stage-bg);
}
.collapsibleChevron {
transition: transform 0.3s ease;
font-size: 0.7rem;
}
.collapsibleChevronOpen {
transform: rotate(90deg);
}
.collapsibleContent {
max-height: 0;
overflow: hidden;
transition: max-height 0.35s ease;
}
.collapsibleContentOpen {
max-height: 600px;
}
.dataTable {
width: 100%;
border-collapse: collapse;
margin-top: 0.75rem;
font-size: 0.85rem;
}
.dataTable th,
.dataTable td {
padding: 0.5rem 0.75rem;
text-align: left;
border-bottom: 1px solid var(--mw-panel-border);
}
.dataTable th {
font-weight: 600;
color: var(--mw-text-secondary);
font-size: 0.75rem;
text-transform: uppercase;
letter-spacing: 0.04em;
}
.dataTable td {
color: var(--mw-text-primary);
font-variant-numeric: tabular-nums;
}
/* ── Reproduce section ── */
.reproduceSection {
margin-top: 1rem;
}
/* ── Done badge ── */
.doneBadge {
display: inline-block;
font-size: 0.75rem;
font-weight: 600;
padding: 0.2rem 0.6rem;
border-radius: 4px;
margin-left: 0.5rem;
}
.doneBadgeBefore {
color: var(--mw-warning-accent);
background: var(--mw-layer-warning-bg);
}
.doneBadgeAfter {
color: var(--mw-success-accent);
background: var(--mw-stage-green-active-bg);
}
/* ── Responsive ── */
@media (max-width: 768px) {
.stageWrapper {
width: 140px;
}
.pipelineTwoCol .stageWrapper {
width: 160px;
}
.benchmarkColumns {
flex-direction: column;
}
.summaryStats {
flex-direction: column;
align-items: center;
}
}

View file

@ -1,6 +1,6 @@
import base64
import time
from typing import TYPE_CHECKING, Any, Dict, List, Literal, Optional, Union, cast
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union, cast
from litellm.types.llms.openai import (
ChatCompletionAssistantContentValue,
@ -326,10 +326,22 @@ class ChunkProcessor:
thinking_blocks: List[
Union["ChatCompletionThinkingBlock", "ChatCompletionRedactedThinkingBlock"]
] = []
combined_thinking_text: Optional[str] = None
data: Optional[str] = None
signature: Optional[str] = None
type: Literal["thinking", "redacted_thinking"] = "thinking"
current_thinking_text_parts: List[str] = []
current_signature: Optional[str] = None
def _flush_thinking_block() -> None:
nonlocal current_thinking_text_parts, current_signature
if len(current_thinking_text_parts) > 0 and current_signature:
thinking_blocks.append(
ChatCompletionThinkingBlock(
type="thinking",
thinking="".join(current_thinking_text_parts),
signature=current_signature,
)
)
current_thinking_text_parts = []
current_signature = None
for chunk in chunks:
choices = chunk["choices"]
for choice in choices:
@ -339,33 +351,25 @@ class ChunkProcessor:
for thinking_block in thinking:
thinking_type = thinking_block.get("type", None)
if thinking_type and thinking_type == "redacted_thinking":
type = "redacted_thinking"
data = thinking_block.get("data", None)
_flush_thinking_block()
redacted_data = thinking_block.get("data", None)
if redacted_data:
thinking_blocks.append(
ChatCompletionRedactedThinkingBlock(
type="redacted_thinking",
data=redacted_data,
)
)
else:
type = "thinking"
thinking_text = thinking_block.get("thinking", None)
if thinking_text:
if combined_thinking_text is None:
combined_thinking_text = ""
combined_thinking_text += thinking_text
current_thinking_text_parts.append(thinking_text)
signature = thinking_block.get("signature", None)
if signature:
current_signature = signature
_flush_thinking_block()
if combined_thinking_text and type == "thinking" and signature:
thinking_blocks.append(
ChatCompletionThinkingBlock(
type=type,
thinking=combined_thinking_text,
signature=signature,
)
)
elif data and type == "redacted_thinking":
thinking_blocks.append(
ChatCompletionRedactedThinkingBlock(
type=type,
data=data,
)
)
_flush_thinking_block()
if len(thinking_blocks) > 0:
return thinking_blocks

View file

@ -1083,36 +1083,6 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
"us.anthropic.claude-opus-4-6-v1:0": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
"cache_read_input_token_cost": 5.5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1.1e-06,
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_above_200k_tokens": 1.1e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.75e-05,
"output_cost_per_token_above_200k_tokens": 4.125e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
"us.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
@ -1263,36 +1233,6 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
"au.anthropic.claude-opus-4-6-v1:0": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
"cache_read_input_token_cost": 5.5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1.1e-06,
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_above_200k_tokens": 1.1e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 200000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.75e-05,
"output_cost_per_token_above_200k_tokens": 4.125e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
"anthropic.claude-sonnet-4-20250514-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_read_input_token_cost": 3e-07,

View file

@ -3095,12 +3095,7 @@ async def list_available_teams(
),
)
if available_teams is None:
raise HTTPException(
status_code=400,
detail={
"error": "No available teams for user to join. See how to set available teams here: https://docs.litellm.ai/docs/proxy/self_serve#all-settings-for-self-serve--sso-flow"
},
)
return []
# filter out teams that the user is already a member of
user_info = await prisma_client.db.litellm_usertable.find_unique(

View file

@ -1083,36 +1083,6 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
"us.anthropic.claude-opus-4-6-v1:0": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
"cache_read_input_token_cost": 5.5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1.1e-06,
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_above_200k_tokens": 1.1e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 1000000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.75e-05,
"output_cost_per_token_above_200k_tokens": 4.125e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
"us.anthropic.claude-opus-4-6-v1": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
@ -1263,36 +1233,6 @@
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
"au.anthropic.claude-opus-4-6-v1:0": {
"cache_creation_input_token_cost": 6.875e-06,
"cache_creation_input_token_cost_above_200k_tokens": 1.375e-05,
"cache_read_input_token_cost": 5.5e-07,
"cache_read_input_token_cost_above_200k_tokens": 1.1e-06,
"input_cost_per_token": 5.5e-06,
"input_cost_per_token_above_200k_tokens": 1.1e-05,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 200000,
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 2.75e-05,
"output_cost_per_token_above_200k_tokens": 4.125e-05,
"search_context_cost_per_query": {
"search_context_size_high": 0.01,
"search_context_size_low": 0.01,
"search_context_size_medium": 0.01
},
"supports_assistant_prefill": false,
"supports_computer_use": true,
"supports_function_calling": true,
"supports_pdf_input": true,
"supports_prompt_caching": true,
"supports_reasoning": true,
"supports_response_schema": true,
"supports_tool_choice": true,
"supports_vision": true,
"tool_use_system_prompt_tokens": 346
},
"anthropic.claude-sonnet-4-20250514-v1:0": {
"cache_creation_input_token_cost": 3.75e-06,
"cache_read_input_token_cost": 3e-07,

View file

@ -158,6 +158,76 @@ def test_get_combined_tool_content():
]
def test_get_combined_thinking_content_preserves_interleaved_blocks():
base_chunk = {
"id": "chatcmpl-123",
"object": "chat.completion.chunk",
"created": 1234567890,
"model": "claude-sonnet-4-20250514",
}
def make_chunk(**delta_kwargs):
return ModelResponseStream(
**base_chunk,
choices=[
StreamingChoices(
index=0,
delta=Delta(**delta_kwargs),
finish_reason=None,
)
],
)
chunks = [
make_chunk(role="assistant", content=None),
make_chunk(
thinking_blocks=[
{"type": "thinking", "thinking": "Step 1 analysis...", "signature": None}
]
),
make_chunk(
thinking_blocks=[
{"type": "thinking", "thinking": None, "signature": "sig_block1"}
]
),
make_chunk(
thinking_blocks=[
{
"type": "redacted_thinking",
"data": "EuoBCoYBGAIi...encrypted...",
}
]
),
make_chunk(
thinking_blocks=[
{"type": "thinking", "thinking": "Step 2 analysis...", "signature": None}
]
),
make_chunk(
thinking_blocks=[
{"type": "thinking", "thinking": None, "signature": "sig_block2"}
]
),
]
thinking_chunks = [
chunk for chunk in chunks if chunk["choices"][0]["delta"].get("thinking_blocks")
]
processor = ChunkProcessor(chunks=chunks)
result = processor.get_combined_thinking_content(thinking_chunks)
assert result is not None
assert len(result) == 3
assert result[0]["type"] == "thinking"
assert result[0]["thinking"] == "Step 1 analysis..."
assert result[0]["signature"] == "sig_block1"
assert result[1]["type"] == "redacted_thinking"
assert result[1]["data"] == "EuoBCoYBGAIi...encrypted..."
assert result[2]["type"] == "thinking"
assert result[2]["thinking"] == "Step 2 analysis..."
assert result[2]["signature"] == "sig_block2"
def test_cache_read_input_tokens_retained():
chunk1 = ModelResponseStream(
id="chatcmpl-95aabb85-c39f-443d-ae96-0370c404d70c",
@ -441,4 +511,4 @@ def test_stream_chunk_builder_anthropic_web_search():
assert usage.prompt_tokens == 50
assert usage.completion_tokens == 27
assert usage.total_tokens == 77
assert usage.server_tool_use['web_search_requests'] == 2
assert usage.server_tool_use['web_search_requests'] == 2

View file

@ -38,6 +38,7 @@ from litellm.proxy.management_endpoints.team_endpoints import (
_transform_teams_to_deleted_records,
_validate_and_populate_member_user_info,
delete_team,
list_available_teams,
router,
team_member_add_duplication_check,
team_member_delete,
@ -5871,3 +5872,37 @@ async def test_validate_and_populate_member_user_info_only_user_id_not_found():
mock_prisma_client.db.litellm_usertable.find_unique.assert_called_once_with(
where={"user_id": "nonexistent-user"}
)
@pytest.mark.asyncio
async def test_list_available_teams_returns_empty_list_when_none_configured():
"""
Test that /team/available returns an empty list when no available teams
are configured, instead of raising an exception.
"""
import litellm
mock_request = MagicMock()
mock_user_key = UserAPIKeyAuth(user_id="test-user", token="fake-token")
with patch(
"litellm.proxy.proxy_server.prisma_client", mock_prisma_client
):
# Case 1: default_internal_user_params is None
original = litellm.default_internal_user_params
litellm.default_internal_user_params = None
result = await list_available_teams(
http_request=mock_request,
user_api_key_dict=mock_user_key,
)
assert result == []
# Case 2: default_internal_user_params exists but has no "available_teams" key
litellm.default_internal_user_params = {"some_other_param": "value"}
result = await list_available_teams(
http_request=mock_request,
user_api_key_dict=mock_user_key,
)
assert result == []
litellm.default_internal_user_params = original

View file

@ -7,13 +7,14 @@ import { columns } from "@/components/molecules/models/columns";
import { getDisplayModelName } from "@/components/view_model/model_name_display";
import { InfoCircleOutlined } from "@ant-design/icons";
import { PaginationState, SortingState } from "@tanstack/react-table";
import { Grid, Select, SelectItem, TabPanel, Text } from "@tremor/react";
import { Skeleton, Spin } from "antd";
import { Grid, TabPanel } from "@tremor/react";
import { Badge, Select, Skeleton, Space, Typography } from "antd";
import debounce from "lodash/debounce";
import { useEffect, useMemo, useState } from "react";
import { useModelsInfo } from "../../hooks/models/useModels";
import { transformModelData } from "../utils/modelDataTransformer";
type ModelViewMode = "all" | "current_team";
const { Text } = Typography;
interface AllModelsTabProps {
selectedModelGroup: string | null;
@ -197,88 +198,95 @@ const AllModelsTab = ({
<div className="flex items-center justify-between">
<div className="flex items-center gap-4">
<Text className="text-lg font-semibold text-gray-900">Current Team:</Text>
{isLoading ? (
<Skeleton.Input active style={{ width: 320, height: 36 }} />
) : (
<Select
className="w-80"
defaultValue="personal"
value={currentTeam === "personal" ? "personal" : currentTeam.team_id}
onValueChange={(value) => {
if (value === "personal") {
setCurrentTeam("personal");
// Reset to page 1 when team changes
setCurrentPage(1);
setPagination((prev: PaginationState) => ({ ...prev, pageIndex: 0 }));
} else {
const team = teams?.find((t) => t.team_id === value);
if (team) {
setCurrentTeam(team);
<div className="w-80">
{isLoading ? (
<Skeleton.Input active block size="large" />
) : (
<Select
style={{ width: "100%" }}
size="large"
defaultValue="personal"
value={currentTeam === "personal" ? "personal" : currentTeam.team_id}
onChange={(value) => {
if (value === "personal") {
setCurrentTeam("personal");
// Reset to page 1 when team changes
setCurrentPage(1);
setPagination((prev: PaginationState) => ({ ...prev, pageIndex: 0 }));
} else {
const team = teams?.find((t) => t.team_id === value);
if (team) {
setCurrentTeam(team);
// Reset to page 1 when team changes
setCurrentPage(1);
setPagination((prev: PaginationState) => ({ ...prev, pageIndex: 0 }));
}
}
}
}}
>
<SelectItem value="personal">
<div className="flex items-center gap-2">
<div className="w-2 h-2 bg-blue-500 rounded-full"></div>
<span className="font-medium">Personal</span>
</div>
</SelectItem>
{isLoadingTeams ? (
<SelectItem value="loading">
<div className="flex items-center gap-2">
<Spin size="small" />
<span className="font-medium text-gray-500">Loading teams...</span>
</div>
</SelectItem>
) : (
teams
?.filter((team) => team.team_id)
.map((team) => (
<SelectItem key={team.team_id} value={team.team_id}>
<div className="flex items-center gap-2">
<div className="w-2 h-2 bg-green-500 rounded-full"></div>
<span className="font-medium">
{team.team_alias
? `${team.team_alias.slice(0, 30)}...`
: `Team ${team.team_id.slice(0, 30)}...`}
</span>
</div>
</SelectItem>
))
)}
</Select>
)}
}}
loading={isLoadingTeams}
options={[
{
value: "personal",
label: (
<Space direction="horizontal" align="center">
<Badge color="blue" size="small" />
<Text style={{ fontSize: 16 }}>Personal</Text>
</Space>
),
},
...(teams
?.filter((team) => team.team_id)
.map((team) => ({
value: team.team_id,
label: (
<Space direction="horizontal" align="center">
<Badge color="green" size="small" />
<Text ellipsis style={{ fontSize: 16 }}>
{team.team_alias ? team.team_alias : team.team_id}
</Text>
</Space>
),
})) ?? []),
]}
/>
)}
</div>
</div>
<div className="flex items-center gap-4">
<Text className="text-lg font-semibold text-gray-900">View:</Text>
{isLoading ? (
<Skeleton.Input active style={{ width: 256, height: 36 }} />
) : (
<Select
className="w-64"
defaultValue="current_team"
value={modelViewMode}
onValueChange={(value) => setModelViewMode(value as "current_team" | "all")}
>
<SelectItem value="current_team">
<div className="flex items-center gap-2">
<div className="w-2 h-2 bg-purple-500 rounded-full"></div>
<span className="font-medium">Current Team Models</span>
</div>
</SelectItem>
<SelectItem value="all">
<div className="flex items-center gap-2">
<div className="w-2 h-2 bg-gray-500 rounded-full"></div>
<span className="font-medium">All Available Models</span>
</div>
</SelectItem>
</Select>
)}
<div className="w-64">
{isLoading ? (
<Skeleton.Input active block size="large" />
) : (
<Select
style={{ width: "100%" }}
size="large"
defaultValue="current_team"
value={modelViewMode}
onChange={(value) => setModelViewMode(value as "current_team" | "all")}
options={[
{
value: "current_team",
label: (
<Space direction="horizontal" align="center">
<Badge color="purple" size="small" />
<Text style={{ fontSize: 16 }}>Current Team Models</Text>
</Space>
),
},
{
value: "all",
label: (
<Space direction="horizontal" align="center">
<Badge color="gray" size="small" />
<Text style={{ fontSize: 16 }}>All Available Models</Text>
</Space>
),
},
]}
/>
)}
</div>
</div>
</div>
@ -382,34 +390,38 @@ const AllModelsTab = ({
{/* Model Name Filter */}
<div className="w-64">
<Select
className="w-full"
value={selectedModelGroup ?? "all"}
onValueChange={(value) => setSelectedModelGroup(value === "all" ? "all" : value)}
onChange={(value) => setSelectedModelGroup(value === "all" ? "all" : value)}
placeholder="Filter by Public Model Name"
>
<SelectItem value="all">All Models</SelectItem>
<SelectItem value="wildcard">Wildcard Models (*)</SelectItem>
{availableModelGroups.map((group, idx) => (
<SelectItem key={idx} value={group}>
{group}
</SelectItem>
))}
</Select>
showSearch
options={[
{ value: "all", label: "All Models" },
{ value: "wildcard", label: "Wildcard Models (*)" },
...availableModelGroups.map((group, idx) => ({
value: group,
label: group,
})),
]}
/>
</div>
{/* Model Access Group Filter */}
<div className="w-64">
<Select
className="w-full"
value={selectedModelAccessGroupFilter ?? "all"}
onValueChange={(value) => setSelectedModelAccessGroupFilter(value === "all" ? null : value)}
onChange={(value) => setSelectedModelAccessGroupFilter(value === "all" ? null : value)}
placeholder="Filter by Model Access Group"
>
<SelectItem value="all">All Model Access Groups</SelectItem>
{availableModelAccessGroups.map((accessGroup, idx) => (
<SelectItem key={idx} value={accessGroup}>
{accessGroup}
</SelectItem>
))}
</Select>
showSearch
options={[
{ value: "all", label: "All Model Access Groups" },
...availableModelAccessGroups.map((accessGroup, idx) => ({
value: accessGroup,
label: accessGroup,
})),
]}
/>
</div>
</div>
)}

View file

@ -50,4 +50,73 @@ describe("transformModelData", () => {
const result = transformModelData(null, mockGetProviderFromModel);
expect(result).toEqual({ data: [] });
});
it("should handle zero cost models correctly", () => {
const rawData = {
data: [
{
model_name: "gemini-2.5-flash",
litellm_params: {
model: "vertex_ai/gemini-2.5-flash",
},
model_info: {
input_cost_per_token: 0.0,
output_cost_per_token: 0.0,
max_tokens: 65535,
max_input_tokens: 1048576,
},
},
],
};
const result = transformModelData(rawData, mockGetProviderFromModel);
// Zero costs should be converted to "0.00" per 1M tokens, not left as 0 or null
expect(result.data[0]).toHaveProperty("input_cost", "0.00");
expect(result.data[0]).toHaveProperty("output_cost", "0.00");
});
it("should handle null cost fields in model_info", () => {
const rawData = {
data: [
{
model_name: "some-model",
litellm_params: {
model: "openai/some-model",
},
model_info: {
input_cost_per_token: null,
output_cost_per_token: null,
max_tokens: 4096,
max_input_tokens: 8192,
},
},
],
};
const result = transformModelData(rawData, mockGetProviderFromModel);
// Null costs should remain null (displayed as "-" in the UI)
expect(result.data[0].input_cost).toBeNull();
expect(result.data[0].output_cost).toBeNull();
});
it("should handle missing model_info", () => {
const rawData = {
data: [
{
model_name: "some-model",
litellm_params: {
model: "openai/some-model",
},
},
],
};
const result = transformModelData(rawData, mockGetProviderFromModel);
// Missing model_info should result in null costs
expect(result.data[0].input_cost).toBeNull();
expect(result.data[0].output_cost).toBeNull();
});
});

View file

@ -15,8 +15,8 @@ export const transformModelData = (rawModelData: any, getProviderFromModel: (mod
let model_info = curr_model?.model_info;
let provider = "";
let input_cost = "Undefined";
let output_cost = "Undefined";
let input_cost: any = null;
let output_cost: any = null;
let max_tokens = "Undefined";
let max_input_tokens = "Undefined";
let cleanedLitellmParams = {};
@ -58,11 +58,11 @@ export const transformModelData = (rawModelData: any, getProviderFromModel: (mod
transformedData[i].litellm_model_name = litellm_model_name;
// Convert Cost in terms of Cost per 1M tokens
if (transformedData[i].input_cost) {
if (transformedData[i].input_cost != null) {
transformedData[i].input_cost = (Number(transformedData[i].input_cost) * 1000000).toFixed(2);
}
if (transformedData[i].output_cost) {
if (transformedData[i].output_cost != null) {
transformedData[i].output_cost = (Number(transformedData[i].output_cost) * 1000000).toFixed(2);
}

View file

@ -211,7 +211,7 @@ export const columns = (
const outputCost = model.output_cost;
// If both costs are missing or undefined, show "-"
if (!inputCost && !outputCost) {
if (inputCost == null && outputCost == null) {
return (
<div className="w-full">
<span className="text-xs text-gray-400">-</span>
@ -223,9 +223,9 @@ export const columns = (
<Tooltip title="Cost per 1M tokens">
<div className="flex flex-col min-w-0 w-full">
{/* Input Cost - Primary */}
{inputCost && <div className="text-xs font-medium text-gray-900 truncate">In: ${inputCost}</div>}
{inputCost != null && <div className="text-xs font-medium text-gray-900 truncate">In: ${inputCost}</div>}
{/* Output Cost - Secondary */}
{outputCost && <div className="text-xs text-gray-500 truncate mt-0.5">Out: ${outputCost}</div>}
{outputCost != null && <div className="text-xs text-gray-500 truncate mt-0.5">Out: ${outputCost}</div>}
</div>
</Tooltip>
);

View file

@ -58,7 +58,8 @@ describe("AvailableTeamsPanel", () => {
renderWithProviders(<AvailableTeamsPanel accessToken="token-123" userID="user-123" />);
await waitFor(() => {
expect(screen.getByText("No available teams to join")).toBeInTheDocument();
expect(screen.getByText(/No available teams to join/i)).toBeInTheDocument();
expect(screen.getByText(/See how to set available teams/i)).toBeInTheDocument();
});
});

View file

@ -113,7 +113,16 @@ const AvailableTeamsPanel: React.FC<AvailableTeamsProps> = ({ accessToken, userI
{availableTeams.length === 0 && (
<TableRow>
<TableCell colSpan={5} className="text-center">
<Text>No available teams to join</Text>
<Text>No available teams to join. See how to set available teams{" "}
<a
href="https://docs.litellm.ai/docs/proxy/self_serve#all-settings-for-self-serve--sso-flow"
target="_blank"
rel="noopener noreferrer"
className="text-blue-500 hover:text-blue-700 underline"
>
here
</a>.
</Text>
</TableCell>
</TableRow>
)}