mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
prototype for the streaming API - file contents
This commit is contained in:
parent
4c06e4379b
commit
6b63082ddf
5 changed files with 967 additions and 0 deletions
307
docs/investigation/oom-file-endpoint.md
Normal file
307
docs/investigation/oom-file-endpoint.md
Normal file
|
|
@ -0,0 +1,307 @@
|
|||
## Issue
|
||||
|
||||
The /v1/files/{file_id}/contents endpoint loads the file in memory before sending it to the client. During Load tests, multiple buffered files in memory lead to high peak memory usage and OOM
|
||||
|
||||
**Summary:**
|
||||
During batch file retrievals with large payloads, worker memory rises sharply and quickly approaches the container limit before worker recycling can occur, indicating that large file responses are being buffered in memory rather than streamed.
|
||||
|
||||
**OOM Error is not reached** but container memory usage reaches peak 98 - 99%. Payload sizes of 65MB used with 1000 requests in parallel, according to the batch_lt.py script provided
|
||||
|
||||
Code Path (file_endpoints.py, line 716):
|
||||
```
|
||||
response = await litellm.afile_content(
|
||||
**{
|
||||
"custom_llm_provider": custom_llm_provider,
|
||||
"file_id": file_id,
|
||||
**data,
|
||||
} # type: ignore
|
||||
)
|
||||
```
|
||||
|
||||
**Steps**
|
||||
- Ran **LiteLLM with 2 worker** and **with** `max_requests_before_restart`.
|
||||
- Use **memray** (e.g. leak mode / `--leaks`) on the **files** path.
|
||||
- **Mocked** file upstream / route so we can send a **~65 MB** payload through the proxy without relying on real provider keys, then attribute memory.
|
||||
|
||||
|
||||
**Potential Resolution**
|
||||
Without changing the current business logic, the potential fix could be:
|
||||
|
||||
- Limitation of the openai SDK which returns BinaryResponse, instead of a streamed response.
|
||||
- We can wrap this API with a StreamingResponse handler, so that this can be mitigated. Again all signs point to this code path in the flamegraph as well.
|
||||
- Needs to be profiled after fix and before release.
|
||||
|
||||
### Before the fix
|
||||
|
||||
**Average Memory Usage**: 3.707 / 4 GiB (92.67%)
|
||||
**Peak Memory Usage**: 3.893 / 4 GiB (97.32%)
|
||||
|
||||
Run Logs
|
||||
```
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 34.76% 3.335GiB / 4GiB 83.37% 201GB / 201GB 269MB / 400MB 80
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 37.94% 3.58GiB / 4GiB 89.50% 206GB / 206GB 269MB / 400MB 83
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 8.29% 3.512GiB / 4GiB 87.81% 206GB / 206GB 269MB / 400MB 83
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 41.50% 3.656GiB / 4GiB 91.39% 214GB / 215GB 269MB / 400MB 83
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 14.97% 3.781GiB / 4GiB 94.52% 214GB / 215GB 269MB / 400MB 83
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 3.68% 3.495GiB / 4GiB 87.37% 236GB / 236GB 318MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 62.47% 3.69GiB / 4GiB 92.24% 236GB / 237GB 318MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 29.04% 3.687GiB / 4GiB 92.18% 240GB / 241GB 318MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 35.19% 3.883GiB / 4GiB 97.08% 243GB / 244GB 318MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 11.32% 3.819GiB / 4GiB 95.47% 243GB / 244GB 318MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 13.22% 3.685GiB / 4GiB 92.12% 244GB / 244GB 318MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 3.89% 3.822GiB / 4GiB 95.54% 244GB / 245GB 318MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 46.25% 3.822GiB / 4GiB 95.54% 244GB / 245GB 318MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 15.11% 3.768GiB / 4GiB 94.19% 246GB / 247GB 319MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 17.94% 3.822GiB / 4GiB 95.54% 246GB / 247GB 319MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 17.71% 3.766GiB / 4GiB 94.15% 251GB / 252GB 319MB / 400MB 82
|
||||
khgokul@instance-20260329-212319:~/litellm-ifood-debug/litellm$ docker stats litellm-litellm-1 --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
c7505b5263d1 litellm-litellm-1 29.78% 3.893GiB / 4GiB 97.32% 252GB / 252GB 319MB / 400MB 82
|
||||
```
|
||||
|
||||
### After the fix
|
||||
|
||||
Test: 1000 concurrent requests with a payload size of 65MB
|
||||
**Average Memory Usage**: 2.56 / 4 GB (64%)
|
||||
**Peak Memory Usage**: 2.621 / 4 GB (65.2%)
|
||||
|
||||
|
||||
Run Logs
|
||||
```
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 19.08% 2.455GiB / 4GiB 61.38% 9.89GB / 9.9GB 43.1MB / 21.1MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 20.34% 2.487GiB / 4GiB 62.18% 11.2GB / 11.3GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 22.11% 2.487GiB / 4GiB 62.18% 11.5GB / 11.6GB 43.1MB / 21.1MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 25.33% 2.492GiB / 4GiB 62.30% 11.7GB / 11.7GB 43.1MB / 21.1MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 32.51% 2.508GiB / 4GiB 62.70% 11.9GB / 11.9GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 24.88% 2.488GiB / 4GiB 62.19% 12GB / 12.1GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 27.51% 2.503GiB / 4GiB 62.56% 12.2GB / 12.3GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 33.72% 2.488GiB / 4GiB 62.19% 12.4GB / 12.4GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 32.68% 2.507GiB / 4GiB 62.68% 12.6GB / 12.6GB 43.1MB / 21.1MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 27.12% 2.513GiB / 4GiB 62.82% 12.8GB / 12.8GB 43.1MB / 21.1MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 26.59% 2.515GiB / 4GiB 62.87% 12.9GB / 12.9GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 31.63% 2.506GiB / 4GiB 62.64% 13.1GB / 13.1GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 19.79% 2.494GiB / 4GiB 62.35% 13.2GB / 13.3GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 56.66% 2.505GiB / 4GiB 62.64% 13.3GB / 13.4GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 22.55% 2.503GiB / 4GiB 62.57% 13.5GB / 13.5GB 43.1MB / 21.1MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 24.58% 2.499GiB / 4GiB 62.48% 13.7GB / 13.7GB 43.1MB / 21.1MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 44.50% 2.526GiB / 4GiB 63.15% 14.4GB / 14.4GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 24.61% 2.525GiB / 4GiB 63.11% 14.5GB / 14.5GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 25.20% 2.514GiB / 4GiB 62.85% 14.6GB / 14.7GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 22.55% 2.524GiB / 4GiB 63.09% 14.7GB / 14.8GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 25.63% 2.53GiB / 4GiB 63.26% 14.9GB / 14.9GB 43.1MB / 21.1MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 21.76% 2.556GiB / 4GiB 63.90% 18.7GB / 18.8GB 43.1MB / 21.1MB 53
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 24.33% 2.552GiB / 4GiB 63.81% 18.8GB / 18.9GB 43.1MB / 21.1MB 53
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 25.40% 2.556GiB / 4GiB 63.91% 19GB / 19.1GB 43.1MB / 21.1MB 53
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 29.29% 2.561GiB / 4GiB 64.03% 19.2GB / 19.3GB 43.1MB / 21.1MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 40.03% 2.56GiB / 4GiB 64.00% 19.5GB / 19.6GB 43.1MB / 21.1MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 25.49% 2.577GiB / 4GiB 64.41% 28.8GB / 29GB 43.1MB / 21.2MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 28.65% 2.575GiB / 4GiB 64.37% 29.1GB / 29.3GB 43.1MB / 21.2MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 25.35% 2.578GiB / 4GiB 64.46% 29.3GB / 29.5GB 43.1MB / 21.2MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 19.23% 2.577GiB / 4GiB 64.44% 34GB / 34.4GB 43.1MB / 21.2MB 57
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 22.20% 2.577GiB / 4GiB 64.43% 34.2GB / 34.5GB 43.1MB / 21.2MB 57
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 27.50% 2.595GiB / 4GiB 64.88% 37.1GB / 37.5GB 43.1MB / 21.2MB 57
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 22.41% 2.588GiB / 4GiB 64.71% 37.3GB / 37.7GB 43.1MB / 21.2MB 57
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 38.88% 2.607GiB / 4GiB 65.17% 38.7GB / 39.1GB 43.1MB / 21.2MB 57
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 30.39% 2.606GiB / 4GiB 65.15% 38.8GB / 39.2GB 43.1MB / 21.2MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 37.61% 2.597GiB / 4GiB 64.92% 39GB / 39.4GB 43.1MB / 21.2MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 43.09% 2.558GiB / 4GiB 63.94% 39.1GB / 39.5GB 43.1MB / 21.2MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 43.19% 2.6GiB / 4GiB 65.01% 39.3GB / 39.7GB 43.1MB / 21.2MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 26.15% 2.584GiB / 4GiB 64.61% 39.4GB / 39.8GB 43.1MB / 21.2MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 29.18% 2.594GiB / 4GiB 64.84% 42.2GB / 42.6GB 43.1MB / 21.2MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 30.01% 2.595GiB / 4GiB 64.88% 42.3GB / 42.8GB 43.1MB / 21.2MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 30.10% 2.604GiB / 4GiB 65.10% 43.4GB / 43.8GB 43.1MB / 21.2MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 37.32% 2.597GiB / 4GiB 64.93% 43.6GB / 44.1GB 43.1MB / 21.2MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 78.75% 2.613GiB / 4GiB 65.32% 44.9GB / 45.4GB 43.1MB / 21.2MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 32.54% 2.615GiB / 4GiB 65.37% 45.1GB / 45.5GB 43.1MB / 21.2MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 24.97% 2.611GiB / 4GiB 65.26% 45.2GB / 45.7GB 43.1MB / 21.2MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 24.59% 2.613GiB / 4GiB 65.33% 45.5GB / 46GB 43.1MB / 21.2MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 21.86% 2.596GiB / 4GiB 64.90% 48.5GB / 49GB 43.1MB / 21.2MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 25.42% 2.59GiB / 4GiB 64.75% 48.7GB / 49.2GB 43.1MB / 21.2MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 30.47% 2.615GiB / 4GiB 65.38% 52.6GB / 53.1GB 43.1MB / 21.3MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 24.78% 2.613GiB / 4GiB 65.32% 52.7GB / 53.3GB 43.1MB / 21.3MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 20.47% 2.617GiB / 4GiB 65.42% 52.8GB / 53.4GB 43.1MB / 21.3MB 52
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 22.36% 2.603GiB / 4GiB 65.09% 53GB / 53.5GB 43.1MB / 21.3MB 53
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 27.24% 2.606GiB / 4GiB 65.16% 53.1GB / 53.7GB 43.1MB / 21.3MB 54
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 20.96% 2.624GiB / 4GiB 65.60% 57.7GB / 58.3GB 43.1MB / 21.3MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 18.32% 2.603GiB / 4GiB 65.07% 57.8GB / 58.5GB 43.1MB / 21.3MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 18.01% 2.621GiB / 4GiB 65.53% 61.9GB / 62.5GB 43.1MB / 21.3MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 14.86% 2.602GiB / 4GiB 65.04% 62GB / 62.7GB 43.1MB / 21.3MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 21.48% 2.611GiB / 4GiB 65.29% 62.4GB / 63.1GB 43.1MB / 21.3MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 19.22% 2.619GiB / 4GiB 65.48% 62.5GB / 63.2GB 43.1MB / 21.3MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 19.12% 2.62GiB / 4GiB 65.51% 67.5GB / 68.3GB 43.1MB / 21.3MB 55
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 21.15% 2.618GiB / 4GiB 65.44% 67.6GB / 68.4GB 43.1MB / 21.3MB 56
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 0.87% 2.496GiB / 4GiB 62.39% 68.5GB / 69.4GB 43.1MB / 21.3MB 52
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 12.21% 2.496GiB / 4GiB 62.39% 68.5GB / 69.4GB 43.1MB / 21.3MB 52
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 0.79% 2.491GiB / 4GiB 62.28% 68.5GB / 69.4GB 44.4MB / 21.4MB 52
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 0.80% 2.491GiB / 4GiB 62.28% 68.5GB / 69.4GB 44.4MB / 21.4MB 52
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 0.79% 2.491GiB / 4GiB 62.28% 68.5GB / 69.4GB 44.4MB / 21.4MB 52
|
||||
(.venv) harishgokul@instance-20260404-194023:~/litellm$ docker compose stats litellm --no-stream
|
||||
CONTAINER ID NAME CPU % MEM USAGE / LIMIT MEM % NET I/O BLOCK I/O PIDS
|
||||
93a259999492 litellm-litellm-1 0.55% 2.491GiB / 4GiB 62.28% 68.5GB / 69.4GB 44.4MB / 21.4MB 52
|
||||
```
|
||||
|
|
@ -21,6 +21,8 @@ from fastapi import (
|
|||
UploadFile,
|
||||
status,
|
||||
)
|
||||
from fastapi.responses import StreamingResponse
|
||||
from openai import OpenAI
|
||||
|
||||
import litellm
|
||||
from litellm import CreateFileRequest, get_secret_str
|
||||
|
|
@ -55,6 +57,7 @@ from .common_utils import (
|
|||
handle_model_based_routing,
|
||||
prepare_data_with_credentials,
|
||||
)
|
||||
import os
|
||||
from .storage_backend_service import StorageBackendFileService
|
||||
|
||||
router = APIRouter()
|
||||
|
|
@ -113,6 +116,11 @@ def get_model_from_json_obj(json_object: dict) -> Optional[str]:
|
|||
return model
|
||||
|
||||
|
||||
def _stream_openai_file_content(file_id: str, client: OpenAI):
|
||||
with client.files.with_streaming_response.content(file_id) as response:
|
||||
yield from response.iter_bytes(chunk_size=1024 * 1024)
|
||||
|
||||
|
||||
async def _deprecated_loadbalanced_create_file(
|
||||
llm_router: Optional[Router],
|
||||
router_model: str,
|
||||
|
|
@ -824,6 +832,118 @@ async def get_file_content( # noqa: PLR0915
|
|||
)
|
||||
|
||||
|
||||
# NOTE: Rough Prototype to check memory usage
|
||||
@router.get(
|
||||
"/{provider}/v2/files/{file_id:path}/content",
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
tags=["files"],
|
||||
)
|
||||
@router.get(
|
||||
"/v2/files/{file_id:path}/content",
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
tags=["files"],
|
||||
)
|
||||
async def get_file_content_v2(
|
||||
request: Request,
|
||||
fastapi_response: Response,
|
||||
file_id: str,
|
||||
provider: Optional[str] = None,
|
||||
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
|
||||
):
|
||||
from litellm.proxy.proxy_server import (
|
||||
general_settings,
|
||||
proxy_config,
|
||||
proxy_logging_obj,
|
||||
version,
|
||||
)
|
||||
|
||||
data: Dict = {"file_id": file_id}
|
||||
try:
|
||||
# Include original request and headers in the data (same as v1 flow)
|
||||
base_llm_response_processor = ProxyBaseLLMRequestProcessing(data=data)
|
||||
(
|
||||
data,
|
||||
litellm_logging_obj,
|
||||
) = await base_llm_response_processor.common_processing_pre_call_logic(
|
||||
request=request,
|
||||
general_settings=general_settings,
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
version=version,
|
||||
proxy_logging_obj=proxy_logging_obj,
|
||||
proxy_config=proxy_config,
|
||||
route_type="afile_content",
|
||||
)
|
||||
|
||||
custom_llm_provider = (
|
||||
provider
|
||||
or get_custom_llm_provider_from_request_headers(request=request)
|
||||
or get_custom_llm_provider_from_request_query(request=request)
|
||||
or await get_custom_llm_provider_from_request_body(request=request)
|
||||
or "openai"
|
||||
)
|
||||
if custom_llm_provider != "openai":
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail={"error": "v2 files content currently only supports openai provider"},
|
||||
)
|
||||
|
||||
client = OpenAI(
|
||||
base_url=os.getenv("OPENAI_BASE_URL"),
|
||||
api_key=os.getenv("OPENAI_API_KEY"),
|
||||
)
|
||||
|
||||
asyncio.create_task(
|
||||
proxy_logging_obj.update_request_status(
|
||||
litellm_call_id=data.get("litellm_call_id", ""), status="success"
|
||||
)
|
||||
)
|
||||
|
||||
fastapi_response.headers.update(
|
||||
ProxyBaseLLMRequestProcessing.get_custom_headers(
|
||||
user_api_key_dict=user_api_key_dict,
|
||||
model_id="",
|
||||
cache_key="",
|
||||
api_base="",
|
||||
version=version,
|
||||
model_region=getattr(user_api_key_dict, "allowed_model_region", ""),
|
||||
)
|
||||
)
|
||||
|
||||
return StreamingResponse(
|
||||
_stream_openai_file_content(file_id, client),
|
||||
media_type="application/octet-stream",
|
||||
headers={
|
||||
"content-disposition": f'attachment; filename="{file_id}.bin"',
|
||||
},
|
||||
)
|
||||
except Exception as e:
|
||||
await proxy_logging_obj.post_call_failure_hook(
|
||||
user_api_key_dict=user_api_key_dict, original_exception=e, request_data=data
|
||||
)
|
||||
verbose_proxy_logger.exception(
|
||||
"litellm.proxy.proxy_server.retrieve_file_content(): Exception occured - {}".format(
|
||||
str(e)
|
||||
)
|
||||
)
|
||||
verbose_proxy_logger.debug(traceback.format_exc())
|
||||
if isinstance(e, HTTPException):
|
||||
raise ProxyException(
|
||||
message=getattr(e, "message", str(e.detail)),
|
||||
type=getattr(e, "type", "None"),
|
||||
param=getattr(e, "param", "None"),
|
||||
code=getattr(e, "status_code", status.HTTP_400_BAD_REQUEST),
|
||||
)
|
||||
else:
|
||||
error_msg = f"{str(e)}"
|
||||
raise ProxyException(
|
||||
message=getattr(e, "message", error_msg),
|
||||
type=getattr(e, "type", "None"),
|
||||
param=getattr(e, "param", "None"),
|
||||
code=getattr(e, "status_code", 500),
|
||||
)
|
||||
|
||||
|
||||
|
||||
@router.get(
|
||||
"/{provider}/v1/files/{file_id:path}",
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
|
|
|
|||
121
scripts/batch_lt.py
Normal file
121
scripts/batch_lt.py
Normal file
|
|
@ -0,0 +1,121 @@
|
|||
import requests
|
||||
import argparse
|
||||
import time
|
||||
import concurrent.futures
|
||||
import threading
|
||||
|
||||
|
||||
LITELLM_BASE_URL = "***/v1/files/test-file-65/content"
|
||||
HEADERS = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": "Bearer sk-1234",
|
||||
}
|
||||
|
||||
def percentile(values, p):
|
||||
"""Returns percentile value using linear interpolation."""
|
||||
if not values:
|
||||
return 0.0
|
||||
if len(values) == 1:
|
||||
return values[0]
|
||||
|
||||
sorted_values = sorted(values)
|
||||
rank = (len(sorted_values) - 1) * (p / 100)
|
||||
lower = int(rank)
|
||||
upper = min(lower + 1, len(sorted_values) - 1)
|
||||
weight = rank - lower
|
||||
return sorted_values[lower] * (1 - weight) + sorted_values[upper] * weight
|
||||
|
||||
|
||||
def render_progress(completed, total):
|
||||
"""Renders a simple terminal progress bar."""
|
||||
bar_width = 32
|
||||
fraction = completed / total if total else 1
|
||||
filled = int(bar_width * fraction)
|
||||
bar = "#" * filled + "-" * (bar_width - filled)
|
||||
percent = fraction * 100
|
||||
return f"[{bar}] {completed}/{total} ({percent:6.2f}%)"
|
||||
|
||||
|
||||
def make_request(session, url):
|
||||
"""Makes a single request and returns timing + success info."""
|
||||
|
||||
start_time = time.time()
|
||||
try:
|
||||
response = session.get(
|
||||
url,
|
||||
headers=HEADERS,
|
||||
)
|
||||
response.raise_for_status()
|
||||
success = True
|
||||
except requests.exceptions.RequestException as e:
|
||||
success = False
|
||||
error_message = str(e)
|
||||
|
||||
response_time = time.time() - start_time
|
||||
if success:
|
||||
return success, response_time, ""
|
||||
|
||||
return success, response_time, error_message
|
||||
|
||||
def run_load_test(num_requests):
|
||||
"""Runs the load test by making a specified number of requests in parallel."""
|
||||
print(f"Starting load test with {num_requests} parallel requests to {LITELLM_BASE_URL}")
|
||||
|
||||
lock = threading.Lock()
|
||||
completed = 0
|
||||
errors = 0
|
||||
latencies = []
|
||||
first_error = None
|
||||
|
||||
with concurrent.futures.ThreadPoolExecutor() as executor:
|
||||
with requests.Session() as session:
|
||||
futures = [executor.submit(make_request, session, LITELLM_BASE_URL) for _ in range(num_requests)]
|
||||
|
||||
for future in concurrent.futures.as_completed(futures):
|
||||
try:
|
||||
success, response_time, error_message = future.result()
|
||||
with lock:
|
||||
completed += 1
|
||||
latencies.append(response_time)
|
||||
if not success:
|
||||
errors += 1
|
||||
if first_error is None:
|
||||
first_error = error_message
|
||||
|
||||
progress = render_progress(completed, num_requests)
|
||||
print(f"\r{progress}", end="", flush=True)
|
||||
except Exception as exc:
|
||||
with lock:
|
||||
completed += 1
|
||||
errors += 1
|
||||
if first_error is None:
|
||||
first_error = str(exc)
|
||||
progress = render_progress(completed, num_requests)
|
||||
print(f"\r{progress}", end="", flush=True)
|
||||
|
||||
print() # move to the next line after progress bar
|
||||
|
||||
p50 = percentile(latencies, 50)
|
||||
p90 = percentile(latencies, 90)
|
||||
p95 = percentile(latencies, 95)
|
||||
p99 = percentile(latencies, 99)
|
||||
|
||||
print("Load test finished.")
|
||||
print(f"Total requests: {num_requests}")
|
||||
print(f"Successful requests: {num_requests - errors}")
|
||||
print(f"Errors: {errors}")
|
||||
print(f"Latency p50: {p50:.3f}s")
|
||||
print(f"Latency p90: {p90:.3f}s")
|
||||
print(f"Latency p95: {p95:.3f}s")
|
||||
print(f"Latency p99: {p99:.3f}s")
|
||||
|
||||
if first_error:
|
||||
print(f"First error: {first_error}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description="A simple script to run a load test against a URL.")
|
||||
parser.add_argument("num_requests", type=int, help="The number of requests to make.")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
run_load_test(args.num_requests)
|
||||
277
scripts/monitor_docker_stats.py
Normal file
277
scripts/monitor_docker_stats.py
Normal file
|
|
@ -0,0 +1,277 @@
|
|||
#!/usr/bin/env python3
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import datetime as dt
|
||||
import math
|
||||
import re
|
||||
import statistics
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import List, Optional
|
||||
|
||||
|
||||
@dataclass
|
||||
class Sample:
|
||||
timestamp: dt.datetime
|
||||
elapsed_s: float
|
||||
cpu_percent: float
|
||||
mem_used_bytes: float
|
||||
|
||||
|
||||
UNIT_FACTORS = {
|
||||
"b": 1,
|
||||
"kb": 1000,
|
||||
"kib": 1024,
|
||||
"mb": 1000**2,
|
||||
"mib": 1024**2,
|
||||
"gb": 1000**3,
|
||||
"gib": 1024**3,
|
||||
"tb": 1000**4,
|
||||
"tib": 1024**4,
|
||||
}
|
||||
|
||||
|
||||
def parse_size_to_bytes(size_str: str) -> float:
|
||||
match = re.fullmatch(r"\s*([0-9]*\.?[0-9]+)\s*([A-Za-z]+)\s*", size_str)
|
||||
if match is None:
|
||||
raise ValueError(f"Unable to parse size: {size_str}")
|
||||
|
||||
value = float(match.group(1))
|
||||
unit = match.group(2).lower()
|
||||
if unit not in UNIT_FACTORS:
|
||||
raise ValueError(f"Unsupported size unit: {unit}")
|
||||
return value * UNIT_FACTORS[unit]
|
||||
|
||||
|
||||
def bytes_to_gib(value_bytes: float) -> float:
|
||||
return value_bytes / (1024**3)
|
||||
|
||||
|
||||
def run_stats_command(compose_cmd: str, service: str) -> str:
|
||||
cmd = compose_cmd.split() + ["stats", service, "--no-stream"]
|
||||
completed = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||
if completed.returncode != 0:
|
||||
stderr = (completed.stderr or "").strip()
|
||||
stdout = (completed.stdout or "").strip()
|
||||
msg = stderr or stdout or "Unknown error"
|
||||
raise RuntimeError(f"Command failed: {' '.join(cmd)}\n{msg}")
|
||||
return completed.stdout
|
||||
|
||||
|
||||
def parse_stats_output(raw_output: str) -> tuple[float, float]:
|
||||
lines = [line for line in raw_output.splitlines() if line.strip()]
|
||||
if len(lines) < 2:
|
||||
raise ValueError("Unexpected docker stats output (no data row found)")
|
||||
|
||||
data_line = lines[-1]
|
||||
|
||||
cpu_match = re.search(r"([0-9]*\.?[0-9]+)%", data_line)
|
||||
mem_match = re.search(
|
||||
r"([0-9]*\.?[0-9]+[A-Za-z]+)\s*/\s*([0-9]*\.?[0-9]+[A-Za-z]+)",
|
||||
data_line,
|
||||
)
|
||||
if cpu_match is None or mem_match is None:
|
||||
raise ValueError(f"Unable to parse stats row: {data_line}")
|
||||
|
||||
cpu_percent = float(cpu_match.group(1))
|
||||
mem_used_bytes = parse_size_to_bytes(mem_match.group(1))
|
||||
return cpu_percent, mem_used_bytes
|
||||
|
||||
|
||||
def summarize(samples: List[Sample]) -> dict:
|
||||
if not samples:
|
||||
raise ValueError("No samples collected")
|
||||
|
||||
cpu_values = [s.cpu_percent for s in samples]
|
||||
mem_gib_values = [bytes_to_gib(s.mem_used_bytes) for s in samples]
|
||||
|
||||
return {
|
||||
"samples": len(samples),
|
||||
"duration_seconds": samples[-1].elapsed_s,
|
||||
"cpu_avg_percent": statistics.mean(cpu_values),
|
||||
"cpu_peak_percent": max(cpu_values),
|
||||
"cpu_variance": statistics.pvariance(cpu_values) if len(cpu_values) > 1 else 0.0,
|
||||
"mem_avg_gib": statistics.mean(mem_gib_values),
|
||||
"mem_peak_gib": max(mem_gib_values),
|
||||
"mem_variance_gib2": (
|
||||
statistics.pvariance(mem_gib_values) if len(mem_gib_values) > 1 else 0.0
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def print_summary(summary: dict) -> None:
|
||||
print("\n=== Docker Stats Summary ===")
|
||||
print(f"Samples: {summary['samples']}")
|
||||
print(f"Duration: {summary['duration_seconds']:.1f}s")
|
||||
print(f"Average CPU: {summary['cpu_avg_percent']:.2f}%")
|
||||
print(f"Peak CPU: {summary['cpu_peak_percent']:.2f}%")
|
||||
print(f"CPU Variance: {summary['cpu_variance']:.4f} (%^2)")
|
||||
print(f"Average Memory: {summary['mem_avg_gib']:.3f} GiB")
|
||||
print(f"Peak Memory: {summary['mem_peak_gib']:.3f} GiB")
|
||||
print(f"Memory Variance: {summary['mem_variance_gib2']:.6f} (GiB^2)")
|
||||
|
||||
|
||||
def write_csv(samples: List[Sample], path: str) -> None:
|
||||
with open(path, "w", newline="", encoding="utf-8") as f:
|
||||
writer = csv.writer(f)
|
||||
writer.writerow(
|
||||
[
|
||||
"timestamp",
|
||||
"elapsed_seconds",
|
||||
"cpu_percent",
|
||||
"mem_used_gib",
|
||||
]
|
||||
)
|
||||
for s in samples:
|
||||
writer.writerow(
|
||||
[
|
||||
s.timestamp.isoformat(),
|
||||
f"{s.elapsed_s:.3f}",
|
||||
f"{s.cpu_percent:.4f}",
|
||||
f"{bytes_to_gib(s.mem_used_bytes):.6f}",
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def plot_memory(samples: List[Sample], output_path: str) -> None:
|
||||
try:
|
||||
import matplotlib.pyplot as plt
|
||||
except Exception as e: # pragma: no cover
|
||||
raise RuntimeError(
|
||||
"matplotlib is required for plotting. Install with: pip install matplotlib"
|
||||
) from e
|
||||
|
||||
x = [s.elapsed_s for s in samples]
|
||||
y = [bytes_to_gib(s.mem_used_bytes) for s in samples]
|
||||
|
||||
plt.figure(figsize=(10, 5))
|
||||
plt.plot(x, y, linewidth=2)
|
||||
plt.xlabel("Time (seconds)")
|
||||
plt.ylabel("Memory usage (GiB)")
|
||||
plt.title("Memory Usage vs Time")
|
||||
plt.grid(True, alpha=0.3)
|
||||
plt.tight_layout()
|
||||
plt.savefig(output_path, dpi=160)
|
||||
plt.close()
|
||||
|
||||
|
||||
def should_stop(start_monotonic: float, duration_s: Optional[float]) -> bool:
|
||||
if duration_s is not None and (time.monotonic() - start_monotonic) >= duration_s:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Sample 'docker compose stats <service> --no-stream' and summarize usage",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--interval",
|
||||
type=float,
|
||||
default=45.0,
|
||||
help="Sampling interval in seconds (default: 45)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--duration",
|
||||
type=float,
|
||||
default=None,
|
||||
help="Total duration in seconds. If omitted, runs until Ctrl+C.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--service",
|
||||
type=str,
|
||||
default="litellm",
|
||||
help="Compose service name (default: litellm)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--compose-cmd",
|
||||
type=str,
|
||||
default="docker compose",
|
||||
help="Compose command prefix (default: 'docker compose')",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--csv",
|
||||
type=str,
|
||||
default=None,
|
||||
help="Optional output CSV path for raw samples",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--plot",
|
||||
type=str,
|
||||
default=None,
|
||||
help="Optional PNG path for memory-vs-time plot",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.interval <= 0:
|
||||
print("--interval must be > 0", file=sys.stderr)
|
||||
return 2
|
||||
|
||||
if args.duration is None:
|
||||
print("Running until Ctrl+C. Use --duration for automatic stop.")
|
||||
|
||||
samples: List[Sample] = []
|
||||
start_monotonic = time.monotonic()
|
||||
next_run = start_monotonic
|
||||
|
||||
try:
|
||||
while True:
|
||||
now = time.monotonic()
|
||||
if now < next_run:
|
||||
time.sleep(next_run - now)
|
||||
|
||||
timestamp = dt.datetime.now(dt.timezone.utc)
|
||||
elapsed_s = time.monotonic() - start_monotonic
|
||||
|
||||
try:
|
||||
output = run_stats_command(args.compose_cmd, args.service)
|
||||
cpu, mem_used_bytes = parse_stats_output(output)
|
||||
sample = Sample(
|
||||
timestamp=timestamp,
|
||||
elapsed_s=elapsed_s,
|
||||
cpu_percent=cpu,
|
||||
mem_used_bytes=mem_used_bytes,
|
||||
)
|
||||
samples.append(sample)
|
||||
print(
|
||||
f"[{timestamp.isoformat()}] sample={len(samples)} "
|
||||
f"cpu={cpu:.2f}% mem={bytes_to_gib(mem_used_bytes):.3f}GiB"
|
||||
)
|
||||
except Exception as e:
|
||||
print(f"Warning: failed to sample stats: {e}", file=sys.stderr)
|
||||
|
||||
if should_stop(start_monotonic, args.duration):
|
||||
break
|
||||
|
||||
next_run += args.interval
|
||||
|
||||
except KeyboardInterrupt:
|
||||
print("\nStopping on Ctrl+C...")
|
||||
|
||||
if not samples:
|
||||
print("No valid samples collected.", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
summary = summarize(samples)
|
||||
print_summary(summary)
|
||||
|
||||
if args.csv:
|
||||
write_csv(samples, args.csv)
|
||||
print(f"Wrote samples CSV: {args.csv}")
|
||||
|
||||
if args.plot:
|
||||
try:
|
||||
plot_memory(samples, args.plot)
|
||||
print(f"Wrote memory plot: {args.plot}")
|
||||
except Exception as e:
|
||||
print(f"Failed to generate plot: {e}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
142
tests/proxy_unit_tests/test_openai_files_v2_endpoint.py
Normal file
142
tests/proxy_unit_tests/test_openai_files_v2_endpoint.py
Normal file
|
|
@ -0,0 +1,142 @@
|
|||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
from fastapi import Request
|
||||
from fastapi.responses import Response, StreamingResponse
|
||||
|
||||
from litellm.proxy.openai_files_endpoints.files_endpoints import get_file_content_v2
|
||||
from litellm.proxy._types import ProxyException
|
||||
|
||||
|
||||
class _FakeStreamingResponse:
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc, tb):
|
||||
return False
|
||||
|
||||
def iter_bytes(self, chunk_size=1024 * 1024):
|
||||
yield b"hello "
|
||||
yield b"world"
|
||||
|
||||
|
||||
class _FakeFilesClient:
|
||||
class _WithStreamingResponse:
|
||||
def content(self, file_id):
|
||||
assert file_id == "file-123"
|
||||
return _FakeStreamingResponse()
|
||||
|
||||
def __init__(self):
|
||||
self.with_streaming_response = self._WithStreamingResponse()
|
||||
|
||||
|
||||
class _FakeOpenAIClient:
|
||||
def __init__(self, *args, **kwargs):
|
||||
self.files = _FakeFilesClient()
|
||||
|
||||
|
||||
class _FakeProxyLogging:
|
||||
async def update_request_status(self, litellm_call_id: str, status: str):
|
||||
return None
|
||||
|
||||
async def post_call_failure_hook(
|
||||
self, user_api_key_dict, original_exception: Exception, request_data
|
||||
):
|
||||
return None
|
||||
|
||||
|
||||
async def _fake_common_processing_pre_call_logic(self, **kwargs):
|
||||
return self.data, MagicMock()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def request_obj() -> Request:
|
||||
scope = {
|
||||
"type": "http",
|
||||
"method": "GET",
|
||||
"path": "/v2/files/file-123/content",
|
||||
"headers": [],
|
||||
"query_string": b"",
|
||||
}
|
||||
return Request(scope)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_file_content_v2_returns_streaming_response(monkeypatch, request_obj):
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.openai_files_endpoints.files_endpoints.OpenAI",
|
||||
_FakeOpenAIClient,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.openai_files_endpoints.files_endpoints.ProxyBaseLLMRequestProcessing.common_processing_pre_call_logic",
|
||||
_fake_common_processing_pre_call_logic,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
_FakeProxyLogging(),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.general_settings",
|
||||
{},
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.proxy_config",
|
||||
None,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.version",
|
||||
"test-version",
|
||||
)
|
||||
|
||||
response = await get_file_content_v2(
|
||||
request=request_obj,
|
||||
fastapi_response=Response(),
|
||||
file_id="file-123",
|
||||
provider="openai",
|
||||
user_api_key_dict=MagicMock(),
|
||||
)
|
||||
|
||||
assert isinstance(response, StreamingResponse)
|
||||
assert response.media_type == "application/octet-stream"
|
||||
|
||||
chunks = []
|
||||
async for chunk in response.body_iterator:
|
||||
chunks.append(chunk)
|
||||
|
||||
assert chunks == [b"hello ", b"world"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_get_file_content_v2_rejects_non_openai_provider(monkeypatch, request_obj):
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.openai_files_endpoints.files_endpoints.ProxyBaseLLMRequestProcessing.common_processing_pre_call_logic",
|
||||
_fake_common_processing_pre_call_logic,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.proxy_logging_obj",
|
||||
_FakeProxyLogging(),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.general_settings",
|
||||
{},
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.proxy_config",
|
||||
None,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.version",
|
||||
"test-version",
|
||||
)
|
||||
|
||||
with pytest.raises(ProxyException) as exc_info:
|
||||
await get_file_content_v2(
|
||||
request=request_obj,
|
||||
fastapi_response=Response(),
|
||||
file_id="file-123",
|
||||
provider="anthropic",
|
||||
user_api_key_dict=MagicMock(),
|
||||
)
|
||||
|
||||
assert exc_info.value.code == str(400)
|
||||
assert "only supports openai provider" in exc_info.value.message
|
||||
Loading…
Add table
Reference in a new issue