diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md
index 807fc85cec3..5692e924e6d 100644
--- a/ARCHITECTURE.md
+++ b/ARCHITECTURE.md
@@ -28,16 +28,24 @@ sequenceDiagram
participant Client
participant ProxyServer as proxy/proxy_server.py
participant Auth as proxy/auth/user_api_key_auth.py
+ participant Redis as Redis Cache
participant Hooks as proxy/hooks/
participant Router as router.py
participant Main as main.py
participant Handler as llms/custom_httpx/llm_http_handler.py
participant Transform as llms/{provider}/chat/transformation.py
participant Provider as LLM Provider API
+ participant CostCalc as litellm.completion_cost()
+ participant DBWriter as db/db_spend_update_writer.py
+ participant Postgres as PostgreSQL
+ %% Request Flow
Client->>ProxyServer: POST /v1/chat/completions
ProxyServer->>Auth: user_api_key_auth()
+ Auth->>Redis: Check API key cache
+ Redis-->>Auth: Key info + spend limits
ProxyServer->>Hooks: max_budget_limiter, parallel_request_limiter
+ Hooks->>Redis: Check/increment rate limit counters
ProxyServer->>Router: route_request()
Router->>Main: litellm.acompletion()
Main->>Handler: BaseLLMHTTPHandler.completion()
@@ -45,8 +53,16 @@ sequenceDiagram
Handler->>Provider: HTTP Request
Provider-->>Handler: Response
Handler->>Transform: ProviderConfig.transform_response()
- Handler-->>Hooks: async_log_success_event()
- Handler-->>Client: ModelResponse
+
+ %% Response Flow with Cost Attribution
+ Handler->>CostCalc: Calculate response cost (tokens × price)
+ CostCalc-->>Handler: response_cost
+ Handler->>Hooks: async_log_success_event()
+ Hooks->>DBWriter: update_database(response_cost)
+ DBWriter->>Redis: Queue spend increment
+ DBWriter->>Postgres: Batch write spend logs (async)
+ Hooks->>Redis: update_cache(token, response_cost)
+ Handler-->>Client: ModelResponse + x-litellm-response-cost header
```
### Proxy Components
@@ -75,12 +91,20 @@ graph TD
Main["main.py"]
end
+ subgraph "Infrastructure"
+ Redis["Redis
(rate limits, caching, spend queue)"]
+ Postgres["PostgreSQL
(keys, teams, spend logs)"]
+ end
+
Client --> Endpoint
Endpoint --> Auth
+ Auth --> Redis
Auth --> PreCall
PreCall --> RouteRequest
RouteRequest --> Router
Router --> Main
+ Main --> Redis
+ Main --> Postgres
Main --> Client
```
@@ -119,6 +143,59 @@ graph TD
To add a new proxy hook, implement `CustomLogger` and register in `PROXY_HOOKS`.
+### Infrastructure Components
+
+The AI Gateway uses external infrastructure for persistence and caching:
+
+```mermaid
+graph LR
+ subgraph "AI Gateway"
+ Proxy["proxy/proxy_server.py"]
+ DBWriter["proxy/db/db_spend_update_writer.py
DBSpendUpdateWriter"]
+ Cache["proxy/utils.py
InternalUsageCache"]
+ CostCallback["proxy/hooks/proxy_track_cost_callback.py
_ProxyDBLogger"]
+ end
+
+ subgraph "Redis (caching/redis_cache.py)"
+ RateLimit["Rate Limit Counters"]
+ SpendQueue["Spend Increment Queue"]
+ KeyCache["API Key Cache (DualCache)"]
+ ResponseCache["LLM Response Cache"]
+ end
+
+ subgraph "PostgreSQL (proxy/schema.prisma)"
+ Keys["LiteLLM_VerificationToken"]
+ Teams["LiteLLM_TeamTable"]
+ SpendLogs["LiteLLM_SpendLogs"]
+ Users["LiteLLM_UserTable"]
+ end
+
+ Proxy --> Cache
+ Cache --> RateLimit
+ Cache --> KeyCache
+ Cache --> ResponseCache
+ CostCallback --> DBWriter
+ DBWriter --> SpendQueue
+ DBWriter --> SpendLogs
+ Proxy --> Keys
+ Proxy --> Teams
+```
+
+| Component | Purpose | Key Files/Classes |
+|-----------|---------|-------------------|
+| **Redis** | Rate limiting, caching, spend queuing | `caching/redis_cache.py` (`RedisCache`), `caching/dual_cache.py` (`DualCache`) |
+| **PostgreSQL** | API keys, teams, users, spend logs | `proxy/utils.py` (`PrismaClient`), `proxy/schema.prisma` |
+| **InternalUsageCache** | In-memory + Redis cache abstraction | `proxy/utils.py` (`InternalUsageCache`) |
+| **DBSpendUpdateWriter** | Batches spend updates to reduce DB writes | `proxy/db/db_spend_update_writer.py` (`DBSpendUpdateWriter`) |
+| **Cost Tracking** | Calculates and logs response costs | `proxy/hooks/proxy_track_cost_callback.py` (`_ProxyDBLogger`) |
+
+**Cost Attribution Flow:**
+1. `litellm.completion_cost()` (`cost_calculator.py`) calculates cost from token usage × model pricing
+2. Cost is added to response headers (`x-litellm-response-cost`) via `proxy/common_request_processing.py`
+3. `_ProxyDBLogger.async_log_success_event()` triggers spend tracking
+4. `DBSpendUpdateWriter.update_database()` queues spend increments
+5. `update_cache()` in `proxy/proxy_server.py` updates Redis for real-time budget enforcement
+
---
## 2. SDK Request Flow
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index 4abbddb0d50..470d598a25f 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -10201,48 +10201,6 @@
"mode": "completion",
"output_cost_per_token": 5e-07
},
- "deepseek-v3-2-251201": {
- "input_cost_per_token": 0.0,
- "litellm_provider": "volcengine",
- "max_input_tokens": 98304,
- "max_output_tokens": 32768,
- "max_tokens": 32768,
- "mode": "chat",
- "output_cost_per_token": 0.0,
- "supports_assistant_prefill": true,
- "supports_function_calling": true,
- "supports_prompt_caching": true,
- "supports_reasoning": true,
- "supports_tool_choice": true
- },
- "glm-4-7-251222": {
- "input_cost_per_token": 0.0,
- "litellm_provider": "volcengine",
- "max_input_tokens": 204800,
- "max_output_tokens": 131072,
- "max_tokens": 131072,
- "mode": "chat",
- "output_cost_per_token": 0.0,
- "supports_assistant_prefill": true,
- "supports_function_calling": true,
- "supports_prompt_caching": true,
- "supports_reasoning": true,
- "supports_tool_choice": true
- },
- "kimi-k2-thinking-251104": {
- "input_cost_per_token": 0.0,
- "litellm_provider": "volcengine",
- "max_input_tokens": 229376,
- "max_output_tokens": 32768,
- "max_tokens": 32768,
- "mode": "chat",
- "output_cost_per_token": 0.0,
- "supports_assistant_prefill": true,
- "supports_function_calling": true,
- "supports_prompt_caching": true,
- "supports_reasoning": true,
- "supports_tool_choice": true
- },
"doubao-embedding": {
"input_cost_per_token": 0.0,
"litellm_provider": "volcengine",