diff --git a/docs/my-website/docs/proxy/litellm_managed_files.md b/docs/my-website/docs/proxy/litellm_managed_files.md index 6e40c6dd449..e1f883fb3e9 100644 --- a/docs/my-website/docs/proxy/litellm_managed_files.md +++ b/docs/my-website/docs/proxy/litellm_managed_files.md @@ -6,6 +6,15 @@ import Image from '@theme/IdealImage'; Reuse the same 'file id' across different providers. +:::info + +This is a LiteLLM Enterprise feature. + +Available for free via the `litellm[proxy]` package or any `litellm` docker image. + +::: + + | Feature | Description | Comments | | --- | --- | --- | | Proxy | ✅ | | diff --git a/docs/my-website/docs/proxy/managed_batches.md b/docs/my-website/docs/proxy/managed_batches.md new file mode 100644 index 00000000000..5eca083f501 --- /dev/null +++ b/docs/my-website/docs/proxy/managed_batches.md @@ -0,0 +1,158 @@ +# [BETA] Unified File ID with Batches + +:::info + +This is a LiteLLM Enterprise feature. + +Available for free via the `litellm[proxy]` package or any `litellm` docker image. + +::: + + +| Feature | Description | Comments | +| --- | --- | --- | +| Proxy | ✅ | | +| SDK | ❌ | Requires postgres DB for storing file ids | +| Available across all [Batch providers](../batches#supported-providers) | ✅ | | + + +## Overview + +Use this to: + +- Loadbalance across multiple Azure Batch deployments +- Easily switch across Azure/OpenAI/VertexAI Batch APIs without needing to reupload files each time +- Control batch model access by key/user/team (same as chat completion models) + +## (Proxy Admin) Usage + +Here's how to give developers access to your Batch models. + +### 1. Setup config.yaml + +- specify `mode: batch` for each model: Allows developers to know this is a batch model. + +```yaml +model_list: + - model_name: "gpt-4o-batch" + litellm_params: + model: azure/gpt-4o-mini-general-deployment + api_base: os.environ/AZURE_API_BASE + api_key: os.environ/AZURE_API_KEY + model_info: + mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model + - model_name: "gpt-4o-batch" + litellm_params: + model: azure/gpt-4o-mini-special-deployment + api_base: os.environ/AZURE_API_BASE_2 + api_key: os.environ/AZURE_API_KEY_2 + model_info: + mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model + +``` + +### 2. Create Virtual Key + +```bash +curl -L -X POST 'https://{PROXY_BASE_URL}/key/generate' \ +-H 'Authorization: Bearer ${PROXY_API_KEY}' \ +-H 'Content-Type: application/json' \ +-d '{"models": ["gpt-4o-batch"]}' +``` + + +You can now use the virtual key to access the batch models (See Developer flow). + +## (Developer) Usage + +### 1. Create request.jsonl + +- Check models available via `/model_group/info` +- See all models with `mode: batch` +- Set `model` in .jsonl to the model from `/model_group/info` + +```json +{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-4o-batch", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 1000}} +{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-4o-batch", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 1000}} +``` + +Expectation: + +- LiteLLM translates this to the azure model specific value + +### 2. Upload File + +Specify `target_model_names: ""` to enable LiteLLM managed files and request validation. + +model-name should be the same as the model-name in the request.jsonl + +```python +from openai import OpenAI + +client = OpenAI( + base_url="http://0.0.0.0:4000", + api_key="sk-1234", +) + +# Upload file +batch_input_file = client.files.create( + file=open("./request.jsonl", "rb"), # {"model": "gpt-4o-batch"} <-> {"model": "gpt-4o-mini-special-deployment"} + purpose="batch", + extra_body={"target_model_names": "gpt-4o-batch"} +) +print(batch_input_file) +``` + +Expectation: + +- This is used to validate if user has model access. +- This is used to write the file to the correct deployments (all gpt-4o-batch deployments will be written to). + +### 3. Create + Retrieve the batch + +```python +... +# Create batch +batch = client.batches.create( + input_file_id=batch_input_file.id, + endpoint="/v1/chat/completions", + completion_window="24h", + metadata={"description": "Test batch job"}, +) +print(batch) + +# Retrieve batch + +batch_response = client.batches.retrieve( # LOG VIRTUAL MODEL NAME + batch_id +) +status = batch_response.status +``` + +### 4. Retrieve Batch Content + +```python +... + +file_id = batch_response.output_file_id + +file_response = client.files.content(file_id) +print(file_response.text) +``` + +### 5. Cancel a batch + +```python +... + +client.batches.cancel(batch_id) +``` + +### 6. List batches + +```python +... + +client.batches.list(limit=10, extra_body={"target_model_names": "gpt-4o-batch"}) +``` + diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 16572ad450d..6fc49332205 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -18,6 +18,7 @@ const sidebars = { // But you can create a sidebar manually tutorialSidebar: [ { type: "doc", id: "index" }, // NEW + { type: "category", label: "LiteLLM Proxy Server", @@ -181,6 +182,95 @@ const sidebars = { "proxy/caching", ] }, + { + type: "category", + label: "Supported Endpoints", + link: { + type: "generated-index", + title: "Supported Endpoints", + description: + "Learn how to deploy + call models from different providers on LiteLLM", + slug: "/supported_endpoints", + }, + items: [ + { + type: "category", + label: "/chat/completions", + link: { + type: "generated-index", + title: "Chat Completions", + description: "Details on the completion() function", + slug: "/completion", + }, + items: [ + "completion/input", + "completion/output", + "completion/usage", + ], + }, + "response_api", + "text_completion", + "embedding/supported_embedding", + "anthropic_unified", + "mcp", + { + type: "category", + label: "/images", + items: [ + "image_generation", + "image_variations", + ] + }, + { + type: "category", + label: "/audio", + "items": [ + "audio_transcription", + "text_to_speech", + ] + }, + { + type: "category", + label: "Pass-through Endpoints (Anthropic SDK, etc.)", + items: [ + "pass_through/intro", + "pass_through/vertex_ai", + "pass_through/google_ai_studio", + "pass_through/cohere", + "pass_through/vllm", + "pass_through/mistral", + "pass_through/openai_passthrough", + "pass_through/anthropic_completion", + "pass_through/bedrock", + "pass_through/assembly_ai", + "pass_through/langfuse", + "proxy/pass_through", + ], + }, + "rerank", + "assistants", + + { + type: "category", + label: "/files", + items: [ + "files_endpoints", + "proxy/litellm_managed_files", + ], + }, + { + type: "category", + label: "/batches", + items: [ + "batches", + "proxy/managed_batches", + ] + }, + "realtime", + "fine_tuning", + "moderation", + ], + }, { type: "category", label: "Supported Models & Providers", @@ -305,88 +395,7 @@ const sidebars = { ] }, - { - type: "category", - label: "Supported Endpoints", - link: { - type: "generated-index", - title: "Supported Endpoints", - description: - "Learn how to deploy + call models from different providers on LiteLLM", - slug: "/supported_endpoints", - }, - items: [ - { - type: "category", - label: "/chat/completions", - link: { - type: "generated-index", - title: "Chat Completions", - description: "Details on the completion() function", - slug: "/completion", - }, - items: [ - "completion/input", - "completion/output", - "completion/usage", - ], - }, - "response_api", - "text_completion", - "embedding/supported_embedding", - "anthropic_unified", - "mcp", - { - type: "category", - label: "/images", - items: [ - "image_generation", - "image_variations", - ] - }, - { - type: "category", - label: "/audio", - "items": [ - "audio_transcription", - "text_to_speech", - ] - }, - { - type: "category", - label: "Pass-through Endpoints (Anthropic SDK, etc.)", - items: [ - "pass_through/intro", - "pass_through/vertex_ai", - "pass_through/google_ai_studio", - "pass_through/cohere", - "pass_through/vllm", - "pass_through/mistral", - "pass_through/openai_passthrough", - "pass_through/anthropic_completion", - "pass_through/bedrock", - "pass_through/assembly_ai", - "pass_through/langfuse", - "proxy/pass_through", - ], - }, - "rerank", - "assistants", - - { - type: "category", - label: "/files", - items: [ - "files_endpoints", - "proxy/litellm_managed_files", - ], - }, - "batches", - "realtime", - "fine_tuning", - "moderation", - ], - }, + { type: "category", label: "Routing, Loadbalancing & Fallbacks", diff --git a/litellm/proxy/_new_secret_config.yaml b/litellm/proxy/_new_secret_config.yaml index eacfe88d6d8..65ef4b89296 100644 --- a/litellm/proxy/_new_secret_config.yaml +++ b/litellm/proxy/_new_secret_config.yaml @@ -63,6 +63,7 @@ model_list: api_key: os.environ/AZURE_API_KEY model_info: id: my-general-azure-deployment + mode: batch - model_name: "gpt-4o-batch" litellm_params: model: azure/gpt-4o-mini @@ -70,3 +71,4 @@ model_list: api_key: 04d22fb7e9ad4d9c8afe7c6abf97a6fc model_info: id: my-unique-azure-deployment + mode: batch \ No newline at end of file