diff --git a/.circleci/config.yml b/.circleci/config.yml index 95984e4e68e..4306fa5cf05 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -2,6 +2,7 @@ version: 2.1 orbs: codecov: codecov/codecov@4.0.1 node: circleci/node@5.1.0 # Add this line to declare the node orb + win: circleci/windows@5.0 # Add Windows orb commands: setup_google_dns: @@ -15,8 +16,41 @@ commands: echo "nameserver 127.0.0.11" | sudo tee /etc/resolv.conf echo "nameserver 8.8.8.8" | sudo tee -a /etc/resolv.conf echo "nameserver 8.8.4.4" | sudo tee -a /etc/resolv.conf + setup_litellm_enterprise_pip: + steps: + - run: + name: "Install local version of litellm-enterprise" + command: | + cd enterprise + python -m pip install -e . + cd .. jobs: + # Add Windows testing job + using_litellm_on_windows: + executor: + name: win/default + shell: powershell.exe + working_directory: ~/project + steps: + - checkout + - run: + name: Install Python + command: | + choco install python --version=3.11.0 -y + refreshenv + python --version + - run: + name: Install Dependencies + command: | + python -m pip install --upgrade pip + pip install pytest + pip install . + - run: + name: Run Windows-specific test + command: | + python -m pytest tests/windows_tests/test_litellm_on_windows.py -v + local_testing: docker: - image: cimg/python:3.11 @@ -85,6 +119,7 @@ jobs: pip install "pytest-xdist==3.6.1" pip install "websockets==13.1.0" pip uninstall posthog -y + - setup_litellm_enterprise_pip - save_cache: paths: - ./venv @@ -107,10 +142,13 @@ jobs: name: Linting Testing command: | cd litellm + pip install "cryptography<40.0.0" python -m pip install types-requests types-setuptools types-redis types-PyYAML - if ! python -m mypy . --ignore-missing-imports; then - echo "mypy detected errors" - exit 1 + if ! python -m mypy . \ + --config-file mypy.ini \ + --ignore-missing-imports; then + echo "mypy detected errors" + exit 1 fi cd .. @@ -202,6 +240,7 @@ jobs: pip install "Pillow==10.3.0" pip install "jsonschema==4.22.0" pip install "websockets==13.1.0" + - setup_litellm_enterprise_pip - save_cache: paths: - ./venv @@ -308,6 +347,7 @@ jobs: pip install "Pillow==10.3.0" pip install "jsonschema==4.22.0" pip install "websockets==13.1.0" + - setup_litellm_enterprise_pip - save_cache: paths: - ./venv @@ -419,6 +459,7 @@ jobs: pip install "pytest-retry==1.6.3" pip install "pytest-asyncio==0.21.1" # Run pytest and generate JUnit XML report + - setup_litellm_enterprise_pip - run: name: Run tests command: | @@ -563,6 +604,7 @@ jobs: pip install "jsonschema==4.22.0" pip install "pytest-postgresql==7.0.1" pip install "fakeredis==2.28.1" + - setup_litellm_enterprise_pip - save_cache: paths: - ./venv @@ -620,6 +662,7 @@ jobs: pip install "pytest-asyncio==0.21.1" pip install "pytest-cov==5.0.0" # Run pytest and generate JUnit XML report + - setup_litellm_enterprise_pip - run: name: Run tests command: | @@ -765,6 +808,51 @@ jobs: paths: - mcp_coverage.xml - mcp_coverage + guardrails_testing: + docker: + - image: cimg/python:3.11 + auth: + username: ${DOCKERHUB_USERNAME} + password: ${DOCKERHUB_PASSWORD} + working_directory: ~/project + + steps: + - checkout + - setup_google_dns + - run: + name: Install Dependencies + command: | + python -m pip install --upgrade pip + python -m pip install -r requirements.txt + pip install "pytest==7.3.1" + pip install "pytest-retry==1.6.3" + pip install "pytest-cov==5.0.0" + pip install "pytest-asyncio==0.21.1" + pip install "respx==0.21.1" + pip install "pydantic==2.10.2" + pip install "boto3==1.34.34" + # Run pytest and generate JUnit XML report + - run: + name: Run tests + command: | + pwd + ls + python -m pytest -vv tests/guardrails_tests --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=5 + no_output_timeout: 120m + - run: + name: Rename the coverage files + command: | + mv coverage.xml guardrails_coverage.xml + mv .coverage guardrails_coverage + + # Store test results + - store_test_results: + path: test-results + - persist_to_workspace: + root: . + paths: + - guardrails_coverage.xml + - guardrails_coverage llm_responses_api_testing: docker: - image: cimg/python:3.11 @@ -835,14 +923,14 @@ jobs: pip install "mcp==1.5.0" pip install "requests-mock>=1.12.1" pip install "responses==0.25.7" - + - setup_litellm_enterprise_pip # Run pytest and generate JUnit XML report - run: name: Run tests command: | pwd ls - python -m pytest -vv tests/litellm tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=5 + python -m pytest -vv tests/litellm tests/enterprise --cov=litellm --cov-report=xml -x -s -v --junitxml=test-results/junit.xml --durations=10 no_output_timeout: 120m - run: name: Rename the coverage files @@ -1062,6 +1150,7 @@ jobs: pip install "google-cloud-aiplatform==1.43.0" pip install "mlflow==2.17.2" # Run pytest and generate JUnit XML report + - setup_litellm_enterprise_pip - run: name: Run tests command: | @@ -1109,6 +1198,7 @@ jobs: pip install "tokenizers==0.20.0" pip install "uvloop==0.21.0" pip install jsonschema + - setup_litellm_enterprise_pip - run: name: Run tests command: | @@ -1457,7 +1547,7 @@ jobs: command: | pwd ls - python -m pytest -s -vv tests/*.py -x --junitxml=test-results/junit.xml --durations=5 --ignore=tests/otel_tests --ignore=tests/spend_tracking_tests --ignore=tests/pass_through_tests --ignore=tests/proxy_admin_ui_tests --ignore=tests/load_tests --ignore=tests/llm_translation --ignore=tests/llm_responses_api_testing --ignore=tests/mcp_tests --ignore=tests/image_gen_tests --ignore=tests/pass_through_unit_tests + python -m pytest -s -vv tests/*.py -x --junitxml=test-results/junit.xml --durations=5 --ignore=tests/otel_tests --ignore=tests/spend_tracking_tests --ignore=tests/pass_through_tests --ignore=tests/proxy_admin_ui_tests --ignore=tests/load_tests --ignore=tests/llm_translation --ignore=tests/llm_responses_api_testing --ignore=tests/mcp_tests --ignore=tests/guardrails_tests --ignore=tests/image_gen_tests --ignore=tests/pass_through_unit_tests no_output_timeout: 120m # Store test results @@ -2316,7 +2406,7 @@ jobs: python -m venv venv . venv/bin/activate pip install coverage - coverage combine llm_translation_coverage llm_responses_api_coverage mcp_coverage logging_coverage litellm_router_coverage local_testing_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_proxy_security_tests_coverage + coverage combine llm_translation_coverage llm_responses_api_coverage mcp_coverage logging_coverage litellm_router_coverage local_testing_coverage litellm_assistants_api_coverage auth_ui_unit_tests_coverage langfuse_coverage caching_coverage litellm_proxy_unit_tests_coverage image_gen_coverage pass_through_unit_tests_coverage batches_coverage litellm_proxy_security_tests_coverage guardrails_coverage coverage xml - codecov/upload: file: ./coverage.xml @@ -2680,6 +2770,12 @@ workflows: version: 2 build_and_test: jobs: + - using_litellm_on_windows: + filters: + branches: + only: + - main + - /litellm_.*/ - local_testing: filters: branches: @@ -2800,6 +2896,12 @@ workflows: only: - main - /litellm_.*/ + - guardrails_testing: + filters: + branches: + only: + - main + - /litellm_.*/ - llm_responses_api_testing: filters: branches: @@ -2846,6 +2948,7 @@ workflows: requires: - llm_translation_testing - mcp_testing + - guardrails_testing - llm_responses_api_testing - litellm_mapped_tests - batches_testing @@ -2937,4 +3040,5 @@ workflows: - proxy_pass_through_endpoint_tests - check_code_and_doc_quality - publish_proxy_extras - + - guardrails_testing + diff --git a/.env.example b/.env.example index c6df78cafef..24c2b608414 100644 --- a/.env.example +++ b/.env.example @@ -20,10 +20,12 @@ REPLICATE_API_TOKEN = "" ANTHROPIC_API_KEY = "" # Infisical INFISICAL_TOKEN = "" +# Novita AI +NOVITA_API_KEY = "" # INFINITY INFINITY_API_KEY = "" # Development Configs LITELLM_MASTER_KEY = "sk-1234" DATABASE_URL = "postgresql://llmproxy:dbpassword9090@db:5432/litellm" -STORE_MODEL_IN_DB = "True" \ No newline at end of file +STORE_MODEL_IN_DB = "True" diff --git a/.github/workflows/test-litellm.yml b/.github/workflows/test-litellm.yml index 12d09725ed1..a2b9e6c7c34 100644 --- a/.github/workflows/test-litellm.yml +++ b/.github/workflows/test-litellm.yml @@ -7,7 +7,7 @@ on: jobs: test: runs-on: ubuntu-latest - timeout-minutes: 5 + timeout-minutes: 8 steps: - uses: actions/checkout@v4 @@ -29,7 +29,11 @@ jobs: run: | poetry install --with dev,proxy-dev --extras proxy poetry run pip install pytest-xdist - + - name: Setup litellm-enterprise as local package + run: | + cd enterprise + python -m pip install -e . + cd .. - name: Run tests run: | poetry run pytest tests/litellm -x -vv -n 4 \ No newline at end of file diff --git a/README.md b/README.md index 1c4e1484437..01a60310522 100644 --- a/README.md +++ b/README.md @@ -299,6 +299,7 @@ curl 'http://0.0.0.0:4000/key/generate' \ | Provider | [Completion](https://docs.litellm.ai/docs/#basic-usage) | [Streaming](https://docs.litellm.ai/docs/completion/stream#streaming-responses) | [Async Completion](https://docs.litellm.ai/docs/completion/stream#async-completion) | [Async Streaming](https://docs.litellm.ai/docs/completion/stream#async-streaming) | [Async Embedding](https://docs.litellm.ai/docs/embedding/supported_embedding) | [Async Image Generation](https://docs.litellm.ai/docs/image_generation) | |-------------------------------------------------------------------------------------|---------------------------------------------------------|---------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------|-------------------------------------------------------------------------------|-------------------------------------------------------------------------| | [openai](https://docs.litellm.ai/docs/providers/openai) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | +| [Meta - Llama API](https://docs.litellm.ai/docs/providers/meta_llama) | ✅ | ✅ | ✅ | ✅ | | | | [azure](https://docs.litellm.ai/docs/providers/azure) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | [AI/ML API](https://docs.litellm.ai/docs/providers/aiml) | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | [aws - sagemaker](https://docs.litellm.ai/docs/providers/aws_sagemaker) | ✅ | ✅ | ✅ | ✅ | ✅ | | @@ -332,7 +333,8 @@ curl 'http://0.0.0.0:4000/key/generate' \ | [xinference [Xorbits Inference]](https://docs.litellm.ai/docs/providers/xinference) | | | | | ✅ | | | [FriendliAI](https://docs.litellm.ai/docs/providers/friendliai) | ✅ | ✅ | ✅ | ✅ | | | | [Galadriel](https://docs.litellm.ai/docs/providers/galadriel) | ✅ | ✅ | ✅ | ✅ | | | - +| [Novita AI](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) | ✅ | ✅ | ✅ | ✅ | | | +| [Featherless AI](https://docs.litellm.ai/docs/providers/featherless_ai) | ✅ | ✅ | ✅ | ✅ | | | [**Read the Docs**](https://docs.litellm.ai/docs/) ## Contributing diff --git a/cookbook/LiteLLM_NovitaAI_Cookbook.ipynb b/cookbook/LiteLLM_NovitaAI_Cookbook.ipynb new file mode 100644 index 00000000000..8fa7d0b987a --- /dev/null +++ b/cookbook/LiteLLM_NovitaAI_Cookbook.ipynb @@ -0,0 +1,97 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "metadata": { + "id": "iFEmsVJI_2BR" + }, + "source": [ + "# LiteLLM NovitaAI Cookbook" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cBlUhCEP_xj4" + }, + "outputs": [], + "source": [ + "!pip install litellm" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "p-MQqWOT_1a7" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "os.environ['NOVITA_API_KEY'] = \"\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Ze8JqMqWAARO" + }, + "outputs": [], + "source": [ + "from litellm import completion\n", + "response = completion(\n", + " model=\"novita/deepseek/deepseek-r1\",\n", + " messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n", + ")\n", + "response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-LnhELrnAM_J" + }, + "outputs": [], + "source": [ + "response = completion(\n", + " model=\"novita/deepseek/deepseek-r1\",\n", + " messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n", + ")\n", + "response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dJBOUYdwCEn1" + }, + "outputs": [], + "source": [ + "response = completion(\n", + " model=\"mistralai/mistral-7b-instruct\",\n", + " messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n", + ")\n", + "response" + ] + } + ], + "metadata": { + "colab": { + "provenance": [] + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + }, + "language_info": { + "name": "python" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/cookbook/LiteLLM_OpenRouter.ipynb b/cookbook/LiteLLM_OpenRouter.ipynb index e0d03e1258f..6444b23b294 100644 --- a/cookbook/LiteLLM_OpenRouter.ipynb +++ b/cookbook/LiteLLM_OpenRouter.ipynb @@ -1,27 +1,13 @@ { - "nbformat": 4, - "nbformat_minor": 0, - "metadata": { - "colab": { - "provenance": [] - }, - "kernelspec": { - "name": "python3", - "display_name": "Python 3" - }, - "language_info": { - "name": "python" - } - }, "cells": [ { "cell_type": "markdown", - "source": [ - "# LiteLLM OpenRouter Cookbook" - ], "metadata": { "id": "iFEmsVJI_2BR" - } + }, + "source": [ + "# LiteLLM OpenRouter Cookbook" + ] }, { "cell_type": "code", @@ -36,27 +22,20 @@ }, { "cell_type": "code", + "execution_count": 14, + "metadata": { + "id": "p-MQqWOT_1a7" + }, + "outputs": [], "source": [ "import os\n", "\n", "os.environ['OPENROUTER_API_KEY'] = \"\"" - ], - "metadata": { - "id": "p-MQqWOT_1a7" - }, - "execution_count": 14, - "outputs": [] + ] }, { "cell_type": "code", - "source": [ - "from litellm import completion\n", - "response = completion(\n", - " model=\"openrouter/google/palm-2-chat-bison\",\n", - " messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n", - ")\n", - "response" - ], + "execution_count": 11, "metadata": { "colab": { "base_uri": "https://localhost:8080/" @@ -64,10 +43,8 @@ "id": "Ze8JqMqWAARO", "outputId": "64f3e836-69fa-4f8e-fb35-088a913bbe98" }, - "execution_count": 11, "outputs": [ { - "output_type": "execute_result", "data": { "text/plain": [ " JSON: {\n", @@ -85,20 +62,23 @@ "}" ] }, + "execution_count": 11, "metadata": {}, - "execution_count": 11 + "output_type": "execute_result" } + ], + "source": [ + "from litellm import completion\n", + "response = completion(\n", + " model=\"openrouter/google/palm-2-chat-bison\",\n", + " messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n", + ")\n", + "response" ] }, { "cell_type": "code", - "source": [ - "response = completion(\n", - " model=\"openrouter/anthropic/claude-2\",\n", - " messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n", - ")\n", - "response" - ], + "execution_count": 12, "metadata": { "colab": { "base_uri": "https://localhost:8080/" @@ -106,10 +86,8 @@ "id": "-LnhELrnAM_J", "outputId": "d51c7ab7-d761-4bd1-f849-1534d9df4cd0" }, - "execution_count": 12, "outputs": [ { - "output_type": "execute_result", "data": { "text/plain": [ " JSON: {\n", @@ -128,20 +106,22 @@ "}" ] }, + "execution_count": 12, "metadata": {}, - "execution_count": 12 + "output_type": "execute_result" } + ], + "source": [ + "response = completion(\n", + " model=\"openrouter/anthropic/claude-2\",\n", + " messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n", + ")\n", + "response" ] }, { "cell_type": "code", - "source": [ - "response = completion(\n", - " model=\"openrouter/meta-llama/llama-2-70b-chat\",\n", - " messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n", - ")\n", - "response" - ], + "execution_count": 13, "metadata": { "colab": { "base_uri": "https://localhost:8080/" @@ -149,10 +129,8 @@ "id": "dJBOUYdwCEn1", "outputId": "ffa18679-ec15-4dad-fe2b-68665cdf36b0" }, - "execution_count": 13, "outputs": [ { - "output_type": "execute_result", "data": { "text/plain": [ " JSON: {\n", @@ -170,10 +148,32 @@ "}" ] }, + "execution_count": 13, "metadata": {}, - "execution_count": 13 + "output_type": "execute_result" } + ], + "source": [ + "response = completion(\n", + " model=\"openrouter/meta-llama/llama-2-70b-chat\",\n", + " messages=[{\"role\": \"user\", \"content\": \"write code for saying hi\"}]\n", + ")\n", + "response" ] } - ] -} \ No newline at end of file + ], + "metadata": { + "colab": { + "provenance": [] + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + }, + "language_info": { + "name": "python" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/cookbook/google_adk_litellm_tutorial.ipynb b/cookbook/google_adk_litellm_tutorial.ipynb new file mode 100644 index 00000000000..27914edbba8 --- /dev/null +++ b/cookbook/google_adk_litellm_tutorial.ipynb @@ -0,0 +1,412 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "7aa8875d", + "metadata": {}, + "source": [ + "# Google ADK with LiteLLM\n", + "\n", + "Use Google ADK with LiteLLM Python SDK, LiteLLM Proxy.\n", + "\n", + "This tutorial shows you how to create intelligent agents using Agent Development Kit (ADK) with support for multiple Large Language Model (LLM) providers through LiteLLM." + ] + }, + { + "cell_type": "markdown", + "id": "a4d249c3", + "metadata": {}, + "source": [ + "## Overview\n", + "\n", + "ADK (Agent Development Kit) allows you to build intelligent agents powered by LLMs. By integrating with LiteLLM, you can:\n", + "\n", + "- Use multiple LLM providers (OpenAI, Anthropic, Google, etc.)\n", + "- Switch easily between models from different providers\n", + "- Connect to a LiteLLM proxy for centralized model management" + ] + }, + { + "cell_type": "markdown", + "id": "a0bbb56b", + "metadata": {}, + "source": [ + "## Prerequisites\n", + "\n", + "- Python environment setup\n", + "- API keys for model providers (OpenAI, Anthropic, Google AI Studio)\n", + "- Basic understanding of LLMs and agent concepts" + ] + }, + { + "cell_type": "markdown", + "id": "7fee50a8", + "metadata": {}, + "source": [ + "## Installation" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "44106a23", + "metadata": {}, + "outputs": [], + "source": [ + "# Install dependencies\n", + "!pip install google-adk litellm" + ] + }, + { + "cell_type": "markdown", + "id": "2171740a", + "metadata": {}, + "source": [ + "## 1. Setting Up Environment" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "6695807e", + "metadata": {}, + "outputs": [], + "source": [ + "# Setup environment and API keys\n", + "import os\n", + "import asyncio\n", + "from google.adk.agents import Agent\n", + "from google.adk.models.lite_llm import LiteLlm # For multi-model support\n", + "from google.adk.sessions import InMemorySessionService\n", + "from google.adk.runners import Runner\n", + "from google.genai import types\n", + "import litellm # Import for proxy configuration\n", + "\n", + "# Set your API keys\n", + "os.environ['GOOGLE_API_KEY'] = 'your-google-api-key' # For Gemini models\n", + "os.environ['OPENAI_API_KEY'] = 'your-openai-api-key' # For OpenAI models\n", + "os.environ['ANTHROPIC_API_KEY'] = 'your-anthropic-api-key' # For Claude models\n", + "\n", + "# Define model constants for cleaner code\n", + "MODEL_GEMINI_PRO = 'gemini-1.5-pro'\n", + "MODEL_GPT_4O = 'openai/gpt-4o'\n", + "MODEL_CLAUDE_SONNET = 'anthropic/claude-3-sonnet-20240229'" + ] + }, + { + "cell_type": "markdown", + "id": "d2b1ed59", + "metadata": {}, + "source": [ + "## 2. Define a Simple Tool" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "04b3ef5b", + "metadata": {}, + "outputs": [], + "source": [ + "# Weather tool implementation\n", + "def get_weather(city: str) -> dict:\n", + " \"\"\"Retrieves the current weather report for a specified city.\"\"\"\n", + " print(f'Tool: get_weather called for city: {city}')\n", + "\n", + " # Mock weather data\n", + " mock_weather_db = {\n", + " 'newyork': {\n", + " 'status': 'success',\n", + " 'report': 'The weather in New York is sunny with a temperature of 25°C.'\n", + " },\n", + " 'london': {\n", + " 'status': 'success',\n", + " 'report': \"It's cloudy in London with a temperature of 15°C.\"\n", + " },\n", + " 'tokyo': {\n", + " 'status': 'success',\n", + " 'report': 'Tokyo is experiencing light rain and a temperature of 18°C.'\n", + " },\n", + " }\n", + "\n", + " city_normalized = city.lower().replace(' ', '')\n", + "\n", + " if city_normalized in mock_weather_db:\n", + " return mock_weather_db[city_normalized]\n", + " else:\n", + " return {\n", + " 'status': 'error',\n", + " 'error_message': f\"Sorry, I don't have weather information for '{city}'.\"\n", + " }" + ] + }, + { + "cell_type": "markdown", + "id": "727b15c9", + "metadata": {}, + "source": [ + "## 3. Helper Function for Agent Interaction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "f77449bf", + "metadata": {}, + "outputs": [], + "source": [ + "# Agent interaction helper function\n", + "async def call_agent_async(query: str, runner, user_id, session_id):\n", + " \"\"\"Sends a query to the agent and prints the final response.\"\"\"\n", + " print(f'\\n>>> User Query: {query}')\n", + "\n", + " content = types.Content(role='user', parts=[types.Part(text=query)])\n", + " final_response_text = 'Agent did not produce a final response.'\n", + "\n", + " async for event in runner.run_async(\n", + " user_id=user_id,\n", + " session_id=session_id,\n", + " new_message=content\n", + " ):\n", + " if event.is_final_response():\n", + " if event.content and event.content.parts:\n", + " final_response_text = event.content.parts[0].text\n", + " break\n", + " print(f'<<< Agent Response: {final_response_text}')" + ] + }, + { + "cell_type": "markdown", + "id": "0ac87987", + "metadata": {}, + "source": [ + "## 4. Using Different Model Providers with ADK\n", + "\n", + "### 4.1 Using OpenAI Models" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "e167d557", + "metadata": {}, + "outputs": [], + "source": [ + "# OpenAI model implementation\n", + "weather_agent_gpt = Agent(\n", + " name='weather_agent_gpt',\n", + " model=LiteLlm(model=MODEL_GPT_4O),\n", + " description='Provides weather information using OpenAI\\'s GPT.',\n", + " instruction=(\n", + " 'You are a helpful weather assistant powered by GPT-4o. '\n", + " \"Use the 'get_weather' tool for city weather requests. \"\n", + " 'Present information clearly.'\n", + " ),\n", + " tools=[get_weather],\n", + ")\n", + "\n", + "session_service_gpt = InMemorySessionService()\n", + "session_gpt = session_service_gpt.create_session(\n", + " app_name='weather_app', user_id='user_1', session_id='session_gpt'\n", + ")\n", + "\n", + "runner_gpt = Runner(\n", + " agent=weather_agent_gpt,\n", + " app_name='weather_app',\n", + " session_service=session_service_gpt,\n", + ")\n", + "\n", + "async def test_gpt_agent():\n", + " print('\\n--- Testing GPT Agent ---')\n", + " await call_agent_async(\n", + " \"What's the weather in London?\",\n", + " runner=runner_gpt,\n", + " user_id='user_1',\n", + " session_id='session_gpt',\n", + " )\n", + "\n", + "# To execute in a notebook cell:\n", + "# await test_gpt_agent()" + ] + }, + { + "cell_type": "markdown", + "id": "f9cb0613", + "metadata": {}, + "source": [ + "### 4.2 Using Anthropic Models" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "1c653665", + "metadata": {}, + "outputs": [], + "source": [ + "# Anthropic model implementation\n", + "weather_agent_claude = Agent(\n", + " name='weather_agent_claude',\n", + " model=LiteLlm(model=MODEL_CLAUDE_SONNET),\n", + " description='Provides weather information using Anthropic\\'s Claude.',\n", + " instruction=(\n", + " 'You are a helpful weather assistant powered by Claude Sonnet. '\n", + " \"Use the 'get_weather' tool for city weather requests. \"\n", + " 'Present information clearly.'\n", + " ),\n", + " tools=[get_weather],\n", + ")\n", + "\n", + "session_service_claude = InMemorySessionService()\n", + "session_claude = session_service_claude.create_session(\n", + " app_name='weather_app', user_id='user_1', session_id='session_claude'\n", + ")\n", + "\n", + "runner_claude = Runner(\n", + " agent=weather_agent_claude,\n", + " app_name='weather_app',\n", + " session_service=session_service_claude,\n", + ")\n", + "\n", + "async def test_claude_agent():\n", + " print('\\n--- Testing Claude Agent ---')\n", + " await call_agent_async(\n", + " \"What's the weather in Tokyo?\",\n", + " runner=runner_claude,\n", + " user_id='user_1',\n", + " session_id='session_claude',\n", + " )\n", + "\n", + "# To execute in a notebook cell:\n", + "# await test_claude_agent()" + ] + }, + { + "cell_type": "markdown", + "id": "bf9d863b", + "metadata": {}, + "source": [ + "### 4.3 Using Google's Gemini Models" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "83f49d0a", + "metadata": {}, + "outputs": [], + "source": [ + "# Gemini model implementation\n", + "weather_agent_gemini = Agent(\n", + " name='weather_agent_gemini',\n", + " model=MODEL_GEMINI_PRO,\n", + " description='Provides weather information using Google\\'s Gemini.',\n", + " instruction=(\n", + " 'You are a helpful weather assistant powered by Gemini Pro. '\n", + " \"Use the 'get_weather' tool for city weather requests. \"\n", + " 'Present information clearly.'\n", + " ),\n", + " tools=[get_weather],\n", + ")\n", + "\n", + "session_service_gemini = InMemorySessionService()\n", + "session_gemini = session_service_gemini.create_session(\n", + " app_name='weather_app', user_id='user_1', session_id='session_gemini'\n", + ")\n", + "\n", + "runner_gemini = Runner(\n", + " agent=weather_agent_gemini,\n", + " app_name='weather_app',\n", + " session_service=session_service_gemini,\n", + ")\n", + "\n", + "async def test_gemini_agent():\n", + " print('\\n--- Testing Gemini Agent ---')\n", + " await call_agent_async(\n", + " \"What's the weather in New York?\",\n", + " runner=runner_gemini,\n", + " user_id='user_1',\n", + " session_id='session_gemini',\n", + " )\n", + "\n", + "# To execute in a notebook cell:\n", + "# await test_gemini_agent()" + ] + }, + { + "cell_type": "markdown", + "id": "93bc5fd0", + "metadata": {}, + "source": [ + "## 5. Using LiteLLM Proxy with ADK" + ] + }, + { + "cell_type": "markdown", + "id": "b4275151", + "metadata": {}, + "source": [ + "| Variable | Description |\n", + "|----------|-------------|\n", + "| `LITELLM_PROXY_API_KEY` | The API key for the LiteLLM proxy |\n", + "| `LITELLM_PROXY_API_BASE` | The base URL for the LiteLLM proxy |\n", + "| `USE_LITELLM_PROXY` or `litellm.use_litellm_proxy` | When set to True, your request will be sent to LiteLLM proxy. |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "256530a6", + "metadata": {}, + "outputs": [], + "source": [ + "# LiteLLM proxy integration\n", + "os.environ['LITELLM_PROXY_API_KEY'] = 'your-litellm-proxy-api-key'\n", + "os.environ['LITELLM_PROXY_API_BASE'] = 'your-litellm-proxy-url' # e.g., 'http://localhost:4000'\n", + "litellm.use_litellm_proxy = True\n", + "\n", + "weather_agent_proxy_env = Agent(\n", + " name='weather_agent_proxy_env',\n", + " model=LiteLlm(model='gpt-4o'),\n", + " description='Provides weather information using a model from LiteLLM proxy.',\n", + " instruction=(\n", + " 'You are a helpful weather assistant. '\n", + " \"Use the 'get_weather' tool for city weather requests. \"\n", + " 'Present information clearly.'\n", + " ),\n", + " tools=[get_weather],\n", + ")\n", + "\n", + "session_service_proxy_env = InMemorySessionService()\n", + "session_proxy_env = session_service_proxy_env.create_session(\n", + " app_name='weather_app', user_id='user_1', session_id='session_proxy_env'\n", + ")\n", + "\n", + "runner_proxy_env = Runner(\n", + " agent=weather_agent_proxy_env,\n", + " app_name='weather_app',\n", + " session_service=session_service_proxy_env,\n", + ")\n", + "\n", + "async def test_proxy_env_agent():\n", + " print('\\n--- Testing Proxy-enabled Agent (Environment Variables) ---')\n", + " await call_agent_async(\n", + " \"What's the weather in London?\",\n", + " runner=runner_proxy_env,\n", + " user_id='user_1',\n", + " session_id='session_proxy_env',\n", + " )\n", + "\n", + "# To execute in a notebook cell:\n", + "# await test_proxy_env_agent()" + ] + } + ], + "metadata": { + "language_info": { + "name": "python" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/docs/my-website/docs/anthropic_unified.md b/docs/my-website/docs/anthropic_unified.md index 92cae9c0aa9..8a34db52482 100644 --- a/docs/my-website/docs/anthropic_unified.md +++ b/docs/my-website/docs/anthropic_unified.md @@ -16,10 +16,10 @@ Use LiteLLM to call all your LLM APIs in the Anthropic `v1/messages` format. | Streaming | ✅ | | | Fallbacks | ✅ | between anthropic models | | Loadbalancing | ✅ | between anthropic models | +| Support llm providers | - `anthropic`
- `bedrock` (only Anthropic models) | | Planned improvement: - Vertex AI Anthropic support -- Bedrock Anthropic support ## Usage --- diff --git a/docs/my-website/docs/apply_guardrail.md b/docs/my-website/docs/apply_guardrail.md new file mode 100644 index 00000000000..740eb232e13 --- /dev/null +++ b/docs/my-website/docs/apply_guardrail.md @@ -0,0 +1,70 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# /guardrails/apply_guardrail + +Use this endpoint to directly call a guardrail configured on your LiteLLM instance. This is useful when you have services that need to directly call a guardrail. + + +## Usage +--- + +In this example `mask_pii` is the guardrail name configured on LiteLLM. + +```bash showLineNumbers title="Example calling the endpoint" +curl -X POST 'http://localhost:4000/guardrails/apply_guardrail' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer your-api-key' \ +-d '{ + "guardrail_name": "mask_pii", + "text": "My name is John Doe and my email is john@example.com", + "language": "en", + "entities": ["NAME", "EMAIL"] +}' +``` + + +## Request Format +--- + +The request body should follow the ApplyGuardrailRequest format. + +#### Example Request Body + +```json +{ + "guardrail_name": "mask_pii", + "text": "My name is John Doe and my email is john@example.com", + "language": "en", + "entities": ["NAME", "EMAIL"] +} +``` + +#### Required Fields +- **guardrail_name** (string): + The identifier for the guardrail to apply (e.g., "mask_pii"). +- **text** (string): + The input text to process through the guardrail. + +#### Optional Fields +- **language** (string): + The language of the input text (e.g., "en" for English). +- **entities** (array of strings): + Specific entities to process or filter (e.g., ["NAME", "EMAIL"]). + +## Response Format +--- + +The response will contain the processed text after applying the guardrail. + +#### Example Response + +```json +{ + "response_text": "My name is [REDACTED] and my email is [REDACTED]" +} +``` + +#### Response Fields +- **response_text** (string): + The text after applying the guardrail. diff --git a/docs/my-website/docs/completion/input.md b/docs/my-website/docs/completion/input.md index a8aa79b8cba..f9751094249 100644 --- a/docs/my-website/docs/completion/input.md +++ b/docs/my-website/docs/completion/input.md @@ -55,6 +55,7 @@ Use `litellm.get_supported_openai_params()` for an updated list of params for ea |Bedrock| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | | | ✅ (model dependent) | | |Sagemaker| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | |TogetherAI| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | ✅ | | | ✅ | | ✅ | ✅ | | | | +|Sambanova| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | ✅ | ✅ | | | | |AlephAlpha| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | |NLP Cloud| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | |Petals| ✅ | ✅ | | ✅ | ✅ | | | | | | @@ -62,6 +63,7 @@ Use `litellm.get_supported_openai_params()` for an updated list of params for ea |Databricks| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | | | | | | | |ClarifAI| ✅ | ✅ | ✅ | |✅ | ✅ | | | | | | | | | | | |Github| ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | | | | | ✅ |✅ (model dependent)|✅ (model dependent)| | | +|Novita AI| ✅ | ✅ | | ✅ | ✅ | ✅ | | ✅ | ✅ | ✅ | ✅ | | | ✅ | | | | | | | | :::note By default, LiteLLM raises an exception if the openai param being passed in isn't supported. diff --git a/docs/my-website/docs/completion/knowledgebase.md b/docs/my-website/docs/completion/knowledgebase.md index e810935a37c..033dccea200 100644 --- a/docs/my-website/docs/completion/knowledgebase.md +++ b/docs/my-website/docs/completion/knowledgebase.md @@ -1,23 +1,61 @@ -# Using Vector Stores (Knowledge Bases) with LiteLLM +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; +import Image from '@theme/IdealImage'; -LiteLLM integrates with AWS Bedrock Knowledge Bases, allowing your models to access your organization's data for more accurate and contextually relevant responses. +# Using Vector Stores (Knowledge Bases) + + +

+ Use Vector Stores with any LiteLLM supported model +

+ + +LiteLLM integrates with vector stores, allowing your models to access your organization's data for more accurate and contextually relevant responses. + +## Supported Vector Stores +- [Bedrock Knowledge Bases](https://aws.amazon.com/bedrock/knowledge-bases/) ## Quick Start -In order to use a Bedrock Knowledge Base with LiteLLM, you need to pass `vector_store_ids` as a parameter to the completion request. Where `vector_store_ids` is a list of Bedrock Knowledge Base IDs. +In order to use a vector store with LiteLLM, you need to + +- Initialize litellm.vector_store_registry +- Pass tools with vector_store_ids to the completion request. Where `vector_store_ids` is a list of vector store ids you initialized in litellm.vector_store_registry ### LiteLLM Python SDK +LiteLLM's allows you to use vector stores in the [OpenAI API spec](https://platform.openai.com/docs/api-reference/chat/create) by passing a tool with vector_store_ids you want to use + ```python showLineNumbers title="Basic Bedrock Knowledge Base Usage" import os import litellm +from litellm.vector_stores.vector_store_registry import VectorStoreRegistry, LiteLLM_ManagedVectorStore + +# Init vector store registry +litellm.vector_store_registry = VectorStoreRegistry( + vector_stores=[ + LiteLLM_ManagedVectorStore( + vector_store_id="T37J8R4WTM", + custom_llm_provider="bedrock" + ) + ] +) + # Make a completion request with vector_store_ids parameter response = await litellm.acompletion( model="anthropic/claude-3-5-sonnet", messages=[{"role": "user", "content": "What is litellm?"}], - vector_store_ids=["YOUR_KNOWLEDGE_BASE_ID"] # e.g., "T37J8R4WTM" + tools=[ + { + "type": "file_search", + "vector_store_ids": ["T37J8R4WTM"] + } + ], ) print(response.choices[0].message.content) @@ -25,7 +63,12 @@ print(response.choices[0].message.content) ### LiteLLM Proxy -#### 1. Configure your proxy +#### 1. Configure your vector_store_registry + +In order to use a vector store with LiteLLM, you need to configure your vector_store_registry. This tells litellm which vector stores to use and api provider to use for the vector store. + + + ```yaml showLineNumbers title="config.yaml" model_list: @@ -34,12 +77,35 @@ model_list: model: anthropic/claude-3-5-sonnet api_key: os.environ/ANTHROPIC_API_KEY +vector_store_registry: + - vector_store_name: "bedrock-litellm-website-knowledgebase" + litellm_params: + vector_store_id: "T37J8R4WTM" + custom_llm_provider: "bedrock" + vector_store_description: "Bedrock vector store for the Litellm website knowledgebase" + vector_store_metadata: + source: "https://www.litellm.com/docs" + ``` -#### 2. Make a request with vector_store_ids parameter + -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; + + +On the LiteLLM UI, Navigate to Experimental > Vector Stores > Create Vector Store. On this page you can create a vector store with a name, vector store id and credentials. + + + + + + + + + +#### 2. Make a request with vector_store_ids parameter @@ -51,7 +117,12 @@ curl http://localhost:4000/v1/chat/completions \ -d '{ "model": "claude-3-5-sonnet", "messages": [{"role": "user", "content": "What is litellm?"}], - "vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"] + "tools": [ + { + "type": "file_search", + "vector_store_ids": ["T37J8R4WTM"] + } + ] }' ``` @@ -72,7 +143,12 @@ client = OpenAI( response = client.chat.completions.create( model="claude-3-5-sonnet", messages=[{"role": "user", "content": "What is litellm?"}], - extra_body={"vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"]} + tools=[ + { + "type": "file_search", + "vector_store_ids": ["T37J8R4WTM"] + } + ] ) print(response.choices[0].message.content) @@ -81,17 +157,98 @@ print(response.choices[0].message.content) + + + +## Advanced + +### Logging Vector Store Usage + +LiteLLM allows you to view your vector store usage in the LiteLLM UI on the `Logs` page. + +After completing a request with a vector store, navigate to the `Logs` page on LiteLLM. Here you should be able to see the query sent to the vector store and corresponding response with scores. + + +

+ LiteLLM Logs Page: Vector Store Usage +

+ + +### Listing available vector stores + +You can list all available vector stores using the /vector_store/list endpoint + +**Request:** +```bash showLineNumbers title="List all available vector stores" +curl -X GET "http://localhost:4000/vector_store/list" \ + -H "Authorization: Bearer $LITELLM_API_KEY" +``` + +**Response:** + +The response will be a list of all vector stores that are available to use with LiteLLM. + +```json +{ + "object": "list", + "data": [ + { + "vector_store_id": "T37J8R4WTM", + "custom_llm_provider": "bedrock", + "vector_store_name": "bedrock-litellm-website-knowledgebase", + "vector_store_description": "Bedrock vector store for the Litellm website knowledgebase", + "vector_store_metadata": { + "source": "https://www.litellm.com/docs" + }, + "created_at": "2023-05-03T18:21:36.462Z", + "updated_at": "2023-05-03T18:21:36.462Z", + "litellm_credential_name": "bedrock_credentials" + } + ], + "total_count": 1, + "current_page": 1, + "total_pages": 1 +} +``` + + +### Always on for a model + +**Use this if you want vector stores to be used by default for a specific model.** + +In this config, we add `vector_store_ids` to the claude-3-5-sonnet-with-vector-store model. This means that any request to the claude-3-5-sonnet-with-vector-store model will always use the vector store with the id `T37J8R4WTM` defined in the `vector_store_registry`. + +```yaml showLineNumbers title="Always on for a model" +model_list: + - model_name: claude-3-5-sonnet-with-vector-store + litellm_params: + model: anthropic/claude-3-5-sonnet + vector_store_ids: ["T37J8R4WTM"] + +vector_store_registry: + - vector_store_name: "bedrock-litellm-website-knowledgebase" + litellm_params: + vector_store_id: "T37J8R4WTM" + custom_llm_provider: "bedrock" + vector_store_description: "Bedrock vector store for the Litellm website knowledgebase" + vector_store_metadata: + source: "https://www.litellm.com/docs" +``` + ## How It Works -LiteLLM implements a `BedrockKnowledgeBaseHook` that intercepts your completion requests for handling the integration with Bedrock Knowledge Bases. +If your request includes a `vector_store_ids` parameter where any of the vector store ids are found in the `vector_store_registry`, LiteLLM will automatically use the vector store for the request. -1. You make a completion request with the `vector_store_ids` parameter +1. You make a completion request with the `vector_store_ids` parameter and any of the vector store ids are found in the `litellm.vector_store_registry` 2. LiteLLM automatically: - Uses your last message as the query to retrieve relevant information from the Knowledge Base - Adds the retrieved context to your conversation - Sends the augmented messages to the model -### Example Transformation +#### Example Transformation When you pass `vector_store_ids=["YOUR_KNOWLEDGE_BASE_ID"]`, your request flows through these steps: @@ -137,4 +294,63 @@ When using the Knowledge Base integration with LiteLLM, you can include the foll | Parameter | Type | Description | |-----------|------|-------------| -| `vector_store_ids` | List[str] | List of Bedrock Knowledge Base IDs to query | +| `vector_store_ids` | List[str] | List of Knowledge Base IDs to query | + +### VectorStoreRegistry + +The `VectorStoreRegistry` is a central component for managing vector stores in LiteLLM. It acts as a registry where you can configure and access your vector stores. + +#### What is VectorStoreRegistry? + +`VectorStoreRegistry` is a class that: +- Maintains a collection of vector stores that LiteLLM can use +- Allows you to register vector stores with their credentials and metadata +- Makes vector stores accessible via their IDs in your completion requests + +#### Using VectorStoreRegistry in Python + +```python +from litellm.vector_stores.vector_store_registry import VectorStoreRegistry, LiteLLM_ManagedVectorStore + +# Initialize the vector store registry with one or more vector stores +litellm.vector_store_registry = VectorStoreRegistry( + vector_stores=[ + LiteLLM_ManagedVectorStore( + vector_store_id="YOUR_VECTOR_STORE_ID", # Required: Unique ID for referencing this store + custom_llm_provider="bedrock" # Required: Provider (e.g., "bedrock") + ) + ] +) +``` + +#### LiteLLM_ManagedVectorStore Parameters + +Each vector store in the registry is configured using a `LiteLLM_ManagedVectorStore` object with these parameters: + +| Parameter | Type | Required | Description | +|-----------|------|----------|-------------| +| `vector_store_id` | str | Yes | Unique identifier for the vector store | +| `custom_llm_provider` | str | Yes | The provider of the vector store (e.g., "bedrock") | +| `vector_store_name` | str | No | A friendly name for the vector store | +| `vector_store_description` | str | No | Description of what the vector store contains | +| `vector_store_metadata` | dict or str | No | Additional metadata about the vector store | +| `litellm_credential_name` | str | No | Name of the credentials to use for this vector store | + +#### Configuring VectorStoreRegistry in config.yaml + +For the LiteLLM Proxy, you can configure the same registry in your `config.yaml` file: + +```yaml showLineNumbers title="Vector store configuration in config.yaml" +vector_store_registry: + - vector_store_name: "bedrock-litellm-website-knowledgebase" # Optional friendly name + litellm_params: + vector_store_id: "T37J8R4WTM" # Required: Unique ID + custom_llm_provider: "bedrock" # Required: Provider + vector_store_description: "Bedrock vector store for the Litellm website knowledgebase" + vector_store_metadata: + source: "https://www.litellm.com/docs" +``` + +The `litellm_params` section accepts all the same parameters as the `LiteLLM_ManagedVectorStore` constructor in the Python SDK. + + diff --git a/docs/my-website/docs/embedding/supported_embedding.md b/docs/my-website/docs/embedding/supported_embedding.md index 06d41073722..6257ca2dba4 100644 --- a/docs/my-website/docs/embedding/supported_embedding.md +++ b/docs/my-website/docs/embedding/supported_embedding.md @@ -225,36 +225,6 @@ response = embedding( | text-embedding-3-large | `embedding('text-embedding-3-large', input)` | `os.environ['OPENAI_API_KEY']` | | text-embedding-ada-002 | `embedding('text-embedding-ada-002', input)` | `os.environ['OPENAI_API_KEY']` | -## Azure OpenAI Embedding Models - -### API keys -This can be set as env variables or passed as **params to litellm.embedding()** -```python -import os -os.environ['AZURE_API_KEY'] = -os.environ['AZURE_API_BASE'] = -os.environ['AZURE_API_VERSION'] = -``` - -### Usage -```python -from litellm import embedding -response = embedding( - model="azure/", - input=["good morning from litellm"], - api_key=api_key, - api_base=api_base, - api_version=api_version, -) -print(response) -``` - -| Model Name | Function Call | -|----------------------|---------------------------------------------| -| text-embedding-ada-002 | `embedding(model="azure/", input=input)` | - -h/t to [Mikko](https://www.linkedin.com/in/mikkolehtimaki/) for this integration - ## OpenAI Compatible Embedding Models Use this for calling `/embedding` endpoints on OpenAI Compatible Servers, example https://github.com/xorbitsai/inference diff --git a/docs/my-website/docs/extras/contributing_code.md b/docs/my-website/docs/extras/contributing_code.md index ee46a330958..747df5b60fc 100644 --- a/docs/my-website/docs/extras/contributing_code.md +++ b/docs/my-website/docs/extras/contributing_code.md @@ -4,20 +4,23 @@ Here are the core requirements for any PR submitted to LiteLLM - +- [ ] Sign the Contributor License Agreement (CLA) - [see details](#contributor-license-agreement-cla) - [ ] Add testing, **Adding at least 1 test is a hard requirement** - [see details](#2-adding-testing-to-your-pr) - [ ] Ensure your PR passes the following tests: - - [ ] [Unit Tests](#3-running-unit-tests) - - [ ] [Formatting / Linting Tests](#35-running-linting-tests) + - [ ] [Unit Tests](#3-running-unit-tests) + - [ ] [Formatting / Linting Tests](#35-running-linting-tests) - [ ] Keep scope as isolated as possible. As a general rule, your changes should address 1 specific problem at a time +## **Contributor License Agreement (CLA)** +Before contributing code to LiteLLM, you must sign our [Contributor License Agreement (CLA)](<(https://cla-assistant.io/BerriAI/litellm)>). This is a legal requirement for all contributions to be merged into the main repository. The CLA helps protect both you and the project by clearly defining the terms under which your contributions are made. + +**Important:** We strongly recommend reviewing and signing the CLA before starting work on your contribution to avoid any delays in the PR process. You can find the CLA [here](https://cla-assistant.io/BerriAI/litellm) and sign it through our CLA management system when you submit your first PR. ## Quick start ## 1. Setup your local dev environment - Here's how to modify the repo locally: Step 1: Clone the repo @@ -71,9 +74,9 @@ LiteLLM uses mypy for linting. On ci/cd we also run `black` for formatting. - push your fork to your GitHub repo - submit a PR from there - ## Advanced -### Building LiteLLM Docker Image + +### Building LiteLLM Docker Image Some people might want to build the LiteLLM docker image themselves. Follow these instructions if you want to build / run the LiteLLM Docker Image yourself. diff --git a/docs/my-website/docs/index.md b/docs/my-website/docs/index.md index 9e4d76b89c0..58cabc81b48 100644 --- a/docs/my-website/docs/index.md +++ b/docs/my-website/docs/index.md @@ -208,6 +208,22 @@ response = completion( ) ``` + + + +```python +from litellm import completion +import os + +## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key +os.environ["NOVITA_API_KEY"] = "novita-api-key" + +response = completion( + model="novita/deepseek/deepseek-r1", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) +``` + @@ -411,6 +427,23 @@ response = completion( ) ``` + + + +```python +from litellm import completion +import os + +## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key +os.environ["NOVITA_API_KEY"] = "novita_api_key" + +response = completion( + model="novita/deepseek/deepseek-r1", + messages = [{ "content": "Hello, how are you?","role": "user"}], + stream=True, +) +``` + diff --git a/docs/my-website/docs/mcp.md b/docs/my-website/docs/mcp.md index 96d035c9387..f04324f965f 100644 --- a/docs/my-website/docs/mcp.md +++ b/docs/my-website/docs/mcp.md @@ -421,3 +421,9 @@ async with stdio_client(server_params) as (read, write): + +### Permission Management + +Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs. + +Join the discussion [here](https://github.com/BerriAI/litellm/discussions/9891) \ No newline at end of file diff --git a/docs/my-website/docs/observability/langsmith_integration.md b/docs/my-website/docs/observability/langsmith_integration.md index 8f55c854db8..cada4122b20 100644 --- a/docs/my-website/docs/observability/langsmith_integration.md +++ b/docs/my-website/docs/observability/langsmith_integration.md @@ -1,4 +1,6 @@ import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; # Langsmith - Logging LLM Input/Output @@ -22,10 +24,13 @@ pip install litellm ## Quick Start Use just 2 lines of code, to instantly log your responses **across all providers** with Langsmith + + ```python -litellm.success_callback = ["langsmith"] +litellm.callbacks = ["langsmith"] ``` + ```python import litellm import os @@ -37,7 +42,7 @@ os.environ["LANGSMITH_DEFAULT_RUN_NAME"] = "" # defaults to LLMRun os.environ['OPENAI_API_KEY']="" # set langsmith as a callback, litellm will send the data to langsmith -litellm.success_callback = ["langsmith"] +litellm.callbacks = ["langsmith"] # openai call response = litellm.completion( @@ -47,8 +52,124 @@ response = litellm.completion( ] ) ``` + + + +1. Setup config.yaml +```yaml +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +litellm_settings: + callbacks: ["langsmith"] +``` + +2. Start LiteLLM Proxy +```bash +litellm --config /path/to/config.yaml +``` + +3. Test it! +```bash +curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \ +-d '{ + "model": "gpt-3.5-turbo", + "messages": [ + { + "role": "user", + "content": "Hey, how are you?" + } + ], + "max_completion_tokens": 250 +}' +``` + + + + ## Advanced + +### Local Testing - Control Batch Size + +Set the size of the batch that Langsmith will process at a time, default is 512. + +Set `langsmith_batch_size=1` when testing locally, to see logs land quickly. + + + + +```python +import litellm +import os + +os.environ["LANGSMITH_API_KEY"] = "" +# LLM API Keys +os.environ['OPENAI_API_KEY']="" + +# set langsmith as a callback, litellm will send the data to langsmith +litellm.callbacks = ["langsmith"] +litellm.langsmith_batch_size = 1 # 👈 KEY CHANGE + +response = litellm.completion( + model="gpt-3.5-turbo", + messages=[ + {"role": "user", "content": "Hi 👋 - i'm openai"} + ] +) +print(response) +``` + + + +1. Setup config.yaml +```yaml +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +litellm_settings: + langsmith_batch_size: 1 + callbacks: ["langsmith"] +``` + +2. Start LiteLLM Proxy +```bash +litellm --config /path/to/config.yaml +``` + +3. Test it! +```bash +curl -L -X POST 'http://0.0.0.0:4000/v1/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-eWkpOhYaHiuIZV-29JDeTQ' \ +-d '{ + "model": "gpt-3.5-turbo", + "messages": [ + { + "role": "user", + "content": "Hey, how are you?" + } + ], + "max_completion_tokens": 250 +}' +``` + + + + + + + + + ### Set Langsmith fields ```python diff --git a/docs/my-website/docs/observability/opentelemetry_integration.md b/docs/my-website/docs/observability/opentelemetry_integration.md index 5df82c93c87..958c33f18e6 100644 --- a/docs/my-website/docs/observability/opentelemetry_integration.md +++ b/docs/my-website/docs/observability/opentelemetry_integration.md @@ -34,8 +34,9 @@ OTEL_HEADERS="Authorization=Bearer%20" ```shell -OTEL_EXPORTER="otlp_http" -OTEL_ENDPOINT="http://0.0.0.0:4318" +OTEL_EXPORTER_OTLP_ENDPOINT="http://0.0.0.0:4318" +OTEL_EXPORTER_OTLP_PROTOCOL=http/json +OTEL_EXPORTER_OTLP_HEADERS="api-key=key,other-config-value=value" ``` @@ -43,8 +44,9 @@ OTEL_ENDPOINT="http://0.0.0.0:4318" ```shell -OTEL_EXPORTER="otlp_grpc" -OTEL_ENDPOINT="http://0.0.0.0:4317" +OTEL_EXPORTER_OTLP_ENDPOINT="http://0.0.0.0:4318" +OTEL_EXPORTER_OTLP_PROTOCOL=grpc +OTEL_EXPORTER_OTLP_HEADERS="api-key=key,other-config-value=value" ``` @@ -98,7 +100,7 @@ LiteLLM emits the user_api_key_metadata - user_id - team_id -for successful + failed requests +for successful + failed requests click under `litellm_request` in the trace diff --git a/docs/my-website/docs/observability/phoenix_integration.md b/docs/my-website/docs/observability/phoenix_integration.md index 7067a5078b6..d15eea9a834 100644 --- a/docs/my-website/docs/observability/phoenix_integration.md +++ b/docs/my-website/docs/observability/phoenix_integration.md @@ -1,6 +1,6 @@ import Image from '@theme/IdealImage'; -# Phoenix OSS +# Arize Phoenix OSS Open source tracing and evaluation platform diff --git a/docs/my-website/docs/projects/GPTLocalhost.md b/docs/my-website/docs/projects/GPTLocalhost.md new file mode 100644 index 00000000000..791217fe765 --- /dev/null +++ b/docs/my-website/docs/projects/GPTLocalhost.md @@ -0,0 +1,3 @@ +# GPTLocalhost + +[GPTLocalhost](https://gptlocalhost.com/demo#LiteLLM) - LiteLLM is supported by GPTLocalhost, a local Word Add-in for you to use models in LiteLLM within Microsoft Word. 100% Private. diff --git a/docs/my-website/docs/providers/anthropic.md b/docs/my-website/docs/providers/anthropic.md index 95323719f0a..990a9b5122e 100644 --- a/docs/my-website/docs/providers/anthropic.md +++ b/docs/my-website/docs/providers/anthropic.md @@ -750,7 +750,11 @@ except Exception as e: s/o @[Shekhar Patnaik](https://www.linkedin.com/in/patnaikshekhar) for requesting this! -### Computer Tools +### Anthropic Hosted Tools (Computer, Text Editor, Web Search) + + + + ```python from litellm import completion @@ -781,6 +785,205 @@ resp = completion( print(resp) ``` + + + + + + +```python +from litellm import completion + +tools = [{ + "type": "text_editor_20250124", + "name": "str_replace_editor" +}] +model = "claude-3-5-sonnet-20241022" +messages = [{"role": "user", "content": "There's a syntax error in my primes.py file. Can you help me fix it?"}] + +resp = completion( + model=model, + messages=messages, + tools=tools, +) + +print(resp) +``` + + + + +1. Setup config.yaml + +```yaml +- model_name: claude-3-5-sonnet-latest + litellm_params: + model: anthropic/claude-3-5-sonnet-latest + api_key: os.environ/ANTHROPIC_API_KEY +``` + +2. Start proxy + +```bash +litellm --config /path/to/config.yaml +``` + +3. Test it! + +```bash +curl http://0.0.0.0:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer $LITELLM_KEY" \ + -d '{ + "model": "claude-3-5-sonnet-latest", + "messages": [{"role": "user", "content": "There's a syntax error in my primes.py file. Can you help me fix it?"}], + "tools": [{"type": "text_editor_20250124", "name": "str_replace_editor"}] + }' +``` + + + + + + +:::info +Live from v1.70.1+ +::: + +LiteLLM maps OpenAI's `search_context_size` param to Anthropic's `max_uses` param. + +| OpenAI | Anthropic | +| --- | --- | +| Low | 1 | +| Medium | 5 | +| High | 10 | + + + + + + + + + +```python +from litellm import completion + +model = "claude-3-5-sonnet-20241022" +messages = [{"role": "user", "content": "What's the weather like today?"}] + +resp = completion( + model=model, + messages=messages, + web_search_options={ + "search_context_size": "medium", + "user_location": { + "type": "approximate", + "approximate": { + "city": "San Francisco", + }, + } + } +) + +print(resp) +``` + + + +```python +from litellm import completion + +tools = [{ + "type": "web_search_20250305", + "name": "web_search", + "max_uses": 5 +}] +model = "claude-3-5-sonnet-20241022" +messages = [{"role": "user", "content": "There's a syntax error in my primes.py file. Can you help me fix it?"}] + +resp = completion( + model=model, + messages=messages, + tools=tools, +) + +print(resp) +``` + + + + + + + +1. Setup config.yaml + +```yaml +- model_name: claude-3-5-sonnet-latest + litellm_params: + model: anthropic/claude-3-5-sonnet-latest + api_key: os.environ/ANTHROPIC_API_KEY +``` + +2. Start proxy + +```bash +litellm --config /path/to/config.yaml +``` + +3. Test it! + + + + + +```bash +curl http://0.0.0.0:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer $LITELLM_KEY" \ + -d '{ + "model": "claude-3-5-sonnet-latest", + "messages": [{"role": "user", "content": "What's the weather like today?"}], + "web_search_options": { + "search_context_size": "medium", + "user_location": { + "type": "approximate", + "approximate": { + "city": "San Francisco", + }, + } + } + }' +``` + + + +```bash +curl http://0.0.0.0:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer $LITELLM_KEY" \ + -d '{ + "model": "claude-3-5-sonnet-latest", + "messages": [{"role": "user", "content": "What's the weather like today?"}], + "tools": [{ + "type": "web_search_20250305", + "name": "web_search", + "max_uses": 5 + }] + }' +``` + + + + + + + + + + + ## Usage - Vision ```python diff --git a/docs/my-website/docs/providers/azure.md b/docs/my-website/docs/providers/azure/azure.md similarity index 99% rename from docs/my-website/docs/providers/azure.md rename to docs/my-website/docs/providers/azure/azure.md index 2ea444b0295..d0b03719868 100644 --- a/docs/my-website/docs/providers/azure.md +++ b/docs/my-website/docs/providers/azure/azure.md @@ -11,7 +11,7 @@ import TabItem from '@theme/TabItem'; |-------|-------| | Description | Azure OpenAI Service provides REST API access to OpenAI's powerful language models including o1, o1-mini, GPT-4o, GPT-4o mini, GPT-4 Turbo with Vision, GPT-4, GPT-3.5-Turbo, and Embeddings model series | | Provider Route on LiteLLM | `azure/`, [`azure/o_series/`](#azure-o-series-models) | -| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/completions`](#azure-instruct-models), [`/embeddings`](../embedding/supported_embedding#azure-openai-embedding-models), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) | +| Supported Operations | [`/chat/completions`](#azure-openai-chat-completion-models), [`/completions`](#azure-instruct-models), [`/embeddings`](./azure_embedding), [`/audio/speech`](#azure-text-to-speech-tts), [`/audio/transcriptions`](../audio_transcription), `/fine_tuning`, [`/batches`](#azure-batches-api), `/files`, [`/images`](../image_generation#azure-openai-image-generation-models) | | Link to Provider Doc | [Azure OpenAI ↗](https://learn.microsoft.com/en-us/azure/ai-services/openai/overview) ## API Keys, Params diff --git a/docs/my-website/docs/providers/azure/azure_embedding.md b/docs/my-website/docs/providers/azure/azure_embedding.md new file mode 100644 index 00000000000..03bb501f36f --- /dev/null +++ b/docs/my-website/docs/providers/azure/azure_embedding.md @@ -0,0 +1,93 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Azure OpenAI Embeddings + +### API keys +This can be set as env variables or passed as **params to litellm.embedding()** +```python +import os +os.environ['AZURE_API_KEY'] = +os.environ['AZURE_API_BASE'] = +os.environ['AZURE_API_VERSION'] = +``` + +### Usage +```python +from litellm import embedding +response = embedding( + model="azure/", + input=["good morning from litellm"], + api_key=api_key, + api_base=api_base, + api_version=api_version, +) +print(response) +``` + +| Model Name | Function Call | +|----------------------|---------------------------------------------| +| text-embedding-ada-002 | `embedding(model="azure/", input=input)` | + +h/t to [Mikko](https://www.linkedin.com/in/mikkolehtimaki/) for this integration + + +## **Usage - LiteLLM Proxy Server** + +Here's how to call Azure OpenAI models with the LiteLLM Proxy Server + +### 1. Save key in your environment + +```bash +export AZURE_API_KEY="" +``` + +### 2. Start the proxy + +```yaml +model_list: + - model_name: text-embedding-ada-002 + litellm_params: + model: azure/my-deployment-name + api_base: https://openai-gpt-4-test-v-1.openai.azure.com/ + api_version: "2023-05-15" + api_key: os.environ/AZURE_API_KEY # The `os.environ/` prefix tells litellm to read this from the env. +``` + +### 3. Test it + + + + +```shell +curl --location 'http://0.0.0.0:4000/embeddings' \ + --header 'Content-Type: application/json' \ + --data ' { + "model": "text-embedding-ada-002", + "input": ["write a litellm poem"] + }' +``` + + + +```python +import openai +from openai import OpenAI + +# set base_url to your proxy server +# set api_key to send to proxy server +client = OpenAI(api_key="", base_url="http://0.0.0.0:4000") + +response = client.embeddings.create( + input=["hello from litellm"], + model="text-embedding-ada-002" +) + +print(response) + +``` + + + + diff --git a/docs/my-website/docs/providers/bedrock.md b/docs/my-website/docs/providers/bedrock.md index 2a9c528a655..8217f429ff3 100644 --- a/docs/my-website/docs/providers/bedrock.md +++ b/docs/my-website/docs/providers/bedrock.md @@ -60,9 +60,9 @@ Here's how to call Bedrock with the LiteLLM Proxy Server ```yaml model_list: - - model_name: bedrock-claude-v1 + - model_name: bedrock-claude-3-5-sonnet litellm_params: - model: bedrock/anthropic.claude-instant-v1 + model: bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0 aws_access_key_id: os.environ/AWS_ACCESS_KEY_ID aws_secret_access_key: os.environ/AWS_SECRET_ACCESS_KEY aws_region_name: os.environ/AWS_REGION_NAME diff --git a/docs/my-website/docs/providers/bedrock_vector_store.md b/docs/my-website/docs/providers/bedrock_vector_store.md new file mode 100644 index 00000000000..779c4fd0417 --- /dev/null +++ b/docs/my-website/docs/providers/bedrock_vector_store.md @@ -0,0 +1,144 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; +import Image from '@theme/IdealImage'; + +# Bedrock Knowledge Bases + +AWS Bedrock Knowledge Bases allows you to connect your LLM's to your organization's data, letting your models retrieve and reference information specific to your business. + +| Property | Details | +|----------|---------| +| Description | Bedrock Knowledge Bases connects your data to LLM's, enabling them to retrieve and reference your organization's information in their responses. | +| Provider Route on LiteLLM | `bedrock` in the litellm vector_store_registry | +| Provider Doc | [AWS Bedrock Knowledge Bases ↗](https://aws.amazon.com/bedrock/knowledge-bases/) | + +## Quick Start + +### LiteLLM Python SDK + +```python showLineNumbers title="Example using LiteLLM Python SDK" +import os +import litellm + +from litellm.vector_stores.vector_store_registry import VectorStoreRegistry, LiteLLM_ManagedVectorStore + +# Init vector store registry with your Bedrock Knowledge Base +litellm.vector_store_registry = VectorStoreRegistry( + vector_stores=[ + LiteLLM_ManagedVectorStore( + vector_store_id="YOUR_KNOWLEDGE_BASE_ID", # KB ID from AWS Bedrock + custom_llm_provider="bedrock" + ) + ] +) + +# Make a completion request using your Knowledge Base +response = await litellm.acompletion( + model="anthropic/claude-3-5-sonnet", + messages=[{"role": "user", "content": "What does our company policy say about remote work?"}], + tools=[ + { + "type": "file_search", + "vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"] + } + ], +) + +print(response.choices[0].message.content) +``` + +### LiteLLM Proxy + +#### 1. Configure your vector_store_registry + + + + +```yaml +model_list: + - model_name: claude-3-5-sonnet + litellm_params: + model: anthropic/claude-3-5-sonnet + api_key: os.environ/ANTHROPIC_API_KEY + +vector_store_registry: + - vector_store_name: "bedrock-company-docs" + litellm_params: + vector_store_id: "YOUR_KNOWLEDGE_BASE_ID" + custom_llm_provider: "bedrock" + vector_store_description: "Bedrock Knowledge Base for company documents" + vector_store_metadata: + source: "Company internal documentation" +``` + + + + + +On the LiteLLM UI, Navigate to Experimental > Vector Stores > Create Vector Store. On this page you can create a vector store with a name, vector store id and credentials. + + + + + + +#### 2. Make a request with vector_store_ids parameter + + + + +```bash +curl http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer $LITELLM_API_KEY" \ + -d '{ + "model": "claude-3-5-sonnet", + "messages": [{"role": "user", "content": "What does our company policy say about remote work?"}], + "tools": [ + { + "type": "file_search", + "vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"] + } + ] + }' +``` + + + + + +```python +from openai import OpenAI + +# Initialize client with your LiteLLM proxy URL +client = OpenAI( + base_url="http://localhost:4000", + api_key="your-litellm-api-key" +) + +# Make a completion request with vector_store_ids parameter +response = client.chat.completions.create( + model="claude-3-5-sonnet", + messages=[{"role": "user", "content": "What does our company policy say about remote work?"}], + tools=[ + { + "type": "file_search", + "vector_store_ids": ["YOUR_KNOWLEDGE_BASE_ID"] + } + ] +) + +print(response.choices[0].message.content) +``` + + + + + +Futher Reading Vector Stores: +- [Always on Vector Stores](https://docs.litellm.ai/docs/completion/knowledgebase#always-on-for-a-model) +- [Listing available vector stores on litellm proxy](https://docs.litellm.ai/docs/completion/knowledgebase#listing-available-vector-stores) +- [How LiteLLM Vector Stores Work](https://docs.litellm.ai/docs/completion/knowledgebase#how-it-works) \ No newline at end of file diff --git a/docs/my-website/docs/providers/featherless_ai.md b/docs/my-website/docs/providers/featherless_ai.md new file mode 100644 index 00000000000..5b9312e435d --- /dev/null +++ b/docs/my-website/docs/providers/featherless_ai.md @@ -0,0 +1,56 @@ +# Featherless AI +https://featherless.ai/ + +:::tip + +**We support ALL Featherless AI models, just set `model=featherless_ai/` as a prefix when sending litellm requests. For the complete supported model list, visit https://featherless.ai/models ** + +::: + + +## API Key +```python +# env variable +os.environ['FEATHERLESS_AI_API_KEY'] +``` + +## Sample Usage +```python +from litellm import completion +import os + +os.environ['FEATHERLESS_AI_API_KEY'] = "" +response = completion( + model="featherless_ai/featherless-ai/Qwerky-72B", + messages=[{"role": "user", "content": "write code for saying hi from LiteLLM"}] +) +``` + +## Sample Usage - Streaming +```python +from litellm import completion +import os + +os.environ['FEATHERLESS_AI_API_KEY'] = "" +response = completion( + model="featherless_ai/featherless-ai/Qwerky-72B", + messages=[{"role": "user", "content": "write code for saying hi from LiteLLM"}], + stream=True +) + +for chunk in response: + print(chunk) +``` + +## Chat Models +| Model Name | Function Call | +|---------------------------------------------|-----------------------------------------------------------------------------------------------| +| featherless-ai/Qwerky-72B | `completion(model="featherless_ai/featherless-ai/Qwerky-72B", messages)` | +| featherless-ai/Qwerky-QwQ-32B | `completion(model="featherless_ai/featherless-ai/Qwerky-QwQ-32B", messages)` | +| Qwen/Qwen2.5-72B-Instruct | `completion(model="featherless_ai/Qwen/Qwen2.5-72B-Instruct", messages)` | +| all-hands/openhands-lm-32b-v0.1 | `completion(model="featherless_ai/all-hands/openhands-lm-32b-v0.1", messages)` | +| Qwen/Qwen2.5-Coder-32B-Instruct | `completion(model="featherless_ai/Qwen/Qwen2.5-Coder-32B-Instruct", messages)` | +| deepseek-ai/DeepSeek-V3-0324 | `completion(model="featherless_ai/deepseek-ai/DeepSeek-V3-0324", messages)` | +| mistralai/Mistral-Small-24B-Instruct-2501 | `completion(model="featherless_ai/mistralai/Mistral-Small-24B-Instruct-2501", messages)` | +| mistralai/Mistral-Nemo-Instruct-2407 | `completion(model="featherless_ai/mistralai/Mistral-Nemo-Instruct-2407", messages)` | +| ProdeusUnity/Stellar-Odyssey-12b-v0.0 | `completion(model="featherless_ai/ProdeusUnity/Stellar-Odyssey-12b-v0.0", messages)` | diff --git a/docs/my-website/docs/providers/github.md b/docs/my-website/docs/providers/github.md index 023eaf7dcbf..7594b6af4c0 100644 --- a/docs/my-website/docs/providers/github.md +++ b/docs/my-website/docs/providers/github.md @@ -7,6 +7,7 @@ https://github.com/marketplace/models :::tip **We support ALL Github models, just set `model=github/` as a prefix when sending litellm requests** +Ignore company prefix: meta/Llama-3.2-11B-Vision-Instruct becomes model=github/Llama-3.2-11B-Vision-Instruct ::: @@ -23,7 +24,7 @@ import os os.environ['GITHUB_API_KEY'] = "" response = completion( - model="github/llama3-8b-8192", + model="github/Llama-3.2-11B-Vision-Instruct", messages=[ {"role": "user", "content": "hello from litellm"} ], @@ -38,7 +39,7 @@ import os os.environ['GITHUB_API_KEY'] = "" response = completion( - model="github/llama3-8b-8192", + model="github/Llama-3.2-11B-Vision-Instruct", messages=[ {"role": "user", "content": "hello from litellm"} ], @@ -57,9 +58,9 @@ for chunk in response: ```yaml model_list: - - model_name: github-llama3-8b-8192 # Model Alias to use for requests + - model_name: github-Llama-3.2-11B-Vision-Instruct # Model Alias to use for requests litellm_params: - model: github/llama3-8b-8192 + model: github/Llama-3.2-11B-Vision-Instruct api_key: "os.environ/GITHUB_API_KEY" # ensure you have `GITHUB_API_KEY` in your .env ``` @@ -80,7 +81,7 @@ Make request to litellm proxy curl --location 'http://0.0.0.0:4000/chat/completions' \ --header 'Content-Type: application/json' \ --data ' { - "model": "github-llama3-8b-8192", + "model": "github-Llama-3.2-11B-Vision-Instruct", "messages": [ { "role": "user", @@ -100,7 +101,7 @@ client = openai.OpenAI( base_url="http://0.0.0.0:4000" ) -response = client.chat.completions.create(model="github-llama3-8b-8192", messages = [ +response = client.chat.completions.create(model="github-Llama-3.2-11B-Vision-Instruct", messages = [ { "role": "user", "content": "this is a test request, write a short poem" @@ -124,7 +125,7 @@ from langchain.schema import HumanMessage, SystemMessage chat = ChatOpenAI( openai_api_base="http://0.0.0.0:4000", # set openai_api_base to the LiteLLM Proxy - model = "github-llama3-8b-8192", + model = "github-Llama-3.2-11B-Vision-Instruct", temperature=0.1 ) @@ -152,7 +153,7 @@ We support ALL Github models, just set `github/` as a prefix when sending comple |--------------------|---------------------------------------------------------| | llama-3.1-8b-instant | `completion(model="github/llama-3.1-8b-instant", messages)` | | llama-3.1-70b-versatile | `completion(model="github/llama-3.1-70b-versatile", messages)` | -| llama3-8b-8192 | `completion(model="github/llama3-8b-8192", messages)` | +| Llama-3.2-11B-Vision-Instruct | `completion(model="github/Llama-3.2-11B-Vision-Instruct", messages)` | | llama3-70b-8192 | `completion(model="github/llama3-70b-8192", messages)` | | llama2-70b-4096 | `completion(model="github/llama2-70b-4096", messages)` | | mixtral-8x7b-32768 | `completion(model="github/mixtral-8x7b-32768", messages)` | @@ -214,7 +215,7 @@ tools = [ } ] response = litellm.completion( - model="github/llama3-8b-8192", + model="github/Llama-3.2-11B-Vision-Instruct", messages=messages, tools=tools, tool_choice="auto", # auto is default, but we'll be explicit @@ -254,7 +255,7 @@ if tool_calls: ) # extend conversation with function response print(f"messages: {messages}") second_response = litellm.completion( - model="github/llama3-8b-8192", messages=messages + model="github/Llama-3.2-11B-Vision-Instruct", messages=messages ) # get a new response from the model where it can see the function response print("second response\n", second_response) ``` diff --git a/docs/my-website/docs/providers/google_ai_studio/realtime.md b/docs/my-website/docs/providers/google_ai_studio/realtime.md new file mode 100644 index 00000000000..50a18e131cc --- /dev/null +++ b/docs/my-website/docs/providers/google_ai_studio/realtime.md @@ -0,0 +1,92 @@ +# Gemini Realtime API - Google AI Studio + +| Feature | Description | Comments | +| --- | --- | --- | +| Proxy | ✅ | | +| SDK | ⌛️ | Experimental access via `litellm._arealtime`. | + + +## Proxy Usage + +### Add model to config + +```yaml +model_list: + - model_name: "gemini-2.0-flash" + litellm_params: + model: gemini/gemini-2.0-flash-live-001 + model_info: + mode: realtime +``` + +### Start proxy + +```bash +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:8000 +``` + +### Test + +Run this script using node - `node test.js` + +```js +// test.js +const WebSocket = require("ws"); + +const url = "ws://0.0.0.0:4000/v1/realtime?model=openai-gemini-2.0-flash"; + +const ws = new WebSocket(url, { + headers: { + "api-key": `${LITELLM_API_KEY}`, + "OpenAI-Beta": "realtime=v1", + }, +}); + +ws.on("open", function open() { + console.log("Connected to server."); + ws.send(JSON.stringify({ + type: "response.create", + response: { + modalities: ["text"], + instructions: "Please assist the user.", + } + })); +}); + +ws.on("message", function incoming(message) { + console.log(JSON.parse(message.toString())); +}); + +ws.on("error", function handleError(error) { + console.error("Error: ", error); +}); +``` + +## Limitations + +- Does not support audio transcription. +- Does not support tool calling + +## Supported OpenAI Realtime Events + +- `session.created` +- `response.created` +- `response.output_item.added` +- `conversation.item.created` +- `response.content_part.added` +- `response.text.delta` +- `response.audio.delta` +- `response.text.done` +- `response.audio.done` +- `response.content_part.done` +- `response.output_item.done` +- `response.done` + + + +## [Supported Session Params](https://github.com/BerriAI/litellm/blob/e87b536d038f77c2a2206fd7433e275c487179ee/litellm/llms/gemini/realtime/transformation.py#L155) + +## More Examples +### [Gemini Realtime API with Audio Input/Output](../../../docs/tutorials/gemini_realtime_with_audio) \ No newline at end of file diff --git a/docs/my-website/docs/providers/litellm_proxy.md b/docs/my-website/docs/providers/litellm_proxy.md index a66423dac54..a9de5d5913d 100644 --- a/docs/my-website/docs/providers/litellm_proxy.md +++ b/docs/my-website/docs/providers/litellm_proxy.md @@ -155,6 +155,53 @@ response = litellm.rerank( api_key="your-litellm-proxy-api-key" ) ``` -## **Usage with Langchain, LLamaindex, OpenAI Js, Anthropic SDK, Instructor** -#### [Follow this doc to see how to use litellm proxy with langchain, llamaindex, anthropic etc](../proxy/user_keys) \ No newline at end of file + +## Integration with Other Libraries + +LiteLLM Proxy works seamlessly with Langchain, LlamaIndex, OpenAI JS, Anthropic SDK, Instructor, and more. + +[Learn how to use LiteLLM proxy with these libraries →](../proxy/user_keys) + +## Send all SDK requests to LiteLLM Proxy + +Use this when calling LiteLLM Proxy from any library / codebase already using the LiteLLM SDK. + +These flags will route all requests through your LiteLLM proxy, regardless of the model specified. + +When enabled, requests will use `LITELLM_PROXY_API_BASE` with `LITELLM_PROXY_API_KEY` as the authentication. + +### Option 1: Set Globally in Code + +```python +# Set the flag globally for all requests +litellm.use_litellm_proxy = True + +response = litellm.completion( + model="vertex_ai/gemini-2.0-flash-001", + messages=[{"role": "user", "content": "Hello, how are you?"}] +) +``` + +### Option 2: Control via Environment Variable + +```python +# Control proxy usage through environment variable +os.environ["USE_LITELLM_PROXY"] = "True" + +response = litellm.completion( + model="vertex_ai/gemini-2.0-flash-001", + messages=[{"role": "user", "content": "Hello, how are you?"}] +) +``` + +### Option 3: Set Per Request + +```python +# Enable proxy for specific requests only +response = litellm.completion( + model="vertex_ai/gemini-2.0-flash-001", + messages=[{"role": "user", "content": "Hello, how are you?"}], + use_litellm_proxy=True +) +``` diff --git a/docs/my-website/docs/providers/lm_studio.md b/docs/my-website/docs/providers/lm_studio.md index 45c546ada68..0cf9acff33d 100644 --- a/docs/my-website/docs/providers/lm_studio.md +++ b/docs/my-website/docs/providers/lm_studio.md @@ -153,3 +153,26 @@ response = embedding( ) print(response) ``` + + +## Structured Output + +LM Studio supports structured outputs via JSON Schema. You can pass a pydantic model or a raw schema using `response_format`. +LiteLLM sends the schema as `{ "type": "json_schema", "json_schema": {"schema": } }`. + +```python +from pydantic import BaseModel +from litellm import completion + +class Book(BaseModel): + title: str + author: str + year: int + +response = completion( + model="lm_studio/llama-3-8b-instruct", + messages=[{"role": "user", "content": "Tell me about The Hobbit"}], + response_format=Book, +) +print(response.choices[0].message.content) +``` \ No newline at end of file diff --git a/docs/my-website/docs/providers/meta_llama.md b/docs/my-website/docs/providers/meta_llama.md new file mode 100644 index 00000000000..8219bef12b2 --- /dev/null +++ b/docs/my-website/docs/providers/meta_llama.md @@ -0,0 +1,205 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Meta Llama + +| Property | Details | +|-------|-------| +| Description | Meta's Llama API provides access to Meta's family of large language models. | +| Provider Route on LiteLLM | `meta_llama/` | +| Supported Endpoints | `/chat/completions`, `/completions`, `/responses` | +| API Reference | [Llama API Reference ↗](https://llama.developer.meta.com?utm_source=partner-litellm&utm_medium=website) | + +## Required Variables + +```python showLineNumbers title="Environment Variables" +os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key +``` + +## Supported Models + +:::info +All models listed here https://llama.developer.meta.com/docs/models/ are supported. We actively maintain the list of models, token window, etc. [here](https://github.com/BerriAI/litellm/blob/main/model_prices_and_context_window.json). + +::: + + +| Model ID | Input context length | Output context length | Input Modalities | Output Modalities | +| --- | --- | --- | --- | --- | +| `Llama-4-Scout-17B-16E-Instruct-FP8` | 128k | 4028 | Text, Image | Text | +| `Llama-4-Maverick-17B-128E-Instruct-FP8` | 128k | 4028 | Text, Image | Text | +| `Llama-3.3-70B-Instruct` | 128k | 4028 | Text | Text | +| `Llama-3.3-8B-Instruct` | 128k | 4028 | Text | Text | + +## Usage - LiteLLM Python SDK + +### Non-streaming + +```python showLineNumbers title="Meta Llama Non-streaming Completion" +import os +import litellm +from litellm import completion + +os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key + +messages = [{"content": "Hello, how are you?", "role": "user"}] + +# Meta Llama call +response = completion(model="meta_llama/Llama-3.3-70B-Instruct", messages=messages) +``` + +### Streaming + +```python showLineNumbers title="Meta Llama Streaming Completion" +import os +import litellm +from litellm import completion + +os.environ["LLAMA_API_KEY"] = "" # your Meta Llama API key + +messages = [{"content": "Hello, how are you?", "role": "user"}] + +# Meta Llama call with streaming +response = completion( + model="meta_llama/Llama-3.3-70B-Instruct", + messages=messages, + stream=True +) + +for chunk in response: + print(chunk) +``` + + +## Usage - LiteLLM Proxy + + +Add the following to your LiteLLM Proxy configuration file: + +```yaml showLineNumbers title="config.yaml" +model_list: + - model_name: meta_llama/Llama-3.3-70B-Instruct + litellm_params: + model: meta_llama/Llama-3.3-70B-Instruct + api_key: os.environ/LLAMA_API_KEY + + - model_name: meta_llama/Llama-3.3-8B-Instruct + litellm_params: + model: meta_llama/Llama-3.3-8B-Instruct + api_key: os.environ/LLAMA_API_KEY +``` + +Start your LiteLLM Proxy server: + +```bash showLineNumbers title="Start LiteLLM Proxy" +litellm --config config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + + + + +```python showLineNumbers title="Meta Llama via Proxy - Non-streaming" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-proxy-api-key" # Your proxy API key +) + +# Non-streaming response +response = client.chat.completions.create( + model="meta_llama/Llama-3.3-70B-Instruct", + messages=[{"role": "user", "content": "Write a short poem about AI."}] +) + +print(response.choices[0].message.content) +``` + +```python showLineNumbers title="Meta Llama via Proxy - Streaming" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-proxy-api-key" # Your proxy API key +) + +# Streaming response +response = client.chat.completions.create( + model="meta_llama/Llama-3.3-70B-Instruct", + messages=[{"role": "user", "content": "Write a short poem about AI."}], + stream=True +) + +for chunk in response: + if chunk.choices[0].delta.content is not None: + print(chunk.choices[0].delta.content, end="") +``` + + + + + +```python showLineNumbers title="Meta Llama via Proxy - LiteLLM SDK" +import litellm + +# Configure LiteLLM to use your proxy +response = litellm.completion( + model="litellm_proxy/meta_llama/Llama-3.3-70B-Instruct", + messages=[{"role": "user", "content": "Write a short poem about AI."}], + api_base="http://localhost:4000", + api_key="your-proxy-api-key" +) + +print(response.choices[0].message.content) +``` + +```python showLineNumbers title="Meta Llama via Proxy - LiteLLM SDK Streaming" +import litellm + +# Configure LiteLLM to use your proxy with streaming +response = litellm.completion( + model="litellm_proxy/meta_llama/Llama-3.3-70B-Instruct", + messages=[{"role": "user", "content": "Write a short poem about AI."}], + api_base="http://localhost:4000", + api_key="your-proxy-api-key", + stream=True +) + +for chunk in response: + if hasattr(chunk.choices[0], 'delta') and chunk.choices[0].delta.content is not None: + print(chunk.choices[0].delta.content, end="") +``` + + + + + +```bash showLineNumbers title="Meta Llama via Proxy - cURL" +curl http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer your-proxy-api-key" \ + -d '{ + "model": "meta_llama/Llama-3.3-70B-Instruct", + "messages": [{"role": "user", "content": "Write a short poem about AI."}] + }' +``` + +```bash showLineNumbers title="Meta Llama via Proxy - cURL Streaming" +curl http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer your-proxy-api-key" \ + -d '{ + "model": "meta_llama/Llama-3.3-70B-Instruct", + "messages": [{"role": "user", "content": "Write a short poem about AI."}], + "stream": true + }' +``` + + + + +For more detailed information on using the LiteLLM Proxy, see the [LiteLLM Proxy documentation](../providers/litellm_proxy). diff --git a/docs/my-website/docs/providers/novita.md b/docs/my-website/docs/providers/novita.md new file mode 100644 index 00000000000..f879ef4abac --- /dev/null +++ b/docs/my-website/docs/providers/novita.md @@ -0,0 +1,234 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Novita AI + +| Property | Details | +|-------|-------| +| Description | Novita AI is an AI cloud platform that helps developers easily deploy AI models through a simple API, backed by affordable and reliable GPU cloud infrastructure. LiteLLM supports all models from [Novita AI](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) | +| Provider Route on LiteLLM | `novita/` | +| Provider Doc | [Novita AI Docs ↗](https://novita.ai/docs/guides/introduction) | +| API Endpoint for Provider | https://api.novita.ai/v3/openai | +| Supported OpenAI Endpoints | `/chat/completions`, `/completions` | + +
+ +## API Keys + +Get your API key [here](https://novita.ai/settings/key-management) +```python +import os +os.environ["NOVITA_API_KEY"] = "your-api-key" +``` + +## Supported OpenAI Params +- max_tokens +- stream +- stream_options +- n +- seed +- frequency_penalty +- presence_penalty +- repetition_penalty +- stop +- temperature +- top_p +- top_k +- min_p +- logit_bias +- logprobs +- top_logprobs +- tools +- response_format +- separate_reasoning + + +## Sample Usage + + + + +```python +import os +from litellm import completion +os.environ["NOVITA_API_KEY"] = "" + +response = completion( + model="novita/deepseek/deepseek-r1-turbo", + messages=[{"role": "user", "content": "List 5 popular cookie recipes."}] +) + +content = response.get('choices', [{}])[0].get('message', {}).get('content') +print(content) +``` + + + + +1. Add model to config.yaml +```yaml +model_list: + - model_name: deepseek-r1-turbo + litellm_params: + model: novita/deepseek/deepseek-r1-turbo + api_key: os.environ/NOVITA_API_KEY +``` + +2. Start Proxy + +``` +$ litellm --config /path/to/config.yaml +``` + +3. Make Request! + +```bash +curl -X POST 'http://0.0.0.0:4000/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk_sujEQQEjTRxGUiMLN3TJh2KadRX4pw2TLWRoIKeoYZ0' \ +-d '{ + "model": "deepseek-r1-turbo", + "messages": [ + {"role": "user", "content": "List 5 popular cookie recipes."} + ] +} +' +``` + + + + + +## Tool Calling + +```python +from litellm import completion +import os +# set env +os.environ["NOVITA_API_KEY"] = "" + +tools = [ + { + "type": "function", + "function": { + "name": "get_current_weather", + "description": "Get the current weather in a given location", + "parameters": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "The city and state, e.g. San Francisco, CA", + }, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, + }, + "required": ["location"], + }, + }, + } +] +messages = [{"role": "user", "content": "What's the weather like in Boston today?"}] + +response = completion( + model="novita/deepseek/deepseek-r1-turbo", + messages=messages, + tools=tools, +) +# Add any assertions, here to check response args +print(response) +assert isinstance(response.choices[0].message.tool_calls[0].function.name, str) +assert isinstance( + response.choices[0].message.tool_calls[0].function.arguments, str +) + +``` + +## JSON Mode + + + + +```python +from litellm import completion +import json +import os + +os.environ['NOVITA_API_KEY'] = "" + +messages = [ + { + "role": "user", + "content": "List 5 popular cookie recipes." + } +] + +completion( + model="novita/deepseek/deepseek-r1-turbo", + messages=messages, + response_format={"type": "json_object"} # 👈 KEY CHANGE +) + +print(json.loads(completion.choices[0].message.content)) +``` + + + + +1. Add model to config.yaml +```yaml +model_list: + - model_name: deepseek-r1-turbo + litellm_params: + model: novita/deepseek/deepseek-r1-turbo + api_key: os.environ/NOVITA_API_KEY +``` + +2. Start Proxy + +``` +$ litellm --config /path/to/config.yaml +``` + +3. Make Request! + +```bash +curl -X POST 'http://0.0.0.0:4000/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-1234' \ +-d '{ + "model": "deepseek-r1-turbo", + "messages": [ + {"role": "user", "content": "List 5 popular cookie recipes."} + ], + "response_format": {"type": "json_object"} +} +' +``` + + + + + +## Chat Models + +🚨 LiteLLM supports ALL Novita AI models, send `model=novita/` to send it to Novita AI. See all Novita AI models [here](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) + +| Model Name | Function Call | +|---------------------------|-----------------------------------------------------| +| novita/deepseek/deepseek-r1-turbo | `completion('novita/deepseek/deepseek-r1-turbo', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/deepseek/deepseek-v3-turbo | `completion('novita/deepseek/deepseek-v3-turbo', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/deepseek/deepseek-v3-0324 | `completion('novita/deepseek/deepseek-v3-0324', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/qwen/qwen3-235b-a22b-fp8 | `completion('novita/qwen/qwen/qwen3-235b-a22b-fp8', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/qwen/qwen3-30b-a3b-fp8 | `completion('novita/qwen/qwen3-30b-a3b-fp8', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/qwen/qwen/qwen3-32b-fp8 | `completion('novita/qwen/qwen3-32b-fp8', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/qwen/qwen3-30b-a3b-fp8 | `completion('novita/qwen/qwen3-30b-a3b-fp8', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/qwen/qwen2.5-vl-72b-instruct | `completion('novita/qwen/qwen2.5-vl-72b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-4-maverick-17b-128e-instruct-fp8 | `completion('novita/meta-llama/llama-4-maverick-17b-128e-instruct-fp8', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.3-70b-instruct | `completion('novita/meta-llama/llama-3.3-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.1-8b-instruct | `completion('novita/meta-llama/llama-3.1-8b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.1-8b-instruct-max | `completion('novita/meta-llama/llama-3.1-8b-instruct-max', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.1-70b-instruct | `completion('novita/meta-llama/llama-3.1-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/gryphe/mythomax-l2-13b | `completion('novita/gryphe/mythomax-l2-13b', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/google/gemma-3-27b-it | `completion('novita/google/gemma-3-27b-it', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/mistralai/mistral-nemo | `completion('novita/mistralai/mistral-nemo', messages)` | `os.environ['NOVITA_API_KEY']` | \ No newline at end of file diff --git a/docs/my-website/docs/providers/nscale.md b/docs/my-website/docs/providers/nscale.md new file mode 100644 index 00000000000..0413253a4be --- /dev/null +++ b/docs/my-website/docs/providers/nscale.md @@ -0,0 +1,180 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Nscale (EU Sovereign) + +https://docs.nscale.com/docs/inference/chat + +:::tip + +**We support ALL Nscale models, just set `model=nscale/` as a prefix when sending litellm requests** + +::: + +| Property | Details | +|-------|-------| +| Description | European-domiciled full-stack AI cloud platform for LLMs and image generation. | +| Provider Route on LiteLLM | `nscale/` | +| Supported Endpoints | `/chat/completions`, `/images/generations` | +| API Reference | [Nscale docs](https://docs.nscale.com/docs/getting-started/overview) | + +## Required Variables + +```python showLineNumbers title="Environment Variables" +os.environ["NSCALE_API_KEY"] = "" # your Nscale API key +``` + +## Explore Available Models + +Explore our full list of text and multimodal AI models — all available at highly competitive pricing: +📚 [Full List of Models](https://docs.nscale.com/docs/inference/serverless-models/current) + + +## Key Features +- **EU Sovereign**: Full data sovereignty and compliance with European regulations +- **Ultra-Low Cost (starting at $0.01 / M tokens)**: Extremely competitive pricing for both text and image generation models +- **Production Grade**: Reliable serverless deployments with full isolation +- **No Setup Required**: Instant access to compute without infrastructure management +- **Full Control**: Your data remains private and isolated + +## Usage - LiteLLM Python SDK + +### Text Generation + +```python showLineNumbers title="Nscale Text Generation" +from litellm import completion +import os + +os.environ["NSCALE_API_KEY"] = "" # your Nscale API key +response = completion( + model="nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct", + messages=[{"role": "user", "content": "What is LiteLLM?"}] +) +print(response) +``` + +```python showLineNumbers title="Nscale Text Generation - Streaming" +from litellm import completion +import os + +os.environ["NSCALE_API_KEY"] = "" # your Nscale API key +stream = completion( + model="nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct", + messages=[{"role": "user", "content": "What is LiteLLM?"}], + stream=True +) + +for chunk in stream: + if chunk.choices[0].delta.content is not None: + print(chunk.choices[0].delta.content, end="") +``` + +### Image Generation + +```python showLineNumbers title="Nscale Image Generation" +from litellm import image_generation +import os + +os.environ["NSCALE_API_KEY"] = "" # your Nscale API key +response = image_generation( + model="nscale/stabilityai/stable-diffusion-xl-base-1.0", + prompt="A beautiful sunset over mountains", + n=1, + size="1024x1024" +) +print(response) +``` + +## Usage - LiteLLM Proxy + +Add the following to your LiteLLM Proxy configuration file: + +```yaml showLineNumbers title="config.yaml" +model_list: + - model_name: nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct + litellm_params: + model: nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct + api_key: os.environ/NSCALE_API_KEY + - model_name: nscale/meta-llama/Llama-3.3-70B-Instruct + litellm_params: + model: nscale/meta-llama/Llama-3.3-70B-Instruct + api_key: os.environ/NSCALE_API_KEY + - model_name: nscale/stabilityai/stable-diffusion-xl-base-1.0 + litellm_params: + model: nscale/stabilityai/stable-diffusion-xl-base-1.0 + api_key: os.environ/NSCALE_API_KEY +``` + +Start your LiteLLM Proxy server: + +```bash showLineNumbers title="Start LiteLLM Proxy" +litellm --config config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + + + + +```python showLineNumbers title="Nscale via Proxy - Non-streaming" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-proxy-api-key" # Your proxy API key +) + +# Non-streaming response +response = client.chat.completions.create( + model="nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct", + messages=[{"role": "user", "content": "What is LiteLLM?"}] +) + +print(response.choices[0].message.content) +``` + + + + + +```python showLineNumbers title="Nscale via Proxy - LiteLLM SDK" +import litellm + +# Configure LiteLLM to use your proxy +response = litellm.completion( + model="litellm_proxy/nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct", + messages=[{"role": "user", "content": "What is LiteLLM?"}], + api_base="http://localhost:4000", + api_key="your-proxy-api-key" +) + +print(response.choices[0].message.content) +``` + + + + + +```bash showLineNumbers title="Nscale via Proxy - cURL" +curl http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer your-proxy-api-key" \ + -d '{ + "model": "nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct", + "messages": [{"role": "user", "content": "What is LiteLLM?"}] + }' +``` + + + + +## Getting Started +1. Create an account at [console.nscale.com](https://console.nscale.com) +2. Claim free credit +3. Create an API key in settings +4. Start making API calls using LiteLLM + +## Additional Resources +- [Nscale Documentation](https://docs.nscale.com/docs/getting-started/overview) +- [Blog: Sovereign Serverless](https://www.nscale.com/blog/sovereign-serverless-how-we-designed-full-isolation-without-sacrificing-performance) diff --git a/docs/my-website/docs/providers/nvidia_nim.md b/docs/my-website/docs/providers/nvidia_nim.md index 04390e7efec..270b356c917 100644 --- a/docs/my-website/docs/providers/nvidia_nim.md +++ b/docs/my-website/docs/providers/nvidia_nim.md @@ -10,10 +10,19 @@ https://docs.api.nvidia.com/nim/reference/ ::: +| Property | Details | +|-------|-------| +| Description | Nvidia NIM is a platform that provides a simple API for deploying and using AI models. LiteLLM supports all models from [Nvidia NIM](https://developer.nvidia.com/nim/) | +| Provider Route on LiteLLM | `nvidia_nim/` | +| Provider Doc | [Nvidia NIM Docs ↗](https://developer.nvidia.com/nim/) | +| API Endpoint for Provider | https://integrate.api.nvidia.com/v1/ | +| Supported OpenAI Endpoints | `/chat/completions`, `/completions`, `/responses`, `/embeddings` | + ## API Key ```python # env variable -os.environ['NVIDIA_NIM_API_KEY'] +os.environ['NVIDIA_NIM_API_KEY'] = "" +os.environ['NVIDIA_NIM_API_BASE'] = "" # [OPTIONAL] - default is https://integrate.api.nvidia.com/v1/ ``` ## Sample Usage @@ -100,6 +109,7 @@ Here's how to call an Nvidia NIM Endpoint with the LiteLLM Proxy Server litellm_params: model: nvidia_nim/ # add nvidia_nim/ prefix to route as Nvidia NIM provider api_key: api-key # api key to send your model + # api_base: "" # [OPTIONAL] - default is https://integrate.api.nvidia.com/v1/ ``` diff --git a/docs/my-website/docs/providers/openai/responses_api.md b/docs/my-website/docs/providers/openai/responses_api.md new file mode 100644 index 00000000000..578ce038f37 --- /dev/null +++ b/docs/my-website/docs/providers/openai/responses_api.md @@ -0,0 +1,320 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# OpenAI - Response API + +## Usage + +### LiteLLM Python SDK + + +#### Non-streaming +```python showLineNumbers title="OpenAI Non-streaming Response" +import litellm + +# Non-streaming response +response = litellm.responses( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + max_output_tokens=100 +) + +print(response) +``` + +#### Streaming +```python showLineNumbers title="OpenAI Streaming Response" +import litellm + +# Streaming response +response = litellm.responses( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + stream=True +) + +for event in response: + print(event) +``` + +#### GET a Response +```python showLineNumbers title="Get Response by ID" +import litellm + +# First, create a response +response = litellm.responses( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + max_output_tokens=100 +) + +# Get the response ID +response_id = response.id + +# Retrieve the response by ID +retrieved_response = litellm.get_responses( + response_id=response_id +) + +print(retrieved_response) + +# For async usage +# retrieved_response = await litellm.aget_responses(response_id=response_id) +``` + +#### DELETE a Response +```python showLineNumbers title="Delete Response by ID" +import litellm + +# First, create a response +response = litellm.responses( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + max_output_tokens=100 +) + +# Get the response ID +response_id = response.id + +# Delete the response by ID +delete_response = litellm.delete_responses( + response_id=response_id +) + +print(delete_response) + +# For async usage +# delete_response = await litellm.adelete_responses(response_id=response_id) +``` + + +### LiteLLM Proxy with OpenAI SDK + +1. Set up config.yaml + +```yaml showLineNumbers title="OpenAI Proxy Configuration" +model_list: + - model_name: openai/o1-pro + litellm_params: + model: openai/o1-pro + api_key: os.environ/OPENAI_API_KEY +``` + +2. Start LiteLLM Proxy Server + +```bash title="Start LiteLLM Proxy Server" +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +3. Use OpenAI SDK with LiteLLM Proxy + +#### Non-streaming +```python showLineNumbers title="OpenAI Proxy Non-streaming Response" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Non-streaming response +response = client.responses.create( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn." +) + +print(response) +``` + +#### Streaming +```python showLineNumbers title="OpenAI Proxy Streaming Response" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Streaming response +response = client.responses.create( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn.", + stream=True +) + +for event in response: + print(event) +``` + +#### GET a Response +```python showLineNumbers title="Get Response by ID with OpenAI SDK" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# First, create a response +response = client.responses.create( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn." +) + +# Get the response ID +response_id = response.id + +# Retrieve the response by ID +retrieved_response = client.responses.retrieve(response_id) + +print(retrieved_response) +``` + +#### DELETE a Response +```python showLineNumbers title="Delete Response by ID with OpenAI SDK" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# First, create a response +response = client.responses.create( + model="openai/o1-pro", + input="Tell me a three sentence bedtime story about a unicorn." +) + +# Get the response ID +response_id = response.id + +# Delete the response by ID +delete_response = client.responses.delete(response_id) + +print(delete_response) +``` + + +## Supported Responses API Parameters + +| Provider | Supported Parameters | +|----------|---------------------| +| `openai` | [All Responses API parameters are supported](https://github.com/BerriAI/litellm/blob/7c3df984da8e4dff9201e4c5353fdc7a2b441831/litellm/llms/openai/responses/transformation.py#L23) | + +## Computer Use + + + + +```python +import litellm + +# Non-streaming response +response = litellm.responses( + model="computer-use-preview", + tools=[{ + "type": "computer_use_preview", + "display_width": 1024, + "display_height": 768, + "environment": "browser" # other possible values: "mac", "windows", "ubuntu" + }], + input=[ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "Check the latest OpenAI news on bing.com." + } + # Optional: include a screenshot of the initial state of the environment + # { + # type: "input_image", + # image_url: f"data:image/png;base64,{screenshot_base64}" + # } + ] + } + ], + reasoning={ + "summary": "concise", + }, + truncation="auto" +) + +print(response.output) +``` + + + + +1. Set up config.yaml + +```yaml showLineNumbers title="OpenAI Proxy Configuration" +model_list: + - model_name: openai/o1-pro + litellm_params: + model: openai/o1-pro + api_key: os.environ/OPENAI_API_KEY +``` + +2. Start LiteLLM Proxy Server + +```bash title="Start LiteLLM Proxy Server" +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +3. Test it! + +```python showLineNumbers title="OpenAI Proxy Non-streaming Response" +from openai import OpenAI + +# Initialize client with your proxy URL +client = OpenAI( + base_url="http://localhost:4000", # Your proxy URL + api_key="your-api-key" # Your proxy API key +) + +# Non-streaming response +response = client.responses.create( + model="computer-use-preview", + tools=[{ + "type": "computer_use_preview", + "display_width": 1024, + "display_height": 768, + "environment": "browser" # other possible values: "mac", "windows", "ubuntu" + }], + input=[ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "Check the latest OpenAI news on bing.com." + } + # Optional: include a screenshot of the initial state of the environment + # { + # type: "input_image", + # image_url: f"data:image/png;base64,{screenshot_base64}" + # } + ] + } + ], + reasoning={ + "summary": "concise", + }, + truncation="auto" +) + +print(response) +``` + + + + diff --git a/docs/my-website/docs/providers/openai/text_to_speech.md b/docs/my-website/docs/providers/openai/text_to_speech.md new file mode 100644 index 00000000000..34cd0f069e6 --- /dev/null +++ b/docs/my-website/docs/providers/openai/text_to_speech.md @@ -0,0 +1,122 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# OpenAI - Text-to-speech + +## **LiteLLM Python SDK Usage** +### Quick Start + +```python +from pathlib import Path +from litellm import speech +import os + +os.environ["OPENAI_API_KEY"] = "sk-.." + +speech_file_path = Path(__file__).parent / "speech.mp3" +response = speech( + model="openai/tts-1", + voice="alloy", + input="the quick brown fox jumped over the lazy dogs", + ) +response.stream_to_file(speech_file_path) +``` + +### Async Usage + +```python +from litellm import aspeech +from pathlib import Path +import os, asyncio + +os.environ["OPENAI_API_KEY"] = "sk-.." + +async def test_async_speech(): + speech_file_path = Path(__file__).parent / "speech.mp3" + response = await litellm.aspeech( + model="openai/tts-1", + voice="alloy", + input="the quick brown fox jumped over the lazy dogs", + api_base=None, + api_key=None, + organization=None, + project=None, + max_retries=1, + timeout=600, + client=None, + optional_params={}, + ) + response.stream_to_file(speech_file_path) + +asyncio.run(test_async_speech()) +``` + +## **LiteLLM Proxy Usage** + +LiteLLM provides an openai-compatible `/audio/speech` endpoint for Text-to-speech calls. + +```bash +curl http://0.0.0.0:4000/v1/audio/speech \ + -H "Authorization: Bearer sk-1234" \ + -H "Content-Type: application/json" \ + -d '{ + "model": "tts-1", + "input": "The quick brown fox jumped over the lazy dog.", + "voice": "alloy" + }' \ + --output speech.mp3 +``` + +**Setup** + +```bash +- model_name: tts + litellm_params: + model: openai/tts-1 + api_key: os.environ/OPENAI_API_KEY +``` + +```bash +litellm --config /path/to/config.yaml + +# RUNNING on http://0.0.0.0:4000 +``` + +## Supported Models + +| Model | Example | +|-------|-------------| +| tts-1 | speech(model="tts-1", voice="alloy", input="Hello, world!") | +| tts-1-hd | speech(model="tts-1-hd", voice="alloy", input="Hello, world!") | +| gpt-4o-mini-tts | speech(model="gpt-4o-mini-tts", voice="alloy", input="Hello, world!") | + + +## ✨ Enterprise LiteLLM Proxy - Set Max Request File Size + +Use this when you want to limit the file size for requests sent to `audio/transcriptions` + +```yaml +- model_name: whisper + litellm_params: + model: whisper-1 + api_key: sk-******* + max_file_size_mb: 0.00001 # 👈 max file size in MB (Set this intentionally very small for testing) + model_info: + mode: audio_transcription +``` + +Make a test Request with a valid file +```shell +curl --location 'http://localhost:4000/v1/audio/transcriptions' \ +--header 'Authorization: Bearer sk-1234' \ +--form 'file=@"/Users/ishaanjaffer/Github/litellm/tests/gettysburg.wav"' \ +--form 'model="whisper"' +``` + + +Expect to see the follow response + +```shell +{"error":{"message":"File size is too large. Please check your file size. Passed file size: 0.7392807006835938 MB. Max file size: 0.0001 MB","type":"bad_request","param":"file","code":500}}% +``` \ No newline at end of file diff --git a/docs/my-website/docs/providers/sambanova.md b/docs/my-website/docs/providers/sambanova.md index 7dd837e1b0a..290b64a1f09 100644 --- a/docs/my-website/docs/providers/sambanova.md +++ b/docs/my-website/docs/providers/sambanova.md @@ -1,8 +1,8 @@ import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; -# Sambanova -https://cloud.sambanova.ai/ +# SambaNova +[https://cloud.sambanova.ai/](http://cloud.sambanova.ai?utm_source=litellm&utm_medium=external&utm_campaign=cloud_signup) :::tip @@ -23,20 +23,17 @@ import os os.environ['SAMBANOVA_API_KEY'] = "" response = completion( - model="sambanova/Meta-Llama-3.1-8B-Instruct", + model="sambanova/Llama-4-Maverick-17B-128E-Instruct", messages=[ { "role": "user", - "content": "What do you know about sambanova.ai. Give your response in json format", + "content": "What do you know about SambaNova Systems", } ], max_tokens=10, - response_format={ "type": "json_object" }, - stop=["\n\n"], + stop=[], temperature=0.2, top_p=0.9, - tool_choice="auto", - tools=[], user="user", ) print(response) @@ -49,17 +46,17 @@ import os os.environ['SAMBANOVA_API_KEY'] = "" response = completion( - model="sambanova/Meta-Llama-3.1-8B-Instruct", + model="sambanova/Llama-4-Maverick-17B-128E-Instruct", messages=[ { "role": "user", - "content": "What do you know about sambanova.ai. Give your response in json format", + "content": "What do you know about SambaNova Systems", } ], stream=True, max_tokens=10, response_format={ "type": "json_object" }, - stop=["\n\n"], + stop=[], temperature=0.2, top_p=0.9, tool_choice="auto", @@ -139,3 +136,174 @@ Here's how to call a Sambanova model with the LiteLLM Proxy Server + +## SambaNova - Tool Calling + +```python +import litellm + +# Example dummy function + +def get_current_weather(location, unit="fahrenheit"): + if unit == "fahrenheit" + return{"location": location, "temperature": "72", "unit": "fahrenheit"} + else: + return{"location": location, "temperature": "22", "unit": "celsius"} + +messages = [{"role": "user", "content": "What's the weather like in San Francisco"}] + +tools = [ + { + "type": "function", + "function": { + "name": "import litellm", + "description": "Get the current weather in a given location", + "parameters": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "The city and state, e.g. San Francisco, CA", + }, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, + }, + "required": ["location"], + }, + }, + } +] + +response = litellm.completion( + model="sambanova/Meta-Llama-3.3-70B-Instruct", + messages=messages, + tools=tools, + tool_choice="auto", # auto is default, but we'll be explicit +) + +print("\nFirst LLM Response:\n", response) +response_message = response.choices[0].message +tool_calls = response_message.tool_calls + +if tool_calls: + # Step 2: check if the model wanted to call a function +if tool_calls: + # Step 3: call the function + # Note: the JSON response may not always be valid; be sure to handle errors + available_functions = { + "get_current_weather": get_current_weather, + } + messages.append( + response_message + ) # extend conversation with assistant's reply + print("Response message\n", response_message) + # Step 4: send the info for each function call and function response to the model + for tool_call in tool_calls: + function_name = tool_call.function.name + function_to_call = available_functions[function_name] + function_args = json.loads(tool_call.function.arguments) + function_response = function_to_call( + location=function_args.get("location"), + unit=function_args.get("unit"), + ) + messages.append( + { + "tool_call_id": tool_call.id, + "role": "tool", + "name": function_name, + "content": function_response, + } + ) # extend conversation with function response + print(f"messages: {messages}") + second_response = litellm.completion( + model="sambanova/Meta-Llama-3.3-70B-Instruct", messages=messages + ) # get a new response from the model where it can see the function response + print("second response\n", second_response) +``` + +## SambaNova - Vision Example + +```python +import litellm + +# Auxiliary function to get b64 images +def data_url_from_image(file_path): + mime_type, _ = mimetypes.guess_type(file_path) + if mime_type is None: + raise ValueError("Could not determine MIME type of the file") + + with open(file_path, "rb") as image_file: + encoded_string = base64.b64encode(image_file.read()).decode("utf-8") + + data_url = f"data:{mime_type};base64,{encoded_string}" + return data_url + +response = litellm.completion( + model = "sambanova/Llama-4-Maverick-17B-128E-Instruct", + messages=[ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "What's in this image?" + }, + { + "type": "image_url", + "image_url": { + "url": data_url_from_image("your_image_path"), + "format": "image/jpeg" + } + } + ] + } + ], + stream=False +) + +print(response.choices[0].message.content) +``` + + +## SambaNova - Structured Output + +```python +import litellm + +response = litellm.completion( + model="sambanova/Meta-Llama-3.3-70B-Instruct", + messages=[ + { + "role": "system", + "content": "You are an expert at structured data extraction. You will be given unstructured text should convert it into the given structure." + }, + { + "role": "user", + "content": "the section 24 has appliances, and videogames" + }, + ], + response_format={ + "type": "json_schema", + "json_schema": { + "title": "data", + "name": "data_extraction", + "schema": { + "type": "object", + "properties": { + "section": { + "type": "string" }, + "products": { + "type": "array", + "items": { "type": "string" } + } + }, + "required": ["section", "products"], + "additionalProperties": False + }, + "strict": False + } + }, + stream=False +) + +print(response.choices[0].message.content)) +``` diff --git a/docs/my-website/docs/providers/vertex.md b/docs/my-website/docs/providers/vertex.md index b328c805770..30887e9f60d 100644 --- a/docs/my-website/docs/providers/vertex.md +++ b/docs/my-website/docs/providers/vertex.md @@ -1284,11 +1284,18 @@ ModelResponse( -## Llama 3 API +## Meta/Llama API | Model Name | Function Call | |------------------|--------------------------------------| +| meta/llama-3.2-90b-vision-instruct-maas | `completion('vertex_ai/meta/llama-3.2-90b-vision-instruct-maas', messages)` | +| meta/llama3-8b-instruct-maas | `completion('vertex_ai/meta/llama3-8b-instruct-maas', messages)` | +| meta/llama3-70b-instruct-maas | `completion('vertex_ai/meta/llama3-70b-instruct-maas', messages)` | | meta/llama3-405b-instruct-maas | `completion('vertex_ai/meta/llama3-405b-instruct-maas', messages)` | +| meta/llama-4-scout-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas', messages)` | +| meta/llama-4-scout-17-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas', messages)` | +| meta/llama-4-maverick-17b-128e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas',messages)` | +| meta/llama-4-maverick-17b-16e-instruct-maas | `completion('vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas',messages)` | ### Usage diff --git a/docs/my-website/docs/proxy/caching.md b/docs/my-website/docs/proxy/caching.md index b60b9966ba2..84e8c5f8d58 100644 --- a/docs/my-website/docs/proxy/caching.md +++ b/docs/my-website/docs/proxy/caching.md @@ -16,6 +16,7 @@ Cache LLM Responses. LiteLLM's caching system stores and reuses LLM responses to ### Supported Caches - In Memory Cache +- Disk Cache - Redis Cache - Qdrant Semantic Cache - Redis Semantic Cache @@ -338,7 +339,7 @@ model_list: litellm_settings: set_verbose: True - cache: True # set cache responses to True, litellm defaults to using a redis cache + cache: True # set cache responses to True cache_params: type: "redis-semantic" similarity_threshold: 0.8 # similarity threshold for semantic cache @@ -369,6 +370,40 @@ $ litellm --config /path/to/config.yaml + + +#### Step 1: Add `cache` to the config.yaml +```yaml +litellm_settings: + cache: True + cache_params: + type: local +``` + +#### Step 2: Run proxy with config +```shell +$ litellm --config /path/to/config.yaml +``` + + + + + +#### Step 1: Add `cache` to the config.yaml +```yaml +litellm_settings: + cache: True + cache_params: + type: disk + disk_cache_dir: /tmp/litellm-cache # OPTIONAL, default to ./.litellm_cache +``` + +#### Step 2: Run proxy with config +```shell +$ litellm --config /path/to/config.yaml +``` + + @@ -932,4 +967,4 @@ general_settings: user_api_key_cache_ttl: #time in seconds ``` -By default this value is set to 60s. \ No newline at end of file +By default this value is set to 60s. diff --git a/docs/my-website/docs/proxy/call_hooks.md b/docs/my-website/docs/proxy/call_hooks.md index a7b0afcc18b..c588ca0d0e6 100644 --- a/docs/my-website/docs/proxy/call_hooks.md +++ b/docs/my-website/docs/proxy/call_hooks.md @@ -44,7 +44,8 @@ class MyCustomHandler(CustomLogger): # https://docs.litellm.ai/docs/observabilit self, request_data: dict, original_exception: Exception, - user_api_key_dict: UserAPIKeyAuth + user_api_key_dict: UserAPIKeyAuth, + traceback_str: Optional[str] = None, ): pass diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md index 01ff24da694..fdd68c953f6 100644 --- a/docs/my-website/docs/proxy/config_settings.md +++ b/docs/my-website/docs/proxy/config_settings.md @@ -1,6 +1,5 @@ # All settings - ```yaml environment_variables: {} @@ -95,6 +94,8 @@ general_settings: allowed_routes: ["route1", "route2"] # list of allowed proxy API routes - a user can access. (currently JWT-Auth only) key_management_system: google_kms # either google_kms or azure_kms master_key: string + maximum_spend_logs_retention_period: 30d # The maximum time to retain spend logs before deletion. + maximum_spend_logs_retention_interval: 1d # interval in which the spend log cleanup task should run in. # Database Settings database_url: string @@ -211,7 +212,8 @@ general_settings: | enable_oauth2_proxy_auth | boolean | (Enterprise Feature) If true, enables oauth2.0 authentication | | forward_openai_org_id | boolean | If true, forwards the OpenAI Organization ID to the backend LLM call (if it's OpenAI). | | forward_client_headers_to_llm_api | boolean | If true, forwards the client headers (any `x-` headers) to the backend LLM call | - +| maximum_spend_logs_retention_period | str | Used to set the max retention time for spend logs in the db, after which they will be auto-purged | +| maximum_spend_logs_retention_interval | str | Used to set the interval in which the spend log cleanup task should run in. | ### router_settings - Reference :::info @@ -331,14 +333,19 @@ router_settings: | AZURE_PASSWORD | Password for Azure services, use in conjunction with AZURE_USERNAME for azure ad token with basic username/password workflow | AZURE_FEDERATED_TOKEN_FILE | File path to Azure federated token | AZURE_KEY_VAULT_URI | URI for Azure Key Vault +| AZURE_OPERATION_POLLING_TIMEOUT | Timeout in seconds for Azure operation polling | AZURE_STORAGE_ACCOUNT_KEY | The Azure Storage Account Key to use for Authentication to Azure Blob Storage logging | AZURE_STORAGE_ACCOUNT_NAME | Name of the Azure Storage Account to use for logging to Azure Blob Storage | AZURE_STORAGE_FILE_SYSTEM | Name of the Azure Storage File System to use for logging to Azure Blob Storage. (Typically the Container name) | AZURE_STORAGE_TENANT_ID | The Application Tenant ID to use for Authentication to Azure Blob Storage logging | AZURE_STORAGE_CLIENT_ID | The Application Client ID to use for Authentication to Azure Blob Storage logging | AZURE_STORAGE_CLIENT_SECRET | The Application Client Secret to use for Authentication to Azure Blob Storage logging +| BATCH_STATUS_POLL_INTERVAL_SECONDS | Interval in seconds for polling batch status. Default is 3600 (1 hour) +| BATCH_STATUS_POLL_MAX_ATTEMPTS | Maximum number of attempts for polling batch status. Default is 24 (for 24 hours) +| BEDROCK_MAX_POLICY_SIZE | Maximum size for Bedrock policy. Default is 75 | BERRISPEND_ACCOUNT_ID | Account ID for BerriSpend service | BRAINTRUST_API_KEY | API key for Braintrust integration +| CACHED_STREAMING_CHUNK_DELAY | Delay in seconds for cached streaming chunks. Default is 0.02 | CIRCLE_OIDC_TOKEN | OpenID Connect token for CircleCI | CIRCLE_OIDC_TOKEN_V2 | Version 2 of the OpenID Connect token for CircleCI | CONFIG_FILE_PATH | File path for configuration file @@ -352,6 +359,9 @@ router_settings: | DATABASE_USER | Username for database connection | DATABASE_USERNAME | Alias for database user | DATABRICKS_API_BASE | Base URL for Databricks API +| DAYS_IN_A_MONTH | Days in a month for calculation purposes. Default is 28 +| DAYS_IN_A_WEEK | Days in a week for calculation purposes. Default is 7 +| DAYS_IN_A_YEAR | Days in a year for calculation purposes. Default is 365 | DD_BASE_URL | Base URL for Datadog integration | DATADOG_BASE_URL | (Alternative to DD_BASE_URL) Base URL for Datadog integration | _DATADOG_BASE_URL | (Alternative to DD_BASE_URL) Base URL for Datadog integration @@ -362,6 +372,39 @@ router_settings: | DD_SERVICE | Service identifier for Datadog logs. Defaults to "litellm-server" | DD_VERSION | Version identifier for Datadog logs. Defaults to "unknown" | DEBUG_OTEL | Enable debug mode for OpenTelemetry +| DEFAULT_ALLOWED_FAILS | Maximum failures allowed before cooling down a model. Default is 3 +| DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS | Default maximum tokens for Anthropic chat completions. Default is 4096 +| DEFAULT_BATCH_SIZE | Default batch size for operations. Default is 512 +| DEFAULT_COOLDOWN_TIME_SECONDS | Duration in seconds to cooldown a model after failures. Default is 5 +| DEFAULT_CRON_JOB_LOCK_TTL_SECONDS | Time-to-live for cron job locks in seconds. Default is 60 (1 minute) +| DEFAULT_FAILURE_THRESHOLD_PERCENT | Threshold percentage of failures to cool down a deployment. Default is 0.5 (50%) +| DEFAULT_FLUSH_INTERVAL_SECONDS | Default interval in seconds for flushing operations. Default is 5 +| DEFAULT_HEALTH_CHECK_INTERVAL | Default interval in seconds for health checks. Default is 300 (5 minutes) +| DEFAULT_IMAGE_HEIGHT | Default height for images. Default is 300 +| DEFAULT_IMAGE_TOKEN_COUNT | Default token count for images. Default is 250 +| DEFAULT_IMAGE_WIDTH | Default width for images. Default is 300 +| DEFAULT_IN_MEMORY_TTL | Default time-to-live for in-memory cache in seconds. Default is 5 +| DEFAULT_MAX_LRU_CACHE_SIZE | Default maximum size for LRU cache. Default is 16 +| DEFAULT_MAX_RECURSE_DEPTH | Default maximum recursion depth. Default is 100 +| DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER | Default maximum recursion depth for sensitive data masker. Default is 10 +| DEFAULT_MAX_RETRIES | Default maximum retry attempts. Default is 2 +| DEFAULT_MAX_TOKENS | Default maximum tokens for LLM calls. Default is 4096 +| DEFAULT_MAX_TOKENS_FOR_TRITON | Default maximum tokens for Triton models. Default is 2000 +| DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT | Default token count for mock response completions. Default is 20 +| DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT | Default token count for mock response prompts. Default is 10 +| DEFAULT_MODEL_CREATED_AT_TIME | Default creation timestamp for models. Default is 1677610602 +| DEFAULT_PROMPT_INJECTION_SIMILARITY_THRESHOLD | Default threshold for prompt injection similarity. Default is 0.7 +| DEFAULT_POLLING_INTERVAL | Default polling interval for schedulers in seconds. Default is 0.03 +| DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET | Default high reasoning effort thinking budget. Default is 4096 +| DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET | Default low reasoning effort thinking budget. Default is 1024 +| DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET | Default medium reasoning effort thinking budget. Default is 2048 +| DEFAULT_REDIS_SYNC_INTERVAL | Default Redis synchronization interval in seconds. Default is 1 +| DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND | Default price per second for Replicate GPU. Default is 0.001400 +| DEFAULT_REPLICATE_POLLING_DELAY_SECONDS | Default delay in seconds for Replicate polling. Default is 1 +| DEFAULT_REPLICATE_POLLING_RETRIES | Default number of retries for Replicate polling. Default is 5 +| DEFAULT_SLACK_ALERTING_THRESHOLD | Default threshold for Slack alerting. Default is 300 +| DEFAULT_SOFT_BUDGET | Default soft budget for LiteLLM proxy keys. Default is 50.0 +| DEFAULT_TRIM_RATIO | Default ratio of tokens to trim from prompt end. Default is 0.75 | DIRECT_URL | Direct URL for service endpoint | DISABLE_ADMIN_UI | Toggle to disable the admin UI | DISABLE_SCHEMA_UPDATE | Toggle to disable schema updates @@ -369,8 +412,19 @@ router_settings: | DOCS_FILTERED | Flag indicating filtered documentation | DOCS_TITLE | Title of the documentation pages | DOCS_URL | The path to the Swagger API documentation. **By default this is "/"** +| EMAIL_LOGO_URL | URL for the logo used in emails | EMAIL_SUPPORT_CONTACT | Support contact email address | EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING | Flag to enable new multi-instance rate limiting. **Default is False** +| FIREWORKS_AI_4_B | Size parameter for Fireworks AI 4B model. Default is 4 +| FIREWORKS_AI_16_B | Size parameter for Fireworks AI 16B model. Default is 16 +| FIREWORKS_AI_56_B_MOE | Size parameter for Fireworks AI 56B MOE model. Default is 56 +| FIREWORKS_AI_80_B | Size parameter for Fireworks AI 80B model. Default is 80 +| FIREWORKS_AI_176_B_MOE | Size parameter for Fireworks AI 176B MOE model. Default is 176 +| FUNCTION_DEFINITION_TOKEN_COUNT | Token count for function definitions. Default is 9 +| GALILEO_BASE_URL | Base URL for Galileo platform +| GALILEO_PASSWORD | Password for Galileo authentication +| GALILEO_PROJECT_ID | Project ID for Galileo usage +| GALILEO_USERNAME | Username for Galileo authentication | GCS_BUCKET_NAME | Name of the Google Cloud Storage bucket | GCS_PATH_SERVICE_ACCOUNT | Path to the Google Cloud service account JSON file | GCS_FLUSH_INTERVAL | Flush interval for GCS logging (in seconds). Specify how often you want a log to be sent to GCS. **Default is 20 seconds** @@ -402,6 +456,7 @@ router_settings: | GOOGLE_CLIENT_ID | Client ID for Google OAuth | GOOGLE_CLIENT_SECRET | Client secret for Google OAuth | GOOGLE_KMS_RESOURCE_NAME | Name of the resource in Google KMS +| HEALTH_CHECK_TIMEOUT_SECONDS | Timeout in seconds for health checks. Default is 60 | HF_API_BASE | Base URL for Hugging Face API | HCP_VAULT_ADDR | Address for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault) | HCP_VAULT_CLIENT_CERT | Path to client certificate for [Hashicorp Vault Secret Manager](../secret.md#hashicorp-vault) @@ -411,9 +466,13 @@ router_settings: | HCP_VAULT_CERT_ROLE | Role for [Hashicorp Vault Secret Manager Auth](../secret.md#hashicorp-vault) | HELICONE_API_KEY | API key for Helicone service | HOSTNAME | Hostname for the server, this will be [emitted to `datadog` logs](https://docs.litellm.ai/docs/proxy/logging#datadog) +| HOURS_IN_A_DAY | Hours in a day for calculation purposes. Default is 24 | HUGGINGFACE_API_BASE | Base URL for Hugging Face API | HUGGINGFACE_API_KEY | API key for Hugging Face API +| HUMANLOOP_PROMPT_CACHE_TTL_SECONDS | Time-to-live in seconds for cached prompts in Humanloop. Default is 60 | IAM_TOKEN_DB_AUTH | IAM token for database authentication +| INITIAL_RETRY_DELAY | Initial delay in seconds for retrying requests. Default is 0.5 +| JITTER | Jitter factor for retry delay calculations. Default is 0.75 | JSON_LOGS | Enable JSON formatted logging | JWT_AUDIENCE | Expected audience for JWT tokens | JWT_PUBLIC_KEY_URL | URL to fetch public key for JWT verification @@ -434,6 +493,7 @@ router_settings: | LANGSMITH_PROJECT | Project name for Langsmith integration | LANGSMITH_SAMPLING_RATE | Sampling rate for Langsmith logging | LANGTRACE_API_KEY | API key for Langtrace service +| LENGTH_OF_LITELLM_GENERATED_KEY | Length of keys generated by LiteLLM. Default is 16 | LITERAL_API_KEY | API key for Literal integration | LITERAL_API_URL | API URL for Literal service | LITERAL_BATCH_SIZE | Batch size for Literal operations @@ -454,6 +514,22 @@ router_settings: | LITELLM_TOKEN | Access token for LiteLLM integration | LITELLM_PRINT_STANDARD_LOGGING_PAYLOAD | If true, prints the standard logging payload to the console - useful for debugging | LOGFIRE_TOKEN | Token for Logfire logging service +| MAX_EXCEPTION_MESSAGE_LENGTH | Maximum length for exception messages. Default is 2000 +| MAX_IN_MEMORY_QUEUE_FLUSH_COUNT | Maximum count for in-memory queue flush operations. Default is 1000 +| MAX_LONG_SIDE_FOR_IMAGE_HIGH_RES | Maximum length for the long side of high-resolution images. Default is 2000 +| MAX_REDIS_BUFFER_DEQUEUE_COUNT | Maximum count for Redis buffer dequeue operations. Default is 100 +| MAX_SHORT_SIDE_FOR_IMAGE_HIGH_RES | Maximum length for the short side of high-resolution images. Default is 768 +| MAX_SIZE_IN_MEMORY_QUEUE | Maximum size for in-memory queue. Default is 10000 +| MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB | Maximum size in KB for each item in memory cache. Default is 512 or 1024 +| MAX_SPENDLOG_ROWS_TO_QUERY | Maximum number of spend log rows to query. Default is 1,000,000 +| MAX_TEAM_LIST_LIMIT | Maximum number of teams to list. Default is 20 +| MAX_TILE_HEIGHT | Maximum height for image tiles. Default is 512 +| MAX_TILE_WIDTH | Maximum width for image tiles. Default is 512 +| MAX_TOKEN_TRIMMING_ATTEMPTS | Maximum number of attempts to trim a token message. Default is 10 +| MAXIMUM_TRACEBACK_LINES_TO_LOG | Maximum number of lines to log in traceback in LiteLLM Logs UI. Default is 100 +| MAX_RETRY_DELAY | Maximum delay in seconds for retrying requests. Default is 8.0 +| MIN_NON_ZERO_TEMPERATURE | Minimum non-zero temperature value. Default is 0.0001 +| MINIMUM_PROMPT_CACHE_TOKEN_COUNT | Minimum token count for caching a prompt. Default is 1024 | MISTRAL_API_BASE | Base URL for Mistral API | MISTRAL_API_KEY | API key for Mistral API | MICROSOFT_CLIENT_ID | Client ID for Microsoft services @@ -462,10 +538,12 @@ router_settings: | MICROSOFT_SERVICE_PRINCIPAL_ID | Service Principal ID for Microsoft Enterprise Application. (This is an advanced feature if you want litellm to auto-assign members to Litellm Teams based on their Microsoft Entra ID Groups) | NO_DOCS | Flag to disable documentation generation | NO_PROXY | List of addresses to bypass proxy +| NON_LLM_CONNECTION_TIMEOUT | Timeout in seconds for non-LLM service connections. Default is 15 | OAUTH_TOKEN_INFO_ENDPOINT | Endpoint for OAuth token info retrieval | OPENAI_BASE_URL | Base URL for OpenAI API | OPENAI_API_BASE | Base URL for OpenAI API | OPENAI_API_KEY | API key for OpenAI services +| OPENAI_FILE_SEARCH_COST_PER_1K_CALLS | Cost per 1000 calls for OpenAI file search. Default is 0.0025 | OPENAI_ORGANIZATION | Organization identifier for OpenAI | OPENID_BASE_URL | Base URL for OpenID Connect services | OPENID_CLIENT_ID | Client ID for OpenID Connect authentication @@ -474,9 +552,12 @@ router_settings: | OPENMETER_API_KEY | API key for OpenMeter services | OPENMETER_EVENT_TYPE | Type of events sent to OpenMeter | OTEL_ENDPOINT | OpenTelemetry endpoint for traces +| OTEL_EXPORTER_OTLP_ENDPOINT | OpenTelemetry endpoint for traces | OTEL_ENVIRONMENT_NAME | Environment name for OpenTelemetry | OTEL_EXPORTER | Exporter type for OpenTelemetry +| OTEL_EXPORTER_OTLP_PROTOCOL | Exporter type for OpenTelemetry | OTEL_HEADERS | Headers for OpenTelemetry requests +| OTEL_EXPORTER_OTLP_HEADERS | Headers for OpenTelemetry requests | OTEL_SERVICE_NAME | Service name identifier for OpenTelemetry | OTEL_TRACER_NAME | Tracer name for OpenTelemetry tracing | PAGERDUTY_API_KEY | API key for PagerDuty Alerting @@ -487,21 +568,37 @@ router_settings: | PREDIBASE_API_BASE | Base URL for Predibase API | PRESIDIO_ANALYZER_API_BASE | Base URL for Presidio Analyzer service | PRESIDIO_ANONYMIZER_API_BASE | Base URL for Presidio Anonymizer service +| PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES | Refresh interval in minutes for Prometheus budget metrics. Default is 5 +| PROMETHEUS_FALLBACK_STATS_SEND_TIME_HOURS | Fallback time in hours for sending stats to Prometheus. Default is 9 | PROMETHEUS_URL | URL for Prometheus service | PROMPTLAYER_API_KEY | API key for PromptLayer integration | PROXY_ADMIN_ID | Admin identifier for proxy server | PROXY_BASE_URL | Base URL for proxy service +| PROXY_BATCH_WRITE_AT | Time in seconds to wait before batch writing spend logs to the database. Default is 10 +| PROXY_BUDGET_RESCHEDULER_MAX_TIME | Maximum time in seconds to wait before checking database for budget resets. Default is 605 +| PROXY_BUDGET_RESCHEDULER_MIN_TIME | Minimum time in seconds to wait before checking database for budget resets. Default is 597 | PROXY_LOGOUT_URL | URL for logging out of the proxy service | LITELLM_MASTER_KEY | Master key for proxy authentication | QDRANT_API_BASE | Base URL for Qdrant API | QDRANT_API_KEY | API key for Qdrant service +| QDRANT_SCALAR_QUANTILE | Scalar quantile for Qdrant operations. Default is 0.99 | QDRANT_URL | Connection URL for Qdrant database +| QDRANT_VECTOR_SIZE | Vector size for Qdrant operations. Default is 1536 +| REDIS_CONNECTION_POOL_TIMEOUT | Timeout in seconds for Redis connection pool. Default is 5 | REDIS_HOST | Hostname for Redis server | REDIS_PASSWORD | Password for Redis service | REDIS_PORT | Port number for Redis server +| REDIS_SOCKET_TIMEOUT | Timeout in seconds for Redis socket operations. Default is 0.1 | REDOC_URL | The path to the Redoc Fast API documentation. **By default this is "/redoc"** +| REPEATED_STREAMING_CHUNK_LIMIT | Limit for repeated streaming chunks to detect looping. Default is 100 +| REPLICATE_MODEL_NAME_WITH_ID_LENGTH | Length of Replicate model names with ID. Default is 64 +| REPLICATE_POLLING_DELAY_SECONDS | Delay in seconds for Replicate polling operations. Default is 0.5 +| REQUEST_TIMEOUT | Timeout in seconds for requests. Default is 6000 +| ROUTER_MAX_FALLBACKS | Maximum number of fallbacks for router. Default is 5 +| SECRET_MANAGER_REFRESH_INTERVAL | Refresh interval in seconds for secret manager. Default is 86400 (24 hours) | SERVER_ROOT_PATH | Root path for the server application | SET_VERBOSE | Flag to enable verbose logging +| SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD | Minimum number of requests to consider "reasonable traffic" for single-deployment cooldown logic. Default is 1000 | SLACK_DAILY_REPORT_FREQUENCY | Frequency of daily Slack reports (e.g., daily, weekly) | SLACK_WEBHOOK_URL | Webhook URL for Slack integration | SMTP_HOST | Hostname for the SMTP server @@ -518,7 +615,17 @@ router_settings: | SUPABASE_KEY | API key for Supabase service | SUPABASE_URL | Base URL for Supabase instance | STORE_MODEL_IN_DB | If true, enables storing model + credential information in the DB. +| SYSTEM_MESSAGE_TOKEN_COUNT | Token count for system messages. Default is 4 | TEST_EMAIL_ADDRESS | Email address used for testing purposes +| TOGETHER_AI_4_B | Size parameter for Together AI 4B model. Default is 4 +| TOGETHER_AI_8_B | Size parameter for Together AI 8B model. Default is 8 +| TOGETHER_AI_21_B | Size parameter for Together AI 21B model. Default is 21 +| TOGETHER_AI_41_B | Size parameter for Together AI 41B model. Default is 41 +| TOGETHER_AI_80_B | Size parameter for Together AI 80B model. Default is 80 +| TOGETHER_AI_110_B | Size parameter for Together AI 110B model. Default is 110 +| TOGETHER_AI_EMBEDDING_150_M | Size parameter for Together AI 150M embedding model. Default is 150 +| TOGETHER_AI_EMBEDDING_350_M | Size parameter for Together AI 350M embedding model. Default is 350 +| TOOL_CHOICE_OBJECT_TOKEN_COUNT | Token count for tool choice objects. Default is 4 | UI_LOGO_PATH | Path to the logo image used in the UI | UI_PASSWORD | Password for accessing the UI | UI_USERNAME | Username for accessing the UI @@ -530,3 +637,4 @@ router_settings: | USE_AWS_KMS | Flag to enable AWS Key Management Service for encryption | USE_PRISMA_MIGRATE | Flag to use prisma migrate instead of prisma db push. Recommended for production environments. | WEBHOOK_URL | URL for receiving webhooks from external services +| SPEND_LOG_RUN_LOOPS | Constant for setting how many runs of 1000 batch deletes should spend_log_cleanup task run \ No newline at end of file diff --git a/docs/my-website/docs/proxy/email.md b/docs/my-website/docs/proxy/email.md index a3f3a41694e..4eb35367dbe 100644 --- a/docs/my-website/docs/proxy/email.md +++ b/docs/my-website/docs/proxy/email.md @@ -1,35 +1,130 @@ import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; # Email Notifications -Send an Email to your users when: -- A Proxy API Key is created for them -- Their API Key crosses it's Budget -- All Team members of a LiteLLM Team -> when the team crosses it's budget + +

+ LiteLLM Email Notifications +

- +## Overview -## Quick Start +Send LiteLLM Proxy users emails for specific events. + +| Category | Details | +|----------|---------| +| Supported Events | • User added as a user on LiteLLM Proxy
• Proxy API Key created for user | +| Supported Email Integrations | • Resend API
• SMTP | + +## Usage + +:::info + +LiteLLM Cloud: This feature is enabled for all LiteLLM Cloud users, there's no need to configure anything. + +::: + +### 1. Configure email integration + + + Get SMTP credentials to set this up + +```yaml showLineNumbers title="proxy_config.yaml" +litellm_settings: + callbacks: ["smtp_email"] +``` + Add the following to your proxy env -```shell +```shell showLineNumbers SMTP_HOST="smtp.resend.com" +SMTP_TLS="True" +SMTP_PORT="587" SMTP_USERNAME="resend" -SMTP_PASSWORD="*******" -SMTP_SENDER_EMAIL="support@alerts.litellm.ai" # email to send alerts from: `support@alerts.litellm.ai` +SMTP_SENDER_EMAIL="notifications@alerts.litellm.ai" +SMTP_PASSWORD="xxxxx" ``` -Add `email` to your proxy config.yaml under `general_settings` + + -```yaml -general_settings: - master_key: sk-1234 - alerting: ["email"] +Add `resend_email` to your proxy config.yaml under `litellm_settings` + +set the following env variables + +```shell showLineNumbers +RESEND_API_KEY="re_1234" ``` -That's it ! start your proxy +```yaml showLineNumbers title="proxy_config.yaml" +litellm_settings: + callbacks: ["resend_email"] +``` + + + + +### 2. Create a new user + +On the LiteLLM Proxy UI, go to users > create a new user. + +After creating a new user, they will receive an email invite a the email you specified when creating the user. + +## Email Templates + + +### 1. User added as a user on LiteLLM Proxy + +This email is send when you create a new user on LiteLLM Proxy. + + + +**How to trigger this event** + +On the LiteLLM Proxy UI, go to Users > Create User > Enter the user's email address > Create User. + + + +### 2. Proxy API Key created for user + +This email is sent when you create a new API key for a user on LiteLLM Proxy. + + + +**How to trigger this event** + +On the LiteLLM Proxy UI, go to Virtual Keys > Create API Key > Select User ID + + + +On the Create Key Modal, Select Advanced Settings > Set Send Email to True. + + + + + ## Customizing Email Branding diff --git a/docs/my-website/docs/proxy/guardrails/bedrock.md b/docs/my-website/docs/proxy/guardrails/bedrock.md index 0da2238bcf0..a0c43d47dec 100644 --- a/docs/my-website/docs/proxy/guardrails/bedrock.md +++ b/docs/my-website/docs/proxy/guardrails/bedrock.md @@ -2,7 +2,7 @@ import Image from '@theme/IdealImage'; import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; -# Bedrock +# Bedrock Guardrails LiteLLM supports Bedrock guardrails via the [Bedrock ApplyGuardrail API](https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_ApplyGuardrail.html). @@ -135,3 +135,48 @@ curl -i http://localhost:4000/v1/chat/completions \ +## PII Masking with Bedrock Guardrails + +Bedrock guardrails support PII detection and masking capabilities. To enable this feature, you need to: + +1. Set `mode` to `pre_call` to run the guardrail check before the LLM call +2. Enable masking by setting `mask_request_content` and/or `mask_response_content` to `true` + +Here's how to configure it in your config.yaml: + +```yaml showLineNumbers title="litellm proxy config.yaml" +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +guardrails: + - guardrail_name: "bedrock-pre-guard" + litellm_params: + guardrail: bedrock + mode: "pre_call" # Important: must use pre_call mode for masking + guardrailIdentifier: wf0hkdb5x07f + guardrailVersion: "DRAFT" + mask_request_content: true # Enable masking in user requests + mask_response_content: true # Enable masking in model responses +``` + +With this configuration, when the bedrock guardrail intervenes, litellm will read the masked output from the guardrail and send it to the model. + +### Example Usage + +When enabled, PII will be automatically masked in the text. For example, if a user sends: + +``` +My email is john.doe@example.com and my phone number is 555-123-4567 +``` + +The text sent to the model might be masked as: + +``` +My email is [EMAIL] and my phone number is [PHONE_NUMBER] +``` + +This helps protect sensitive information while still allowing the model to understand the context of the request. + diff --git a/docs/my-website/docs/proxy/guardrails/lakera_ai.md b/docs/my-website/docs/proxy/guardrails/lakera_ai.md index ba1ca0b2183..e66329dcb0c 100644 --- a/docs/my-website/docs/proxy/guardrails/lakera_ai.md +++ b/docs/my-website/docs/proxy/guardrails/lakera_ai.md @@ -8,7 +8,8 @@ import TabItem from '@theme/TabItem'; ### 1. Define Guardrails on your LiteLLM config.yaml Define your guardrails under the `guardrails` section -```yaml + +```yaml showLineNumbers title="litellm config.yaml" model_list: - model_name: gpt-3.5-turbo litellm_params: @@ -18,13 +19,13 @@ model_list: guardrails: - guardrail_name: "lakera-guard" litellm_params: - guardrail: lakera # supported values: "aporia", "bedrock", "lakera" + guardrail: lakera_v2 # supported values: "aporia", "bedrock", "lakera" mode: "during_call" api_key: os.environ/LAKERA_API_KEY api_base: os.environ/LAKERA_API_BASE - guardrail_name: "lakera-pre-guard" litellm_params: - guardrail: lakera # supported values: "aporia", "bedrock", "lakera" + guardrail: lakera_v2 # supported values: "aporia", "bedrock", "lakera" mode: "pre_call" api_key: os.environ/LAKERA_API_KEY api_base: os.environ/LAKERA_API_BASE @@ -53,7 +54,7 @@ litellm --config config.yaml --detailed_debug Expect this to fail since since `ishaan@berri.ai` in the request is PII -```shell +```shell showLineNumbers title="Curl Request" curl -i http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ @@ -108,7 +109,7 @@ Expected response on failure -```shell +```shell showLineNumbers title="Curl Request" curl -i http://localhost:4000/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ @@ -125,31 +126,3 @@ curl -i http://localhost:4000/v1/chat/completions \ - -## Advanced -### Set category-based thresholds. - -Lakera has 2 categories for prompt_injection attacks: -- jailbreak -- prompt_injection - -```yaml -model_list: - - model_name: fake-openai-endpoint - litellm_params: - model: openai/fake - api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ - -guardrails: - - guardrail_name: "lakera-guard" - litellm_params: - guardrail: lakera # supported values: "aporia", "bedrock", "lakera" - mode: "during_call" - api_key: os.environ/LAKERA_API_KEY - api_base: os.environ/LAKERA_API_BASE - category_thresholds: - prompt_injection: 0.1 - jailbreak: 0.1 - -``` \ No newline at end of file diff --git a/docs/my-website/docs/proxy/guardrails/pii_masking_v2.md b/docs/my-website/docs/proxy/guardrails/pii_masking_v2.md index 59690666ee4..427308cf221 100644 --- a/docs/my-website/docs/proxy/guardrails/pii_masking_v2.md +++ b/docs/my-website/docs/proxy/guardrails/pii_masking_v2.md @@ -2,16 +2,60 @@ import Image from '@theme/IdealImage'; import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; -# PII Masking - Presidio +# PII, PHI Masking - Presidio + +## Overview + +| Property | Details | +|-------|-------| +| Description | Use this guardrail to mask PII (Personally Identifiable Information), PHI (Protected Health Information), and other sensitive data. | +| Provider | [Microsoft Presidio](https://github.com/microsoft/presidio/) | +| Supported Entity Types | All Presidio Entity Types | +| Supported Actions | `MASK`, `BLOCK` | +| Supported Modes | `pre_call`, `during_call`, `post_call`, `logging_only` | + +## Deployment options + +For this guardrail you need a deployed Presidio Analyzer and Presido Anonymizer containers. + +| Deployment Option | Details | +|------------------|----------| +| Deploy Presidio Docker Containers | - [Presidio Analyzer Docker Container](https://hub.docker.com/r/microsoft/presidio-analyzer)
- [Presidio Anonymizer Docker Container](https://hub.docker.com/r/microsoft/presidio-anonymizer) | ## Quick Start -LiteLLM supports [Microsoft Presidio](https://github.com/microsoft/presidio/) for PII masking. + + -### 1. Define Guardrails on your LiteLLM config.yaml +### 1. Create a PII, PHI Masking Guardrail + +On the LiteLLM UI, navigate to Guardrails. Click "Add Guardrail". On this dropdown select "Presidio PII" and enter your presidio analyzer and anonymizer endpoints. + + + +
+
+ +#### 1.2 Configure Entity Types + +Now select the entity types you want to mask. See the [supported actions here](#supported-actions) + + + +
+ + + Define your guardrails under the `guardrails` section -```yaml + +```yaml title="config.yaml" showLineNumbers model_list: - model_name: gpt-3.5-turbo litellm_params: @@ -19,7 +63,7 @@ model_list: api_key: os.environ/OPENAI_API_KEY guardrails: - - guardrail_name: "presidio-pre-guard" + - guardrail_name: "presidio-pii" litellm_params: guardrail: presidio # supported values: "aporia", "bedrock", "lakera", "presidio" mode: "pre_call" @@ -27,7 +71,7 @@ guardrails: Set the following env vars -```bash +```bash title="Setup Environment Variables" showLineNumbers export PRESIDIO_ANALYZER_API_BASE="http://localhost:5002" export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001" ``` @@ -38,15 +82,36 @@ export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001" - `post_call` Run **after** LLM call, on **input & output** - `logging_only` Run **after** LLM call, only apply PII Masking before logging to Langfuse, etc. Not on the actual llm api request / response. - ### 2. Start LiteLLM Gateway - -```shell +```shell title="Start Gateway" showLineNumbers litellm --config config.yaml --detailed_debug ``` -### 3. Test request + +
+ + +### 3. Test it! + +#### 3.1 LiteLLM UI + +On the litellm UI, navigate to the 'Test Keys' page, select the guardrail you created and send the following messaged filled with PII data. + +```text title="PII Request" showLineNumbers +My credit card is 4111-1111-1111-1111 and my email is test@example.com. +``` + + + +
+ +#### 3.2 Test in code + +In order to apply a guardrail for a request send `guardrails=["presidio-pii"]` in the request body. **[Langchain, OpenAI SDK Usage Examples](../proxy/user_keys#request-format)** @@ -55,7 +120,7 @@ litellm --config config.yaml --detailed_debug Expect this to mask `Jane Doe` since it's PII -```shell +```shell title="Masked PII Request" showLineNumbers curl http://localhost:4000/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ @@ -64,13 +129,13 @@ curl http://localhost:4000/chat/completions \ "messages": [ {"role": "user", "content": "Hello my name is Jane Doe"} ], - "guardrails": ["presidio-pre-guard"], + "guardrails": ["presidio-pii"], }' ``` Expected response on failure -```shell +```shell title="Response with Masked PII" showLineNumbers { "id": "chatcmpl-A3qSC39K7imjGbZ8xCDacGJZBoTJQ", "choices": [ @@ -102,7 +167,7 @@ Expected response on failure -```shell +```shell title="No PII Request" showLineNumbers curl http://localhost:4000/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ @@ -111,13 +176,150 @@ curl http://localhost:4000/chat/completions \ "messages": [ {"role": "user", "content": "Hello good morning"} ], - "guardrails": ["presidio-pre-guard"], + "guardrails": ["presidio-pii"], }' ``` + +## Tracing Guardrail requests + +Once your guardrail is live in production, you will also be able to trace your guardrail on LiteLLM Logs, Langfuse, Arize Phoenix, etc, all LiteLLM logging integrations. + +### LiteLLM UI + +On the LiteLLM logs page you can see that the PII content was masked for this specific request. And you can see detailed tracing for the guardrail. This allows you to monitor entity types masked with their corresponding confidence score and the duration of the guardrail execution. + + + +### Langfuse + +When connecting Litellm to Langfuse, you can see the guardrail information on the Langfuse Trace. + + + +## Entity Type Configuration + +You can configure specific entity types for PII detection and decide how to handle each entity type (mask or block). + +### Configure Entity Types in config.yaml + +Define your guardrails with specific entity type configuration: + +```yaml title="config.yaml with Entity Types" showLineNumbers +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +guardrails: + - guardrail_name: "presidio-mask-guard" + litellm_params: + guardrail: presidio + mode: "pre_call" + pii_entities_config: + CREDIT_CARD: "MASK" # Will mask credit card numbers + EMAIL_ADDRESS: "MASK" # Will mask email addresses + + - guardrail_name: "presidio-block-guard" + litellm_params: + guardrail: presidio + mode: "pre_call" + pii_entities_config: + CREDIT_CARD: "BLOCK" # Will block requests containing credit card numbers +``` + +### Supported Entity Types + +LiteLLM Supports all Presidio entity types. See the complete list of presidio entity types [here](https://microsoft.github.io/presidio/supported_entities/). + +### Supported Actions + +For each entity type, you can specify one of the following actions: + +- `MASK`: Replace the entity with a placeholder (e.g., ``) +- `BLOCK`: Block the request entirely if this entity type is detected + +### Test request with Entity Type Configuration + + + + +When using the masking configuration, entities will be replaced with placeholders: + +```shell title="Masking PII Request" showLineNumbers +curl http://localhost:4000/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-1234" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [ + {"role": "user", "content": "My credit card is 4111-1111-1111-1111 and my email is test@example.com"} + ], + "guardrails": ["presidio-mask-guard"] + }' +``` + +Example response with masked entities: + +```json +{ + "id": "chatcmpl-123abc", + "choices": [ + { + "message": { + "content": "I can see you provided a and an . For security reasons, I recommend not sharing this sensitive information.", + "role": "assistant" + }, + "index": 0, + "finish_reason": "stop" + } + ], + // ... other response fields +} +``` + + + + + +When using the blocking configuration, requests containing the configured entity types will be blocked completely with an exception: + +```shell title="Blocking PII Request" showLineNumbers +curl http://localhost:4000/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-1234" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [ + {"role": "user", "content": "My credit card is 4111-1111-1111-1111"} + ], + "guardrails": ["presidio-block-guard"] + }' +``` + +When running this request, the proxy will raise a `BlockedPiiEntityError` exception. + +```json +{ + "error": { + "message": "Blocked PII entity detected: CREDIT_CARD by Guardrail: presidio-block-guard." + } +} +``` + +The exception includes the entity type that was blocked (`CREDIT_CARD` in this case) and the guardrail name that caused the blocking. + + ## Advanced @@ -129,7 +331,7 @@ The Presidio API [supports passing the `language` param](https://microsoft.githu -```shell +```shell title="Language Parameter - curl" showLineNumbers curl http://localhost:4000/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-1234" \ @@ -148,8 +350,7 @@ curl http://localhost:4000/chat/completions \ -```python - +```python title="Language Parameter - Python" showLineNumbers import openai client = openai.OpenAI( api_key="anything", @@ -179,7 +380,6 @@ print(response) - ### Output parsing @@ -188,7 +388,7 @@ LLM responses can sometimes contain the masked tokens. For presidio 'replace' operations, LiteLLM can check the LLM response and replace the masked token with the user-submitted values. Define your guardrails under the `guardrails` section -```yaml +```yaml title="Output Parsing Config" showLineNumbers model_list: - model_name: gpt-3.5-turbo litellm_params: @@ -223,7 +423,7 @@ Send ad-hoc recognizers to presidio `/analyze` by passing a json file to the pro #### Define ad-hoc recognizer on your LiteLLM config.yaml Define your guardrails under the `guardrails` section -```yaml +```yaml title="Ad Hoc Recognizers Config" showLineNumbers model_list: - model_name: gpt-3.5-turbo litellm_params: @@ -240,7 +440,7 @@ guardrails: Set the following env vars -```bash +```bash title="Ad Hoc Recognizers Environment Variables" showLineNumbers export PRESIDIO_ANALYZER_API_BASE="http://localhost:5002" export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001" ``` @@ -248,13 +448,13 @@ export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001" You can see this working, when you run the proxy: -```bash +```bash title="Run Proxy with Debug" showLineNumbers litellm --config /path/to/config.yaml --debug ``` Make a chat completions request, example: -``` +```json title="Custom PII Request" showLineNumbers { "model": "azure-gpt-3.5", "messages": [{"role": "user", "content": "John Smith AHV number is 756.3026.0705.92. Zip code: 1334023"}] @@ -262,7 +462,7 @@ Make a chat completions request, example: ``` And search for any log starting with `Presidio PII Masking`, example: -``` +```text title="PII Masking Log" showLineNumbers Presidio PII Masking: Redacted pii message: AHV number is . Zip code: ``` @@ -283,7 +483,7 @@ This is currently only applied for 1. Define mode: `logging_only` on your LiteLLM config.yaml Define your guardrails under the `guardrails` section -```yaml +```yaml title="Logging Only Config" showLineNumbers model_list: - model_name: gpt-3.5-turbo litellm_params: @@ -299,7 +499,7 @@ guardrails: Set the following env vars -```bash +```bash title="Logging Only Environment Variables" showLineNumbers export PRESIDIO_ANALYZER_API_BASE="http://localhost:5002" export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001" ``` @@ -307,13 +507,13 @@ export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001" 2. Start proxy -```bash +```bash title="Start Proxy" showLineNumbers litellm --config /path/to/config.yaml ``` 3. Test it! -```bash +```bash title="Test Logging Only" showLineNumbers curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ @@ -331,7 +531,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ **Expected Logged Response** -``` +```text title="Logged Response with Masked PII" showLineNumbers Hi, my name is ! ``` diff --git a/docs/my-website/docs/proxy/litellm_managed_files.md b/docs/my-website/docs/proxy/litellm_managed_files.md index 6e40c6dd449..de4bd1b9bfc 100644 --- a/docs/my-website/docs/proxy/litellm_managed_files.md +++ b/docs/my-website/docs/proxy/litellm_managed_files.md @@ -2,9 +2,18 @@ import TabItem from '@theme/TabItem'; import Tabs from '@theme/Tabs'; import Image from '@theme/IdealImage'; -# [BETA] Unified File ID +# [BETA] LiteLLM Managed Files + +Reuse the same file across different providers. + +:::info + +This is a free LiteLLM Enterprise feature. + +Available via the `litellm[proxy]` package or any `litellm` docker image. + +::: -Reuse the same 'file id' across different providers. | Feature | Description | Comments | | --- | --- | --- | @@ -15,8 +24,7 @@ Reuse the same 'file id' across different providers. Limitations of LiteLLM Managed Files: -- Only works for `/chat/completions` requests. -- Assumes just 1 model configured per model_name. +- Only works for `/chat/completions` and `/batch` requests. Follow [here](https://github.com/BerriAI/litellm/discussions/9632) for multiple models, batches support. diff --git a/docs/my-website/docs/proxy/logging.md b/docs/my-website/docs/proxy/logging.md index ad4cababc08..e6285ec31ee 100644 --- a/docs/my-website/docs/proxy/logging.md +++ b/docs/my-website/docs/proxy/logging.md @@ -11,7 +11,7 @@ Log Proxy input, output, and exceptions using: - GCS, s3, Azure (Blob) Buckets - Lunary - MLflow -- Custom Callbacks +- Custom Callbacks - Custom code and API endpoints - Langsmith - DataDog - DynamoDB @@ -1850,103 +1850,88 @@ ModelResponse( ## Custom Callback APIs [Async] + +

+ Send LiteLLM logs to a custom API endpoint +

+ :::info This is an Enterprise only feature [Get Started with Enterprise here](https://github.com/BerriAI/litellm/tree/main/enterprise) ::: +| Property | Details | +|----------|---------| +| Description | Log LLM Input/Output to a custom API endpoint | +| Logged Payload | `List[StandardLoggingPayload]` LiteLLM logs a list of [`StandardLoggingPayload` objects](https://docs.litellm.ai/docs/proxy/logging_spec) to your endpoint | + + + Use this if you: - Want to use custom callbacks written in a non Python programming language - Want your callbacks to run on a different microservice -#### Step 1. Create your generic logging API endpoint +#### Usage -Set up a generic API endpoint that can receive data in JSON format. The data will be included within a "data" field. +1. Set `success_callback: ["generic_api"]` on litellm config.yaml -Your server should support the following Request format: - -```shell -curl --location https://your-domain.com/log-event \ - --request POST \ - --header "Content-Type: application/json" \ - --data '{ - "data": { - "id": "chatcmpl-8sgE89cEQ4q9biRtxMvDfQU1O82PT", - "call_type": "acompletion", - "cache_hit": "None", - "startTime": "2024-02-15 16:18:44.336280", - "endTime": "2024-02-15 16:18:45.045539", - "model": "gpt-3.5-turbo", - "user": "ishaan-2", - "modelParameters": "{'temperature': 0.7, 'max_tokens': 10, 'user': 'ishaan-2', 'extra_body': {}}", - "messages": "[{'role': 'user', 'content': 'This is a test'}]", - "response": "ModelResponse(id='chatcmpl-8sgE89cEQ4q9biRtxMvDfQU1O82PT', choices=[Choices(finish_reason='length', index=0, message=Message(content='Great! How can I assist you with this test', role='assistant'))], created=1708042724, model='gpt-3.5-turbo-0613', object='chat.completion', system_fingerprint=None, usage=Usage(completion_tokens=10, prompt_tokens=11, total_tokens=21))", - "usage": "Usage(completion_tokens=10, prompt_tokens=11, total_tokens=21)", - "metadata": "{}", - "cost": "3.65e-05" - } - }' -``` - -Reference FastAPI Python Server - -Here's a reference FastAPI Server that is compatible with LiteLLM Proxy: - -```python -# this is an example endpoint to receive data from litellm -from fastapi import FastAPI, HTTPException, Request - -app = FastAPI() - - -@app.post("/log-event") -async def log_event(request: Request): - try: - print("Received /log-event request") - # Assuming the incoming request has JSON data - data = await request.json() - print("Received request data:") - print(data) - - # Your additional logic can go here - # For now, just printing the received data - - return {"message": "Request received successfully"} - except Exception as e: - print(f"Error processing request: {str(e)}") - import traceback - - traceback.print_exc() - raise HTTPException(status_code=500, detail="Internal Server Error") - - -if __name__ == "__main__": - import uvicorn - uvicorn.run(app, host="127.0.0.1", port=4000) -``` - -#### Step 2. Set your `GENERIC_LOGGER_ENDPOINT` to the endpoint + route we should send callback logs to - -```shell -os.environ["GENERIC_LOGGER_ENDPOINT"] = "http://localhost:4000/log-event" -``` - -#### Step 3. Create a `config.yaml` file and set `litellm_settings`: `success_callback` = ["generic"] - -Example litellm proxy config.yaml - -```yaml +```yaml showLineNumbers title="litellm config.yaml" model_list: - - model_name: gpt-3.5-turbo + - model_name: openai/gpt-4o litellm_params: - model: gpt-3.5-turbo + model: openai/gpt-4o + api_key: os.environ/OPENAI_API_KEY + litellm_settings: - success_callback: ["generic"] + success_callback: ["generic_api"] ``` -Start the LiteLLM Proxy and make a test request to verify the logs reached your callback API +2. Set Environment Variables for the custom API endpoint + +| Environment Variable | Details | Required | +|----------|---------|----------| +| `GENERIC_LOGGER_ENDPOINT` | The endpoint + route we should send callback logs to | Yes | +| `GENERIC_LOGGER_HEADERS` | Optional: Set headers to be sent to the custom API endpoint | No, this is optional | + +```shell showLineNumbers title=".env" +GENERIC_LOGGER_ENDPOINT="https://webhook-test.com/30343bc33591bc5e6dc44217ceae3e0a" + + +# Optional: Set headers to be sent to the custom API endpoint +GENERIC_LOGGER_HEADERS="Authorization=Bearer " +# if multiple headers, separate by commas +GENERIC_LOGGER_HEADERS="Authorization=Bearer ,X-Custom-Header=custom-header-value" +``` + +3. Start the proxy + +```shell +litellm --config /path/to/config.yaml +``` + +4. Make a test request + +```shell +curl -i --location 'http://0.0.0.0:4000/chat/completions' \ + --header 'Content-Type: application/json' \ + --header 'Authorization: Bearer sk-1234' \ + --data '{ + "model": "openai/gpt-4o", + "messages": [ + { + "role": "user", + "content": "what llm are you" + } + ] +}' +``` + + ## Langsmith diff --git a/docs/my-website/docs/proxy/logging_spec.md b/docs/my-website/docs/proxy/logging_spec.md index b314dd350b6..a39a62318e7 100644 --- a/docs/my-website/docs/proxy/logging_spec.md +++ b/docs/my-website/docs/proxy/logging_spec.md @@ -59,6 +59,22 @@ Inherits from `StandardLoggingUserAPIKeyMetadata` and adds: | `spend_logs_metadata` | `Optional[dict]` | Key-value pairs for spend logging | | `requester_ip_address` | `Optional[str]` | Requester's IP address | | `requester_metadata` | `Optional[dict]` | Additional requester metadata | +| `vector_store_request_metadata` | `Optional[List[StandardLoggingVectorStoreRequest]]` | Vector store request metadata | +| `requester_custom_headers` | Dict[str, str] | Any custom (`x-`) headers sent by the client to the proxy. | +| `guardrail_information` | `Optional[StandardLoggingGuardrailInformation]` | Guardrail information | + + +## StandardLoggingVectorStoreRequest + +| Field | Type | Description | +|-------|------|-------------| +| vector_store_id | Optional[str] | ID of the vector store | +| custom_llm_provider | Optional[str] | Custom LLM provider the vector store is associated with (e.g., bedrock, openai, anthropic) | +| query | Optional[str] | Query to the vector store | +| vector_store_search_response | Optional[VectorStoreSearchResponse] | OpenAI format vector store search response | +| start_time | Optional[float] | Start time of the vector store request | +| end_time | Optional[float] | End time of the vector store request | + ## StandardLoggingAdditionalHeaders @@ -113,4 +129,20 @@ Inherits from `StandardLoggingUserAPIKeyMetadata` and adds: A literal type with two possible values: - `"success"` -- `"failure"` \ No newline at end of file +- `"failure"` + +## StandardLoggingGuardrailInformation + +| Field | Type | Description | +|-------|------|-------------| +| `guardrail_name` | `Optional[str]` | Guardrail name | +| `guardrail_mode` | `Optional[Union[GuardrailEventHooks, List[GuardrailEventHooks]]]` | Guardrail mode | +| `guardrail_request` | `Optional[dict]` | Guardrail request | +| `guardrail_response` | `Optional[Union[dict, str, List[dict]]]` | Guardrail response | +| `guardrail_status` | `Literal["success", "failure"]` | Guardrail status | +| `start_time` | `Optional[float]` | Start time of the guardrail | +| `end_time` | `Optional[float]` | End time of the guardrail | +| `duration` | `Optional[float]` | Duration of the guardrail in seconds | +| `masked_entity_count` | `Optional[Dict[str, int]]` | Count of masked entities | + + diff --git a/docs/my-website/docs/proxy/managed_batches.md b/docs/my-website/docs/proxy/managed_batches.md new file mode 100644 index 00000000000..1b9b71c1779 --- /dev/null +++ b/docs/my-website/docs/proxy/managed_batches.md @@ -0,0 +1,263 @@ +# [BETA] LiteLLM Managed Files with Batches + +:::info + +This is a free LiteLLM Enterprise feature. + +Available via the `litellm[proxy]` package or any `litellm` docker image. + +::: + + +| Feature | Description | Comments | +| --- | --- | --- | +| Proxy | ✅ | | +| SDK | ❌ | Requires postgres DB for storing file ids | +| Available across all [Batch providers](../batches#supported-providers) | ✅ | | + + +## Overview + +Use this to: + +- Loadbalance across multiple Azure Batch deployments +- Control batch model access by key/user/team (same as chat completion models) + + +## (Proxy Admin) Usage + +Here's how to give developers access to your Batch models. + +### 1. Setup config.yaml + +- specify `mode: batch` for each model: Allows developers to know this is a batch model. + +```yaml showLineNumbers title="litellm_config.yaml" +model_list: + - model_name: "gpt-4o-batch" + litellm_params: + model: azure/gpt-4o-mini-general-deployment + api_base: os.environ/AZURE_API_BASE + api_key: os.environ/AZURE_API_KEY + model_info: + mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model + - model_name: "gpt-4o-batch" + litellm_params: + model: azure/gpt-4o-mini-special-deployment + api_base: os.environ/AZURE_API_BASE_2 + api_key: os.environ/AZURE_API_KEY_2 + model_info: + mode: batch # 👈 SPECIFY MODE AS BATCH, to tell user this is a batch model + +``` + +### 2. Create Virtual Key + +```bash showLineNumbers title="create_virtual_key.sh" +curl -L -X POST 'https://{PROXY_BASE_URL}/key/generate' \ +-H 'Authorization: Bearer ${PROXY_API_KEY}' \ +-H 'Content-Type: application/json' \ +-d '{"models": ["gpt-4o-batch"]}' +``` + + +You can now use the virtual key to access the batch models (See Developer flow). + +## (Developer) Usage + +Here's how to create a LiteLLM managed file and execute Batch CRUD operations with the file. + +### 1. Create request.jsonl + +- Check models available via `/model_group/info` +- See all models with `mode: batch` +- Set `model` in .jsonl to the model from `/model_group/info` + +```json showLineNumbers title="request.jsonl" +{"custom_id": "request-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-4o-batch", "messages": [{"role": "system", "content": "You are a helpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 1000}} +{"custom_id": "request-2", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "gpt-4o-batch", "messages": [{"role": "system", "content": "You are an unhelpful assistant."},{"role": "user", "content": "Hello world!"}],"max_tokens": 1000}} +``` + +Expectation: + +- LiteLLM translates this to the azure deployment specific value (e.g. `gpt-4o-mini-general-deployment`) + +### 2. Upload File + +Specify `target_model_names: ""` to enable LiteLLM managed files and request validation. + +model-name should be the same as the model-name in the request.jsonl + +```python showLineNumbers title="create_batch.py" +from openai import OpenAI + +client = OpenAI( + base_url="http://0.0.0.0:4000", + api_key="sk-1234", +) + +# Upload file +batch_input_file = client.files.create( + file=open("./request.jsonl", "rb"), # {"model": "gpt-4o-batch"} <-> {"model": "gpt-4o-mini-special-deployment"} + purpose="batch", + extra_body={"target_model_names": "gpt-4o-batch"} +) +print(batch_input_file) +``` + + +**Where is the file written?**: + +All gpt-4o-batch deployments (gpt-4o-mini-general-deployment, gpt-4o-mini-special-deployment) will be written to. This enables loadbalancing across all gpt-4o-batch deployments in Step 3. + +### 3. Create + Retrieve the batch + +```python showLineNumbers title="create_batch.py" +... +# Create batch +batch = client.batches.create( + input_file_id=batch_input_file.id, + endpoint="/v1/chat/completions", + completion_window="24h", + metadata={"description": "Test batch job"}, +) +print(batch) + +# Retrieve batch + +batch_response = client.batches.retrieve( + batch_id +) +status = batch_response.status +``` + +### 4. Retrieve Batch Content + +```python showLineNumbers title="create_batch.py" +... + +file_id = batch_response.output_file_id + +file_response = client.files.content(file_id) +print(file_response.text) +``` + +### 5. List batches + +```python showLineNumbers title="create_batch.py" +... + +client.batches.list(limit=10, extra_body={"target_model_names": "gpt-4o-batch"}) +``` + +### [Coming Soon] Cancel a batch + +```python showLineNumbers title="create_batch.py" +... + +client.batches.cancel(batch_id) +``` + + + +## E2E Example + +```python showLineNumbers title="create_batch.py" +import json +from pathlib import Path +from openai import OpenAI + +""" +litellm yaml: + +model_list: + - model_name: gpt-4o-batch + litellm_params: + model: azure/gpt-4o-my-special-deployment + api_key: .. + api_base: .. + +--- +request.jsonl: +{ + { + ..., + "body":{"model": "gpt-4o-batch", ...}} + } +} +""" + +client = OpenAI( + base_url="http://0.0.0.0:4000", + api_key="sk-1234", +) + +# Upload file +batch_input_file = client.files.create( + file=open("./request.jsonl", "rb"), + purpose="batch", + extra_body={"target_model_names": "gpt-4o-batch"} +) +print(batch_input_file) + + +# Create batch +batch = client.batches.create( # UPDATE BATCH ID TO FILE ID + input_file_id=batch_input_file.id, + endpoint="/v1/chat/completions", + completion_window="24h", + metadata={"description": "Test batch job"}, +) +print(batch) +batch_id = batch.id + +# Retrieve batch + +batch_response = client.batches.retrieve( # LOG VIRTUAL MODEL NAME + batch_id +) +status = batch_response.status + +print(f"status: {status}, output_file_id: {batch_response.output_file_id}") + +# Download file +output_file_id = batch_response.output_file_id +print(f"output_file_id: {output_file_id}") +if not output_file_id: + output_file_id = batch_response.error_file_id + +if output_file_id: + file_response = client.files.content( + output_file_id + ) + raw_responses = file_response.text.strip().split("\n") + + with open( + Path.cwd().parent / "unified_batch_output.json", "w" + ) as output_file: + for raw_response in raw_responses: + json.dump(json.loads(raw_response), output_file) + output_file.write("\n") +## List Batch + +list_batch_response = client.batches.list( # LOG VIRTUAL MODEL NAME + extra_query={"target_model_names": "gpt-4o-batch"} +) + +## Cancel Batch + +batch_response = client.batches.cancel( # LOG VIRTUAL MODEL NAME + batch_id +) +status = batch_response.status + +print(f"status: {status}") +``` + +## FAQ + +### Where are my files written? + +When a `target_model_names` is specified, the file is written to all deployments that match the `target_model_names`. + +No additional infrastructure is required. \ No newline at end of file diff --git a/docs/my-website/docs/proxy/management_cli.md b/docs/my-website/docs/proxy/management_cli.md new file mode 100644 index 00000000000..962831f6a35 --- /dev/null +++ b/docs/my-website/docs/proxy/management_cli.md @@ -0,0 +1,275 @@ +# LiteLLM Proxy CLI + +The `litellm-proxy` CLI is a command-line tool for managing your LiteLLM proxy +server. It provides commands for managing models, credentials, API keys, users, +and more, as well as making chat and HTTP requests to the proxy server. + +| Feature | What you can do | +|------------------------|-------------------------------------------------| +| Models Management | List, add, update, and delete models | +| Credentials Management | Manage provider credentials | +| Keys Management | Generate, list, and delete API keys | +| User Management | Create, list, and delete users | +| Chat Completions | Run chat completions | +| HTTP Requests | Make custom HTTP requests to the proxy server | + +## Quick Start + +1. **Install the CLI** + + If you have [uv](https://github.com/astral-sh/uv) installed, you can try this: + + ```shell + uvx --from=litellm[proxy] litellm-proxy + ``` + + and if things are working, you should see something like this: + + ```shell + Usage: litellm-proxy [OPTIONS] COMMAND [ARGS]... + + LiteLLM Proxy CLI - Manage your LiteLLM proxy server + + Options: + --base-url TEXT Base URL of the LiteLLM proxy server [env var: + LITELLM_PROXY_URL] + --api-key TEXT API key for authentication [env var: + LITELLM_PROXY_API_KEY] + --help Show this message and exit. + + Commands: + chat Chat with models through the LiteLLM proxy server + credentials Manage credentials for the LiteLLM proxy server + http Make HTTP requests to the LiteLLM proxy server + keys Manage API keys for the LiteLLM proxy server + models Manage models on your LiteLLM proxy server + ``` + + If this works, you can make use of the tool more convenient by doing: + + ```shell + uv tool install litellm[proxy] + ``` + + If that works, you'll see something like this: + + ```shell + ... + Installed 2 executables: litellm, litellm-proxy + ``` + + and now you can use the tool by just typing `litellm-proxy` in your terminal: + + ```shell + litellm-proxy + ``` + + In the future if you want to upgrade, you can do so with: + + ```shell + uv tool upgrade litellm[proxy] + ``` + + or if you want to uninstall, you can do so with: + + ```shell + uv tool uninstall litellm + ``` + + If you don't have uv or otherwise want to use pip, you can activate a virtual + environment and install the package manually: + + ```bash + pip install 'litellm[proxy]' + ``` + +2. **Set up environment variables** + + ```bash + export LITELLM_PROXY_URL=http://localhost:4000 + export LITELLM_PROXY_API_KEY=sk-your-key + ``` + + *(Replace with your actual proxy URL and API key)* + +3. **Make your first request (list models)** + + ```bash + litellm-proxy models list + ``` + + If the CLI is set up correctly, you should see a list of available models or a table output. + +4. **Troubleshooting** + + - If you see an error, check your environment variables and proxy server status. + +## Configuration + +You can configure the CLI using environment variables or command-line options: + +- `LITELLM_PROXY_URL`: Base URL of the LiteLLM proxy server (default: http://localhost:4000) +- `LITELLM_PROXY_API_KEY`: API key for authentication + +## Main Commands + +### Models Management + +- List, add, update, get, and delete models on the proxy. +- Example: + + ```bash + litellm-proxy models list + litellm-proxy models add gpt-4 \ + --param api_key=sk-123 \ + --param max_tokens=2048 + litellm-proxy models update -p temperature=0.7 + litellm-proxy models delete + ``` + + [API used (OpenAPI)](https://litellm-api.up.railway.app/#/model%20management) + +### Credentials Management + +- List, create, get, and delete credentials for LLM providers. +- Example: + + ```bash + litellm-proxy credentials list + litellm-proxy credentials create azure-prod \ + --info='{"custom_llm_provider": "azure"}' \ + --values='{"api_key": "sk-123", "api_base": "https://prod.azure.openai.com"}' + litellm-proxy credentials get azure-cred + litellm-proxy credentials delete azure-cred + ``` + + [API used (OpenAPI)](https://litellm-api.up.railway.app/#/credential%20management) + +### Keys Management + +- List, generate, get info, and delete API keys. +- Example: + + ```bash + litellm-proxy keys list + litellm-proxy keys generate \ + --models=gpt-4 \ + --spend=100 \ + --duration=24h \ + --key-alias=my-key + litellm-proxy keys info --key sk-key1 + litellm-proxy keys delete --keys sk-key1,sk-key2 --key-aliases alias1,alias2 + ``` + + [API used (OpenAPI)](https://litellm-api.up.railway.app/#/key%20management) + +### User Management + +- List, create, get info, and delete users. +- Example: + + ```bash + litellm-proxy users list + litellm-proxy users create \ + --email=user@example.com \ + --role=internal_user \ + --alias="Alice" \ + --team=team1 \ + --max-budget=100.0 + litellm-proxy users get --id + litellm-proxy users delete + ``` + + [API used (OpenAPI)](https://litellm-api.up.railway.app/#/Internal%20User%20management) + +### Chat Completions + +- Ask for chat completions from the proxy server. +- Example: + + ```bash + litellm-proxy chat completions gpt-4 -m "user:Hello, how are you?" + ``` + + [API used (OpenAPI)](https://litellm-api.up.railway.app/#/chat%2Fcompletions) + +### General HTTP Requests + +- Make direct HTTP requests to the proxy server. +- Example: + + ```bash + litellm-proxy http request \ + POST /chat/completions \ + --json '{"model": "gpt-4", "messages": [{"role": "user", "content": "Hello"}]}' + ``` + + [All APIs (OpenAPI)](https://litellm-api.up.railway.app/#/) + +## Environment Variables + +- `LITELLM_PROXY_URL`: Base URL of the proxy server +- `LITELLM_PROXY_API_KEY`: API key for authentication + +## Examples + +1. **List all models:** + + ```bash + litellm-proxy models list + ``` + +2. **Add a new model:** + + ```bash + litellm-proxy models add gpt-4 \ + --param api_key=sk-123 \ + --param max_tokens=2048 + ``` + +3. **Create a credential:** + + ```bash + litellm-proxy credentials create azure-prod \ + --info='{"custom_llm_provider": "azure"}' \ + --values='{"api_key": "sk-123", "api_base": "https://prod.azure.openai.com"}' + ``` + +4. **Generate an API key:** + + ```bash + litellm-proxy keys generate \ + --models=gpt-4 \ + --spend=100 \ + --duration=24h \ + --key-alias=my-key + ``` + +5. **Chat completion:** + + ```bash + litellm-proxy chat completions gpt-4 \ + -m "user:Write a story" + ``` + +6. **Custom HTTP request:** + + ```bash + litellm-proxy http request \ + POST /chat/completions \ + --json '{"model": "gpt-4", "messages": [{"role": "user", "content": "Hello"}]}' + ``` + +## Error Handling + +The CLI will display error messages for: + +- Server not accessible +- Authentication failures +- Invalid parameters or JSON +- Nonexistent models/credentials +- Any other operation failures + +Use the `--debug` flag for detailed debugging output. + +For full command reference and advanced usage, see the [CLI README](https://github.com/BerriAI/litellm/blob/main/litellm/proxy/client/cli/README.md). diff --git a/docs/my-website/docs/proxy/pii_masking.md b/docs/my-website/docs/proxy/pii_masking.md deleted file mode 100644 index 83e4965a495..00000000000 --- a/docs/my-website/docs/proxy/pii_masking.md +++ /dev/null @@ -1,246 +0,0 @@ -import Image from '@theme/IdealImage'; -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; - -# PII Masking - LiteLLM Gateway (Deprecated Version) - -:::warning - -This is deprecated, please use [our new Presidio pii masking integration](./guardrails/pii_masking_v2) - -::: - -LiteLLM supports [Microsoft Presidio](https://github.com/microsoft/presidio/) for PII masking. - - -## Quick Start -### Step 1. Add env - -```bash -export PRESIDIO_ANALYZER_API_BASE="http://localhost:5002" -export PRESIDIO_ANONYMIZER_API_BASE="http://localhost:5001" -``` - -### Step 2. Set it as a callback in config.yaml - -```yaml -litellm_settings: - callbacks = ["presidio", ...] # e.g. ["presidio", custom_callbacks.proxy_handler_instance] -``` - -### Step 3. Start proxy - - -``` -litellm --config /path/to/config.yaml -``` - - -This will mask the input going to the llm provider - - - -## Output parsing - -LLM responses can sometimes contain the masked tokens. - -For presidio 'replace' operations, LiteLLM can check the LLM response and replace the masked token with the user-submitted values. - -Just set `litellm.output_parse_pii = True`, to enable this. - - -```yaml -litellm_settings: - output_parse_pii: true -``` - -**Expected Flow: ** - -1. User Input: "hello world, my name is Jane Doe. My number is: 034453334" - -2. LLM Input: "hello world, my name is [PERSON]. My number is: [PHONE_NUMBER]" - -3. LLM Response: "Hey [PERSON], nice to meet you!" - -4. User Response: "Hey Jane Doe, nice to meet you!" - -## Ad-hoc recognizers - -Send ad-hoc recognizers to presidio `/analyze` by passing a json file to the proxy - -[**Example** ad-hoc recognizer](../../../../litellm/proxy/hooks/example_presidio_ad_hoc_recognizer.json) - -```yaml -litellm_settings: - callbacks: ["presidio"] - presidio_ad_hoc_recognizers: "./hooks/example_presidio_ad_hoc_recognizer.json" -``` - -You can see this working, when you run the proxy: - -```bash -litellm --config /path/to/config.yaml --debug -``` - -Make a chat completions request, example: - -``` -{ - "model": "azure-gpt-3.5", - "messages": [{"role": "user", "content": "John Smith AHV number is 756.3026.0705.92. Zip code: 1334023"}] -} -``` - -And search for any log starting with `Presidio PII Masking`, example: -``` -Presidio PII Masking: Redacted pii message: AHV number is . Zip code: -``` - - -## Turn on/off per key - -Turn off PII masking for a given key. - -Do this by setting `permissions: {"pii": false}`, when generating a key. - -```shell -curl --location 'http://0.0.0.0:4000/key/generate' \ ---header 'Authorization: Bearer sk-1234' \ ---header 'Content-Type: application/json' \ ---data '{ - "permissions": {"pii": false} -}' -``` - - -## Turn on/off per request - -The proxy support 2 request-level PII controls: - -- *no-pii*: Optional(bool) - Allow user to turn off pii masking per request. -- *output_parse_pii*: Optional(bool) - Allow user to turn off pii output parsing per request. - -### Usage - -**Step 1. Create key with pii permissions** - -Set `allow_pii_controls` to true for a given key. This will allow the user to set request-level PII controls. - -```bash -curl --location 'http://0.0.0.0:4000/key/generate' \ ---header 'Authorization: Bearer my-master-key' \ ---header 'Content-Type: application/json' \ ---data '{ - "permissions": {"allow_pii_controls": true} -}' -``` - -**Step 2. Turn off pii output parsing** - -```python -import os -from openai import OpenAI - -client = OpenAI( - # This is the default and can be omitted - api_key=os.environ.get("OPENAI_API_KEY"), - base_url="http://0.0.0.0:4000" -) - -chat_completion = client.chat.completions.create( - messages=[ - { - "role": "user", - "content": "My name is Jane Doe, my number is 8382043839", - } - ], - model="gpt-3.5-turbo", - extra_body={ - "content_safety": {"output_parse_pii": False} - } -) -``` - -**Step 3: See response** - -``` -{ - "id": "chatcmpl-8c5qbGTILZa1S4CK3b31yj5N40hFN", - "choices": [ - { - "finish_reason": "stop", - "index": 0, - "message": { - "content": "Hi [PERSON], what can I help you with?", - "role": "assistant" - } - } - ], - "created": 1704089632, - "model": "gpt-35-turbo", - "object": "chat.completion", - "system_fingerprint": null, - "usage": { - "completion_tokens": 47, - "prompt_tokens": 12, - "total_tokens": 59 - }, - "_response_ms": 1753.426 -} -``` - - -## Turn on for logging only - -Only apply PII Masking before logging to Langfuse, etc. - -Not on the actual llm api request / response. - -:::note -This is currently only applied for -- `/chat/completion` requests -- on 'success' logging - -::: - -1. Setup config.yaml -```yaml -litellm_settings: - presidio_logging_only: true - -model_list: - - model_name: gpt-3.5-turbo - litellm_params: - model: gpt-3.5-turbo - api_key: os.environ/OPENAI_API_KEY -``` - -2. Start proxy - -```bash -litellm --config /path/to/config.yaml -``` - -3. Test it! - -```bash -curl -X POST 'http://0.0.0.0:4000/chat/completions' \ --H 'Content-Type: application/json' \ --H 'Authorization: Bearer sk-1234' \ --D '{ - "model": "gpt-3.5-turbo", - "messages": [ - { - "role": "user", - "content": "Hi, my name is Jane!" - } - ] - }' -``` - - -**Expected Logged Response** - -``` -Hi, my name is ! -``` \ No newline at end of file diff --git a/docs/my-website/docs/proxy/release_cycle.md b/docs/my-website/docs/proxy/release_cycle.md index c5782087f21..10dd6d8b3c5 100644 --- a/docs/my-website/docs/proxy/release_cycle.md +++ b/docs/my-website/docs/proxy/release_cycle.md @@ -18,3 +18,8 @@ Follow our release notes [here](https://github.com/BerriAI/litellm/releases). Stable releases come out every week (typically Sunday) +### What is considered a 'minor' bump vs. 'patch' bump? + +- 'patch' bumps: extremely minor addition that doesn't affect any existing functionality or add any user-facing features. (e.g. a 'created_at' column in a database table) +- 'minor' bumps: add a new feature or a new database table that is backward compatible. +- 'major' bumps: break backward compatibility. \ No newline at end of file diff --git a/docs/my-website/docs/proxy/reliability.md b/docs/my-website/docs/proxy/reliability.md index 654c2618c2e..32b35e4bd24 100644 --- a/docs/my-website/docs/proxy/reliability.md +++ b/docs/my-website/docs/proxy/reliability.md @@ -117,7 +117,7 @@ response = router.completion( curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ --D '{ +-d '{ "model": "my-bad-model", "messages": [ { @@ -628,7 +628,7 @@ litellm_settings: curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ --D '{ +-d '{ "model": "gpt-4", "messages": [ { @@ -655,7 +655,7 @@ Check if your fallbacks are working as expected. curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ --D '{ +-d '{ "model": "my-bad-model", "messages": [ { @@ -674,7 +674,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ --D '{ +-d '{ "model": "my-bad-model", "messages": [ { @@ -693,7 +693,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ --D '{ +-d '{ "model": "my-bad-model", "messages": [ { @@ -1050,4 +1050,4 @@ curl -L -X POST 'http://0.0.0.0:4000/key/generate' \ ```
- \ No newline at end of file + diff --git a/docs/my-website/docs/proxy/spend_logs_deletion.md b/docs/my-website/docs/proxy/spend_logs_deletion.md new file mode 100644 index 00000000000..5b980e61eac --- /dev/null +++ b/docs/my-website/docs/proxy/spend_logs_deletion.md @@ -0,0 +1,91 @@ +# ✨ Maximum Retention Period for Spend Logs + +This walks through how to set the maximum retention period for spend logs. This helps manage database size by deleting old logs automatically. + +:::info + +✨ This is on LiteLLM Enterprise + +[Enterprise Pricing](https://www.litellm.ai/#pricing) + +[Get free 7-day trial key](https://www.litellm.ai/#trial) + +::: + +### Requirements + +- **Postgres** (for log storage) +- **Redis** *(optional)* — required only if you're running multiple proxy instances and want to enable distributed locking + +## Usage + +### Setup + +Add this to your `proxy_config.yaml` under `general_settings`: + +```yaml title="proxy_config.yaml" +general_settings: + maximum_spend_logs_retention_period: "7d" # Keep logs for 7 days + + # Optional: set how frequently cleanup should run - default is daily + maximum_spend_logs_retention_interval: "1d" # Run cleanup daily + +litellm_settings: + cache: true + cache_params: + type: redis +``` + +### Configuration Options + +#### `maximum_spend_logs_retention_period` (required) + +How long logs should be kept before deletion. Supported formats: + +- `"7d"` – 7 days +- `"24h"` – 24 hours +- `"60m"` – 60 minutes +- `"3600s"` – 3600 seconds + +#### `maximum_spend_logs_retention_interval` (optional) + +How often the cleanup job should run. Uses the same format as above. If not set, cleanup will run every 24 hours if and only if `maximum_spend_logs_retention_period` is set. + +## How it works + +### Step 1. Lock Acquisition (Optional with Redis) + +If Redis is enabled, LiteLLM uses it to make sure only one instance runs the cleanup at a time. + +- If the lock is acquired: + - This instance proceeds with cleanup + - Others skip it +- If no lock is present: + - Cleanup still runs (useful for single-node setups) + +![Working of spend log deletions](../../img/spend_log_deletion_working.png) +*Working of spend log deletions* + +### Step 2. Batch Deletion + +Once cleanup starts: + +- It calculates the cutoff date using the configured retention period +- Deletes logs older than the cutoff in **batches of 1000** +- Adds a short delay between batches to avoid overloading the database + +### Default settings: +- **Batch size**: 1000 logs +- **Max batches per run**: 500 +- **Max deletions per run**: 500,000 logs + +You can change the number of batches using an environment variable: + +```bash +SPEND_LOG_RUN_LOOPS=200 +``` + +This would allow up to 200,000 logs to be deleted in one run. + +![Batch deletion of old logs](../../img/spend_log_deletion_multi_pod.jpg) +*Batch deletion of old logs* diff --git a/docs/my-website/docs/proxy/ui_logs.md b/docs/my-website/docs/proxy/ui_logs.md index c6cbbe6e7b7..bca50a2165b 100644 --- a/docs/my-website/docs/proxy/ui_logs.md +++ b/docs/my-website/docs/proxy/ui_logs.md @@ -52,3 +52,30 @@ If you do not want to store spend logs in DB, you can opt out with this setting general_settings: disable_spend_logs: True # Disable writing spend logs to DB ``` + +## Automatically Deleting Old Spend Logs + +If you're storing spend logs, it might be a good idea to delete them regularly to keep the database fast. + +LiteLLM lets you configure this in your `proxy_config.yaml`: + +```yaml +general_settings: + maximum_spend_logs_retention_period: "7d" # Delete logs older than 7 days + + # Optional: how often to run cleanup + maximum_spend_logs_retention_interval: "1d" # Run once per day +``` + +You can control how many logs are deleted per run using this environment variable: + +`SPEND_LOG_RUN_LOOPS=200 # Deletes up to 200,000 logs in one run (batch size = 1000)` + +For detailed architecture and how it works, see [Spend Logs Deletion](../proxy/spend_logs_deletion). + + + + + + + diff --git a/docs/my-website/docs/proxy/users.md b/docs/my-website/docs/proxy/users.md index 92ea73b9d2e..b4457b8d553 100644 --- a/docs/my-website/docs/proxy/users.md +++ b/docs/my-website/docs/proxy/users.md @@ -786,6 +786,17 @@ Expected Response: } } ``` + +### [BETA] Multi-instance rate limiting + +Enable multi-instance rate limiting with the env var `EXPERIMENTAL_MULTI_INSTANCE_RATE_LIMITING="True"` + +Changes: +- This moves to using async_increment instead of async_set_cache when updating current requests/tokens. +- The in-memory cache is synced with redis every 0.01s, to avoid calling redis for every request. +- In testing, this was found to be 2x faster than the previous implementation, and reduced drift between expected and actual fails to at most 10 requests at high-traffic (100 RPS across 3 instances). + + ## Grant Access to new model Use model access groups to give users access to select models, and add new ones to it over time (e.g. mistral, llama-2, etc.). diff --git a/docs/my-website/docs/routing.md b/docs/my-website/docs/routing.md index 967d5ad483e..fa784a719c2 100644 --- a/docs/my-website/docs/routing.md +++ b/docs/my-website/docs/routing.md @@ -25,7 +25,7 @@ If you want a server to load balance across different LLM APIs, use our [LiteLLM ### Quick Start -Loadbalance across multiple [azure](./providers/azure.md)/[bedrock](./providers/bedrock.md)/[provider](./providers/) deployments. LiteLLM will handle retrying in different regions if a call fails. +Loadbalance across multiple [azure](./providers/azure)/[bedrock](./providers/bedrock.md)/[provider](./providers/) deployments. LiteLLM will handle retrying in different regions if a call fails. diff --git a/docs/my-website/docs/tutorials/gemini_realtime_with_audio.md b/docs/my-website/docs/tutorials/gemini_realtime_with_audio.md new file mode 100644 index 00000000000..e6814c56900 --- /dev/null +++ b/docs/my-website/docs/tutorials/gemini_realtime_with_audio.md @@ -0,0 +1,136 @@ +# Call Gemini Realtime API with Audio Input/Output + +:::info +Requires LiteLLM Proxy v1.70.1+ +::: + +1. Setup config.yaml for LiteLLM Proxy + +```yaml +model_list: + - model_name: "gemini-2.0-flash" + litellm_params: + model: gemini/gemini-2.0-flash-live-001 + model_info: + mode: realtime +``` + +2. Start LiteLLM Proxy + +```bash +litellm-proxy start +``` + +3. Run test script + +```python +import asyncio +import websockets +import json +import base64 +from dotenv import load_dotenv +import wave +import base64 +import soundfile as sf +import sounddevice as sd +import io +import numpy as np + +# Load environment variables + +OPENAI_API_KEY = "sk-1234" # Replace with your LiteLLM API key +OPENAI_API_URL = 'ws://{PROXY_URL}/v1/realtime?model=gemini-2.0-flash' # REPLACE WITH `wss://{PROXY_URL}/v1/realtime?model=gemini-2.0-flash` for secure connection +WAV_FILE_PATH = "/path/to/audio.wav" # Replace with your .wav file path + +async def send_session_update(ws): + session_update = { + "type": "session.update", + "session": { + "conversation_id": "123456", + "language": "en-US", + "transcription_mode": "fast", + "modalities": ["text"] + } + } + await ws.send(json.dumps(session_update)) + +async def send_audio_file(ws, file_path): + with wave.open(file_path, 'rb') as wav_file: + chunk_size = 1024 # Adjust as needed + while True: + chunk = wav_file.readframes(chunk_size) + if not chunk: + break + base64_audio = base64.b64encode(chunk).decode('utf-8') + audio_message = { + "type": "input_audio_buffer.append", + "audio": base64_audio + } + await ws.send(json.dumps(audio_message)) + await asyncio.sleep(0.1) # Add a small delay to simulate real-time streaming + + # Send end of audio stream message + await ws.send(json.dumps({"type": "input_audio_buffer.end"})) + +def play_base64_audio(base64_string, sample_rate=24000, channels=1): + # Decode the base64 string + audio_data = base64.b64decode(base64_string) + + # Convert to numpy array + audio_np = np.frombuffer(audio_data, dtype=np.int16) + + # Reshape if stereo + if channels == 2: + audio_np = audio_np.reshape(-1, 2) + + # Normalize + audio_float = audio_np.astype(np.float32) / 32768.0 + + # Play the audio + sd.play(audio_float, sample_rate) + sd.wait() + + +def combine_base64_audio(base64_strings): + # Step 1: Decode base64 strings to binary + binary_data = [base64.b64decode(s) for s in base64_strings] + + # Step 2: Concatenate binary data + combined_binary = b''.join(binary_data) + + # Step 3: Encode combined binary back to base64 + combined_base64 = base64.b64encode(combined_binary).decode('utf-8') + + return combined_base64 + +async def listen_in_background(ws): + combined_b64_audio_str = [] + try: + while True: + response = await ws.recv() + message_json = json.loads(response) + print(f"message_json: {message_json}") + + if message_json['type'] == 'response.audio.delta' and message_json.get('delta'): + play_base64_audio(message_json["delta"]) + except Exception: + print("END OF STREAM") + +async def main(): + async with websockets.connect( + OPENAI_API_URL, + additional_headers={ + "Authorization": f"Bearer {OPENAI_API_KEY}", + "OpenAI-Beta": "realtime=v1" + } + ) as ws: + asyncio.create_task(listen_in_background(ws=ws)) + await send_session_update(ws) + await send_audio_file(ws, WAV_FILE_PATH) + + + +if __name__ == "__main__": + asyncio.run(main()) +``` + diff --git a/docs/my-website/docs/tutorials/google_adk.md b/docs/my-website/docs/tutorials/google_adk.md new file mode 100644 index 00000000000..81a3dacc153 --- /dev/null +++ b/docs/my-website/docs/tutorials/google_adk.md @@ -0,0 +1,324 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; +import Image from '@theme/IdealImage'; + + +# Google ADK with LiteLLM + + +

+ Use Google ADK with LiteLLM Python SDK, LiteLLM Proxy +

+ + +This tutorial shows you how to create intelligent agents using Agent Development Kit (ADK) with support for multiple Large Language Model (LLM) providers with LiteLLM. + + + +## Overview + +ADK (Agent Development Kit) allows you to build intelligent agents powered by LLMs. By integrating with LiteLLM, you can: + +- Use multiple LLM providers (OpenAI, Anthropic, Google, etc.) +- Switch easily between models from different providers +- Connect to a LiteLLM proxy for centralized model management + +## Prerequisites + +- Python environment setup +- API keys for model providers (OpenAI, Anthropic, Google AI Studio) +- Basic understanding of LLMs and agent concepts + +## Installation + +```bash showLineNumbers title="Install dependencies" +pip install google-adk litellm +``` + +## 1. Setting Up Environment + +First, import the necessary libraries and set up your API keys: + +```python showLineNumbers title="Setup environment and API keys" +import os +import asyncio +from google.adk.agents import Agent +from google.adk.models.lite_llm import LiteLlm # For multi-model support +from google.adk.sessions import InMemorySessionService +from google.adk.runners import Runner +from google.genai import types +import litellm # Import for proxy configuration + +# Set your API keys +os.environ["GOOGLE_API_KEY"] = "your-google-api-key" # For Gemini models +os.environ["OPENAI_API_KEY"] = "your-openai-api-key" # For OpenAI models +os.environ["ANTHROPIC_API_KEY"] = "your-anthropic-api-key" # For Claude models + +# Define model constants for cleaner code +MODEL_GEMINI_PRO = "gemini-1.5-pro" +MODEL_GPT_4O = "openai/gpt-4o" +MODEL_CLAUDE_SONNET = "anthropic/claude-3-sonnet-20240229" +``` + +## 2. Define a Simple Tool + +Create a tool that your agent can use: + +```python showLineNumbers title="Weather tool implementation" +def get_weather(city: str) -> dict: + """Retrieves the current weather report for a specified city. + + Args: + city (str): The name of the city (e.g., "New York", "London", "Tokyo"). + + Returns: + dict: A dictionary containing the weather information. + Includes a 'status' key ('success' or 'error'). + If 'success', includes a 'report' key with weather details. + If 'error', includes an 'error_message' key. + """ + print(f"Tool: get_weather called for city: {city}") + + # Mock weather data + mock_weather_db = { + "newyork": {"status": "success", "report": "The weather in New York is sunny with a temperature of 25°C."}, + "london": {"status": "success", "report": "It's cloudy in London with a temperature of 15°C."}, + "tokyo": {"status": "success", "report": "Tokyo is experiencing light rain and a temperature of 18°C."}, + } + + city_normalized = city.lower().replace(" ", "") + + if city_normalized in mock_weather_db: + return mock_weather_db[city_normalized] + else: + return {"status": "error", "error_message": f"Sorry, I don't have weather information for '{city}'."} +``` + +## 3. Helper Function for Agent Interaction + +Create a helper function to facilitate agent interaction: + +```python showLineNumbers title="Agent interaction helper function" +async def call_agent_async(query: str, runner, user_id, session_id): + """Sends a query to the agent and prints the final response.""" + print(f"\n>>> User Query: {query}") + + # Prepare the user's message in ADK format + content = types.Content(role='user', parts=[types.Part(text=query)]) + + final_response_text = "Agent did not produce a final response." + + # Execute the agent and find the final response + async for event in runner.run_async( + user_id=user_id, + session_id=session_id, + new_message=content + ): + if event.is_final_response(): + if event.content and event.content.parts: + final_response_text = event.content.parts[0].text + break + + print(f"<<< Agent Response: {final_response_text}") +``` + +## 4. Using Different Model Providers with ADK + +### 4.1 Using OpenAI Models + +```python showLineNumbers title="OpenAI model implementation" +# Create an agent powered by OpenAI's GPT model +weather_agent_gpt = Agent( + name="weather_agent_gpt", + model=LiteLlm(model=MODEL_GPT_4O), # Use OpenAI's GPT model + description="Provides weather information using OpenAI's GPT.", + instruction="You are a helpful weather assistant powered by GPT-4o. " + "Use the 'get_weather' tool for city weather requests. " + "Present information clearly.", + tools=[get_weather], +) + +# Set up session and runner +session_service_gpt = InMemorySessionService() +session_gpt = session_service_gpt.create_session( + app_name="weather_app", + user_id="user_1", + session_id="session_gpt" +) + +runner_gpt = Runner( + agent=weather_agent_gpt, + app_name="weather_app", + session_service=session_service_gpt +) + +# Test the GPT agent +async def test_gpt_agent(): + print("\n--- Testing GPT Agent ---") + await call_agent_async( + "What's the weather in London?", + runner=runner_gpt, + user_id="user_1", + session_id="session_gpt" + ) + +# Execute the conversation with the GPT agent +await test_gpt_agent() + +# Or if running as a standard Python script: +# if __name__ == "__main__": +# asyncio.run(test_gpt_agent()) +``` + +### 4.2 Using Anthropic Models + +```python showLineNumbers title="Anthropic model implementation" +# Create an agent powered by Anthropic's Claude model +weather_agent_claude = Agent( + name="weather_agent_claude", + model=LiteLlm(model=MODEL_CLAUDE_SONNET), # Use Anthropic's Claude model + description="Provides weather information using Anthropic's Claude.", + instruction="You are a helpful weather assistant powered by Claude Sonnet. " + "Use the 'get_weather' tool for city weather requests. " + "Present information clearly.", + tools=[get_weather], +) + +# Set up session and runner +session_service_claude = InMemorySessionService() +session_claude = session_service_claude.create_session( + app_name="weather_app", + user_id="user_1", + session_id="session_claude" +) + +runner_claude = Runner( + agent=weather_agent_claude, + app_name="weather_app", + session_service=session_service_claude +) + +# Test the Claude agent +async def test_claude_agent(): + print("\n--- Testing Claude Agent ---") + await call_agent_async( + "What's the weather in Tokyo?", + runner=runner_claude, + user_id="user_1", + session_id="session_claude" + ) + +# Execute the conversation with the Claude agent +await test_claude_agent() + +# Or if running as a standard Python script: +# if __name__ == "__main__": +# asyncio.run(test_claude_agent()) +``` + +### 4.3 Using Google's Gemini Models + +```python showLineNumbers title="Gemini model implementation" +# Create an agent powered by Google's Gemini model +weather_agent_gemini = Agent( + name="weather_agent_gemini", + model=MODEL_GEMINI_PRO, # Use Gemini model directly (no LiteLlm wrapper needed) + description="Provides weather information using Google's Gemini.", + instruction="You are a helpful weather assistant powered by Gemini Pro. " + "Use the 'get_weather' tool for city weather requests. " + "Present information clearly.", + tools=[get_weather], +) + +# Set up session and runner +session_service_gemini = InMemorySessionService() +session_gemini = session_service_gemini.create_session( + app_name="weather_app", + user_id="user_1", + session_id="session_gemini" +) + +runner_gemini = Runner( + agent=weather_agent_gemini, + app_name="weather_app", + session_service=session_service_gemini +) + +# Test the Gemini agent +async def test_gemini_agent(): + print("\n--- Testing Gemini Agent ---") + await call_agent_async( + "What's the weather in New York?", + runner=runner_gemini, + user_id="user_1", + session_id="session_gemini" + ) + +# Execute the conversation with the Gemini agent +await test_gemini_agent() + +# Or if running as a standard Python script: +# if __name__ == "__main__": +# asyncio.run(test_gemini_agent()) +``` + +## 5. Using LiteLLM Proxy with ADK + +LiteLLM proxy provides a unified API endpoint for multiple models, simplifying deployment and centralized management. + +Required settings for using litellm proxy + +| Variable | Description | +|----------|-------------| +| `LITELLM_PROXY_API_KEY` | The API key for the LiteLLM proxy | +| `LITELLM_PROXY_API_BASE` | The base URL for the LiteLLM proxy | +| `USE_LITELLM_PROXY` or `litellm.use_litellm_proxy` | When set to True, your request will be sent to litellm proxy. | + +```python showLineNumbers title="LiteLLM proxy integration" +# Set your LiteLLM Proxy credentials as environment variables +os.environ["LITELLM_PROXY_API_KEY"] = "your-litellm-proxy-api-key" +os.environ["LITELLM_PROXY_API_BASE"] = "your-litellm-proxy-url" # e.g., "http://localhost:4000" +# Enable the use_litellm_proxy flag +litellm.use_litellm_proxy = True + +# Create a proxy-enabled agent (using environment variables) +weather_agent_proxy_env = Agent( + name="weather_agent_proxy_env", + model=LiteLlm(model="gpt-4o"), # this will call the `gpt-4o` model on LiteLLM proxy + description="Provides weather information using a model from LiteLLM proxy.", + instruction="You are a helpful weather assistant. " + "Use the 'get_weather' tool for city weather requests. " + "Present information clearly.", + tools=[get_weather], +) + +# Set up session and runner +session_service_proxy_env = InMemorySessionService() +session_proxy_env = session_service_proxy_env.create_session( + app_name="weather_app", + user_id="user_1", + session_id="session_proxy_env" +) + +runner_proxy_env = Runner( + agent=weather_agent_proxy_env, + app_name="weather_app", + session_service=session_service_proxy_env +) + +# Test the proxy-enabled agent (environment variables method) +async def test_proxy_env_agent(): + print("\n--- Testing Proxy-enabled Agent (Environment Variables) ---") + await call_agent_async( + "What's the weather in London?", + runner=runner_proxy_env, + user_id="user_1", + session_id="session_proxy_env" + ) + +# Execute the conversation +await test_proxy_env_agent() +``` diff --git a/docs/my-website/docs/tutorials/instructor.md b/docs/my-website/docs/tutorials/instructor.md index d972aff9151..073215b47be 100644 --- a/docs/my-website/docs/tutorials/instructor.md +++ b/docs/my-website/docs/tutorials/instructor.md @@ -1,80 +1,73 @@ -# Instructor - Function Calling +# Instructor -Use LiteLLM with [jxnl's instructor library](https://github.com/jxnl/instructor) for function calling in prod. +Combine LiteLLM with [jxnl's instructor library](https://github.com/jxnl/instructor) for more robust structured outputs. Outputs are automatically validated into Pydantic types and validation errors are provided back to the model to increase the chance of a successful response in the retries. -## Usage +## Usage (Sync) ```python -import os - import instructor from litellm import completion from pydantic import BaseModel -os.environ["LITELLM_LOG"] = "DEBUG" # 👈 print DEBUG LOGS client = instructor.from_litellm(completion) -# import dotenv -# dotenv.load_dotenv() - -class UserDetail(BaseModel): +class User(BaseModel): name: str age: int -user = client.chat.completions.create( - model="gpt-4o-mini", - response_model=UserDetail, - messages=[ - {"role": "user", "content": "Extract Jason is 25 years old"}, - ], -) +def extract_user(text: str): + return client.chat.completions.create( + model="gpt-4o-mini", + response_model=User, + messages=[ + {"role": "user", "content": text}, + ], + max_retries=3, + ) -assert isinstance(user, UserDetail) +user = extract_user("Jason is 25 years old") + +assert isinstance(user, User) assert user.name == "Jason" assert user.age == 25 - -print(f"user: {user}") +print(f"{user=}") ``` -## Async Calls +## Usage (Async) ```python import asyncio + import instructor -from litellm import Router +from litellm import acompletion from pydantic import BaseModel -aclient = instructor.patch( - Router( - model_list=[ - { - "model_name": "gpt-4o-mini", - "litellm_params": {"model": "gpt-4o-mini"}, - } - ], - default_litellm_params={"acompletion": True}, # 👈 IMPORTANT - tells litellm to route to async completion function. - ) -) + +client = instructor.from_litellm(acompletion) -class UserExtract(BaseModel): +class User(BaseModel): name: str age: int -async def main(): - model = await aclient.chat.completions.create( +async def extract(text: str) -> User: + return await client.chat.completions.create( model="gpt-4o-mini", - response_model=UserExtract, + response_model=User, messages=[ - {"role": "user", "content": "Extract jason is 25 years old"}, + {"role": "user", "content": text}, ], + max_retries=3, ) - print(f"model: {model}") +user = asyncio.run(extract("Alice is 30 years old")) -asyncio.run(main()) -``` \ No newline at end of file +assert isinstance(user, User) +assert user.name == "Alice" +assert user.age == 30 +print(f"{user=}") +``` diff --git a/docs/my-website/docs/tutorials/openweb_ui.md b/docs/my-website/docs/tutorials/openweb_ui.md index b2c1204069c..82ff475add9 100644 --- a/docs/my-website/docs/tutorials/openweb_ui.md +++ b/docs/my-website/docs/tutorials/openweb_ui.md @@ -2,35 +2,35 @@ import Image from '@theme/IdealImage'; import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; -# OpenWeb UI with LiteLLM +# Open WebUI with LiteLLM -This guide walks you through connecting OpenWeb UI to LiteLLM. Using LiteLLM with OpenWeb UI allows teams to -- Access 100+ LLMs on OpenWeb UI +This guide walks you through connecting Open WebUI to LiteLLM. Using LiteLLM with Open WebUI allows teams to +- Access 100+ LLMs on Open WebUI - Track Spend / Usage, Set Budget Limits - Send Request/Response Logs to logging destinations like langfuse, s3, gcs buckets, etc. -- Set access controls eg. Control what models OpenWebUI can access. +- Set access controls eg. Control what models Open WebUI can access. ## Quickstart - Make sure to setup LiteLLM with the [LiteLLM Getting Started Guide](https://docs.litellm.ai/docs/proxy/docker_quick_start) -## 1. Start LiteLLM & OpenWebUI +## 1. Start LiteLLM & Open WebUI -- OpenWebUI starts running on [http://localhost:3000](http://localhost:3000) +- Open WebUI starts running on [http://localhost:3000](http://localhost:3000) - LiteLLM starts running on [http://localhost:4000](http://localhost:4000) ## 2. Create a Virtual Key on LiteLLM -Virtual Keys are API Keys that allow you to authenticate to LiteLLM Proxy. We will create a Virtual Key that will allow OpenWebUI to access LiteLLM. +Virtual Keys are API Keys that allow you to authenticate to LiteLLM Proxy. We will create a Virtual Key that will allow Open WebUI to access LiteLLM. ### 2.1 LiteLLM User Management Hierarchy On LiteLLM, you can create Organizations, Teams, Users and Virtual Keys. For this tutorial, we will create a Team and a Virtual Key. - `Organization` - An Organization is a group of Teams. (US Engineering, EU Developer Tools) -- `Team` - A Team is a group of Users. (OpenWeb UI Team, Data Science Team, etc.) +- `Team` - A Team is a group of Users. (Open WebUI Team, Data Science Team, etc.) - `User` - A User is an individual user (employee, developer, eg. `krrish@litellm.ai`) - `Virtual Key` - A Virtual Key is an API Key that allows you to authenticate to LiteLLM Proxy. A Virtual Key is associated with a User or Team. @@ -46,13 +46,13 @@ Navigate to [http://localhost:4000/ui](http://localhost:4000/ui) and create a ne Navigate to [http://localhost:4000/ui](http://localhost:4000/ui) and create a new virtual Key. -LiteLLM allows you to specify what models are available on OpenWeb UI (by specifying the models the key will have access to). +LiteLLM allows you to specify what models are available on Open WebUI (by specifying the models the key will have access to). -## 3. Connect OpenWeb UI to LiteLLM +## 3. Connect Open WebUI to LiteLLM -On OpenWeb UI, navigate to Settings -> Connections and create a new connection to LiteLLM +On Open WebUI, navigate to Settings -> Connections and create a new connection to LiteLLM Enter the following details: - URL: `http://localhost:4000` (your litellm proxy base url) @@ -68,17 +68,52 @@ Once you selected a model, enter your message content and click on `Submit` -### 3.2 Tracking Spend / Usage +### 3.2 Tracking Usage & Spend -After your request is made, navigate to `Logs` on the LiteLLM UI, you can see Team, Key, Model, Usage and Cost. +#### Basic Tracking - +After making requests, navigate to the `Logs` section in the LiteLLM UI to view Model, Usage and Cost information. + +#### Per-User Tracking + +To track spend and usage for each Open WebUI user, configure both Open WebUI and LiteLLM: + +1. **Enable User Info Headers in Open WebUI** + + Set the following environment variable for Open WebUI to enable user information in request headers: + ```dotenv + ENABLE_FORWARD_USER_INFO_HEADERS=True + ``` + + For more details, see the [Environment Variable Configuration Guide](https://docs.openwebui.com/getting-started/env-configuration/#enable_forward_user_info_headers). + +2. **Configure LiteLLM to Parse User Headers** + + Add the following to your LiteLLM `config.yaml` to specify a header to use for user tracking: + + ```yaml + general_settings: + user_header_name: X-OpenWebUI-User-Id + ``` + + ⓘ Available tracking options + + You can use any of the following headers for `user_header_name`: + - `X-OpenWebUI-User-Id` + - `X-OpenWebUI-User-Email` + - `X-OpenWebUI-User-Name` + + These may offer better readability and easier mental attribution when hosting for a small group of users that you know well. + + Choose based on your needs, but note that in Open WebUI: + - Users can modify their own usernames + - Administrators can modify both usernames and emails of any account -## Render `thinking` content on OpenWeb UI +## Render `thinking` content on Open WebUI -OpenWebUI requires reasoning/thinking content to be rendered with `` tags. In order to render this for specific models, you can use the `merge_reasoning_content_in_choices` litellm parameter. +Open WebUI requires reasoning/thinking content to be rendered with `` tags. In order to render this for specific models, you can use the `merge_reasoning_content_in_choices` litellm parameter. Example litellm config.yaml: @@ -92,11 +127,11 @@ model_list: merge_reasoning_content_in_choices: true ``` -### Test it on OpenWeb UI +### Test it on Open WebUI On the models dropdown select `thinking-anthropic-claude-3-7-sonnet` ## Additional Resources -- Running LiteLLM and OpenWebUI on Windows Localhost: A Comprehensive Guide [https://www.tanyongsheng.com/note/running-litellm-and-openwebui-on-windows-localhost-a-comprehensive-guide/](https://www.tanyongsheng.com/note/running-litellm-and-openwebui-on-windows-localhost-a-comprehensive-guide/) \ No newline at end of file +- Running LiteLLM and Open WebUI on Windows Localhost: A Comprehensive Guide [https://www.tanyongsheng.com/note/running-litellm-and-openwebui-on-windows-localhost-a-comprehensive-guide/](https://www.tanyongsheng.com/note/running-litellm-and-openwebui-on-windows-localhost-a-comprehensive-guide/) diff --git a/docs/my-website/img/callback_api.png b/docs/my-website/img/callback_api.png new file mode 100644 index 00000000000..b123dae2620 Binary files /dev/null and b/docs/my-website/img/callback_api.png differ diff --git a/docs/my-website/img/delete_spend_logs.jpg b/docs/my-website/img/delete_spend_logs.jpg new file mode 100644 index 00000000000..6fa0f04b657 Binary files /dev/null and b/docs/my-website/img/delete_spend_logs.jpg differ diff --git a/docs/my-website/img/email_2.png b/docs/my-website/img/email_2.png new file mode 100644 index 00000000000..d686022824e Binary files /dev/null and b/docs/my-website/img/email_2.png differ diff --git a/docs/my-website/img/email_2_0.png b/docs/my-website/img/email_2_0.png new file mode 100644 index 00000000000..3e2c5d59db9 Binary files /dev/null and b/docs/my-website/img/email_2_0.png differ diff --git a/docs/my-website/img/email_event_1.png b/docs/my-website/img/email_event_1.png new file mode 100644 index 00000000000..edbb0a80931 Binary files /dev/null and b/docs/my-website/img/email_event_1.png differ diff --git a/docs/my-website/img/email_event_2.png b/docs/my-website/img/email_event_2.png new file mode 100644 index 00000000000..b4ef1f49c2e Binary files /dev/null and b/docs/my-website/img/email_event_2.png differ diff --git a/docs/my-website/img/gemini_realtime.png b/docs/my-website/img/gemini_realtime.png new file mode 100644 index 00000000000..2311a63f7d3 Binary files /dev/null and b/docs/my-website/img/gemini_realtime.png differ diff --git a/docs/my-website/img/kb.png b/docs/my-website/img/kb.png new file mode 100644 index 00000000000..ba35e7a8a0f Binary files /dev/null and b/docs/my-website/img/kb.png differ diff --git a/docs/my-website/img/kb_2.png b/docs/my-website/img/kb_2.png new file mode 100644 index 00000000000..0cce544a9fe Binary files /dev/null and b/docs/my-website/img/kb_2.png differ diff --git a/docs/my-website/img/kb_3.png b/docs/my-website/img/kb_3.png new file mode 100644 index 00000000000..5e169e16f40 Binary files /dev/null and b/docs/my-website/img/kb_3.png differ diff --git a/docs/my-website/img/kb_4.png b/docs/my-website/img/kb_4.png new file mode 100644 index 00000000000..7927a7f2e1d Binary files /dev/null and b/docs/my-website/img/kb_4.png differ diff --git a/docs/my-website/img/key_email.png b/docs/my-website/img/key_email.png new file mode 100644 index 00000000000..c4108b7a743 Binary files /dev/null and b/docs/my-website/img/key_email.png differ diff --git a/docs/my-website/img/key_email_2.png b/docs/my-website/img/key_email_2.png new file mode 100644 index 00000000000..d591ce03e8a Binary files /dev/null and b/docs/my-website/img/key_email_2.png differ diff --git a/docs/my-website/img/litellm_adk.png b/docs/my-website/img/litellm_adk.png new file mode 100644 index 00000000000..7d79b94f3b1 Binary files /dev/null and b/docs/my-website/img/litellm_adk.png differ diff --git a/docs/my-website/img/multi_instance_rate_limiting.png b/docs/my-website/img/multi_instance_rate_limiting.png new file mode 100644 index 00000000000..56e944ddbf1 Binary files /dev/null and b/docs/my-website/img/multi_instance_rate_limiting.png differ diff --git a/docs/my-website/img/new_user_email.png b/docs/my-website/img/new_user_email.png new file mode 100644 index 00000000000..1a4d44523b2 Binary files /dev/null and b/docs/my-website/img/new_user_email.png differ diff --git a/docs/my-website/img/pii_masking_v2.png b/docs/my-website/img/pii_masking_v2.png new file mode 100644 index 00000000000..597dc403fa6 Binary files /dev/null and b/docs/my-website/img/pii_masking_v2.png differ diff --git a/docs/my-website/img/presidio_1.png b/docs/my-website/img/presidio_1.png new file mode 100644 index 00000000000..6cc13cfacf2 Binary files /dev/null and b/docs/my-website/img/presidio_1.png differ diff --git a/docs/my-website/img/presidio_2.png b/docs/my-website/img/presidio_2.png new file mode 100644 index 00000000000..2bdab8821bd Binary files /dev/null and b/docs/my-website/img/presidio_2.png differ diff --git a/docs/my-website/img/presidio_3.png b/docs/my-website/img/presidio_3.png new file mode 100644 index 00000000000..7e6e0039d3a Binary files /dev/null and b/docs/my-website/img/presidio_3.png differ diff --git a/docs/my-website/img/presidio_4.png b/docs/my-website/img/presidio_4.png new file mode 100644 index 00000000000..b7732ba0fe1 Binary files /dev/null and b/docs/my-website/img/presidio_4.png differ diff --git a/docs/my-website/img/presidio_5.png b/docs/my-website/img/presidio_5.png new file mode 100644 index 00000000000..a0d903f8edc Binary files /dev/null and b/docs/my-website/img/presidio_5.png differ diff --git a/docs/my-website/img/release_notes/bedrock_kb.png b/docs/my-website/img/release_notes/bedrock_kb.png new file mode 100644 index 00000000000..86efa5ecb6c Binary files /dev/null and b/docs/my-website/img/release_notes/bedrock_kb.png differ diff --git a/docs/my-website/img/release_notes/lb_batch.png b/docs/my-website/img/release_notes/lb_batch.png new file mode 100644 index 00000000000..05e430ef49f Binary files /dev/null and b/docs/my-website/img/release_notes/lb_batch.png differ diff --git a/docs/my-website/img/spend_log_deletion_multi_pod.jpg b/docs/my-website/img/spend_log_deletion_multi_pod.jpg new file mode 100644 index 00000000000..52cf22c1a35 Binary files /dev/null and b/docs/my-website/img/spend_log_deletion_multi_pod.jpg differ diff --git a/docs/my-website/img/spend_log_deletion_working.png b/docs/my-website/img/spend_log_deletion_working.png new file mode 100644 index 00000000000..f0dca082611 Binary files /dev/null and b/docs/my-website/img/spend_log_deletion_working.png differ diff --git a/docs/my-website/package-lock.json b/docs/my-website/package-lock.json index e6f20d567bc..5c619ad2c28 100644 --- a/docs/my-website/package-lock.json +++ b/docs/my-website/package-lock.json @@ -21071,9 +21071,9 @@ } }, "node_modules/undici": { - "version": "6.21.1", - "resolved": "https://registry.npmjs.org/undici/-/undici-6.21.1.tgz", - "integrity": "sha512-q/1rj5D0/zayJB2FraXdaWxbhWiNKDvu8naDT2dl1yTlvJp4BLtOcp2a5BvgGNQpYYJzau7tf1WgKv3b+7mqpQ==", + "version": "6.21.3", + "resolved": "https://registry.npmjs.org/undici/-/undici-6.21.3.tgz", + "integrity": "sha512-gBLkYIlEnSp8pFbT64yFgGE6UIB9tAkhukC23PmMDCe5Nd+cRqKxSjw5y54MK2AZMgZfJWMaNE4nYUHgi1XEOw==", "license": "MIT", "engines": { "node": ">=18.17" diff --git a/docs/my-website/release_notes/v1.68.0-stable/index.md b/docs/my-website/release_notes/v1.68.0-stable/index.md new file mode 100644 index 00000000000..4d456d9c853 --- /dev/null +++ b/docs/my-website/release_notes/v1.68.0-stable/index.md @@ -0,0 +1,182 @@ +--- +title: v1.68.0-stable +slug: v1.68.0-stable +date: 2025-05-03T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1749686400&v=beta&t=Hkl3U8Ps0VtvNxX0BNNq24b4dtX5wQaPFp6oiKCIHD8 + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg + +hide_table_of_contents: false +--- +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + + + +## Deploy this version + + + + +``` showLineNumbers title="docker run litellm" +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.68.0-stable +``` + + + + +``` showLineNumbers title="pip install litellm" +pip install litellm==1.68.0.post1 +``` + + + +## Key Highlights + +LiteLLM v1.68.0-stable will be live soon. Here are the key highlights of this release: + +- **Bedrock Knowledge Base**: You can now call query your Bedrock Knowledge Base with all LiteLLM models via `/chat/completion` or `/responses` API. +- **Rate Limits**: This release brings accurate rate limiting across multiple instances, reducing spillover to at most 10 additional requests in high traffic. +- **Meta Llama API**: Added support for Meta Llama API [Get Started](https://docs.litellm.ai/docs/providers/meta_llama) +- **LlamaFile**: Added support for LlamaFile [Get Started](https://docs.litellm.ai/docs/providers/llamafile) + +## Bedrock Knowledge Base (Vector Store) + + +
+ +This release adds support for Bedrock vector stores (knowledge bases) in LiteLLM. With this update, you can: + +- Use Bedrock vector stores in the OpenAI /chat/completions spec with all LiteLLM supported models. +- View all available vector stores through the LiteLLM UI or API. +- Configure vector stores to be always active for specific models. +- Track vector store usage in LiteLLM Logs. + +For the next release we plan on allowing you to set key, user, team, org permissions for vector stores. + +[Read more here](https://docs.litellm.ai/docs/completion/knowledgebase) + +## Rate Limiting + + +
+ + +This release brings accurate multi-instance rate limiting across keys/users/teams. Outlining key engineering changes below: + +- **Change**: Instances now increment cache value instead of setting it. To avoid calling Redis on each request, this is synced every 0.01s. +- **Accuracy**: In testing, we saw a maximum spill over from expected of 10 requests, in high traffic (100 RPS, 3 instances), vs. current 189 request spillover +- **Performance**: Our load tests show this to reduce median response time by 100ms in high traffic  + +This is currently behind a feature flag, and we plan to have this be the default by next week. To enable this today, just add this environment variable: + +``` +export LITELLM_RATE_LIMIT_ACCURACY=true +``` + +[Read more here](../../docs/proxy/users#beta-multi-instance-rate-limiting) + + + +## New Models / Updated Models +- **Gemini ([VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) + [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))** + - Handle more json schema - openapi schema conversion edge cases [PR](https://github.com/BerriAI/litellm/pull/10351) + - Tool calls - return ‘finish_reason=“tool_calls”’ on gemini tool calling response [PR](https://github.com/BerriAI/litellm/pull/10485) +- **[VertexAI](../../docs/providers/vertex#metallama-api)** + - Meta/llama-4 model support [PR](https://github.com/BerriAI/litellm/pull/10492) + - Meta/llama3 - handle tool call result in content [PR](https://github.com/BerriAI/litellm/pull/10492) + - Meta/* - return ‘finish_reason=“tool_calls”’ on tool calling response [PR](https://github.com/BerriAI/litellm/pull/10492) +- **[Bedrock](../../docs/providers/bedrock#litellm-proxy-usage)** + - [Image Generation](../../docs/providers/bedrock#image-generation) - Support new ‘stable-image-core’ models - [PR](https://github.com/BerriAI/litellm/pull/10351) + - [Knowledge Bases](../../docs/completion/knowledgebase) - support using Bedrock knowledge bases with `/chat/completions` [PR](https://github.com/BerriAI/litellm/pull/10413) + - [Anthropic](../../docs/providers/bedrock#litellm-proxy-usage) - add ‘supports_pdf_input’ for claude-3.7-bedrock models [PR](https://github.com/BerriAI/litellm/pull/9917), [Get Started](../../docs/completion/document_understanding#checking-if-a-model-supports-pdf-input) +- **[OpenAI](../../docs/providers/openai)** + - Support OPENAI_BASE_URL in addition to OPENAI_API_BASE [PR](https://github.com/BerriAI/litellm/pull/10423) + - Correctly re-raise 504 timeout errors [PR](https://github.com/BerriAI/litellm/pull/10462) + - Native Gpt-4o-mini-tts support [PR](https://github.com/BerriAI/litellm/pull/10462) +- 🆕 **[Meta Llama API](../../docs/providers/meta_llama)** provider [PR](https://github.com/BerriAI/litellm/pull/10451) +- 🆕 **[LlamaFile](../../docs/providers/llamafile)** provider [PR](https://github.com/BerriAI/litellm/pull/10482) + +## LLM API Endpoints +- **[Response API](../../docs/response_api)** + - Fix for handling multi turn sessions [PR](https://github.com/BerriAI/litellm/pull/10415) +- **[Embeddings](../../docs/embedding/supported_embedding)** + - Caching fixes - [PR](https://github.com/BerriAI/litellm/pull/10424) + - handle str -> list cache + - Return usage tokens for cache hit + - Combine usage tokens on partial cache hits +- 🆕 **[Vector Stores](../../docs/completion/knowledgebase)** + - Allow defining Vector Store Configs - [PR](https://github.com/BerriAI/litellm/pull/10448) + - New StandardLoggingPayload field for requests made when a vector store is used - [PR](https://github.com/BerriAI/litellm/pull/10509) + - Show Vector Store / KB Request on LiteLLM Logs Page - [PR](https://github.com/BerriAI/litellm/pull/10514) + - Allow using vector store in OpenAI API spec with tools - [PR](https://github.com/BerriAI/litellm/pull/10516) +- **[MCP](../../docs/mcp)** + - Ensure Non-Admin virtual keys can access /mcp routes - [PR](https://github.com/BerriAI/litellm/pull/10473) + + **Note:** Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs. Follow [here](https://github.com/BerriAI/litellm/discussions/9891) for updates. +- **Moderations** + - Add logging callback support for `/moderations` API - [PR](https://github.com/BerriAI/litellm/pull/10390) + + +## Spend Tracking / Budget Improvements +- **[OpenAI](../../docs/providers/openai)** + - [computer-use-preview](../../docs/providers/openai/responses_api#computer-use) cost tracking / pricing [PR](https://github.com/BerriAI/litellm/pull/10422) + - [gpt-4o-mini-tts](../../docs/providers/openai/text_to_speech) input cost tracking - [PR](https://github.com/BerriAI/litellm/pull/10462) +- **[Fireworks AI](../../docs/providers/fireworks_ai)** - pricing updates - new `0-4b` model pricing tier + llama4 model pricing +- **[Budgets](../../docs/proxy/users#set-budgets)** + - [Budget resets](../../docs/proxy/users#reset-budgets) now happen as start of day/week/month - [PR](https://github.com/BerriAI/litellm/pull/10333) + - Trigger [Soft Budget Alerts](../../docs/proxy/alerting#soft-budget-alerts-for-virtual-keys) When Key Crosses Threshold - [PR](https://github.com/BerriAI/litellm/pull/10491) +- **[Token Counting](../../docs/completion/token_usage#3-token_counter)** + - Rewrite of token_counter() function to handle to prevent undercounting tokens - [PR](https://github.com/BerriAI/litellm/pull/10409) + + +## Management Endpoints / UI +- **Virtual Keys** + - Fix filtering on key alias - [PR](https://github.com/BerriAI/litellm/pull/10455) + - Support global filtering on keys - [PR](https://github.com/BerriAI/litellm/pull/10455) + - Pagination - fix clicking on next/back buttons on table - [PR](https://github.com/BerriAI/litellm/pull/10528) +- **Models** + - Triton - Support adding model/provider on UI - [PR](https://github.com/BerriAI/litellm/pull/10456) + - VertexAI - Fix adding vertex models with reusable credentials - [PR](https://github.com/BerriAI/litellm/pull/10528) + - LLM Credentials - show existing credentials for easy editing - [PR](https://github.com/BerriAI/litellm/pull/10519) +- **Teams** + - Allow reassigning team to other org - [PR](https://github.com/BerriAI/litellm/pull/10527) +- **Organizations** + - Fix showing org budget on table - [PR](https://github.com/BerriAI/litellm/pull/10528) + + + +## Logging / Guardrail Integrations +- **[Langsmith](../../docs/observability/langsmith_integration)** + - Respect [langsmith_batch_size](../../docs/observability/langsmith_integration#local-testing---control-batch-size) param - [PR](https://github.com/BerriAI/litellm/pull/10411) + +## Performance / Loadbalancing / Reliability improvements +- **[Redis](../../docs/proxy/caching)** + - Ensure all redis queues are periodically flushed, this fixes an issue where redis queue size was growing indefinitely when request tags were used - [PR](https://github.com/BerriAI/litellm/pull/10393) +- **[Rate Limits](../../docs/proxy/users#set-rate-limit)** + - [Multi-instance rate limiting](../../docs/proxy/users#beta-multi-instance-rate-limiting) support across keys/teams/users/customers - [PR](https://github.com/BerriAI/litellm/pull/10458), [PR](https://github.com/BerriAI/litellm/pull/10497), [PR](https://github.com/BerriAI/litellm/pull/10500) +- **[Azure OpenAI OIDC](../../docs/providers/azure#entra-id---use-azure_ad_token)** + - allow using litellm defined params for [OIDC Auth](../../docs/providers/azure#entra-id---use-azure_ad_token) - [PR](https://github.com/BerriAI/litellm/pull/10394) + + +## General Proxy Improvements +- **Security** + - Allow [blocking web crawlers](../../docs/proxy/enterprise#blocking-web-crawlers) - [PR](https://github.com/BerriAI/litellm/pull/10420) +- **Auth** + - Support [`x-litellm-api-key` header param by default](../../docs/pass_through/vertex_ai#use-with-virtual-keys), this fixes an issue from the prior release where `x-litellm-api-key` was not being used on vertex ai passthrough requests - [PR](https://github.com/BerriAI/litellm/pull/10392) + - Allow key at max budget to call non-llm api endpoints - [PR](https://github.com/BerriAI/litellm/pull/10392) +- 🆕 **[Python Client Library](../../docs/proxy/management_cli) for LiteLLM Proxy management endpoints** + - Initial PR - [PR](https://github.com/BerriAI/litellm/pull/10445) + - Support for doing HTTP requests - [PR](https://github.com/BerriAI/litellm/pull/10452) +- **Dependencies** + - Don’t require uvloop for windows - [PR](https://github.com/BerriAI/litellm/pull/10483) diff --git a/docs/my-website/release_notes/v1.69.0-stable/index.md b/docs/my-website/release_notes/v1.69.0-stable/index.md new file mode 100644 index 00000000000..3f8ce7a29c4 --- /dev/null +++ b/docs/my-website/release_notes/v1.69.0-stable/index.md @@ -0,0 +1,200 @@ +--- +title: v1.69.0-stable - Loadbalance Batch API Models +slug: v1.69.0-stable +date: 2025-05-10T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1749686400&v=beta&t=Hkl3U8Ps0VtvNxX0BNNq24b4dtX5wQaPFp6oiKCIHD8 + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg + +hide_table_of_contents: false +--- +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + + + +## Deploy this version + + + + +``` showLineNumbers title="docker run litellm" +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.69.0-stable +``` + + + + +``` showLineNumbers title="pip install litellm" +pip install litellm==1.69.0.post1 +``` + + + +## Key Highlights + +LiteLLM v1.69.0-stable brings the following key improvements: + +- **Loadbalance Batch API Models**: Easily loadbalance across multiple azure batch deployments using LiteLLM Managed Files +- **Email Invites 2.0**: Send new users onboarded to LiteLLM an email invite. +- **Nscale**: LLM API for compliance with European regulations. +- **Bedrock /v1/messages**: Use Bedrock Anthropic models with Anthropic's /v1/messages. + +## Batch API Load Balancing + + + + +This release brings LiteLLM Managed File support to Batches. This is great for: + +- Proxy Admins: You can now control which Batch models users can call. +- Developers: You no longer need to know the Azure deployment name when creating your batch .jsonl files - just specify the model your LiteLLM key has access to. + +Over time, we expect LiteLLM Managed Files to be the way most teams use Files across `/chat/completions`, `/batch`, `/fine_tuning` endpoints. + +[Read more here](https://docs.litellm.ai/docs/proxy/managed_batches) + + +## Email Invites + + + +This release brings the following improvements to our email invite integration: +- New templates for user invited and key created events. +- Fixes for using SMTP email providers. +- Native support for Resend API. +- Ability for Proxy Admins to control email events. + +For LiteLLM Cloud Users, please reach out to us if you want this enabled for your instance. + +[Read more here](https://docs.litellm.ai/docs/proxy/email) + + +## New Models / Updated Models +- **Gemini ([VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) + [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))** + - Added `gemini-2.5-pro-preview-05-06` models with pricing and context window info - [PR](https://github.com/BerriAI/litellm/pull/10597) + - Set correct context window length for all Gemini 2.5 variants - [PR](https://github.com/BerriAI/litellm/pull/10690) +- **[Perplexity](../../docs/providers/perplexity)**: + - Added new Perplexity models - [PR](https://github.com/BerriAI/litellm/pull/10652) + - Added sonar-deep-research model pricing - [PR](https://github.com/BerriAI/litellm/pull/10537) +- **[Azure OpenAI](../../docs/providers/azure)**: + - Fixed passing through of azure_ad_token_provider parameter - [PR](https://github.com/BerriAI/litellm/pull/10694) +- **[OpenAI](../../docs/providers/openai)**: + - Added support for pdf url's in 'file' parameter - [PR](https://github.com/BerriAI/litellm/pull/10640) +- **[Sagemaker](../../docs/providers/aws_sagemaker)**: + - Fix content length for `sagemaker_chat` provider - [PR](https://github.com/BerriAI/litellm/pull/10607) +- **[Azure AI Foundry](../../docs/providers/azure_ai)**: + - Added cost tracking for the following models [PR](https://github.com/BerriAI/litellm/pull/9956) + - DeepSeek V3 0324 + - Llama 4 Scout + - Llama 4 Maverick +- **[Bedrock](../../docs/providers/bedrock)**: + - Added cost tracking for Bedrock Llama 4 models - [PR](https://github.com/BerriAI/litellm/pull/10582) + - Fixed template conversion for Llama 4 models in Bedrock - [PR](https://github.com/BerriAI/litellm/pull/10582) + - Added support for using Bedrock Anthropic models with /v1/messages format - [PR](https://github.com/BerriAI/litellm/pull/10681) + - Added streaming support for Bedrock Anthropic models with /v1/messages format - [PR](https://github.com/BerriAI/litellm/pull/10710) +- **[OpenAI](../../docs/providers/openai)**: Added `reasoning_effort` support for `o3` models - [PR](https://github.com/BerriAI/litellm/pull/10591) +- **[Databricks](../../docs/providers/databricks)**: + - Fixed issue when Databricks uses external model and delta could be empty - [PR](https://github.com/BerriAI/litellm/pull/10540) +- **[Cerebras](../../docs/providers/cerebras)**: Fixed Llama-3.1-70b model pricing and context window - [PR](https://github.com/BerriAI/litellm/pull/10648) +- **[Ollama](../../docs/providers/ollama)**: + - Fixed custom price cost tracking and added 'max_completion_token' support - [PR](https://github.com/BerriAI/litellm/pull/10636) + - Fixed KeyError when using JSON response format - [PR](https://github.com/BerriAI/litellm/pull/10611) +- 🆕 **[Nscale](../../docs/providers/nscale)**: + - Added support for chat, image generation endpoints - [PR](https://github.com/BerriAI/litellm/pull/10638) + +## LLM API Endpoints +- **[Messages API](../../docs/anthropic_unified)**: + - 🆕 Added support for using Bedrock Anthropic models with /v1/messages format - [PR](https://github.com/BerriAI/litellm/pull/10681) and streaming support - [PR](https://github.com/BerriAI/litellm/pull/10710) +- **[Moderations API](../../docs/moderations)**: + - Fixed bug to allow using LiteLLM UI credentials for /moderations API - [PR](https://github.com/BerriAI/litellm/pull/10723) +- **[Realtime API](../../docs/realtime)**: + - Fixed setting 'headers' in scope for websocket auth requests and infinite loop issues - [PR](https://github.com/BerriAI/litellm/pull/10679) +- **[Files API](../../docs/proxy/litellm_managed_files)**: + - Unified File ID output support - [PR](https://github.com/BerriAI/litellm/pull/10713) + - Support for writing files to all deployments - [PR](https://github.com/BerriAI/litellm/pull/10708) + - Added target model name validation - [PR](https://github.com/BerriAI/litellm/pull/10722) +- **[Batches API](../../docs/batches)**: + - Complete unified batch ID support - replacing model in jsonl to be deployment model name - [PR](https://github.com/BerriAI/litellm/pull/10719) + - Beta support for unified file ID (managed files) for batches - [PR](https://github.com/BerriAI/litellm/pull/10650) + + +## Spend Tracking / Budget Improvements +- Bug Fix - PostgreSQL Integer Overflow Error in DB Spend Tracking - [PR](https://github.com/BerriAI/litellm/pull/10697) + +## Management Endpoints / UI +- **Models** + - Fixed model info overwriting when editing a model on UI - [PR](https://github.com/BerriAI/litellm/pull/10726) + - Fixed team admin model updates and organization creation with specific models - [PR](https://github.com/BerriAI/litellm/pull/10539) +- **Logs**: + - Bug Fix - copying Request/Response on Logs Page - [PR](https://github.com/BerriAI/litellm/pull/10720) + - Bug Fix - log did not remain in focus on QA Logs page + text overflow on error logs - [PR](https://github.com/BerriAI/litellm/pull/10725) + - Added index for session_id on LiteLLM_SpendLogs for better query performance - [PR](https://github.com/BerriAI/litellm/pull/10727) +- **User Management**: + - Added user management functionality to Python client library & CLI - [PR](https://github.com/BerriAI/litellm/pull/10627) + - Bug Fix - Fixed SCIM token creation on Admin UI - [PR](https://github.com/BerriAI/litellm/pull/10628) + - Bug Fix - Added 404 response when trying to delete verification tokens that don't exist - [PR](https://github.com/BerriAI/litellm/pull/10605) + +## Logging / Guardrail Integrations +- **Custom Logger API**: v2 Custom Callback API (send llm logs to custom api) - [PR](https://github.com/BerriAI/litellm/pull/10575), [Get Started](https://docs.litellm.ai/docs/proxy/logging#custom-callback-apis-async) +- **OpenTelemetry**: + - Fixed OpenTelemetry to follow genai semantic conventions + support for 'instructions' param for TTS - [PR](https://github.com/BerriAI/litellm/pull/10608) +- ** Bedrock PII**: + - Add support for PII Masking with bedrock guardrails - [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/bedrock#pii-masking-with-bedrock-guardrails), [PR](https://github.com/BerriAI/litellm/pull/10608) +- **Documentation**: + - Added documentation for StandardLoggingVectorStoreRequest - [PR](https://github.com/BerriAI/litellm/pull/10535) + +## Performance / Reliability Improvements +- **Python Compatibility**: + - Added support for Python 3.11- (fixed datetime UTC handling) - [PR](https://github.com/BerriAI/litellm/pull/10701) + - Fixed UnicodeDecodeError: 'charmap' on Windows during litellm import - [PR](https://github.com/BerriAI/litellm/pull/10542) +- **Caching**: + - Fixed embedding string caching result - [PR](https://github.com/BerriAI/litellm/pull/10700) + - Fixed cache miss for Gemini models with response_format - [PR](https://github.com/BerriAI/litellm/pull/10635) + +## General Proxy Improvements +- **Proxy CLI**: + - Added `--version` flag to `litellm-proxy` CLI - [PR](https://github.com/BerriAI/litellm/pull/10704) + - Added dedicated `litellm-proxy` CLI - [PR](https://github.com/BerriAI/litellm/pull/10578) +- **Alerting**: + - Fixed Slack alerting not working when using a DB - [PR](https://github.com/BerriAI/litellm/pull/10370) +- **Email Invites**: + - Added V2 Emails with fixes for sending emails when creating keys + Resend API support - [PR](https://github.com/BerriAI/litellm/pull/10602) + - Added user invitation emails - [PR](https://github.com/BerriAI/litellm/pull/10615) + - Added endpoints to manage email settings - [PR](https://github.com/BerriAI/litellm/pull/10646) +- **General**: + - Fixed bug where duplicate JSON logs were getting emitted - [PR](https://github.com/BerriAI/litellm/pull/10580) + + +## New Contributors +- [@zoltan-ongithub](https://github.com/zoltan-ongithub) made their first contribution in [PR #10568](https://github.com/BerriAI/litellm/pull/10568) +- [@mkavinkumar1](https://github.com/mkavinkumar1) made their first contribution in [PR #10548](https://github.com/BerriAI/litellm/pull/10548) +- [@thomelane](https://github.com/thomelane) made their first contribution in [PR #10549](https://github.com/BerriAI/litellm/pull/10549) +- [@frankzye](https://github.com/frankzye) made their first contribution in [PR #10540](https://github.com/BerriAI/litellm/pull/10540) +- [@aholmberg](https://github.com/aholmberg) made their first contribution in [PR #10591](https://github.com/BerriAI/litellm/pull/10591) +- [@aravindkarnam](https://github.com/aravindkarnam) made their first contribution in [PR #10611](https://github.com/BerriAI/litellm/pull/10611) +- [@xsg22](https://github.com/xsg22) made their first contribution in [PR #10648](https://github.com/BerriAI/litellm/pull/10648) +- [@casparhsws](https://github.com/casparhsws) made their first contribution in [PR #10635](https://github.com/BerriAI/litellm/pull/10635) +- [@hypermoose](https://github.com/hypermoose) made their first contribution in [PR #10370](https://github.com/BerriAI/litellm/pull/10370) +- [@tomukmatthews](https://github.com/tomukmatthews) made their first contribution in [PR #10638](https://github.com/BerriAI/litellm/pull/10638) +- [@keyute](https://github.com/keyute) made their first contribution in [PR #10652](https://github.com/BerriAI/litellm/pull/10652) +- [@GPTLocalhost](https://github.com/GPTLocalhost) made their first contribution in [PR #10687](https://github.com/BerriAI/litellm/pull/10687) +- [@husnain7766](https://github.com/husnain7766) made their first contribution in [PR #10697](https://github.com/BerriAI/litellm/pull/10697) +- [@claralp](https://github.com/claralp) made their first contribution in [PR #10694](https://github.com/BerriAI/litellm/pull/10694) +- [@mollux](https://github.com/mollux) made their first contribution in [PR #10690](https://github.com/BerriAI/litellm/pull/10690) diff --git a/docs/my-website/release_notes/v1.70.1-stable/index.md b/docs/my-website/release_notes/v1.70.1-stable/index.md new file mode 100644 index 00000000000..c55ac8b9c61 --- /dev/null +++ b/docs/my-website/release_notes/v1.70.1-stable/index.md @@ -0,0 +1,248 @@ +--- +title: v1.70.1-stable - Gemini Realtime API Support +slug: v1.70.1-stable +date: 2025-05-17T10:00:00 +authors: + - name: Krrish Dholakia + title: CEO, LiteLLM + url: https://www.linkedin.com/in/krish-d/ + image_url: https://media.licdn.com/dms/image/v2/D4D03AQGrlsJ3aqpHmQ/profile-displayphoto-shrink_400_400/B4DZSAzgP7HYAg-/0/1737327772964?e=1749686400&v=beta&t=Hkl3U8Ps0VtvNxX0BNNq24b4dtX5wQaPFp6oiKCIHD8 + - name: Ishaan Jaffer + title: CTO, LiteLLM + url: https://www.linkedin.com/in/reffajnaahsi/ + image_url: https://pbs.twimg.com/profile_images/1613813310264340481/lz54oEiB_400x400.jpg + +hide_table_of_contents: false +--- + +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + + + +## Deploy this version + + + + +``` showLineNumbers title="docker run litellm" +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.70.1-stable +``` + + + + +``` showLineNumbers title="pip install litellm" +pip install litellm==1.70.1 +``` + + + + +## Key Highlights + +LiteLLM v1.70.1-stable is live now. Here are the key highlights of this release: + +- **Gemini Realtime API**: You can now call Gemini's Live API via the OpenAI /v1/realtime API +- **Spend Logs Retention Period**: Enable deleting spend logs older than a certain period. +- **PII Masking 2.0**: Easily configure masking or blocking specific PII/PHI entities on the UI + +## Gemini Realtime API + + + + +This release brings support for calling Gemini's realtime models (e.g. gemini-2.0-flash-live) via OpenAI's /v1/realtime API. This is great for developers as it lets them easily switch from OpenAI to Gemini by just changing the model name. + +Key Highlights: +- Support for text + audio input/output +- Support for setting session configurations (modality, instructions, activity detection) in the OpenAI format +- Support for logging + usage tracking for realtime sessions + +This is currently supported via Google AI Studio. We plan to release VertexAI support over the coming week. + +[**Read more**](../../docs/providers/google_ai_studio/realtime) + +## Spend Logs Retention Period + + + + + +This release enables deleting LiteLLM Spend Logs older than a certain period. Since we now enable storing the raw request/response in the logs, deleting old logs ensures the database remains performant in production. + +[**Read more**](../../docs/proxy/spend_logs_deletion) + +## PII Masking 2.0 + + + +This release brings improvements to our Presidio PII Integration. As a Proxy Admin, you now have the ability to: + +- Mask or block specific entities (e.g., block medical licenses while masking other entities like emails). +- Monitor guardrails in production. LiteLLM Logs will now show you the guardrail run, the entities it detected, and its confidence score for each entity. + +[**Read more**](../../docs/proxy/guardrails/pii_masking_v2) + +## New Models / Updated Models + +- **Gemini ([VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) + [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))** + - `/chat/completion` + - Handle audio input - [PR](https://github.com/BerriAI/litellm/pull/10739) + - Fixes maximum recursion depth issue when using deeply nested response schemas with Vertex AI by Increasing DEFAULT_MAX_RECURSE_DEPTH from 10 to 100 in constants. [PR](https://github.com/BerriAI/litellm/pull/10798) + - Capture reasoning tokens in streaming mode - [PR](https://github.com/BerriAI/litellm/pull/10789) +- **[Google AI Studio](../../docs/providers/google_ai_studio/realtime)** + - `/realtime` + - Gemini Multimodal Live API support + - Audio input/output support, optional param mapping, accurate usage calculation - [PR](https://github.com/BerriAI/litellm/pull/10909) +- **[VertexAI](../../docs/providers/vertex#metallama-api)** + - `/chat/completion` + - Fix llama streaming error - where model response was nested in returned streaming chunk - [PR](https://github.com/BerriAI/litellm/pull/10878) +- **[Ollama](../../docs/providers/ollama)** + - `/chat/completion` + - structure responses fix - [PR](https://github.com/BerriAI/litellm/pull/10617) +- **[Bedrock](../../docs/providers/bedrock#litellm-proxy-usage)** + - [`/chat/completion`](../../docs/providers/bedrock#litellm-proxy-usage) + - Handle thinking_blocks when assistant.content is None - [PR](https://github.com/BerriAI/litellm/pull/10688) + - Fixes to only allow accepted fields for tool json schema - [PR](https://github.com/BerriAI/litellm/pull/10062) + - Add bedrock sonnet prompt caching cost information + - Mistral Pixtral support - [PR](https://github.com/BerriAI/litellm/pull/10439) + - Tool caching support - [PR](https://github.com/BerriAI/litellm/pull/10897) + - [`/messages`](../../docs/anthropic_unified) + - allow using dynamic AWS Params - [PR](https://github.com/BerriAI/litellm/pull/10769) +- **[Nvidia NIM](../../docs/providers/nvidia_nim)** + - [`/chat/completion`](../../docs/providers/nvidia_nim#usage---litellm-proxy-server) + - Add tools, tool_choice, parallel_tool_calls support - [PR](https://github.com/BerriAI/litellm/pull/10763) +- **[Novita AI](../../docs/providers/novita)** + - New Provider added for `/chat/completion` routes - [PR](https://github.com/BerriAI/litellm/pull/9527) +- **[Azure](../../docs/providers/azure)** + - [`/image/generation`](../../docs/providers/azure#image-generation) + - Fix azure dall e 3 call with custom model name - [PR](https://github.com/BerriAI/litellm/pull/10776) +- **[Cohere](../../docs/providers/cohere)** + - [`/embeddings`](../../docs/providers/cohere#embedding) + - Migrate embedding to use `/v2/embed` - adds support for output_dimensions param - [PR](https://github.com/BerriAI/litellm/pull/10809) +- **[Anthropic](../../docs/providers/anthropic)** + - [`/chat/completion`](../../docs/providers/anthropic#usage-with-litellm-proxy) + - Web search tool support - native + openai format - [Get Started](../../docs/providers/anthropic#anthropic-hosted-tools-computer-text-editor-web-search) +- **[VLLM](../../docs/providers/vllm)** + - [`/embeddings`](../../docs/providers/vllm#embeddings) + - Support embedding input as list of integers +- **[OpenAI](../../docs/providers/openai)** + - [`/chat/completion`](../../docs/providers/openai#usage---litellm-proxy-server) + - Fix - b64 file data input handling - [Get Started](../../docs/providers/openai#pdf-file-parsing) + - Add ‘supports_pdf_input’ to all vision models - [PR](https://github.com/BerriAI/litellm/pull/10897) + +## LLM API Endpoints +- [**Responses API**](../../docs/response_api) + - Fix delete API support - [PR](https://github.com/BerriAI/litellm/pull/10845) +- [**Rerank API**](../../docs/rerank) + - `/v2/rerank` now registered as ‘llm_api_route’ - enabling non-admins to call it - [PR](https://github.com/BerriAI/litellm/pull/10861) + +## Spend Tracking Improvements +- **`/chat/completion`, `/messages`** + - Anthropic - web search tool cost tracking - [PR](https://github.com/BerriAI/litellm/pull/10846) + - Groq - update model max tokens + cost information - [PR](https://github.com/BerriAI/litellm/pull/10077) +- **`/audio/transcription`** + - Azure - Add gpt-4o-mini-tts pricing - [PR](https://github.com/BerriAI/litellm/pull/10807) + - Proxy - Fix tracking spend by tag - [PR](https://github.com/BerriAI/litellm/pull/10832) +- **`/embeddings`** + - Azure AI - Add cohere embed v4 pricing - [PR](https://github.com/BerriAI/litellm/pull/10806) + +## Management Endpoints / UI +- **Models** + - Ollama - adds api base param to UI +- **Logs** + - Add team id, key alias, key hash filter on logs - https://github.com/BerriAI/litellm/pull/10831 + - Guardrail tracing now in Logs UI - https://github.com/BerriAI/litellm/pull/10893 +- **Teams** + - Patch for updating team info when team in org and members not in org - https://github.com/BerriAI/litellm/pull/10835 +- **Guardrails** + - Add Bedrock, Presidio, Lakers guardrails on UI - https://github.com/BerriAI/litellm/pull/10874 + - See guardrail info page - https://github.com/BerriAI/litellm/pull/10904 + - Allow editing guardrails on UI - https://github.com/BerriAI/litellm/pull/10907 +- **Test Key** + - select guardrails to test on UI + + + +## Logging / Alerting Integrations +- **[StandardLoggingPayload](../../docs/proxy/logging_spec)** + - Log any `x-` headers in requester metadata - [Get Started](../../docs/proxy/logging_spec#standardloggingmetadata) + - Guardrail tracing now in standard logging payload - [Get Started](../../docs/proxy/logging_spec#standardloggingguardrailinformation) +- **[Generic API Logger](../../docs/proxy/logging#custom-callback-apis-async)** + - Support passing application/json header +- **[Arize Phoenix](../../docs/observability/phoenix_integration)** + - fix: URL encode OTEL_EXPORTER_OTLP_TRACES_HEADERS for Phoenix Integration - [PR](https://github.com/BerriAI/litellm/pull/10654) + - add guardrail tracing to OTEL, Arize phoenix - [PR](https://github.com/BerriAI/litellm/pull/10896) +- **[PagerDuty](../../docs/proxy/pagerduty)** + - Pagerduty is now a free feature - [PR](https://github.com/BerriAI/litellm/pull/10857) +- **[Alerting](../../docs/proxy/alerting)** + - Sending slack alerts on virtual key/user/team updates is now free - [PR](https://github.com/BerriAI/litellm/pull/10863) + + +## Guardrails +- **Guardrails** + - New `/apply_guardrail` endpoint for directly testing a guardrail - [PR](https://github.com/BerriAI/litellm/pull/10867) +- **[Lakera](../../docs/proxy/guardrails/lakera_ai)** + - `/v2` endpoints support - [PR](https://github.com/BerriAI/litellm/pull/10880) +- **[Presidio](../../docs/proxy/guardrails/pii_masking_v2)** + - Fixes handling of message content on presidio guardrail integration - [PR](https://github.com/BerriAI/litellm/pull/10197) + - Allow specifying PII Entities Config - [PR](https://github.com/BerriAI/litellm/pull/10810) +- **[Aim Security](../../docs/proxy/guardrails/aim_security)** + - Support for anonymization in AIM Guardrails - [PR](https://github.com/BerriAI/litellm/pull/10757) + + + +## Performance / Loadbalancing / Reliability improvements +- **Allow overriding all constants using a .env variable** - [PR](https://github.com/BerriAI/litellm/pull/10803) +- **[Maximum retention period for spend logs](../../docs/proxy/spend_logs_deletion)** + - Add retention flag to config - [PR](https://github.com/BerriAI/litellm/pull/10815) + - Support for cleaning up logs based on configured time period - [PR](https://github.com/BerriAI/litellm/pull/10872) + +## General Proxy Improvements +- **Authentication** + - Handle Bearer $LITELLM_API_KEY in x-litellm-api-key custom header [PR](https://github.com/BerriAI/litellm/pull/10776) +- **New Enterprise pip package** - `litellm-enterprise` - fixes issue where `enterprise` folder was not found when using pip package +- **[Proxy CLI](../../docs/proxy/management_cli)** + - Add `models import` command - [PR](https://github.com/BerriAI/litellm/pull/10581) +- **[OpenWebUI](../../docs/tutorials/openweb_ui#per-user-tracking)** + - Configure LiteLLM to Parse User Headers from Open Web UI +- **[LiteLLM Proxy w/ LiteLLM SDK](../../docs/providers/litellm_proxy#send-all-sdk-requests-to-litellm-proxy)** + - Option to force/always use the litellm proxy when calling via LiteLLM SDK + + +## New Contributors +* [@imdigitalashish](https://github.com/imdigitalashish) made their first contribution in PR [#10617](https://github.com/BerriAI/litellm/pull/10617) +* [@LouisShark](https://github.com/LouisShark) made their first contribution in PR [#10688](https://github.com/BerriAI/litellm/pull/10688) +* [@OscarSavNS](https://github.com/OscarSavNS) made their first contribution in PR [#10764](https://github.com/BerriAI/litellm/pull/10764) +* [@arizedatngo](https://github.com/arizedatngo) made their first contribution in PR [#10654](https://github.com/BerriAI/litellm/pull/10654) +* [@jugaldb](https://github.com/jugaldb) made their first contribution in PR [#10805](https://github.com/BerriAI/litellm/pull/10805) +* [@daikeren](https://github.com/daikeren) made their first contribution in PR [#10781](https://github.com/BerriAI/litellm/pull/10781) +* [@naliotopier](https://github.com/naliotopier) made their first contribution in PR [#10077](https://github.com/BerriAI/litellm/pull/10077) +* [@damienpontifex](https://github.com/damienpontifex) made their first contribution in PR [#10813](https://github.com/BerriAI/litellm/pull/10813) +* [@Dima-Mediator](https://github.com/Dima-Mediator) made their first contribution in PR [#10789](https://github.com/BerriAI/litellm/pull/10789) +* [@igtm](https://github.com/igtm) made their first contribution in PR [#10814](https://github.com/BerriAI/litellm/pull/10814) +* [@shibaboy](https://github.com/shibaboy) made their first contribution in PR [#10752](https://github.com/BerriAI/litellm/pull/10752) +* [@camfarineau](https://github.com/camfarineau) made their first contribution in PR [#10629](https://github.com/BerriAI/litellm/pull/10629) +* [@ajac-zero](https://github.com/ajac-zero) made their first contribution in PR [#10439](https://github.com/BerriAI/litellm/pull/10439) +* [@damgem](https://github.com/damgem) made their first contribution in PR [#9802](https://github.com/BerriAI/litellm/pull/9802) +* [@hxdror](https://github.com/hxdror) made their first contribution in PR [#10757](https://github.com/BerriAI/litellm/pull/10757) +* [@wwwillchen](https://github.com/wwwillchen) made their first contribution in PR [#10894](https://github.com/BerriAI/litellm/pull/10894) + + +## Demo Instance + +Here's a Demo Instance to test changes: + +- Instance: https://demo.litellm.ai/ +- Login Credentials: + - Username: admin + - Password: sk-1234 + + +## [Git Diff](https://github.com/BerriAI/litellm/releases) + diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index e278b36ed74..59bf42c9302 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -18,6 +18,7 @@ const sidebars = { // But you can create a sidebar manually tutorialSidebar: [ { type: "doc", id: "index" }, // NEW + { type: "category", label: "LiteLLM Proxy Server", @@ -53,7 +54,7 @@ const sidebars = { { type: "category", label: "Architecture", - items: ["proxy/architecture", "proxy/db_info", "proxy/db_deadlocks", "router_architecture", "proxy/user_management_heirarchy", "proxy/jwt_auth_arch", "proxy/image_handling"], + items: ["proxy/architecture", "proxy/db_info", "proxy/db_deadlocks", "router_architecture", "proxy/user_management_heirarchy", "proxy/jwt_auth_arch", "proxy/image_handling", "proxy/spend_logs_deletion"], }, { type: "link", @@ -61,6 +62,7 @@ const sidebars = { href: "https://litellm-api.up.railway.app/", }, "proxy/enterprise", + "proxy/management_cli", { type: "category", label: "Making LLM Requests", @@ -179,113 +181,6 @@ const sidebars = { "proxy/caching", ] }, - { - type: "category", - label: "Supported Models & Providers", - link: { - type: "generated-index", - title: "Providers", - description: - "Learn how to deploy + call models from different providers on LiteLLM", - slug: "/providers", - }, - items: [ - "providers/openai", - "providers/text_completion_openai", - "providers/openai_compatible", - "providers/azure", - "providers/azure_ai", - "providers/aiml", - "providers/vertex", - - { - type: "category", - label: "Google AI Studio", - items: [ - "providers/gemini", - "providers/google_ai_studio/files", - ] - }, - "providers/anthropic", - "providers/aws_sagemaker", - "providers/bedrock", - "providers/litellm_proxy", - "providers/mistral", - "providers/codestral", - "providers/cohere", - "providers/anyscale", - "providers/huggingface", - "providers/databricks", - "providers/deepgram", - "providers/watsonx", - "providers/predibase", - "providers/nvidia_nim", - "providers/xai", - "providers/lm_studio", - "providers/cerebras", - "providers/volcano", - "providers/triton-inference-server", - "providers/ollama", - "providers/perplexity", - "providers/friendliai", - "providers/galadriel", - "providers/topaz", - "providers/groq", - "providers/github", - "providers/deepseek", - "providers/fireworks_ai", - "providers/clarifai", - "providers/vllm", - "providers/llamafile", - "providers/infinity", - "providers/xinference", - "providers/cloudflare_workers", - "providers/deepinfra", - "providers/ai21", - "providers/nlp_cloud", - "providers/replicate", - "providers/togetherai", - "providers/voyage", - "providers/jina_ai", - "providers/aleph_alpha", - "providers/baseten", - "providers/openrouter", - "providers/sambanova", - "providers/custom_llm_server", - "providers/petals", - "providers/snowflake" - ], - }, - { - type: "category", - label: "Guides", - items: [ - "exception_mapping", - "completion/provider_specific_params", - "guides/finetuned_models", - "guides/security_settings", - "completion/audio", - "completion/web_search", - "completion/document_understanding", - "completion/vision", - "completion/json_mode", - "reasoning_content", - "completion/prompt_caching", - "completion/predict_outputs", - "completion/knowledgebase", - "completion/prefix", - "completion/drop_params", - "completion/prompt_formatting", - "completion/stream", - "completion/message_trimming", - "completion/function_call", - "completion/model_alias", - "completion/batching", - "completion/mock_requests", - "completion/reliable_completions", - - ] - }, { type: "category", label: "Supported Endpoints", @@ -362,12 +257,154 @@ const sidebars = { "proxy/litellm_managed_files", ], }, - "batches", + { + type: "category", + label: "/batches", + items: [ + "batches", + "proxy/managed_batches", + ] + }, "realtime", "fine_tuning", "moderation", + "apply_guardrail", ], }, + { + type: "category", + label: "Supported Models & Providers", + link: { + type: "generated-index", + title: "Providers", + description: + "Learn how to deploy + call models from different providers on LiteLLM", + slug: "/providers", + }, + items: [ + { + type: "category", + label: "OpenAI", + items: [ + "providers/openai", + "providers/openai/responses_api", + "providers/openai/text_to_speech", + ] + }, + "providers/text_completion_openai", + "providers/openai_compatible", + { + type: "category", + label: "Azure OpenAI", + items: [ + "providers/azure/azure", + "providers/azure/azure_embedding", + ] + }, + "providers/azure_ai", + "providers/aiml", + "providers/vertex", + { + type: "category", + label: "Google AI Studio", + items: [ + "providers/gemini", + "providers/google_ai_studio/files", + "providers/google_ai_studio/realtime", + ] + }, + "providers/anthropic", + "providers/aws_sagemaker", + { + type: "category", + label: "Bedrock", + items: [ + "providers/bedrock", + "providers/bedrock_vector_store", + ] + }, + "providers/litellm_proxy", + "providers/meta_llama", + "providers/mistral", + "providers/codestral", + "providers/cohere", + "providers/anyscale", + "providers/huggingface", + "providers/databricks", + "providers/deepgram", + "providers/watsonx", + "providers/predibase", + "providers/nvidia_nim", + { type: "doc", id: "providers/nscale", label: "Nscale (EU Sovereign)" }, + "providers/xai", + "providers/lm_studio", + "providers/cerebras", + "providers/volcano", + "providers/triton-inference-server", + "providers/ollama", + "providers/perplexity", + "providers/friendliai", + "providers/galadriel", + "providers/topaz", + "providers/groq", + "providers/github", + "providers/deepseek", + "providers/fireworks_ai", + "providers/clarifai", + "providers/vllm", + "providers/llamafile", + "providers/infinity", + "providers/xinference", + "providers/cloudflare_workers", + "providers/deepinfra", + "providers/ai21", + "providers/nlp_cloud", + "providers/replicate", + "providers/togetherai", + "providers/novita", + "providers/voyage", + "providers/jina_ai", + "providers/aleph_alpha", + "providers/baseten", + "providers/openrouter", + "providers/sambanova", + "providers/custom_llm_server", + "providers/petals", + "providers/snowflake", + "providers/featherless_ai" + ], + }, + { + type: "category", + label: "Guides", + items: [ + "exception_mapping", + "completion/provider_specific_params", + "guides/finetuned_models", + "guides/security_settings", + "completion/audio", + "completion/web_search", + "completion/document_understanding", + "completion/vision", + "completion/json_mode", + "reasoning_content", + "completion/prompt_caching", + "completion/predict_outputs", + "completion/knowledgebase", + "completion/prefix", + "completion/drop_params", + "completion/prompt_formatting", + "completion/stream", + "completion/message_trimming", + "completion/function_call", + "completion/model_alias", + "completion/batching", + "completion/mock_requests", + "completion/reliable_completions", + + ] + }, + { type: "category", label: "Routing, Loadbalancing & Fallbacks", @@ -462,11 +499,12 @@ const sidebars = { "tutorials/prompt_caching", "tutorials/tag_management", 'tutorials/litellm_proxy_aporia', + "tutorials/gemini_realtime_with_audio", { type: "category", label: "LiteLLM Python SDK Tutorials", items: [ - + 'tutorials/google_adk', 'tutorials/azure_openai', 'tutorials/instructor', "tutorials/gradio_integration", @@ -534,9 +572,9 @@ const sidebars = { "projects/LiteLLM Proxy", "projects/llm_cord", "projects/pgai", + "projects/GPTLocalhost", ], }, - "proxy/pii_masking", "extras/code_quality", "rules", "proxy/team_based_routing", diff --git a/docs/my-website/src/pages/completion/input.md b/docs/my-website/src/pages/completion/input.md index 86546bbbaef..ff9a3f0f0a5 100644 --- a/docs/my-website/src/pages/completion/input.md +++ b/docs/my-website/src/pages/completion/input.md @@ -1,6 +1,6 @@ # Completion Function - completion() The Input params are **exactly the same** as the -OpenAI Create chat completion, and let you call **Azure OpenAI, Anthropic, Cohere, Replicate, OpenRouter** models in the same format. +OpenAI Create chat completion, and let you call **Azure OpenAI, Anthropic, Cohere, Replicate, OpenRouter, Novita AI** models in the same format. In addition, liteLLM allows you to pass in the following **Optional** liteLLM args: `force_timeout`, `azure`, `logger_fn`, `verbose` diff --git a/docs/my-website/src/pages/completion/supported.md b/docs/my-website/src/pages/completion/supported.md index 2599353aa3f..097af2bb4cb 100644 --- a/docs/my-website/src/pages/completion/supported.md +++ b/docs/my-website/src/pages/completion/supported.md @@ -70,4 +70,28 @@ All the text models from [OpenRouter](https://openrouter.ai/docs) are supported | google/palm-2-chat-bison | `completion('google/palm-2-chat-bison', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` | | google/palm-2-codechat-bison | `completion('google/palm-2-codechat-bison', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` | | meta-llama/llama-2-13b-chat | `completion('meta-llama/llama-2-13b-chat', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` | -| meta-llama/llama-2-70b-chat | `completion('meta-llama/llama-2-70b-chat', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` | \ No newline at end of file +| meta-llama/llama-2-70b-chat | `completion('meta-llama/llama-2-70b-chat', messages)` | `os.environ['OR_SITE_URL']`,`os.environ['OR_APP_NAME']`,`os.environ['OR_API_KEY']` | + +## Novita AI Completion Models + +🚨 LiteLLM supports ALL Novita AI models, send `model=novita/` to send it to Novita AI. See all Novita AI models [here](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) + +| Model Name | Function Call | Required OS Variables | +|------------------|--------------------------------------------|--------------------------------------| +| novita/deepseek/deepseek-r1 | `completion('novita/deepseek/deepseek-r1', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/deepseek/deepseek_v3 | `completion('novita/deepseek/deepseek_v3', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.3-70b-instruct | `completion('novita/meta-llama/llama-3.3-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.1-8b-instruct | `completion('novita/meta-llama/llama-3.1-8b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.1-8b-instruct-max | `completion('novita/meta-llama/llama-3.1-8b-instruct-max', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.1-70b-instruct | `completion('novita/meta-llama/llama-3.1-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3-8b-instruct | `completion('novita/meta-llama/llama-3-8b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3-70b-instruct | `completion('novita/meta-llama/llama-3-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.2-1b-instruct | `completion('novita/meta-llama/llama-3.2-1b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.2-11b-vision-instruct | `completion('novita/meta-llama/llama-3.2-11b-vision-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.2-3b-instruct | `completion('novita/meta-llama/llama-3.2-3b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/gryphe/mythomax-l2-13b | `completion('novita/gryphe/mythomax-l2-13b', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/google/gemma-2-9b-it | `completion('novita/google/gemma-2-9b-it', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/mistralai/mistral-nemo | `completion('novita/mistralai/mistral-nemo', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/mistralai/mistral-7b-instruct | `completion('novita/mistralai/mistral-7b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/qwen/qwen-2.5-72b-instruct | `completion('novita/qwen/qwen-2.5-72b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/qwen/qwen-2-vl-72b-instruct | `completion('novita/qwen/qwen-2-vl-72b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | \ No newline at end of file diff --git a/docs/my-website/src/pages/index.md b/docs/my-website/src/pages/index.md index 4a2e5203e31..2c89d28a626 100644 --- a/docs/my-website/src/pages/index.md +++ b/docs/my-website/src/pages/index.md @@ -194,6 +194,22 @@ response = completion( ) ``` +
+ + +```python +from litellm import completion +import os + +## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key +os.environ["NOVITA_API_KEY"] = "novita-api-key" + +response = completion( + model="novita/deepseek/deepseek-r1", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) +``` +
@@ -347,7 +363,23 @@ response = completion( ``` + +```python +from litellm import completion +import os + +## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key +os.environ["NOVITA_API_KEY"] = "novita_api_key" + +response = completion( + model="novita/deepseek/deepseek-r1", + messages = [{ "content": "Hello, how are you?","role": "user"}], + stream=True, +) +``` + + ### Exception handling diff --git a/docs/my-website/static/llms-full.txt b/docs/my-website/static/llms-full.txt new file mode 100644 index 00000000000..30cc424f855 --- /dev/null +++ b/docs/my-website/static/llms-full.txt @@ -0,0 +1,9164 @@ +# https://docs.litellm.ai/ llms-full.txt + +## LiteLLM Overview +[Skip to main content](https://docs.litellm.ai/#__docusaurus_skipToContent_fallback) + +# LiteLLM - Getting Started + +[https://github.com/BerriAI/litellm](https://github.com/BerriAI/litellm) + +## **Call 100+ LLMs using the OpenAI Input/Output Format** [​](https://docs.litellm.ai/\#call-100-llms-using-the-openai-inputoutput-format "Direct link to call-100-llms-using-the-openai-inputoutput-format") + +- Translate inputs to provider's `completion`, `embedding`, and `image_generation` endpoints +- [Consistent output](https://docs.litellm.ai/docs/completion/output), text responses will always be available at `['choices'][0]['message']['content']` +- Retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - [Router](https://docs.litellm.ai/docs/routing) +- Track spend & set budgets per project [LiteLLM Proxy Server](https://docs.litellm.ai/docs/simple_proxy) + +## How to use LiteLLM [​](https://docs.litellm.ai/\#how-to-use-litellm "Direct link to How to use LiteLLM") + +You can use litellm through either: + +1. [LiteLLM Proxy Server](https://docs.litellm.ai/#litellm-proxy-server-llm-gateway) \- Server (LLM Gateway) to call 100+ LLMs, load balance, cost tracking across projects +2. [LiteLLM python SDK](https://docs.litellm.ai/#basic-usage) \- Python Client to call 100+ LLMs, load balance, cost tracking + +### **When to use LiteLLM Proxy Server (LLM Gateway)** [​](https://docs.litellm.ai/\#when-to-use-litellm-proxy-server-llm-gateway "Direct link to when-to-use-litellm-proxy-server-llm-gateway") + +tip + +Use LiteLLM Proxy Server if you want a **central service (LLM Gateway) to access multiple LLMs** + +Typically used by Gen AI Enablement / ML PLatform Teams + +- LiteLLM Proxy gives you a unified interface to access multiple LLMs (100+ LLMs) +- Track LLM Usage and setup guardrails +- Customize Logging, Guardrails, Caching per project + +### **When to use LiteLLM Python SDK** [​](https://docs.litellm.ai/\#when-to-use-litellm-python-sdk "Direct link to when-to-use-litellm-python-sdk") + +tip + +Use LiteLLM Python SDK if you want to use LiteLLM in your **python code** + +Typically used by developers building llm projects + +- LiteLLM SDK gives you a unified interface to access multiple LLMs (100+ LLMs) +- Retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - [Router](https://docs.litellm.ai/docs/routing) + +## **LiteLLM Python SDK** [​](https://docs.litellm.ai/\#litellm-python-sdk "Direct link to litellm-python-sdk") + +### Basic usage [​](https://docs.litellm.ai/\#basic-usage "Direct link to Basic usage") + +[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/BerriAI/litellm/blob/main/cookbook/liteLLM_Getting_Started.ipynb) + +```codeBlockLines_e6Vv +pip install litellm + +``` + +- OpenAI +- Anthropic +- VertexAI +- NVIDIA +- HuggingFace +- Azure OpenAI +- Ollama +- Openrouter +- Novita AI + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["OPENAI_API_KEY"] = "your-api-key" + +response = completion( + model="gpt-3.5-turbo", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["ANTHROPIC_API_KEY"] = "your-api-key" + +response = completion( + model="claude-2", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +# auth: run 'gcloud auth application-default' +os.environ["VERTEX_PROJECT"] = "hardy-device-386718" +os.environ["VERTEX_LOCATION"] = "us-central1" + +response = completion( + model="chat-bison", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["NVIDIA_NIM_API_KEY"] = "nvidia_api_key" +os.environ["NVIDIA_NIM_API_BASE"] = "nvidia_nim_endpoint_url" + +response = completion( + model="nvidia_nim/", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +os.environ["HUGGINGFACE_API_KEY"] = "huggingface_api_key" + +# e.g. Call 'WizardLM/WizardCoder-Python-34B-V1.0' hosted on HF Inference endpoints +response = completion( + model="huggingface/WizardLM/WizardCoder-Python-34B-V1.0", + messages=[{ "content": "Hello, how are you?","role": "user"}], + api_base="https://my-endpoint.huggingface.cloud" +) + +print(response) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["AZURE_API_KEY"] = "" +os.environ["AZURE_API_BASE"] = "" +os.environ["AZURE_API_VERSION"] = "" + +# azure call +response = completion( + "azure/", + messages = [{ "content": "Hello, how are you?","role": "user"}] +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion + +response = completion( + model="ollama/llama2", + messages = [{ "content": "Hello, how are you?","role": "user"}], + api_base="http://localhost:11434" +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["OPENROUTER_API_KEY"] = "openrouter_api_key" + +response = completion( + model="openrouter/google/palm-2-chat-bison", + messages = [{ "content": "Hello, how are you?","role": "user"}], +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key +os.environ["NOVITA_API_KEY"] = "novita-api-key" + +response = completion( + model="novita/deepseek/deepseek-r1", + messages=[{ "content": "Hello, how are you?","role": "user"}] +) + +``` + +### Streaming [​](https://docs.litellm.ai/\#streaming "Direct link to Streaming") + +Set `stream=True` in the `completion` args. + +- OpenAI +- Anthropic +- VertexAI +- NVIDIA +- HuggingFace +- Azure OpenAI +- Ollama +- Openrouter +- Novita AI + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["OPENAI_API_KEY"] = "your-api-key" + +response = completion( + model="gpt-3.5-turbo", + messages=[{ "content": "Hello, how are you?","role": "user"}], + stream=True, +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["ANTHROPIC_API_KEY"] = "your-api-key" + +response = completion( + model="claude-2", + messages=[{ "content": "Hello, how are you?","role": "user"}], + stream=True, +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +# auth: run 'gcloud auth application-default' +os.environ["VERTEX_PROJECT"] = "hardy-device-386718" +os.environ["VERTEX_LOCATION"] = "us-central1" + +response = completion( + model="chat-bison", + messages=[{ "content": "Hello, how are you?","role": "user"}], + stream=True, +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["NVIDIA_NIM_API_KEY"] = "nvidia_api_key" +os.environ["NVIDIA_NIM_API_BASE"] = "nvidia_nim_endpoint_url" + +response = completion( + model="nvidia_nim/", + messages=[{ "content": "Hello, how are you?","role": "user"}] + stream=True, +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +os.environ["HUGGINGFACE_API_KEY"] = "huggingface_api_key" + +# e.g. Call 'WizardLM/WizardCoder-Python-34B-V1.0' hosted on HF Inference endpoints +response = completion( + model="huggingface/WizardLM/WizardCoder-Python-34B-V1.0", + messages=[{ "content": "Hello, how are you?","role": "user"}], + api_base="https://my-endpoint.huggingface.cloud", + stream=True, +) + +print(response) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["AZURE_API_KEY"] = "" +os.environ["AZURE_API_BASE"] = "" +os.environ["AZURE_API_VERSION"] = "" + +# azure call +response = completion( + "azure/", + messages = [{ "content": "Hello, how are you?","role": "user"}], + stream=True, +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion + +response = completion( + model="ollama/llama2", + messages = [{ "content": "Hello, how are you?","role": "user"}], + api_base="http://localhost:11434", + stream=True, +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables +os.environ["OPENROUTER_API_KEY"] = "openrouter_api_key" + +response = completion( + model="openrouter/google/palm-2-chat-bison", + messages = [{ "content": "Hello, how are you?","role": "user"}], + stream=True, +) + +``` + +```codeBlockLines_e6Vv +from litellm import completion +import os + +## set ENV variables. Visit https://novita.ai/settings/key-management to get your API key +os.environ["NOVITA_API_KEY"] = "novita_api_key" + +response = completion( + model="novita/deepseek/deepseek-r1", + messages = [{ "content": "Hello, how are you?","role": "user"}], + stream=True, +) + +``` + +### Exception handling [​](https://docs.litellm.ai/\#exception-handling "Direct link to Exception handling") + +LiteLLM maps exceptions across all supported providers to the OpenAI exceptions. All our exceptions inherit from OpenAI's exception types, so any error-handling you have for that, should work out of the box with LiteLLM. + +```codeBlockLines_e6Vv +from openai.error import OpenAIError +from litellm import completion + +os.environ["ANTHROPIC_API_KEY"] = "bad-key" +try: + # some code + completion(model="claude-instant-1", messages=[{"role": "user", "content": "Hey, how's it going?"}]) +except OpenAIError as e: + print(e) + +``` + +### Logging Observability - Log LLM Input/Output ( [Docs](https://docs.litellm.ai/docs/observability/callbacks)) [​](https://docs.litellm.ai/\#logging-observability---log-llm-inputoutput-docs "Direct link to logging-observability---log-llm-inputoutput-docs") + +LiteLLM exposes pre defined callbacks to send data to MLflow, Lunary, Langfuse, Helicone, Promptlayer, Traceloop, Slack + +```codeBlockLines_e6Vv +from litellm import completion + +## set env variables for logging tools (API key set up is not required when using MLflow) +os.environ["LUNARY_PUBLIC_KEY"] = "your-lunary-public-key" # get your key at https://app.lunary.ai/settings +os.environ["HELICONE_API_KEY"] = "your-helicone-key" +os.environ["LANGFUSE_PUBLIC_KEY"] = "" +os.environ["LANGFUSE_SECRET_KEY"] = "" + +os.environ["OPENAI_API_KEY"] + +# set callbacks +litellm.success_callback = ["lunary", "mlflow", "langfuse", "helicone"] # log input/output to lunary, mlflow, langfuse, helicone + +#openai call +response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) + +``` + +### Track Costs, Usage, Latency for streaming [​](https://docs.litellm.ai/\#track-costs-usage-latency-for-streaming "Direct link to Track Costs, Usage, Latency for streaming") + +Use a callback function for this - more info on custom callbacks: [https://docs.litellm.ai/docs/observability/custom\_callback](https://docs.litellm.ai/docs/observability/custom_callback) + +```codeBlockLines_e6Vv +import litellm + +# track_cost_callback +def track_cost_callback( + kwargs, # kwargs to completion + completion_response, # response from completion + start_time, end_time # start/end time +): + try: + response_cost = kwargs.get("response_cost", 0) + print("streaming response_cost", response_cost) + except: + pass +# set callback +litellm.success_callback = [track_cost_callback] # set custom callback function + +# litellm.completion() call +response = completion( + model="gpt-3.5-turbo", + messages=[\ + {\ + "role": "user",\ + "content": "Hi 👋 - i'm openai"\ + }\ + ], + stream=True +) + +``` + +## **LiteLLM Proxy Server (LLM Gateway)** [​](https://docs.litellm.ai/\#litellm-proxy-server-llm-gateway "Direct link to litellm-proxy-server-llm-gateway") + +Track spend across multiple projects/people + +![ui_3](https://github.com/BerriAI/litellm/assets/29436595/47c97d5e-b9be-4839-b28c-43d7f4f10033) + +The proxy provides: + +1. [Hooks for auth](https://docs.litellm.ai/docs/proxy/virtual_keys#custom-auth) +2. [Hooks for logging](https://docs.litellm.ai/docs/proxy/logging#step-1---create-your-custom-litellm-callback-class) +3. [Cost tracking](https://docs.litellm.ai/docs/proxy/virtual_keys#tracking-spend) +4. [Rate Limiting](https://docs.litellm.ai/docs/proxy/users#set-rate-limits) + +### 📖 Proxy Endpoints - [Swagger Docs](https://litellm-api.up.railway.app/) [​](https://docs.litellm.ai/\#-proxy-endpoints---swagger-docs "Direct link to -proxy-endpoints---swagger-docs") + +Go here for a complete tutorial with keys + rate limits - [**here**](https://docs.litellm.ai/proxy/docker_quick_start.md) + +### Quick Start Proxy - CLI [​](https://docs.litellm.ai/\#quick-start-proxy---cli "Direct link to Quick Start Proxy - CLI") + +```codeBlockLines_e6Vv +pip install 'litellm[proxy]' + +``` + +#### Step 1: Start litellm proxy [​](https://docs.litellm.ai/\#step-1-start-litellm-proxy "Direct link to Step 1: Start litellm proxy") + +- pip package +- Docker container + +```codeBlockLines_e6Vv +$ litellm --model huggingface/bigcode/starcoder + +#INFO: Proxy running on http://0.0.0.0:4000 + +``` + +### Step 1. CREATE config.yaml [​](https://docs.litellm.ai/\#step-1-create-configyaml "Direct link to Step 1. CREATE config.yaml") + +Example `litellm_config.yaml` + +```codeBlockLines_e6Vv +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: azure/ + api_base: os.environ/AZURE_API_BASE # runs os.getenv("AZURE_API_BASE") + api_key: os.environ/AZURE_API_KEY # runs os.getenv("AZURE_API_KEY") + api_version: "2023-07-01-preview" + +``` + +### Step 2. RUN Docker Image [​](https://docs.litellm.ai/\#step-2-run-docker-image "Direct link to Step 2. RUN Docker Image") + +```codeBlockLines_e6Vv +docker run \ + -v $(pwd)/litellm_config.yaml:/app/config.yaml \ + -e AZURE_API_KEY=d6*********** \ + -e AZURE_API_BASE=https://openai-***********/ \ + -p 4000:4000 \ + ghcr.io/berriai/litellm:main-latest \ + --config /app/config.yaml --detailed_debug + +``` + +#### Step 2: Make ChatCompletions Request to Proxy [​](https://docs.litellm.ai/\#step-2-make-chatcompletions-request-to-proxy "Direct link to Step 2: Make ChatCompletions Request to Proxy") + +```codeBlockLines_e6Vv +import openai # openai v1.0.0+ +client = openai.OpenAI(api_key="anything",base_url="http://0.0.0.0:4000") # set proxy to base_url +# request sent to model set on litellm proxy, `litellm --model` +response = client.chat.completions.create(model="gpt-3.5-turbo", messages = [\ + {\ + "role": "user",\ + "content": "this is a test request, write a short poem"\ + }\ +]) + +print(response) + +``` + +## More details [​](https://docs.litellm.ai/\#more-details "Direct link to More details") + +- [exception mapping](https://docs.litellm.ai/docs/exception_mapping) +- [E2E Tutorial for LiteLLM Proxy Server](https://docs.litellm.ai/docs/proxy/docker_quick_start) +- [proxy virtual keys & spend management](https://docs.litellm.ai/docs/proxy/virtual_keys) + +- [**Call 100+ LLMs using the OpenAI Input/Output Format**](https://docs.litellm.ai/#call-100-llms-using-the-openai-inputoutput-format) +- [How to use LiteLLM](https://docs.litellm.ai/#how-to-use-litellm) + - [**When to use LiteLLM Proxy Server (LLM Gateway)**](https://docs.litellm.ai/#when-to-use-litellm-proxy-server-llm-gateway) + - [**When to use LiteLLM Python SDK**](https://docs.litellm.ai/#when-to-use-litellm-python-sdk) +- [**LiteLLM Python SDK**](https://docs.litellm.ai/#litellm-python-sdk) + - [Basic usage](https://docs.litellm.ai/#basic-usage) + - [Streaming](https://docs.litellm.ai/#streaming) + - [Exception handling](https://docs.litellm.ai/#exception-handling) + - [Logging Observability - Log LLM Input/Output (Docs)](https://docs.litellm.ai/#logging-observability---log-llm-inputoutput-docs) + - [Track Costs, Usage, Latency for streaming](https://docs.litellm.ai/#track-costs-usage-latency-for-streaming) +- [**LiteLLM Proxy Server (LLM Gateway)**](https://docs.litellm.ai/#litellm-proxy-server-llm-gateway) + - [📖 Proxy Endpoints - Swagger Docs](https://docs.litellm.ai/#-proxy-endpoints---swagger-docs) + - [Quick Start Proxy - CLI](https://docs.litellm.ai/#quick-start-proxy---cli) + - [Step 1. CREATE config.yaml](https://docs.litellm.ai/#step-1-create-configyaml) + - [Step 2. RUN Docker Image](https://docs.litellm.ai/#step-2-run-docker-image) +- [More details](https://docs.litellm.ai/#more-details) + +## Completion Function Guide +[Skip to main content](https://docs.litellm.ai/completion/input#__docusaurus_skipToContent_fallback) + +# Completion Function - completion() + +The Input params are **exactly the same** as the + +[OpenAI Create chat completion](https://platform.openai.com/docs/api-reference/chat/create), and let you call \*\*Azure OpenAI, Anthropic, Cohere, Replicate, OpenRouter, Novita AI\*\* models in the same format. + +In addition, liteLLM allows you to pass in the following **Optional** liteLLM args: +`force_timeout`, `azure`, `logger_fn`, `verbose` + +## Input - Request Body [​](https://docs.litellm.ai/completion/input\#input---request-body "Direct link to Input - Request Body") + +# Request Body + +**Required Fields** + +- `model`: _string_ \- ID of the model to use. Refer to the model endpoint compatibility table for details on which models work with the Chat API. +- `messages`: _array_ \- A list of messages comprising the conversation so far. + +_Note_ \- Each message in the array contains the following properties: + +```codeBlockLines_e6Vv +- `role`: *string* - The role of the message's author. Roles can be: system, user, assistant, or function. + +- `content`: *string or null* - The contents of the message. It is required for all messages, but may be null for assistant messages with function calls. + +- `name`: *string (optional)* - The name of the author of the message. It is required if the role is "function". The name should match the name of the function represented in the content. It can contain characters (a-z, A-Z, 0-9), and underscores, with a maximum length of 64 characters. + +- `function_call`: *object (optional)* - The name and arguments of a function that should be called, as generated by the model. + +``` + +**Optional Fields** + +- `functions`: _array_ \- A list of functions that the model may use to generate JSON inputs. Each function should have the following properties: + + - `name`: _string_ \- The name of the function to be called. It should contain a-z, A-Z, 0-9, underscores and dashes, with a maximum length of 64 characters. + - `description`: _string (optional)_ \- A description explaining what the function does. It helps the model to decide when and how to call the function. + - `parameters`: _object_ \- The parameters that the function accepts, described as a JSON Schema object. + - `function_call`: _string or object (optional)_ \- Controls how the model responds to function calls. +- `temperature`: _number or null (optional)_ \- The sampling temperature to be used, between 0 and 2. Higher values like 0.8 produce more random outputs, while lower values like 0.2 make outputs more focused and deterministic. + +- `top_p`: _number or null (optional)_ \- An alternative to sampling with temperature. It instructs the model to consider the results of the tokens with top\_p probability. For example, 0.1 means only the tokens comprising the top 10% probability mass are considered. + +- `n`: _integer or null (optional)_ \- The number of chat completion choices to generate for each input message. + +- `stream`: _boolean or null (optional)_ \- If set to true, it sends partial message deltas. Tokens will be sent as they become available, with the stream terminated by a \[DONE\] message. + +- `stop`: _string/ array/ null (optional)_ \- Up to 4 sequences where the API will stop generating further tokens. + +- `max_tokens`: _integer (optional)_ \- The maximum number of tokens to generate in the chat completion. + +- `presence_penalty`: _number or null (optional)_ \- It is used to penalize new tokens based on their existence in the text so far. + +- `frequency_penalty`: _number or null (optional)_ \- It is used to penalize new tokens based on their frequency in the text so far. + +- `logit_bias`: _map (optional)_ \- Used to modify the probability of specific tokens appearing in the completion. + +- `user`: _string (optional)_ \- A unique identifier representing your end-user. This can help OpenAI to monitor and detect abuse. + + +- [Input - Request Body](https://docs.litellm.ai/completion/input#input---request-body) + +## Litellm Completion Function +[Skip to main content](https://docs.litellm.ai/completion/output#__docusaurus_skipToContent_fallback) + +# Completion Function - completion() + +Here's the exact json output you can expect from a litellm `completion` call: + +```codeBlockLines_e6Vv +{'choices': [{'finish_reason': 'stop',\ + 'index': 0,\ + 'message': {'role': 'assistant',\ + 'content': " I'm doing well, thank you for asking. I am Claude, an AI assistant created by Anthropic."}}], + 'created': 1691429984.3852863, + 'model': 'claude-instant-1', + 'usage': {'prompt_tokens': 18, 'completion_tokens': 23, 'total_tokens': 41}} + +``` + +## AI Completion Models +[Skip to main content](https://docs.litellm.ai/completion/supported#__docusaurus_skipToContent_fallback) + +# Generation/Completion/Chat Completion Models + +### OpenAI Chat Completion Models [​](https://docs.litellm.ai/completion/supported\#openai-chat-completion-models "Direct link to OpenAI Chat Completion Models") + +| Model Name | Function Call | Required OS Variables | +| --- | --- | --- | +| gpt-3.5-turbo | `completion('gpt-3.5-turbo', messages)` | `os.environ['OPENAI_API_KEY']` | +| gpt-3.5-turbo-16k | `completion('gpt-3.5-turbo-16k', messages)` | `os.environ['OPENAI_API_KEY']` | +| gpt-3.5-turbo-16k-0613 | `completion('gpt-3.5-turbo-16k-0613', messages)` | `os.environ['OPENAI_API_KEY']` | +| gpt-4 | `completion('gpt-4', messages)` | `os.environ['OPENAI_API_KEY']` | + +## Azure OpenAI Chat Completion Models [​](https://docs.litellm.ai/completion/supported\#azure-openai-chat-completion-models "Direct link to Azure OpenAI Chat Completion Models") + +For Azure calls add the `azure/` prefix to `model`. If your azure deployment name is `gpt-v-2` set `model` = `azure/gpt-v-2` + +| Model Name | Function Call | Required OS Variables | +| --- | --- | --- | +| gpt-3.5-turbo | `completion('azure/gpt-3.5-turbo-deployment', messages)` | `os.environ['AZURE_API_KEY']`, `os.environ['AZURE_API_BASE']`, `os.environ['AZURE_API_VERSION']` | +| gpt-4 | `completion('azure/gpt-4-deployment', messages)` | `os.environ['AZURE_API_KEY']`, `os.environ['AZURE_API_BASE']`, `os.environ['AZURE_API_VERSION']` | + +### OpenAI Text Completion Models [​](https://docs.litellm.ai/completion/supported\#openai-text-completion-models "Direct link to OpenAI Text Completion Models") + +| Model Name | Function Call | Required OS Variables | +| --- | --- | --- | +| text-davinci-003 | `completion('text-davinci-003', messages)` | `os.environ['OPENAI_API_KEY']` | + +### Cohere Models [​](https://docs.litellm.ai/completion/supported\#cohere-models "Direct link to Cohere Models") + +| Model Name | Function Call | Required OS Variables | +| --- | --- | --- | +| command-nightly | `completion('command-nightly', messages)` | `os.environ['COHERE_API_KEY']` | + +### Anthropic Models [​](https://docs.litellm.ai/completion/supported\#anthropic-models "Direct link to Anthropic Models") + +| Model Name | Function Call | Required OS Variables | +| --- | --- | --- | +| claude-instant-1 | `completion('claude-instant-1', messages)` | `os.environ['ANTHROPIC_API_KEY']` | +| claude-2 | `completion('claude-2', messages)` | `os.environ['ANTHROPIC_API_KEY']` | + +### Hugging Face Inference API [​](https://docs.litellm.ai/completion/supported\#hugging-face-inference-api "Direct link to Hugging Face Inference API") + +All [`text2text-generation`](https://huggingface.co/models?library=transformers&pipeline_tag=text2text-generation&sort=downloads) and [`text-generation`](https://huggingface.co/models?library=transformers&pipeline_tag=text-generation&sort=downloads) models are supported by liteLLM. You can use any text model from Hugging Face with the following steps: + +- Copy the `model repo` URL from Hugging Face and set it as the `model` parameter in the completion call. +- Set `hugging_face` parameter to `True`. +- Make sure to set the hugging face API key + +Here are some examples of supported models: +**Note that the models mentioned in the table are examples, and you can use any text model available on Hugging Face by following the steps above.** + +| Model Name | Function Call | Required OS Variables | +| --- | --- | --- | +| [stabilityai/stablecode-completion-alpha-3b-4k](https://huggingface.co/stabilityai/stablecode-completion-alpha-3b-4k) | `completion(model="stabilityai/stablecode-completion-alpha-3b-4k", messages=messages, hugging_face=True)` | `os.environ['HF_TOKEN']` | +| [bigcode/starcoder](https://huggingface.co/bigcode/starcoder) | `completion(model="bigcode/starcoder", messages=messages, hugging_face=True)` | `os.environ['HF_TOKEN']` | +| [google/flan-t5-xxl](https://huggingface.co/google/flan-t5-xxl) | `completion(model="google/flan-t5-xxl", messages=messages, hugging_face=True)` | `os.environ['HF_TOKEN']` | +| [google/flan-t5-large](https://huggingface.co/google/flan-t5-large) | `completion(model="google/flan-t5-large", messages=messages, hugging_face=True)` | `os.environ['HF_TOKEN']` | + +### OpenRouter Completion Models [​](https://docs.litellm.ai/completion/supported\#openrouter-completion-models "Direct link to OpenRouter Completion Models") + +All the text models from [OpenRouter](https://openrouter.ai/docs) are supported by liteLLM. + +| Model Name | Function Call | Required OS Variables | +| --- | --- | --- | +| openai/gpt-3.5-turbo | `completion('openai/gpt-3.5-turbo', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| openai/gpt-3.5-turbo-16k | `completion('openai/gpt-3.5-turbo-16k', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| openai/gpt-4 | `completion('openai/gpt-4', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| openai/gpt-4-32k | `completion('openai/gpt-4-32k', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| anthropic/claude-2 | `completion('anthropic/claude-2', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| anthropic/claude-instant-v1 | `completion('anthropic/claude-instant-v1', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| google/palm-2-chat-bison | `completion('google/palm-2-chat-bison', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| google/palm-2-codechat-bison | `completion('google/palm-2-codechat-bison', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| meta-llama/llama-2-13b-chat | `completion('meta-llama/llama-2-13b-chat', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | +| meta-llama/llama-2-70b-chat | `completion('meta-llama/llama-2-70b-chat', messages)` | `os.environ['OR_SITE_URL']`, `os.environ['OR_APP_NAME']`, `os.environ['OR_API_KEY']` | + +## Novita AI Completion Models [​](https://docs.litellm.ai/completion/supported\#novita-ai-completion-models "Direct link to Novita AI Completion Models") + +🚨 LiteLLM supports ALL Novita AI models, send `model=novita/` to send it to Novita AI. See all Novita AI models [here](https://novita.ai/models/llm?utm_source=github_litellm&utm_medium=github_readme&utm_campaign=github_link) + +| Model Name | Function Call | Required OS Variables | +| --- | --- | --- | +| novita/deepseek/deepseek-r1 | `completion('novita/deepseek/deepseek-r1', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/deepseek/deepseek\_v3 | `completion('novita/deepseek/deepseek_v3', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.3-70b-instruct | `completion('novita/meta-llama/llama-3.3-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.1-8b-instruct | `completion('novita/meta-llama/llama-3.1-8b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.1-8b-instruct-max | `completion('novita/meta-llama/llama-3.1-8b-instruct-max', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.1-70b-instruct | `completion('novita/meta-llama/llama-3.1-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3-8b-instruct | `completion('novita/meta-llama/llama-3-8b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3-70b-instruct | `completion('novita/meta-llama/llama-3-70b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.2-1b-instruct | `completion('novita/meta-llama/llama-3.2-1b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.2-11b-vision-instruct | `completion('novita/meta-llama/llama-3.2-11b-vision-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/meta-llama/llama-3.2-3b-instruct | `completion('novita/meta-llama/llama-3.2-3b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/gryphe/mythomax-l2-13b | `completion('novita/gryphe/mythomax-l2-13b', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/google/gemma-2-9b-it | `completion('novita/google/gemma-2-9b-it', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/mistralai/mistral-nemo | `completion('novita/mistralai/mistral-nemo', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/mistralai/mistral-7b-instruct | `completion('novita/mistralai/mistral-7b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/qwen/qwen-2.5-72b-instruct | `completion('novita/qwen/qwen-2.5-72b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | +| novita/qwen/qwen-2-vl-72b-instruct | `completion('novita/qwen/qwen-2-vl-72b-instruct', messages)` | `os.environ['NOVITA_API_KEY']` | + +- [OpenAI Chat Completion Models](https://docs.litellm.ai/completion/supported#openai-chat-completion-models) +- [Azure OpenAI Chat Completion Models](https://docs.litellm.ai/completion/supported#azure-openai-chat-completion-models) + - [OpenAI Text Completion Models](https://docs.litellm.ai/completion/supported#openai-text-completion-models) + - [Cohere Models](https://docs.litellm.ai/completion/supported#cohere-models) + - [Anthropic Models](https://docs.litellm.ai/completion/supported#anthropic-models) + - [Hugging Face Inference API](https://docs.litellm.ai/completion/supported#hugging-face-inference-api) + - [OpenRouter Completion Models](https://docs.litellm.ai/completion/supported#openrouter-completion-models) +- [Novita AI Completion Models](https://docs.litellm.ai/completion/supported#novita-ai-completion-models) + +## Contact Litellm +[Skip to main content](https://docs.litellm.ai/contact#__docusaurus_skipToContent_fallback) + +# Contact Us + +[![](https://dcbadge.vercel.app/api/server/wuPM9dRgDw)](https://discord.gg/wuPM9dRgDw) + +- [Meet with us 👋](https://calendly.com/d/4mp-gd3-k5k/berriai-1-1-onboarding-litellm-hosted-version) +- Contact us at [ishaan@berri.ai](mailto:ishaan@berri.ai) / [krrish@berri.ai](mailto:krrish@berri.ai) + +## Contributing to Documentation +[Skip to main content](https://docs.litellm.ai/contributing#__docusaurus_skipToContent_fallback) + +# Contributing to Documentation + +Clone litellm + +```codeBlockLines_e6Vv +git clone https://github.com/BerriAI/litellm.git + +``` + +### Local setup for locally running docs [​](https://docs.litellm.ai/contributing\#local-setup-for-locally-running-docs "Direct link to Local setup for locally running docs") + +#### Installation [​](https://docs.litellm.ai/contributing\#installation "Direct link to Installation") + +```codeBlockLines_e6Vv +pip install mkdocs + +``` + +#### Locally Serving Docs [​](https://docs.litellm.ai/contributing\#locally-serving-docs "Direct link to Locally Serving Docs") + +```codeBlockLines_e6Vv +mkdocs serve + +``` + +If you see `command not found: mkdocs` try running the following + +```codeBlockLines_e6Vv +python3 -m mkdocs serve + +``` + +This command builds your Markdown files into HTML and starts a development server to browse your documentation. Open up [http://127.0.0.1:8000/](http://127.0.0.1:8000/) in your web browser to see your documentation. You can make changes to your Markdown files and your docs will automatically rebuild. + +[Full tutorial here](https://docs.readthedocs.io/en/stable/intro/getting-started-with-mkdocs.html) + +### Making changes to Docs [​](https://docs.litellm.ai/contributing\#making-changes-to-docs "Direct link to Making changes to Docs") + +- All the docs are placed under the `docs` directory +- If you are adding a new `.md` file or editing the hierarchy edit `mkdocs.yml` in the root of the project +- After testing your changes, make a change to the `main` branch of [github.com/BerriAI/litellm](https://github.com/BerriAI/litellm) + +- [Local setup for locally running docs](https://docs.litellm.ai/contributing#local-setup-for-locally-running-docs) +- [Making changes to Docs](https://docs.litellm.ai/contributing#making-changes-to-docs) + +## Supported Embedding Models +[Skip to main content](https://docs.litellm.ai/embedding/supported_embedding#__docusaurus_skipToContent_fallback) + +# Embedding Models + +| Model Name | Function Call | Required OS Variables | +| --- | --- | --- | +| text-embedding-ada-002 | `embedding('text-embedding-ada-002', input)` | `os.environ['OPENAI_API_KEY']` | + +## Docusaurus Setup Guide +[Skip to main content](https://docs.litellm.ai/intro#__docusaurus_skipToContent_fallback) + +# Tutorial Intro + +Let's discover **Docusaurus in less than 5 minutes**. + +## Getting Started [​](https://docs.litellm.ai/intro\#getting-started "Direct link to Getting Started") + +Get started by **creating a new site**. + +Or **try Docusaurus immediately** with **[docusaurus.new](https://docusaurus.new/)**. + +### What you'll need [​](https://docs.litellm.ai/intro\#what-youll-need "Direct link to What you'll need") + +- [Node.js](https://nodejs.org/en/download/) version 16.14 or above: + - When installing Node.js, you are recommended to check all checkboxes related to dependencies. + +## Generate a new site [​](https://docs.litellm.ai/intro\#generate-a-new-site "Direct link to Generate a new site") + +Generate a new Docusaurus site using the **classic template**. + +The classic template will automatically be added to your project after you run the command: + +```codeBlockLines_e6Vv +npm init docusaurus@latest my-website classic + +``` + +You can type this command into Command Prompt, Powershell, Terminal, or any other integrated terminal of your code editor. + +The command also installs all necessary dependencies you need to run Docusaurus. + +## Start your site [​](https://docs.litellm.ai/intro\#start-your-site "Direct link to Start your site") + +Run the development server: + +```codeBlockLines_e6Vv +cd my-website +npm run start + +``` + +The `cd` command changes the directory you're working with. In order to work with your newly created Docusaurus site, you'll need to navigate the terminal there. + +The `npm run start` command builds your website locally and serves it through a development server, ready for you to view at http://localhost:3000/. + +Open `docs/intro.md` (this page) and edit some lines: the site **reloads automatically** and displays your changes. + +- [Getting Started](https://docs.litellm.ai/intro#getting-started) + - [What you'll need](https://docs.litellm.ai/intro#what-youll-need) +- [Generate a new site](https://docs.litellm.ai/intro#generate-a-new-site) +- [Start your site](https://docs.litellm.ai/intro#start-your-site) + +## Callbacks for Data Output +[Skip to main content](https://docs.litellm.ai/observability/callbacks#__docusaurus_skipToContent_fallback) + +# Callbacks + +## Use Callbacks to send Output Data to Posthog, Sentry etc [​](https://docs.litellm.ai/observability/callbacks\#use-callbacks-to-send-output-data-to-posthog-sentry-etc "Direct link to Use Callbacks to send Output Data to Posthog, Sentry etc") + +liteLLM provides `success_callbacks` and `failure_callbacks`, making it easy for you to send data to a particular provider depending on the status of your responses. + +liteLLM supports: + +- [Lunary](https://lunary.ai/docs) +- [Helicone](https://docs.helicone.ai/introduction) +- [Sentry](https://docs.sentry.io/platforms/python/) +- [PostHog](https://posthog.com/docs/libraries/python) +- [Slack](https://slack.dev/bolt-python/concepts) + +### Quick Start [​](https://docs.litellm.ai/observability/callbacks\#quick-start "Direct link to Quick Start") + +```codeBlockLines_e6Vv +from litellm import completion + +# set callbacks +litellm.success_callback=["posthog", "helicone", "lunary"] +litellm.failure_callback=["sentry", "lunary"] + +## set env variables +os.environ['SENTRY_DSN'], os.environ['SENTRY_API_TRACE_RATE']= "" +os.environ['POSTHOG_API_KEY'], os.environ['POSTHOG_API_URL'] = "api-key", "api-url" +os.environ["HELICONE_API_KEY"] = "" + +response = completion(model="gpt-3.5-turbo", messages=messages) + +``` + +- [Use Callbacks to send Output Data to Posthog, Sentry etc](https://docs.litellm.ai/observability/callbacks#use-callbacks-to-send-output-data-to-posthog-sentry-etc) + - [Quick Start](https://docs.litellm.ai/observability/callbacks#quick-start) + +## Helicone Integration Guide +[Skip to main content](https://docs.litellm.ai/observability/helicone_integration#__docusaurus_skipToContent_fallback) + +# Helicone Tutorial + +[Helicone](https://helicone.ai/) is an open source observability platform that proxies your OpenAI traffic and provides you key insights into your spend, latency and usage. + +## Use Helicone to log requests across all LLM Providers (OpenAI, Azure, Anthropic, Cohere, Replicate, PaLM) [​](https://docs.litellm.ai/observability/helicone_integration\#use-helicone-to-log-requests-across-all-llm-providers-openai-azure-anthropic-cohere-replicate-palm "Direct link to Use Helicone to log requests across all LLM Providers (OpenAI, Azure, Anthropic, Cohere, Replicate, PaLM)") + +liteLLM provides `success_callbacks` and `failure_callbacks`, making it easy for you to send data to a particular provider depending on the status of your responses. + +In this case, we want to log requests to Helicone when a request succeeds. + +### Approach 1: Use Callbacks [​](https://docs.litellm.ai/observability/helicone_integration\#approach-1-use-callbacks "Direct link to Approach 1: Use Callbacks") + +Use just 1 line of code, to instantly log your responses **across all providers** with helicone: + +```codeBlockLines_e6Vv +litellm.success_callback=["helicone"] + +``` + +Complete code + +```codeBlockLines_e6Vv +from litellm import completion + +## set env variables +os.environ["HELICONE_API_KEY"] = "your-helicone-key" +os.environ["OPENAI_API_KEY"], os.environ["COHERE_API_KEY"] = "", "" + +# set callbacks +litellm.success_callback=["helicone"] + +#openai call +response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) + +#cohere call +response = completion(model="command-nightly", messages=[{"role": "user", "content": "Hi 👋 - i'm cohere"}]) + +``` + +### Approach 2: \[OpenAI + Azure only\] Use Helicone as a proxy [​](https://docs.litellm.ai/observability/helicone_integration\#approach-2-openai--azure-only-use-helicone-as-a-proxy "Direct link to approach-2-openai--azure-only-use-helicone-as-a-proxy") + +Helicone provides advanced functionality like caching, etc. Helicone currently supports this for Azure and OpenAI. + +If you want to use Helicone to proxy your OpenAI/Azure requests, then you can - + +- Set helicone as your base url via: `litellm.api_url` +- Pass in helicone request headers via: `litellm.headers` + +Complete Code + +```codeBlockLines_e6Vv +import litellm +from litellm import completion + +litellm.api_base = "https://oai.hconeai.com/v1" +litellm.headers = {"Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}"} + +response = litellm.completion( + model="gpt-3.5-turbo", + messages=[{"role": "user", "content": "how does a court case get to the Supreme Court?"}] +) + +print(response) + +``` + +- [Use Helicone to log requests across all LLM Providers (OpenAI, Azure, Anthropic, Cohere, Replicate, PaLM)](https://docs.litellm.ai/observability/helicone_integration#use-helicone-to-log-requests-across-all-llm-providers-openai-azure-anthropic-cohere-replicate-palm) + - [Approach 1: Use Callbacks](https://docs.litellm.ai/observability/helicone_integration#approach-1-use-callbacks) + - [Approach 2: OpenAI + Azure only Use Helicone as a proxy](https://docs.litellm.ai/observability/helicone_integration#approach-2-openai--azure-only-use-helicone-as-a-proxy) + +## Supabase Integration Guide +[Skip to main content](https://docs.litellm.ai/observability/supabase_integration#__docusaurus_skipToContent_fallback) + +# Supabase Tutorial + +[Supabase](https://supabase.com/) is an open source Firebase alternative. +Start your project with a Postgres database, Authentication, instant APIs, Edge Functions, Realtime subscriptions, Storage, and Vector embeddings. + +## Use Supabase to log requests and see total spend across all LLM Providers (OpenAI, Azure, Anthropic, Cohere, Replicate, PaLM) [​](https://docs.litellm.ai/observability/supabase_integration\#use-supabase-to-log-requests-and-see-total-spend-across-all-llm-providers-openai-azure-anthropic-cohere-replicate-palm "Direct link to Use Supabase to log requests and see total spend across all LLM Providers (OpenAI, Azure, Anthropic, Cohere, Replicate, PaLM)") + +liteLLM provides `success_callbacks` and `failure_callbacks`, making it easy for you to send data to a particular provider depending on the status of your responses. + +In this case, we want to log requests to Supabase in both scenarios - when it succeeds and fails. + +### Create a supabase table [​](https://docs.litellm.ai/observability/supabase_integration\#create-a-supabase-table "Direct link to Create a supabase table") + +Go to your Supabase project > go to the [Supabase SQL Editor](https://supabase.com/dashboard/projects) and create a new table with this configuration. + +Note: You can change the table name. Just don't change the column names. + +```codeBlockLines_e6Vv +create table + public.request_logs ( + id bigint generated by default as identity, + created_at timestamp with time zone null default now(), + model text null default ''::text, + messages json null default '{}'::json, + response json null default '{}'::json, + end_user text null default ''::text, + error json null default '{}'::json, + response_time real null default '0'::real, + total_cost real null, + additional_details json null default '{}'::json, + constraint request_logs_pkey primary key (id) + ) tablespace pg_default; + +``` + +### Use Callbacks [​](https://docs.litellm.ai/observability/supabase_integration\#use-callbacks "Direct link to Use Callbacks") + +Use just 2 lines of code, to instantly see costs and log your responses **across all providers** with Supabase: + +```codeBlockLines_e6Vv +litellm.success_callback=["supabase"] +litellm.failure_callback=["supabase"] + +``` + +Complete code + +```codeBlockLines_e6Vv +from litellm import completion + +## set env variables +### SUPABASE +os.environ["SUPABASE_URL"] = "your-supabase-url" +os.environ["SUPABASE_KEY"] = "your-supabase-key" + +## LLM API KEY +os.environ["OPENAI_API_KEY"] = "" + +# set callbacks +litellm.success_callback=["supabase"] +litellm.failure_callback=["supabase"] + +#openai call +response = completion(model="gpt-3.5-turbo", messages=[{"role": "user", "content": "Hi 👋 - i'm openai"}]) + +#bad call +response = completion(model="chatgpt-test", messages=[{"role": "user", "content": "Hi 👋 - i'm a bad call to test error logging"}]) + +``` + +### Additional Controls [​](https://docs.litellm.ai/observability/supabase_integration\#additional-controls "Direct link to Additional Controls") + +**Different Table name** + +If you modified your table name, here's how to pass the new name. + +```codeBlockLines_e6Vv +litellm.modify_integration("supabase",{"table_name": "litellm_logs"}) + +``` + +**Identify end-user** + +Here's how to map your llm call to an end-user + +```codeBlockLines_e6Vv +litellm.identify({"end_user": "krrish@berri.ai"}) + +``` + +- [Use Supabase to log requests and see total spend across all LLM Providers (OpenAI, Azure, Anthropic, Cohere, Replicate, PaLM)](https://docs.litellm.ai/observability/supabase_integration#use-supabase-to-log-requests-and-see-total-spend-across-all-llm-providers-openai-azure-anthropic-cohere-replicate-palm) + - [Create a supabase table](https://docs.litellm.ai/observability/supabase_integration#create-a-supabase-table) + - [Use Callbacks](https://docs.litellm.ai/observability/supabase_integration#use-callbacks) + - [Additional Controls](https://docs.litellm.ai/observability/supabase_integration#additional-controls) + +## LiteLLM Release Notes +[Skip to main content](https://docs.litellm.ai/release_notes#__docusaurus_skipToContent_fallback) + +## Deploy this version [​](https://docs.litellm.ai/release_notes\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.70.1-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.70.1 + +``` + +## Key Highlights [​](https://docs.litellm.ai/release_notes\#key-highlights "Direct link to Key Highlights") + +LiteLLM v1.70.1-stable is live now. Here are the key highlights of this release: + +- **Gemini Realtime API**: You can now call Gemini's Live API via the OpenAI /v1/realtime API +- **Spend Logs Retention Period**: Enable deleting spend logs older than a certain period. +- **PII Masking 2.0**: Easily configure masking or blocking specific PII/PHI entities on the UI + +## Gemini Realtime API [​](https://docs.litellm.ai/release_notes\#gemini-realtime-api "Direct link to Gemini Realtime API") + +![](https://docs.litellm.ai/assets/ideal-img/gemini_realtime.c8e974c.1920.png) + +This release brings support for calling Gemini's realtime models (e.g. gemini-2.0-flash-live) via OpenAI's /v1/realtime API. This is great for developers as it lets them easily switch from OpenAI to Gemini by just changing the model name. + +Key Highlights: + +- Support for text + audio input/output +- Support for setting session configurations (modality, instructions, activity detection) in the OpenAI format +- Support for logging + usage tracking for realtime sessions + +This is currently supported via Google AI Studio. We plan to release VertexAI support over the coming week. + +[**Read more**](https://docs.litellm.ai/docs/providers/google_ai_studio/realtime) + +## Spend Logs Retention Period [​](https://docs.litellm.ai/release_notes\#spend-logs-retention-period "Direct link to Spend Logs Retention Period") + +![](https://docs.litellm.ai/assets/ideal-img/delete_spend_logs.158ab9b.1920.jpg) + +This release enables deleting LiteLLM Spend Logs older than a certain period. Since we now enable storing the raw request/response in the logs, deleting old logs ensures the database remains performant in production. + +[**Read more**](https://docs.litellm.ai/docs/proxy/spend_logs_deletion) + +## PII Masking 2.0 [​](https://docs.litellm.ai/release_notes\#pii-masking-20 "Direct link to PII Masking 2.0") + +![](https://docs.litellm.ai/assets/ideal-img/pii_masking_v2.8bb7c2d.1920.png) + +This release brings improvements to our Presidio PII Integration. As a Proxy Admin, you now have the ability to: + +- Mask or block specific entities (e.g., block medical licenses while masking other entities like emails). +- Monitor guardrails in production. LiteLLM Logs will now show you the guardrail run, the entities it detected, and its confidence score for each entity. + +[**Read more**](https://docs.litellm.ai/docs/proxy/guardrails/pii_masking_v2) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **Gemini ( [VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) \+ [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))** + - `/chat/completion` + - Handle audio input - [PR](https://github.com/BerriAI/litellm/pull/10739) + - Fixes maximum recursion depth issue when using deeply nested response schemas with Vertex AI by Increasing DEFAULT\_MAX\_RECURSE\_DEPTH from 10 to 100 in constants. [PR](https://github.com/BerriAI/litellm/pull/10798) + - Capture reasoning tokens in streaming mode - [PR](https://github.com/BerriAI/litellm/pull/10789) +- **[Google AI Studio](https://docs.litellm.ai/docs/providers/google_ai_studio/realtime)** + - `/realtime` + - Gemini Multimodal Live API support + - Audio input/output support, optional param mapping, accurate usage calculation - [PR](https://github.com/BerriAI/litellm/pull/10909) +- **[VertexAI](https://docs.litellm.ai/docs/providers/vertex#metallama-api)** + - `/chat/completion` + - Fix llama streaming error - where model response was nested in returned streaming chunk - [PR](https://github.com/BerriAI/litellm/pull/10878) +- **[Ollama](https://docs.litellm.ai/docs/providers/ollama)** + - `/chat/completion` + - structure responses fix - [PR](https://github.com/BerriAI/litellm/pull/10617) +- **[Bedrock](https://docs.litellm.ai/docs/providers/bedrock#litellm-proxy-usage)** + - [`/chat/completion`](https://docs.litellm.ai/docs/providers/bedrock#litellm-proxy-usage) + - Handle thinking\_blocks when assistant.content is None - [PR](https://github.com/BerriAI/litellm/pull/10688) + - Fixes to only allow accepted fields for tool json schema - [PR](https://github.com/BerriAI/litellm/pull/10062) + - Add bedrock sonnet prompt caching cost information + - Mistral Pixtral support - [PR](https://github.com/BerriAI/litellm/pull/10439) + - Tool caching support - [PR](https://github.com/BerriAI/litellm/pull/10897) + - [`/messages`](https://docs.litellm.ai/docs/anthropic_unified) + - allow using dynamic AWS Params - [PR](https://github.com/BerriAI/litellm/pull/10769) +- **[Nvidia NIM](https://docs.litellm.ai/docs/providers/nvidia_nim)** + - [`/chat/completion`](https://docs.litellm.ai/docs/providers/nvidia_nim#usage---litellm-proxy-server)\[NEED DOCS ON SUPPORTED PARAMS\] + - Add tools, tool\_choice, parallel\_tool\_calls support - [PR](https://github.com/BerriAI/litellm/pull/10763) +- **[Novita AI](https://docs.litellm.ai/docs/providers/novita)** + - New Provider added for `/chat/completion` routes - [PR](https://github.com/BerriAI/litellm/pull/9527) +- **[Azure](https://docs.litellm.ai/docs/providers/azure)** + - [`/image/generation`](https://docs.litellm.ai/docs/providers/azure#image-generation) + - Fix azure dall e 3 call with custom model name - [PR](https://github.com/BerriAI/litellm/pull/10776) +- **[Cohere](https://docs.litellm.ai/docs/providers/cohere)** + - [`/embeddings`](https://docs.litellm.ai/docs/providers/cohere#embedding) + - Migrate embedding to use `/v2/embed` \- adds support for output\_dimensions param - [PR](https://github.com/BerriAI/litellm/pull/10809) +- **[Anthropic](https://docs.litellm.ai/docs/providers/anthropic)** + - [`/chat/completion`](https://docs.litellm.ai/docs/providers/anthropic#usage-with-litellm-proxy) + - Web search tool support - native + openai format - [Get Started](https://docs.litellm.ai/docs/providers/anthropic#anthropic-hosted-tools-computer-text-editor-web-search) +- **[VLLM](https://docs.litellm.ai/docs/providers/vllm)** + - [`/embeddings`](https://docs.litellm.ai/docs/providers/vllm#embeddings) + - Support embedding input as list of integers +- **[OpenAI](https://docs.litellm.ai/docs/providers/openai)** + - [`/chat/completion`](https://docs.litellm.ai/docs/providers/openai#usage---litellm-proxy-server) + - Fix - b64 file data input handling - [Get Started](https://docs.litellm.ai/docs/providers/openai#pdf-file-parsing) + - Add ‘supports\_pdf\_input’ to all vision models - [PR](https://github.com/BerriAI/litellm/pull/10897) + +## LLM API Endpoints [​](https://docs.litellm.ai/release_notes\#llm-api-endpoints "Direct link to LLM API Endpoints") + +- [**Responses API**](https://docs.litellm.ai/docs/response_api) + - Fix delete API support - [PR](https://github.com/BerriAI/litellm/pull/10845) +- [**Rerank API**](https://docs.litellm.ai/docs/rerank) + - `/v2/rerank` now registered as ‘llm\_api\_route’ - enabling non-admins to call it - [PR](https://github.com/BerriAI/litellm/pull/10861) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **`/chat/completion`, `/messages`** + - Anthropic - web search tool cost tracking - [PR](https://github.com/BerriAI/litellm/pull/10846) + - Groq - update model max tokens + cost information - [PR](https://github.com/BerriAI/litellm/pull/10077) +- **`/audio/transcription`** + - Azure - Add gpt-4o-mini-tts pricing - [PR](https://github.com/BerriAI/litellm/pull/10807) + - Proxy - Fix tracking spend by tag - [PR](https://github.com/BerriAI/litellm/pull/10832) +- **`/embeddings`** + - Azure AI - Add cohere embed v4 pricing - [PR](https://github.com/BerriAI/litellm/pull/10806) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +- **Models** + - Ollama - adds api base param to UI +- **Logs** + - Add team id, key alias, key hash filter on logs - [https://github.com/BerriAI/litellm/pull/10831](https://github.com/BerriAI/litellm/pull/10831) + - Guardrail tracing now in Logs UI - [https://github.com/BerriAI/litellm/pull/10893](https://github.com/BerriAI/litellm/pull/10893) +- **Teams** + - Patch for updating team info when team in org and members not in org - [https://github.com/BerriAI/litellm/pull/10835](https://github.com/BerriAI/litellm/pull/10835) +- **Guardrails** + - Add Bedrock, Presidio, Lakers guardrails on UI - [https://github.com/BerriAI/litellm/pull/10874](https://github.com/BerriAI/litellm/pull/10874) + - See guardrail info page - [https://github.com/BerriAI/litellm/pull/10904](https://github.com/BerriAI/litellm/pull/10904) + - Allow editing guardrails on UI - [https://github.com/BerriAI/litellm/pull/10907](https://github.com/BerriAI/litellm/pull/10907) +- **Test Key** + - select guardrails to test on UI + +## Logging / Alerting Integrations [​](https://docs.litellm.ai/release_notes\#logging--alerting-integrations "Direct link to Logging / Alerting Integrations") + +- **[StandardLoggingPayload](https://docs.litellm.ai/docs/proxy/logging_spec)** + - Log any `x-` headers in requester metadata - [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec#standardloggingmetadata) + - Guardrail tracing now in standard logging payload - [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec#standardloggingguardrailinformation) +- **[Generic API Logger](https://docs.litellm.ai/docs/proxy/logging#custom-callback-apis-async)** + - Support passing application/json header +- **[Arize Phoenix](https://docs.litellm.ai/docs/observability/phoenix_integration)** + - fix: URL encode OTEL\_EXPORTER\_OTLP\_TRACES\_HEADERS for Phoenix Integration - [PR](https://github.com/BerriAI/litellm/pull/10654) + - add guardrail tracing to OTEL, Arize phoenix - [PR](https://github.com/BerriAI/litellm/pull/10896) +- **[PagerDuty](https://docs.litellm.ai/docs/proxy/pagerduty)** + - Pagerduty is now a free feature - [PR](https://github.com/BerriAI/litellm/pull/10857) +- **[Alerting](https://docs.litellm.ai/docs/proxy/alerting)** + - Sending slack alerts on virtual key/user/team updates is now free - [PR](https://github.com/BerriAI/litellm/pull/10863) + +## Guardrails [​](https://docs.litellm.ai/release_notes\#guardrails "Direct link to Guardrails") + +- **Guardrails** + - New `/apply_guardrail` endpoint for directly testing a guardrail - [PR](https://github.com/BerriAI/litellm/pull/10867) +- **[Lakera](https://docs.litellm.ai/docs/proxy/guardrails/lakera_ai)** + - `/v2` endpoints support - [PR](https://github.com/BerriAI/litellm/pull/10880) +- **[Presidio](https://docs.litellm.ai/docs/proxy/guardrails/pii_masking_v2)** + - Fixes handling of message content on presidio guardrail integration - [PR](https://github.com/BerriAI/litellm/pull/10197) + - Allow specifying PII Entities Config - [PR](https://github.com/BerriAI/litellm/pull/10810) +- **[Aim Security](https://docs.litellm.ai/docs/proxy/guardrails/aim_security)** + - Support for anonymization in AIM Guardrails - [PR](https://github.com/BerriAI/litellm/pull/10757) + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +- **Allow overriding all constants using a .env variable** \- [PR](https://github.com/BerriAI/litellm/pull/10803) +- **[Maximum retention period for spend logs](https://docs.litellm.ai/docs/proxy/spend_logs_deletion)** + - Add retention flag to config - [PR](https://github.com/BerriAI/litellm/pull/10815) + - Support for cleaning up logs based on configured time period - [PR](https://github.com/BerriAI/litellm/pull/10872) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Authentication** + - Handle Bearer $LITELLM\_API\_KEY in x-litellm-api-key custom header [PR](https://github.com/BerriAI/litellm/pull/10776) +- **New Enterprise pip package** \- `litellm-enterprise` \- fixes issue where `enterprise` folder was not found when using pip package +- **[Proxy CLI](https://docs.litellm.ai/docs/proxy/management_cli)** + - Add `models import` command - [PR](https://github.com/BerriAI/litellm/pull/10581) +- **[OpenWebUI](https://docs.litellm.ai/docs/tutorials/openweb_ui#per-user-tracking)** + - Configure LiteLLM to Parse User Headers from Open Web UI +- **[LiteLLM Proxy w/ LiteLLM SDK](https://docs.litellm.ai/docs/providers/litellm_proxy#send-all-sdk-requests-to-litellm-proxy)** + - Option to force/always use the litellm proxy when calling via LiteLLM SDK + +## New Contributors [​](https://docs.litellm.ai/release_notes\#new-contributors "Direct link to New Contributors") + +- [@imdigitalashish](https://github.com/imdigitalashish) made their first contribution in PR [#10617](https://github.com/BerriAI/litellm/pull/10617) +- [@LouisShark](https://github.com/LouisShark) made their first contribution in PR [#10688](https://github.com/BerriAI/litellm/pull/10688) +- [@OscarSavNS](https://github.com/OscarSavNS) made their first contribution in PR [#10764](https://github.com/BerriAI/litellm/pull/10764) +- [@arizedatngo](https://github.com/arizedatngo) made their first contribution in PR [#10654](https://github.com/BerriAI/litellm/pull/10654) +- [@jugaldb](https://github.com/jugaldb) made their first contribution in PR [#10805](https://github.com/BerriAI/litellm/pull/10805) +- [@daikeren](https://github.com/daikeren) made their first contribution in PR [#10781](https://github.com/BerriAI/litellm/pull/10781) +- [@naliotopier](https://github.com/naliotopier) made their first contribution in PR [#10077](https://github.com/BerriAI/litellm/pull/10077) +- [@damienpontifex](https://github.com/damienpontifex) made their first contribution in PR [#10813](https://github.com/BerriAI/litellm/pull/10813) +- [@Dima-Mediator](https://github.com/Dima-Mediator) made their first contribution in PR [#10789](https://github.com/BerriAI/litellm/pull/10789) +- [@igtm](https://github.com/igtm) made their first contribution in PR [#10814](https://github.com/BerriAI/litellm/pull/10814) +- [@shibaboy](https://github.com/shibaboy) made their first contribution in PR [#10752](https://github.com/BerriAI/litellm/pull/10752) +- [@camfarineau](https://github.com/camfarineau) made their first contribution in PR [#10629](https://github.com/BerriAI/litellm/pull/10629) +- [@ajac-zero](https://github.com/ajac-zero) made their first contribution in PR [#10439](https://github.com/BerriAI/litellm/pull/10439) +- [@damgem](https://github.com/damgem) made their first contribution in PR [#9802](https://github.com/BerriAI/litellm/pull/9802) +- [@hxdror](https://github.com/hxdror) made their first contribution in PR [#10757](https://github.com/BerriAI/litellm/pull/10757) +- [@wwwillchen](https://github.com/wwwillchen) made their first contribution in PR [#10894](https://github.com/BerriAI/litellm/pull/10894) + +## Demo Instance [​](https://docs.litellm.ai/release_notes\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## [Git Diff](https://github.com/BerriAI/litellm/releases) [​](https://docs.litellm.ai/release_notes\#git-diff "Direct link to git-diff") + +## Deploy this version [​](https://docs.litellm.ai/release_notes\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.69.0-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.69.0.post1 + +``` + +## Key Highlights [​](https://docs.litellm.ai/release_notes\#key-highlights "Direct link to Key Highlights") + +LiteLLM v1.69.0-stable brings the following key improvements: + +- **Loadbalance Batch API Models**: Easily loadbalance across multiple azure batch deployments using LiteLLM Managed Files +- **Email Invites 2.0**: Send new users onboarded to LiteLLM an email invite. +- **Nscale**: LLM API for compliance with European regulations. +- **Bedrock /v1/messages**: Use Bedrock Anthropic models with Anthropic's /v1/messages. + +## Batch API Load Balancing [​](https://docs.litellm.ai/release_notes\#batch-api-load-balancing "Direct link to Batch API Load Balancing") + +![](https://docs.litellm.ai/assets/ideal-img/lb_batch.40626de.1920.png) + +This release brings LiteLLM Managed File support to Batches. This is great for: + +- Proxy Admins: You can now control which Batch models users can call. +- Developers: You no longer need to know the Azure deployment name when creating your batch .jsonl files - just specify the model your LiteLLM key has access to. + +Over time, we expect LiteLLM Managed Files to be the way most teams use Files across `/chat/completions`, `/batch`, `/fine_tuning` endpoints. + +[Read more here](https://docs.litellm.ai/docs/proxy/managed_batches) + +## Email Invites [​](https://docs.litellm.ai/release_notes\#email-invites "Direct link to Email Invites") + +![](https://docs.litellm.ai/assets/ideal-img/email_2_0.61b79ad.1920.png) + +This release brings the following improvements to our email invite integration: + +- New templates for user invited and key created events. +- Fixes for using SMTP email providers. +- Native support for Resend API. +- Ability for Proxy Admins to control email events. + +For LiteLLM Cloud Users, please reach out to us if you want this enabled for your instance. + +[Read more here](https://docs.litellm.ai/docs/proxy/email) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **Gemini ( [VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) \+ [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))** + - Added `gemini-2.5-pro-preview-05-06` models with pricing and context window info - [PR](https://github.com/BerriAI/litellm/pull/10597) + - Set correct context window length for all Gemini 2.5 variants - [PR](https://github.com/BerriAI/litellm/pull/10690) +- **[Perplexity](https://docs.litellm.ai/docs/providers/perplexity)**: + - Added new Perplexity models - [PR](https://github.com/BerriAI/litellm/pull/10652) + - Added sonar-deep-research model pricing - [PR](https://github.com/BerriAI/litellm/pull/10537) +- **[Azure OpenAI](https://docs.litellm.ai/docs/providers/azure)**: + - Fixed passing through of azure\_ad\_token\_provider parameter - [PR](https://github.com/BerriAI/litellm/pull/10694) +- **[OpenAI](https://docs.litellm.ai/docs/providers/openai)**: + - Added support for pdf url's in 'file' parameter - [PR](https://github.com/BerriAI/litellm/pull/10640) +- **[Sagemaker](https://docs.litellm.ai/docs/providers/aws_sagemaker)**: + - Fix content length for `sagemaker_chat` provider - [PR](https://github.com/BerriAI/litellm/pull/10607) +- **[Azure AI Foundry](https://docs.litellm.ai/docs/providers/azure_ai)**: + - Added cost tracking for the following models [PR](https://github.com/BerriAI/litellm/pull/9956) + - DeepSeek V3 0324 + - Llama 4 Scout + - Llama 4 Maverick +- **[Bedrock](https://docs.litellm.ai/docs/providers/bedrock)**: + - Added cost tracking for Bedrock Llama 4 models - [PR](https://github.com/BerriAI/litellm/pull/10582) + - Fixed template conversion for Llama 4 models in Bedrock - [PR](https://github.com/BerriAI/litellm/pull/10582) + - Added support for using Bedrock Anthropic models with /v1/messages format - [PR](https://github.com/BerriAI/litellm/pull/10681) + - Added streaming support for Bedrock Anthropic models with /v1/messages format - [PR](https://github.com/BerriAI/litellm/pull/10710) +- **[OpenAI](https://docs.litellm.ai/docs/providers/openai)**: Added `reasoning_effort` support for `o3` models - [PR](https://github.com/BerriAI/litellm/pull/10591) +- **[Databricks](https://docs.litellm.ai/docs/providers/databricks)**: + - Fixed issue when Databricks uses external model and delta could be empty - [PR](https://github.com/BerriAI/litellm/pull/10540) +- **[Cerebras](https://docs.litellm.ai/docs/providers/cerebras)**: Fixed Llama-3.1-70b model pricing and context window - [PR](https://github.com/BerriAI/litellm/pull/10648) +- **[Ollama](https://docs.litellm.ai/docs/providers/ollama)**: + - Fixed custom price cost tracking and added 'max\_completion\_token' support - [PR](https://github.com/BerriAI/litellm/pull/10636) + - Fixed KeyError when using JSON response format - [PR](https://github.com/BerriAI/litellm/pull/10611) +- 🆕 **[Nscale](https://docs.litellm.ai/docs/providers/nscale)**: + - Added support for chat, image generation endpoints - [PR](https://github.com/BerriAI/litellm/pull/10638) + +## LLM API Endpoints [​](https://docs.litellm.ai/release_notes\#llm-api-endpoints "Direct link to LLM API Endpoints") + +- **[Messages API](https://docs.litellm.ai/docs/anthropic_unified)**: + - 🆕 Added support for using Bedrock Anthropic models with /v1/messages format - [PR](https://github.com/BerriAI/litellm/pull/10681) and streaming support - [PR](https://github.com/BerriAI/litellm/pull/10710) +- **[Moderations API](https://docs.litellm.ai/docs/moderations)**: + - Fixed bug to allow using LiteLLM UI credentials for /moderations API - [PR](https://github.com/BerriAI/litellm/pull/10723) +- **[Realtime API](https://docs.litellm.ai/docs/realtime)**: + - Fixed setting 'headers' in scope for websocket auth requests and infinite loop issues - [PR](https://github.com/BerriAI/litellm/pull/10679) +- **[Files API](https://docs.litellm.ai/docs/proxy/litellm_managed_files)**: + - Unified File ID output support - [PR](https://github.com/BerriAI/litellm/pull/10713) + - Support for writing files to all deployments - [PR](https://github.com/BerriAI/litellm/pull/10708) + - Added target model name validation - [PR](https://github.com/BerriAI/litellm/pull/10722) +- **[Batches API](https://docs.litellm.ai/docs/batches)**: + - Complete unified batch ID support - replacing model in jsonl to be deployment model name - [PR](https://github.com/BerriAI/litellm/pull/10719) + - Beta support for unified file ID (managed files) for batches - [PR](https://github.com/BerriAI/litellm/pull/10650) + +## Spend Tracking / Budget Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking--budget-improvements "Direct link to Spend Tracking / Budget Improvements") + +- Bug Fix - PostgreSQL Integer Overflow Error in DB Spend Tracking - [PR](https://github.com/BerriAI/litellm/pull/10697) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +- **Models** + - Fixed model info overwriting when editing a model on UI - [PR](https://github.com/BerriAI/litellm/pull/10726) + - Fixed team admin model updates and organization creation with specific models - [PR](https://github.com/BerriAI/litellm/pull/10539) +- **Logs**: + - Bug Fix - copying Request/Response on Logs Page - [PR](https://github.com/BerriAI/litellm/pull/10720) + - Bug Fix - log did not remain in focus on QA Logs page + text overflow on error logs - [PR](https://github.com/BerriAI/litellm/pull/10725) + - Added index for session\_id on LiteLLM\_SpendLogs for better query performance - [PR](https://github.com/BerriAI/litellm/pull/10727) +- **User Management**: + - Added user management functionality to Python client library & CLI - [PR](https://github.com/BerriAI/litellm/pull/10627) + - Bug Fix - Fixed SCIM token creation on Admin UI - [PR](https://github.com/BerriAI/litellm/pull/10628) + - Bug Fix - Added 404 response when trying to delete verification tokens that don't exist - [PR](https://github.com/BerriAI/litellm/pull/10605) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +- **Custom Logger API**: v2 Custom Callback API (send llm logs to custom api) - [PR](https://github.com/BerriAI/litellm/pull/10575), [Get Started](https://docs.litellm.ai/docs/proxy/logging#custom-callback-apis-async) +- **OpenTelemetry**: + - Fixed OpenTelemetry to follow genai semantic conventions + support for 'instructions' param for TTS - [PR](https://github.com/BerriAI/litellm/pull/10608) +- **Bedrock PII**: + - Add support for PII Masking with bedrock guardrails - [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/bedrock#pii-masking-with-bedrock-guardrails), [PR](https://github.com/BerriAI/litellm/pull/10608) +- **Documentation**: + - Added documentation for StandardLoggingVectorStoreRequest - [PR](https://github.com/BerriAI/litellm/pull/10535) + +## Performance / Reliability Improvements [​](https://docs.litellm.ai/release_notes\#performance--reliability-improvements "Direct link to Performance / Reliability Improvements") + +- **Python Compatibility**: + - Added support for Python 3.11- (fixed datetime UTC handling) - [PR](https://github.com/BerriAI/litellm/pull/10701) + - Fixed UnicodeDecodeError: 'charmap' on Windows during litellm import - [PR](https://github.com/BerriAI/litellm/pull/10542) +- **Caching**: + - Fixed embedding string caching result - [PR](https://github.com/BerriAI/litellm/pull/10700) + - Fixed cache miss for Gemini models with response\_format - [PR](https://github.com/BerriAI/litellm/pull/10635) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Proxy CLI**: + - Added `--version` flag to `litellm-proxy` CLI - [PR](https://github.com/BerriAI/litellm/pull/10704) + - Added dedicated `litellm-proxy` CLI - [PR](https://github.com/BerriAI/litellm/pull/10578) +- **Alerting**: + - Fixed Slack alerting not working when using a DB - [PR](https://github.com/BerriAI/litellm/pull/10370) +- **Email Invites**: + - Added V2 Emails with fixes for sending emails when creating keys + Resend API support - [PR](https://github.com/BerriAI/litellm/pull/10602) + - Added user invitation emails - [PR](https://github.com/BerriAI/litellm/pull/10615) + - Added endpoints to manage email settings - [PR](https://github.com/BerriAI/litellm/pull/10646) +- **General**: + - Fixed bug where duplicate JSON logs were getting emitted - [PR](https://github.com/BerriAI/litellm/pull/10580) + +## New Contributors [​](https://docs.litellm.ai/release_notes\#new-contributors "Direct link to New Contributors") + +- [@zoltan-ongithub](https://github.com/zoltan-ongithub) made their first contribution in [PR #10568](https://github.com/BerriAI/litellm/pull/10568) +- [@mkavinkumar1](https://github.com/mkavinkumar1) made their first contribution in [PR #10548](https://github.com/BerriAI/litellm/pull/10548) +- [@thomelane](https://github.com/thomelane) made their first contribution in [PR #10549](https://github.com/BerriAI/litellm/pull/10549) +- [@frankzye](https://github.com/frankzye) made their first contribution in [PR #10540](https://github.com/BerriAI/litellm/pull/10540) +- [@aholmberg](https://github.com/aholmberg) made their first contribution in [PR #10591](https://github.com/BerriAI/litellm/pull/10591) +- [@aravindkarnam](https://github.com/aravindkarnam) made their first contribution in [PR #10611](https://github.com/BerriAI/litellm/pull/10611) +- [@xsg22](https://github.com/xsg22) made their first contribution in [PR #10648](https://github.com/BerriAI/litellm/pull/10648) +- [@casparhsws](https://github.com/casparhsws) made their first contribution in [PR #10635](https://github.com/BerriAI/litellm/pull/10635) +- [@hypermoose](https://github.com/hypermoose) made their first contribution in [PR #10370](https://github.com/BerriAI/litellm/pull/10370) +- [@tomukmatthews](https://github.com/tomukmatthews) made their first contribution in [PR #10638](https://github.com/BerriAI/litellm/pull/10638) +- [@keyute](https://github.com/keyute) made their first contribution in [PR #10652](https://github.com/BerriAI/litellm/pull/10652) +- [@GPTLocalhost](https://github.com/GPTLocalhost) made their first contribution in [PR #10687](https://github.com/BerriAI/litellm/pull/10687) +- [@husnain7766](https://github.com/husnain7766) made their first contribution in [PR #10697](https://github.com/BerriAI/litellm/pull/10697) +- [@claralp](https://github.com/claralp) made their first contribution in [PR #10694](https://github.com/BerriAI/litellm/pull/10694) +- [@mollux](https://github.com/mollux) made their first contribution in [PR #10690](https://github.com/BerriAI/litellm/pull/10690) + +## Deploy this version [​](https://docs.litellm.ai/release_notes\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.68.0-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.68.0.post1 + +``` + +## Key Highlights [​](https://docs.litellm.ai/release_notes\#key-highlights "Direct link to Key Highlights") + +LiteLLM v1.68.0-stable will be live soon. Here are the key highlights of this release: + +- **Bedrock Knowledge Base**: You can now call query your Bedrock Knowledge Base with all LiteLLM models via `/chat/completion` or `/responses` API. +- **Rate Limits**: This release brings accurate rate limiting across multiple instances, reducing spillover to at most 10 additional requests in high traffic. +- **Meta Llama API**: Added support for Meta Llama API [Get Started](https://docs.litellm.ai/docs/providers/meta_llama) +- **LlamaFile**: Added support for LlamaFile [Get Started](https://docs.litellm.ai/docs/providers/llamafile) + +## Bedrock Knowledge Base (Vector Store) [​](https://docs.litellm.ai/release_notes\#bedrock-knowledge-base-vector-store "Direct link to Bedrock Knowledge Base (Vector Store)") + +![](https://docs.litellm.ai/assets/ideal-img/bedrock_kb.0b661ae.1920.png) + +This release adds support for Bedrock vector stores (knowledge bases) in LiteLLM. With this update, you can: + +- Use Bedrock vector stores in the OpenAI /chat/completions spec with all LiteLLM supported models. +- View all available vector stores through the LiteLLM UI or API. +- Configure vector stores to be always active for specific models. +- Track vector store usage in LiteLLM Logs. + +For the next release we plan on allowing you to set key, user, team, org permissions for vector stores. + +[Read more here](https://docs.litellm.ai/docs/completion/knowledgebase) + +## Rate Limiting [​](https://docs.litellm.ai/release_notes\#rate-limiting "Direct link to Rate Limiting") + +![](https://docs.litellm.ai/assets/ideal-img/multi_instance_rate_limiting.06ee750.1800.png) + +This release brings accurate multi-instance rate limiting across keys/users/teams. Outlining key engineering changes below: + +- **Change**: Instances now increment cache value instead of setting it. To avoid calling Redis on each request, this is synced every 0.01s. +- **Accuracy**: In testing, we saw a maximum spill over from expected of 10 requests, in high traffic (100 RPS, 3 instances), vs. current 189 request spillover +- **Performance**: Our load tests show this to reduce median response time by 100ms in high traffic + +This is currently behind a feature flag, and we plan to have this be the default by next week. To enable this today, just add this environment variable: + +```codeBlockLines_e6Vv +export LITELLM_RATE_LIMIT_ACCURACY=true + +``` + +[Read more here](https://docs.litellm.ai/docs/proxy/users#beta-multi-instance-rate-limiting) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **Gemini ( [VertexAI](https://docs.litellm.ai/docs/providers/vertex#usage-with-litellm-proxy-server) \+ [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini))** + - Handle more json schema - openapi schema conversion edge cases [PR](https://github.com/BerriAI/litellm/pull/10351) + - Tool calls - return ‘finish\_reason=“tool\_calls”’ on gemini tool calling response [PR](https://github.com/BerriAI/litellm/pull/10485) +- **[VertexAI](https://docs.litellm.ai/docs/providers/vertex#metallama-api)** + - Meta/llama-4 model support [PR](https://github.com/BerriAI/litellm/pull/10492) + - Meta/llama3 - handle tool call result in content [PR](https://github.com/BerriAI/litellm/pull/10492) + - Meta/\* - return ‘finish\_reason=“tool\_calls”’ on tool calling response [PR](https://github.com/BerriAI/litellm/pull/10492) +- **[Bedrock](https://docs.litellm.ai/docs/providers/bedrock#litellm-proxy-usage)** + - [Image Generation](https://docs.litellm.ai/docs/providers/bedrock#image-generation) \- Support new ‘stable-image-core’ models - [PR](https://github.com/BerriAI/litellm/pull/10351) + - [Knowledge Bases](https://docs.litellm.ai/docs/completion/knowledgebase) \- support using Bedrock knowledge bases with `/chat/completions` [PR](https://github.com/BerriAI/litellm/pull/10413) + - [Anthropic](https://docs.litellm.ai/docs/providers/bedrock#litellm-proxy-usage) \- add ‘supports\_pdf\_input’ for claude-3.7-bedrock models [PR](https://github.com/BerriAI/litellm/pull/9917), [Get Started](https://docs.litellm.ai/docs/completion/document_understanding#checking-if-a-model-supports-pdf-input) +- **[OpenAI](https://docs.litellm.ai/docs/providers/openai)** + - Support OPENAI\_BASE\_URL in addition to OPENAI\_API\_BASE [PR](https://github.com/BerriAI/litellm/pull/10423) + - Correctly re-raise 504 timeout errors [PR](https://github.com/BerriAI/litellm/pull/10462) + - Native Gpt-4o-mini-tts support [PR](https://github.com/BerriAI/litellm/pull/10462) +- 🆕 **[Meta Llama API](https://docs.litellm.ai/docs/providers/meta_llama)** provider [PR](https://github.com/BerriAI/litellm/pull/10451) +- 🆕 **[LlamaFile](https://docs.litellm.ai/docs/providers/llamafile)** provider [PR](https://github.com/BerriAI/litellm/pull/10482) + +## LLM API Endpoints [​](https://docs.litellm.ai/release_notes\#llm-api-endpoints "Direct link to LLM API Endpoints") + +- **[Response API](https://docs.litellm.ai/docs/response_api)** + - Fix for handling multi turn sessions [PR](https://github.com/BerriAI/litellm/pull/10415) +- **[Embeddings](https://docs.litellm.ai/docs/embedding/supported_embedding)** + - Caching fixes - [PR](https://github.com/BerriAI/litellm/pull/10424) + - handle str -> list cache + - Return usage tokens for cache hit + - Combine usage tokens on partial cache hits +- 🆕 **[Vector Stores](https://docs.litellm.ai/docs/completion/knowledgebase)** + - Allow defining Vector Store Configs - [PR](https://github.com/BerriAI/litellm/pull/10448) + - New StandardLoggingPayload field for requests made when a vector store is used - [PR](https://github.com/BerriAI/litellm/pull/10509) + - Show Vector Store / KB Request on LiteLLM Logs Page - [PR](https://github.com/BerriAI/litellm/pull/10514) + - Allow using vector store in OpenAI API spec with tools - [PR](https://github.com/BerriAI/litellm/pull/10516) +- **[MCP](https://docs.litellm.ai/docs/mcp)** + - Ensure Non-Admin virtual keys can access /mcp routes - [PR](https://github.com/BerriAI/litellm/pull/10473) + + **Note:** Currently, all Virtual Keys are able to access the MCP endpoints. We are working on a feature to allow restricting MCP access by keys/teams/users/orgs. Follow [here](https://github.com/BerriAI/litellm/discussions/9891) for updates. +- **Moderations** + - Add logging callback support for `/moderations` API - [PR](https://github.com/BerriAI/litellm/pull/10390) + +## Spend Tracking / Budget Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking--budget-improvements "Direct link to Spend Tracking / Budget Improvements") + +- **[OpenAI](https://docs.litellm.ai/docs/providers/openai)** + - [computer-use-preview](https://docs.litellm.ai/docs/providers/openai/responses_api#computer-use) cost tracking / pricing [PR](https://github.com/BerriAI/litellm/pull/10422) + - [gpt-4o-mini-tts](https://docs.litellm.ai/docs/providers/openai/text_to_speech) input cost tracking - [PR](https://github.com/BerriAI/litellm/pull/10462) +- **[Fireworks AI](https://docs.litellm.ai/docs/providers/fireworks_ai)** \- pricing updates - new `0-4b` model pricing tier + llama4 model pricing +- **[Budgets](https://docs.litellm.ai/docs/proxy/users#set-budgets)** + - [Budget resets](https://docs.litellm.ai/docs/proxy/users#reset-budgets) now happen as start of day/week/month - [PR](https://github.com/BerriAI/litellm/pull/10333) + - Trigger [Soft Budget Alerts](https://docs.litellm.ai/docs/proxy/alerting#soft-budget-alerts-for-virtual-keys) When Key Crosses Threshold - [PR](https://github.com/BerriAI/litellm/pull/10491) +- **[Token Counting](https://docs.litellm.ai/docs/completion/token_usage#3-token_counter)** + - Rewrite of token\_counter() function to handle to prevent undercounting tokens - [PR](https://github.com/BerriAI/litellm/pull/10409) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +- **Virtual Keys** + - Fix filtering on key alias - [PR](https://github.com/BerriAI/litellm/pull/10455) + - Support global filtering on keys - [PR](https://github.com/BerriAI/litellm/pull/10455) + - Pagination - fix clicking on next/back buttons on table - [PR](https://github.com/BerriAI/litellm/pull/10528) +- **Models** + - Triton - Support adding model/provider on UI - [PR](https://github.com/BerriAI/litellm/pull/10456) + - VertexAI - Fix adding vertex models with reusable credentials - [PR](https://github.com/BerriAI/litellm/pull/10528) + - LLM Credentials - show existing credentials for easy editing - [PR](https://github.com/BerriAI/litellm/pull/10519) +- **Teams** + - Allow reassigning team to other org - [PR](https://github.com/BerriAI/litellm/pull/10527) +- **Organizations** + - Fix showing org budget on table - [PR](https://github.com/BerriAI/litellm/pull/10528) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +- **[Langsmith](https://docs.litellm.ai/docs/observability/langsmith_integration)** + - Respect [langsmith\_batch\_size](https://docs.litellm.ai/docs/observability/langsmith_integration#local-testing---control-batch-size) param - [PR](https://github.com/BerriAI/litellm/pull/10411) + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +- **[Redis](https://docs.litellm.ai/docs/proxy/caching)** + - Ensure all redis queues are periodically flushed, this fixes an issue where redis queue size was growing indefinitely when request tags were used - [PR](https://github.com/BerriAI/litellm/pull/10393) +- **[Rate Limits](https://docs.litellm.ai/docs/proxy/users#set-rate-limit)** + - [Multi-instance rate limiting](https://docs.litellm.ai/docs/proxy/users#beta-multi-instance-rate-limiting) support across keys/teams/users/customers - [PR](https://github.com/BerriAI/litellm/pull/10458), [PR](https://github.com/BerriAI/litellm/pull/10497), [PR](https://github.com/BerriAI/litellm/pull/10500) +- **[Azure OpenAI OIDC](https://docs.litellm.ai/docs/providers/azure#entra-id---use-azure_ad_token)** + - allow using litellm defined params for [OIDC Auth](https://docs.litellm.ai/docs/providers/azure#entra-id---use-azure_ad_token) \- [PR](https://github.com/BerriAI/litellm/pull/10394) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Security** + - Allow [blocking web crawlers](https://docs.litellm.ai/docs/proxy/enterprise#blocking-web-crawlers) \- [PR](https://github.com/BerriAI/litellm/pull/10420) +- **Auth** + - Support [`x-litellm-api-key` header param by default](https://docs.litellm.ai/docs/pass_through/vertex_ai#use-with-virtual-keys), this fixes an issue from the prior release where `x-litellm-api-key` was not being used on vertex ai passthrough requests - [PR](https://github.com/BerriAI/litellm/pull/10392) + - Allow key at max budget to call non-llm api endpoints - [PR](https://github.com/BerriAI/litellm/pull/10392) +- 🆕 **[Python Client Library](https://docs.litellm.ai/docs/proxy/management_cli) for LiteLLM Proxy management endpoints** + - Initial PR - [PR](https://github.com/BerriAI/litellm/pull/10445) + - Support for doing HTTP requests - [PR](https://github.com/BerriAI/litellm/pull/10452) +- **Dependencies** + - Don’t require uvloop for windows - [PR](https://github.com/BerriAI/litellm/pull/10483) + +## Deploy this version [​](https://docs.litellm.ai/release_notes\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.67.4-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.67.4.post1 + +``` + +## Key Highlights [​](https://docs.litellm.ai/release_notes\#key-highlights "Direct link to Key Highlights") + +- **Improved User Management**: This release enables search and filtering across users, keys, teams, and models. +- **Responses API Load Balancing**: Route requests across provider regions and ensure session continuity. +- **UI Session Logs**: Group several requests to LiteLLM into a session. + +## Improved User Management [​](https://docs.litellm.ai/release_notes\#improved-user-management "Direct link to Improved User Management") + +![](https://docs.litellm.ai/assets/ideal-img/ui_search_users.7472bdc.1920.png) + +This release makes it easier to manage users and keys on LiteLLM. You can now search and filter across users, keys, teams, and models, and control user settings more easily. + +New features include: + +- Search for users by email, ID, role, or team. +- See all of a user's models, teams, and keys in one place. +- Change user roles and model access right from the Users Tab. + +These changes help you spend less time on user setup and management on LiteLLM. + +## Responses API Load Balancing [​](https://docs.litellm.ai/release_notes\#responses-api-load-balancing "Direct link to Responses API Load Balancing") + +![](https://docs.litellm.ai/assets/ideal-img/ui_responses_lb.1e64cec.1204.png) + +This release introduces load balancing for the Responses API, allowing you to route requests across provider regions and ensure session continuity. It works as follows: + +- If a `previous_response_id` is provided, LiteLLM will route the request to the original deployment that generated the prior response — ensuring session continuity. +- If no `previous_response_id` is provided, LiteLLM will load-balance requests across your available deployments. + +[Read more](https://docs.litellm.ai/docs/response_api#load-balancing-with-session-continuity) + +## UI Session Logs [​](https://docs.litellm.ai/release_notes\#ui-session-logs "Direct link to UI Session Logs") + +![](https://docs.litellm.ai/assets/ideal-img/ui_session_logs.926dffc.1920.png) + +This release allow you to group requests to LiteLLM proxy into a session. If you specify a litellm\_session\_id in your request LiteLLM will automatically group all logs in the same session. This allows you to easily track usage and request content per session. + +[Read more](https://docs.litellm.ai/docs/proxy/ui_logs_sessions) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **OpenAI** +1. Added `gpt-image-1` cost tracking [Get Started](https://docs.litellm.ai/docs/image_generation) +2. Bug fix: added cost tracking for gpt-image-1 when quality is unspecified [PR](https://github.com/BerriAI/litellm/pull/10247) +- **Azure** +1. Fixed timestamp granularities passing to whisper in Azure [Get Started](https://docs.litellm.ai/docs/audio_transcription) +2. Added azure/gpt-image-1 pricing [Get Started](https://docs.litellm.ai/docs/image_generation), [PR](https://github.com/BerriAI/litellm/pull/10327) +3. Added cost tracking for `azure/computer-use-preview`, `azure/gpt-4o-audio-preview-2024-12-17`, `azure/gpt-4o-mini-audio-preview-2024-12-17` [PR](https://github.com/BerriAI/litellm/pull/10178) +- **Bedrock** +1. Added support for all compatible Bedrock parameters when model="arn:.." (Bedrock application inference profile models) [Get started](https://docs.litellm.ai/docs/providers/bedrock#bedrock-application-inference-profile), [PR](https://github.com/BerriAI/litellm/pull/10256) +2. Fixed wrong system prompt transformation [PR](https://github.com/BerriAI/litellm/pull/10120) +- **VertexAI / Google AI Studio** +1. Allow setting `budget_tokens=0` for `gemini-2.5-flash` [Get Started](https://docs.litellm.ai/docs/providers/gemini#usage---thinking--reasoning_content), [PR](https://github.com/BerriAI/litellm/pull/10198) +2. Ensure returned `usage` includes thinking token usage [PR](https://github.com/BerriAI/litellm/pull/10198) +3. Added cost tracking for `gemini-2.5-pro-preview-03-25` [PR](https://github.com/BerriAI/litellm/pull/10178) +- **Cohere** +1. Added support for cohere command-a-03-2025 [Get Started](https://docs.litellm.ai/docs/providers/cohere), [PR](https://github.com/BerriAI/litellm/pull/10295) +- **SageMaker** +1. Added support for max\_completion\_tokens parameter [Get Started](https://docs.litellm.ai/docs/providers/sagemaker), [PR](https://github.com/BerriAI/litellm/pull/10300) +- **Responses API** +1. Added support for GET and DELETE operations - `/v1/responses/{response_id}` [Get Started](https://docs.litellm.ai/docs/response_api) +2. Added session management support for non-OpenAI models [PR](https://github.com/BerriAI/litellm/pull/10321) +3. Added routing affinity to maintain model consistency within sessions [Get Started](https://docs.litellm.ai/docs/response_api#load-balancing-with-routing-affinity), [PR](https://github.com/BerriAI/litellm/pull/10193) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **Bug Fix**: Fixed spend tracking bug, ensuring default litellm params aren't modified in memory [PR](https://github.com/BerriAI/litellm/pull/10167) +- **Deprecation Dates**: Added deprecation dates for Azure, VertexAI models [PR](https://github.com/BerriAI/litellm/pull/10308) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +#### Users [​](https://docs.litellm.ai/release_notes\#users "Direct link to Users") + +- **Filtering and Searching**: + + + - Filter users by user\_id, role, team, sso\_id + - Search users by email + +![](https://docs.litellm.ai/assets/ideal-img/user_filters.e2b4a8c.1920.png) + +- **User Info Panel**: Added a new user information pane [PR](https://github.com/BerriAI/litellm/pull/10213) + + - View teams, keys, models associated with User + - Edit user role, model permissions + +#### Teams [​](https://docs.litellm.ai/release_notes\#teams "Direct link to Teams") + +- **Filtering and Searching**: + + + - Filter teams by Organization, Team ID [PR](https://github.com/BerriAI/litellm/pull/10324) + - Search teams by Team Name [PR](https://github.com/BerriAI/litellm/pull/10324) + +![](https://docs.litellm.ai/assets/ideal-img/team_filters.c9c085b.1920.png) + +#### Keys [​](https://docs.litellm.ai/release_notes\#keys "Direct link to Keys") + +- **Key Management**: + - Support for cross-filtering and filtering by key hash [PR](https://github.com/BerriAI/litellm/pull/10322) + - Fixed key alias reset when resetting filters [PR](https://github.com/BerriAI/litellm/pull/10099) + - Fixed table rendering on key creation [PR](https://github.com/BerriAI/litellm/pull/10224) + +#### UI Logs Page [​](https://docs.litellm.ai/release_notes\#ui-logs-page "Direct link to UI Logs Page") + +- **Session Logs**: Added UI Session Logs [Get Started](https://docs.litellm.ai/docs/proxy/ui_logs_sessions) + +#### UI Authentication & Security [​](https://docs.litellm.ai/release_notes\#ui-authentication--security "Direct link to UI Authentication & Security") + +- **Required Authentication**: Authentication now required for all dashboard pages [PR](https://github.com/BerriAI/litellm/pull/10229) +- **SSO Fixes**: Fixed SSO user login invalid token error [PR](https://github.com/BerriAI/litellm/pull/10298) +- \[BETA\] **Encrypted Tokens**: Moved UI to encrypted token usage [PR](https://github.com/BerriAI/litellm/pull/10302) +- **Token Expiry**: Support token refresh by re-routing to login page (fixes issue where expired token would show a blank page) [PR](https://github.com/BerriAI/litellm/pull/10250) + +#### UI General fixes [​](https://docs.litellm.ai/release_notes\#ui-general-fixes "Direct link to UI General fixes") + +- **Fixed UI Flicker**: Addressed UI flickering issues in Dashboard [PR](https://github.com/BerriAI/litellm/pull/10261) +- **Improved Terminology**: Better loading and no-data states on Keys and Tools pages [PR](https://github.com/BerriAI/litellm/pull/10253) +- **Azure Model Support**: Fixed editing Azure public model names and changing model names after creation [PR](https://github.com/BerriAI/litellm/pull/10249) +- **Team Model Selector**: Bug fix for team model selection [PR](https://github.com/BerriAI/litellm/pull/10171) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +- **Datadog**: +1. Fixed Datadog LLM observability logging [Get Started](https://docs.litellm.ai/docs/proxy/logging#datadog), [PR](https://github.com/BerriAI/litellm/pull/10206) +- **Prometheus / Grafana**: +1. Enable datasource selection on LiteLLM Grafana Template [Get Started](https://docs.litellm.ai/docs/proxy/prometheus#-litellm-maintained-grafana-dashboards-), [PR](https://github.com/BerriAI/litellm/pull/10257) +- **AgentOps**: +1. Added AgentOps Integration [Get Started](https://docs.litellm.ai/docs/observability/agentops_integration), [PR](https://github.com/BerriAI/litellm/pull/9685) +- **Arize**: +1. Added missing attributes for Arize & Phoenix Integration [Get Started](https://docs.litellm.ai/docs/observability/arize_integration), [PR](https://github.com/BerriAI/litellm/pull/10215) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Caching**: Fixed caching to account for `thinking` or `reasoning_effort` when calculating cache key [PR](https://github.com/BerriAI/litellm/pull/10140) +- **Model Groups**: Fixed handling for cases where user sets model\_group inside model\_info [PR](https://github.com/BerriAI/litellm/pull/10191) +- **Passthrough Endpoints**: Ensured `PassthroughStandardLoggingPayload` is logged with method, URL, request/response body [PR](https://github.com/BerriAI/litellm/pull/10194) +- **Fix SQL Injection**: Fixed potential SQL injection vulnerability in spend\_management\_endpoints.py [PR](https://github.com/BerriAI/litellm/pull/9878) + +## Helm [​](https://docs.litellm.ai/release_notes\#helm "Direct link to Helm") + +- Fixed serviceAccountName on migration job [PR](https://github.com/BerriAI/litellm/pull/10258) + +## Full Changelog [​](https://docs.litellm.ai/release_notes\#full-changelog "Direct link to Full Changelog") + +The complete list of changes can be found in the [GitHub release notes](https://github.com/BerriAI/litellm/compare/v1.67.0-stable...v1.67.4-stable). + +## Key Highlights [​](https://docs.litellm.ai/release_notes\#key-highlights "Direct link to Key Highlights") + +- **SCIM Integration**: Enables identity providers (Okta, Azure AD, OneLogin, etc.) to automate user and team (group) provisioning, updates, and deprovisioning +- **Team and Tag based usage tracking**: You can now see usage and spend by team and tag at 1M+ spend logs. +- **Unified Responses API**: Support for calling Anthropic, Gemini, Groq, etc. via OpenAI's new Responses API. + +Let's dive in. + +## SCIM Integration [​](https://docs.litellm.ai/release_notes\#scim-integration "Direct link to SCIM Integration") + +![](https://docs.litellm.ai/assets/ideal-img/scim_integration.01959e2.1200.png) + +This release adds SCIM support to LiteLLM. This allows your SSO provider (Okta, Azure AD, etc) to automatically create/delete users, teams, and memberships on LiteLLM. This means that when you remove a team on your SSO provider, your SSO provider will automatically delete the corresponding team on LiteLLM. + +[Read more](https://docs.litellm.ai/docs/tutorials/scim_litellm) + +## Team and Tag based usage tracking [​](https://docs.litellm.ai/release_notes\#team-and-tag-based-usage-tracking "Direct link to Team and Tag based usage tracking") + +![](https://docs.litellm.ai/assets/ideal-img/new_team_usage_highlight.60482cc.1920.jpg) + +This release improves team and tag based usage tracking at 1m+ spend logs, making it easy to monitor your LLM API Spend in production. This covers: + +- View **daily spend** by teams + tags +- View **usage / spend by key**, within teams +- View **spend by multiple tags** +- Allow **internal users** to view spend of teams they're a member of + +[Read more](https://docs.litellm.ai/release_notes#management-endpoints--ui) + +## Unified Responses API [​](https://docs.litellm.ai/release_notes\#unified-responses-api "Direct link to Unified Responses API") + +This release allows you to call Azure OpenAI, Anthropic, AWS Bedrock, and Google Vertex AI models via the POST /v1/responses endpoint on LiteLLM. This means you can now use popular tools like [OpenAI Codex](https://docs.litellm.ai/docs/tutorials/openai_codex) with your own models. + +![](https://docs.litellm.ai/assets/ideal-img/unified_responses_api_rn.0acc91a.1920.png) + +[Read more](https://docs.litellm.ai/docs/response_api) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **OpenAI** +1. gpt-4.1, gpt-4.1-mini, gpt-4.1-nano, o3, o3-mini, o4-mini pricing - [Get Started](https://docs.litellm.ai/docs/providers/openai#usage), [PR](https://github.com/BerriAI/litellm/pull/9990) +2. o4 - correctly map o4 to openai o\_series model +- **Azure AI** +1. Phi-4 output cost per token fix - [PR](https://github.com/BerriAI/litellm/pull/9880) +2. Responses API support [Get Started](https://docs.litellm.ai/docs/providers/azure#azure-responses-api), [PR](https://github.com/BerriAI/litellm/pull/10116) +- **Anthropic** +1. redacted message thinking support - [Get Started](https://docs.litellm.ai/docs/providers/anthropic#usage---thinking--reasoning_content), [PR](https://github.com/BerriAI/litellm/pull/10129) +- **Cohere** +1. `/v2/chat` Passthrough endpoint support w/ cost tracking - [Get Started](https://docs.litellm.ai/docs/pass_through/cohere), [PR](https://github.com/BerriAI/litellm/pull/9997) +- **Azure** +1. Support azure tenant\_id/client\_id env vars - [Get Started](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret), [PR](https://github.com/BerriAI/litellm/pull/9993) +2. Fix response\_format check for 2025+ api versions - [PR](https://github.com/BerriAI/litellm/pull/9993) +3. Add gpt-4.1, gpt-4.1-mini, gpt-4.1-nano, o3, o3-mini, o4-mini pricing +- **VLLM** +1. Files - Support 'file' message type for VLLM video url's - [Get Started](https://docs.litellm.ai/docs/providers/vllm#send-video-url-to-vllm), [PR](https://github.com/BerriAI/litellm/pull/10129) +2. Passthrough - new `/vllm/` passthrough endpoint support [Get Started](https://docs.litellm.ai/docs/pass_through/vllm), [PR](https://github.com/BerriAI/litellm/pull/10002) +- **Mistral** +1. new `/mistral` passthrough endpoint support [Get Started](https://docs.litellm.ai/docs/pass_through/mistral), [PR](https://github.com/BerriAI/litellm/pull/10002) +- **AWS** +1. New mapped bedrock regions - [PR](https://github.com/BerriAI/litellm/pull/9430) +- **VertexAI / Google AI Studio** +1. Gemini - Response format - Retain schema field ordering for google gemini and vertex by specifying propertyOrdering - [Get Started](https://docs.litellm.ai/docs/providers/vertex#json-schema), [PR](https://github.com/BerriAI/litellm/pull/9828) +2. Gemini-2.5-flash - return reasoning content [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini#usage---thinking--reasoning_content), [Vertex AI](https://docs.litellm.ai/docs/providers/vertex#thinking--reasoning_content) +3. Gemini-2.5-flash - pricing + model information [PR](https://github.com/BerriAI/litellm/pull/10125) +4. Passthrough - new `/vertex_ai/discovery` route - enables calling AgentBuilder API routes [Get Started](https://docs.litellm.ai/docs/pass_through/vertex_ai#supported-api-endpoints), [PR](https://github.com/BerriAI/litellm/pull/10084) +- **Fireworks AI** +1. return tool calling responses in `tool_calls` field (fireworks incorrectly returns this as a json str in content) [PR](https://github.com/BerriAI/litellm/pull/10130) +- **Triton** +1. Remove fixed remove bad\_words / stop words from `/generate` call - [Get Started](https://docs.litellm.ai/docs/providers/triton-inference-server#triton-generate---chat-completion), [PR](https://github.com/BerriAI/litellm/pull/10163) +- **Other** +1. Support for all litellm providers on Responses API (works with Codex) - [Get Started](https://docs.litellm.ai/docs/tutorials/openai_codex), [PR](https://github.com/BerriAI/litellm/pull/10132) +2. Fix combining multiple tool calls in streaming response - [Get Started](https://docs.litellm.ai/docs/completion/stream#helper-function), [PR](https://github.com/BerriAI/litellm/pull/10040) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **Cost Control** \- inject cache control points in prompt for cost reduction [Get Started](https://docs.litellm.ai/docs/tutorials/prompt_caching), [PR](https://github.com/BerriAI/litellm/pull/10000) +- **Spend Tags** \- spend tags in headers - support x-litellm-tags even if tag based routing not enabled [Get Started](https://docs.litellm.ai/docs/proxy/request_headers#litellm-headers), [PR](https://github.com/BerriAI/litellm/pull/10000) +- **Gemini-2.5-flash** \- support cost calculation for reasoning tokens [PR](https://github.com/BerriAI/litellm/pull/10141) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +- **Users** + +1. Show created\_at and updated\_at on users page - [PR](https://github.com/BerriAI/litellm/pull/10033) +- **Virtual Keys** + +1. Filter by key alias - [https://github.com/BerriAI/litellm/pull/10085](https://github.com/BerriAI/litellm/pull/10085) +- **Usage Tab** + +1. Team based usage + + + - New `LiteLLM_DailyTeamSpend` Table for aggregate team based usage logging - [PR](https://github.com/BerriAI/litellm/pull/10039) + + - New Team based usage dashboard + new `/team/daily/activity` API - [PR](https://github.com/BerriAI/litellm/pull/10081) + + - Return team alias on /team/daily/activity API - [PR](https://github.com/BerriAI/litellm/pull/10157) + + - allow internal user view spend for teams they belong to - [PR](https://github.com/BerriAI/litellm/pull/10157) + + - allow viewing top keys by team - [PR](https://github.com/BerriAI/litellm/pull/10157) + + +![](https://docs.litellm.ai/assets/ideal-img/new_team_usage.9237b43.1754.png) + +2. Tag Based Usage + + - New `LiteLLM_DailyTagSpend` Table for aggregate tag based usage logging - [PR](https://github.com/BerriAI/litellm/pull/10071) + - Restrict to only Proxy Admins - [PR](https://github.com/BerriAI/litellm/pull/10157) + - allow viewing top keys by tag + - Return tags passed in request (i.e. dynamic tags) on `/tag/list` API - [PR](https://github.com/BerriAI/litellm/pull/10157) + ![](https://docs.litellm.ai/assets/ideal-img/new_tag_usage.cd55b64.1863.png) +3. Track prompt caching metrics in daily user, team, tag tables - [PR](https://github.com/BerriAI/litellm/pull/10029) + +4. Show usage by key (on all up, team, and tag usage dashboards) - [PR](https://github.com/BerriAI/litellm/pull/10157) + +5. swap old usage with new usage tab +- **Models** + +1. Make columns resizable/hideable - [PR](https://github.com/BerriAI/litellm/pull/10119) +- **API Playground** + +1. Allow internal user to call api playground - [PR](https://github.com/BerriAI/litellm/pull/10157) +- **SCIM** + +1. Add LiteLLM SCIM Integration for Team and User management - [Get Started](https://docs.litellm.ai/docs/tutorials/scim_litellm), [PR](https://github.com/BerriAI/litellm/pull/10072) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +- **GCS** +1. Fix gcs pub sub logging with env var GCS\_PROJECT\_ID - [Get Started](https://docs.litellm.ai/docs/observability/gcs_bucket_integration#usage), [PR](https://github.com/BerriAI/litellm/pull/10042) +- **AIM** +1. Add litellm call id passing to Aim guardrails on pre and post-hooks calls - [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/aim_security), [PR](https://github.com/BerriAI/litellm/pull/10021) +- **Azure blob storage** +1. Ensure logging works in high throughput scenarios - [Get Started](https://docs.litellm.ai/docs/proxy/logging#azure-blob-storage), [PR](https://github.com/BerriAI/litellm/pull/9962) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Support setting `litellm.modify_params` via env var** [PR](https://github.com/BerriAI/litellm/pull/9964) +- **Model Discovery** \- Check provider’s `/models` endpoints when calling proxy’s `/v1/models` endpoint - [Get Started](https://docs.litellm.ai/docs/proxy/model_discovery), [PR](https://github.com/BerriAI/litellm/pull/9958) +- **`/utils/token_counter`** \- fix retrieving custom tokenizer for db models - [Get Started](https://docs.litellm.ai/docs/proxy/configs#set-custom-tokenizer), [PR](https://github.com/BerriAI/litellm/pull/10047) +- **Prisma migrate** \- handle existing columns in db table - [PR](https://github.com/BerriAI/litellm/pull/10138) + +## Deploy this version [​](https://docs.litellm.ai/release_notes\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.66.0-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.66.0.post1 + +``` + +v1.66.0-stable is live now, here are the key highlights of this release + +## Key Highlights [​](https://docs.litellm.ai/release_notes\#key-highlights "Direct link to Key Highlights") + +- **Realtime API Cost Tracking**: Track cost of realtime API calls +- **Microsoft SSO Auto-sync**: Auto-sync groups and group members from Azure Entra ID to LiteLLM +- **xAI grok-3**: Added support for `xai/grok-3` models +- **Security Fixes**: Fixed [CVE-2025-0330](https://www.cve.org/CVERecord?id=CVE-2025-0330) and [CVE-2024-6825](https://www.cve.org/CVERecord?id=CVE-2024-6825) vulnerabilities + +Let's dive in. + +## Realtime API Cost Tracking [​](https://docs.litellm.ai/release_notes\#realtime-api-cost-tracking "Direct link to Realtime API Cost Tracking") + +![](https://docs.litellm.ai/assets/ideal-img/realtime_api.960b38e.1920.png) + +This release adds Realtime API logging + cost tracking. + +- **Logging**: LiteLLM now logs the complete response from realtime calls to all logging integrations (DB, S3, Langfuse, etc.) +- **Cost Tracking**: You can now set 'base\_model' and custom pricing for realtime models. [Custom Pricing](https://docs.litellm.ai/docs/proxy/custom_pricing) +- **Budgets**: Your key/user/team budgets now work for realtime models as well. + +Start [here](https://docs.litellm.ai/docs/realtime) + +## Microsoft SSO Auto-sync [​](https://docs.litellm.ai/release_notes\#microsoft-sso-auto-sync "Direct link to Microsoft SSO Auto-sync") + +![](https://docs.litellm.ai/assets/ideal-img/sso_sync.2f79062.1414.png) + +Auto-sync groups and members from Azure Entra ID to LiteLLM + +This release adds support for auto-syncing groups and members on Microsoft Entra ID with LiteLLM. This means that LiteLLM proxy administrators can spend less time managing teams and members and LiteLLM handles the following: + +- Auto-create teams that exist on Microsoft Entra ID +- Sync team members on Microsoft Entra ID with LiteLLM teams + +Get started with this [here](https://docs.litellm.ai/docs/tutorials/msft_sso) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **xAI** + +1. Added reasoning\_effort support for `xai/grok-3-mini-beta` [Get Started](https://docs.litellm.ai/docs/providers/xai#reasoning-usage) +2. Added cost tracking for `xai/grok-3` models [PR](https://github.com/BerriAI/litellm/pull/9920) +- **Hugging Face** + +1. Added inference providers support [Get Started](https://docs.litellm.ai/docs/providers/huggingface#serverless-inference-providers) +- **Azure** + +1. Added azure/gpt-4o-realtime-audio cost tracking [PR](https://github.com/BerriAI/litellm/pull/9893) +- **VertexAI** + +1. Added enterpriseWebSearch tool support [Get Started](https://docs.litellm.ai/docs/providers/vertex#grounding---web-search) +2. Moved to only passing keys accepted by the Vertex AI response schema [PR](https://github.com/BerriAI/litellm/pull/8992) +- **Google AI Studio** + +1. Added cost tracking for `gemini-2.5-pro` [PR](https://github.com/BerriAI/litellm/pull/9837) +2. Fixed pricing for 'gemini/gemini-2.5-pro-preview-03-25' [PR](https://github.com/BerriAI/litellm/pull/9896) +3. Fixed handling file\_data being passed in [PR](https://github.com/BerriAI/litellm/pull/9786) +- **Azure** + +1. Updated Azure Phi-4 pricing [PR](https://github.com/BerriAI/litellm/pull/9862) +2. Added azure/gpt-4o-realtime-audio cost tracking [PR](https://github.com/BerriAI/litellm/pull/9893) +- **Databricks** + +1. Removed reasoning\_effort from parameters [PR](https://github.com/BerriAI/litellm/pull/9811) +2. Fixed custom endpoint check for Databricks [PR](https://github.com/BerriAI/litellm/pull/9925) +- **General** + +1. Added litellm.supports\_reasoning() util to track if an llm supports reasoning [Get Started](https://docs.litellm.ai/docs/providers/anthropic#reasoning) +2. Function Calling - Handle pydantic base model in message tool calls, handle tools = \[\], and support fake streaming on tool calls for meta.llama3-3-70b-instruct-v1:0 [PR](https://github.com/BerriAI/litellm/pull/9774) +3. LiteLLM Proxy - Allow passing `thinking` param to litellm proxy via client sdk [PR](https://github.com/BerriAI/litellm/pull/9386) +4. Fixed correctly translating 'thinking' param for litellm [PR](https://github.com/BerriAI/litellm/pull/9904) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **OpenAI, Azure** +1. Realtime API Cost tracking with token usage metrics in spend logs [Get Started](https://docs.litellm.ai/docs/realtime) +- **Anthropic** +1. Fixed Claude Haiku cache read pricing per token [PR](https://github.com/BerriAI/litellm/pull/9834) +2. Added cost tracking for Claude responses with base\_model [PR](https://github.com/BerriAI/litellm/pull/9897) +3. Fixed Anthropic prompt caching cost calculation and trimmed logged message in db [PR](https://github.com/BerriAI/litellm/pull/9838) +- **General** +1. Added token tracking and log usage object in spend logs [PR](https://github.com/BerriAI/litellm/pull/9843) +2. Handle custom pricing at deployment level [PR](https://github.com/BerriAI/litellm/pull/9855) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +- **Test Key Tab** + +1. Added rendering of Reasoning content, ttft, usage metrics on test key page [PR](https://github.com/BerriAI/litellm/pull/9931) + + ![](https://docs.litellm.ai/assets/ideal-img/chat_metrics.c59fcfe.1920.png) + + View input, output, reasoning tokens, ttft metrics. +- **Tag / Policy Management** + +1. Added Tag/Policy Management. Create routing rules based on request metadata. This allows you to enforce that requests with `tags="private"` only go to specific models. [Get Started](https://docs.litellm.ai/docs/tutorials/tag_management) + + + + ![](https://docs.litellm.ai/assets/ideal-img/tag_management.5bf985c.1920.png) + + Create and manage tags. +- **Redesigned Login Screen** + +1. Polished login screen [PR](https://github.com/BerriAI/litellm/pull/9778) +- **Microsoft SSO Auto-Sync** + +1. Added debug route to allow admins to debug SSO JWT fields [PR](https://github.com/BerriAI/litellm/pull/9835) +2. Added ability to use MSFT Graph API to assign users to teams [PR](https://github.com/BerriAI/litellm/pull/9865) +3. Connected litellm to Azure Entra ID Enterprise Application [PR](https://github.com/BerriAI/litellm/pull/9872) +4. Added ability for admins to set `default_team_params` for when litellm SSO creates default teams [PR](https://github.com/BerriAI/litellm/pull/9895) +5. Fixed MSFT SSO to use correct field for user email [PR](https://github.com/BerriAI/litellm/pull/9886) +6. Added UI support for setting Default Team setting when litellm SSO auto creates teams [PR](https://github.com/BerriAI/litellm/pull/9918) +- **UI Bug Fixes** + +1. Prevented team, key, org, model numerical values changing on scrolling [PR](https://github.com/BerriAI/litellm/pull/9776) +2. Instantly reflect key and team updates in UI [PR](https://github.com/BerriAI/litellm/pull/9825) + +## Logging / Guardrail Improvements [​](https://docs.litellm.ai/release_notes\#logging--guardrail-improvements "Direct link to Logging / Guardrail Improvements") + +- **Prometheus** +1. Emit Key and Team Budget metrics on a cron job schedule [Get Started](https://docs.litellm.ai/docs/proxy/prometheus#initialize-budget-metrics-on-startup) + +## Security Fixes [​](https://docs.litellm.ai/release_notes\#security-fixes "Direct link to Security Fixes") + +- Fixed [CVE-2025-0330](https://www.cve.org/CVERecord?id=CVE-2025-0330) \- Leakage of Langfuse API keys in team exception handling [PR](https://github.com/BerriAI/litellm/pull/9830) +- Fixed [CVE-2024-6825](https://www.cve.org/CVERecord?id=CVE-2024-6825) \- Remote code execution in post call rules [PR](https://github.com/BerriAI/litellm/pull/9826) + +## Helm [​](https://docs.litellm.ai/release_notes\#helm "Direct link to Helm") + +- Added service annotations to litellm-helm chart [PR](https://github.com/BerriAI/litellm/pull/9840) +- Added extraEnvVars to the helm deployment [PR](https://github.com/BerriAI/litellm/pull/9292) + +## Demo [​](https://docs.litellm.ai/release_notes\#demo "Direct link to Demo") + +Try this on the demo instance [today](https://docs.litellm.ai/docs/proxy/demo) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes\#complete-git-diff "Direct link to Complete Git Diff") + +See the complete git diff since v1.65.4-stable, [here](https://github.com/BerriAI/litellm/releases/tag/v1.66.0-stable) + +## Deploy this version [​](https://docs.litellm.ai/release_notes\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.65.4-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.65.4.post1 + +``` + +v1.65.4-stable is live. Here are the improvements since v1.65.0-stable. + +## Key Highlights [​](https://docs.litellm.ai/release_notes\#key-highlights "Direct link to Key Highlights") + +- **Preventing DB Deadlocks**: Fixes a high-traffic issue when multiple instances were writing to the DB at the same time. +- **New Usage Tab**: Enables viewing spend by model and customizing date range + +Let's dive in. + +### Preventing DB Deadlocks [​](https://docs.litellm.ai/release_notes\#preventing-db-deadlocks "Direct link to Preventing DB Deadlocks") + +![](https://docs.litellm.ai/assets/ideal-img/prevent_deadlocks.779afdb.1920.jpg) + +This release fixes the DB deadlocking issue that users faced in high traffic (10K+ RPS). This is great because it enables user/key/team spend tracking works at that scale. + +Read more about the new architecture [here](https://docs.litellm.ai/docs/proxy/db_deadlocks) + +### New Usage Tab [​](https://docs.litellm.ai/release_notes\#new-usage-tab "Direct link to New Usage Tab") + +![](https://docs.litellm.ai/assets/ideal-img/spend_by_model.5023558.1920.jpg) + +The new Usage tab now brings the ability to track daily spend by model. This makes it easier to catch any spend tracking or token counting errors, when combined with the ability to view successful requests, and token usage. + +To test this out, just go to Experimental > New Usage > Activity. + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Databricks - claude-3-7-sonnet cost tracking [PR](https://github.com/BerriAI/litellm/blob/52b35cd8093b9ad833987b24f494586a1e923209/model_prices_and_context_window.json#L10350) +2. VertexAI - `gemini-2.5-pro-exp-03-25` cost tracking [PR](https://github.com/BerriAI/litellm/blob/52b35cd8093b9ad833987b24f494586a1e923209/model_prices_and_context_window.json#L4492) +3. VertexAI - `gemini-2.0-flash` cost tracking [PR](https://github.com/BerriAI/litellm/blob/52b35cd8093b9ad833987b24f494586a1e923209/model_prices_and_context_window.json#L4689) +4. Groq - add whisper ASR models to model cost map [PR](https://github.com/BerriAI/litellm/blob/52b35cd8093b9ad833987b24f494586a1e923209/model_prices_and_context_window.json#L3324) +5. IBM - Add watsonx/ibm/granite-3-8b-instruct to model cost map [PR](https://github.com/BerriAI/litellm/blob/52b35cd8093b9ad833987b24f494586a1e923209/model_prices_and_context_window.json#L91) +6. Google AI Studio - add gemini/gemini-2.5-pro-preview-03-25 to model cost map [PR](https://github.com/BerriAI/litellm/blob/52b35cd8093b9ad833987b24f494586a1e923209/model_prices_and_context_window.json#L4850) + +## LLM Translation [​](https://docs.litellm.ai/release_notes\#llm-translation "Direct link to LLM Translation") + +01. Vertex AI - Support anyOf param for OpenAI json schema translation [Get Started](https://docs.litellm.ai/docs/providers/vertex#json-schema) +02. Anthropic- response\_format + thinking param support (works across Anthropic API, Bedrock, Vertex) [Get Started](https://docs.litellm.ai/docs/reasoning_content) +03. Anthropic - if thinking token is specified and max tokens is not - ensure max token to anthropic is higher than thinking tokens (works across Anthropic API, Bedrock, Vertex) [PR](https://github.com/BerriAI/litellm/pull/9594) +04. Bedrock - latency optimized inference support [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---latency-optimized-inference) +05. Sagemaker - handle special tokens + multibyte character code in response [Get Started](https://docs.litellm.ai/docs/providers/aws_sagemaker) +06. MCP - add support for using SSE MCP servers [Get Started](https://docs.litellm.ai/docs/mcp#usage) +07. Anthropic - new `litellm.messages.create` interface for calling Anthropic `/v1/messages` via passthrough [Get Started](https://docs.litellm.ai/docs/anthropic_unified#usage) +08. Anthropic - support ‘file’ content type in message param (works across Anthropic API, Bedrock, Vertex) [Get Started](https://docs.litellm.ai/docs/providers/anthropic#usage---pdf) +09. Anthropic - map openai 'reasoning\_effort' to anthropic 'thinking' param (works across Anthropic API, Bedrock, Vertex) [Get Started](https://docs.litellm.ai/docs/providers/anthropic#usage---thinking--reasoning_content) +10. Google AI Studio (Gemini) - \[BETA\] `/v1/files` upload support [Get Started](https://docs.litellm.ai/docs/providers/google_ai_studio/files) +11. Azure - fix o-series tool calling [Get Started](https://docs.litellm.ai/docs/providers/azure#tool-calling--function-calling) +12. Unified file id - \[ALPHA\] allow calling multiple providers with same file id [PR](https://github.com/BerriAI/litellm/pull/9718) + - This is experimental, and not recommended for production use. + - We plan to have a production-ready implementation by next week. +13. Google AI Studio (Gemini) - return logprobs [PR](https://github.com/BerriAI/litellm/pull/9713) +14. Anthropic - Support prompt caching for Anthropic tool calls [Get Started](https://docs.litellm.ai/docs/completion/prompt_caching) +15. OpenRouter - unwrap extra body on open router calls [PR](https://github.com/BerriAI/litellm/pull/9747) +16. VertexAI - fix credential caching issue [PR](https://github.com/BerriAI/litellm/pull/9756) +17. XAI - filter out 'name' param for XAI [PR](https://github.com/BerriAI/litellm/pull/9761) +18. Gemini - image generation output support [Get Started](https://docs.litellm.ai/docs/providers/gemini#image-generation) +19. Databricks - support claude-3-7-sonnet w/ thinking + response\_format [Get Started](https://docs.litellm.ai/docs/providers/databricks#usage---thinking--reasoning_content) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Reliability fix - Check sent and received model for cost calculation [PR](https://github.com/BerriAI/litellm/pull/9669) +2. Vertex AI - Multimodal embedding cost tracking [Get Started](https://docs.litellm.ai/docs/providers/vertex#multi-modal-embeddings), [PR](https://github.com/BerriAI/litellm/pull/9623) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +![](https://docs.litellm.ai/assets/ideal-img/new_activity_tab.1668e74.1920.png) + +1. New Usage Tab + - Report 'total\_tokens' + report success/failure calls + - Remove double bars on scroll + - Ensure ‘daily spend’ chart ordered from earliest to latest date + - showing spend per model per day + - show key alias on usage tab + - Allow non-admins to view their activity + - Add date picker to new usage tab +2. Virtual Keys Tab + - remove 'default key' on user signup + - fix showing user models available for personal key creation +3. Test Key Tab + - Allow testing image generation models +4. Models Tab + - Fix bulk adding models + - support reusable credentials for passthrough endpoints + - Allow team members to see team models +5. Teams Tab + - Fix json serialization error on update team metadata +6. Request Logs Tab + - Add reasoning\_content token tracking across all providers on streaming +7. API + - return key alias on /user/daily/activity [Get Started](https://docs.litellm.ai/docs/proxy/cost_tracking#daily-spend-breakdown-api) +8. SSO + - Allow assigning SSO users to teams on MSFT SSO [PR](https://github.com/BerriAI/litellm/pull/9745) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Console Logs - Add json formatting for uncaught exceptions [PR](https://github.com/BerriAI/litellm/pull/9619) +2. Guardrails - AIM Guardrails support for virtual key based policies [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/aim_security) +3. Logging - fix completion start time tracking [PR](https://github.com/BerriAI/litellm/pull/9688) +4. Prometheus + - Allow adding authentication on Prometheus /metrics endpoints [PR](https://github.com/BerriAI/litellm/pull/9766) + - Distinguish LLM Provider Exception vs. LiteLLM Exception in metric naming [PR](https://github.com/BerriAI/litellm/pull/9760) + - Emit operational metrics for new DB Transaction architecture [PR](https://github.com/BerriAI/litellm/pull/9719) + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Preventing Deadlocks + - Reduce DB Deadlocks by storing spend updates in Redis and then committing to DB [PR](https://github.com/BerriAI/litellm/pull/9608) + - Ensure no deadlocks occur when updating DailyUserSpendTransaction [PR](https://github.com/BerriAI/litellm/pull/9690) + - High Traffic fix - ensure new DB + Redis architecture accurately tracks spend [PR](https://github.com/BerriAI/litellm/pull/9673) + - Use Redis for PodLock Manager instead of PG (ensures no deadlocks occur) [PR](https://github.com/BerriAI/litellm/pull/9715) + - v2 DB Deadlock Reduction Architecture – Add Max Size for In-Memory Queue + Backpressure Mechanism [PR](https://github.com/BerriAI/litellm/pull/9759) +2. Prisma Migrations [Get Started](https://docs.litellm.ai/docs/proxy/prod#9-use-prisma-migrate-deploy) + - connects litellm proxy to litellm's prisma migration files + - Handle db schema updates from new `litellm-proxy-extras` sdk +3. Redis - support password for sync sentinel clients [PR](https://github.com/BerriAI/litellm/pull/9622) +4. Fix "Circular reference detected" error when max\_parallel\_requests = 0 [PR](https://github.com/BerriAI/litellm/pull/9671) +5. Code QA - Ban hardcoded numbers [PR](https://github.com/BerriAI/litellm/pull/9709) + +## Helm [​](https://docs.litellm.ai/release_notes\#helm "Direct link to Helm") + +1. fix: wrong indentation of ttlSecondsAfterFinished in chart [PR](https://github.com/BerriAI/litellm/pull/9611) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Fix - only apply service\_account\_settings.enforced\_params on service accounts [PR](https://github.com/BerriAI/litellm/pull/9683) +2. Fix - handle metadata null on `/chat/completion` [PR](https://github.com/BerriAI/litellm/issues/9717) +3. Fix - Move daily user transaction logging outside of 'disable\_spend\_logs' flag, as they’re unrelated [PR](https://github.com/BerriAI/litellm/pull/9772) + +## Demo [​](https://docs.litellm.ai/release_notes\#demo "Direct link to Demo") + +Try this on the demo instance [today](https://docs.litellm.ai/docs/proxy/demo) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes\#complete-git-diff "Direct link to Complete Git Diff") + +See the complete git diff since v1.65.0-stable, [here](https://github.com/BerriAI/litellm/releases/tag/v1.65.4-stable) + +v1.65.0-stable is live now. Here are the key highlights of this release: + +- **MCP Support**: Support for adding and using MCP servers on the LiteLLM proxy. +- **UI view total usage after 1M+ logs**: You can now view usage analytics after crossing 1M+ logs in DB. + +## Model Context Protocol (MCP) [​](https://docs.litellm.ai/release_notes\#model-context-protocol-mcp "Direct link to Model Context Protocol (MCP)") + +This release introduces support for centrally adding MCP servers on LiteLLM. This allows you to add MCP server endpoints and your developers can `list` and `call` MCP tools through LiteLLM. + +Read more about MCP [here](https://docs.litellm.ai/docs/mcp). + +![](https://docs.litellm.ai/assets/ideal-img/mcp_ui.4a5216a.1920.png) + +Expose and use MCP servers through LiteLLM + +## UI view total usage after 1M+ logs [​](https://docs.litellm.ai/release_notes\#ui-view-total-usage-after-1m-logs "Direct link to UI view total usage after 1M+ logs") + +This release brings the ability to view total usage analytics even after exceeding 1M+ logs in your database. We've implemented a scalable architecture that stores only aggregate usage data, resulting in significantly more efficient queries and reduced database CPU utilization. + +![](https://docs.litellm.ai/assets/ideal-img/ui_usage.3ffdba3.1200.png) + +View total usage after 1M+ logs + +- How this works: + + - We now aggregate usage data into a dedicated DailyUserSpend table, significantly reducing query load and CPU usage even beyond 1M+ logs. +- Daily Spend Breakdown API: + + - Retrieve granular daily usage data (by model, provider, and API key) with a single endpoint. + Example Request: + + + + Daily Spend Breakdown API + + + + + + ```codeBlockLines_e6Vv codeBlockLinesWithNumbering_o6Pm + curl -L -X GET 'http://localhost:4000/user/daily/activity?start_date=2025-03-20&end_date=2025-03-27' \ + -H 'Authorization: Bearer sk-...' + + ``` + + + + + + + + + + + + Daily Spend Breakdown API Response + + + + + + ```codeBlockLines_e6Vv codeBlockLinesWithNumbering_o6Pm + { + "results": [\ + {\ + "date": "2025-03-27",\ + "metrics": {\ + "spend": 0.0177072,\ + "prompt_tokens": 111,\ + "completion_tokens": 1711,\ + "total_tokens": 1822,\ + "api_requests": 11\ + },\ + "breakdown": {\ + "models": {\ + "gpt-4o-mini": {\ + "spend": 1.095e-05,\ + "prompt_tokens": 37,\ + "completion_tokens": 9,\ + "total_tokens": 46,\ + "api_requests": 1\ + },\ + "providers": { "openai": { ... }, "azure_ai": { ... } },\ + "api_keys": { "3126b6eaf1...": { ... } }\ + }\ + }\ + ], + "metadata": { + "total_spend": 0.7274667, + "total_prompt_tokens": 280990, + "total_completion_tokens": 376674, + "total_api_requests": 14 + } + } + + ``` + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Support for Vertex AI gemini-2.0-flash-lite & Google AI Studio gemini-2.0-flash-lite [PR](https://github.com/BerriAI/litellm/pull/9523) +- Support for Vertex AI Fine-Tuned LLMs [PR](https://github.com/BerriAI/litellm/pull/9542) +- Nova Canvas image generation support [PR](https://github.com/BerriAI/litellm/pull/9525) +- OpenAI gpt-4o-transcribe support [PR](https://github.com/BerriAI/litellm/pull/9517) +- Added new Vertex AI text embedding model [PR](https://github.com/BerriAI/litellm/pull/9476) + +## LLM Translation [​](https://docs.litellm.ai/release_notes\#llm-translation "Direct link to LLM Translation") + +- OpenAI Web Search Tool Call Support [PR](https://github.com/BerriAI/litellm/pull/9465) +- Vertex AI topLogprobs support [PR](https://github.com/BerriAI/litellm/pull/9518) +- Support for sending images and video to Vertex AI multimodal embedding [Doc](https://docs.litellm.ai/docs/providers/vertex#multi-modal-embeddings) +- Support litellm.api\_base for Vertex AI + Gemini across completion, embedding, image\_generation [PR](https://github.com/BerriAI/litellm/pull/9516) +- Bug fix for returning `response_cost` when using litellm python SDK with LiteLLM Proxy [PR](https://github.com/BerriAI/litellm/commit/6fd18651d129d606182ff4b980e95768fc43ca3d) +- Support for `max_completion_tokens` on Mistral API [PR](https://github.com/BerriAI/litellm/pull/9606) +- Refactored Vertex AI passthrough routes - fixes unpredictable behaviour with auto-setting default\_vertex\_region on router model add [PR](https://github.com/BerriAI/litellm/pull/9467) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- Log 'api\_base' on spend logs [PR](https://github.com/BerriAI/litellm/pull/9509) +- Support for Gemini audio token cost tracking [PR](https://github.com/BerriAI/litellm/pull/9535) +- Fixed OpenAI audio input token cost tracking [PR](https://github.com/BerriAI/litellm/pull/9535) + +## UI [​](https://docs.litellm.ai/release_notes\#ui "Direct link to UI") + +### Model Management [​](https://docs.litellm.ai/release_notes\#model-management "Direct link to Model Management") + +- Allowed team admins to add/update/delete models on UI [PR](https://github.com/BerriAI/litellm/pull/9572) +- Added render supports\_web\_search on model hub [PR](https://github.com/BerriAI/litellm/pull/9469) + +### Request Logs [​](https://docs.litellm.ai/release_notes\#request-logs "Direct link to Request Logs") + +- Show API base and model ID on request logs [PR](https://github.com/BerriAI/litellm/pull/9572) +- Allow viewing keyinfo on request logs [PR](https://github.com/BerriAI/litellm/pull/9568) + +### Usage Tab [​](https://docs.litellm.ai/release_notes\#usage-tab "Direct link to Usage Tab") + +- Added Daily User Spend Aggregate view - allows UI Usage tab to work > 1m rows [PR](https://github.com/BerriAI/litellm/pull/9538) +- Connected UI to "LiteLLM\_DailyUserSpend" spend table [PR](https://github.com/BerriAI/litellm/pull/9603) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes\#logging-integrations "Direct link to Logging Integrations") + +- Fixed StandardLoggingPayload for GCS Pub Sub Logging Integration [PR](https://github.com/BerriAI/litellm/pull/9508) +- Track `litellm_model_name` on `StandardLoggingPayload` [Docs](https://docs.litellm.ai/docs/proxy/logging_spec#standardlogginghiddenparams) + +## Performance / Reliability Improvements [​](https://docs.litellm.ai/release_notes\#performance--reliability-improvements "Direct link to Performance / Reliability Improvements") + +- LiteLLM Redis semantic caching implementation [PR](https://github.com/BerriAI/litellm/pull/9356) +- Gracefully handle exceptions when DB is having an outage [PR](https://github.com/BerriAI/litellm/pull/9533) +- Allow Pods to startup + passing /health/readiness when allow\_requests\_on\_db\_unavailable: True and DB is down [PR](https://github.com/BerriAI/litellm/pull/9569) + +## General Improvements [​](https://docs.litellm.ai/release_notes\#general-improvements "Direct link to General Improvements") + +- Support for exposing MCP tools on litellm proxy [PR](https://github.com/BerriAI/litellm/pull/9426) +- Support discovering Gemini, Anthropic, xAI models by calling their /v1/model endpoint [PR](https://github.com/BerriAI/litellm/pull/9530) +- Fixed route check for non-proxy admins on JWT auth [PR](https://github.com/BerriAI/litellm/pull/9454) +- Added baseline Prisma database migrations [PR](https://github.com/BerriAI/litellm/pull/9565) +- View all wildcard models on /model/info [PR](https://github.com/BerriAI/litellm/pull/9572) + +## Security [​](https://docs.litellm.ai/release_notes\#security "Direct link to Security") + +- Bumped next from 14.2.21 to 14.2.25 in UI dashboard [PR](https://github.com/BerriAI/litellm/pull/9458) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.14-stable.patch1...v1.65.0-stable) + +v1.65.0 updates the `/model/new` endpoint to prevent non-team admins from creating team models. + +This means that only proxy admins or team admins can create team models. + +## Additional Changes [​](https://docs.litellm.ai/release_notes\#additional-changes "Direct link to Additional Changes") + +- Allows team admins to call `/model/update` to update team models. +- Allows team admins to call `/model/delete` to delete team models. +- Introduces new `user_models_only` param to `/v2/model/info` \- only return models added by this user. + +These changes enable team admins to add and manage models for their team on the LiteLLM UI + API. + +![](https://docs.litellm.ai/assets/ideal-img/team_model_add.1ddd404.1251.png) + +These are the changes since `v1.63.11-stable`. + +This release brings: + +- LLM Translation Improvements (MCP Support and Bedrock Application Profiles) +- Perf improvements for Usage-based Routing +- Streaming guardrail support via websockets +- Azure OpenAI client perf fix (from previous release) + +## Docker Run LiteLLM Proxy [​](https://docs.litellm.ai/release_notes\#docker-run-litellm-proxy "Direct link to Docker Run LiteLLM Proxy") + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.63.14-stable.patch1 + +``` + +## Demo Instance [​](https://docs.litellm.ai/release_notes\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Azure gpt-4o - fixed pricing to latest global pricing - [PR](https://github.com/BerriAI/litellm/pull/9361) +- O1-Pro - add pricing + model information - [PR](https://github.com/BerriAI/litellm/pull/9397) +- Azure AI - mistral 3.1 small pricing added - [PR](https://github.com/BerriAI/litellm/pull/9453) +- Azure - gpt-4.5-preview pricing added - [PR](https://github.com/BerriAI/litellm/pull/9453) + +## LLM Translation [​](https://docs.litellm.ai/release_notes\#llm-translation "Direct link to LLM Translation") + +1. **New LLM Features** + +- Bedrock: Support bedrock application inference profiles [Docs](https://docs.litellm.ai/docs/providers/bedrock#bedrock-application-inference-profile) + - Infer aws region from bedrock application profile id - ( `arn:aws:bedrock:us-east-1:...`) +- Ollama - support calling via `/v1/completions` [Get Started](https://docs.litellm.ai/docs/providers/ollama#using-ollama-fim-on-v1completions) +- Bedrock - support `us.deepseek.r1-v1:0` model name [Docs](https://docs.litellm.ai/docs/providers/bedrock#supported-aws-bedrock-models) +- OpenRouter - `OPENROUTER_API_BASE` env var support [Docs](https://docs.litellm.ai/docs/providers/openrouter.md) +- Azure - add audio model parameter support - [Docs](https://docs.litellm.ai/docs/providers/azure#azure-audio-model) +- OpenAI - PDF File support [Docs](https://docs.litellm.ai/docs/completion/document_understanding#openai-file-message-type) +- OpenAI - o1-pro Responses API streaming support [Docs](https://docs.litellm.ai/docs/response_api.md#streaming) +- \[BETA\] MCP - Use MCP Tools with LiteLLM SDK [Docs](https://docs.litellm.ai/docs/mcp) + +2. **Bug Fixes** + +- Voyage: prompt token on embedding tracking fix - [PR](https://github.com/BerriAI/litellm/commit/56d3e75b330c3c3862dc6e1c51c1210e48f1068e) +- Sagemaker - Fix ‘Too little data for declared Content-Length’ error - [PR](https://github.com/BerriAI/litellm/pull/9326) +- OpenAI-compatible models - fix issue when calling openai-compatible models w/ custom\_llm\_provider set - [PR](https://github.com/BerriAI/litellm/pull/9355) +- VertexAI - Embedding ‘outputDimensionality’ support - [PR](https://github.com/BerriAI/litellm/commit/437dbe724620675295f298164a076cbd8019d304) +- Anthropic - return consistent json response format on streaming/non-streaming - [PR](https://github.com/BerriAI/litellm/pull/9437) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- `litellm_proxy/` \- support reading litellm response cost header from proxy, when using client sdk +- Reset Budget Job - fix budget reset error on keys/teams/users [PR](https://github.com/BerriAI/litellm/pull/9329) +- Streaming - Prevents final chunk w/ usage from being ignored (impacted bedrock streaming + cost tracking) [PR](https://github.com/BerriAI/litellm/pull/9314) + +## UI [​](https://docs.litellm.ai/release_notes\#ui "Direct link to UI") + +1. Users Page + - Feature: Control default internal user settings [PR](https://github.com/BerriAI/litellm/pull/9328) +2. Icons: + - Feature: Replace external "artificialanalysis.ai" icons by local svg [PR](https://github.com/BerriAI/litellm/pull/9374) +3. Sign In/Sign Out + - Fix: Default login when `default_user_id` user does not exist in DB [PR](https://github.com/BerriAI/litellm/pull/9395) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes\#logging-integrations "Direct link to Logging Integrations") + +- Support post-call guardrails for streaming responses [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/custom_guardrail#1-write-a-customguardrail-class) +- Arize [Get Started](https://docs.litellm.ai/docs/observability/arize_integration) + - fix invalid package import [PR](https://github.com/BerriAI/litellm/pull/9338) + - migrate to using standardloggingpayload for metadata, ensures spans land successfully [PR](https://github.com/BerriAI/litellm/pull/9338) + - fix logging to just log the LLM I/O [PR](https://github.com/BerriAI/litellm/pull/9353) + - Dynamic API Key/Space param support [Get Started](https://docs.litellm.ai/docs/observability/arize_integration#pass-arize-spacekey-per-request) +- StandardLoggingPayload - Log litellm\_model\_name in payload. Allows knowing what the model sent to API provider was [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec#standardlogginghiddenparams) +- Prompt Management - Allow building custom prompt management integration [Get Started](https://docs.litellm.ai/docs/proxy/custom_prompt_management.md) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +- Redis Caching - add 5s default timeout, prevents hanging redis connection from impacting llm calls [PR](https://github.com/BerriAI/litellm/commit/db92956ae33ed4c4e3233d7e1b0c7229817159bf) +- Allow disabling all spend updates / writes to DB - patch to allow disabling all spend updates to DB with a flag [PR](https://github.com/BerriAI/litellm/pull/9331) +- Azure OpenAI - correctly re-use azure openai client, fixes perf issue from previous Stable release [PR](https://github.com/BerriAI/litellm/commit/f2026ef907c06d94440930917add71314b901413) +- Azure OpenAI - uses litellm.ssl\_verify on Azure/OpenAI clients [PR](https://github.com/BerriAI/litellm/commit/f2026ef907c06d94440930917add71314b901413) +- Usage-based routing - Wildcard model support [Get Started](https://docs.litellm.ai/docs/proxy/usage_based_routing#wildcard-model-support) +- Usage-based routing - Support batch writing increments to redis - reduces latency to same as ‘simple-shuffle’ [PR](https://github.com/BerriAI/litellm/pull/9357) +- Router - show reason for model cooldown on ‘no healthy deployments available error’ [PR](https://github.com/BerriAI/litellm/pull/9438) +- Caching - add max value limit to an item in in-memory cache (1MB) - prevents OOM errors on large image url’s being sent through proxy [PR](https://github.com/BerriAI/litellm/pull/9448) + +## General Improvements [​](https://docs.litellm.ai/release_notes\#general-improvements "Direct link to General Improvements") + +- Passthrough Endpoints - support returning api-base on pass-through endpoints Response Headers [Docs](https://docs.litellm.ai/docs/proxy/response_headers#litellm-specific-headers) +- SSL - support reading ssl security level from env var - Allows user to specify lower security settings [Get Started](https://docs.litellm.ai/docs/guides/security_settings) +- Credentials - only poll Credentials table when `STORE_MODEL_IN_DB` is True [PR](https://github.com/BerriAI/litellm/pull/9376) +- Image URL Handling - new architecture doc on image url handling [Docs](https://docs.litellm.ai/docs/proxy/image_handling) +- OpenAI - bump to pip install "openai==1.68.2" [PR](https://github.com/BerriAI/litellm/commit/e85e3bc52a9de86ad85c3dbb12d87664ee567a5a) +- Gunicorn - security fix - bump gunicorn==23.0.0 [PR](https://github.com/BerriAI/litellm/commit/7e9fc92f5c7fea1e7294171cd3859d55384166eb) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.11-stable...v1.63.14.rc) + +These are the changes since `v1.63.2-stable`. + +This release is primarily focused on: + +- \[Beta\] Responses API Support +- Snowflake Cortex Support, Amazon Nova Image Generation +- UI - Credential Management, re-use credentials when adding new models +- UI - Test Connection to LLM Provider before adding a model + +## Known Issues [​](https://docs.litellm.ai/release_notes\#known-issues "Direct link to Known Issues") + +- 🚨 Known issue on Azure OpenAI - We don't recommend upgrading if you use Azure OpenAI. This version failed our Azure OpenAI load test + +## Docker Run LiteLLM Proxy [​](https://docs.litellm.ai/release_notes\#docker-run-litellm-proxy "Direct link to Docker Run LiteLLM Proxy") + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.63.11-stable + +``` + +## Demo Instance [​](https://docs.litellm.ai/release_notes\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Image Generation support for Amazon Nova Canvas [Getting Started](https://docs.litellm.ai/docs/providers/bedrock#image-generation) +- Add pricing for Jamba new models [PR](https://github.com/BerriAI/litellm/pull/9032/files) +- Add pricing for Amazon EU models [PR](https://github.com/BerriAI/litellm/pull/9056/files) +- Add Bedrock Deepseek R1 model pricing [PR](https://github.com/BerriAI/litellm/pull/9108/files) +- Update Gemini pricing: Gemma 3, Flash 2 thinking update, LearnLM [PR](https://github.com/BerriAI/litellm/pull/9190/files) +- Mark Cohere Embedding 3 models as Multimodal [PR](https://github.com/BerriAI/litellm/pull/9176/commits/c9a576ce4221fc6e50dc47cdf64ab62736c9da41) +- Add Azure Data Zone pricing [PR](https://github.com/BerriAI/litellm/pull/9185/files#diff-19ad91c53996e178c1921cbacadf6f3bae20cfe062bd03ee6bfffb72f847ee37) + - LiteLLM Tracks cost for `azure/eu` and `azure/us` models + +## LLM Translation [​](https://docs.litellm.ai/release_notes\#llm-translation "Direct link to LLM Translation") + +![](https://docs.litellm.ai/assets/ideal-img/responses_api.01dd45d.1200.png) + +1. **New Endpoints** + +- \[Beta\] POST `/responses` API. [Getting Started](https://docs.litellm.ai/docs/response_api) + +2. **New LLM Providers** + +- Snowflake Cortex [Getting Started](https://docs.litellm.ai/docs/providers/snowflake) + +3. **New LLM Features** + +- Support OpenRouter `reasoning_content` on streaming [Getting Started](https://docs.litellm.ai/docs/reasoning_content) + +4. **Bug Fixes** + +- OpenAI: Return `code`, `param` and `type` on bad request error [More information on litellm exceptions](https://docs.litellm.ai/docs/exception_mapping) +- Bedrock: Fix converse chunk parsing to only return empty dict on tool use [PR](https://github.com/BerriAI/litellm/pull/9166) +- Bedrock: Support extra\_headers [PR](https://github.com/BerriAI/litellm/pull/9113) +- Azure: Fix Function Calling Bug & Update Default API Version to `2025-02-01-preview` [PR](https://github.com/BerriAI/litellm/pull/9191) +- Azure: Fix AI services URL [PR](https://github.com/BerriAI/litellm/pull/9185) +- Vertex AI: Handle HTTP 201 status code in response [PR](https://github.com/BerriAI/litellm/pull/9193) +- Perplexity: Fix incorrect streaming response [PR](https://github.com/BerriAI/litellm/pull/9081) +- Triton: Fix streaming completions bug [PR](https://github.com/BerriAI/litellm/pull/8386) +- Deepgram: Support bytes.IO when handling audio files for transcription [PR](https://github.com/BerriAI/litellm/pull/9071) +- Ollama: Fix "system" role has become unacceptable [PR](https://github.com/BerriAI/litellm/pull/9261) +- All Providers (Streaming): Fix String `data:` stripped from entire content in streamed responses [PR](https://github.com/BerriAI/litellm/pull/9070) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Support Bedrock converse cache token tracking [Getting Started](https://docs.litellm.ai/docs/completion/prompt_caching) +2. Cost Tracking for Responses API [Getting Started](https://docs.litellm.ai/docs/response_api) +3. Fix Azure Whisper cost tracking [Getting Started](https://docs.litellm.ai/docs/audio_transcription) + +## UI [​](https://docs.litellm.ai/release_notes\#ui "Direct link to UI") + +### Re-Use Credentials on UI [​](https://docs.litellm.ai/release_notes\#re-use-credentials-on-ui "Direct link to Re-Use Credentials on UI") + +You can now onboard LLM provider credentials on LiteLLM UI. Once these credentials are added you can re-use them when adding new models [Getting Started](https://docs.litellm.ai/docs/proxy/ui_credentials) + +![](https://docs.litellm.ai/assets/ideal-img/credentials.8f19ffb.1920.jpg) + +### Test Connections before adding models [​](https://docs.litellm.ai/release_notes\#test-connections-before-adding-models "Direct link to Test Connections before adding models") + +Before adding a model you can test the connection to the LLM provider to verify you have setup your API Base + API Key correctly + +![](https://docs.litellm.ai/assets/images/litellm_test_connection-029765a2de4dcabccfe3be9a8d33dbdd.gif) + +### General UI Improvements [​](https://docs.litellm.ai/release_notes\#general-ui-improvements "Direct link to General UI Improvements") + +1. Add Models Page + - Allow adding Cerebras, Sambanova, Perplexity, Fireworks, Openrouter, TogetherAI Models, Text-Completion OpenAI on Admin UI + - Allow adding EU OpenAI models + - Fix: Instantly show edit + deletes to models +2. Keys Page + - Fix: Instantly show newly created keys on Admin UI (don't require refresh) + - Fix: Allow clicking into Top Keys when showing users Top API Key + - Fix: Allow Filter Keys by Team Alias, Key Alias and Org + - UI Improvements: Show 100 Keys Per Page, Use full height, increase width of key alias +3. Users Page + - Fix: Show correct count of internal user keys on Users Page + - Fix: Metadata not updating in Team UI +4. Logs Page + - UI Improvements: Keep expanded log in focus on LiteLLM UI + - UI Improvements: Minor improvements to logs page + - Fix: Allow internal user to query their own logs + - Allow switching off storing Error Logs in DB [Getting Started](https://docs.litellm.ai/docs/proxy/ui_logs) +5. Sign In/Sign Out + - Fix: Correctly use `PROXY_LOGOUT_URL` when set [Getting Started](https://docs.litellm.ai/docs/proxy/self_serve#setting-custom-logout-urls) + +## Security [​](https://docs.litellm.ai/release_notes\#security "Direct link to Security") + +1. Support for Rotating Master Keys [Getting Started](https://docs.litellm.ai/docs/proxy/master_key_rotations) +2. Fix: Internal User Viewer Permissions, don't allow `internal_user_viewer` role to see `Test Key Page` or `Create Key Button` [More information on role based access controls](https://docs.litellm.ai/docs/proxy/access_control) +3. Emit audit logs on All user + model Create/Update/Delete endpoints [Getting Started](https://docs.litellm.ai/docs/proxy/multiple_admins) +4. JWT + - Support multiple JWT OIDC providers [Getting Started](https://docs.litellm.ai/docs/proxy/token_auth) + - Fix JWT access with Groups not working when team is assigned All Proxy Models access +5. Using K/V pairs in 1 AWS Secret [Getting Started](https://docs.litellm.ai/docs/secret#using-kv-pairs-in-1-aws-secret) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes\#logging-integrations "Direct link to Logging Integrations") + +1. Prometheus: Track Azure LLM API latency metric [Getting Started](https://docs.litellm.ai/docs/proxy/prometheus#request-latency-metrics) +2. Athina: Added tags, user\_feedback and model\_options to additional\_keys which can be sent to Athina [Getting Started](https://docs.litellm.ai/docs/observability/athina_integration) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +1. Redis + litellm router - Fix Redis cluster mode for litellm router [PR](https://github.com/BerriAI/litellm/pull/9010) + +## General Improvements [​](https://docs.litellm.ai/release_notes\#general-improvements "Direct link to General Improvements") + +1. OpenWebUI Integration - display `thinking` tokens + +- Guide on getting started with LiteLLM x OpenWebUI. [Getting Started](https://docs.litellm.ai/docs/tutorials/openweb_ui) +- Display `thinking` tokens on OpenWebUI (Bedrock, Anthropic, Deepseek) [Getting Started](https://docs.litellm.ai/docs/tutorials/openweb_ui#render-thinking-content-on-openweb-ui) + +![](https://docs.litellm.ai/assets/images/litellm_thinking_openweb-5ec7dddb7e7b6a10252694c27cfc177d.gif) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.2-stable...v1.63.11-stable) + +These are the changes since `v1.61.20-stable`. + +This release is primarily focused on: + +- LLM Translation improvements (more `thinking` content improvements) +- UI improvements (Error logs now shown on UI) + +info + +This release will be live on 03/09/2025 + +![](https://docs.litellm.ai/assets/ideal-img/v1632_release.7b42da1.1920.jpg) + +## Demo Instance [​](https://docs.litellm.ai/release_notes\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Add `supports_pdf_input` for specific Bedrock Claude models [PR](https://github.com/BerriAI/litellm/commit/f63cf0030679fe1a43d03fb196e815a0f28dae92) +2. Add pricing for amazon `eu` models [PR](https://github.com/BerriAI/litellm/commits/main/model_prices_and_context_window.json) +3. Fix Azure O1 mini pricing [PR](https://github.com/BerriAI/litellm/commit/52de1949ef2f76b8572df751f9c868a016d4832c) + +## LLM Translation [​](https://docs.litellm.ai/release_notes\#llm-translation "Direct link to LLM Translation") + +![](https://docs.litellm.ai/assets/ideal-img/anthropic_thinking.3bef9d6.1920.jpg) + +01. Support `/openai/` passthrough for Assistant endpoints. [Get Started](https://docs.litellm.ai/docs/pass_through/openai_passthrough) +02. Bedrock Claude - fix tool calling transformation on invoke route. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---function-calling--tool-calling) +03. Bedrock Claude - response\_format support for claude on invoke route. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---structured-output--json-mode) +04. Bedrock - pass `description` if set in response\_format. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---structured-output--json-mode) +05. Bedrock - Fix passing response\_format: {"type": "text"}. [PR](https://github.com/BerriAI/litellm/commit/c84b489d5897755139aa7d4e9e54727ebe0fa540) +06. OpenAI - Handle sending image\_url as str to openai. [Get Started](https://docs.litellm.ai/docs/completion/vision) +07. Deepseek - return 'reasoning\_content' missing on streaming. [Get Started](https://docs.litellm.ai/docs/reasoning_content) +08. Caching - Support caching on reasoning content. [Get Started](https://docs.litellm.ai/docs/proxy/caching) +09. Bedrock - handle thinking blocks in assistant message. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) +10. Anthropic - Return `signature` on streaming. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) + +- Note: We've also migrated from `signature_delta` to `signature`. [Read more](https://docs.litellm.ai/release_notes/v1.63.0) + +11. Support format param for specifying image type. [Get Started](https://docs.litellm.ai/docs/completion/vision.md#explicitly-specify-image-type) +12. Anthropic - `/v1/messages` endpoint - `thinking` param support. [Get Started](https://docs.litellm.ai/docs/anthropic_unified.md) + +- Note: this refactors the \[BETA\] unified `/v1/messages` endpoint, to just work for the Anthropic API. + +13. Vertex AI - handle $id in response schema when calling vertex ai. [Get Started](https://docs.litellm.ai/docs/providers/vertex#json-schema) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Batches API - Fix cost calculation to run on retrieve\_batch. [Get Started](https://docs.litellm.ai/docs/batches) +2. Batches API - Log batch models in spend logs / standard logging payload. [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec.md#standardlogginghiddenparams) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +![](https://docs.litellm.ai/assets/ideal-img/error_logs.63c5dc9.1920.jpg) + +1. Virtual Keys Page + - Allow team/org filters to be searchable on the Create Key Page + - Add created\_by and updated\_by fields to Keys table + - Show 'user\_email' on key table + - Show 100 Keys Per Page, Use full height, increase width of key alias +2. Logs Page + - Show Error Logs on LiteLLM UI + - Allow Internal Users to View their own logs +3. Internal Users Page + - Allow admin to control default model access for internal users +4. Fix session handling with cookies + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Fix prometheus metrics w/ custom metrics, when keys containing team\_id make requests. [PR](https://github.com/BerriAI/litellm/pull/8935) + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Cooldowns - Support cooldowns on models called with client side credentials. [Get Started](https://docs.litellm.ai/docs/proxy/clientside_auth#pass-user-llm-api-keys--api-base) +2. Tag-based Routing - ensures tag-based routing across all endpoints ( `/embeddings`, `/image_generation`, etc.). [Get Started](https://docs.litellm.ai/docs/proxy/tag_routing) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Raise BadRequestError when unknown model passed in request +2. Enforce model access restrictions on Azure OpenAI proxy route +3. Reliability fix - Handle emoji’s in text - fix orjson error +4. Model Access Patch - don't overwrite litellm.anthropic\_models when running auth checks +5. Enable setting timezone information in docker image + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.61.20-stable...v1.63.2-stable) + +v1.63.0 fixes Anthropic 'thinking' response on streaming to return the `signature` block. [Github Issue](https://github.com/BerriAI/litellm/issues/8964) + +It also moves the response structure from `signature_delta` to `signature` to be the same as Anthropic. [Anthropic Docs](https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking#implementing-extended-thinking) + +## Diff [​](https://docs.litellm.ai/release_notes\#diff "Direct link to Diff") + +```codeBlockLines_e6Vv +"message": { + ... + "reasoning_content": "The capital of France is Paris.", + "thinking_blocks": [\ + {\ + "type": "thinking",\ + "thinking": "The capital of France is Paris.",\ +- "signature_delta": "EqoBCkgIARABGAIiQL2UoU0b1OHYi+..." # 👈 OLD FORMAT\ ++ "signature": "EqoBCkgIARABGAIiQL2UoU0b1OHYi+..." # 👈 KEY CHANGE\ + }\ + ] +} + +``` + +These are the changes since `v1.61.13-stable`. + +This release is primarily focused on: + +- LLM Translation improvements (claude-3-7-sonnet + 'thinking'/'reasoning\_content' support) +- UI improvements (add model flow, user management, etc) + +## Demo Instance [​](https://docs.litellm.ai/release_notes\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Anthropic 3-7 sonnet support + cost tracking (Anthropic API + Bedrock + Vertex AI + OpenRouter) +1. Anthropic API [Start here](https://docs.litellm.ai/docs/providers/anthropic#usage---thinking--reasoning_content) +2. Bedrock API [Start here](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) +3. Vertex AI API [See here](https://docs.litellm.ai/docs/providers/vertex#usage---thinking--reasoning_content) +4. OpenRouter [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L5626) +2. Gpt-4.5-preview support + cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L79) +3. Azure AI - Phi-4 cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L1773) +4. Claude-3.5-sonnet - vision support updated on Anthropic API [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2888) +5. Bedrock llama vision support [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L7714) +6. Cerebras llama3.3-70b pricing [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2697) + +## LLM Translation [​](https://docs.litellm.ai/release_notes\#llm-translation "Direct link to LLM Translation") + +1. Infinity Rerank - support returning documents when return\_documents=True [Start here](https://docs.litellm.ai/docs/providers/infinity#usage---returning-documents) +2. Amazon Deepseek - `` param extraction into ‘reasoning\_content’ [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-imported-models-deepseek-deepseek-r1) +3. Amazon Titan Embeddings - filter out ‘aws\_’ params from request body [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-embedding) +4. Anthropic ‘thinking’ + ‘reasoning\_content’ translation support (Anthropic API, Bedrock, Vertex AI) [Start here](https://docs.litellm.ai/docs/reasoning_content) +5. VLLM - support ‘video\_url’ [Start here](https://docs.litellm.ai/docs/providers/vllm#send-video-url-to-vllm) +6. Call proxy via litellm SDK: Support `litellm_proxy/` for embedding, image\_generation, transcription, speech, rerank [Start here](https://docs.litellm.ai/docs/providers/litellm_proxy) +7. OpenAI Pass-through - allow using Assistants GET, DELETE on /openai pass through routes [Start here](https://docs.litellm.ai/docs/pass_through/openai_passthrough) +8. Message Translation - fix openai message for assistant msg if role is missing - openai allows this +9. O1/O3 - support ‘drop\_params’ for o3-mini and o1 parallel\_tool\_calls param (not supported currently) [See here](https://docs.litellm.ai/docs/completion/drop_params) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Cost tracking for rerank via Bedrock [See PR](https://github.com/BerriAI/litellm/commit/b682dc4ec8fd07acf2f4c981d2721e36ae2a49c5) +2. Anthropic pass-through - fix race condition causing cost to not be tracked [See PR](https://github.com/BerriAI/litellm/pull/8874) +3. Anthropic pass-through: Ensure accurate token counting [See PR](https://github.com/BerriAI/litellm/pull/8880) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +01. Models Page - Allow sorting models by ‘created at’ +02. Models Page - Edit Model Flow Improvements +03. Models Page - Fix Adding Azure, Azure AI Studio models on UI +04. Internal Users Page - Allow Bulk Adding Internal Users on UI +05. Internal Users Page - Allow sorting users by ‘created at’ +06. Virtual Keys Page - Allow searching for UserIDs on the dropdown when assigning a user to a team [See PR](https://github.com/BerriAI/litellm/pull/8844) +07. Virtual Keys Page - allow creating a user when assigning keys to users [See PR](https://github.com/BerriAI/litellm/pull/8844) +08. Model Hub Page - fix text overflow issue [See PR](https://github.com/BerriAI/litellm/pull/8749) +09. Admin Settings Page - Allow adding MSFT SSO on UI +10. Backend - don't allow creating duplicate internal users in DB + +## Helm [​](https://docs.litellm.ai/release_notes\#helm "Direct link to Helm") + +1. support ttlSecondsAfterFinished on the migration job - [See PR](https://github.com/BerriAI/litellm/pull/8593) +2. enhance migrations job with additional configurable properties - [See PR](https://github.com/BerriAI/litellm/pull/8636) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Arize Phoenix support +2. ‘No-log’ - fix ‘no-log’ param support on embedding calls + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Single Deployment Cooldown logic - Use allowed\_fails or allowed\_fail\_policy if set [Start here](https://docs.litellm.ai/docs/routing#advanced-custom-retries-cooldowns-based-on-error-type) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Hypercorn - fix reading / parsing request body +2. Windows - fix running proxy in windows +3. DD-Trace - fix dd-trace enablement on proxy + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes\#complete-git-diff "Direct link to Complete Git Diff") + +View the complete git diff [here](https://github.com/BerriAI/litellm/compare/v1.61.13-stable...v1.61.20-stable). + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. New OpenAI `/image/variations` endpoint BETA support [Docs](https://docs.litellm.ai/docs/image_variations) +2. Topaz API support on OpenAI `/image/variations` BETA endpoint [Docs](https://docs.litellm.ai/docs/providers/topaz) +3. Deepseek - r1 support w/ reasoning\_content ( [Deepseek API](https://docs.litellm.ai/docs/providers/deepseek#reasoning-models), [Vertex AI](https://docs.litellm.ai/docs/providers/vertex#model-garden), [Bedrock](https://docs.litellm.ai/docs/providers/bedrock#deepseek)) +4. Azure - Add azure o1 pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L952) +5. Anthropic - handle `-latest` tag in model for cost calculation +6. Gemini-2.0-flash-thinking - add model pricing (it’s 0.0) [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L3393) +7. Bedrock - add stability sd3 model pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6814) (s/o [Marty Sullivan](https://github.com/marty-sullivan)) +8. Bedrock - add us.amazon.nova-lite-v1:0 to model cost map [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L5619) +9. TogetherAI - add new together\_ai llama3.3 models [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6985) + +## LLM Translation [​](https://docs.litellm.ai/release_notes\#llm-translation "Direct link to LLM Translation") + +01. LM Studio -> fix async embedding call +02. Gpt 4o models - fix response\_format translation +03. Bedrock nova - expand supported document types to include .md, .csv, etc. [Start Here](https://docs.litellm.ai/docs/providers/bedrock#usage---pdf--document-understanding) +04. Bedrock - docs on IAM role based access for bedrock - [Start Here](https://docs.litellm.ai/docs/providers/bedrock#sts-role-based-auth) +05. Bedrock - cache IAM role credentials when used +06. Google AI Studio ( `gemini/`) \- support gemini 'frequency\_penalty' and 'presence\_penalty' +07. Azure O1 - fix model name check +08. WatsonX - ZenAPIKey support for WatsonX [Docs](https://docs.litellm.ai/docs/providers/watsonx) +09. Ollama Chat - support json schema response format [Start Here](https://docs.litellm.ai/docs/providers/ollama#json-schema-support) +10. Bedrock - return correct bedrock status code and error message if error during streaming +11. Anthropic - Supported nested json schema on anthropic calls +12. OpenAI - `metadata` param preview support + 1. SDK - enable via `litellm.enable_preview_features = True` + 2. PROXY - enable via `litellm_settings::enable_preview_features: true` +13. Replicate - retry completion response on status=processing + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Bedrock - QA asserts all bedrock regional models have same `supported_` as base model +2. Bedrock - fix bedrock converse cost tracking w/ region name specified +3. Spend Logs reliability fix - when `user` passed in request body is int instead of string +4. Ensure ‘base\_model’ cost tracking works across all endpoints +5. Fixes for Image generation cost tracking +6. Anthropic - fix anthropic end user cost tracking +7. JWT / OIDC Auth - add end user id tracking from jwt auth + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +01. allows team member to become admin post-add (ui + endpoints) +02. New edit/delete button for updating team membership on UI +03. If team admin - show all team keys +04. Model Hub - clarify cost of models is per 1m tokens +05. Invitation Links - fix invalid url generated +06. New - SpendLogs Table Viewer - allows proxy admin to view spend logs on UI + 1. New spend logs - allow proxy admin to ‘opt in’ to logging request/response in spend logs table - enables easier abuse detection + 2. Show country of origin in spend logs + 3. Add pagination + filtering by key name/team name +07. `/key/delete` \- allow team admin to delete team keys +08. Internal User ‘view’ - fix spend calculation when team selected +09. Model Analytics is now on Free +10. Usage page - shows days when spend = 0, and round spend on charts to 2 sig figs +11. Public Teams - allow admins to expose teams for new users to ‘join’ on UI - [Start Here](https://docs.litellm.ai/docs/proxy/public_teams) +12. Guardrails + 1. set/edit guardrails on a virtual key + 2. Allow setting guardrails on a team + 3. Set guardrails on team create + edit page +13. Support temporary budget increases on `/key/update` \- new `temp_budget_increase` and `temp_budget_expiry` fields - [Start Here](https://docs.litellm.ai/docs/proxy/virtual_keys#temporary-budget-increase) +14. Support writing new key alias to AWS Secret Manager - on key rotation [Start Here](https://docs.litellm.ai/docs/secret#aws-secret-manager) + +## Helm [​](https://docs.litellm.ai/release_notes\#helm "Direct link to Helm") + +1. add securityContext and pull policy values to migration job (s/o [https://github.com/Hexoplon](https://github.com/Hexoplon)) +2. allow specifying envVars on values.yaml +3. new helm lint test + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Log the used prompt when prompt management used. [Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) +2. Support s3 logging with team alias prefixes - [Start Here](https://docs.litellm.ai/docs/proxy/logging#team-alias-prefix-in-object-key) +3. Prometheus [Start Here](https://docs.litellm.ai/docs/proxy/prometheus) +1. fix litellm\_llm\_api\_time\_to\_first\_token\_metric not populating for bedrock models +2. emit remaining team budget metric on regular basis (even when call isn’t made) - allows for more stable metrics on Grafana/etc. +3. add key and team level budget metrics +4. emit `litellm_overhead_latency_metric` +5. Emit `litellm_team_budget_reset_at_metric` and `litellm_api_key_budget_remaining_hours_metric` +4. Datadog - support logging spend tags to Datadog. [Start Here](https://docs.litellm.ai/docs/proxy/enterprise#tracking-spend-for-custom-tags) +5. Langfuse - fix logging request tags, read from standard logging payload +6. GCS - don’t truncate payload on logging +7. New GCS Pub/Sub logging support [Start Here](https://docs.litellm.ai/docs/proxy/logging#google-cloud-storage---pubsub-topic) +8. Add AIM Guardrails support [Start Here](https://docs.litellm.ai/docs/proxy/guardrails/aim_security) + +## Security [​](https://docs.litellm.ai/release_notes\#security "Direct link to Security") + +1. New Enterprise SLA for patching security vulnerabilities. [See Here](https://docs.litellm.ai/docs/enterprise#slas--professional-support) +2. Hashicorp - support using vault namespace for TLS auth. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) +3. Azure - DefaultAzureCredential support + +## Health Checks [​](https://docs.litellm.ai/release_notes\#health-checks "Direct link to Health Checks") + +1. Cleanup pricing-only model names from wildcard route list - prevent bad health checks +2. Allow specifying a health check model for wildcard routes - [https://docs.litellm.ai/docs/proxy/health#wildcard-routes](https://docs.litellm.ai/docs/proxy/health#wildcard-routes) +3. New ‘health\_check\_timeout ‘ param with default 1min upperbound to prevent bad model from health check to hang and cause pod restarts. [Start Here](https://docs.litellm.ai/docs/proxy/health#health-check-timeout) +4. Datadog - add data dog service health check + expose new `/health/services` endpoint. [Start Here](https://docs.litellm.ai/docs/proxy/health#healthservices) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +01. 3x increase in RPS - moving to orjson for reading request body +02. LLM Routing speedup - using cached get model group info +03. SDK speedup - using cached get model info helper - reduces CPU work to get model info +04. Proxy speedup - only read request body 1 time per request +05. Infinite loop detection scripts added to codebase +06. Bedrock - pure async image transformation requests +07. Cooldowns - single deployment model group if 100% calls fail in high traffic - prevents an o1 outage from impacting other calls +08. Response Headers - return + 1. `x-litellm-timeout` + 2. `x-litellm-attempted-retries` + 3. `x-litellm-overhead-duration-ms` + 4. `x-litellm-response-duration-ms` +09. ensure duplicate callbacks are not added to proxy +10. Requirements.txt - bump certifi version + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. JWT / OIDC Auth - new `enforce_rbac` param,allows proxy admin to prevent any unmapped yet authenticated jwt tokens from calling proxy. [Start Here](https://docs.litellm.ai/docs/proxy/token_auth#enforce-role-based-access-control-rbac) +2. fix custom openapi schema generation for customized swagger’s +3. Request Headers - support reading `x-litellm-timeout` param from request headers. Enables model timeout control when using Vercel’s AI SDK + LiteLLM Proxy. [Start Here](https://docs.litellm.ai/docs/proxy/request_headers#litellm-headers) +4. JWT / OIDC Auth - new `role` based permissions for model authentication. [See Here](https://docs.litellm.ai/docs/proxy/jwt_auth_arch) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes\#complete-git-diff "Direct link to Complete Git Diff") + +This is the diff between v1.57.8-stable and v1.59.8-stable. + +Use this to see the changes in the codebase. + +[**Git Diff**](https://github.com/BerriAI/litellm/compare/v1.57.8-stable...v1.59.8-stable) + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## UI Improvements [​](https://docs.litellm.ai/release_notes\#ui-improvements "Direct link to UI Improvements") + +### \[Opt In\] Admin UI - view messages / responses [​](https://docs.litellm.ai/release_notes\#opt-in-admin-ui---view-messages--responses "Direct link to opt-in-admin-ui---view-messages--responses") + +You can now view messages and response logs on Admin UI. + +![](https://docs.litellm.ai/assets/ideal-img/ui_logs.17b0459.1497.png) + +How to enable it - add `store_prompts_in_spend_logs: true` to your `proxy_config.yaml` + +Once this flag is enabled, your `messages` and `responses` will be stored in the `LiteLLM_Spend_Logs` table. + +```codeBlockLines_e6Vv +general_settings: + store_prompts_in_spend_logs: true + +``` + +## DB Schema Change [​](https://docs.litellm.ai/release_notes\#db-schema-change "Direct link to DB Schema Change") + +Added `messages` and `responses` to the `LiteLLM_Spend_Logs` table. + +**By default this is not logged.** If you want `messages` and `responses` to be logged, you need to opt in with this setting + +```codeBlockLines_e6Vv +general_settings: + store_prompts_in_spend_logs: true + +``` + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +`langfuse`, `management endpoints`, `ui`, `prometheus`, `secret management` + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +Langfuse Prompt Management is being labelled as BETA. This allows us to iterate quickly on the feedback we're receiving, and making the status clearer to users. We expect to make this feature to be stable by next month (February 2025). + +Changes: + +- Include the client message in the LLM API Request. (Previously only the prompt template was sent, and the client message was ignored). +- Log the prompt template in the logged request (e.g. to s3/langfuse). +- Log the 'prompt\_id' and 'prompt\_variables' in the logged request (e.g. to s3/langfuse). + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Team/Organization Management + UI Improvements [​](https://docs.litellm.ai/release_notes\#teamorganization-management--ui-improvements "Direct link to Team/Organization Management + UI Improvements") + +Managing teams and organizations on the UI is now easier. + +Changes: + +- Support for editing user role within team on UI. +- Support updating team member role to admin via api - `/team/member_update` +- Show team admins all keys for their team. +- Add organizations with budgets +- Assign teams to orgs on the UI +- Auto-assign SSO users to teams + +[Start Here](https://docs.litellm.ai/docs/proxy/self_serve) + +## Hashicorp Vault Support [​](https://docs.litellm.ai/release_notes\#hashicorp-vault-support "Direct link to Hashicorp Vault Support") + +We now support writing LiteLLM Virtual API keys to Hashicorp Vault. + +[Start Here](https://docs.litellm.ai/docs/proxy/vault) + +## Custom Prometheus Metrics [​](https://docs.litellm.ai/release_notes\#custom-prometheus-metrics "Direct link to Custom Prometheus Metrics") + +Define custom prometheus metrics, and track usage/latency/no. of requests against them + +This allows for more fine-grained tracking - e.g. on prompt template passed in request metadata + +[Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +`docker image`, `security`, `vulnerability` + +# 0 Critical/High Vulnerabilities + +![](https://docs.litellm.ai/assets/ideal-img/security.8eb0218.1200.png) + +## What changed? [​](https://docs.litellm.ai/release_notes\#what-changed "Direct link to What changed?") + +- LiteLLMBase image now uses `cgr.dev/chainguard/python:latest-dev` + +## Why the change? [​](https://docs.litellm.ai/release_notes\#why-the-change "Direct link to Why the change?") + +To ensure there are 0 critical/high vulnerabilities on LiteLLM Docker Image + +## Migration Guide [​](https://docs.litellm.ai/release_notes\#migration-guide "Direct link to Migration Guide") + +- If you use a custom dockerfile with litellm as a base image + `apt-get` + +Instead of `apt-get` use `apk`, the base litellm image will no longer have `apt-get` installed. + +**You are only impacted if you use `apt-get` in your Dockerfile** + +```codeBlockLines_e6Vv +# Use the provided base image +FROM ghcr.io/berriai/litellm:main-latest + +# Set the working directory +WORKDIR /app + +# Install dependencies - CHANGE THIS to `apk` +RUN apt-get update && apt-get install -y dumb-init + +``` + +Before Change + +```codeBlockLines_e6Vv +RUN apt-get update && apt-get install -y dumb-init + +``` + +After Change + +```codeBlockLines_e6Vv +RUN apk update && apk add --no-cache dumb-init + +``` + +`deepgram`, `fireworks ai`, `vision`, `admin ui`, `dependency upgrades` + +## New Models [​](https://docs.litellm.ai/release_notes\#new-models "Direct link to New Models") + +### **Deepgram Speech to Text** [​](https://docs.litellm.ai/release_notes\#deepgram-speech-to-text "Direct link to deepgram-speech-to-text") + +New Speech to Text support for Deepgram models. [**Start Here**](https://docs.litellm.ai/docs/providers/deepgram) + +```codeBlockLines_e6Vv +from litellm import transcription +import os + +# set api keys +os.environ["DEEPGRAM_API_KEY"] = "" +audio_file = open("/path/to/audio.mp3", "rb") + +response = transcription(model="deepgram/nova-2", file=audio_file) + +print(f"response: {response}") + +``` + +### **Fireworks AI - Vision** support for all models [​](https://docs.litellm.ai/release_notes\#fireworks-ai---vision-support-for-all-models "Direct link to fireworks-ai---vision-support-for-all-models") + +LiteLLM supports document inlining for Fireworks AI models. This is useful for models that are not vision models, but still need to parse documents/images/etc. +LiteLLM will add `#transform=inline` to the url of the image\_url, if the model is not a vision model [See Code](https://github.com/BerriAI/litellm/blob/1ae9d45798bdaf8450f2dfdec703369f3d2212b7/litellm/llms/fireworks_ai/chat/transformation.py#L114) + +## Proxy Admin UI [​](https://docs.litellm.ai/release_notes\#proxy-admin-ui "Direct link to Proxy Admin UI") + +- `Test Key` Tab displays `model` used in response + +![](https://docs.litellm.ai/assets/ideal-img/ui_model.72a8982.1920.png) + +- `Test Key` Tab renders content in `.md`, `.py` (any code/markdown format) + +![](https://docs.litellm.ai/assets/ideal-img/ui_format.337282b.1920.png) + +## Dependency Upgrades [​](https://docs.litellm.ai/release_notes\#dependency-upgrades "Direct link to Dependency Upgrades") + +- (Security fix) Upgrade to `fastapi==0.115.5` [https://github.com/BerriAI/litellm/pull/7447](https://github.com/BerriAI/litellm/pull/7447) + +## Bug Fixes [​](https://docs.litellm.ai/release_notes\#bug-fixes "Direct link to Bug Fixes") + +- Add health check support for realtime models [Here](https://docs.litellm.ai/docs/proxy/health#realtime-models) +- Health check error with audio\_transcription model [https://github.com/BerriAI/litellm/issues/5999](https://github.com/BerriAI/litellm/issues/5999) + +`guardrails`, `logging`, `virtual key management`, `new models` + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## New Features [​](https://docs.litellm.ai/release_notes\#new-features "Direct link to New Features") + +### ✨ Log Guardrail Traces [​](https://docs.litellm.ai/release_notes\#-log-guardrail-traces "Direct link to ✨ Log Guardrail Traces") + +Track guardrail failure rate and if a guardrail is going rogue and failing requests. [Start here](https://docs.litellm.ai/docs/proxy/guardrails/quick_start) + +#### Traced Guardrail Success [​](https://docs.litellm.ai/release_notes\#traced-guardrail-success "Direct link to Traced Guardrail Success") + +#### Traced Guardrail Failure [​](https://docs.litellm.ai/release_notes\#traced-guardrail-failure "Direct link to Traced Guardrail Failure") + +### `/guardrails/list` [​](https://docs.litellm.ai/release_notes\#guardrailslist "Direct link to guardrailslist") + +`/guardrails/list` allows clients to view available guardrails + supported guardrail params + +```codeBlockLines_e6Vv +curl -X GET 'http://0.0.0.0:4000/guardrails/list' + +``` + +Expected response + +```codeBlockLines_e6Vv +{ + "guardrails": [\ + {\ + "guardrail_name": "aporia-post-guard",\ + "guardrail_info": {\ + "params": [\ + {\ + "name": "toxicity_score",\ + "type": "float",\ + "description": "Score between 0-1 indicating content toxicity level"\ + },\ + {\ + "name": "pii_detection",\ + "type": "boolean"\ + }\ + ]\ + }\ + }\ + ] +} + +``` + +### ✨ Guardrails with Mock LLM [​](https://docs.litellm.ai/release_notes\#-guardrails-with-mock-llm "Direct link to ✨ Guardrails with Mock LLM") + +Send `mock_response` to test guardrails without making an LLM call. More info on `mock_response` [here](https://docs.litellm.ai/docs/proxy/guardrails/quick_start) + +```codeBlockLines_e6Vv +curl -i http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [\ + {"role": "user", "content": "hi my email is ishaan@berri.ai"}\ + ], + "mock_response": "This is a mock response", + "guardrails": ["aporia-pre-guard", "aporia-post-guard"] + }' + +``` + +### Assign Keys to Users [​](https://docs.litellm.ai/release_notes\#assign-keys-to-users "Direct link to Assign Keys to Users") + +You can now assign keys to users via Proxy UI + +## New Models [​](https://docs.litellm.ai/release_notes\#new-models "Direct link to New Models") + +- `openrouter/openai/o1` +- `vertex_ai/mistral-large@2411` + +## Fixes [​](https://docs.litellm.ai/release_notes\#fixes "Direct link to Fixes") + +- Fix `vertex_ai/` mistral model pricing: [https://github.com/BerriAI/litellm/pull/7345](https://github.com/BerriAI/litellm/pull/7345) +- Missing model\_group field in logs for aspeech call types [https://github.com/BerriAI/litellm/pull/7392](https://github.com/BerriAI/litellm/pull/7392) + +`key management`, `budgets/rate limits`, `logging`, `guardrails` + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## ✨ Budget / Rate Limit Tiers [​](https://docs.litellm.ai/release_notes\#-budget--rate-limit-tiers "Direct link to ✨ Budget / Rate Limit Tiers") + +Define tiers with rate limits. Assign them to keys. + +Use this to control access and budgets across a lot of keys. + +**[Start here](https://docs.litellm.ai/docs/proxy/rate_limit_tiers)** + +```codeBlockLines_e6Vv +curl -L -X POST 'http://0.0.0.0:4000/budget/new' \ +-H 'Authorization: Bearer sk-1234' \ +-H 'Content-Type: application/json' \ +-d '{ + "budget_id": "high-usage-tier", + "model_max_budget": { + "gpt-4o": {"rpm_limit": 1000000} + } +}' + +``` + +## OTEL Bug Fix [​](https://docs.litellm.ai/release_notes\#otel-bug-fix "Direct link to OTEL Bug Fix") + +LiteLLM was double logging litellm\_request span. This is now fixed. + +[Relevant PR](https://github.com/BerriAI/litellm/pull/7435) + +## Logging for Finetuning Endpoints [​](https://docs.litellm.ai/release_notes\#logging-for-finetuning-endpoints "Direct link to Logging for Finetuning Endpoints") + +Logs for finetuning requests are now available on all logging providers (e.g. Datadog). + +What's logged per request: + +- file\_id +- finetuning\_job\_id +- any key/team metadata + +**Start Here:** + +- [Setup Finetuning](https://docs.litellm.ai/docs/fine_tuning) +- [Setup Logging](https://docs.litellm.ai/docs/proxy/logging#datadog) + +## Dynamic Params for Guardrails [​](https://docs.litellm.ai/release_notes\#dynamic-params-for-guardrails "Direct link to Dynamic Params for Guardrails") + +You can now set custom parameters (like success threshold) for your guardrails in each request. + +[See guardrails spec for more details](https://docs.litellm.ai/docs/proxy/guardrails/custom_guardrail#-pass-additional-parameters-to-guardrail) + +`batches`, `guardrails`, `team management`, `custom auth` + +info + +Get a free 7-day LiteLLM Enterprise trial here. [Start here](https://www.litellm.ai/#trial) + +**No call needed** + +## ✨ Cost Tracking, Logging for Batches API ( `/batches`) [​](https://docs.litellm.ai/release_notes\#-cost-tracking-logging-for-batches-api-batches "Direct link to -cost-tracking-logging-for-batches-api-batches") + +Track cost, usage for Batch Creation Jobs. [Start here](https://docs.litellm.ai/docs/batches) + +## ✨ `/guardrails/list` endpoint [​](https://docs.litellm.ai/release_notes\#-guardrailslist-endpoint "Direct link to -guardrailslist-endpoint") + +Show available guardrails to users. [Start here](https://litellm-api.up.railway.app/#/Guardrails) + +## ✨ Allow teams to add models [​](https://docs.litellm.ai/release_notes\#-allow-teams-to-add-models "Direct link to ✨ Allow teams to add models") + +This enables team admins to call their own finetuned models via litellm proxy. [Start here](https://docs.litellm.ai/docs/proxy/team_model_add) + +## ✨ Common checks for custom auth [​](https://docs.litellm.ai/release_notes\#-common-checks-for-custom-auth "Direct link to ✨ Common checks for custom auth") + +Calling the internal common\_checks function in custom auth is now enforced as an enterprise feature. This allows admins to use litellm's default budget/auth checks within their custom auth implementation. [Start here](https://docs.litellm.ai/docs/proxy/virtual_keys#custom-auth) + +## ✨ Assigning team admins [​](https://docs.litellm.ai/release_notes\#-assigning-team-admins "Direct link to ✨ Assigning team admins") + +Team admins is graduating from beta and moving to our enterprise tier. This allows proxy admins to allow others to manage keys/models for their own teams (useful for projects in production). [Start here](https://docs.litellm.ai/docs/proxy/virtual_keys#restricting-key-generation) + +A new LiteLLM Stable release [just went out](https://github.com/BerriAI/litellm/releases/tag/v1.55.8-stable). Here are 5 updates since v1.52.2-stable. + +`langfuse`, `fallbacks`, `new models`, `azure_storage` + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +This makes it easy to run experiments or change the specific models `gpt-4o` to `gpt-4o-mini` on Langfuse, instead of making changes in your applications. [Start here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Control fallback prompts client-side [​](https://docs.litellm.ai/release_notes\#control-fallback-prompts-client-side "Direct link to Control fallback prompts client-side") + +> Claude prompts are different than OpenAI + +Pass in prompts specific to model when doing fallbacks. [Start here](https://docs.litellm.ai/docs/proxy/reliability#control-fallback-prompts) + +## New Providers / Models [​](https://docs.litellm.ai/release_notes\#new-providers--models "Direct link to New Providers / Models") + +- [NVIDIA Triton](https://developer.nvidia.com/triton-inference-server) `/infer` endpoint. [Start here](https://docs.litellm.ai/docs/providers/triton-inference-server) +- [Infinity](https://github.com/michaelfeil/infinity) Rerank Models [Start here](https://docs.litellm.ai/docs/providers/infinity) + +## ✨ Azure Data Lake Storage Support [​](https://docs.litellm.ai/release_notes\#-azure-data-lake-storage-support "Direct link to ✨ Azure Data Lake Storage Support") + +Send LLM usage (spend, tokens) data to [Azure Data Lake](https://learn.microsoft.com/en-us/azure/storage/blobs/data-lake-storage-introduction). This makes it easy to consume usage data on other services (eg. Databricks) +[Start here](https://docs.litellm.ai/docs/proxy/logging#azure-blob-storage) + +## Docker Run LiteLLM [​](https://docs.litellm.ai/release_notes\#docker-run-litellm "Direct link to Docker Run LiteLLM") + +```codeBlockLines_e6Vv +docker run \ +-e STORE_MODEL_IN_DB=True \ +-p 4000:4000 \ +ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.55.8-stable + +``` + +## Get Daily Updates [​](https://docs.litellm.ai/release_notes\#get-daily-updates "Direct link to Get Daily Updates") + +LiteLLM ships new releases every day. [Follow us on LinkedIn](https://www.linkedin.com/company/berri-ai/) to get daily updates. + +## LiteLLM Release Notes +[Skip to main content](https://docs.litellm.ai/release_notes/archive#__docusaurus_skipToContent_fallback) + +### 2024 + +- [December 29, 2024 \- v1.56.4](https://docs.litellm.ai/release_notes/v1.56.4) +- [December 28, 2024 \- v1.56.3](https://docs.litellm.ai/release_notes/v1.56.3) +- [December 27, 2024 \- v1.56.1](https://docs.litellm.ai/release_notes/v1.56.1) +- [December 24, 2024 \- v1.55.10](https://docs.litellm.ai/release_notes/v1.55.10) +- [December 22, 2024 \- v1.55.8-stable](https://docs.litellm.ai/release_notes/v1.55.8-stable) + +### 2025 + +- [May 17, 2025 \- v1.70.1-stable - Gemini Realtime API Support](https://docs.litellm.ai/release_notes/v1.70.1-stable) +- [May 10, 2025 \- v1.69.0-stable - Loadbalance Batch API Models](https://docs.litellm.ai/release_notes/v1.69.0-stable) +- [May 3, 2025 \- v1.68.0-stable](https://docs.litellm.ai/release_notes/v1.68.0-stable) +- [April 26, 2025 \- v1.67.4-stable - Improved User Management](https://docs.litellm.ai/release_notes/v1.67.4-stable) +- [April 19, 2025 \- v1.67.0-stable - SCIM Integration](https://docs.litellm.ai/release_notes/v1.67.0-stable) +- [April 12, 2025 \- v1.66.0-stable - Realtime API Cost Tracking](https://docs.litellm.ai/release_notes/v1.66.0-stable) +- [April 5, 2025 \- v1.65.4-stable](https://docs.litellm.ai/release_notes/v1.65.4-stable) +- [March 30, 2025 \- v1.65.0-stable - Model Context Protocol](https://docs.litellm.ai/release_notes/v1.65.0-stable) +- [March 28, 2025 \- v1.65.0 - Team Model Add - update](https://docs.litellm.ai/release_notes/v1.65.0) +- [March 22, 2025 \- v1.63.14-stable](https://docs.litellm.ai/release_notes/v1.63.14-stable) +- [March 15, 2025 \- v1.63.11-stable](https://docs.litellm.ai/release_notes/v1.63.11-stable) +- [March 8, 2025 \- v1.63.2-stable](https://docs.litellm.ai/release_notes/v1.63.2-stable) +- [March 5, 2025 \- v1.63.0 - Anthropic 'thinking' response update](https://docs.litellm.ai/release_notes/v1.63.0) +- [March 1, 2025 \- v1.61.20-stable](https://docs.litellm.ai/release_notes/v1.61.20-stable) +- [January 31, 2025 \- v1.59.8-stable](https://docs.litellm.ai/release_notes/v1.59.8-stable) +- [January 17, 2025 \- v1.59.0](https://docs.litellm.ai/release_notes/v1.59.0) +- [January 11, 2025 \- v1.57.8-stable](https://docs.litellm.ai/release_notes/v1.57.8-stable) +- [January 10, 2025 \- v1.57.7](https://docs.litellm.ai/release_notes/v1.57.7) +- [January 8, 2025 \- v1.57.3 - New Base Docker Image](https://docs.litellm.ai/release_notes/v1.57.3) + +## LiteLLM Release Tags +[Skip to main content](https://docs.litellm.ai/release_notes/tags#__docusaurus_skipToContent_fallback) + +# Tags + +## A + +- [admin ui3](https://docs.litellm.ai/release_notes/tags/admin-ui) +- [alerting1](https://docs.litellm.ai/release_notes/tags/alerting) +- [azure\_storage1](https://docs.litellm.ai/release_notes/tags/azure-storage) + +* * * + +## B + +- [batch1](https://docs.litellm.ai/release_notes/tags/batch) +- [batches1](https://docs.litellm.ai/release_notes/tags/batches) +- [budgets/rate limits1](https://docs.litellm.ai/release_notes/tags/budgets-rate-limits) + +* * * + +## C + +- [claude-3-7-sonnet3](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet) +- [cost\_tracking2](https://docs.litellm.ai/release_notes/tags/cost-tracking) +- [credential management2](https://docs.litellm.ai/release_notes/tags/credential-management) +- [custom auth1](https://docs.litellm.ai/release_notes/tags/custom-auth) +- [custom\_prompt\_management1](https://docs.litellm.ai/release_notes/tags/custom-prompt-management) + +* * * + +## D + +- [db schema2](https://docs.litellm.ai/release_notes/tags/db-schema) +- [deepgram1](https://docs.litellm.ai/release_notes/tags/deepgram) +- [dependency upgrades1](https://docs.litellm.ai/release_notes/tags/dependency-upgrades) +- [docker image1](https://docs.litellm.ai/release_notes/tags/docker-image) + +* * * + +## F + +- [fallbacks1](https://docs.litellm.ai/release_notes/tags/fallbacks) +- [finetuning1](https://docs.litellm.ai/release_notes/tags/finetuning) +- [fireworks ai1](https://docs.litellm.ai/release_notes/tags/fireworks-ai) + +* * * + +## G + +- [guardrails3](https://docs.litellm.ai/release_notes/tags/guardrails) + +* * * + +## H + +- [humanloop1](https://docs.litellm.ai/release_notes/tags/humanloop) + +* * * + +## K + +- [key management1](https://docs.litellm.ai/release_notes/tags/key-management) + +* * * + +## L + +- [langfuse3](https://docs.litellm.ai/release_notes/tags/langfuse) +- [llm translation3](https://docs.litellm.ai/release_notes/tags/llm-translation) +- [logging4](https://docs.litellm.ai/release_notes/tags/logging) + +* * * + +## M + +- [management endpoints3](https://docs.litellm.ai/release_notes/tags/management-endpoints) +- [mcp1](https://docs.litellm.ai/release_notes/tags/mcp) + +* * * + +## N + +- [new models2](https://docs.litellm.ai/release_notes/tags/new-models) + +* * * + +## P + +- [prometheus2](https://docs.litellm.ai/release_notes/tags/prometheus) +- [prompt management1](https://docs.litellm.ai/release_notes/tags/prompt-management) + +* * * + +## R + +- [reasoning\_content3](https://docs.litellm.ai/release_notes/tags/reasoning-content) +- [rerank1](https://docs.litellm.ai/release_notes/tags/rerank) +- [responses\_api3](https://docs.litellm.ai/release_notes/tags/responses-api) + +* * * + +## S + +- [secret management2](https://docs.litellm.ai/release_notes/tags/secret-management) +- [security4](https://docs.litellm.ai/release_notes/tags/security) +- [session\_management1](https://docs.litellm.ai/release_notes/tags/session-management) +- [snowflake2](https://docs.litellm.ai/release_notes/tags/snowflake) +- [sso2](https://docs.litellm.ai/release_notes/tags/sso) + +* * * + +## T + +- [team management1](https://docs.litellm.ai/release_notes/tags/team-management) +- [team models1](https://docs.litellm.ai/release_notes/tags/team-models) +- [thinking3](https://docs.litellm.ai/release_notes/tags/thinking) +- [thinking content2](https://docs.litellm.ai/release_notes/tags/thinking-content) + +* * * + +## U + +- [ui4](https://docs.litellm.ai/release_notes/tags/ui) +- [ui\_improvements1](https://docs.litellm.ai/release_notes/tags/ui-improvements) +- [unified\_file\_id2](https://docs.litellm.ai/release_notes/tags/unified-file-id) + +* * * + +## V + +- [virtual key management1](https://docs.litellm.ai/release_notes/tags/virtual-key-management) +- [vision1](https://docs.litellm.ai/release_notes/tags/vision) +- [vulnerability1](https://docs.litellm.ai/release_notes/tags/vulnerability) + +* * * + +## LiteLLM Admin UI Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/admin-ui#__docusaurus_skipToContent_fallback) + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. New OpenAI `/image/variations` endpoint BETA support [Docs](https://docs.litellm.ai/docs/image_variations) +2. Topaz API support on OpenAI `/image/variations` BETA endpoint [Docs](https://docs.litellm.ai/docs/providers/topaz) +3. Deepseek - r1 support w/ reasoning\_content ( [Deepseek API](https://docs.litellm.ai/docs/providers/deepseek#reasoning-models), [Vertex AI](https://docs.litellm.ai/docs/providers/vertex#model-garden), [Bedrock](https://docs.litellm.ai/docs/providers/bedrock#deepseek)) +4. Azure - Add azure o1 pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L952) +5. Anthropic - handle `-latest` tag in model for cost calculation +6. Gemini-2.0-flash-thinking - add model pricing (it’s 0.0) [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L3393) +7. Bedrock - add stability sd3 model pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6814) (s/o [Marty Sullivan](https://github.com/marty-sullivan)) +8. Bedrock - add us.amazon.nova-lite-v1:0 to model cost map [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L5619) +9. TogetherAI - add new together\_ai llama3.3 models [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6985) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#llm-translation "Direct link to LLM Translation") + +01. LM Studio -> fix async embedding call +02. Gpt 4o models - fix response\_format translation +03. Bedrock nova - expand supported document types to include .md, .csv, etc. [Start Here](https://docs.litellm.ai/docs/providers/bedrock#usage---pdf--document-understanding) +04. Bedrock - docs on IAM role based access for bedrock - [Start Here](https://docs.litellm.ai/docs/providers/bedrock#sts-role-based-auth) +05. Bedrock - cache IAM role credentials when used +06. Google AI Studio ( `gemini/`) \- support gemini 'frequency\_penalty' and 'presence\_penalty' +07. Azure O1 - fix model name check +08. WatsonX - ZenAPIKey support for WatsonX [Docs](https://docs.litellm.ai/docs/providers/watsonx) +09. Ollama Chat - support json schema response format [Start Here](https://docs.litellm.ai/docs/providers/ollama#json-schema-support) +10. Bedrock - return correct bedrock status code and error message if error during streaming +11. Anthropic - Supported nested json schema on anthropic calls +12. OpenAI - `metadata` param preview support + 1. SDK - enable via `litellm.enable_preview_features = True` + 2. PROXY - enable via `litellm_settings::enable_preview_features: true` +13. Replicate - retry completion response on status=processing + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Bedrock - QA asserts all bedrock regional models have same `supported_` as base model +2. Bedrock - fix bedrock converse cost tracking w/ region name specified +3. Spend Logs reliability fix - when `user` passed in request body is int instead of string +4. Ensure ‘base\_model’ cost tracking works across all endpoints +5. Fixes for Image generation cost tracking +6. Anthropic - fix anthropic end user cost tracking +7. JWT / OIDC Auth - add end user id tracking from jwt auth + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +01. allows team member to become admin post-add (ui + endpoints) +02. New edit/delete button for updating team membership on UI +03. If team admin - show all team keys +04. Model Hub - clarify cost of models is per 1m tokens +05. Invitation Links - fix invalid url generated +06. New - SpendLogs Table Viewer - allows proxy admin to view spend logs on UI + 1. New spend logs - allow proxy admin to ‘opt in’ to logging request/response in spend logs table - enables easier abuse detection + 2. Show country of origin in spend logs + 3. Add pagination + filtering by key name/team name +07. `/key/delete` \- allow team admin to delete team keys +08. Internal User ‘view’ - fix spend calculation when team selected +09. Model Analytics is now on Free +10. Usage page - shows days when spend = 0, and round spend on charts to 2 sig figs +11. Public Teams - allow admins to expose teams for new users to ‘join’ on UI - [Start Here](https://docs.litellm.ai/docs/proxy/public_teams) +12. Guardrails + 1. set/edit guardrails on a virtual key + 2. Allow setting guardrails on a team + 3. Set guardrails on team create + edit page +13. Support temporary budget increases on `/key/update` \- new `temp_budget_increase` and `temp_budget_expiry` fields - [Start Here](https://docs.litellm.ai/docs/proxy/virtual_keys#temporary-budget-increase) +14. Support writing new key alias to AWS Secret Manager - on key rotation [Start Here](https://docs.litellm.ai/docs/secret#aws-secret-manager) + +## Helm [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#helm "Direct link to Helm") + +1. add securityContext and pull policy values to migration job (s/o [https://github.com/Hexoplon](https://github.com/Hexoplon)) +2. allow specifying envVars on values.yaml +3. new helm lint test + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Log the used prompt when prompt management used. [Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) +2. Support s3 logging with team alias prefixes - [Start Here](https://docs.litellm.ai/docs/proxy/logging#team-alias-prefix-in-object-key) +3. Prometheus [Start Here](https://docs.litellm.ai/docs/proxy/prometheus) +1. fix litellm\_llm\_api\_time\_to\_first\_token\_metric not populating for bedrock models +2. emit remaining team budget metric on regular basis (even when call isn’t made) - allows for more stable metrics on Grafana/etc. +3. add key and team level budget metrics +4. emit `litellm_overhead_latency_metric` +5. Emit `litellm_team_budget_reset_at_metric` and `litellm_api_key_budget_remaining_hours_metric` +4. Datadog - support logging spend tags to Datadog. [Start Here](https://docs.litellm.ai/docs/proxy/enterprise#tracking-spend-for-custom-tags) +5. Langfuse - fix logging request tags, read from standard logging payload +6. GCS - don’t truncate payload on logging +7. New GCS Pub/Sub logging support [Start Here](https://docs.litellm.ai/docs/proxy/logging#google-cloud-storage---pubsub-topic) +8. Add AIM Guardrails support [Start Here](https://docs.litellm.ai/docs/proxy/guardrails/aim_security) + +## Security [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#security "Direct link to Security") + +1. New Enterprise SLA for patching security vulnerabilities. [See Here](https://docs.litellm.ai/docs/enterprise#slas--professional-support) +2. Hashicorp - support using vault namespace for TLS auth. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) +3. Azure - DefaultAzureCredential support + +## Health Checks [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#health-checks "Direct link to Health Checks") + +1. Cleanup pricing-only model names from wildcard route list - prevent bad health checks +2. Allow specifying a health check model for wildcard routes - [https://docs.litellm.ai/docs/proxy/health#wildcard-routes](https://docs.litellm.ai/docs/proxy/health#wildcard-routes) +3. New ‘health\_check\_timeout ‘ param with default 1min upperbound to prevent bad model from health check to hang and cause pod restarts. [Start Here](https://docs.litellm.ai/docs/proxy/health#health-check-timeout) +4. Datadog - add data dog service health check + expose new `/health/services` endpoint. [Start Here](https://docs.litellm.ai/docs/proxy/health#healthservices) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +01. 3x increase in RPS - moving to orjson for reading request body +02. LLM Routing speedup - using cached get model group info +03. SDK speedup - using cached get model info helper - reduces CPU work to get model info +04. Proxy speedup - only read request body 1 time per request +05. Infinite loop detection scripts added to codebase +06. Bedrock - pure async image transformation requests +07. Cooldowns - single deployment model group if 100% calls fail in high traffic - prevents an o1 outage from impacting other calls +08. Response Headers - return + 1. `x-litellm-timeout` + 2. `x-litellm-attempted-retries` + 3. `x-litellm-overhead-duration-ms` + 4. `x-litellm-response-duration-ms` +09. ensure duplicate callbacks are not added to proxy +10. Requirements.txt - bump certifi version + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. JWT / OIDC Auth - new `enforce_rbac` param,allows proxy admin to prevent any unmapped yet authenticated jwt tokens from calling proxy. [Start Here](https://docs.litellm.ai/docs/proxy/token_auth#enforce-role-based-access-control-rbac) +2. fix custom openapi schema generation for customized swagger’s +3. Request Headers - support reading `x-litellm-timeout` param from request headers. Enables model timeout control when using Vercel’s AI SDK + LiteLLM Proxy. [Start Here](https://docs.litellm.ai/docs/proxy/request_headers#litellm-headers) +4. JWT / OIDC Auth - new `role` based permissions for model authentication. [See Here](https://docs.litellm.ai/docs/proxy/jwt_auth_arch) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#complete-git-diff "Direct link to Complete Git Diff") + +This is the diff between v1.57.8-stable and v1.59.8-stable. + +Use this to see the changes in the codebase. + +[**Git Diff**](https://github.com/BerriAI/litellm/compare/v1.57.8-stable...v1.59.8-stable) + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## UI Improvements [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#ui-improvements "Direct link to UI Improvements") + +### \[Opt In\] Admin UI - view messages / responses [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#opt-in-admin-ui---view-messages--responses "Direct link to opt-in-admin-ui---view-messages--responses") + +You can now view messages and response logs on Admin UI. + +![](https://docs.litellm.ai/assets/ideal-img/ui_logs.17b0459.1497.png) + +How to enable it - add `store_prompts_in_spend_logs: true` to your `proxy_config.yaml` + +Once this flag is enabled, your `messages` and `responses` will be stored in the `LiteLLM_Spend_Logs` table. + +```codeBlockLines_e6Vv +general_settings: + store_prompts_in_spend_logs: true + +``` + +## DB Schema Change [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#db-schema-change "Direct link to DB Schema Change") + +Added `messages` and `responses` to the `LiteLLM_Spend_Logs` table. + +**By default this is not logged.** If you want `messages` and `responses` to be logged, you need to opt in with this setting + +```codeBlockLines_e6Vv +general_settings: + store_prompts_in_spend_logs: true + +``` + +`deepgram`, `fireworks ai`, `vision`, `admin ui`, `dependency upgrades` + +## New Models [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#new-models "Direct link to New Models") + +### **Deepgram Speech to Text** [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#deepgram-speech-to-text "Direct link to deepgram-speech-to-text") + +New Speech to Text support for Deepgram models. [**Start Here**](https://docs.litellm.ai/docs/providers/deepgram) + +```codeBlockLines_e6Vv +from litellm import transcription +import os + +# set api keys +os.environ["DEEPGRAM_API_KEY"] = "" +audio_file = open("/path/to/audio.mp3", "rb") + +response = transcription(model="deepgram/nova-2", file=audio_file) + +print(f"response: {response}") + +``` + +### **Fireworks AI - Vision** support for all models [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#fireworks-ai---vision-support-for-all-models "Direct link to fireworks-ai---vision-support-for-all-models") + +LiteLLM supports document inlining for Fireworks AI models. This is useful for models that are not vision models, but still need to parse documents/images/etc. +LiteLLM will add `#transform=inline` to the url of the image\_url, if the model is not a vision model [See Code](https://github.com/BerriAI/litellm/blob/1ae9d45798bdaf8450f2dfdec703369f3d2212b7/litellm/llms/fireworks_ai/chat/transformation.py#L114) + +## Proxy Admin UI [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#proxy-admin-ui "Direct link to Proxy Admin UI") + +- `Test Key` Tab displays `model` used in response + +![](https://docs.litellm.ai/assets/ideal-img/ui_model.72a8982.1920.png) + +- `Test Key` Tab renders content in `.md`, `.py` (any code/markdown format) + +![](https://docs.litellm.ai/assets/ideal-img/ui_format.337282b.1920.png) + +## Dependency Upgrades [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#dependency-upgrades "Direct link to Dependency Upgrades") + +- (Security fix) Upgrade to `fastapi==0.115.5` [https://github.com/BerriAI/litellm/pull/7447](https://github.com/BerriAI/litellm/pull/7447) + +## Bug Fixes [​](https://docs.litellm.ai/release_notes/tags/admin-ui\#bug-fixes "Direct link to Bug Fixes") + +- Add health check support for realtime models [Here](https://docs.litellm.ai/docs/proxy/health#realtime-models) +- Health check error with audio\_transcription model [https://github.com/BerriAI/litellm/issues/5999](https://github.com/BerriAI/litellm/issues/5999) + +## Alerting Features Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/alerting#__docusaurus_skipToContent_fallback) + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes/tags/alerting\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/alerting\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes/tags/alerting\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes/tags/alerting\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes/tags/alerting\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes/tags/alerting\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes/tags/alerting\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes/tags/alerting\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes/tags/alerting\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes/tags/alerting\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes/tags/alerting\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes/tags/alerting\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +## LiteLLM Azure Storage Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/azure-storage#__docusaurus_skipToContent_fallback) + +A new LiteLLM Stable release [just went out](https://github.com/BerriAI/litellm/releases/tag/v1.55.8-stable). Here are 5 updates since v1.52.2-stable. + +`langfuse`, `fallbacks`, `new models`, `azure_storage` + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes/tags/azure-storage\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +This makes it easy to run experiments or change the specific models `gpt-4o` to `gpt-4o-mini` on Langfuse, instead of making changes in your applications. [Start here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Control fallback prompts client-side [​](https://docs.litellm.ai/release_notes/tags/azure-storage\#control-fallback-prompts-client-side "Direct link to Control fallback prompts client-side") + +> Claude prompts are different than OpenAI + +Pass in prompts specific to model when doing fallbacks. [Start here](https://docs.litellm.ai/docs/proxy/reliability#control-fallback-prompts) + +## New Providers / Models [​](https://docs.litellm.ai/release_notes/tags/azure-storage\#new-providers--models "Direct link to New Providers / Models") + +- [NVIDIA Triton](https://developer.nvidia.com/triton-inference-server) `/infer` endpoint. [Start here](https://docs.litellm.ai/docs/providers/triton-inference-server) +- [Infinity](https://github.com/michaelfeil/infinity) Rerank Models [Start here](https://docs.litellm.ai/docs/providers/infinity) + +## ✨ Azure Data Lake Storage Support [​](https://docs.litellm.ai/release_notes/tags/azure-storage\#-azure-data-lake-storage-support "Direct link to ✨ Azure Data Lake Storage Support") + +Send LLM usage (spend, tokens) data to [Azure Data Lake](https://learn.microsoft.com/en-us/azure/storage/blobs/data-lake-storage-introduction). This makes it easy to consume usage data on other services (eg. Databricks) +[Start here](https://docs.litellm.ai/docs/proxy/logging#azure-blob-storage) + +## Docker Run LiteLLM [​](https://docs.litellm.ai/release_notes/tags/azure-storage\#docker-run-litellm "Direct link to Docker Run LiteLLM") + +```codeBlockLines_e6Vv +docker run \ +-e STORE_MODEL_IN_DB=True \ +-p 4000:4000 \ +ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.55.8-stable + +``` + +## Get Daily Updates [​](https://docs.litellm.ai/release_notes/tags/azure-storage\#get-daily-updates "Direct link to Get Daily Updates") + +LiteLLM ships new releases every day. [Follow us on LinkedIn](https://www.linkedin.com/company/berri-ai/) to get daily updates. + +## Batch Processing Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/batch#__docusaurus_skipToContent_fallback) + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes/tags/batch\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/batch\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes/tags/batch\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes/tags/batch\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes/tags/batch\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes/tags/batch\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes/tags/batch\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes/tags/batch\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes/tags/batch\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes/tags/batch\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes/tags/batch\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes/tags/batch\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +## Batches API Features +[Skip to main content](https://docs.litellm.ai/release_notes/tags/batches#__docusaurus_skipToContent_fallback) + +`batches`, `guardrails`, `team management`, `custom auth` + +![](https://docs.litellm.ai/assets/ideal-img/batches_cost_tracking.8fc9663.1208.png) + +info + +Get a free 7-day LiteLLM Enterprise trial here. [Start here](https://www.litellm.ai/#trial) + +**No call needed** + +## ✨ Cost Tracking, Logging for Batches API ( `/batches`) [​](https://docs.litellm.ai/release_notes/tags/batches\#-cost-tracking-logging-for-batches-api-batches "Direct link to -cost-tracking-logging-for-batches-api-batches") + +Track cost, usage for Batch Creation Jobs. [Start here](https://docs.litellm.ai/docs/batches) + +## ✨ `/guardrails/list` endpoint [​](https://docs.litellm.ai/release_notes/tags/batches\#-guardrailslist-endpoint "Direct link to -guardrailslist-endpoint") + +Show available guardrails to users. [Start here](https://litellm-api.up.railway.app/#/Guardrails) + +## ✨ Allow teams to add models [​](https://docs.litellm.ai/release_notes/tags/batches\#-allow-teams-to-add-models "Direct link to ✨ Allow teams to add models") + +This enables team admins to call their own finetuned models via litellm proxy. [Start here](https://docs.litellm.ai/docs/proxy/team_model_add) + +## ✨ Common checks for custom auth [​](https://docs.litellm.ai/release_notes/tags/batches\#-common-checks-for-custom-auth "Direct link to ✨ Common checks for custom auth") + +Calling the internal common\_checks function in custom auth is now enforced as an enterprise feature. This allows admins to use litellm's default budget/auth checks within their custom auth implementation. [Start here](https://docs.litellm.ai/docs/proxy/virtual_keys#custom-auth) + +## ✨ Assigning team admins [​](https://docs.litellm.ai/release_notes/tags/batches\#-assigning-team-admins "Direct link to ✨ Assigning team admins") + +Team admins is graduating from beta and moving to our enterprise tier. This allows proxy admins to allow others to manage keys/models for their own teams (useful for projects in production). [Start here](https://docs.litellm.ai/docs/proxy/virtual_keys#restricting-key-generation) + +## Budgets and Rate Limits +[Skip to main content](https://docs.litellm.ai/release_notes/tags/budgets-rate-limits#__docusaurus_skipToContent_fallback) + +`key management`, `budgets/rate limits`, `logging`, `guardrails` + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## ✨ Budget / Rate Limit Tiers [​](https://docs.litellm.ai/release_notes/tags/budgets-rate-limits\#-budget--rate-limit-tiers "Direct link to ✨ Budget / Rate Limit Tiers") + +Define tiers with rate limits. Assign them to keys. + +Use this to control access and budgets across a lot of keys. + +**[Start here](https://docs.litellm.ai/docs/proxy/rate_limit_tiers)** + +```codeBlockLines_e6Vv +curl -L -X POST 'http://0.0.0.0:4000/budget/new' \ +-H 'Authorization: Bearer sk-1234' \ +-H 'Content-Type: application/json' \ +-d '{ + "budget_id": "high-usage-tier", + "model_max_budget": { + "gpt-4o": {"rpm_limit": 1000000} + } +}' + +``` + +## OTEL Bug Fix [​](https://docs.litellm.ai/release_notes/tags/budgets-rate-limits\#otel-bug-fix "Direct link to OTEL Bug Fix") + +LiteLLM was double logging litellm\_request span. This is now fixed. + +[Relevant PR](https://github.com/BerriAI/litellm/pull/7435) + +## Logging for Finetuning Endpoints [​](https://docs.litellm.ai/release_notes/tags/budgets-rate-limits\#logging-for-finetuning-endpoints "Direct link to Logging for Finetuning Endpoints") + +Logs for finetuning requests are now available on all logging providers (e.g. Datadog). + +What's logged per request: + +- file\_id +- finetuning\_job\_id +- any key/team metadata + +**Start Here:** + +- [Setup Finetuning](https://docs.litellm.ai/docs/fine_tuning) +- [Setup Logging](https://docs.litellm.ai/docs/proxy/logging#datadog) + +## Dynamic Params for Guardrails [​](https://docs.litellm.ai/release_notes/tags/budgets-rate-limits\#dynamic-params-for-guardrails "Direct link to Dynamic Params for Guardrails") + +You can now set custom parameters (like success threshold) for your guardrails in each request. + +[See guardrails spec for more details](https://docs.litellm.ai/docs/proxy/guardrails/custom_guardrail#-pass-additional-parameters-to-guardrail) + +## Claude 3.7 Sonnet Release +[Skip to main content](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet#__docusaurus_skipToContent_fallback) + +These are the changes since `v1.61.20-stable`. + +This release is primarily focused on: + +- LLM Translation improvements (more `thinking` content improvements) +- UI improvements (Error logs now shown on UI) + +info + +This release will be live on 03/09/2025 + +![](https://docs.litellm.ai/assets/ideal-img/v1632_release.7b42da1.1920.jpg) + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Add `supports_pdf_input` for specific Bedrock Claude models [PR](https://github.com/BerriAI/litellm/commit/f63cf0030679fe1a43d03fb196e815a0f28dae92) +2. Add pricing for amazon `eu` models [PR](https://github.com/BerriAI/litellm/commits/main/model_prices_and_context_window.json) +3. Fix Azure O1 mini pricing [PR](https://github.com/BerriAI/litellm/commit/52de1949ef2f76b8572df751f9c868a016d4832c) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#llm-translation "Direct link to LLM Translation") + +![](https://docs.litellm.ai/assets/ideal-img/anthropic_thinking.3bef9d6.1920.jpg) + +01. Support `/openai/` passthrough for Assistant endpoints. [Get Started](https://docs.litellm.ai/docs/pass_through/openai_passthrough) +02. Bedrock Claude - fix tool calling transformation on invoke route. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---function-calling--tool-calling) +03. Bedrock Claude - response\_format support for claude on invoke route. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---structured-output--json-mode) +04. Bedrock - pass `description` if set in response\_format. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---structured-output--json-mode) +05. Bedrock - Fix passing response\_format: {"type": "text"}. [PR](https://github.com/BerriAI/litellm/commit/c84b489d5897755139aa7d4e9e54727ebe0fa540) +06. OpenAI - Handle sending image\_url as str to openai. [Get Started](https://docs.litellm.ai/docs/completion/vision) +07. Deepseek - return 'reasoning\_content' missing on streaming. [Get Started](https://docs.litellm.ai/docs/reasoning_content) +08. Caching - Support caching on reasoning content. [Get Started](https://docs.litellm.ai/docs/proxy/caching) +09. Bedrock - handle thinking blocks in assistant message. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) +10. Anthropic - Return `signature` on streaming. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) + +- Note: We've also migrated from `signature_delta` to `signature`. [Read more](https://docs.litellm.ai/release_notes/v1.63.0) + +11. Support format param for specifying image type. [Get Started](https://docs.litellm.ai/docs/completion/vision.md#explicitly-specify-image-type) +12. Anthropic - `/v1/messages` endpoint - `thinking` param support. [Get Started](https://docs.litellm.ai/docs/anthropic_unified.md) + +- Note: this refactors the \[BETA\] unified `/v1/messages` endpoint, to just work for the Anthropic API. + +13. Vertex AI - handle $id in response schema when calling vertex ai. [Get Started](https://docs.litellm.ai/docs/providers/vertex#json-schema) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Batches API - Fix cost calculation to run on retrieve\_batch. [Get Started](https://docs.litellm.ai/docs/batches) +2. Batches API - Log batch models in spend logs / standard logging payload. [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec.md#standardlogginghiddenparams) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +![](https://docs.litellm.ai/assets/ideal-img/error_logs.63c5dc9.1920.jpg) + +1. Virtual Keys Page + - Allow team/org filters to be searchable on the Create Key Page + - Add created\_by and updated\_by fields to Keys table + - Show 'user\_email' on key table + - Show 100 Keys Per Page, Use full height, increase width of key alias +2. Logs Page + - Show Error Logs on LiteLLM UI + - Allow Internal Users to View their own logs +3. Internal Users Page + - Allow admin to control default model access for internal users +4. Fix session handling with cookies + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Fix prometheus metrics w/ custom metrics, when keys containing team\_id make requests. [PR](https://github.com/BerriAI/litellm/pull/8935) + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Cooldowns - Support cooldowns on models called with client side credentials. [Get Started](https://docs.litellm.ai/docs/proxy/clientside_auth#pass-user-llm-api-keys--api-base) +2. Tag-based Routing - ensures tag-based routing across all endpoints ( `/embeddings`, `/image_generation`, etc.). [Get Started](https://docs.litellm.ai/docs/proxy/tag_routing) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Raise BadRequestError when unknown model passed in request +2. Enforce model access restrictions on Azure OpenAI proxy route +3. Reliability fix - Handle emoji’s in text - fix orjson error +4. Model Access Patch - don't overwrite litellm.anthropic\_models when running auth checks +5. Enable setting timezone information in docker image + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.61.20-stable...v1.63.2-stable) + +v1.63.0 fixes Anthropic 'thinking' response on streaming to return the `signature` block. [Github Issue](https://github.com/BerriAI/litellm/issues/8964) + +It also moves the response structure from `signature_delta` to `signature` to be the same as Anthropic. [Anthropic Docs](https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking#implementing-extended-thinking) + +## Diff [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#diff "Direct link to Diff") + +```codeBlockLines_e6Vv +"message": { + ... + "reasoning_content": "The capital of France is Paris.", + "thinking_blocks": [\ + {\ + "type": "thinking",\ + "thinking": "The capital of France is Paris.",\ +- "signature_delta": "EqoBCkgIARABGAIiQL2UoU0b1OHYi+..." # 👈 OLD FORMAT\ ++ "signature": "EqoBCkgIARABGAIiQL2UoU0b1OHYi+..." # 👈 KEY CHANGE\ + }\ + ] +} + +``` + +These are the changes since `v1.61.13-stable`. + +This release is primarily focused on: + +- LLM Translation improvements (claude-3-7-sonnet + 'thinking'/'reasoning\_content' support) +- UI improvements (add model flow, user management, etc) + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Anthropic 3-7 sonnet support + cost tracking (Anthropic API + Bedrock + Vertex AI + OpenRouter) +1. Anthropic API [Start here](https://docs.litellm.ai/docs/providers/anthropic#usage---thinking--reasoning_content) +2. Bedrock API [Start here](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) +3. Vertex AI API [See here](https://docs.litellm.ai/docs/providers/vertex#usage---thinking--reasoning_content) +4. OpenRouter [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L5626) +2. Gpt-4.5-preview support + cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L79) +3. Azure AI - Phi-4 cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L1773) +4. Claude-3.5-sonnet - vision support updated on Anthropic API [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2888) +5. Bedrock llama vision support [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L7714) +6. Cerebras llama3.3-70b pricing [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2697) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#llm-translation "Direct link to LLM Translation") + +1. Infinity Rerank - support returning documents when return\_documents=True [Start here](https://docs.litellm.ai/docs/providers/infinity#usage---returning-documents) +2. Amazon Deepseek - `` param extraction into ‘reasoning\_content’ [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-imported-models-deepseek-deepseek-r1) +3. Amazon Titan Embeddings - filter out ‘aws\_’ params from request body [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-embedding) +4. Anthropic ‘thinking’ + ‘reasoning\_content’ translation support (Anthropic API, Bedrock, Vertex AI) [Start here](https://docs.litellm.ai/docs/reasoning_content) +5. VLLM - support ‘video\_url’ [Start here](https://docs.litellm.ai/docs/providers/vllm#send-video-url-to-vllm) +6. Call proxy via litellm SDK: Support `litellm_proxy/` for embedding, image\_generation, transcription, speech, rerank [Start here](https://docs.litellm.ai/docs/providers/litellm_proxy) +7. OpenAI Pass-through - allow using Assistants GET, DELETE on /openai pass through routes [Start here](https://docs.litellm.ai/docs/pass_through/openai_passthrough) +8. Message Translation - fix openai message for assistant msg if role is missing - openai allows this +9. O1/O3 - support ‘drop\_params’ for o3-mini and o1 parallel\_tool\_calls param (not supported currently) [See here](https://docs.litellm.ai/docs/completion/drop_params) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Cost tracking for rerank via Bedrock [See PR](https://github.com/BerriAI/litellm/commit/b682dc4ec8fd07acf2f4c981d2721e36ae2a49c5) +2. Anthropic pass-through - fix race condition causing cost to not be tracked [See PR](https://github.com/BerriAI/litellm/pull/8874) +3. Anthropic pass-through: Ensure accurate token counting [See PR](https://github.com/BerriAI/litellm/pull/8880) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +01. Models Page - Allow sorting models by ‘created at’ +02. Models Page - Edit Model Flow Improvements +03. Models Page - Fix Adding Azure, Azure AI Studio models on UI +04. Internal Users Page - Allow Bulk Adding Internal Users on UI +05. Internal Users Page - Allow sorting users by ‘created at’ +06. Virtual Keys Page - Allow searching for UserIDs on the dropdown when assigning a user to a team [See PR](https://github.com/BerriAI/litellm/pull/8844) +07. Virtual Keys Page - allow creating a user when assigning keys to users [See PR](https://github.com/BerriAI/litellm/pull/8844) +08. Model Hub Page - fix text overflow issue [See PR](https://github.com/BerriAI/litellm/pull/8749) +09. Admin Settings Page - Allow adding MSFT SSO on UI +10. Backend - don't allow creating duplicate internal users in DB + +## Helm [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#helm "Direct link to Helm") + +1. support ttlSecondsAfterFinished on the migration job - [See PR](https://github.com/BerriAI/litellm/pull/8593) +2. enhance migrations job with additional configurable properties - [See PR](https://github.com/BerriAI/litellm/pull/8636) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Arize Phoenix support +2. ‘No-log’ - fix ‘no-log’ param support on embedding calls + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Single Deployment Cooldown logic - Use allowed\_fails or allowed\_fail\_policy if set [Start here](https://docs.litellm.ai/docs/routing#advanced-custom-retries-cooldowns-based-on-error-type) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Hypercorn - fix reading / parsing request body +2. Windows - fix running proxy in windows +3. DD-Trace - fix dd-trace enablement on proxy + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet\#complete-git-diff "Direct link to Complete Git Diff") + +View the complete git diff [here](https://github.com/BerriAI/litellm/compare/v1.61.13-stable...v1.61.20-stable). + +## Cost Tracking Features +[Skip to main content](https://docs.litellm.ai/release_notes/tags/cost-tracking#__docusaurus_skipToContent_fallback) + +## Key Highlights [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#key-highlights "Direct link to Key Highlights") + +- **SCIM Integration**: Enables identity providers (Okta, Azure AD, OneLogin, etc.) to automate user and team (group) provisioning, updates, and deprovisioning +- **Team and Tag based usage tracking**: You can now see usage and spend by team and tag at 1M+ spend logs. +- **Unified Responses API**: Support for calling Anthropic, Gemini, Groq, etc. via OpenAI's new Responses API. + +Let's dive in. + +## SCIM Integration [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#scim-integration "Direct link to SCIM Integration") + +![](https://docs.litellm.ai/assets/ideal-img/scim_integration.01959e2.1200.png) + +This release adds SCIM support to LiteLLM. This allows your SSO provider (Okta, Azure AD, etc) to automatically create/delete users, teams, and memberships on LiteLLM. This means that when you remove a team on your SSO provider, your SSO provider will automatically delete the corresponding team on LiteLLM. + +[Read more](https://docs.litellm.ai/docs/tutorials/scim_litellm) + +## Team and Tag based usage tracking [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#team-and-tag-based-usage-tracking "Direct link to Team and Tag based usage tracking") + +![](https://docs.litellm.ai/assets/ideal-img/new_team_usage_highlight.60482cc.1920.jpg) + +This release improves team and tag based usage tracking at 1m+ spend logs, making it easy to monitor your LLM API Spend in production. This covers: + +- View **daily spend** by teams + tags +- View **usage / spend by key**, within teams +- View **spend by multiple tags** +- Allow **internal users** to view spend of teams they're a member of + +[Read more](https://docs.litellm.ai/release_notes/tags/cost-tracking#management-endpoints--ui) + +## Unified Responses API [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#unified-responses-api "Direct link to Unified Responses API") + +This release allows you to call Azure OpenAI, Anthropic, AWS Bedrock, and Google Vertex AI models via the POST /v1/responses endpoint on LiteLLM. This means you can now use popular tools like [OpenAI Codex](https://docs.litellm.ai/docs/tutorials/openai_codex) with your own models. + +![](https://docs.litellm.ai/assets/ideal-img/unified_responses_api_rn.0acc91a.1920.png) + +[Read more](https://docs.litellm.ai/docs/response_api) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **OpenAI** +1. gpt-4.1, gpt-4.1-mini, gpt-4.1-nano, o3, o3-mini, o4-mini pricing - [Get Started](https://docs.litellm.ai/docs/providers/openai#usage), [PR](https://github.com/BerriAI/litellm/pull/9990) +2. o4 - correctly map o4 to openai o\_series model +- **Azure AI** +1. Phi-4 output cost per token fix - [PR](https://github.com/BerriAI/litellm/pull/9880) +2. Responses API support [Get Started](https://docs.litellm.ai/docs/providers/azure#azure-responses-api), [PR](https://github.com/BerriAI/litellm/pull/10116) +- **Anthropic** +1. redacted message thinking support - [Get Started](https://docs.litellm.ai/docs/providers/anthropic#usage---thinking--reasoning_content), [PR](https://github.com/BerriAI/litellm/pull/10129) +- **Cohere** +1. `/v2/chat` Passthrough endpoint support w/ cost tracking - [Get Started](https://docs.litellm.ai/docs/pass_through/cohere), [PR](https://github.com/BerriAI/litellm/pull/9997) +- **Azure** +1. Support azure tenant\_id/client\_id env vars - [Get Started](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret), [PR](https://github.com/BerriAI/litellm/pull/9993) +2. Fix response\_format check for 2025+ api versions - [PR](https://github.com/BerriAI/litellm/pull/9993) +3. Add gpt-4.1, gpt-4.1-mini, gpt-4.1-nano, o3, o3-mini, o4-mini pricing +- **VLLM** +1. Files - Support 'file' message type for VLLM video url's - [Get Started](https://docs.litellm.ai/docs/providers/vllm#send-video-url-to-vllm), [PR](https://github.com/BerriAI/litellm/pull/10129) +2. Passthrough - new `/vllm/` passthrough endpoint support [Get Started](https://docs.litellm.ai/docs/pass_through/vllm), [PR](https://github.com/BerriAI/litellm/pull/10002) +- **Mistral** +1. new `/mistral` passthrough endpoint support [Get Started](https://docs.litellm.ai/docs/pass_through/mistral), [PR](https://github.com/BerriAI/litellm/pull/10002) +- **AWS** +1. New mapped bedrock regions - [PR](https://github.com/BerriAI/litellm/pull/9430) +- **VertexAI / Google AI Studio** +1. Gemini - Response format - Retain schema field ordering for google gemini and vertex by specifying propertyOrdering - [Get Started](https://docs.litellm.ai/docs/providers/vertex#json-schema), [PR](https://github.com/BerriAI/litellm/pull/9828) +2. Gemini-2.5-flash - return reasoning content [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini#usage---thinking--reasoning_content), [Vertex AI](https://docs.litellm.ai/docs/providers/vertex#thinking--reasoning_content) +3. Gemini-2.5-flash - pricing + model information [PR](https://github.com/BerriAI/litellm/pull/10125) +4. Passthrough - new `/vertex_ai/discovery` route - enables calling AgentBuilder API routes [Get Started](https://docs.litellm.ai/docs/pass_through/vertex_ai#supported-api-endpoints), [PR](https://github.com/BerriAI/litellm/pull/10084) +- **Fireworks AI** +1. return tool calling responses in `tool_calls` field (fireworks incorrectly returns this as a json str in content) [PR](https://github.com/BerriAI/litellm/pull/10130) +- **Triton** +1. Remove fixed remove bad\_words / stop words from `/generate` call - [Get Started](https://docs.litellm.ai/docs/providers/triton-inference-server#triton-generate---chat-completion), [PR](https://github.com/BerriAI/litellm/pull/10163) +- **Other** +1. Support for all litellm providers on Responses API (works with Codex) - [Get Started](https://docs.litellm.ai/docs/tutorials/openai_codex), [PR](https://github.com/BerriAI/litellm/pull/10132) +2. Fix combining multiple tool calls in streaming response - [Get Started](https://docs.litellm.ai/docs/completion/stream#helper-function), [PR](https://github.com/BerriAI/litellm/pull/10040) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **Cost Control** \- inject cache control points in prompt for cost reduction [Get Started](https://docs.litellm.ai/docs/tutorials/prompt_caching), [PR](https://github.com/BerriAI/litellm/pull/10000) +- **Spend Tags** \- spend tags in headers - support x-litellm-tags even if tag based routing not enabled [Get Started](https://docs.litellm.ai/docs/proxy/request_headers#litellm-headers), [PR](https://github.com/BerriAI/litellm/pull/10000) +- **Gemini-2.5-flash** \- support cost calculation for reasoning tokens [PR](https://github.com/BerriAI/litellm/pull/10141) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +- **Users** + +1. Show created\_at and updated\_at on users page - [PR](https://github.com/BerriAI/litellm/pull/10033) +- **Virtual Keys** + +1. Filter by key alias - [https://github.com/BerriAI/litellm/pull/10085](https://github.com/BerriAI/litellm/pull/10085) +- **Usage Tab** + +1. Team based usage + + + - New `LiteLLM_DailyTeamSpend` Table for aggregate team based usage logging - [PR](https://github.com/BerriAI/litellm/pull/10039) + + - New Team based usage dashboard + new `/team/daily/activity` API - [PR](https://github.com/BerriAI/litellm/pull/10081) + + - Return team alias on /team/daily/activity API - [PR](https://github.com/BerriAI/litellm/pull/10157) + + - allow internal user view spend for teams they belong to - [PR](https://github.com/BerriAI/litellm/pull/10157) + + - allow viewing top keys by team - [PR](https://github.com/BerriAI/litellm/pull/10157) + + +![](https://docs.litellm.ai/assets/ideal-img/new_team_usage.9237b43.1754.png) + +2. Tag Based Usage + + - New `LiteLLM_DailyTagSpend` Table for aggregate tag based usage logging - [PR](https://github.com/BerriAI/litellm/pull/10071) + - Restrict to only Proxy Admins - [PR](https://github.com/BerriAI/litellm/pull/10157) + - allow viewing top keys by tag + - Return tags passed in request (i.e. dynamic tags) on `/tag/list` API - [PR](https://github.com/BerriAI/litellm/pull/10157) + ![](https://docs.litellm.ai/assets/ideal-img/new_tag_usage.cd55b64.1863.png) +3. Track prompt caching metrics in daily user, team, tag tables - [PR](https://github.com/BerriAI/litellm/pull/10029) + +4. Show usage by key (on all up, team, and tag usage dashboards) - [PR](https://github.com/BerriAI/litellm/pull/10157) + +5. swap old usage with new usage tab +- **Models** + +1. Make columns resizable/hideable - [PR](https://github.com/BerriAI/litellm/pull/10119) +- **API Playground** + +1. Allow internal user to call api playground - [PR](https://github.com/BerriAI/litellm/pull/10157) +- **SCIM** + +1. Add LiteLLM SCIM Integration for Team and User management - [Get Started](https://docs.litellm.ai/docs/tutorials/scim_litellm), [PR](https://github.com/BerriAI/litellm/pull/10072) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +- **GCS** +1. Fix gcs pub sub logging with env var GCS\_PROJECT\_ID - [Get Started](https://docs.litellm.ai/docs/observability/gcs_bucket_integration#usage), [PR](https://github.com/BerriAI/litellm/pull/10042) +- **AIM** +1. Add litellm call id passing to Aim guardrails on pre and post-hooks calls - [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/aim_security), [PR](https://github.com/BerriAI/litellm/pull/10021) +- **Azure blob storage** +1. Ensure logging works in high throughput scenarios - [Get Started](https://docs.litellm.ai/docs/proxy/logging#azure-blob-storage), [PR](https://github.com/BerriAI/litellm/pull/9962) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Support setting `litellm.modify_params` via env var** [PR](https://github.com/BerriAI/litellm/pull/9964) +- **Model Discovery** \- Check provider’s `/models` endpoints when calling proxy’s `/v1/models` endpoint - [Get Started](https://docs.litellm.ai/docs/proxy/model_discovery), [PR](https://github.com/BerriAI/litellm/pull/9958) +- **`/utils/token_counter`** \- fix retrieving custom tokenizer for db models - [Get Started](https://docs.litellm.ai/docs/proxy/configs#set-custom-tokenizer), [PR](https://github.com/BerriAI/litellm/pull/10047) +- **Prisma migrate** \- handle existing columns in db table - [PR](https://github.com/BerriAI/litellm/pull/10138) + +## Deploy this version [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.66.0-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.66.0.post1 + +``` + +v1.66.0-stable is live now, here are the key highlights of this release + +## Key Highlights [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#key-highlights "Direct link to Key Highlights") + +- **Realtime API Cost Tracking**: Track cost of realtime API calls +- **Microsoft SSO Auto-sync**: Auto-sync groups and group members from Azure Entra ID to LiteLLM +- **xAI grok-3**: Added support for `xai/grok-3` models +- **Security Fixes**: Fixed [CVE-2025-0330](https://www.cve.org/CVERecord?id=CVE-2025-0330) and [CVE-2024-6825](https://www.cve.org/CVERecord?id=CVE-2024-6825) vulnerabilities + +Let's dive in. + +## Realtime API Cost Tracking [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#realtime-api-cost-tracking "Direct link to Realtime API Cost Tracking") + +![](https://docs.litellm.ai/assets/ideal-img/realtime_api.960b38e.1920.png) + +This release adds Realtime API logging + cost tracking. + +- **Logging**: LiteLLM now logs the complete response from realtime calls to all logging integrations (DB, S3, Langfuse, etc.) +- **Cost Tracking**: You can now set 'base\_model' and custom pricing for realtime models. [Custom Pricing](https://docs.litellm.ai/docs/proxy/custom_pricing) +- **Budgets**: Your key/user/team budgets now work for realtime models as well. + +Start [here](https://docs.litellm.ai/docs/realtime) + +## Microsoft SSO Auto-sync [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#microsoft-sso-auto-sync "Direct link to Microsoft SSO Auto-sync") + +![](https://docs.litellm.ai/assets/ideal-img/sso_sync.2f79062.1414.png) + +Auto-sync groups and members from Azure Entra ID to LiteLLM + +This release adds support for auto-syncing groups and members on Microsoft Entra ID with LiteLLM. This means that LiteLLM proxy administrators can spend less time managing teams and members and LiteLLM handles the following: + +- Auto-create teams that exist on Microsoft Entra ID +- Sync team members on Microsoft Entra ID with LiteLLM teams + +Get started with this [here](https://docs.litellm.ai/docs/tutorials/msft_sso) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **xAI** + +1. Added reasoning\_effort support for `xai/grok-3-mini-beta` [Get Started](https://docs.litellm.ai/docs/providers/xai#reasoning-usage) +2. Added cost tracking for `xai/grok-3` models [PR](https://github.com/BerriAI/litellm/pull/9920) +- **Hugging Face** + +1. Added inference providers support [Get Started](https://docs.litellm.ai/docs/providers/huggingface#serverless-inference-providers) +- **Azure** + +1. Added azure/gpt-4o-realtime-audio cost tracking [PR](https://github.com/BerriAI/litellm/pull/9893) +- **VertexAI** + +1. Added enterpriseWebSearch tool support [Get Started](https://docs.litellm.ai/docs/providers/vertex#grounding---web-search) +2. Moved to only passing keys accepted by the Vertex AI response schema [PR](https://github.com/BerriAI/litellm/pull/8992) +- **Google AI Studio** + +1. Added cost tracking for `gemini-2.5-pro` [PR](https://github.com/BerriAI/litellm/pull/9837) +2. Fixed pricing for 'gemini/gemini-2.5-pro-preview-03-25' [PR](https://github.com/BerriAI/litellm/pull/9896) +3. Fixed handling file\_data being passed in [PR](https://github.com/BerriAI/litellm/pull/9786) +- **Azure** + +1. Updated Azure Phi-4 pricing [PR](https://github.com/BerriAI/litellm/pull/9862) +2. Added azure/gpt-4o-realtime-audio cost tracking [PR](https://github.com/BerriAI/litellm/pull/9893) +- **Databricks** + +1. Removed reasoning\_effort from parameters [PR](https://github.com/BerriAI/litellm/pull/9811) +2. Fixed custom endpoint check for Databricks [PR](https://github.com/BerriAI/litellm/pull/9925) +- **General** + +1. Added litellm.supports\_reasoning() util to track if an llm supports reasoning [Get Started](https://docs.litellm.ai/docs/providers/anthropic#reasoning) +2. Function Calling - Handle pydantic base model in message tool calls, handle tools = \[\], and support fake streaming on tool calls for meta.llama3-3-70b-instruct-v1:0 [PR](https://github.com/BerriAI/litellm/pull/9774) +3. LiteLLM Proxy - Allow passing `thinking` param to litellm proxy via client sdk [PR](https://github.com/BerriAI/litellm/pull/9386) +4. Fixed correctly translating 'thinking' param for litellm [PR](https://github.com/BerriAI/litellm/pull/9904) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **OpenAI, Azure** +1. Realtime API Cost tracking with token usage metrics in spend logs [Get Started](https://docs.litellm.ai/docs/realtime) +- **Anthropic** +1. Fixed Claude Haiku cache read pricing per token [PR](https://github.com/BerriAI/litellm/pull/9834) +2. Added cost tracking for Claude responses with base\_model [PR](https://github.com/BerriAI/litellm/pull/9897) +3. Fixed Anthropic prompt caching cost calculation and trimmed logged message in db [PR](https://github.com/BerriAI/litellm/pull/9838) +- **General** +1. Added token tracking and log usage object in spend logs [PR](https://github.com/BerriAI/litellm/pull/9843) +2. Handle custom pricing at deployment level [PR](https://github.com/BerriAI/litellm/pull/9855) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +- **Test Key Tab** + +1. Added rendering of Reasoning content, ttft, usage metrics on test key page [PR](https://github.com/BerriAI/litellm/pull/9931) + + ![](https://docs.litellm.ai/assets/ideal-img/chat_metrics.c59fcfe.1920.png) + + View input, output, reasoning tokens, ttft metrics. +- **Tag / Policy Management** + +1. Added Tag/Policy Management. Create routing rules based on request metadata. This allows you to enforce that requests with `tags="private"` only go to specific models. [Get Started](https://docs.litellm.ai/docs/tutorials/tag_management) + + + + ![](https://docs.litellm.ai/assets/ideal-img/tag_management.5bf985c.1920.png) + + Create and manage tags. +- **Redesigned Login Screen** + +1. Polished login screen [PR](https://github.com/BerriAI/litellm/pull/9778) +- **Microsoft SSO Auto-Sync** + +1. Added debug route to allow admins to debug SSO JWT fields [PR](https://github.com/BerriAI/litellm/pull/9835) +2. Added ability to use MSFT Graph API to assign users to teams [PR](https://github.com/BerriAI/litellm/pull/9865) +3. Connected litellm to Azure Entra ID Enterprise Application [PR](https://github.com/BerriAI/litellm/pull/9872) +4. Added ability for admins to set `default_team_params` for when litellm SSO creates default teams [PR](https://github.com/BerriAI/litellm/pull/9895) +5. Fixed MSFT SSO to use correct field for user email [PR](https://github.com/BerriAI/litellm/pull/9886) +6. Added UI support for setting Default Team setting when litellm SSO auto creates teams [PR](https://github.com/BerriAI/litellm/pull/9918) +- **UI Bug Fixes** + +1. Prevented team, key, org, model numerical values changing on scrolling [PR](https://github.com/BerriAI/litellm/pull/9776) +2. Instantly reflect key and team updates in UI [PR](https://github.com/BerriAI/litellm/pull/9825) + +## Logging / Guardrail Improvements [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#logging--guardrail-improvements "Direct link to Logging / Guardrail Improvements") + +- **Prometheus** +1. Emit Key and Team Budget metrics on a cron job schedule [Get Started](https://docs.litellm.ai/docs/proxy/prometheus#initialize-budget-metrics-on-startup) + +## Security Fixes [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#security-fixes "Direct link to Security Fixes") + +- Fixed [CVE-2025-0330](https://www.cve.org/CVERecord?id=CVE-2025-0330) \- Leakage of Langfuse API keys in team exception handling [PR](https://github.com/BerriAI/litellm/pull/9830) +- Fixed [CVE-2024-6825](https://www.cve.org/CVERecord?id=CVE-2024-6825) \- Remote code execution in post call rules [PR](https://github.com/BerriAI/litellm/pull/9826) + +## Helm [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#helm "Direct link to Helm") + +- Added service annotations to litellm-helm chart [PR](https://github.com/BerriAI/litellm/pull/9840) +- Added extraEnvVars to the helm deployment [PR](https://github.com/BerriAI/litellm/pull/9292) + +## Demo [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#demo "Direct link to Demo") + +Try this on the demo instance [today](https://docs.litellm.ai/docs/proxy/demo) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/cost-tracking\#complete-git-diff "Direct link to Complete Git Diff") + +See the complete git diff since v1.65.4-stable, [here](https://github.com/BerriAI/litellm/releases/tag/v1.66.0-stable) + +## Credential Management Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/credential-management#__docusaurus_skipToContent_fallback) + +These are the changes since `v1.63.11-stable`. + +This release brings: + +- LLM Translation Improvements (MCP Support and Bedrock Application Profiles) +- Perf improvements for Usage-based Routing +- Streaming guardrail support via websockets +- Azure OpenAI client perf fix (from previous release) + +## Docker Run LiteLLM Proxy [​](https://docs.litellm.ai/release_notes/tags/credential-management\#docker-run-litellm-proxy "Direct link to Docker Run LiteLLM Proxy") + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.63.14-stable.patch1 + +``` + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/credential-management\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/credential-management\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Azure gpt-4o - fixed pricing to latest global pricing - [PR](https://github.com/BerriAI/litellm/pull/9361) +- O1-Pro - add pricing + model information - [PR](https://github.com/BerriAI/litellm/pull/9397) +- Azure AI - mistral 3.1 small pricing added - [PR](https://github.com/BerriAI/litellm/pull/9453) +- Azure - gpt-4.5-preview pricing added - [PR](https://github.com/BerriAI/litellm/pull/9453) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/credential-management\#llm-translation "Direct link to LLM Translation") + +1. **New LLM Features** + +- Bedrock: Support bedrock application inference profiles [Docs](https://docs.litellm.ai/docs/providers/bedrock#bedrock-application-inference-profile) + - Infer aws region from bedrock application profile id - ( `arn:aws:bedrock:us-east-1:...`) +- Ollama - support calling via `/v1/completions` [Get Started](https://docs.litellm.ai/docs/providers/ollama#using-ollama-fim-on-v1completions) +- Bedrock - support `us.deepseek.r1-v1:0` model name [Docs](https://docs.litellm.ai/docs/providers/bedrock#supported-aws-bedrock-models) +- OpenRouter - `OPENROUTER_API_BASE` env var support [Docs](https://docs.litellm.ai/docs/providers/openrouter.md) +- Azure - add audio model parameter support - [Docs](https://docs.litellm.ai/docs/providers/azure#azure-audio-model) +- OpenAI - PDF File support [Docs](https://docs.litellm.ai/docs/completion/document_understanding#openai-file-message-type) +- OpenAI - o1-pro Responses API streaming support [Docs](https://docs.litellm.ai/docs/response_api.md#streaming) +- \[BETA\] MCP - Use MCP Tools with LiteLLM SDK [Docs](https://docs.litellm.ai/docs/mcp) + +2. **Bug Fixes** + +- Voyage: prompt token on embedding tracking fix - [PR](https://github.com/BerriAI/litellm/commit/56d3e75b330c3c3862dc6e1c51c1210e48f1068e) +- Sagemaker - Fix ‘Too little data for declared Content-Length’ error - [PR](https://github.com/BerriAI/litellm/pull/9326) +- OpenAI-compatible models - fix issue when calling openai-compatible models w/ custom\_llm\_provider set - [PR](https://github.com/BerriAI/litellm/pull/9355) +- VertexAI - Embedding ‘outputDimensionality’ support - [PR](https://github.com/BerriAI/litellm/commit/437dbe724620675295f298164a076cbd8019d304) +- Anthropic - return consistent json response format on streaming/non-streaming - [PR](https://github.com/BerriAI/litellm/pull/9437) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/credential-management\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- `litellm_proxy/` \- support reading litellm response cost header from proxy, when using client sdk +- Reset Budget Job - fix budget reset error on keys/teams/users [PR](https://github.com/BerriAI/litellm/pull/9329) +- Streaming - Prevents final chunk w/ usage from being ignored (impacted bedrock streaming + cost tracking) [PR](https://github.com/BerriAI/litellm/pull/9314) + +## UI [​](https://docs.litellm.ai/release_notes/tags/credential-management\#ui "Direct link to UI") + +1. Users Page + - Feature: Control default internal user settings [PR](https://github.com/BerriAI/litellm/pull/9328) +2. Icons: + - Feature: Replace external "artificialanalysis.ai" icons by local svg [PR](https://github.com/BerriAI/litellm/pull/9374) +3. Sign In/Sign Out + - Fix: Default login when `default_user_id` user does not exist in DB [PR](https://github.com/BerriAI/litellm/pull/9395) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes/tags/credential-management\#logging-integrations "Direct link to Logging Integrations") + +- Support post-call guardrails for streaming responses [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/custom_guardrail#1-write-a-customguardrail-class) +- Arize [Get Started](https://docs.litellm.ai/docs/observability/arize_integration) + - fix invalid package import [PR](https://github.com/BerriAI/litellm/pull/9338) + - migrate to using standardloggingpayload for metadata, ensures spans land successfully [PR](https://github.com/BerriAI/litellm/pull/9338) + - fix logging to just log the LLM I/O [PR](https://github.com/BerriAI/litellm/pull/9353) + - Dynamic API Key/Space param support [Get Started](https://docs.litellm.ai/docs/observability/arize_integration#pass-arize-spacekey-per-request) +- StandardLoggingPayload - Log litellm\_model\_name in payload. Allows knowing what the model sent to API provider was [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec#standardlogginghiddenparams) +- Prompt Management - Allow building custom prompt management integration [Get Started](https://docs.litellm.ai/docs/proxy/custom_prompt_management.md) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/credential-management\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +- Redis Caching - add 5s default timeout, prevents hanging redis connection from impacting llm calls [PR](https://github.com/BerriAI/litellm/commit/db92956ae33ed4c4e3233d7e1b0c7229817159bf) +- Allow disabling all spend updates / writes to DB - patch to allow disabling all spend updates to DB with a flag [PR](https://github.com/BerriAI/litellm/pull/9331) +- Azure OpenAI - correctly re-use azure openai client, fixes perf issue from previous Stable release [PR](https://github.com/BerriAI/litellm/commit/f2026ef907c06d94440930917add71314b901413) +- Azure OpenAI - uses litellm.ssl\_verify on Azure/OpenAI clients [PR](https://github.com/BerriAI/litellm/commit/f2026ef907c06d94440930917add71314b901413) +- Usage-based routing - Wildcard model support [Get Started](https://docs.litellm.ai/docs/proxy/usage_based_routing#wildcard-model-support) +- Usage-based routing - Support batch writing increments to redis - reduces latency to same as ‘simple-shuffle’ [PR](https://github.com/BerriAI/litellm/pull/9357) +- Router - show reason for model cooldown on ‘no healthy deployments available error’ [PR](https://github.com/BerriAI/litellm/pull/9438) +- Caching - add max value limit to an item in in-memory cache (1MB) - prevents OOM errors on large image url’s being sent through proxy [PR](https://github.com/BerriAI/litellm/pull/9448) + +## General Improvements [​](https://docs.litellm.ai/release_notes/tags/credential-management\#general-improvements "Direct link to General Improvements") + +- Passthrough Endpoints - support returning api-base on pass-through endpoints Response Headers [Docs](https://docs.litellm.ai/docs/proxy/response_headers#litellm-specific-headers) +- SSL - support reading ssl security level from env var - Allows user to specify lower security settings [Get Started](https://docs.litellm.ai/docs/guides/security_settings) +- Credentials - only poll Credentials table when `STORE_MODEL_IN_DB` is True [PR](https://github.com/BerriAI/litellm/pull/9376) +- Image URL Handling - new architecture doc on image url handling [Docs](https://docs.litellm.ai/docs/proxy/image_handling) +- OpenAI - bump to pip install "openai==1.68.2" [PR](https://github.com/BerriAI/litellm/commit/e85e3bc52a9de86ad85c3dbb12d87664ee567a5a) +- Gunicorn - security fix - bump gunicorn==23.0.0 [PR](https://github.com/BerriAI/litellm/commit/7e9fc92f5c7fea1e7294171cd3859d55384166eb) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/credential-management\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.11-stable...v1.63.14.rc) + +These are the changes since `v1.63.2-stable`. + +This release is primarily focused on: + +- \[Beta\] Responses API Support +- Snowflake Cortex Support, Amazon Nova Image Generation +- UI - Credential Management, re-use credentials when adding new models +- UI - Test Connection to LLM Provider before adding a model + +## Known Issues [​](https://docs.litellm.ai/release_notes/tags/credential-management\#known-issues "Direct link to Known Issues") + +- 🚨 Known issue on Azure OpenAI - We don't recommend upgrading if you use Azure OpenAI. This version failed our Azure OpenAI load test + +## Docker Run LiteLLM Proxy [​](https://docs.litellm.ai/release_notes/tags/credential-management\#docker-run-litellm-proxy "Direct link to Docker Run LiteLLM Proxy") + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.63.11-stable + +``` + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/credential-management\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/credential-management\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Image Generation support for Amazon Nova Canvas [Getting Started](https://docs.litellm.ai/docs/providers/bedrock#image-generation) +- Add pricing for Jamba new models [PR](https://github.com/BerriAI/litellm/pull/9032/files) +- Add pricing for Amazon EU models [PR](https://github.com/BerriAI/litellm/pull/9056/files) +- Add Bedrock Deepseek R1 model pricing [PR](https://github.com/BerriAI/litellm/pull/9108/files) +- Update Gemini pricing: Gemma 3, Flash 2 thinking update, LearnLM [PR](https://github.com/BerriAI/litellm/pull/9190/files) +- Mark Cohere Embedding 3 models as Multimodal [PR](https://github.com/BerriAI/litellm/pull/9176/commits/c9a576ce4221fc6e50dc47cdf64ab62736c9da41) +- Add Azure Data Zone pricing [PR](https://github.com/BerriAI/litellm/pull/9185/files#diff-19ad91c53996e178c1921cbacadf6f3bae20cfe062bd03ee6bfffb72f847ee37) + - LiteLLM Tracks cost for `azure/eu` and `azure/us` models + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/credential-management\#llm-translation "Direct link to LLM Translation") + +1. **New Endpoints** + +- \[Beta\] POST `/responses` API. [Getting Started](https://docs.litellm.ai/docs/response_api) + +2. **New LLM Providers** + +- Snowflake Cortex [Getting Started](https://docs.litellm.ai/docs/providers/snowflake) + +3. **New LLM Features** + +- Support OpenRouter `reasoning_content` on streaming [Getting Started](https://docs.litellm.ai/docs/reasoning_content) + +4. **Bug Fixes** + +- OpenAI: Return `code`, `param` and `type` on bad request error [More information on litellm exceptions](https://docs.litellm.ai/docs/exception_mapping) +- Bedrock: Fix converse chunk parsing to only return empty dict on tool use [PR](https://github.com/BerriAI/litellm/pull/9166) +- Bedrock: Support extra\_headers [PR](https://github.com/BerriAI/litellm/pull/9113) +- Azure: Fix Function Calling Bug & Update Default API Version to `2025-02-01-preview` [PR](https://github.com/BerriAI/litellm/pull/9191) +- Azure: Fix AI services URL [PR](https://github.com/BerriAI/litellm/pull/9185) +- Vertex AI: Handle HTTP 201 status code in response [PR](https://github.com/BerriAI/litellm/pull/9193) +- Perplexity: Fix incorrect streaming response [PR](https://github.com/BerriAI/litellm/pull/9081) +- Triton: Fix streaming completions bug [PR](https://github.com/BerriAI/litellm/pull/8386) +- Deepgram: Support bytes.IO when handling audio files for transcription [PR](https://github.com/BerriAI/litellm/pull/9071) +- Ollama: Fix "system" role has become unacceptable [PR](https://github.com/BerriAI/litellm/pull/9261) +- All Providers (Streaming): Fix String `data:` stripped from entire content in streamed responses [PR](https://github.com/BerriAI/litellm/pull/9070) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/credential-management\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Support Bedrock converse cache token tracking [Getting Started](https://docs.litellm.ai/docs/completion/prompt_caching) +2. Cost Tracking for Responses API [Getting Started](https://docs.litellm.ai/docs/response_api) +3. Fix Azure Whisper cost tracking [Getting Started](https://docs.litellm.ai/docs/audio_transcription) + +## UI [​](https://docs.litellm.ai/release_notes/tags/credential-management\#ui "Direct link to UI") + +### Re-Use Credentials on UI [​](https://docs.litellm.ai/release_notes/tags/credential-management\#re-use-credentials-on-ui "Direct link to Re-Use Credentials on UI") + +You can now onboard LLM provider credentials on LiteLLM UI. Once these credentials are added you can re-use them when adding new models [Getting Started](https://docs.litellm.ai/docs/proxy/ui_credentials) + +![](https://docs.litellm.ai/assets/ideal-img/credentials.8f19ffb.1920.jpg) + +### Test Connections before adding models [​](https://docs.litellm.ai/release_notes/tags/credential-management\#test-connections-before-adding-models "Direct link to Test Connections before adding models") + +Before adding a model you can test the connection to the LLM provider to verify you have setup your API Base + API Key correctly + +![](https://docs.litellm.ai/assets/images/litellm_test_connection-029765a2de4dcabccfe3be9a8d33dbdd.gif) + +### General UI Improvements [​](https://docs.litellm.ai/release_notes/tags/credential-management\#general-ui-improvements "Direct link to General UI Improvements") + +1. Add Models Page + - Allow adding Cerebras, Sambanova, Perplexity, Fireworks, Openrouter, TogetherAI Models, Text-Completion OpenAI on Admin UI + - Allow adding EU OpenAI models + - Fix: Instantly show edit + deletes to models +2. Keys Page + - Fix: Instantly show newly created keys on Admin UI (don't require refresh) + - Fix: Allow clicking into Top Keys when showing users Top API Key + - Fix: Allow Filter Keys by Team Alias, Key Alias and Org + - UI Improvements: Show 100 Keys Per Page, Use full height, increase width of key alias +3. Users Page + - Fix: Show correct count of internal user keys on Users Page + - Fix: Metadata not updating in Team UI +4. Logs Page + - UI Improvements: Keep expanded log in focus on LiteLLM UI + - UI Improvements: Minor improvements to logs page + - Fix: Allow internal user to query their own logs + - Allow switching off storing Error Logs in DB [Getting Started](https://docs.litellm.ai/docs/proxy/ui_logs) +5. Sign In/Sign Out + - Fix: Correctly use `PROXY_LOGOUT_URL` when set [Getting Started](https://docs.litellm.ai/docs/proxy/self_serve#setting-custom-logout-urls) + +## Security [​](https://docs.litellm.ai/release_notes/tags/credential-management\#security "Direct link to Security") + +1. Support for Rotating Master Keys [Getting Started](https://docs.litellm.ai/docs/proxy/master_key_rotations) +2. Fix: Internal User Viewer Permissions, don't allow `internal_user_viewer` role to see `Test Key Page` or `Create Key Button` [More information on role based access controls](https://docs.litellm.ai/docs/proxy/access_control) +3. Emit audit logs on All user + model Create/Update/Delete endpoints [Getting Started](https://docs.litellm.ai/docs/proxy/multiple_admins) +4. JWT + - Support multiple JWT OIDC providers [Getting Started](https://docs.litellm.ai/docs/proxy/token_auth) + - Fix JWT access with Groups not working when team is assigned All Proxy Models access +5. Using K/V pairs in 1 AWS Secret [Getting Started](https://docs.litellm.ai/docs/secret#using-kv-pairs-in-1-aws-secret) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes/tags/credential-management\#logging-integrations "Direct link to Logging Integrations") + +1. Prometheus: Track Azure LLM API latency metric [Getting Started](https://docs.litellm.ai/docs/proxy/prometheus#request-latency-metrics) +2. Athina: Added tags, user\_feedback and model\_options to additional\_keys which can be sent to Athina [Getting Started](https://docs.litellm.ai/docs/observability/athina_integration) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/credential-management\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +1. Redis + litellm router - Fix Redis cluster mode for litellm router [PR](https://github.com/BerriAI/litellm/pull/9010) + +## General Improvements [​](https://docs.litellm.ai/release_notes/tags/credential-management\#general-improvements "Direct link to General Improvements") + +1. OpenWebUI Integration - display `thinking` tokens + +- Guide on getting started with LiteLLM x OpenWebUI. [Getting Started](https://docs.litellm.ai/docs/tutorials/openweb_ui) +- Display `thinking` tokens on OpenWebUI (Bedrock, Anthropic, Deepseek) [Getting Started](https://docs.litellm.ai/docs/tutorials/openweb_ui#render-thinking-content-on-openweb-ui) + +![](https://docs.litellm.ai/assets/images/litellm_thinking_openweb-5ec7dddb7e7b6a10252694c27cfc177d.gif) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/credential-management\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.2-stable...v1.63.11-stable) + +## Custom Auth Features +[Skip to main content](https://docs.litellm.ai/release_notes/tags/custom-auth#__docusaurus_skipToContent_fallback) + +`batches`, `guardrails`, `team management`, `custom auth` + +![](https://docs.litellm.ai/assets/ideal-img/batches_cost_tracking.8fc9663.1208.png) + +info + +Get a free 7-day LiteLLM Enterprise trial here. [Start here](https://www.litellm.ai/#trial) + +**No call needed** + +## ✨ Cost Tracking, Logging for Batches API ( `/batches`) [​](https://docs.litellm.ai/release_notes/tags/custom-auth\#-cost-tracking-logging-for-batches-api-batches "Direct link to -cost-tracking-logging-for-batches-api-batches") + +Track cost, usage for Batch Creation Jobs. [Start here](https://docs.litellm.ai/docs/batches) + +## ✨ `/guardrails/list` endpoint [​](https://docs.litellm.ai/release_notes/tags/custom-auth\#-guardrailslist-endpoint "Direct link to -guardrailslist-endpoint") + +Show available guardrails to users. [Start here](https://litellm-api.up.railway.app/#/Guardrails) + +## ✨ Allow teams to add models [​](https://docs.litellm.ai/release_notes/tags/custom-auth\#-allow-teams-to-add-models "Direct link to ✨ Allow teams to add models") + +This enables team admins to call their own finetuned models via litellm proxy. [Start here](https://docs.litellm.ai/docs/proxy/team_model_add) + +## ✨ Common checks for custom auth [​](https://docs.litellm.ai/release_notes/tags/custom-auth\#-common-checks-for-custom-auth "Direct link to ✨ Common checks for custom auth") + +Calling the internal common\_checks function in custom auth is now enforced as an enterprise feature. This allows admins to use litellm's default budget/auth checks within their custom auth implementation. [Start here](https://docs.litellm.ai/docs/proxy/virtual_keys#custom-auth) + +## ✨ Assigning team admins [​](https://docs.litellm.ai/release_notes/tags/custom-auth\#-assigning-team-admins "Direct link to ✨ Assigning team admins") + +Team admins is graduating from beta and moving to our enterprise tier. This allows proxy admins to allow others to manage keys/models for their own teams (useful for projects in production). [Start here](https://docs.litellm.ai/docs/proxy/virtual_keys#restricting-key-generation) + +## LiteLLM v1.65.0 Release +[Skip to main content](https://docs.litellm.ai/release_notes/tags/custom-prompt-management#__docusaurus_skipToContent_fallback) + +v1.65.0-stable is live now. Here are the key highlights of this release: + +- **MCP Support**: Support for adding and using MCP servers on the LiteLLM proxy. +- **UI view total usage after 1M+ logs**: You can now view usage analytics after crossing 1M+ logs in DB. + +## Model Context Protocol (MCP) [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#model-context-protocol-mcp "Direct link to Model Context Protocol (MCP)") + +This release introduces support for centrally adding MCP servers on LiteLLM. This allows you to add MCP server endpoints and your developers can `list` and `call` MCP tools through LiteLLM. + +Read more about MCP [here](https://docs.litellm.ai/docs/mcp). + +![](https://docs.litellm.ai/assets/ideal-img/mcp_ui.4a5216a.1920.png) + +Expose and use MCP servers through LiteLLM + +## UI view total usage after 1M+ logs [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#ui-view-total-usage-after-1m-logs "Direct link to UI view total usage after 1M+ logs") + +This release brings the ability to view total usage analytics even after exceeding 1M+ logs in your database. We've implemented a scalable architecture that stores only aggregate usage data, resulting in significantly more efficient queries and reduced database CPU utilization. + +![](https://docs.litellm.ai/assets/ideal-img/ui_usage.3ffdba3.1200.png) + +View total usage after 1M+ logs + +- How this works: + + - We now aggregate usage data into a dedicated DailyUserSpend table, significantly reducing query load and CPU usage even beyond 1M+ logs. +- Daily Spend Breakdown API: + + - Retrieve granular daily usage data (by model, provider, and API key) with a single endpoint. + Example Request: + + + + Daily Spend Breakdown API + + + + + + ```codeBlockLines_e6Vv codeBlockLinesWithNumbering_o6Pm + curl -L -X GET 'http://localhost:4000/user/daily/activity?start_date=2025-03-20&end_date=2025-03-27' \ + -H 'Authorization: Bearer sk-...' + + ``` + + + + + + + + + + + + Daily Spend Breakdown API Response + + + + + + ```codeBlockLines_e6Vv codeBlockLinesWithNumbering_o6Pm + { + "results": [\ + {\ + "date": "2025-03-27",\ + "metrics": {\ + "spend": 0.0177072,\ + "prompt_tokens": 111,\ + "completion_tokens": 1711,\ + "total_tokens": 1822,\ + "api_requests": 11\ + },\ + "breakdown": {\ + "models": {\ + "gpt-4o-mini": {\ + "spend": 1.095e-05,\ + "prompt_tokens": 37,\ + "completion_tokens": 9,\ + "total_tokens": 46,\ + "api_requests": 1\ + },\ + "providers": { "openai": { ... }, "azure_ai": { ... } },\ + "api_keys": { "3126b6eaf1...": { ... } }\ + }\ + }\ + ], + "metadata": { + "total_spend": 0.7274667, + "total_prompt_tokens": 280990, + "total_completion_tokens": 376674, + "total_api_requests": 14 + } + } + + ``` + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Support for Vertex AI gemini-2.0-flash-lite & Google AI Studio gemini-2.0-flash-lite [PR](https://github.com/BerriAI/litellm/pull/9523) +- Support for Vertex AI Fine-Tuned LLMs [PR](https://github.com/BerriAI/litellm/pull/9542) +- Nova Canvas image generation support [PR](https://github.com/BerriAI/litellm/pull/9525) +- OpenAI gpt-4o-transcribe support [PR](https://github.com/BerriAI/litellm/pull/9517) +- Added new Vertex AI text embedding model [PR](https://github.com/BerriAI/litellm/pull/9476) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#llm-translation "Direct link to LLM Translation") + +- OpenAI Web Search Tool Call Support [PR](https://github.com/BerriAI/litellm/pull/9465) +- Vertex AI topLogprobs support [PR](https://github.com/BerriAI/litellm/pull/9518) +- Support for sending images and video to Vertex AI multimodal embedding [Doc](https://docs.litellm.ai/docs/providers/vertex#multi-modal-embeddings) +- Support litellm.api\_base for Vertex AI + Gemini across completion, embedding, image\_generation [PR](https://github.com/BerriAI/litellm/pull/9516) +- Bug fix for returning `response_cost` when using litellm python SDK with LiteLLM Proxy [PR](https://github.com/BerriAI/litellm/commit/6fd18651d129d606182ff4b980e95768fc43ca3d) +- Support for `max_completion_tokens` on Mistral API [PR](https://github.com/BerriAI/litellm/pull/9606) +- Refactored Vertex AI passthrough routes - fixes unpredictable behaviour with auto-setting default\_vertex\_region on router model add [PR](https://github.com/BerriAI/litellm/pull/9467) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- Log 'api\_base' on spend logs [PR](https://github.com/BerriAI/litellm/pull/9509) +- Support for Gemini audio token cost tracking [PR](https://github.com/BerriAI/litellm/pull/9535) +- Fixed OpenAI audio input token cost tracking [PR](https://github.com/BerriAI/litellm/pull/9535) + +## UI [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#ui "Direct link to UI") + +### Model Management [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#model-management "Direct link to Model Management") + +- Allowed team admins to add/update/delete models on UI [PR](https://github.com/BerriAI/litellm/pull/9572) +- Added render supports\_web\_search on model hub [PR](https://github.com/BerriAI/litellm/pull/9469) + +### Request Logs [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#request-logs "Direct link to Request Logs") + +- Show API base and model ID on request logs [PR](https://github.com/BerriAI/litellm/pull/9572) +- Allow viewing keyinfo on request logs [PR](https://github.com/BerriAI/litellm/pull/9568) + +### Usage Tab [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#usage-tab "Direct link to Usage Tab") + +- Added Daily User Spend Aggregate view - allows UI Usage tab to work > 1m rows [PR](https://github.com/BerriAI/litellm/pull/9538) +- Connected UI to "LiteLLM\_DailyUserSpend" spend table [PR](https://github.com/BerriAI/litellm/pull/9603) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#logging-integrations "Direct link to Logging Integrations") + +- Fixed StandardLoggingPayload for GCS Pub Sub Logging Integration [PR](https://github.com/BerriAI/litellm/pull/9508) +- Track `litellm_model_name` on `StandardLoggingPayload` [Docs](https://docs.litellm.ai/docs/proxy/logging_spec#standardlogginghiddenparams) + +## Performance / Reliability Improvements [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#performance--reliability-improvements "Direct link to Performance / Reliability Improvements") + +- LiteLLM Redis semantic caching implementation [PR](https://github.com/BerriAI/litellm/pull/9356) +- Gracefully handle exceptions when DB is having an outage [PR](https://github.com/BerriAI/litellm/pull/9533) +- Allow Pods to startup + passing /health/readiness when allow\_requests\_on\_db\_unavailable: True and DB is down [PR](https://github.com/BerriAI/litellm/pull/9569) + +## General Improvements [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#general-improvements "Direct link to General Improvements") + +- Support for exposing MCP tools on litellm proxy [PR](https://github.com/BerriAI/litellm/pull/9426) +- Support discovering Gemini, Anthropic, xAI models by calling their /v1/model endpoint [PR](https://github.com/BerriAI/litellm/pull/9530) +- Fixed route check for non-proxy admins on JWT auth [PR](https://github.com/BerriAI/litellm/pull/9454) +- Added baseline Prisma database migrations [PR](https://github.com/BerriAI/litellm/pull/9565) +- View all wildcard models on /model/info [PR](https://github.com/BerriAI/litellm/pull/9572) + +## Security [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#security "Direct link to Security") + +- Bumped next from 14.2.21 to 14.2.25 in UI dashboard [PR](https://github.com/BerriAI/litellm/pull/9458) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/custom-prompt-management\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.14-stable.patch1...v1.65.0-stable) + +## LiteLLM Release Notes +[Skip to main content](https://docs.litellm.ai/release_notes/tags/db-schema#__docusaurus_skipToContent_fallback) + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/db-schema\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. New OpenAI `/image/variations` endpoint BETA support [Docs](https://docs.litellm.ai/docs/image_variations) +2. Topaz API support on OpenAI `/image/variations` BETA endpoint [Docs](https://docs.litellm.ai/docs/providers/topaz) +3. Deepseek - r1 support w/ reasoning\_content ( [Deepseek API](https://docs.litellm.ai/docs/providers/deepseek#reasoning-models), [Vertex AI](https://docs.litellm.ai/docs/providers/vertex#model-garden), [Bedrock](https://docs.litellm.ai/docs/providers/bedrock#deepseek)) +4. Azure - Add azure o1 pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L952) +5. Anthropic - handle `-latest` tag in model for cost calculation +6. Gemini-2.0-flash-thinking - add model pricing (it’s 0.0) [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L3393) +7. Bedrock - add stability sd3 model pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6814) (s/o [Marty Sullivan](https://github.com/marty-sullivan)) +8. Bedrock - add us.amazon.nova-lite-v1:0 to model cost map [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L5619) +9. TogetherAI - add new together\_ai llama3.3 models [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6985) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/db-schema\#llm-translation "Direct link to LLM Translation") + +01. LM Studio -> fix async embedding call +02. Gpt 4o models - fix response\_format translation +03. Bedrock nova - expand supported document types to include .md, .csv, etc. [Start Here](https://docs.litellm.ai/docs/providers/bedrock#usage---pdf--document-understanding) +04. Bedrock - docs on IAM role based access for bedrock - [Start Here](https://docs.litellm.ai/docs/providers/bedrock#sts-role-based-auth) +05. Bedrock - cache IAM role credentials when used +06. Google AI Studio ( `gemini/`) \- support gemini 'frequency\_penalty' and 'presence\_penalty' +07. Azure O1 - fix model name check +08. WatsonX - ZenAPIKey support for WatsonX [Docs](https://docs.litellm.ai/docs/providers/watsonx) +09. Ollama Chat - support json schema response format [Start Here](https://docs.litellm.ai/docs/providers/ollama#json-schema-support) +10. Bedrock - return correct bedrock status code and error message if error during streaming +11. Anthropic - Supported nested json schema on anthropic calls +12. OpenAI - `metadata` param preview support + 1. SDK - enable via `litellm.enable_preview_features = True` + 2. PROXY - enable via `litellm_settings::enable_preview_features: true` +13. Replicate - retry completion response on status=processing + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/db-schema\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Bedrock - QA asserts all bedrock regional models have same `supported_` as base model +2. Bedrock - fix bedrock converse cost tracking w/ region name specified +3. Spend Logs reliability fix - when `user` passed in request body is int instead of string +4. Ensure ‘base\_model’ cost tracking works across all endpoints +5. Fixes for Image generation cost tracking +6. Anthropic - fix anthropic end user cost tracking +7. JWT / OIDC Auth - add end user id tracking from jwt auth + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/db-schema\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +01. allows team member to become admin post-add (ui + endpoints) +02. New edit/delete button for updating team membership on UI +03. If team admin - show all team keys +04. Model Hub - clarify cost of models is per 1m tokens +05. Invitation Links - fix invalid url generated +06. New - SpendLogs Table Viewer - allows proxy admin to view spend logs on UI + 1. New spend logs - allow proxy admin to ‘opt in’ to logging request/response in spend logs table - enables easier abuse detection + 2. Show country of origin in spend logs + 3. Add pagination + filtering by key name/team name +07. `/key/delete` \- allow team admin to delete team keys +08. Internal User ‘view’ - fix spend calculation when team selected +09. Model Analytics is now on Free +10. Usage page - shows days when spend = 0, and round spend on charts to 2 sig figs +11. Public Teams - allow admins to expose teams for new users to ‘join’ on UI - [Start Here](https://docs.litellm.ai/docs/proxy/public_teams) +12. Guardrails + 1. set/edit guardrails on a virtual key + 2. Allow setting guardrails on a team + 3. Set guardrails on team create + edit page +13. Support temporary budget increases on `/key/update` \- new `temp_budget_increase` and `temp_budget_expiry` fields - [Start Here](https://docs.litellm.ai/docs/proxy/virtual_keys#temporary-budget-increase) +14. Support writing new key alias to AWS Secret Manager - on key rotation [Start Here](https://docs.litellm.ai/docs/secret#aws-secret-manager) + +## Helm [​](https://docs.litellm.ai/release_notes/tags/db-schema\#helm "Direct link to Helm") + +1. add securityContext and pull policy values to migration job (s/o [https://github.com/Hexoplon](https://github.com/Hexoplon)) +2. allow specifying envVars on values.yaml +3. new helm lint test + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/db-schema\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Log the used prompt when prompt management used. [Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) +2. Support s3 logging with team alias prefixes - [Start Here](https://docs.litellm.ai/docs/proxy/logging#team-alias-prefix-in-object-key) +3. Prometheus [Start Here](https://docs.litellm.ai/docs/proxy/prometheus) +1. fix litellm\_llm\_api\_time\_to\_first\_token\_metric not populating for bedrock models +2. emit remaining team budget metric on regular basis (even when call isn’t made) - allows for more stable metrics on Grafana/etc. +3. add key and team level budget metrics +4. emit `litellm_overhead_latency_metric` +5. Emit `litellm_team_budget_reset_at_metric` and `litellm_api_key_budget_remaining_hours_metric` +4. Datadog - support logging spend tags to Datadog. [Start Here](https://docs.litellm.ai/docs/proxy/enterprise#tracking-spend-for-custom-tags) +5. Langfuse - fix logging request tags, read from standard logging payload +6. GCS - don’t truncate payload on logging +7. New GCS Pub/Sub logging support [Start Here](https://docs.litellm.ai/docs/proxy/logging#google-cloud-storage---pubsub-topic) +8. Add AIM Guardrails support [Start Here](https://docs.litellm.ai/docs/proxy/guardrails/aim_security) + +## Security [​](https://docs.litellm.ai/release_notes/tags/db-schema\#security "Direct link to Security") + +1. New Enterprise SLA for patching security vulnerabilities. [See Here](https://docs.litellm.ai/docs/enterprise#slas--professional-support) +2. Hashicorp - support using vault namespace for TLS auth. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) +3. Azure - DefaultAzureCredential support + +## Health Checks [​](https://docs.litellm.ai/release_notes/tags/db-schema\#health-checks "Direct link to Health Checks") + +1. Cleanup pricing-only model names from wildcard route list - prevent bad health checks +2. Allow specifying a health check model for wildcard routes - [https://docs.litellm.ai/docs/proxy/health#wildcard-routes](https://docs.litellm.ai/docs/proxy/health#wildcard-routes) +3. New ‘health\_check\_timeout ‘ param with default 1min upperbound to prevent bad model from health check to hang and cause pod restarts. [Start Here](https://docs.litellm.ai/docs/proxy/health#health-check-timeout) +4. Datadog - add data dog service health check + expose new `/health/services` endpoint. [Start Here](https://docs.litellm.ai/docs/proxy/health#healthservices) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/db-schema\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +01. 3x increase in RPS - moving to orjson for reading request body +02. LLM Routing speedup - using cached get model group info +03. SDK speedup - using cached get model info helper - reduces CPU work to get model info +04. Proxy speedup - only read request body 1 time per request +05. Infinite loop detection scripts added to codebase +06. Bedrock - pure async image transformation requests +07. Cooldowns - single deployment model group if 100% calls fail in high traffic - prevents an o1 outage from impacting other calls +08. Response Headers - return + 1. `x-litellm-timeout` + 2. `x-litellm-attempted-retries` + 3. `x-litellm-overhead-duration-ms` + 4. `x-litellm-response-duration-ms` +09. ensure duplicate callbacks are not added to proxy +10. Requirements.txt - bump certifi version + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/db-schema\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. JWT / OIDC Auth - new `enforce_rbac` param,allows proxy admin to prevent any unmapped yet authenticated jwt tokens from calling proxy. [Start Here](https://docs.litellm.ai/docs/proxy/token_auth#enforce-role-based-access-control-rbac) +2. fix custom openapi schema generation for customized swagger’s +3. Request Headers - support reading `x-litellm-timeout` param from request headers. Enables model timeout control when using Vercel’s AI SDK + LiteLLM Proxy. [Start Here](https://docs.litellm.ai/docs/proxy/request_headers#litellm-headers) +4. JWT / OIDC Auth - new `role` based permissions for model authentication. [See Here](https://docs.litellm.ai/docs/proxy/jwt_auth_arch) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/db-schema\#complete-git-diff "Direct link to Complete Git Diff") + +This is the diff between v1.57.8-stable and v1.59.8-stable. + +Use this to see the changes in the codebase. + +[**Git Diff**](https://github.com/BerriAI/litellm/compare/v1.57.8-stable...v1.59.8-stable) + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## UI Improvements [​](https://docs.litellm.ai/release_notes/tags/db-schema\#ui-improvements "Direct link to UI Improvements") + +### \[Opt In\] Admin UI - view messages / responses [​](https://docs.litellm.ai/release_notes/tags/db-schema\#opt-in-admin-ui---view-messages--responses "Direct link to opt-in-admin-ui---view-messages--responses") + +You can now view messages and response logs on Admin UI. + +![](https://docs.litellm.ai/assets/ideal-img/ui_logs.17b0459.1497.png) + +How to enable it - add `store_prompts_in_spend_logs: true` to your `proxy_config.yaml` + +Once this flag is enabled, your `messages` and `responses` will be stored in the `LiteLLM_Spend_Logs` table. + +```codeBlockLines_e6Vv +general_settings: + store_prompts_in_spend_logs: true + +``` + +## DB Schema Change [​](https://docs.litellm.ai/release_notes/tags/db-schema\#db-schema-change "Direct link to DB Schema Change") + +Added `messages` and `responses` to the `LiteLLM_Spend_Logs` table. + +**By default this is not logged.** If you want `messages` and `responses` to be logged, you need to opt in with this setting + +```codeBlockLines_e6Vv +general_settings: + store_prompts_in_spend_logs: true + +``` + +## Deepgram Release Notes +[Skip to main content](https://docs.litellm.ai/release_notes/tags/deepgram#__docusaurus_skipToContent_fallback) + +`deepgram`, `fireworks ai`, `vision`, `admin ui`, `dependency upgrades` + +## New Models [​](https://docs.litellm.ai/release_notes/tags/deepgram\#new-models "Direct link to New Models") + +### **Deepgram Speech to Text** [​](https://docs.litellm.ai/release_notes/tags/deepgram\#deepgram-speech-to-text "Direct link to deepgram-speech-to-text") + +New Speech to Text support for Deepgram models. [**Start Here**](https://docs.litellm.ai/docs/providers/deepgram) + +```codeBlockLines_e6Vv +from litellm import transcription +import os + +# set api keys +os.environ["DEEPGRAM_API_KEY"] = "" +audio_file = open("/path/to/audio.mp3", "rb") + +response = transcription(model="deepgram/nova-2", file=audio_file) + +print(f"response: {response}") + +``` + +### **Fireworks AI - Vision** support for all models [​](https://docs.litellm.ai/release_notes/tags/deepgram\#fireworks-ai---vision-support-for-all-models "Direct link to fireworks-ai---vision-support-for-all-models") + +LiteLLM supports document inlining for Fireworks AI models. This is useful for models that are not vision models, but still need to parse documents/images/etc. +LiteLLM will add `#transform=inline` to the url of the image\_url, if the model is not a vision model [See Code](https://github.com/BerriAI/litellm/blob/1ae9d45798bdaf8450f2dfdec703369f3d2212b7/litellm/llms/fireworks_ai/chat/transformation.py#L114) + +## Proxy Admin UI [​](https://docs.litellm.ai/release_notes/tags/deepgram\#proxy-admin-ui "Direct link to Proxy Admin UI") + +- `Test Key` Tab displays `model` used in response + +![](https://docs.litellm.ai/assets/ideal-img/ui_model.72a8982.1920.png) + +- `Test Key` Tab renders content in `.md`, `.py` (any code/markdown format) + +![](https://docs.litellm.ai/assets/ideal-img/ui_format.337282b.1920.png) + +## Dependency Upgrades [​](https://docs.litellm.ai/release_notes/tags/deepgram\#dependency-upgrades "Direct link to Dependency Upgrades") + +- (Security fix) Upgrade to `fastapi==0.115.5` [https://github.com/BerriAI/litellm/pull/7447](https://github.com/BerriAI/litellm/pull/7447) + +## Bug Fixes [​](https://docs.litellm.ai/release_notes/tags/deepgram\#bug-fixes "Direct link to Bug Fixes") + +- Add health check support for realtime models [Here](https://docs.litellm.ai/docs/proxy/health#realtime-models) +- Health check error with audio\_transcription model [https://github.com/BerriAI/litellm/issues/5999](https://github.com/BerriAI/litellm/issues/5999) + +## Dependency Upgrades +[Skip to main content](https://docs.litellm.ai/release_notes/tags/dependency-upgrades#__docusaurus_skipToContent_fallback) + +`deepgram`, `fireworks ai`, `vision`, `admin ui`, `dependency upgrades` + +## New Models [​](https://docs.litellm.ai/release_notes/tags/dependency-upgrades\#new-models "Direct link to New Models") + +### **Deepgram Speech to Text** [​](https://docs.litellm.ai/release_notes/tags/dependency-upgrades\#deepgram-speech-to-text "Direct link to deepgram-speech-to-text") + +New Speech to Text support for Deepgram models. [**Start Here**](https://docs.litellm.ai/docs/providers/deepgram) + +```codeBlockLines_e6Vv +from litellm import transcription +import os + +# set api keys +os.environ["DEEPGRAM_API_KEY"] = "" +audio_file = open("/path/to/audio.mp3", "rb") + +response = transcription(model="deepgram/nova-2", file=audio_file) + +print(f"response: {response}") + +``` + +### **Fireworks AI - Vision** support for all models [​](https://docs.litellm.ai/release_notes/tags/dependency-upgrades\#fireworks-ai---vision-support-for-all-models "Direct link to fireworks-ai---vision-support-for-all-models") + +LiteLLM supports document inlining for Fireworks AI models. This is useful for models that are not vision models, but still need to parse documents/images/etc. +LiteLLM will add `#transform=inline` to the url of the image\_url, if the model is not a vision model [See Code](https://github.com/BerriAI/litellm/blob/1ae9d45798bdaf8450f2dfdec703369f3d2212b7/litellm/llms/fireworks_ai/chat/transformation.py#L114) + +## Proxy Admin UI [​](https://docs.litellm.ai/release_notes/tags/dependency-upgrades\#proxy-admin-ui "Direct link to Proxy Admin UI") + +- `Test Key` Tab displays `model` used in response + +- `Test Key` Tab renders content in `.md`, `.py` (any code/markdown format) + +## Dependency Upgrades [​](https://docs.litellm.ai/release_notes/tags/dependency-upgrades\#dependency-upgrades "Direct link to Dependency Upgrades") + +- (Security fix) Upgrade to `fastapi==0.115.5` [https://github.com/BerriAI/litellm/pull/7447](https://github.com/BerriAI/litellm/pull/7447) + +## Bug Fixes [​](https://docs.litellm.ai/release_notes/tags/dependency-upgrades\#bug-fixes "Direct link to Bug Fixes") + +- Add health check support for realtime models [Here](https://docs.litellm.ai/docs/proxy/health#realtime-models) +- Health check error with audio\_transcription model [https://github.com/BerriAI/litellm/issues/5999](https://github.com/BerriAI/litellm/issues/5999) + +## Docker Image Release Notes +[Skip to main content](https://docs.litellm.ai/release_notes/tags/docker-image#__docusaurus_skipToContent_fallback) + +`docker image`, `security`, `vulnerability` + +# 0 Critical/High Vulnerabilities + +![](https://docs.litellm.ai/assets/ideal-img/security.8eb0218.1200.png) + +## What changed? [​](https://docs.litellm.ai/release_notes/tags/docker-image\#what-changed "Direct link to What changed?") + +- LiteLLMBase image now uses `cgr.dev/chainguard/python:latest-dev` + +## Why the change? [​](https://docs.litellm.ai/release_notes/tags/docker-image\#why-the-change "Direct link to Why the change?") + +To ensure there are 0 critical/high vulnerabilities on LiteLLM Docker Image + +## Migration Guide [​](https://docs.litellm.ai/release_notes/tags/docker-image\#migration-guide "Direct link to Migration Guide") + +- If you use a custom dockerfile with litellm as a base image + `apt-get` + +Instead of `apt-get` use `apk`, the base litellm image will no longer have `apt-get` installed. + +**You are only impacted if you use `apt-get` in your Dockerfile** + +```codeBlockLines_e6Vv +# Use the provided base image +FROM ghcr.io/berriai/litellm:main-latest + +# Set the working directory +WORKDIR /app + +# Install dependencies - CHANGE THIS to `apk` +RUN apt-get update && apt-get install -y dumb-init + +``` + +Before Change + +```codeBlockLines_e6Vv +RUN apt-get update && apt-get install -y dumb-init + +``` + +After Change + +```codeBlockLines_e6Vv +RUN apk update && apk add --no-cache dumb-init + +``` + +## LiteLLM Release Notes +[Skip to main content](https://docs.litellm.ai/release_notes/tags/fallbacks#__docusaurus_skipToContent_fallback) + +A new LiteLLM Stable release [just went out](https://github.com/BerriAI/litellm/releases/tag/v1.55.8-stable). Here are 5 updates since v1.52.2-stable. + +`langfuse`, `fallbacks`, `new models`, `azure_storage` + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes/tags/fallbacks\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +This makes it easy to run experiments or change the specific models `gpt-4o` to `gpt-4o-mini` on Langfuse, instead of making changes in your applications. [Start here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Control fallback prompts client-side [​](https://docs.litellm.ai/release_notes/tags/fallbacks\#control-fallback-prompts-client-side "Direct link to Control fallback prompts client-side") + +> Claude prompts are different than OpenAI + +Pass in prompts specific to model when doing fallbacks. [Start here](https://docs.litellm.ai/docs/proxy/reliability#control-fallback-prompts) + +## New Providers / Models [​](https://docs.litellm.ai/release_notes/tags/fallbacks\#new-providers--models "Direct link to New Providers / Models") + +- [NVIDIA Triton](https://developer.nvidia.com/triton-inference-server) `/infer` endpoint. [Start here](https://docs.litellm.ai/docs/providers/triton-inference-server) +- [Infinity](https://github.com/michaelfeil/infinity) Rerank Models [Start here](https://docs.litellm.ai/docs/providers/infinity) + +## ✨ Azure Data Lake Storage Support [​](https://docs.litellm.ai/release_notes/tags/fallbacks\#-azure-data-lake-storage-support "Direct link to ✨ Azure Data Lake Storage Support") + +Send LLM usage (spend, tokens) data to [Azure Data Lake](https://learn.microsoft.com/en-us/azure/storage/blobs/data-lake-storage-introduction). This makes it easy to consume usage data on other services (eg. Databricks) +[Start here](https://docs.litellm.ai/docs/proxy/logging#azure-blob-storage) + +## Docker Run LiteLLM [​](https://docs.litellm.ai/release_notes/tags/fallbacks\#docker-run-litellm "Direct link to Docker Run LiteLLM") + +```codeBlockLines_e6Vv +docker run \ +-e STORE_MODEL_IN_DB=True \ +-p 4000:4000 \ +ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.55.8-stable + +``` + +## Get Daily Updates [​](https://docs.litellm.ai/release_notes/tags/fallbacks\#get-daily-updates "Direct link to Get Daily Updates") + +LiteLLM ships new releases every day. [Follow us on LinkedIn](https://www.linkedin.com/company/berri-ai/) to get daily updates. + +## Finetuning Updates and Improvements +[Skip to main content](https://docs.litellm.ai/release_notes/tags/finetuning#__docusaurus_skipToContent_fallback) + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes/tags/finetuning\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/finetuning\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes/tags/finetuning\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes/tags/finetuning\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes/tags/finetuning\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes/tags/finetuning\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes/tags/finetuning\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes/tags/finetuning\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes/tags/finetuning\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes/tags/finetuning\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes/tags/finetuning\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes/tags/finetuning\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +## Fireworks AI Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/fireworks-ai#__docusaurus_skipToContent_fallback) + +`deepgram`, `fireworks ai`, `vision`, `admin ui`, `dependency upgrades` + +## New Models [​](https://docs.litellm.ai/release_notes/tags/fireworks-ai\#new-models "Direct link to New Models") + +### **Deepgram Speech to Text** [​](https://docs.litellm.ai/release_notes/tags/fireworks-ai\#deepgram-speech-to-text "Direct link to deepgram-speech-to-text") + +New Speech to Text support for Deepgram models. [**Start Here**](https://docs.litellm.ai/docs/providers/deepgram) + +```codeBlockLines_e6Vv +from litellm import transcription +import os + +# set api keys +os.environ["DEEPGRAM_API_KEY"] = "" +audio_file = open("/path/to/audio.mp3", "rb") + +response = transcription(model="deepgram/nova-2", file=audio_file) + +print(f"response: {response}") + +``` + +### **Fireworks AI - Vision** support for all models [​](https://docs.litellm.ai/release_notes/tags/fireworks-ai\#fireworks-ai---vision-support-for-all-models "Direct link to fireworks-ai---vision-support-for-all-models") + +LiteLLM supports document inlining for Fireworks AI models. This is useful for models that are not vision models, but still need to parse documents/images/etc. +LiteLLM will add `#transform=inline` to the url of the image\_url, if the model is not a vision model [See Code](https://github.com/BerriAI/litellm/blob/1ae9d45798bdaf8450f2dfdec703369f3d2212b7/litellm/llms/fireworks_ai/chat/transformation.py#L114) + +## Proxy Admin UI [​](https://docs.litellm.ai/release_notes/tags/fireworks-ai\#proxy-admin-ui "Direct link to Proxy Admin UI") + +- `Test Key` Tab displays `model` used in response + +![](https://docs.litellm.ai/assets/ideal-img/ui_model.72a8982.1920.png) + +- `Test Key` Tab renders content in `.md`, `.py` (any code/markdown format) + +![](https://docs.litellm.ai/assets/ideal-img/ui_format.337282b.1920.png) + +## Dependency Upgrades [​](https://docs.litellm.ai/release_notes/tags/fireworks-ai\#dependency-upgrades "Direct link to Dependency Upgrades") + +- (Security fix) Upgrade to `fastapi==0.115.5` [https://github.com/BerriAI/litellm/pull/7447](https://github.com/BerriAI/litellm/pull/7447) + +## Bug Fixes [​](https://docs.litellm.ai/release_notes/tags/fireworks-ai\#bug-fixes "Direct link to Bug Fixes") + +- Add health check support for realtime models [Here](https://docs.litellm.ai/docs/proxy/health#realtime-models) +- Health check error with audio\_transcription model [https://github.com/BerriAI/litellm/issues/5999](https://github.com/BerriAI/litellm/issues/5999) + +## Guardrails and Logging Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/guardrails#__docusaurus_skipToContent_fallback) + +`guardrails`, `logging`, `virtual key management`, `new models` + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## New Features [​](https://docs.litellm.ai/release_notes/tags/guardrails\#new-features "Direct link to New Features") + +### ✨ Log Guardrail Traces [​](https://docs.litellm.ai/release_notes/tags/guardrails\#-log-guardrail-traces "Direct link to ✨ Log Guardrail Traces") + +Track guardrail failure rate and if a guardrail is going rogue and failing requests. [Start here](https://docs.litellm.ai/docs/proxy/guardrails/quick_start) + +#### Traced Guardrail Success [​](https://docs.litellm.ai/release_notes/tags/guardrails\#traced-guardrail-success "Direct link to Traced Guardrail Success") + +![](https://docs.litellm.ai/assets/ideal-img/gd_success.02a2daf.1862.png) + +#### Traced Guardrail Failure [​](https://docs.litellm.ai/release_notes/tags/guardrails\#traced-guardrail-failure "Direct link to Traced Guardrail Failure") + +![](https://docs.litellm.ai/assets/ideal-img/gd_fail.457338e.1848.png) + +### `/guardrails/list` [​](https://docs.litellm.ai/release_notes/tags/guardrails\#guardrailslist "Direct link to guardrailslist") + +`/guardrails/list` allows clients to view available guardrails + supported guardrail params + +```codeBlockLines_e6Vv +curl -X GET 'http://0.0.0.0:4000/guardrails/list' + +``` + +Expected response + +```codeBlockLines_e6Vv +{ + "guardrails": [\ + {\ + "guardrail_name": "aporia-post-guard",\ + "guardrail_info": {\ + "params": [\ + {\ + "name": "toxicity_score",\ + "type": "float",\ + "description": "Score between 0-1 indicating content toxicity level"\ + },\ + {\ + "name": "pii_detection",\ + "type": "boolean"\ + }\ + ]\ + }\ + }\ + ] +} + +``` + +### ✨ Guardrails with Mock LLM [​](https://docs.litellm.ai/release_notes/tags/guardrails\#-guardrails-with-mock-llm "Direct link to ✨ Guardrails with Mock LLM") + +Send `mock_response` to test guardrails without making an LLM call. More info on `mock_response` [here](https://docs.litellm.ai/docs/proxy/guardrails/quick_start) + +```codeBlockLines_e6Vv +curl -i http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [\ + {"role": "user", "content": "hi my email is ishaan@berri.ai"}\ + ], + "mock_response": "This is a mock response", + "guardrails": ["aporia-pre-guard", "aporia-post-guard"] + }' + +``` + +### Assign Keys to Users [​](https://docs.litellm.ai/release_notes/tags/guardrails\#assign-keys-to-users "Direct link to Assign Keys to Users") + +You can now assign keys to users via Proxy UI + +![](https://docs.litellm.ai/assets/ideal-img/ui_key.9642332.1212.png) + +## New Models [​](https://docs.litellm.ai/release_notes/tags/guardrails\#new-models "Direct link to New Models") + +- `openrouter/openai/o1` +- `vertex_ai/mistral-large@2411` + +## Fixes [​](https://docs.litellm.ai/release_notes/tags/guardrails\#fixes "Direct link to Fixes") + +- Fix `vertex_ai/` mistral model pricing: [https://github.com/BerriAI/litellm/pull/7345](https://github.com/BerriAI/litellm/pull/7345) +- Missing model\_group field in logs for aspeech call types [https://github.com/BerriAI/litellm/pull/7392](https://github.com/BerriAI/litellm/pull/7392) + +`key management`, `budgets/rate limits`, `logging`, `guardrails` + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## ✨ Budget / Rate Limit Tiers [​](https://docs.litellm.ai/release_notes/tags/guardrails\#-budget--rate-limit-tiers "Direct link to ✨ Budget / Rate Limit Tiers") + +Define tiers with rate limits. Assign them to keys. + +Use this to control access and budgets across a lot of keys. + +**[Start here](https://docs.litellm.ai/docs/proxy/rate_limit_tiers)** + +```codeBlockLines_e6Vv +curl -L -X POST 'http://0.0.0.0:4000/budget/new' \ +-H 'Authorization: Bearer sk-1234' \ +-H 'Content-Type: application/json' \ +-d '{ + "budget_id": "high-usage-tier", + "model_max_budget": { + "gpt-4o": {"rpm_limit": 1000000} + } +}' + +``` + +## OTEL Bug Fix [​](https://docs.litellm.ai/release_notes/tags/guardrails\#otel-bug-fix "Direct link to OTEL Bug Fix") + +LiteLLM was double logging litellm\_request span. This is now fixed. + +[Relevant PR](https://github.com/BerriAI/litellm/pull/7435) + +## Logging for Finetuning Endpoints [​](https://docs.litellm.ai/release_notes/tags/guardrails\#logging-for-finetuning-endpoints "Direct link to Logging for Finetuning Endpoints") + +Logs for finetuning requests are now available on all logging providers (e.g. Datadog). + +What's logged per request: + +- file\_id +- finetuning\_job\_id +- any key/team metadata + +**Start Here:** + +- [Setup Finetuning](https://docs.litellm.ai/docs/fine_tuning) +- [Setup Logging](https://docs.litellm.ai/docs/proxy/logging#datadog) + +## Dynamic Params for Guardrails [​](https://docs.litellm.ai/release_notes/tags/guardrails\#dynamic-params-for-guardrails "Direct link to Dynamic Params for Guardrails") + +You can now set custom parameters (like success threshold) for your guardrails in each request. + +[See guardrails spec for more details](https://docs.litellm.ai/docs/proxy/guardrails/custom_guardrail#-pass-additional-parameters-to-guardrail) + +`batches`, `guardrails`, `team management`, `custom auth` + +![](https://docs.litellm.ai/assets/ideal-img/batches_cost_tracking.8fc9663.1208.png) + +info + +Get a free 7-day LiteLLM Enterprise trial here. [Start here](https://www.litellm.ai/#trial) + +**No call needed** + +## ✨ Cost Tracking, Logging for Batches API ( `/batches`) [​](https://docs.litellm.ai/release_notes/tags/guardrails\#-cost-tracking-logging-for-batches-api-batches "Direct link to -cost-tracking-logging-for-batches-api-batches") + +Track cost, usage for Batch Creation Jobs. [Start here](https://docs.litellm.ai/docs/batches) + +## ✨ `/guardrails/list` endpoint [​](https://docs.litellm.ai/release_notes/tags/guardrails\#-guardrailslist-endpoint "Direct link to -guardrailslist-endpoint") + +Show available guardrails to users. [Start here](https://litellm-api.up.railway.app/#/Guardrails) + +## ✨ Allow teams to add models [​](https://docs.litellm.ai/release_notes/tags/guardrails\#-allow-teams-to-add-models "Direct link to ✨ Allow teams to add models") + +This enables team admins to call their own finetuned models via litellm proxy. [Start here](https://docs.litellm.ai/docs/proxy/team_model_add) + +## ✨ Common checks for custom auth [​](https://docs.litellm.ai/release_notes/tags/guardrails\#-common-checks-for-custom-auth "Direct link to ✨ Common checks for custom auth") + +Calling the internal common\_checks function in custom auth is now enforced as an enterprise feature. This allows admins to use litellm's default budget/auth checks within their custom auth implementation. [Start here](https://docs.litellm.ai/docs/proxy/virtual_keys#custom-auth) + +## ✨ Assigning team admins [​](https://docs.litellm.ai/release_notes/tags/guardrails\#-assigning-team-admins "Direct link to ✨ Assigning team admins") + +Team admins is graduating from beta and moving to our enterprise tier. This allows proxy admins to allow others to manage keys/models for their own teams (useful for projects in production). [Start here](https://docs.litellm.ai/docs/proxy/virtual_keys#restricting-key-generation) + +## LLM Features and Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/humanloop#__docusaurus_skipToContent_fallback) + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes/tags/humanloop\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/humanloop\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes/tags/humanloop\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes/tags/humanloop\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes/tags/humanloop\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes/tags/humanloop\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes/tags/humanloop\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes/tags/humanloop\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes/tags/humanloop\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes/tags/humanloop\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes/tags/humanloop\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes/tags/humanloop\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +## Key Management Overview +[Skip to main content](https://docs.litellm.ai/release_notes/tags/key-management#__docusaurus_skipToContent_fallback) + +`key management`, `budgets/rate limits`, `logging`, `guardrails` + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## ✨ Budget / Rate Limit Tiers [​](https://docs.litellm.ai/release_notes/tags/key-management\#-budget--rate-limit-tiers "Direct link to ✨ Budget / Rate Limit Tiers") + +Define tiers with rate limits. Assign them to keys. + +Use this to control access and budgets across a lot of keys. + +**[Start here](https://docs.litellm.ai/docs/proxy/rate_limit_tiers)** + +```codeBlockLines_e6Vv +curl -L -X POST 'http://0.0.0.0:4000/budget/new' \ +-H 'Authorization: Bearer sk-1234' \ +-H 'Content-Type: application/json' \ +-d '{ + "budget_id": "high-usage-tier", + "model_max_budget": { + "gpt-4o": {"rpm_limit": 1000000} + } +}' + +``` + +## OTEL Bug Fix [​](https://docs.litellm.ai/release_notes/tags/key-management\#otel-bug-fix "Direct link to OTEL Bug Fix") + +LiteLLM was double logging litellm\_request span. This is now fixed. + +[Relevant PR](https://github.com/BerriAI/litellm/pull/7435) + +## Logging for Finetuning Endpoints [​](https://docs.litellm.ai/release_notes/tags/key-management\#logging-for-finetuning-endpoints "Direct link to Logging for Finetuning Endpoints") + +Logs for finetuning requests are now available on all logging providers (e.g. Datadog). + +What's logged per request: + +- file\_id +- finetuning\_job\_id +- any key/team metadata + +**Start Here:** + +- [Setup Finetuning](https://docs.litellm.ai/docs/fine_tuning) +- [Setup Logging](https://docs.litellm.ai/docs/proxy/logging#datadog) + +## Dynamic Params for Guardrails [​](https://docs.litellm.ai/release_notes/tags/key-management\#dynamic-params-for-guardrails "Direct link to Dynamic Params for Guardrails") + +You can now set custom parameters (like success threshold) for your guardrails in each request. + +[See guardrails spec for more details](https://docs.litellm.ai/docs/proxy/guardrails/custom_guardrail#-pass-additional-parameters-to-guardrail) + +## LiteLLM Release Notes +[Skip to main content](https://docs.litellm.ai/release_notes/tags/langfuse#__docusaurus_skipToContent_fallback) + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes/tags/langfuse\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/langfuse\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes/tags/langfuse\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes/tags/langfuse\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes/tags/langfuse\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes/tags/langfuse\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes/tags/langfuse\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes/tags/langfuse\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes/tags/langfuse\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes/tags/langfuse\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes/tags/langfuse\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes/tags/langfuse\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +`langfuse`, `management endpoints`, `ui`, `prometheus`, `secret management` + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes/tags/langfuse\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +Langfuse Prompt Management is being labelled as BETA. This allows us to iterate quickly on the feedback we're receiving, and making the status clearer to users. We expect to make this feature to be stable by next month (February 2025). + +Changes: + +- Include the client message in the LLM API Request. (Previously only the prompt template was sent, and the client message was ignored). +- Log the prompt template in the logged request (e.g. to s3/langfuse). +- Log the 'prompt\_id' and 'prompt\_variables' in the logged request (e.g. to s3/langfuse). + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Team/Organization Management + UI Improvements [​](https://docs.litellm.ai/release_notes/tags/langfuse\#teamorganization-management--ui-improvements "Direct link to Team/Organization Management + UI Improvements") + +Managing teams and organizations on the UI is now easier. + +Changes: + +- Support for editing user role within team on UI. +- Support updating team member role to admin via api - `/team/member_update` +- Show team admins all keys for their team. +- Add organizations with budgets +- Assign teams to orgs on the UI +- Auto-assign SSO users to teams + +[Start Here](https://docs.litellm.ai/docs/proxy/self_serve) + +## Hashicorp Vault Support [​](https://docs.litellm.ai/release_notes/tags/langfuse\#hashicorp-vault-support "Direct link to Hashicorp Vault Support") + +We now support writing LiteLLM Virtual API keys to Hashicorp Vault. + +[Start Here](https://docs.litellm.ai/docs/proxy/vault) + +## Custom Prometheus Metrics [​](https://docs.litellm.ai/release_notes/tags/langfuse\#custom-prometheus-metrics "Direct link to Custom Prometheus Metrics") + +Define custom prometheus metrics, and track usage/latency/no. of requests against them + +This allows for more fine-grained tracking - e.g. on prompt template passed in request metadata + +[Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +A new LiteLLM Stable release [just went out](https://github.com/BerriAI/litellm/releases/tag/v1.55.8-stable). Here are 5 updates since v1.52.2-stable. + +`langfuse`, `fallbacks`, `new models`, `azure_storage` + +![](https://docs.litellm.ai/assets/ideal-img/langfuse_prmpt_mgmt.19b8982.1920.png) + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes/tags/langfuse\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +This makes it easy to run experiments or change the specific models `gpt-4o` to `gpt-4o-mini` on Langfuse, instead of making changes in your applications. [Start here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Control fallback prompts client-side [​](https://docs.litellm.ai/release_notes/tags/langfuse\#control-fallback-prompts-client-side "Direct link to Control fallback prompts client-side") + +> Claude prompts are different than OpenAI + +Pass in prompts specific to model when doing fallbacks. [Start here](https://docs.litellm.ai/docs/proxy/reliability#control-fallback-prompts) + +## New Providers / Models [​](https://docs.litellm.ai/release_notes/tags/langfuse\#new-providers--models "Direct link to New Providers / Models") + +- [NVIDIA Triton](https://developer.nvidia.com/triton-inference-server) `/infer` endpoint. [Start here](https://docs.litellm.ai/docs/providers/triton-inference-server) +- [Infinity](https://github.com/michaelfeil/infinity) Rerank Models [Start here](https://docs.litellm.ai/docs/providers/infinity) + +## ✨ Azure Data Lake Storage Support [​](https://docs.litellm.ai/release_notes/tags/langfuse\#-azure-data-lake-storage-support "Direct link to ✨ Azure Data Lake Storage Support") + +Send LLM usage (spend, tokens) data to [Azure Data Lake](https://learn.microsoft.com/en-us/azure/storage/blobs/data-lake-storage-introduction). This makes it easy to consume usage data on other services (eg. Databricks) +[Start here](https://docs.litellm.ai/docs/proxy/logging#azure-blob-storage) + +## Docker Run LiteLLM [​](https://docs.litellm.ai/release_notes/tags/langfuse\#docker-run-litellm "Direct link to Docker Run LiteLLM") + +```codeBlockLines_e6Vv +docker run \ +-e STORE_MODEL_IN_DB=True \ +-p 4000:4000 \ +ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.55.8-stable + +``` + +## Get Daily Updates [​](https://docs.litellm.ai/release_notes/tags/langfuse\#get-daily-updates "Direct link to Get Daily Updates") + +LiteLLM ships new releases every day. [Follow us on LinkedIn](https://www.linkedin.com/company/berri-ai/) to get daily updates. + +## LLM Translation Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/llm-translation#__docusaurus_skipToContent_fallback) + +These are the changes since `v1.61.20-stable`. + +This release is primarily focused on: + +- LLM Translation improvements (more `thinking` content improvements) +- UI improvements (Error logs now shown on UI) + +info + +This release will be live on 03/09/2025 + +![](https://docs.litellm.ai/assets/ideal-img/v1632_release.7b42da1.1920.jpg) + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Add `supports_pdf_input` for specific Bedrock Claude models [PR](https://github.com/BerriAI/litellm/commit/f63cf0030679fe1a43d03fb196e815a0f28dae92) +2. Add pricing for amazon `eu` models [PR](https://github.com/BerriAI/litellm/commits/main/model_prices_and_context_window.json) +3. Fix Azure O1 mini pricing [PR](https://github.com/BerriAI/litellm/commit/52de1949ef2f76b8572df751f9c868a016d4832c) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#llm-translation "Direct link to LLM Translation") + +![](https://docs.litellm.ai/assets/ideal-img/anthropic_thinking.3bef9d6.1920.jpg) + +01. Support `/openai/` passthrough for Assistant endpoints. [Get Started](https://docs.litellm.ai/docs/pass_through/openai_passthrough) +02. Bedrock Claude - fix tool calling transformation on invoke route. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---function-calling--tool-calling) +03. Bedrock Claude - response\_format support for claude on invoke route. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---structured-output--json-mode) +04. Bedrock - pass `description` if set in response\_format. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---structured-output--json-mode) +05. Bedrock - Fix passing response\_format: {"type": "text"}. [PR](https://github.com/BerriAI/litellm/commit/c84b489d5897755139aa7d4e9e54727ebe0fa540) +06. OpenAI - Handle sending image\_url as str to openai. [Get Started](https://docs.litellm.ai/docs/completion/vision) +07. Deepseek - return 'reasoning\_content' missing on streaming. [Get Started](https://docs.litellm.ai/docs/reasoning_content) +08. Caching - Support caching on reasoning content. [Get Started](https://docs.litellm.ai/docs/proxy/caching) +09. Bedrock - handle thinking blocks in assistant message. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) +10. Anthropic - Return `signature` on streaming. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) + +- Note: We've also migrated from `signature_delta` to `signature`. [Read more](https://docs.litellm.ai/release_notes/v1.63.0) + +11. Support format param for specifying image type. [Get Started](https://docs.litellm.ai/docs/completion/vision.md#explicitly-specify-image-type) +12. Anthropic - `/v1/messages` endpoint - `thinking` param support. [Get Started](https://docs.litellm.ai/docs/anthropic_unified.md) + +- Note: this refactors the \[BETA\] unified `/v1/messages` endpoint, to just work for the Anthropic API. + +13. Vertex AI - handle $id in response schema when calling vertex ai. [Get Started](https://docs.litellm.ai/docs/providers/vertex#json-schema) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Batches API - Fix cost calculation to run on retrieve\_batch. [Get Started](https://docs.litellm.ai/docs/batches) +2. Batches API - Log batch models in spend logs / standard logging payload. [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec.md#standardlogginghiddenparams) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +![](https://docs.litellm.ai/assets/ideal-img/error_logs.63c5dc9.1920.jpg) + +1. Virtual Keys Page + - Allow team/org filters to be searchable on the Create Key Page + - Add created\_by and updated\_by fields to Keys table + - Show 'user\_email' on key table + - Show 100 Keys Per Page, Use full height, increase width of key alias +2. Logs Page + - Show Error Logs on LiteLLM UI + - Allow Internal Users to View their own logs +3. Internal Users Page + - Allow admin to control default model access for internal users +4. Fix session handling with cookies + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Fix prometheus metrics w/ custom metrics, when keys containing team\_id make requests. [PR](https://github.com/BerriAI/litellm/pull/8935) + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Cooldowns - Support cooldowns on models called with client side credentials. [Get Started](https://docs.litellm.ai/docs/proxy/clientside_auth#pass-user-llm-api-keys--api-base) +2. Tag-based Routing - ensures tag-based routing across all endpoints ( `/embeddings`, `/image_generation`, etc.). [Get Started](https://docs.litellm.ai/docs/proxy/tag_routing) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Raise BadRequestError when unknown model passed in request +2. Enforce model access restrictions on Azure OpenAI proxy route +3. Reliability fix - Handle emoji’s in text - fix orjson error +4. Model Access Patch - don't overwrite litellm.anthropic\_models when running auth checks +5. Enable setting timezone information in docker image + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.61.20-stable...v1.63.2-stable) + +v1.63.0 fixes Anthropic 'thinking' response on streaming to return the `signature` block. [Github Issue](https://github.com/BerriAI/litellm/issues/8964) + +It also moves the response structure from `signature_delta` to `signature` to be the same as Anthropic. [Anthropic Docs](https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking#implementing-extended-thinking) + +## Diff [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#diff "Direct link to Diff") + +```codeBlockLines_e6Vv +"message": { + ... + "reasoning_content": "The capital of France is Paris.", + "thinking_blocks": [\ + {\ + "type": "thinking",\ + "thinking": "The capital of France is Paris.",\ +- "signature_delta": "EqoBCkgIARABGAIiQL2UoU0b1OHYi+..." # 👈 OLD FORMAT\ ++ "signature": "EqoBCkgIARABGAIiQL2UoU0b1OHYi+..." # 👈 KEY CHANGE\ + }\ + ] +} + +``` + +These are the changes since `v1.61.13-stable`. + +This release is primarily focused on: + +- LLM Translation improvements (claude-3-7-sonnet + 'thinking'/'reasoning\_content' support) +- UI improvements (add model flow, user management, etc) + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Anthropic 3-7 sonnet support + cost tracking (Anthropic API + Bedrock + Vertex AI + OpenRouter) +1. Anthropic API [Start here](https://docs.litellm.ai/docs/providers/anthropic#usage---thinking--reasoning_content) +2. Bedrock API [Start here](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) +3. Vertex AI API [See here](https://docs.litellm.ai/docs/providers/vertex#usage---thinking--reasoning_content) +4. OpenRouter [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L5626) +2. Gpt-4.5-preview support + cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L79) +3. Azure AI - Phi-4 cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L1773) +4. Claude-3.5-sonnet - vision support updated on Anthropic API [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2888) +5. Bedrock llama vision support [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L7714) +6. Cerebras llama3.3-70b pricing [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2697) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#llm-translation "Direct link to LLM Translation") + +1. Infinity Rerank - support returning documents when return\_documents=True [Start here](https://docs.litellm.ai/docs/providers/infinity#usage---returning-documents) +2. Amazon Deepseek - `` param extraction into ‘reasoning\_content’ [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-imported-models-deepseek-deepseek-r1) +3. Amazon Titan Embeddings - filter out ‘aws\_’ params from request body [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-embedding) +4. Anthropic ‘thinking’ + ‘reasoning\_content’ translation support (Anthropic API, Bedrock, Vertex AI) [Start here](https://docs.litellm.ai/docs/reasoning_content) +5. VLLM - support ‘video\_url’ [Start here](https://docs.litellm.ai/docs/providers/vllm#send-video-url-to-vllm) +6. Call proxy via litellm SDK: Support `litellm_proxy/` for embedding, image\_generation, transcription, speech, rerank [Start here](https://docs.litellm.ai/docs/providers/litellm_proxy) +7. OpenAI Pass-through - allow using Assistants GET, DELETE on /openai pass through routes [Start here](https://docs.litellm.ai/docs/pass_through/openai_passthrough) +8. Message Translation - fix openai message for assistant msg if role is missing - openai allows this +9. O1/O3 - support ‘drop\_params’ for o3-mini and o1 parallel\_tool\_calls param (not supported currently) [See here](https://docs.litellm.ai/docs/completion/drop_params) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Cost tracking for rerank via Bedrock [See PR](https://github.com/BerriAI/litellm/commit/b682dc4ec8fd07acf2f4c981d2721e36ae2a49c5) +2. Anthropic pass-through - fix race condition causing cost to not be tracked [See PR](https://github.com/BerriAI/litellm/pull/8874) +3. Anthropic pass-through: Ensure accurate token counting [See PR](https://github.com/BerriAI/litellm/pull/8880) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +01. Models Page - Allow sorting models by ‘created at’ +02. Models Page - Edit Model Flow Improvements +03. Models Page - Fix Adding Azure, Azure AI Studio models on UI +04. Internal Users Page - Allow Bulk Adding Internal Users on UI +05. Internal Users Page - Allow sorting users by ‘created at’ +06. Virtual Keys Page - Allow searching for UserIDs on the dropdown when assigning a user to a team [See PR](https://github.com/BerriAI/litellm/pull/8844) +07. Virtual Keys Page - allow creating a user when assigning keys to users [See PR](https://github.com/BerriAI/litellm/pull/8844) +08. Model Hub Page - fix text overflow issue [See PR](https://github.com/BerriAI/litellm/pull/8749) +09. Admin Settings Page - Allow adding MSFT SSO on UI +10. Backend - don't allow creating duplicate internal users in DB + +## Helm [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#helm "Direct link to Helm") + +1. support ttlSecondsAfterFinished on the migration job - [See PR](https://github.com/BerriAI/litellm/pull/8593) +2. enhance migrations job with additional configurable properties - [See PR](https://github.com/BerriAI/litellm/pull/8636) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Arize Phoenix support +2. ‘No-log’ - fix ‘no-log’ param support on embedding calls + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Single Deployment Cooldown logic - Use allowed\_fails or allowed\_fail\_policy if set [Start here](https://docs.litellm.ai/docs/routing#advanced-custom-retries-cooldowns-based-on-error-type) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Hypercorn - fix reading / parsing request body +2. Windows - fix running proxy in windows +3. DD-Trace - fix dd-trace enablement on proxy + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/llm-translation\#complete-git-diff "Direct link to Complete Git Diff") + +View the complete git diff [here](https://github.com/BerriAI/litellm/compare/v1.61.13-stable...v1.61.20-stable). + +## LiteLLM Logging Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/logging#__docusaurus_skipToContent_fallback) + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/logging\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. New OpenAI `/image/variations` endpoint BETA support [Docs](https://docs.litellm.ai/docs/image_variations) +2. Topaz API support on OpenAI `/image/variations` BETA endpoint [Docs](https://docs.litellm.ai/docs/providers/topaz) +3. Deepseek - r1 support w/ reasoning\_content ( [Deepseek API](https://docs.litellm.ai/docs/providers/deepseek#reasoning-models), [Vertex AI](https://docs.litellm.ai/docs/providers/vertex#model-garden), [Bedrock](https://docs.litellm.ai/docs/providers/bedrock#deepseek)) +4. Azure - Add azure o1 pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L952) +5. Anthropic - handle `-latest` tag in model for cost calculation +6. Gemini-2.0-flash-thinking - add model pricing (it’s 0.0) [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L3393) +7. Bedrock - add stability sd3 model pricing [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6814) (s/o [Marty Sullivan](https://github.com/marty-sullivan)) +8. Bedrock - add us.amazon.nova-lite-v1:0 to model cost map [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L5619) +9. TogetherAI - add new together\_ai llama3.3 models [See Here](https://github.com/BerriAI/litellm/blob/b8b927f23bc336862dacb89f59c784a8d62aaa15/model_prices_and_context_window.json#L6985) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/logging\#llm-translation "Direct link to LLM Translation") + +01. LM Studio -> fix async embedding call +02. Gpt 4o models - fix response\_format translation +03. Bedrock nova - expand supported document types to include .md, .csv, etc. [Start Here](https://docs.litellm.ai/docs/providers/bedrock#usage---pdf--document-understanding) +04. Bedrock - docs on IAM role based access for bedrock - [Start Here](https://docs.litellm.ai/docs/providers/bedrock#sts-role-based-auth) +05. Bedrock - cache IAM role credentials when used +06. Google AI Studio ( `gemini/`) \- support gemini 'frequency\_penalty' and 'presence\_penalty' +07. Azure O1 - fix model name check +08. WatsonX - ZenAPIKey support for WatsonX [Docs](https://docs.litellm.ai/docs/providers/watsonx) +09. Ollama Chat - support json schema response format [Start Here](https://docs.litellm.ai/docs/providers/ollama#json-schema-support) +10. Bedrock - return correct bedrock status code and error message if error during streaming +11. Anthropic - Supported nested json schema on anthropic calls +12. OpenAI - `metadata` param preview support + 1. SDK - enable via `litellm.enable_preview_features = True` + 2. PROXY - enable via `litellm_settings::enable_preview_features: true` +13. Replicate - retry completion response on status=processing + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/logging\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Bedrock - QA asserts all bedrock regional models have same `supported_` as base model +2. Bedrock - fix bedrock converse cost tracking w/ region name specified +3. Spend Logs reliability fix - when `user` passed in request body is int instead of string +4. Ensure ‘base\_model’ cost tracking works across all endpoints +5. Fixes for Image generation cost tracking +6. Anthropic - fix anthropic end user cost tracking +7. JWT / OIDC Auth - add end user id tracking from jwt auth + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/logging\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +01. allows team member to become admin post-add (ui + endpoints) +02. New edit/delete button for updating team membership on UI +03. If team admin - show all team keys +04. Model Hub - clarify cost of models is per 1m tokens +05. Invitation Links - fix invalid url generated +06. New - SpendLogs Table Viewer - allows proxy admin to view spend logs on UI + 1. New spend logs - allow proxy admin to ‘opt in’ to logging request/response in spend logs table - enables easier abuse detection + 2. Show country of origin in spend logs + 3. Add pagination + filtering by key name/team name +07. `/key/delete` \- allow team admin to delete team keys +08. Internal User ‘view’ - fix spend calculation when team selected +09. Model Analytics is now on Free +10. Usage page - shows days when spend = 0, and round spend on charts to 2 sig figs +11. Public Teams - allow admins to expose teams for new users to ‘join’ on UI - [Start Here](https://docs.litellm.ai/docs/proxy/public_teams) +12. Guardrails + 1. set/edit guardrails on a virtual key + 2. Allow setting guardrails on a team + 3. Set guardrails on team create + edit page +13. Support temporary budget increases on `/key/update` \- new `temp_budget_increase` and `temp_budget_expiry` fields - [Start Here](https://docs.litellm.ai/docs/proxy/virtual_keys#temporary-budget-increase) +14. Support writing new key alias to AWS Secret Manager - on key rotation [Start Here](https://docs.litellm.ai/docs/secret#aws-secret-manager) + +## Helm [​](https://docs.litellm.ai/release_notes/tags/logging\#helm "Direct link to Helm") + +1. add securityContext and pull policy values to migration job (s/o [https://github.com/Hexoplon](https://github.com/Hexoplon)) +2. allow specifying envVars on values.yaml +3. new helm lint test + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/logging\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Log the used prompt when prompt management used. [Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) +2. Support s3 logging with team alias prefixes - [Start Here](https://docs.litellm.ai/docs/proxy/logging#team-alias-prefix-in-object-key) +3. Prometheus [Start Here](https://docs.litellm.ai/docs/proxy/prometheus) +1. fix litellm\_llm\_api\_time\_to\_first\_token\_metric not populating for bedrock models +2. emit remaining team budget metric on regular basis (even when call isn’t made) - allows for more stable metrics on Grafana/etc. +3. add key and team level budget metrics +4. emit `litellm_overhead_latency_metric` +5. Emit `litellm_team_budget_reset_at_metric` and `litellm_api_key_budget_remaining_hours_metric` +4. Datadog - support logging spend tags to Datadog. [Start Here](https://docs.litellm.ai/docs/proxy/enterprise#tracking-spend-for-custom-tags) +5. Langfuse - fix logging request tags, read from standard logging payload +6. GCS - don’t truncate payload on logging +7. New GCS Pub/Sub logging support [Start Here](https://docs.litellm.ai/docs/proxy/logging#google-cloud-storage---pubsub-topic) +8. Add AIM Guardrails support [Start Here](https://docs.litellm.ai/docs/proxy/guardrails/aim_security) + +## Security [​](https://docs.litellm.ai/release_notes/tags/logging\#security "Direct link to Security") + +1. New Enterprise SLA for patching security vulnerabilities. [See Here](https://docs.litellm.ai/docs/enterprise#slas--professional-support) +2. Hashicorp - support using vault namespace for TLS auth. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) +3. Azure - DefaultAzureCredential support + +## Health Checks [​](https://docs.litellm.ai/release_notes/tags/logging\#health-checks "Direct link to Health Checks") + +1. Cleanup pricing-only model names from wildcard route list - prevent bad health checks +2. Allow specifying a health check model for wildcard routes - [https://docs.litellm.ai/docs/proxy/health#wildcard-routes](https://docs.litellm.ai/docs/proxy/health#wildcard-routes) +3. New ‘health\_check\_timeout ‘ param with default 1min upperbound to prevent bad model from health check to hang and cause pod restarts. [Start Here](https://docs.litellm.ai/docs/proxy/health#health-check-timeout) +4. Datadog - add data dog service health check + expose new `/health/services` endpoint. [Start Here](https://docs.litellm.ai/docs/proxy/health#healthservices) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/logging\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +01. 3x increase in RPS - moving to orjson for reading request body +02. LLM Routing speedup - using cached get model group info +03. SDK speedup - using cached get model info helper - reduces CPU work to get model info +04. Proxy speedup - only read request body 1 time per request +05. Infinite loop detection scripts added to codebase +06. Bedrock - pure async image transformation requests +07. Cooldowns - single deployment model group if 100% calls fail in high traffic - prevents an o1 outage from impacting other calls +08. Response Headers - return + 1. `x-litellm-timeout` + 2. `x-litellm-attempted-retries` + 3. `x-litellm-overhead-duration-ms` + 4. `x-litellm-response-duration-ms` +09. ensure duplicate callbacks are not added to proxy +10. Requirements.txt - bump certifi version + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/logging\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. JWT / OIDC Auth - new `enforce_rbac` param,allows proxy admin to prevent any unmapped yet authenticated jwt tokens from calling proxy. [Start Here](https://docs.litellm.ai/docs/proxy/token_auth#enforce-role-based-access-control-rbac) +2. fix custom openapi schema generation for customized swagger’s +3. Request Headers - support reading `x-litellm-timeout` param from request headers. Enables model timeout control when using Vercel’s AI SDK + LiteLLM Proxy. [Start Here](https://docs.litellm.ai/docs/proxy/request_headers#litellm-headers) +4. JWT / OIDC Auth - new `role` based permissions for model authentication. [See Here](https://docs.litellm.ai/docs/proxy/jwt_auth_arch) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/logging\#complete-git-diff "Direct link to Complete Git Diff") + +This is the diff between v1.57.8-stable and v1.59.8-stable. + +Use this to see the changes in the codebase. + +[**Git Diff**](https://github.com/BerriAI/litellm/compare/v1.57.8-stable...v1.59.8-stable) + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## UI Improvements [​](https://docs.litellm.ai/release_notes/tags/logging\#ui-improvements "Direct link to UI Improvements") + +### \[Opt In\] Admin UI - view messages / responses [​](https://docs.litellm.ai/release_notes/tags/logging\#opt-in-admin-ui---view-messages--responses "Direct link to opt-in-admin-ui---view-messages--responses") + +You can now view messages and response logs on Admin UI. + +![](https://docs.litellm.ai/assets/ideal-img/ui_logs.17b0459.1497.png) + +How to enable it - add `store_prompts_in_spend_logs: true` to your `proxy_config.yaml` + +Once this flag is enabled, your `messages` and `responses` will be stored in the `LiteLLM_Spend_Logs` table. + +```codeBlockLines_e6Vv +general_settings: + store_prompts_in_spend_logs: true + +``` + +## DB Schema Change [​](https://docs.litellm.ai/release_notes/tags/logging\#db-schema-change "Direct link to DB Schema Change") + +Added `messages` and `responses` to the `LiteLLM_Spend_Logs` table. + +**By default this is not logged.** If you want `messages` and `responses` to be logged, you need to opt in with this setting + +```codeBlockLines_e6Vv +general_settings: + store_prompts_in_spend_logs: true + +``` + +`guardrails`, `logging`, `virtual key management`, `new models` + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## New Features [​](https://docs.litellm.ai/release_notes/tags/logging\#new-features "Direct link to New Features") + +### ✨ Log Guardrail Traces [​](https://docs.litellm.ai/release_notes/tags/logging\#-log-guardrail-traces "Direct link to ✨ Log Guardrail Traces") + +Track guardrail failure rate and if a guardrail is going rogue and failing requests. [Start here](https://docs.litellm.ai/docs/proxy/guardrails/quick_start) + +#### Traced Guardrail Success [​](https://docs.litellm.ai/release_notes/tags/logging\#traced-guardrail-success "Direct link to Traced Guardrail Success") + +![](https://docs.litellm.ai/assets/ideal-img/gd_success.02a2daf.1862.png) + +#### Traced Guardrail Failure [​](https://docs.litellm.ai/release_notes/tags/logging\#traced-guardrail-failure "Direct link to Traced Guardrail Failure") + +![](https://docs.litellm.ai/assets/ideal-img/gd_fail.457338e.1848.png) + +### `/guardrails/list` [​](https://docs.litellm.ai/release_notes/tags/logging\#guardrailslist "Direct link to guardrailslist") + +`/guardrails/list` allows clients to view available guardrails + supported guardrail params + +```codeBlockLines_e6Vv +curl -X GET 'http://0.0.0.0:4000/guardrails/list' + +``` + +Expected response + +```codeBlockLines_e6Vv +{ + "guardrails": [\ + {\ + "guardrail_name": "aporia-post-guard",\ + "guardrail_info": {\ + "params": [\ + {\ + "name": "toxicity_score",\ + "type": "float",\ + "description": "Score between 0-1 indicating content toxicity level"\ + },\ + {\ + "name": "pii_detection",\ + "type": "boolean"\ + }\ + ]\ + }\ + }\ + ] +} + +``` + +### ✨ Guardrails with Mock LLM [​](https://docs.litellm.ai/release_notes/tags/logging\#-guardrails-with-mock-llm "Direct link to ✨ Guardrails with Mock LLM") + +Send `mock_response` to test guardrails without making an LLM call. More info on `mock_response` [here](https://docs.litellm.ai/docs/proxy/guardrails/quick_start) + +```codeBlockLines_e6Vv +curl -i http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [\ + {"role": "user", "content": "hi my email is ishaan@berri.ai"}\ + ], + "mock_response": "This is a mock response", + "guardrails": ["aporia-pre-guard", "aporia-post-guard"] + }' + +``` + +### Assign Keys to Users [​](https://docs.litellm.ai/release_notes/tags/logging\#assign-keys-to-users "Direct link to Assign Keys to Users") + +You can now assign keys to users via Proxy UI + +![](https://docs.litellm.ai/assets/ideal-img/ui_key.9642332.1212.png) + +## New Models [​](https://docs.litellm.ai/release_notes/tags/logging\#new-models "Direct link to New Models") + +- `openrouter/openai/o1` +- `vertex_ai/mistral-large@2411` + +## Fixes [​](https://docs.litellm.ai/release_notes/tags/logging\#fixes "Direct link to Fixes") + +- Fix `vertex_ai/` mistral model pricing: [https://github.com/BerriAI/litellm/pull/7345](https://github.com/BerriAI/litellm/pull/7345) +- Missing model\_group field in logs for aspeech call types [https://github.com/BerriAI/litellm/pull/7392](https://github.com/BerriAI/litellm/pull/7392) + +`key management`, `budgets/rate limits`, `logging`, `guardrails` + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## ✨ Budget / Rate Limit Tiers [​](https://docs.litellm.ai/release_notes/tags/logging\#-budget--rate-limit-tiers "Direct link to ✨ Budget / Rate Limit Tiers") + +Define tiers with rate limits. Assign them to keys. + +Use this to control access and budgets across a lot of keys. + +**[Start here](https://docs.litellm.ai/docs/proxy/rate_limit_tiers)** + +```codeBlockLines_e6Vv +curl -L -X POST 'http://0.0.0.0:4000/budget/new' \ +-H 'Authorization: Bearer sk-1234' \ +-H 'Content-Type: application/json' \ +-d '{ + "budget_id": "high-usage-tier", + "model_max_budget": { + "gpt-4o": {"rpm_limit": 1000000} + } +}' + +``` + +## OTEL Bug Fix [​](https://docs.litellm.ai/release_notes/tags/logging\#otel-bug-fix "Direct link to OTEL Bug Fix") + +LiteLLM was double logging litellm\_request span. This is now fixed. + +[Relevant PR](https://github.com/BerriAI/litellm/pull/7435) + +## Logging for Finetuning Endpoints [​](https://docs.litellm.ai/release_notes/tags/logging\#logging-for-finetuning-endpoints "Direct link to Logging for Finetuning Endpoints") + +Logs for finetuning requests are now available on all logging providers (e.g. Datadog). + +What's logged per request: + +- file\_id +- finetuning\_job\_id +- any key/team metadata + +**Start Here:** + +- [Setup Finetuning](https://docs.litellm.ai/docs/fine_tuning) +- [Setup Logging](https://docs.litellm.ai/docs/proxy/logging#datadog) + +## Dynamic Params for Guardrails [​](https://docs.litellm.ai/release_notes/tags/logging\#dynamic-params-for-guardrails "Direct link to Dynamic Params for Guardrails") + +You can now set custom parameters (like success threshold) for your guardrails in each request. + +[See guardrails spec for more details](https://docs.litellm.ai/docs/proxy/guardrails/custom_guardrail#-pass-additional-parameters-to-guardrail) + +## Management Endpoints Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/management-endpoints#__docusaurus_skipToContent_fallback) + +v1.65.0 updates the `/model/new` endpoint to prevent non-team admins from creating team models. + +This means that only proxy admins or team admins can create team models. + +## Additional Changes [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#additional-changes "Direct link to Additional Changes") + +- Allows team admins to call `/model/update` to update team models. +- Allows team admins to call `/model/delete` to delete team models. +- Introduces new `user_models_only` param to `/v2/model/info` \- only return models added by this user. + +These changes enable team admins to add and manage models for their team on the LiteLLM UI + API. + +![](https://docs.litellm.ai/assets/ideal-img/team_model_add.1ddd404.1251.png) + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +`langfuse`, `management endpoints`, `ui`, `prometheus`, `secret management` + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +Langfuse Prompt Management is being labelled as BETA. This allows us to iterate quickly on the feedback we're receiving, and making the status clearer to users. We expect to make this feature to be stable by next month (February 2025). + +Changes: + +- Include the client message in the LLM API Request. (Previously only the prompt template was sent, and the client message was ignored). +- Log the prompt template in the logged request (e.g. to s3/langfuse). +- Log the 'prompt\_id' and 'prompt\_variables' in the logged request (e.g. to s3/langfuse). + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Team/Organization Management + UI Improvements [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#teamorganization-management--ui-improvements "Direct link to Team/Organization Management + UI Improvements") + +Managing teams and organizations on the UI is now easier. + +Changes: + +- Support for editing user role within team on UI. +- Support updating team member role to admin via api - `/team/member_update` +- Show team admins all keys for their team. +- Add organizations with budgets +- Assign teams to orgs on the UI +- Auto-assign SSO users to teams + +[Start Here](https://docs.litellm.ai/docs/proxy/self_serve) + +## Hashicorp Vault Support [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#hashicorp-vault-support "Direct link to Hashicorp Vault Support") + +We now support writing LiteLLM Virtual API keys to Hashicorp Vault. + +[Start Here](https://docs.litellm.ai/docs/proxy/vault) + +## Custom Prometheus Metrics [​](https://docs.litellm.ai/release_notes/tags/management-endpoints\#custom-prometheus-metrics "Direct link to Custom Prometheus Metrics") + +Define custom prometheus metrics, and track usage/latency/no. of requests against them + +This allows for more fine-grained tracking - e.g. on prompt template passed in request metadata + +[Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## MCP Support Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/mcp#__docusaurus_skipToContent_fallback) + +v1.65.0-stable is live now. Here are the key highlights of this release: + +- **MCP Support**: Support for adding and using MCP servers on the LiteLLM proxy. +- **UI view total usage after 1M+ logs**: You can now view usage analytics after crossing 1M+ logs in DB. + +## Model Context Protocol (MCP) [​](https://docs.litellm.ai/release_notes/tags/mcp\#model-context-protocol-mcp "Direct link to Model Context Protocol (MCP)") + +This release introduces support for centrally adding MCP servers on LiteLLM. This allows you to add MCP server endpoints and your developers can `list` and `call` MCP tools through LiteLLM. + +Read more about MCP [here](https://docs.litellm.ai/docs/mcp). + +![](https://docs.litellm.ai/assets/ideal-img/mcp_ui.4a5216a.1920.png) + +Expose and use MCP servers through LiteLLM + +## UI view total usage after 1M+ logs [​](https://docs.litellm.ai/release_notes/tags/mcp\#ui-view-total-usage-after-1m-logs "Direct link to UI view total usage after 1M+ logs") + +This release brings the ability to view total usage analytics even after exceeding 1M+ logs in your database. We've implemented a scalable architecture that stores only aggregate usage data, resulting in significantly more efficient queries and reduced database CPU utilization. + +![](https://docs.litellm.ai/assets/ideal-img/ui_usage.3ffdba3.1200.png) + +View total usage after 1M+ logs + +- How this works: + + - We now aggregate usage data into a dedicated DailyUserSpend table, significantly reducing query load and CPU usage even beyond 1M+ logs. +- Daily Spend Breakdown API: + + - Retrieve granular daily usage data (by model, provider, and API key) with a single endpoint. + Example Request: + + + + Daily Spend Breakdown API + + + + + + ```codeBlockLines_e6Vv codeBlockLinesWithNumbering_o6Pm + curl -L -X GET 'http://localhost:4000/user/daily/activity?start_date=2025-03-20&end_date=2025-03-27' \ + -H 'Authorization: Bearer sk-...' + + ``` + + + + + + + + + + + + Daily Spend Breakdown API Response + + + + + + ```codeBlockLines_e6Vv codeBlockLinesWithNumbering_o6Pm + { + "results": [\ + {\ + "date": "2025-03-27",\ + "metrics": {\ + "spend": 0.0177072,\ + "prompt_tokens": 111,\ + "completion_tokens": 1711,\ + "total_tokens": 1822,\ + "api_requests": 11\ + },\ + "breakdown": {\ + "models": {\ + "gpt-4o-mini": {\ + "spend": 1.095e-05,\ + "prompt_tokens": 37,\ + "completion_tokens": 9,\ + "total_tokens": 46,\ + "api_requests": 1\ + },\ + "providers": { "openai": { ... }, "azure_ai": { ... } },\ + "api_keys": { "3126b6eaf1...": { ... } }\ + }\ + }\ + ], + "metadata": { + "total_spend": 0.7274667, + "total_prompt_tokens": 280990, + "total_completion_tokens": 376674, + "total_api_requests": 14 + } + } + + ``` + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/mcp\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Support for Vertex AI gemini-2.0-flash-lite & Google AI Studio gemini-2.0-flash-lite [PR](https://github.com/BerriAI/litellm/pull/9523) +- Support for Vertex AI Fine-Tuned LLMs [PR](https://github.com/BerriAI/litellm/pull/9542) +- Nova Canvas image generation support [PR](https://github.com/BerriAI/litellm/pull/9525) +- OpenAI gpt-4o-transcribe support [PR](https://github.com/BerriAI/litellm/pull/9517) +- Added new Vertex AI text embedding model [PR](https://github.com/BerriAI/litellm/pull/9476) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/mcp\#llm-translation "Direct link to LLM Translation") + +- OpenAI Web Search Tool Call Support [PR](https://github.com/BerriAI/litellm/pull/9465) +- Vertex AI topLogprobs support [PR](https://github.com/BerriAI/litellm/pull/9518) +- Support for sending images and video to Vertex AI multimodal embedding [Doc](https://docs.litellm.ai/docs/providers/vertex#multi-modal-embeddings) +- Support litellm.api\_base for Vertex AI + Gemini across completion, embedding, image\_generation [PR](https://github.com/BerriAI/litellm/pull/9516) +- Bug fix for returning `response_cost` when using litellm python SDK with LiteLLM Proxy [PR](https://github.com/BerriAI/litellm/commit/6fd18651d129d606182ff4b980e95768fc43ca3d) +- Support for `max_completion_tokens` on Mistral API [PR](https://github.com/BerriAI/litellm/pull/9606) +- Refactored Vertex AI passthrough routes - fixes unpredictable behaviour with auto-setting default\_vertex\_region on router model add [PR](https://github.com/BerriAI/litellm/pull/9467) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/mcp\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- Log 'api\_base' on spend logs [PR](https://github.com/BerriAI/litellm/pull/9509) +- Support for Gemini audio token cost tracking [PR](https://github.com/BerriAI/litellm/pull/9535) +- Fixed OpenAI audio input token cost tracking [PR](https://github.com/BerriAI/litellm/pull/9535) + +## UI [​](https://docs.litellm.ai/release_notes/tags/mcp\#ui "Direct link to UI") + +### Model Management [​](https://docs.litellm.ai/release_notes/tags/mcp\#model-management "Direct link to Model Management") + +- Allowed team admins to add/update/delete models on UI [PR](https://github.com/BerriAI/litellm/pull/9572) +- Added render supports\_web\_search on model hub [PR](https://github.com/BerriAI/litellm/pull/9469) + +### Request Logs [​](https://docs.litellm.ai/release_notes/tags/mcp\#request-logs "Direct link to Request Logs") + +- Show API base and model ID on request logs [PR](https://github.com/BerriAI/litellm/pull/9572) +- Allow viewing keyinfo on request logs [PR](https://github.com/BerriAI/litellm/pull/9568) + +### Usage Tab [​](https://docs.litellm.ai/release_notes/tags/mcp\#usage-tab "Direct link to Usage Tab") + +- Added Daily User Spend Aggregate view - allows UI Usage tab to work > 1m rows [PR](https://github.com/BerriAI/litellm/pull/9538) +- Connected UI to "LiteLLM\_DailyUserSpend" spend table [PR](https://github.com/BerriAI/litellm/pull/9603) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes/tags/mcp\#logging-integrations "Direct link to Logging Integrations") + +- Fixed StandardLoggingPayload for GCS Pub Sub Logging Integration [PR](https://github.com/BerriAI/litellm/pull/9508) +- Track `litellm_model_name` on `StandardLoggingPayload` [Docs](https://docs.litellm.ai/docs/proxy/logging_spec#standardlogginghiddenparams) + +## Performance / Reliability Improvements [​](https://docs.litellm.ai/release_notes/tags/mcp\#performance--reliability-improvements "Direct link to Performance / Reliability Improvements") + +- LiteLLM Redis semantic caching implementation [PR](https://github.com/BerriAI/litellm/pull/9356) +- Gracefully handle exceptions when DB is having an outage [PR](https://github.com/BerriAI/litellm/pull/9533) +- Allow Pods to startup + passing /health/readiness when allow\_requests\_on\_db\_unavailable: True and DB is down [PR](https://github.com/BerriAI/litellm/pull/9569) + +## General Improvements [​](https://docs.litellm.ai/release_notes/tags/mcp\#general-improvements "Direct link to General Improvements") + +- Support for exposing MCP tools on litellm proxy [PR](https://github.com/BerriAI/litellm/pull/9426) +- Support discovering Gemini, Anthropic, xAI models by calling their /v1/model endpoint [PR](https://github.com/BerriAI/litellm/pull/9530) +- Fixed route check for non-proxy admins on JWT auth [PR](https://github.com/BerriAI/litellm/pull/9454) +- Added baseline Prisma database migrations [PR](https://github.com/BerriAI/litellm/pull/9565) +- View all wildcard models on /model/info [PR](https://github.com/BerriAI/litellm/pull/9572) + +## Security [​](https://docs.litellm.ai/release_notes/tags/mcp\#security "Direct link to Security") + +- Bumped next from 14.2.21 to 14.2.25 in UI dashboard [PR](https://github.com/BerriAI/litellm/pull/9458) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/mcp\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.14-stable.patch1...v1.65.0-stable) + +## LiteLLM New Features +[Skip to main content](https://docs.litellm.ai/release_notes/tags/new-models#__docusaurus_skipToContent_fallback) + +`guardrails`, `logging`, `virtual key management`, `new models` + +info + +Get a 7 day free trial for LiteLLM Enterprise [here](https://litellm.ai/#trial). + +**no call needed** + +## New Features [​](https://docs.litellm.ai/release_notes/tags/new-models\#new-features "Direct link to New Features") + +### ✨ Log Guardrail Traces [​](https://docs.litellm.ai/release_notes/tags/new-models\#-log-guardrail-traces "Direct link to ✨ Log Guardrail Traces") + +Track guardrail failure rate and if a guardrail is going rogue and failing requests. [Start here](https://docs.litellm.ai/docs/proxy/guardrails/quick_start) + +#### Traced Guardrail Success [​](https://docs.litellm.ai/release_notes/tags/new-models\#traced-guardrail-success "Direct link to Traced Guardrail Success") + +![](https://docs.litellm.ai/assets/ideal-img/gd_success.02a2daf.1862.png) + +#### Traced Guardrail Failure [​](https://docs.litellm.ai/release_notes/tags/new-models\#traced-guardrail-failure "Direct link to Traced Guardrail Failure") + +![](https://docs.litellm.ai/assets/ideal-img/gd_fail.457338e.1848.png) + +### `/guardrails/list` [​](https://docs.litellm.ai/release_notes/tags/new-models\#guardrailslist "Direct link to guardrailslist") + +`/guardrails/list` allows clients to view available guardrails + supported guardrail params + +```codeBlockLines_e6Vv +curl -X GET 'http://0.0.0.0:4000/guardrails/list' + +``` + +Expected response + +```codeBlockLines_e6Vv +{ + "guardrails": [\ + {\ + "guardrail_name": "aporia-post-guard",\ + "guardrail_info": {\ + "params": [\ + {\ + "name": "toxicity_score",\ + "type": "float",\ + "description": "Score between 0-1 indicating content toxicity level"\ + },\ + {\ + "name": "pii_detection",\ + "type": "boolean"\ + }\ + ]\ + }\ + }\ + ] +} + +``` + +### ✨ Guardrails with Mock LLM [​](https://docs.litellm.ai/release_notes/tags/new-models\#-guardrails-with-mock-llm "Direct link to ✨ Guardrails with Mock LLM") + +Send `mock_response` to test guardrails without making an LLM call. More info on `mock_response` [here](https://docs.litellm.ai/docs/proxy/guardrails/quick_start) + +```codeBlockLines_e6Vv +curl -i http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer sk-npnwjPQciVRok5yNZgKmFQ" \ + -d '{ + "model": "gpt-3.5-turbo", + "messages": [\ + {"role": "user", "content": "hi my email is ishaan@berri.ai"}\ + ], + "mock_response": "This is a mock response", + "guardrails": ["aporia-pre-guard", "aporia-post-guard"] + }' + +``` + +### Assign Keys to Users [​](https://docs.litellm.ai/release_notes/tags/new-models\#assign-keys-to-users "Direct link to Assign Keys to Users") + +You can now assign keys to users via Proxy UI + +![](https://docs.litellm.ai/assets/ideal-img/ui_key.9642332.1212.png) + +## New Models [​](https://docs.litellm.ai/release_notes/tags/new-models\#new-models "Direct link to New Models") + +- `openrouter/openai/o1` +- `vertex_ai/mistral-large@2411` + +## Fixes [​](https://docs.litellm.ai/release_notes/tags/new-models\#fixes "Direct link to Fixes") + +- Fix `vertex_ai/` mistral model pricing: [https://github.com/BerriAI/litellm/pull/7345](https://github.com/BerriAI/litellm/pull/7345) +- Missing model\_group field in logs for aspeech call types [https://github.com/BerriAI/litellm/pull/7392](https://github.com/BerriAI/litellm/pull/7392) + +A new LiteLLM Stable release [just went out](https://github.com/BerriAI/litellm/releases/tag/v1.55.8-stable). Here are 5 updates since v1.52.2-stable. + +`langfuse`, `fallbacks`, `new models`, `azure_storage` + +![](https://docs.litellm.ai/assets/ideal-img/langfuse_prmpt_mgmt.19b8982.1920.png) + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes/tags/new-models\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +This makes it easy to run experiments or change the specific models `gpt-4o` to `gpt-4o-mini` on Langfuse, instead of making changes in your applications. [Start here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Control fallback prompts client-side [​](https://docs.litellm.ai/release_notes/tags/new-models\#control-fallback-prompts-client-side "Direct link to Control fallback prompts client-side") + +> Claude prompts are different than OpenAI + +Pass in prompts specific to model when doing fallbacks. [Start here](https://docs.litellm.ai/docs/proxy/reliability#control-fallback-prompts) + +## New Providers / Models [​](https://docs.litellm.ai/release_notes/tags/new-models\#new-providers--models "Direct link to New Providers / Models") + +- [NVIDIA Triton](https://developer.nvidia.com/triton-inference-server) `/infer` endpoint. [Start here](https://docs.litellm.ai/docs/providers/triton-inference-server) +- [Infinity](https://github.com/michaelfeil/infinity) Rerank Models [Start here](https://docs.litellm.ai/docs/providers/infinity) + +## ✨ Azure Data Lake Storage Support [​](https://docs.litellm.ai/release_notes/tags/new-models\#-azure-data-lake-storage-support "Direct link to ✨ Azure Data Lake Storage Support") + +Send LLM usage (spend, tokens) data to [Azure Data Lake](https://learn.microsoft.com/en-us/azure/storage/blobs/data-lake-storage-introduction). This makes it easy to consume usage data on other services (eg. Databricks) +[Start here](https://docs.litellm.ai/docs/proxy/logging#azure-blob-storage) + +## Docker Run LiteLLM [​](https://docs.litellm.ai/release_notes/tags/new-models\#docker-run-litellm "Direct link to Docker Run LiteLLM") + +```codeBlockLines_e6Vv +docker run \ +-e STORE_MODEL_IN_DB=True \ +-p 4000:4000 \ +ghcr.io/berriai/litellm:litellm_stable_release_branch-v1.55.8-stable + +``` + +## Get Daily Updates [​](https://docs.litellm.ai/release_notes/tags/new-models\#get-daily-updates "Direct link to Get Daily Updates") + +LiteLLM ships new releases every day. [Follow us on LinkedIn](https://www.linkedin.com/company/berri-ai/) to get daily updates. + +## Prometheus Integration Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/prometheus#__docusaurus_skipToContent_fallback) + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes/tags/prometheus\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/prometheus\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes/tags/prometheus\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes/tags/prometheus\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes/tags/prometheus\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes/tags/prometheus\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes/tags/prometheus\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes/tags/prometheus\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes/tags/prometheus\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes/tags/prometheus\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes/tags/prometheus\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes/tags/prometheus\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +`langfuse`, `management endpoints`, `ui`, `prometheus`, `secret management` + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes/tags/prometheus\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +Langfuse Prompt Management is being labelled as BETA. This allows us to iterate quickly on the feedback we're receiving, and making the status clearer to users. We expect to make this feature to be stable by next month (February 2025). + +Changes: + +- Include the client message in the LLM API Request. (Previously only the prompt template was sent, and the client message was ignored). +- Log the prompt template in the logged request (e.g. to s3/langfuse). +- Log the 'prompt\_id' and 'prompt\_variables' in the logged request (e.g. to s3/langfuse). + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Team/Organization Management + UI Improvements [​](https://docs.litellm.ai/release_notes/tags/prometheus\#teamorganization-management--ui-improvements "Direct link to Team/Organization Management + UI Improvements") + +Managing teams and organizations on the UI is now easier. + +Changes: + +- Support for editing user role within team on UI. +- Support updating team member role to admin via api - `/team/member_update` +- Show team admins all keys for their team. +- Add organizations with budgets +- Assign teams to orgs on the UI +- Auto-assign SSO users to teams + +[Start Here](https://docs.litellm.ai/docs/proxy/self_serve) + +## Hashicorp Vault Support [​](https://docs.litellm.ai/release_notes/tags/prometheus\#hashicorp-vault-support "Direct link to Hashicorp Vault Support") + +We now support writing LiteLLM Virtual API keys to Hashicorp Vault. + +[Start Here](https://docs.litellm.ai/docs/proxy/vault) + +## Custom Prometheus Metrics [​](https://docs.litellm.ai/release_notes/tags/prometheus\#custom-prometheus-metrics "Direct link to Custom Prometheus Metrics") + +Define custom prometheus metrics, and track usage/latency/no. of requests against them + +This allows for more fine-grained tracking - e.g. on prompt template passed in request metadata + +[Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## Prompt Management Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/prompt-management#__docusaurus_skipToContent_fallback) + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes/tags/prompt-management\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +## LLM Translation Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/reasoning-content#__docusaurus_skipToContent_fallback) + +These are the changes since `v1.61.20-stable`. + +This release is primarily focused on: + +- LLM Translation improvements (more `thinking` content improvements) +- UI improvements (Error logs now shown on UI) + +info + +This release will be live on 03/09/2025 + +![](https://docs.litellm.ai/assets/ideal-img/v1632_release.7b42da1.1920.jpg) + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Add `supports_pdf_input` for specific Bedrock Claude models [PR](https://github.com/BerriAI/litellm/commit/f63cf0030679fe1a43d03fb196e815a0f28dae92) +2. Add pricing for amazon `eu` models [PR](https://github.com/BerriAI/litellm/commits/main/model_prices_and_context_window.json) +3. Fix Azure O1 mini pricing [PR](https://github.com/BerriAI/litellm/commit/52de1949ef2f76b8572df751f9c868a016d4832c) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#llm-translation "Direct link to LLM Translation") + +![](https://docs.litellm.ai/assets/ideal-img/anthropic_thinking.3bef9d6.1920.jpg) + +01. Support `/openai/` passthrough for Assistant endpoints. [Get Started](https://docs.litellm.ai/docs/pass_through/openai_passthrough) +02. Bedrock Claude - fix tool calling transformation on invoke route. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---function-calling--tool-calling) +03. Bedrock Claude - response\_format support for claude on invoke route. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---structured-output--json-mode) +04. Bedrock - pass `description` if set in response\_format. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---structured-output--json-mode) +05. Bedrock - Fix passing response\_format: {"type": "text"}. [PR](https://github.com/BerriAI/litellm/commit/c84b489d5897755139aa7d4e9e54727ebe0fa540) +06. OpenAI - Handle sending image\_url as str to openai. [Get Started](https://docs.litellm.ai/docs/completion/vision) +07. Deepseek - return 'reasoning\_content' missing on streaming. [Get Started](https://docs.litellm.ai/docs/reasoning_content) +08. Caching - Support caching on reasoning content. [Get Started](https://docs.litellm.ai/docs/proxy/caching) +09. Bedrock - handle thinking blocks in assistant message. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) +10. Anthropic - Return `signature` on streaming. [Get Started](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) + +- Note: We've also migrated from `signature_delta` to `signature`. [Read more](https://docs.litellm.ai/release_notes/v1.63.0) + +11. Support format param for specifying image type. [Get Started](https://docs.litellm.ai/docs/completion/vision.md#explicitly-specify-image-type) +12. Anthropic - `/v1/messages` endpoint - `thinking` param support. [Get Started](https://docs.litellm.ai/docs/anthropic_unified.md) + +- Note: this refactors the \[BETA\] unified `/v1/messages` endpoint, to just work for the Anthropic API. + +13. Vertex AI - handle $id in response schema when calling vertex ai. [Get Started](https://docs.litellm.ai/docs/providers/vertex#json-schema) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Batches API - Fix cost calculation to run on retrieve\_batch. [Get Started](https://docs.litellm.ai/docs/batches) +2. Batches API - Log batch models in spend logs / standard logging payload. [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec.md#standardlogginghiddenparams) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +![](https://docs.litellm.ai/assets/ideal-img/error_logs.63c5dc9.1920.jpg) + +1. Virtual Keys Page + - Allow team/org filters to be searchable on the Create Key Page + - Add created\_by and updated\_by fields to Keys table + - Show 'user\_email' on key table + - Show 100 Keys Per Page, Use full height, increase width of key alias +2. Logs Page + - Show Error Logs on LiteLLM UI + - Allow Internal Users to View their own logs +3. Internal Users Page + - Allow admin to control default model access for internal users +4. Fix session handling with cookies + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Fix prometheus metrics w/ custom metrics, when keys containing team\_id make requests. [PR](https://github.com/BerriAI/litellm/pull/8935) + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Cooldowns - Support cooldowns on models called with client side credentials. [Get Started](https://docs.litellm.ai/docs/proxy/clientside_auth#pass-user-llm-api-keys--api-base) +2. Tag-based Routing - ensures tag-based routing across all endpoints ( `/embeddings`, `/image_generation`, etc.). [Get Started](https://docs.litellm.ai/docs/proxy/tag_routing) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Raise BadRequestError when unknown model passed in request +2. Enforce model access restrictions on Azure OpenAI proxy route +3. Reliability fix - Handle emoji’s in text - fix orjson error +4. Model Access Patch - don't overwrite litellm.anthropic\_models when running auth checks +5. Enable setting timezone information in docker image + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.61.20-stable...v1.63.2-stable) + +v1.63.0 fixes Anthropic 'thinking' response on streaming to return the `signature` block. [Github Issue](https://github.com/BerriAI/litellm/issues/8964) + +It also moves the response structure from `signature_delta` to `signature` to be the same as Anthropic. [Anthropic Docs](https://docs.anthropic.com/en/docs/build-with-claude/extended-thinking#implementing-extended-thinking) + +## Diff [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#diff "Direct link to Diff") + +```codeBlockLines_e6Vv +"message": { + ... + "reasoning_content": "The capital of France is Paris.", + "thinking_blocks": [\ + {\ + "type": "thinking",\ + "thinking": "The capital of France is Paris.",\ +- "signature_delta": "EqoBCkgIARABGAIiQL2UoU0b1OHYi+..." # 👈 OLD FORMAT\ ++ "signature": "EqoBCkgIARABGAIiQL2UoU0b1OHYi+..." # 👈 KEY CHANGE\ + }\ + ] +} + +``` + +These are the changes since `v1.61.13-stable`. + +This release is primarily focused on: + +- LLM Translation improvements (claude-3-7-sonnet + 'thinking'/'reasoning\_content' support) +- UI improvements (add model flow, user management, etc) + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Anthropic 3-7 sonnet support + cost tracking (Anthropic API + Bedrock + Vertex AI + OpenRouter) +1. Anthropic API [Start here](https://docs.litellm.ai/docs/providers/anthropic#usage---thinking--reasoning_content) +2. Bedrock API [Start here](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) +3. Vertex AI API [See here](https://docs.litellm.ai/docs/providers/vertex#usage---thinking--reasoning_content) +4. OpenRouter [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L5626) +2. Gpt-4.5-preview support + cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L79) +3. Azure AI - Phi-4 cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L1773) +4. Claude-3.5-sonnet - vision support updated on Anthropic API [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2888) +5. Bedrock llama vision support [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L7714) +6. Cerebras llama3.3-70b pricing [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2697) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#llm-translation "Direct link to LLM Translation") + +1. Infinity Rerank - support returning documents when return\_documents=True [Start here](https://docs.litellm.ai/docs/providers/infinity#usage---returning-documents) +2. Amazon Deepseek - `` param extraction into ‘reasoning\_content’ [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-imported-models-deepseek-deepseek-r1) +3. Amazon Titan Embeddings - filter out ‘aws\_’ params from request body [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-embedding) +4. Anthropic ‘thinking’ + ‘reasoning\_content’ translation support (Anthropic API, Bedrock, Vertex AI) [Start here](https://docs.litellm.ai/docs/reasoning_content) +5. VLLM - support ‘video\_url’ [Start here](https://docs.litellm.ai/docs/providers/vllm#send-video-url-to-vllm) +6. Call proxy via litellm SDK: Support `litellm_proxy/` for embedding, image\_generation, transcription, speech, rerank [Start here](https://docs.litellm.ai/docs/providers/litellm_proxy) +7. OpenAI Pass-through - allow using Assistants GET, DELETE on /openai pass through routes [Start here](https://docs.litellm.ai/docs/pass_through/openai_passthrough) +8. Message Translation - fix openai message for assistant msg if role is missing - openai allows this +9. O1/O3 - support ‘drop\_params’ for o3-mini and o1 parallel\_tool\_calls param (not supported currently) [See here](https://docs.litellm.ai/docs/completion/drop_params) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Cost tracking for rerank via Bedrock [See PR](https://github.com/BerriAI/litellm/commit/b682dc4ec8fd07acf2f4c981d2721e36ae2a49c5) +2. Anthropic pass-through - fix race condition causing cost to not be tracked [See PR](https://github.com/BerriAI/litellm/pull/8874) +3. Anthropic pass-through: Ensure accurate token counting [See PR](https://github.com/BerriAI/litellm/pull/8880) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +01. Models Page - Allow sorting models by ‘created at’ +02. Models Page - Edit Model Flow Improvements +03. Models Page - Fix Adding Azure, Azure AI Studio models on UI +04. Internal Users Page - Allow Bulk Adding Internal Users on UI +05. Internal Users Page - Allow sorting users by ‘created at’ +06. Virtual Keys Page - Allow searching for UserIDs on the dropdown when assigning a user to a team [See PR](https://github.com/BerriAI/litellm/pull/8844) +07. Virtual Keys Page - allow creating a user when assigning keys to users [See PR](https://github.com/BerriAI/litellm/pull/8844) +08. Model Hub Page - fix text overflow issue [See PR](https://github.com/BerriAI/litellm/pull/8749) +09. Admin Settings Page - Allow adding MSFT SSO on UI +10. Backend - don't allow creating duplicate internal users in DB + +## Helm [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#helm "Direct link to Helm") + +1. support ttlSecondsAfterFinished on the migration job - [See PR](https://github.com/BerriAI/litellm/pull/8593) +2. enhance migrations job with additional configurable properties - [See PR](https://github.com/BerriAI/litellm/pull/8636) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Arize Phoenix support +2. ‘No-log’ - fix ‘no-log’ param support on embedding calls + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Single Deployment Cooldown logic - Use allowed\_fails or allowed\_fail\_policy if set [Start here](https://docs.litellm.ai/docs/routing#advanced-custom-retries-cooldowns-based-on-error-type) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Hypercorn - fix reading / parsing request body +2. Windows - fix running proxy in windows +3. DD-Trace - fix dd-trace enablement on proxy + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/reasoning-content\#complete-git-diff "Direct link to Complete Git Diff") + +View the complete git diff [here](https://github.com/BerriAI/litellm/compare/v1.61.13-stable...v1.61.20-stable). + +## Release Notes Overview +[Skip to main content](https://docs.litellm.ai/release_notes/tags/rerank#__docusaurus_skipToContent_fallback) + +These are the changes since `v1.61.13-stable`. + +This release is primarily focused on: + +- LLM Translation improvements (claude-3-7-sonnet + 'thinking'/'reasoning\_content' support) +- UI improvements (add model flow, user management, etc) + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/rerank\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/rerank\#new-models--updated-models "Direct link to New Models / Updated Models") + +1. Anthropic 3-7 sonnet support + cost tracking (Anthropic API + Bedrock + Vertex AI + OpenRouter) +1. Anthropic API [Start here](https://docs.litellm.ai/docs/providers/anthropic#usage---thinking--reasoning_content) +2. Bedrock API [Start here](https://docs.litellm.ai/docs/providers/bedrock#usage---thinking--reasoning-content) +3. Vertex AI API [See here](https://docs.litellm.ai/docs/providers/vertex#usage---thinking--reasoning_content) +4. OpenRouter [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L5626) +2. Gpt-4.5-preview support + cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L79) +3. Azure AI - Phi-4 cost tracking [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L1773) +4. Claude-3.5-sonnet - vision support updated on Anthropic API [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2888) +5. Bedrock llama vision support [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L7714) +6. Cerebras llama3.3-70b pricing [See here](https://github.com/BerriAI/litellm/blob/ba5bdce50a0b9bc822de58c03940354f19a733ed/model_prices_and_context_window.json#L2697) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/rerank\#llm-translation "Direct link to LLM Translation") + +1. Infinity Rerank - support returning documents when return\_documents=True [Start here](https://docs.litellm.ai/docs/providers/infinity#usage---returning-documents) +2. Amazon Deepseek - `` param extraction into ‘reasoning\_content’ [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-imported-models-deepseek-deepseek-r1) +3. Amazon Titan Embeddings - filter out ‘aws\_’ params from request body [Start here](https://docs.litellm.ai/docs/providers/bedrock#bedrock-embedding) +4. Anthropic ‘thinking’ + ‘reasoning\_content’ translation support (Anthropic API, Bedrock, Vertex AI) [Start here](https://docs.litellm.ai/docs/reasoning_content) +5. VLLM - support ‘video\_url’ [Start here](https://docs.litellm.ai/docs/providers/vllm#send-video-url-to-vllm) +6. Call proxy via litellm SDK: Support `litellm_proxy/` for embedding, image\_generation, transcription, speech, rerank [Start here](https://docs.litellm.ai/docs/providers/litellm_proxy) +7. OpenAI Pass-through - allow using Assistants GET, DELETE on /openai pass through routes [Start here](https://docs.litellm.ai/docs/pass_through/openai_passthrough) +8. Message Translation - fix openai message for assistant msg if role is missing - openai allows this +9. O1/O3 - support ‘drop\_params’ for o3-mini and o1 parallel\_tool\_calls param (not supported currently) [See here](https://docs.litellm.ai/docs/completion/drop_params) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/rerank\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Cost tracking for rerank via Bedrock [See PR](https://github.com/BerriAI/litellm/commit/b682dc4ec8fd07acf2f4c981d2721e36ae2a49c5) +2. Anthropic pass-through - fix race condition causing cost to not be tracked [See PR](https://github.com/BerriAI/litellm/pull/8874) +3. Anthropic pass-through: Ensure accurate token counting [See PR](https://github.com/BerriAI/litellm/pull/8880) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/rerank\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +01. Models Page - Allow sorting models by ‘created at’ +02. Models Page - Edit Model Flow Improvements +03. Models Page - Fix Adding Azure, Azure AI Studio models on UI +04. Internal Users Page - Allow Bulk Adding Internal Users on UI +05. Internal Users Page - Allow sorting users by ‘created at’ +06. Virtual Keys Page - Allow searching for UserIDs on the dropdown when assigning a user to a team [See PR](https://github.com/BerriAI/litellm/pull/8844) +07. Virtual Keys Page - allow creating a user when assigning keys to users [See PR](https://github.com/BerriAI/litellm/pull/8844) +08. Model Hub Page - fix text overflow issue [See PR](https://github.com/BerriAI/litellm/pull/8749) +09. Admin Settings Page - Allow adding MSFT SSO on UI +10. Backend - don't allow creating duplicate internal users in DB + +## Helm [​](https://docs.litellm.ai/release_notes/tags/rerank\#helm "Direct link to Helm") + +1. support ttlSecondsAfterFinished on the migration job - [See PR](https://github.com/BerriAI/litellm/pull/8593) +2. enhance migrations job with additional configurable properties - [See PR](https://github.com/BerriAI/litellm/pull/8636) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/rerank\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +1. Arize Phoenix support +2. ‘No-log’ - fix ‘no-log’ param support on embedding calls + +## Performance / Loadbalancing / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/rerank\#performance--loadbalancing--reliability-improvements "Direct link to Performance / Loadbalancing / Reliability improvements") + +1. Single Deployment Cooldown logic - Use allowed\_fails or allowed\_fail\_policy if set [Start here](https://docs.litellm.ai/docs/routing#advanced-custom-retries-cooldowns-based-on-error-type) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/rerank\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Hypercorn - fix reading / parsing request body +2. Windows - fix running proxy in windows +3. DD-Trace - fix dd-trace enablement on proxy + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/rerank\#complete-git-diff "Direct link to Complete Git Diff") + +View the complete git diff [here](https://github.com/BerriAI/litellm/compare/v1.61.13-stable...v1.61.20-stable). + +## Responses API Release Notes +[Skip to main content](https://docs.litellm.ai/release_notes/tags/responses-api#__docusaurus_skipToContent_fallback) + +## Deploy this version [​](https://docs.litellm.ai/release_notes/tags/responses-api\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.67.4-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.67.4.post1 + +``` + +## Key Highlights [​](https://docs.litellm.ai/release_notes/tags/responses-api\#key-highlights "Direct link to Key Highlights") + +- **Improved User Management**: This release enables search and filtering across users, keys, teams, and models. +- **Responses API Load Balancing**: Route requests across provider regions and ensure session continuity. +- **UI Session Logs**: Group several requests to LiteLLM into a session. + +## Improved User Management [​](https://docs.litellm.ai/release_notes/tags/responses-api\#improved-user-management "Direct link to Improved User Management") + +![](https://docs.litellm.ai/assets/ideal-img/ui_search_users.7472bdc.1920.png) + +This release makes it easier to manage users and keys on LiteLLM. You can now search and filter across users, keys, teams, and models, and control user settings more easily. + +New features include: + +- Search for users by email, ID, role, or team. +- See all of a user's models, teams, and keys in one place. +- Change user roles and model access right from the Users Tab. + +These changes help you spend less time on user setup and management on LiteLLM. + +## Responses API Load Balancing [​](https://docs.litellm.ai/release_notes/tags/responses-api\#responses-api-load-balancing "Direct link to Responses API Load Balancing") + +![](https://docs.litellm.ai/assets/ideal-img/ui_responses_lb.1e64cec.1204.png) + +This release introduces load balancing for the Responses API, allowing you to route requests across provider regions and ensure session continuity. It works as follows: + +- If a `previous_response_id` is provided, LiteLLM will route the request to the original deployment that generated the prior response — ensuring session continuity. +- If no `previous_response_id` is provided, LiteLLM will load-balance requests across your available deployments. + +[Read more](https://docs.litellm.ai/docs/response_api#load-balancing-with-session-continuity) + +## UI Session Logs [​](https://docs.litellm.ai/release_notes/tags/responses-api\#ui-session-logs "Direct link to UI Session Logs") + +![](https://docs.litellm.ai/assets/ideal-img/ui_session_logs.926dffc.1920.png) + +This release allow you to group requests to LiteLLM proxy into a session. If you specify a litellm\_session\_id in your request LiteLLM will automatically group all logs in the same session. This allows you to easily track usage and request content per session. + +[Read more](https://docs.litellm.ai/docs/proxy/ui_logs_sessions) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/responses-api\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **OpenAI** +1. Added `gpt-image-1` cost tracking [Get Started](https://docs.litellm.ai/docs/image_generation) +2. Bug fix: added cost tracking for gpt-image-1 when quality is unspecified [PR](https://github.com/BerriAI/litellm/pull/10247) +- **Azure** +1. Fixed timestamp granularities passing to whisper in Azure [Get Started](https://docs.litellm.ai/docs/audio_transcription) +2. Added azure/gpt-image-1 pricing [Get Started](https://docs.litellm.ai/docs/image_generation), [PR](https://github.com/BerriAI/litellm/pull/10327) +3. Added cost tracking for `azure/computer-use-preview`, `azure/gpt-4o-audio-preview-2024-12-17`, `azure/gpt-4o-mini-audio-preview-2024-12-17` [PR](https://github.com/BerriAI/litellm/pull/10178) +- **Bedrock** +1. Added support for all compatible Bedrock parameters when model="arn:.." (Bedrock application inference profile models) [Get started](https://docs.litellm.ai/docs/providers/bedrock#bedrock-application-inference-profile), [PR](https://github.com/BerriAI/litellm/pull/10256) +2. Fixed wrong system prompt transformation [PR](https://github.com/BerriAI/litellm/pull/10120) +- **VertexAI / Google AI Studio** +1. Allow setting `budget_tokens=0` for `gemini-2.5-flash` [Get Started](https://docs.litellm.ai/docs/providers/gemini#usage---thinking--reasoning_content), [PR](https://github.com/BerriAI/litellm/pull/10198) +2. Ensure returned `usage` includes thinking token usage [PR](https://github.com/BerriAI/litellm/pull/10198) +3. Added cost tracking for `gemini-2.5-pro-preview-03-25` [PR](https://github.com/BerriAI/litellm/pull/10178) +- **Cohere** +1. Added support for cohere command-a-03-2025 [Get Started](https://docs.litellm.ai/docs/providers/cohere), [PR](https://github.com/BerriAI/litellm/pull/10295) +- **SageMaker** +1. Added support for max\_completion\_tokens parameter [Get Started](https://docs.litellm.ai/docs/providers/sagemaker), [PR](https://github.com/BerriAI/litellm/pull/10300) +- **Responses API** +1. Added support for GET and DELETE operations - `/v1/responses/{response_id}` [Get Started](https://docs.litellm.ai/docs/response_api) +2. Added session management support for non-OpenAI models [PR](https://github.com/BerriAI/litellm/pull/10321) +3. Added routing affinity to maintain model consistency within sessions [Get Started](https://docs.litellm.ai/docs/response_api#load-balancing-with-routing-affinity), [PR](https://github.com/BerriAI/litellm/pull/10193) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/responses-api\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **Bug Fix**: Fixed spend tracking bug, ensuring default litellm params aren't modified in memory [PR](https://github.com/BerriAI/litellm/pull/10167) +- **Deprecation Dates**: Added deprecation dates for Azure, VertexAI models [PR](https://github.com/BerriAI/litellm/pull/10308) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/responses-api\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +#### Users [​](https://docs.litellm.ai/release_notes/tags/responses-api\#users "Direct link to Users") + +- **Filtering and Searching**: + + + - Filter users by user\_id, role, team, sso\_id + - Search users by email + +![](https://docs.litellm.ai/assets/ideal-img/user_filters.e2b4a8c.1920.png) + +- **User Info Panel**: Added a new user information pane [PR](https://github.com/BerriAI/litellm/pull/10213) + + - View teams, keys, models associated with User + - Edit user role, model permissions + +#### Teams [​](https://docs.litellm.ai/release_notes/tags/responses-api\#teams "Direct link to Teams") + +- **Filtering and Searching**: + + + - Filter teams by Organization, Team ID [PR](https://github.com/BerriAI/litellm/pull/10324) + - Search teams by Team Name [PR](https://github.com/BerriAI/litellm/pull/10324) + +![](https://docs.litellm.ai/assets/ideal-img/team_filters.c9c085b.1920.png) + +#### Keys [​](https://docs.litellm.ai/release_notes/tags/responses-api\#keys "Direct link to Keys") + +- **Key Management**: + - Support for cross-filtering and filtering by key hash [PR](https://github.com/BerriAI/litellm/pull/10322) + - Fixed key alias reset when resetting filters [PR](https://github.com/BerriAI/litellm/pull/10099) + - Fixed table rendering on key creation [PR](https://github.com/BerriAI/litellm/pull/10224) + +#### UI Logs Page [​](https://docs.litellm.ai/release_notes/tags/responses-api\#ui-logs-page "Direct link to UI Logs Page") + +- **Session Logs**: Added UI Session Logs [Get Started](https://docs.litellm.ai/docs/proxy/ui_logs_sessions) + +#### UI Authentication & Security [​](https://docs.litellm.ai/release_notes/tags/responses-api\#ui-authentication--security "Direct link to UI Authentication & Security") + +- **Required Authentication**: Authentication now required for all dashboard pages [PR](https://github.com/BerriAI/litellm/pull/10229) +- **SSO Fixes**: Fixed SSO user login invalid token error [PR](https://github.com/BerriAI/litellm/pull/10298) +- \[BETA\] **Encrypted Tokens**: Moved UI to encrypted token usage [PR](https://github.com/BerriAI/litellm/pull/10302) +- **Token Expiry**: Support token refresh by re-routing to login page (fixes issue where expired token would show a blank page) [PR](https://github.com/BerriAI/litellm/pull/10250) + +#### UI General fixes [​](https://docs.litellm.ai/release_notes/tags/responses-api\#ui-general-fixes "Direct link to UI General fixes") + +- **Fixed UI Flicker**: Addressed UI flickering issues in Dashboard [PR](https://github.com/BerriAI/litellm/pull/10261) +- **Improved Terminology**: Better loading and no-data states on Keys and Tools pages [PR](https://github.com/BerriAI/litellm/pull/10253) +- **Azure Model Support**: Fixed editing Azure public model names and changing model names after creation [PR](https://github.com/BerriAI/litellm/pull/10249) +- **Team Model Selector**: Bug fix for team model selection [PR](https://github.com/BerriAI/litellm/pull/10171) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/responses-api\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +- **Datadog**: +1. Fixed Datadog LLM observability logging [Get Started](https://docs.litellm.ai/docs/proxy/logging#datadog), [PR](https://github.com/BerriAI/litellm/pull/10206) +- **Prometheus / Grafana**: +1. Enable datasource selection on LiteLLM Grafana Template [Get Started](https://docs.litellm.ai/docs/proxy/prometheus#-litellm-maintained-grafana-dashboards-), [PR](https://github.com/BerriAI/litellm/pull/10257) +- **AgentOps**: +1. Added AgentOps Integration [Get Started](https://docs.litellm.ai/docs/observability/agentops_integration), [PR](https://github.com/BerriAI/litellm/pull/9685) +- **Arize**: +1. Added missing attributes for Arize & Phoenix Integration [Get Started](https://docs.litellm.ai/docs/observability/arize_integration), [PR](https://github.com/BerriAI/litellm/pull/10215) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/responses-api\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Caching**: Fixed caching to account for `thinking` or `reasoning_effort` when calculating cache key [PR](https://github.com/BerriAI/litellm/pull/10140) +- **Model Groups**: Fixed handling for cases where user sets model\_group inside model\_info [PR](https://github.com/BerriAI/litellm/pull/10191) +- **Passthrough Endpoints**: Ensured `PassthroughStandardLoggingPayload` is logged with method, URL, request/response body [PR](https://github.com/BerriAI/litellm/pull/10194) +- **Fix SQL Injection**: Fixed potential SQL injection vulnerability in spend\_management\_endpoints.py [PR](https://github.com/BerriAI/litellm/pull/9878) + +## Helm [​](https://docs.litellm.ai/release_notes/tags/responses-api\#helm "Direct link to Helm") + +- Fixed serviceAccountName on migration job [PR](https://github.com/BerriAI/litellm/pull/10258) + +## Full Changelog [​](https://docs.litellm.ai/release_notes/tags/responses-api\#full-changelog "Direct link to Full Changelog") + +The complete list of changes can be found in the [GitHub release notes](https://github.com/BerriAI/litellm/compare/v1.67.0-stable...v1.67.4-stable). + +These are the changes since `v1.63.11-stable`. + +This release brings: + +- LLM Translation Improvements (MCP Support and Bedrock Application Profiles) +- Perf improvements for Usage-based Routing +- Streaming guardrail support via websockets +- Azure OpenAI client perf fix (from previous release) + +## Docker Run LiteLLM Proxy [​](https://docs.litellm.ai/release_notes/tags/responses-api\#docker-run-litellm-proxy "Direct link to Docker Run LiteLLM Proxy") + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.63.14-stable.patch1 + +``` + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/responses-api\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/responses-api\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Azure gpt-4o - fixed pricing to latest global pricing - [PR](https://github.com/BerriAI/litellm/pull/9361) +- O1-Pro - add pricing + model information - [PR](https://github.com/BerriAI/litellm/pull/9397) +- Azure AI - mistral 3.1 small pricing added - [PR](https://github.com/BerriAI/litellm/pull/9453) +- Azure - gpt-4.5-preview pricing added - [PR](https://github.com/BerriAI/litellm/pull/9453) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/responses-api\#llm-translation "Direct link to LLM Translation") + +1. **New LLM Features** + +- Bedrock: Support bedrock application inference profiles [Docs](https://docs.litellm.ai/docs/providers/bedrock#bedrock-application-inference-profile) + - Infer aws region from bedrock application profile id - ( `arn:aws:bedrock:us-east-1:...`) +- Ollama - support calling via `/v1/completions` [Get Started](https://docs.litellm.ai/docs/providers/ollama#using-ollama-fim-on-v1completions) +- Bedrock - support `us.deepseek.r1-v1:0` model name [Docs](https://docs.litellm.ai/docs/providers/bedrock#supported-aws-bedrock-models) +- OpenRouter - `OPENROUTER_API_BASE` env var support [Docs](https://docs.litellm.ai/docs/providers/openrouter.md) +- Azure - add audio model parameter support - [Docs](https://docs.litellm.ai/docs/providers/azure#azure-audio-model) +- OpenAI - PDF File support [Docs](https://docs.litellm.ai/docs/completion/document_understanding#openai-file-message-type) +- OpenAI - o1-pro Responses API streaming support [Docs](https://docs.litellm.ai/docs/response_api.md#streaming) +- \[BETA\] MCP - Use MCP Tools with LiteLLM SDK [Docs](https://docs.litellm.ai/docs/mcp) + +2. **Bug Fixes** + +- Voyage: prompt token on embedding tracking fix - [PR](https://github.com/BerriAI/litellm/commit/56d3e75b330c3c3862dc6e1c51c1210e48f1068e) +- Sagemaker - Fix ‘Too little data for declared Content-Length’ error - [PR](https://github.com/BerriAI/litellm/pull/9326) +- OpenAI-compatible models - fix issue when calling openai-compatible models w/ custom\_llm\_provider set - [PR](https://github.com/BerriAI/litellm/pull/9355) +- VertexAI - Embedding ‘outputDimensionality’ support - [PR](https://github.com/BerriAI/litellm/commit/437dbe724620675295f298164a076cbd8019d304) +- Anthropic - return consistent json response format on streaming/non-streaming - [PR](https://github.com/BerriAI/litellm/pull/9437) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/responses-api\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- `litellm_proxy/` \- support reading litellm response cost header from proxy, when using client sdk +- Reset Budget Job - fix budget reset error on keys/teams/users [PR](https://github.com/BerriAI/litellm/pull/9329) +- Streaming - Prevents final chunk w/ usage from being ignored (impacted bedrock streaming + cost tracking) [PR](https://github.com/BerriAI/litellm/pull/9314) + +## UI [​](https://docs.litellm.ai/release_notes/tags/responses-api\#ui "Direct link to UI") + +1. Users Page + - Feature: Control default internal user settings [PR](https://github.com/BerriAI/litellm/pull/9328) +2. Icons: + - Feature: Replace external "artificialanalysis.ai" icons by local svg [PR](https://github.com/BerriAI/litellm/pull/9374) +3. Sign In/Sign Out + - Fix: Default login when `default_user_id` user does not exist in DB [PR](https://github.com/BerriAI/litellm/pull/9395) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes/tags/responses-api\#logging-integrations "Direct link to Logging Integrations") + +- Support post-call guardrails for streaming responses [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/custom_guardrail#1-write-a-customguardrail-class) +- Arize [Get Started](https://docs.litellm.ai/docs/observability/arize_integration) + - fix invalid package import [PR](https://github.com/BerriAI/litellm/pull/9338) + - migrate to using standardloggingpayload for metadata, ensures spans land successfully [PR](https://github.com/BerriAI/litellm/pull/9338) + - fix logging to just log the LLM I/O [PR](https://github.com/BerriAI/litellm/pull/9353) + - Dynamic API Key/Space param support [Get Started](https://docs.litellm.ai/docs/observability/arize_integration#pass-arize-spacekey-per-request) +- StandardLoggingPayload - Log litellm\_model\_name in payload. Allows knowing what the model sent to API provider was [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec#standardlogginghiddenparams) +- Prompt Management - Allow building custom prompt management integration [Get Started](https://docs.litellm.ai/docs/proxy/custom_prompt_management.md) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/responses-api\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +- Redis Caching - add 5s default timeout, prevents hanging redis connection from impacting llm calls [PR](https://github.com/BerriAI/litellm/commit/db92956ae33ed4c4e3233d7e1b0c7229817159bf) +- Allow disabling all spend updates / writes to DB - patch to allow disabling all spend updates to DB with a flag [PR](https://github.com/BerriAI/litellm/pull/9331) +- Azure OpenAI - correctly re-use azure openai client, fixes perf issue from previous Stable release [PR](https://github.com/BerriAI/litellm/commit/f2026ef907c06d94440930917add71314b901413) +- Azure OpenAI - uses litellm.ssl\_verify on Azure/OpenAI clients [PR](https://github.com/BerriAI/litellm/commit/f2026ef907c06d94440930917add71314b901413) +- Usage-based routing - Wildcard model support [Get Started](https://docs.litellm.ai/docs/proxy/usage_based_routing#wildcard-model-support) +- Usage-based routing - Support batch writing increments to redis - reduces latency to same as ‘simple-shuffle’ [PR](https://github.com/BerriAI/litellm/pull/9357) +- Router - show reason for model cooldown on ‘no healthy deployments available error’ [PR](https://github.com/BerriAI/litellm/pull/9438) +- Caching - add max value limit to an item in in-memory cache (1MB) - prevents OOM errors on large image url’s being sent through proxy [PR](https://github.com/BerriAI/litellm/pull/9448) + +## General Improvements [​](https://docs.litellm.ai/release_notes/tags/responses-api\#general-improvements "Direct link to General Improvements") + +- Passthrough Endpoints - support returning api-base on pass-through endpoints Response Headers [Docs](https://docs.litellm.ai/docs/proxy/response_headers#litellm-specific-headers) +- SSL - support reading ssl security level from env var - Allows user to specify lower security settings [Get Started](https://docs.litellm.ai/docs/guides/security_settings) +- Credentials - only poll Credentials table when `STORE_MODEL_IN_DB` is True [PR](https://github.com/BerriAI/litellm/pull/9376) +- Image URL Handling - new architecture doc on image url handling [Docs](https://docs.litellm.ai/docs/proxy/image_handling) +- OpenAI - bump to pip install "openai==1.68.2" [PR](https://github.com/BerriAI/litellm/commit/e85e3bc52a9de86ad85c3dbb12d87664ee567a5a) +- Gunicorn - security fix - bump gunicorn==23.0.0 [PR](https://github.com/BerriAI/litellm/commit/7e9fc92f5c7fea1e7294171cd3859d55384166eb) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/responses-api\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.11-stable...v1.63.14.rc) + +These are the changes since `v1.63.2-stable`. + +This release is primarily focused on: + +- \[Beta\] Responses API Support +- Snowflake Cortex Support, Amazon Nova Image Generation +- UI - Credential Management, re-use credentials when adding new models +- UI - Test Connection to LLM Provider before adding a model + +## Known Issues [​](https://docs.litellm.ai/release_notes/tags/responses-api\#known-issues "Direct link to Known Issues") + +- 🚨 Known issue on Azure OpenAI - We don't recommend upgrading if you use Azure OpenAI. This version failed our Azure OpenAI load test + +## Docker Run LiteLLM Proxy [​](https://docs.litellm.ai/release_notes/tags/responses-api\#docker-run-litellm-proxy "Direct link to Docker Run LiteLLM Proxy") + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.63.11-stable + +``` + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/responses-api\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/responses-api\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Image Generation support for Amazon Nova Canvas [Getting Started](https://docs.litellm.ai/docs/providers/bedrock#image-generation) +- Add pricing for Jamba new models [PR](https://github.com/BerriAI/litellm/pull/9032/files) +- Add pricing for Amazon EU models [PR](https://github.com/BerriAI/litellm/pull/9056/files) +- Add Bedrock Deepseek R1 model pricing [PR](https://github.com/BerriAI/litellm/pull/9108/files) +- Update Gemini pricing: Gemma 3, Flash 2 thinking update, LearnLM [PR](https://github.com/BerriAI/litellm/pull/9190/files) +- Mark Cohere Embedding 3 models as Multimodal [PR](https://github.com/BerriAI/litellm/pull/9176/commits/c9a576ce4221fc6e50dc47cdf64ab62736c9da41) +- Add Azure Data Zone pricing [PR](https://github.com/BerriAI/litellm/pull/9185/files#diff-19ad91c53996e178c1921cbacadf6f3bae20cfe062bd03ee6bfffb72f847ee37) + - LiteLLM Tracks cost for `azure/eu` and `azure/us` models + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/responses-api\#llm-translation "Direct link to LLM Translation") + +![](https://docs.litellm.ai/assets/ideal-img/responses_api.01dd45d.1200.png) + +1. **New Endpoints** + +- \[Beta\] POST `/responses` API. [Getting Started](https://docs.litellm.ai/docs/response_api) + +2. **New LLM Providers** + +- Snowflake Cortex [Getting Started](https://docs.litellm.ai/docs/providers/snowflake) + +3. **New LLM Features** + +- Support OpenRouter `reasoning_content` on streaming [Getting Started](https://docs.litellm.ai/docs/reasoning_content) + +4. **Bug Fixes** + +- OpenAI: Return `code`, `param` and `type` on bad request error [More information on litellm exceptions](https://docs.litellm.ai/docs/exception_mapping) +- Bedrock: Fix converse chunk parsing to only return empty dict on tool use [PR](https://github.com/BerriAI/litellm/pull/9166) +- Bedrock: Support extra\_headers [PR](https://github.com/BerriAI/litellm/pull/9113) +- Azure: Fix Function Calling Bug & Update Default API Version to `2025-02-01-preview` [PR](https://github.com/BerriAI/litellm/pull/9191) +- Azure: Fix AI services URL [PR](https://github.com/BerriAI/litellm/pull/9185) +- Vertex AI: Handle HTTP 201 status code in response [PR](https://github.com/BerriAI/litellm/pull/9193) +- Perplexity: Fix incorrect streaming response [PR](https://github.com/BerriAI/litellm/pull/9081) +- Triton: Fix streaming completions bug [PR](https://github.com/BerriAI/litellm/pull/8386) +- Deepgram: Support bytes.IO when handling audio files for transcription [PR](https://github.com/BerriAI/litellm/pull/9071) +- Ollama: Fix "system" role has become unacceptable [PR](https://github.com/BerriAI/litellm/pull/9261) +- All Providers (Streaming): Fix String `data:` stripped from entire content in streamed responses [PR](https://github.com/BerriAI/litellm/pull/9070) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/responses-api\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Support Bedrock converse cache token tracking [Getting Started](https://docs.litellm.ai/docs/completion/prompt_caching) +2. Cost Tracking for Responses API [Getting Started](https://docs.litellm.ai/docs/response_api) +3. Fix Azure Whisper cost tracking [Getting Started](https://docs.litellm.ai/docs/audio_transcription) + +## UI [​](https://docs.litellm.ai/release_notes/tags/responses-api\#ui "Direct link to UI") + +### Re-Use Credentials on UI [​](https://docs.litellm.ai/release_notes/tags/responses-api\#re-use-credentials-on-ui "Direct link to Re-Use Credentials on UI") + +You can now onboard LLM provider credentials on LiteLLM UI. Once these credentials are added you can re-use them when adding new models [Getting Started](https://docs.litellm.ai/docs/proxy/ui_credentials) + +![](https://docs.litellm.ai/assets/ideal-img/credentials.8f19ffb.1920.jpg) + +### Test Connections before adding models [​](https://docs.litellm.ai/release_notes/tags/responses-api\#test-connections-before-adding-models "Direct link to Test Connections before adding models") + +Before adding a model you can test the connection to the LLM provider to verify you have setup your API Base + API Key correctly + +![](https://docs.litellm.ai/assets/images/litellm_test_connection-029765a2de4dcabccfe3be9a8d33dbdd.gif) + +### General UI Improvements [​](https://docs.litellm.ai/release_notes/tags/responses-api\#general-ui-improvements "Direct link to General UI Improvements") + +1. Add Models Page + - Allow adding Cerebras, Sambanova, Perplexity, Fireworks, Openrouter, TogetherAI Models, Text-Completion OpenAI on Admin UI + - Allow adding EU OpenAI models + - Fix: Instantly show edit + deletes to models +2. Keys Page + - Fix: Instantly show newly created keys on Admin UI (don't require refresh) + - Fix: Allow clicking into Top Keys when showing users Top API Key + - Fix: Allow Filter Keys by Team Alias, Key Alias and Org + - UI Improvements: Show 100 Keys Per Page, Use full height, increase width of key alias +3. Users Page + - Fix: Show correct count of internal user keys on Users Page + - Fix: Metadata not updating in Team UI +4. Logs Page + - UI Improvements: Keep expanded log in focus on LiteLLM UI + - UI Improvements: Minor improvements to logs page + - Fix: Allow internal user to query their own logs + - Allow switching off storing Error Logs in DB [Getting Started](https://docs.litellm.ai/docs/proxy/ui_logs) +5. Sign In/Sign Out + - Fix: Correctly use `PROXY_LOGOUT_URL` when set [Getting Started](https://docs.litellm.ai/docs/proxy/self_serve#setting-custom-logout-urls) + +## Security [​](https://docs.litellm.ai/release_notes/tags/responses-api\#security "Direct link to Security") + +1. Support for Rotating Master Keys [Getting Started](https://docs.litellm.ai/docs/proxy/master_key_rotations) +2. Fix: Internal User Viewer Permissions, don't allow `internal_user_viewer` role to see `Test Key Page` or `Create Key Button` [More information on role based access controls](https://docs.litellm.ai/docs/proxy/access_control) +3. Emit audit logs on All user + model Create/Update/Delete endpoints [Getting Started](https://docs.litellm.ai/docs/proxy/multiple_admins) +4. JWT + - Support multiple JWT OIDC providers [Getting Started](https://docs.litellm.ai/docs/proxy/token_auth) + - Fix JWT access with Groups not working when team is assigned All Proxy Models access +5. Using K/V pairs in 1 AWS Secret [Getting Started](https://docs.litellm.ai/docs/secret#using-kv-pairs-in-1-aws-secret) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes/tags/responses-api\#logging-integrations "Direct link to Logging Integrations") + +1. Prometheus: Track Azure LLM API latency metric [Getting Started](https://docs.litellm.ai/docs/proxy/prometheus#request-latency-metrics) +2. Athina: Added tags, user\_feedback and model\_options to additional\_keys which can be sent to Athina [Getting Started](https://docs.litellm.ai/docs/observability/athina_integration) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/responses-api\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +1. Redis + litellm router - Fix Redis cluster mode for litellm router [PR](https://github.com/BerriAI/litellm/pull/9010) + +## General Improvements [​](https://docs.litellm.ai/release_notes/tags/responses-api\#general-improvements "Direct link to General Improvements") + +1. OpenWebUI Integration - display `thinking` tokens + +- Guide on getting started with LiteLLM x OpenWebUI. [Getting Started](https://docs.litellm.ai/docs/tutorials/openweb_ui) +- Display `thinking` tokens on OpenWebUI (Bedrock, Anthropic, Deepseek) [Getting Started](https://docs.litellm.ai/docs/tutorials/openweb_ui#render-thinking-content-on-openweb-ui) + +![](https://docs.litellm.ai/assets/images/litellm_thinking_openweb-5ec7dddb7e7b6a10252694c27cfc177d.gif) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/responses-api\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.2-stable...v1.63.11-stable) + +## Secret Management Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/secret-management#__docusaurus_skipToContent_fallback) + +`alerting`, `prometheus`, `secret management`, `management endpoints`, `ui`, `prompt management`, `finetuning`, `batch` + +## New / Updated Models [​](https://docs.litellm.ai/release_notes/tags/secret-management\#new--updated-models "Direct link to New / Updated Models") + +1. Mistral large pricing - [https://github.com/BerriAI/litellm/pull/7452](https://github.com/BerriAI/litellm/pull/7452) +2. Cohere command-r7b-12-2024 pricing - [https://github.com/BerriAI/litellm/pull/7553/files](https://github.com/BerriAI/litellm/pull/7553/files) +3. Voyage - new models, prices and context window information - [https://github.com/BerriAI/litellm/pull/7472](https://github.com/BerriAI/litellm/pull/7472) +4. Anthropic - bump Bedrock claude-3-5-haiku max\_output\_tokens to 8192 + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/secret-management\#general-proxy-improvements "Direct link to General Proxy Improvements") + +1. Health check support for realtime models +2. Support calling Azure realtime routes via virtual keys +3. Support custom tokenizer on `/utils/token_counter` \- useful when checking token count for self-hosted models +4. Request Prioritization - support on `/v1/completion` endpoint as well + +## LLM Translation Improvements [​](https://docs.litellm.ai/release_notes/tags/secret-management\#llm-translation-improvements "Direct link to LLM Translation Improvements") + +1. Deepgram STT support. [Start Here](https://docs.litellm.ai/docs/providers/deepgram) +2. OpenAI Moderations - `omni-moderation-latest` support. [Start Here](https://docs.litellm.ai/docs/moderation) +3. Azure O1 - fake streaming support. This ensures if a `stream=true` is passed, the response is streamed. [Start Here](https://docs.litellm.ai/docs/providers/azure) +4. Anthropic - non-whitespace char stop sequence handling - [PR](https://github.com/BerriAI/litellm/pull/7484) +5. Azure OpenAI - support Entra ID username + password based auth. [Start Here](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret) +6. LM Studio - embedding route support. [Start Here](https://docs.litellm.ai/docs/providers/lm-studio) +7. WatsonX - ZenAPIKeyAuth support. [Start Here](https://docs.litellm.ai/docs/providers/watsonx) + +## Prompt Management Improvements [​](https://docs.litellm.ai/release_notes/tags/secret-management\#prompt-management-improvements "Direct link to Prompt Management Improvements") + +1. Langfuse integration +2. HumanLoop integration +3. Support for using load balanced models +4. Support for loading optional params from prompt manager + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Finetuning + Batch APIs Improvements [​](https://docs.litellm.ai/release_notes/tags/secret-management\#finetuning--batch-apis-improvements "Direct link to Finetuning + Batch APIs Improvements") + +1. Improved unified endpoint support for Vertex AI finetuning - [PR](https://github.com/BerriAI/litellm/pull/7487) +2. Add support for retrieving vertex api batch jobs - [PR](https://github.com/BerriAI/litellm/commit/13f364682d28a5beb1eb1b57f07d83d5ef50cbdc) + +## _NEW_ Alerting Integration [​](https://docs.litellm.ai/release_notes/tags/secret-management\#new-alerting-integration "Direct link to new-alerting-integration") + +PagerDuty Alerting Integration. + +Handles two types of alerts: + +- High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. +- High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +[Start Here](https://docs.litellm.ai/docs/proxy/pagerduty) + +## Prometheus Improvements [​](https://docs.litellm.ai/release_notes/tags/secret-management\#prometheus-improvements "Direct link to Prometheus Improvements") + +Added support for tracking latency/spend/tokens based on custom metrics. [Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## _NEW_ Hashicorp Secret Manager Support [​](https://docs.litellm.ai/release_notes/tags/secret-management\#new-hashicorp-secret-manager-support "Direct link to new-hashicorp-secret-manager-support") + +Support for reading credentials + writing LLM API keys. [Start Here](https://docs.litellm.ai/docs/secret#hashicorp-vault) + +## Management Endpoints / UI Improvements [​](https://docs.litellm.ai/release_notes/tags/secret-management\#management-endpoints--ui-improvements "Direct link to Management Endpoints / UI Improvements") + +1. Create and view organizations + assign org admins on the Proxy UI +2. Support deleting keys by key\_alias +3. Allow assigning teams to org on UI +4. Disable using ui session token for 'test key' pane +5. Show model used in 'test key' pane +6. Support markdown output in 'test key' pane + +## Helm Improvements [​](https://docs.litellm.ai/release_notes/tags/secret-management\#helm-improvements "Direct link to Helm Improvements") + +1. Prevent istio injection for db migrations cron job +2. allow using migrationJob.enabled variable within job + +## Logging Improvements [​](https://docs.litellm.ai/release_notes/tags/secret-management\#logging-improvements "Direct link to Logging Improvements") + +1. braintrust logging: respect project\_id, add more metrics - [https://github.com/BerriAI/litellm/pull/7613](https://github.com/BerriAI/litellm/pull/7613) +2. Athina - support base url - `ATHINA_BASE_URL` +3. Lunary - Allow passing custom parent run id to LLM Calls + +## Git Diff [​](https://docs.litellm.ai/release_notes/tags/secret-management\#git-diff "Direct link to Git Diff") + +This is the diff between v1.56.3-stable and v1.57.8-stable. + +Use this to see the changes in the codebase. + +[Git Diff](https://github.com/BerriAI/litellm/compare/v1.56.3-stable...189b67760011ea313ca58b1f8bd43aa74fbd7f55) + +`langfuse`, `management endpoints`, `ui`, `prometheus`, `secret management` + +## Langfuse Prompt Management [​](https://docs.litellm.ai/release_notes/tags/secret-management\#langfuse-prompt-management "Direct link to Langfuse Prompt Management") + +Langfuse Prompt Management is being labelled as BETA. This allows us to iterate quickly on the feedback we're receiving, and making the status clearer to users. We expect to make this feature to be stable by next month (February 2025). + +Changes: + +- Include the client message in the LLM API Request. (Previously only the prompt template was sent, and the client message was ignored). +- Log the prompt template in the logged request (e.g. to s3/langfuse). +- Log the 'prompt\_id' and 'prompt\_variables' in the logged request (e.g. to s3/langfuse). + +[Start Here](https://docs.litellm.ai/docs/proxy/prompt_management) + +## Team/Organization Management + UI Improvements [​](https://docs.litellm.ai/release_notes/tags/secret-management\#teamorganization-management--ui-improvements "Direct link to Team/Organization Management + UI Improvements") + +Managing teams and organizations on the UI is now easier. + +Changes: + +- Support for editing user role within team on UI. +- Support updating team member role to admin via api - `/team/member_update` +- Show team admins all keys for their team. +- Add organizations with budgets +- Assign teams to orgs on the UI +- Auto-assign SSO users to teams + +[Start Here](https://docs.litellm.ai/docs/proxy/self_serve) + +## Hashicorp Vault Support [​](https://docs.litellm.ai/release_notes/tags/secret-management\#hashicorp-vault-support "Direct link to Hashicorp Vault Support") + +We now support writing LiteLLM Virtual API keys to Hashicorp Vault. + +[Start Here](https://docs.litellm.ai/docs/proxy/vault) + +## Custom Prometheus Metrics [​](https://docs.litellm.ai/release_notes/tags/secret-management\#custom-prometheus-metrics "Direct link to Custom Prometheus Metrics") + +Define custom prometheus metrics, and track usage/latency/no. of requests against them + +This allows for more fine-grained tracking - e.g. on prompt template passed in request metadata + +[Start Here](https://docs.litellm.ai/docs/proxy/prometheus#beta-custom-metrics) + +## LiteLLM Security Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/security#__docusaurus_skipToContent_fallback) + +## Deploy this version [​](https://docs.litellm.ai/release_notes/tags/security\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.67.4-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.67.4.post1 + +``` + +## Key Highlights [​](https://docs.litellm.ai/release_notes/tags/security\#key-highlights "Direct link to Key Highlights") + +- **Improved User Management**: This release enables search and filtering across users, keys, teams, and models. +- **Responses API Load Balancing**: Route requests across provider regions and ensure session continuity. +- **UI Session Logs**: Group several requests to LiteLLM into a session. + +## Improved User Management [​](https://docs.litellm.ai/release_notes/tags/security\#improved-user-management "Direct link to Improved User Management") + +![](https://docs.litellm.ai/assets/ideal-img/ui_search_users.7472bdc.1920.png) + +This release makes it easier to manage users and keys on LiteLLM. You can now search and filter across users, keys, teams, and models, and control user settings more easily. + +New features include: + +- Search for users by email, ID, role, or team. +- See all of a user's models, teams, and keys in one place. +- Change user roles and model access right from the Users Tab. + +These changes help you spend less time on user setup and management on LiteLLM. + +## Responses API Load Balancing [​](https://docs.litellm.ai/release_notes/tags/security\#responses-api-load-balancing "Direct link to Responses API Load Balancing") + +![](https://docs.litellm.ai/assets/ideal-img/ui_responses_lb.1e64cec.1204.png) + +This release introduces load balancing for the Responses API, allowing you to route requests across provider regions and ensure session continuity. It works as follows: + +- If a `previous_response_id` is provided, LiteLLM will route the request to the original deployment that generated the prior response — ensuring session continuity. +- If no `previous_response_id` is provided, LiteLLM will load-balance requests across your available deployments. + +[Read more](https://docs.litellm.ai/docs/response_api#load-balancing-with-session-continuity) + +## UI Session Logs [​](https://docs.litellm.ai/release_notes/tags/security\#ui-session-logs "Direct link to UI Session Logs") + +![](https://docs.litellm.ai/assets/ideal-img/ui_session_logs.926dffc.1920.png) + +This release allow you to group requests to LiteLLM proxy into a session. If you specify a litellm\_session\_id in your request LiteLLM will automatically group all logs in the same session. This allows you to easily track usage and request content per session. + +[Read more](https://docs.litellm.ai/docs/proxy/ui_logs_sessions) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/security\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **OpenAI** +1. Added `gpt-image-1` cost tracking [Get Started](https://docs.litellm.ai/docs/image_generation) +2. Bug fix: added cost tracking for gpt-image-1 when quality is unspecified [PR](https://github.com/BerriAI/litellm/pull/10247) +- **Azure** +1. Fixed timestamp granularities passing to whisper in Azure [Get Started](https://docs.litellm.ai/docs/audio_transcription) +2. Added azure/gpt-image-1 pricing [Get Started](https://docs.litellm.ai/docs/image_generation), [PR](https://github.com/BerriAI/litellm/pull/10327) +3. Added cost tracking for `azure/computer-use-preview`, `azure/gpt-4o-audio-preview-2024-12-17`, `azure/gpt-4o-mini-audio-preview-2024-12-17` [PR](https://github.com/BerriAI/litellm/pull/10178) +- **Bedrock** +1. Added support for all compatible Bedrock parameters when model="arn:.." (Bedrock application inference profile models) [Get started](https://docs.litellm.ai/docs/providers/bedrock#bedrock-application-inference-profile), [PR](https://github.com/BerriAI/litellm/pull/10256) +2. Fixed wrong system prompt transformation [PR](https://github.com/BerriAI/litellm/pull/10120) +- **VertexAI / Google AI Studio** +1. Allow setting `budget_tokens=0` for `gemini-2.5-flash` [Get Started](https://docs.litellm.ai/docs/providers/gemini#usage---thinking--reasoning_content), [PR](https://github.com/BerriAI/litellm/pull/10198) +2. Ensure returned `usage` includes thinking token usage [PR](https://github.com/BerriAI/litellm/pull/10198) +3. Added cost tracking for `gemini-2.5-pro-preview-03-25` [PR](https://github.com/BerriAI/litellm/pull/10178) +- **Cohere** +1. Added support for cohere command-a-03-2025 [Get Started](https://docs.litellm.ai/docs/providers/cohere), [PR](https://github.com/BerriAI/litellm/pull/10295) +- **SageMaker** +1. Added support for max\_completion\_tokens parameter [Get Started](https://docs.litellm.ai/docs/providers/sagemaker), [PR](https://github.com/BerriAI/litellm/pull/10300) +- **Responses API** +1. Added support for GET and DELETE operations - `/v1/responses/{response_id}` [Get Started](https://docs.litellm.ai/docs/response_api) +2. Added session management support for non-OpenAI models [PR](https://github.com/BerriAI/litellm/pull/10321) +3. Added routing affinity to maintain model consistency within sessions [Get Started](https://docs.litellm.ai/docs/response_api#load-balancing-with-routing-affinity), [PR](https://github.com/BerriAI/litellm/pull/10193) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/security\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **Bug Fix**: Fixed spend tracking bug, ensuring default litellm params aren't modified in memory [PR](https://github.com/BerriAI/litellm/pull/10167) +- **Deprecation Dates**: Added deprecation dates for Azure, VertexAI models [PR](https://github.com/BerriAI/litellm/pull/10308) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/security\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +#### Users [​](https://docs.litellm.ai/release_notes/tags/security\#users "Direct link to Users") + +- **Filtering and Searching**: + + + - Filter users by user\_id, role, team, sso\_id + - Search users by email + +![](https://docs.litellm.ai/assets/ideal-img/user_filters.e2b4a8c.1920.png) + +- **User Info Panel**: Added a new user information pane [PR](https://github.com/BerriAI/litellm/pull/10213) + + - View teams, keys, models associated with User + - Edit user role, model permissions + +#### Teams [​](https://docs.litellm.ai/release_notes/tags/security\#teams "Direct link to Teams") + +- **Filtering and Searching**: + + + - Filter teams by Organization, Team ID [PR](https://github.com/BerriAI/litellm/pull/10324) + - Search teams by Team Name [PR](https://github.com/BerriAI/litellm/pull/10324) + +![](https://docs.litellm.ai/assets/ideal-img/team_filters.c9c085b.1920.png) + +#### Keys [​](https://docs.litellm.ai/release_notes/tags/security\#keys "Direct link to Keys") + +- **Key Management**: + - Support for cross-filtering and filtering by key hash [PR](https://github.com/BerriAI/litellm/pull/10322) + - Fixed key alias reset when resetting filters [PR](https://github.com/BerriAI/litellm/pull/10099) + - Fixed table rendering on key creation [PR](https://github.com/BerriAI/litellm/pull/10224) + +#### UI Logs Page [​](https://docs.litellm.ai/release_notes/tags/security\#ui-logs-page "Direct link to UI Logs Page") + +- **Session Logs**: Added UI Session Logs [Get Started](https://docs.litellm.ai/docs/proxy/ui_logs_sessions) + +#### UI Authentication & Security [​](https://docs.litellm.ai/release_notes/tags/security\#ui-authentication--security "Direct link to UI Authentication & Security") + +- **Required Authentication**: Authentication now required for all dashboard pages [PR](https://github.com/BerriAI/litellm/pull/10229) +- **SSO Fixes**: Fixed SSO user login invalid token error [PR](https://github.com/BerriAI/litellm/pull/10298) +- \[BETA\] **Encrypted Tokens**: Moved UI to encrypted token usage [PR](https://github.com/BerriAI/litellm/pull/10302) +- **Token Expiry**: Support token refresh by re-routing to login page (fixes issue where expired token would show a blank page) [PR](https://github.com/BerriAI/litellm/pull/10250) + +#### UI General fixes [​](https://docs.litellm.ai/release_notes/tags/security\#ui-general-fixes "Direct link to UI General fixes") + +- **Fixed UI Flicker**: Addressed UI flickering issues in Dashboard [PR](https://github.com/BerriAI/litellm/pull/10261) +- **Improved Terminology**: Better loading and no-data states on Keys and Tools pages [PR](https://github.com/BerriAI/litellm/pull/10253) +- **Azure Model Support**: Fixed editing Azure public model names and changing model names after creation [PR](https://github.com/BerriAI/litellm/pull/10249) +- **Team Model Selector**: Bug fix for team model selection [PR](https://github.com/BerriAI/litellm/pull/10171) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/security\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +- **Datadog**: +1. Fixed Datadog LLM observability logging [Get Started](https://docs.litellm.ai/docs/proxy/logging#datadog), [PR](https://github.com/BerriAI/litellm/pull/10206) +- **Prometheus / Grafana**: +1. Enable datasource selection on LiteLLM Grafana Template [Get Started](https://docs.litellm.ai/docs/proxy/prometheus#-litellm-maintained-grafana-dashboards-), [PR](https://github.com/BerriAI/litellm/pull/10257) +- **AgentOps**: +1. Added AgentOps Integration [Get Started](https://docs.litellm.ai/docs/observability/agentops_integration), [PR](https://github.com/BerriAI/litellm/pull/9685) +- **Arize**: +1. Added missing attributes for Arize & Phoenix Integration [Get Started](https://docs.litellm.ai/docs/observability/arize_integration), [PR](https://github.com/BerriAI/litellm/pull/10215) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/security\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Caching**: Fixed caching to account for `thinking` or `reasoning_effort` when calculating cache key [PR](https://github.com/BerriAI/litellm/pull/10140) +- **Model Groups**: Fixed handling for cases where user sets model\_group inside model\_info [PR](https://github.com/BerriAI/litellm/pull/10191) +- **Passthrough Endpoints**: Ensured `PassthroughStandardLoggingPayload` is logged with method, URL, request/response body [PR](https://github.com/BerriAI/litellm/pull/10194) +- **Fix SQL Injection**: Fixed potential SQL injection vulnerability in spend\_management\_endpoints.py [PR](https://github.com/BerriAI/litellm/pull/9878) + +## Helm [​](https://docs.litellm.ai/release_notes/tags/security\#helm "Direct link to Helm") + +- Fixed serviceAccountName on migration job [PR](https://github.com/BerriAI/litellm/pull/10258) + +## Full Changelog [​](https://docs.litellm.ai/release_notes/tags/security\#full-changelog "Direct link to Full Changelog") + +The complete list of changes can be found in the [GitHub release notes](https://github.com/BerriAI/litellm/compare/v1.67.0-stable...v1.67.4-stable). + +## Key Highlights [​](https://docs.litellm.ai/release_notes/tags/security\#key-highlights "Direct link to Key Highlights") + +- **SCIM Integration**: Enables identity providers (Okta, Azure AD, OneLogin, etc.) to automate user and team (group) provisioning, updates, and deprovisioning +- **Team and Tag based usage tracking**: You can now see usage and spend by team and tag at 1M+ spend logs. +- **Unified Responses API**: Support for calling Anthropic, Gemini, Groq, etc. via OpenAI's new Responses API. + +Let's dive in. + +## SCIM Integration [​](https://docs.litellm.ai/release_notes/tags/security\#scim-integration "Direct link to SCIM Integration") + +![](https://docs.litellm.ai/assets/ideal-img/scim_integration.01959e2.1200.png) + +This release adds SCIM support to LiteLLM. This allows your SSO provider (Okta, Azure AD, etc) to automatically create/delete users, teams, and memberships on LiteLLM. This means that when you remove a team on your SSO provider, your SSO provider will automatically delete the corresponding team on LiteLLM. + +[Read more](https://docs.litellm.ai/docs/tutorials/scim_litellm) + +## Team and Tag based usage tracking [​](https://docs.litellm.ai/release_notes/tags/security\#team-and-tag-based-usage-tracking "Direct link to Team and Tag based usage tracking") + +![](https://docs.litellm.ai/assets/ideal-img/new_team_usage_highlight.60482cc.1920.jpg) + +This release improves team and tag based usage tracking at 1m+ spend logs, making it easy to monitor your LLM API Spend in production. This covers: + +- View **daily spend** by teams + tags +- View **usage / spend by key**, within teams +- View **spend by multiple tags** +- Allow **internal users** to view spend of teams they're a member of + +[Read more](https://docs.litellm.ai/release_notes/tags/security#management-endpoints--ui) + +## Unified Responses API [​](https://docs.litellm.ai/release_notes/tags/security\#unified-responses-api "Direct link to Unified Responses API") + +This release allows you to call Azure OpenAI, Anthropic, AWS Bedrock, and Google Vertex AI models via the POST /v1/responses endpoint on LiteLLM. This means you can now use popular tools like [OpenAI Codex](https://docs.litellm.ai/docs/tutorials/openai_codex) with your own models. + +![](https://docs.litellm.ai/assets/ideal-img/unified_responses_api_rn.0acc91a.1920.png) + +[Read more](https://docs.litellm.ai/docs/response_api) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/security\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **OpenAI** +1. gpt-4.1, gpt-4.1-mini, gpt-4.1-nano, o3, o3-mini, o4-mini pricing - [Get Started](https://docs.litellm.ai/docs/providers/openai#usage), [PR](https://github.com/BerriAI/litellm/pull/9990) +2. o4 - correctly map o4 to openai o\_series model +- **Azure AI** +1. Phi-4 output cost per token fix - [PR](https://github.com/BerriAI/litellm/pull/9880) +2. Responses API support [Get Started](https://docs.litellm.ai/docs/providers/azure#azure-responses-api), [PR](https://github.com/BerriAI/litellm/pull/10116) +- **Anthropic** +1. redacted message thinking support - [Get Started](https://docs.litellm.ai/docs/providers/anthropic#usage---thinking--reasoning_content), [PR](https://github.com/BerriAI/litellm/pull/10129) +- **Cohere** +1. `/v2/chat` Passthrough endpoint support w/ cost tracking - [Get Started](https://docs.litellm.ai/docs/pass_through/cohere), [PR](https://github.com/BerriAI/litellm/pull/9997) +- **Azure** +1. Support azure tenant\_id/client\_id env vars - [Get Started](https://docs.litellm.ai/docs/providers/azure#entra-id---use-tenant_id-client_id-client_secret), [PR](https://github.com/BerriAI/litellm/pull/9993) +2. Fix response\_format check for 2025+ api versions - [PR](https://github.com/BerriAI/litellm/pull/9993) +3. Add gpt-4.1, gpt-4.1-mini, gpt-4.1-nano, o3, o3-mini, o4-mini pricing +- **VLLM** +1. Files - Support 'file' message type for VLLM video url's - [Get Started](https://docs.litellm.ai/docs/providers/vllm#send-video-url-to-vllm), [PR](https://github.com/BerriAI/litellm/pull/10129) +2. Passthrough - new `/vllm/` passthrough endpoint support [Get Started](https://docs.litellm.ai/docs/pass_through/vllm), [PR](https://github.com/BerriAI/litellm/pull/10002) +- **Mistral** +1. new `/mistral` passthrough endpoint support [Get Started](https://docs.litellm.ai/docs/pass_through/mistral), [PR](https://github.com/BerriAI/litellm/pull/10002) +- **AWS** +1. New mapped bedrock regions - [PR](https://github.com/BerriAI/litellm/pull/9430) +- **VertexAI / Google AI Studio** +1. Gemini - Response format - Retain schema field ordering for google gemini and vertex by specifying propertyOrdering - [Get Started](https://docs.litellm.ai/docs/providers/vertex#json-schema), [PR](https://github.com/BerriAI/litellm/pull/9828) +2. Gemini-2.5-flash - return reasoning content [Google AI Studio](https://docs.litellm.ai/docs/providers/gemini#usage---thinking--reasoning_content), [Vertex AI](https://docs.litellm.ai/docs/providers/vertex#thinking--reasoning_content) +3. Gemini-2.5-flash - pricing + model information [PR](https://github.com/BerriAI/litellm/pull/10125) +4. Passthrough - new `/vertex_ai/discovery` route - enables calling AgentBuilder API routes [Get Started](https://docs.litellm.ai/docs/pass_through/vertex_ai#supported-api-endpoints), [PR](https://github.com/BerriAI/litellm/pull/10084) +- **Fireworks AI** +1. return tool calling responses in `tool_calls` field (fireworks incorrectly returns this as a json str in content) [PR](https://github.com/BerriAI/litellm/pull/10130) +- **Triton** +1. Remove fixed remove bad\_words / stop words from `/generate` call - [Get Started](https://docs.litellm.ai/docs/providers/triton-inference-server#triton-generate---chat-completion), [PR](https://github.com/BerriAI/litellm/pull/10163) +- **Other** +1. Support for all litellm providers on Responses API (works with Codex) - [Get Started](https://docs.litellm.ai/docs/tutorials/openai_codex), [PR](https://github.com/BerriAI/litellm/pull/10132) +2. Fix combining multiple tool calls in streaming response - [Get Started](https://docs.litellm.ai/docs/completion/stream#helper-function), [PR](https://github.com/BerriAI/litellm/pull/10040) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/security\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **Cost Control** \- inject cache control points in prompt for cost reduction [Get Started](https://docs.litellm.ai/docs/tutorials/prompt_caching), [PR](https://github.com/BerriAI/litellm/pull/10000) +- **Spend Tags** \- spend tags in headers - support x-litellm-tags even if tag based routing not enabled [Get Started](https://docs.litellm.ai/docs/proxy/request_headers#litellm-headers), [PR](https://github.com/BerriAI/litellm/pull/10000) +- **Gemini-2.5-flash** \- support cost calculation for reasoning tokens [PR](https://github.com/BerriAI/litellm/pull/10141) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/security\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +- **Users** + +1. Show created\_at and updated\_at on users page - [PR](https://github.com/BerriAI/litellm/pull/10033) +- **Virtual Keys** + +1. Filter by key alias - [https://github.com/BerriAI/litellm/pull/10085](https://github.com/BerriAI/litellm/pull/10085) +- **Usage Tab** + +1. Team based usage + + + - New `LiteLLM_DailyTeamSpend` Table for aggregate team based usage logging - [PR](https://github.com/BerriAI/litellm/pull/10039) + + - New Team based usage dashboard + new `/team/daily/activity` API - [PR](https://github.com/BerriAI/litellm/pull/10081) + + - Return team alias on /team/daily/activity API - [PR](https://github.com/BerriAI/litellm/pull/10157) + + - allow internal user view spend for teams they belong to - [PR](https://github.com/BerriAI/litellm/pull/10157) + + - allow viewing top keys by team - [PR](https://github.com/BerriAI/litellm/pull/10157) + + +![](https://docs.litellm.ai/assets/ideal-img/new_team_usage.9237b43.1754.png) + +2. Tag Based Usage + + - New `LiteLLM_DailyTagSpend` Table for aggregate tag based usage logging - [PR](https://github.com/BerriAI/litellm/pull/10071) + - Restrict to only Proxy Admins - [PR](https://github.com/BerriAI/litellm/pull/10157) + - allow viewing top keys by tag + - Return tags passed in request (i.e. dynamic tags) on `/tag/list` API - [PR](https://github.com/BerriAI/litellm/pull/10157) + ![](https://docs.litellm.ai/assets/ideal-img/new_tag_usage.cd55b64.1863.png) +3. Track prompt caching metrics in daily user, team, tag tables - [PR](https://github.com/BerriAI/litellm/pull/10029) + +4. Show usage by key (on all up, team, and tag usage dashboards) - [PR](https://github.com/BerriAI/litellm/pull/10157) + +5. swap old usage with new usage tab +- **Models** + +1. Make columns resizable/hideable - [PR](https://github.com/BerriAI/litellm/pull/10119) +- **API Playground** + +1. Allow internal user to call api playground - [PR](https://github.com/BerriAI/litellm/pull/10157) +- **SCIM** + +1. Add LiteLLM SCIM Integration for Team and User management - [Get Started](https://docs.litellm.ai/docs/tutorials/scim_litellm), [PR](https://github.com/BerriAI/litellm/pull/10072) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/security\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +- **GCS** +1. Fix gcs pub sub logging with env var GCS\_PROJECT\_ID - [Get Started](https://docs.litellm.ai/docs/observability/gcs_bucket_integration#usage), [PR](https://github.com/BerriAI/litellm/pull/10042) +- **AIM** +1. Add litellm call id passing to Aim guardrails on pre and post-hooks calls - [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/aim_security), [PR](https://github.com/BerriAI/litellm/pull/10021) +- **Azure blob storage** +1. Ensure logging works in high throughput scenarios - [Get Started](https://docs.litellm.ai/docs/proxy/logging#azure-blob-storage), [PR](https://github.com/BerriAI/litellm/pull/9962) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/security\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Support setting `litellm.modify_params` via env var** [PR](https://github.com/BerriAI/litellm/pull/9964) +- **Model Discovery** \- Check provider’s `/models` endpoints when calling proxy’s `/v1/models` endpoint - [Get Started](https://docs.litellm.ai/docs/proxy/model_discovery), [PR](https://github.com/BerriAI/litellm/pull/9958) +- **`/utils/token_counter`** \- fix retrieving custom tokenizer for db models - [Get Started](https://docs.litellm.ai/docs/proxy/configs#set-custom-tokenizer), [PR](https://github.com/BerriAI/litellm/pull/10047) +- **Prisma migrate** \- handle existing columns in db table - [PR](https://github.com/BerriAI/litellm/pull/10138) + +## Deploy this version [​](https://docs.litellm.ai/release_notes/tags/security\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.66.0-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.66.0.post1 + +``` + +v1.66.0-stable is live now, here are the key highlights of this release + +## Key Highlights [​](https://docs.litellm.ai/release_notes/tags/security\#key-highlights "Direct link to Key Highlights") + +- **Realtime API Cost Tracking**: Track cost of realtime API calls +- **Microsoft SSO Auto-sync**: Auto-sync groups and group members from Azure Entra ID to LiteLLM +- **xAI grok-3**: Added support for `xai/grok-3` models +- **Security Fixes**: Fixed [CVE-2025-0330](https://www.cve.org/CVERecord?id=CVE-2025-0330) and [CVE-2024-6825](https://www.cve.org/CVERecord?id=CVE-2024-6825) vulnerabilities + +Let's dive in. + +## Realtime API Cost Tracking [​](https://docs.litellm.ai/release_notes/tags/security\#realtime-api-cost-tracking "Direct link to Realtime API Cost Tracking") + +![](https://docs.litellm.ai/assets/ideal-img/realtime_api.960b38e.1920.png) + +This release adds Realtime API logging + cost tracking. + +- **Logging**: LiteLLM now logs the complete response from realtime calls to all logging integrations (DB, S3, Langfuse, etc.) +- **Cost Tracking**: You can now set 'base\_model' and custom pricing for realtime models. [Custom Pricing](https://docs.litellm.ai/docs/proxy/custom_pricing) +- **Budgets**: Your key/user/team budgets now work for realtime models as well. + +Start [here](https://docs.litellm.ai/docs/realtime) + +## Microsoft SSO Auto-sync [​](https://docs.litellm.ai/release_notes/tags/security\#microsoft-sso-auto-sync "Direct link to Microsoft SSO Auto-sync") + +![](https://docs.litellm.ai/assets/ideal-img/sso_sync.2f79062.1414.png) + +Auto-sync groups and members from Azure Entra ID to LiteLLM + +This release adds support for auto-syncing groups and members on Microsoft Entra ID with LiteLLM. This means that LiteLLM proxy administrators can spend less time managing teams and members and LiteLLM handles the following: + +- Auto-create teams that exist on Microsoft Entra ID +- Sync team members on Microsoft Entra ID with LiteLLM teams + +Get started with this [here](https://docs.litellm.ai/docs/tutorials/msft_sso) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/security\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **xAI** + +1. Added reasoning\_effort support for `xai/grok-3-mini-beta` [Get Started](https://docs.litellm.ai/docs/providers/xai#reasoning-usage) +2. Added cost tracking for `xai/grok-3` models [PR](https://github.com/BerriAI/litellm/pull/9920) +- **Hugging Face** + +1. Added inference providers support [Get Started](https://docs.litellm.ai/docs/providers/huggingface#serverless-inference-providers) +- **Azure** + +1. Added azure/gpt-4o-realtime-audio cost tracking [PR](https://github.com/BerriAI/litellm/pull/9893) +- **VertexAI** + +1. Added enterpriseWebSearch tool support [Get Started](https://docs.litellm.ai/docs/providers/vertex#grounding---web-search) +2. Moved to only passing keys accepted by the Vertex AI response schema [PR](https://github.com/BerriAI/litellm/pull/8992) +- **Google AI Studio** + +1. Added cost tracking for `gemini-2.5-pro` [PR](https://github.com/BerriAI/litellm/pull/9837) +2. Fixed pricing for 'gemini/gemini-2.5-pro-preview-03-25' [PR](https://github.com/BerriAI/litellm/pull/9896) +3. Fixed handling file\_data being passed in [PR](https://github.com/BerriAI/litellm/pull/9786) +- **Azure** + +1. Updated Azure Phi-4 pricing [PR](https://github.com/BerriAI/litellm/pull/9862) +2. Added azure/gpt-4o-realtime-audio cost tracking [PR](https://github.com/BerriAI/litellm/pull/9893) +- **Databricks** + +1. Removed reasoning\_effort from parameters [PR](https://github.com/BerriAI/litellm/pull/9811) +2. Fixed custom endpoint check for Databricks [PR](https://github.com/BerriAI/litellm/pull/9925) +- **General** + +1. Added litellm.supports\_reasoning() util to track if an llm supports reasoning [Get Started](https://docs.litellm.ai/docs/providers/anthropic#reasoning) +2. Function Calling - Handle pydantic base model in message tool calls, handle tools = \[\], and support fake streaming on tool calls for meta.llama3-3-70b-instruct-v1:0 [PR](https://github.com/BerriAI/litellm/pull/9774) +3. LiteLLM Proxy - Allow passing `thinking` param to litellm proxy via client sdk [PR](https://github.com/BerriAI/litellm/pull/9386) +4. Fixed correctly translating 'thinking' param for litellm [PR](https://github.com/BerriAI/litellm/pull/9904) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/security\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **OpenAI, Azure** +1. Realtime API Cost tracking with token usage metrics in spend logs [Get Started](https://docs.litellm.ai/docs/realtime) +- **Anthropic** +1. Fixed Claude Haiku cache read pricing per token [PR](https://github.com/BerriAI/litellm/pull/9834) +2. Added cost tracking for Claude responses with base\_model [PR](https://github.com/BerriAI/litellm/pull/9897) +3. Fixed Anthropic prompt caching cost calculation and trimmed logged message in db [PR](https://github.com/BerriAI/litellm/pull/9838) +- **General** +1. Added token tracking and log usage object in spend logs [PR](https://github.com/BerriAI/litellm/pull/9843) +2. Handle custom pricing at deployment level [PR](https://github.com/BerriAI/litellm/pull/9855) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/security\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +- **Test Key Tab** + +1. Added rendering of Reasoning content, ttft, usage metrics on test key page [PR](https://github.com/BerriAI/litellm/pull/9931) + + ![](https://docs.litellm.ai/assets/ideal-img/chat_metrics.c59fcfe.1920.png) + + View input, output, reasoning tokens, ttft metrics. +- **Tag / Policy Management** + +1. Added Tag/Policy Management. Create routing rules based on request metadata. This allows you to enforce that requests with `tags="private"` only go to specific models. [Get Started](https://docs.litellm.ai/docs/tutorials/tag_management) + + + + ![](https://docs.litellm.ai/assets/ideal-img/tag_management.5bf985c.1920.png) + + Create and manage tags. +- **Redesigned Login Screen** + +1. Polished login screen [PR](https://github.com/BerriAI/litellm/pull/9778) +- **Microsoft SSO Auto-Sync** + +1. Added debug route to allow admins to debug SSO JWT fields [PR](https://github.com/BerriAI/litellm/pull/9835) +2. Added ability to use MSFT Graph API to assign users to teams [PR](https://github.com/BerriAI/litellm/pull/9865) +3. Connected litellm to Azure Entra ID Enterprise Application [PR](https://github.com/BerriAI/litellm/pull/9872) +4. Added ability for admins to set `default_team_params` for when litellm SSO creates default teams [PR](https://github.com/BerriAI/litellm/pull/9895) +5. Fixed MSFT SSO to use correct field for user email [PR](https://github.com/BerriAI/litellm/pull/9886) +6. Added UI support for setting Default Team setting when litellm SSO auto creates teams [PR](https://github.com/BerriAI/litellm/pull/9918) +- **UI Bug Fixes** + +1. Prevented team, key, org, model numerical values changing on scrolling [PR](https://github.com/BerriAI/litellm/pull/9776) +2. Instantly reflect key and team updates in UI [PR](https://github.com/BerriAI/litellm/pull/9825) + +## Logging / Guardrail Improvements [​](https://docs.litellm.ai/release_notes/tags/security\#logging--guardrail-improvements "Direct link to Logging / Guardrail Improvements") + +- **Prometheus** +1. Emit Key and Team Budget metrics on a cron job schedule [Get Started](https://docs.litellm.ai/docs/proxy/prometheus#initialize-budget-metrics-on-startup) + +## Security Fixes [​](https://docs.litellm.ai/release_notes/tags/security\#security-fixes "Direct link to Security Fixes") + +- Fixed [CVE-2025-0330](https://www.cve.org/CVERecord?id=CVE-2025-0330) \- Leakage of Langfuse API keys in team exception handling [PR](https://github.com/BerriAI/litellm/pull/9830) +- Fixed [CVE-2024-6825](https://www.cve.org/CVERecord?id=CVE-2024-6825) \- Remote code execution in post call rules [PR](https://github.com/BerriAI/litellm/pull/9826) + +## Helm [​](https://docs.litellm.ai/release_notes/tags/security\#helm "Direct link to Helm") + +- Added service annotations to litellm-helm chart [PR](https://github.com/BerriAI/litellm/pull/9840) +- Added extraEnvVars to the helm deployment [PR](https://github.com/BerriAI/litellm/pull/9292) + +## Demo [​](https://docs.litellm.ai/release_notes/tags/security\#demo "Direct link to Demo") + +Try this on the demo instance [today](https://docs.litellm.ai/docs/proxy/demo) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/security\#complete-git-diff "Direct link to Complete Git Diff") + +See the complete git diff since v1.65.4-stable, [here](https://github.com/BerriAI/litellm/releases/tag/v1.66.0-stable) + +`docker image`, `security`, `vulnerability` + +# 0 Critical/High Vulnerabilities + +![](https://docs.litellm.ai/assets/ideal-img/security.8eb0218.1200.png) + +## What changed? [​](https://docs.litellm.ai/release_notes/tags/security\#what-changed "Direct link to What changed?") + +- LiteLLMBase image now uses `cgr.dev/chainguard/python:latest-dev` + +## Why the change? [​](https://docs.litellm.ai/release_notes/tags/security\#why-the-change "Direct link to Why the change?") + +To ensure there are 0 critical/high vulnerabilities on LiteLLM Docker Image + +## Migration Guide [​](https://docs.litellm.ai/release_notes/tags/security\#migration-guide "Direct link to Migration Guide") + +- If you use a custom dockerfile with litellm as a base image + `apt-get` + +Instead of `apt-get` use `apk`, the base litellm image will no longer have `apt-get` installed. + +**You are only impacted if you use `apt-get` in your Dockerfile** + +```codeBlockLines_e6Vv +# Use the provided base image +FROM ghcr.io/berriai/litellm:main-latest + +# Set the working directory +WORKDIR /app + +# Install dependencies - CHANGE THIS to `apk` +RUN apt-get update && apt-get install -y dumb-init + +``` + +Before Change + +```codeBlockLines_e6Vv +RUN apt-get update && apt-get install -y dumb-init + +``` + +After Change + +```codeBlockLines_e6Vv +RUN apk update && apk add --no-cache dumb-init + +``` + +## Session Management Updates +[Skip to main content](https://docs.litellm.ai/release_notes/tags/session-management#__docusaurus_skipToContent_fallback) + +## Deploy this version [​](https://docs.litellm.ai/release_notes/tags/session-management\#deploy-this-version "Direct link to Deploy this version") + +- Docker +- Pip + +docker run litellm + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.67.4-stable + +``` + +pip install litellm + +```codeBlockLines_e6Vv +pip install litellm==1.67.4.post1 + +``` + +## Key Highlights [​](https://docs.litellm.ai/release_notes/tags/session-management\#key-highlights "Direct link to Key Highlights") + +- **Improved User Management**: This release enables search and filtering across users, keys, teams, and models. +- **Responses API Load Balancing**: Route requests across provider regions and ensure session continuity. +- **UI Session Logs**: Group several requests to LiteLLM into a session. + +## Improved User Management [​](https://docs.litellm.ai/release_notes/tags/session-management\#improved-user-management "Direct link to Improved User Management") + +![](https://docs.litellm.ai/assets/ideal-img/ui_search_users.7472bdc.1920.png) + +This release makes it easier to manage users and keys on LiteLLM. You can now search and filter across users, keys, teams, and models, and control user settings more easily. + +New features include: + +- Search for users by email, ID, role, or team. +- See all of a user's models, teams, and keys in one place. +- Change user roles and model access right from the Users Tab. + +These changes help you spend less time on user setup and management on LiteLLM. + +## Responses API Load Balancing [​](https://docs.litellm.ai/release_notes/tags/session-management\#responses-api-load-balancing "Direct link to Responses API Load Balancing") + +![](https://docs.litellm.ai/assets/ideal-img/ui_responses_lb.1e64cec.1204.png) + +This release introduces load balancing for the Responses API, allowing you to route requests across provider regions and ensure session continuity. It works as follows: + +- If a `previous_response_id` is provided, LiteLLM will route the request to the original deployment that generated the prior response — ensuring session continuity. +- If no `previous_response_id` is provided, LiteLLM will load-balance requests across your available deployments. + +[Read more](https://docs.litellm.ai/docs/response_api#load-balancing-with-session-continuity) + +## UI Session Logs [​](https://docs.litellm.ai/release_notes/tags/session-management\#ui-session-logs "Direct link to UI Session Logs") + +![](https://docs.litellm.ai/assets/ideal-img/ui_session_logs.926dffc.1920.png) + +This release allow you to group requests to LiteLLM proxy into a session. If you specify a litellm\_session\_id in your request LiteLLM will automatically group all logs in the same session. This allows you to easily track usage and request content per session. + +[Read more](https://docs.litellm.ai/docs/proxy/ui_logs_sessions) + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/session-management\#new-models--updated-models "Direct link to New Models / Updated Models") + +- **OpenAI** +1. Added `gpt-image-1` cost tracking [Get Started](https://docs.litellm.ai/docs/image_generation) +2. Bug fix: added cost tracking for gpt-image-1 when quality is unspecified [PR](https://github.com/BerriAI/litellm/pull/10247) +- **Azure** +1. Fixed timestamp granularities passing to whisper in Azure [Get Started](https://docs.litellm.ai/docs/audio_transcription) +2. Added azure/gpt-image-1 pricing [Get Started](https://docs.litellm.ai/docs/image_generation), [PR](https://github.com/BerriAI/litellm/pull/10327) +3. Added cost tracking for `azure/computer-use-preview`, `azure/gpt-4o-audio-preview-2024-12-17`, `azure/gpt-4o-mini-audio-preview-2024-12-17` [PR](https://github.com/BerriAI/litellm/pull/10178) +- **Bedrock** +1. Added support for all compatible Bedrock parameters when model="arn:.." (Bedrock application inference profile models) [Get started](https://docs.litellm.ai/docs/providers/bedrock#bedrock-application-inference-profile), [PR](https://github.com/BerriAI/litellm/pull/10256) +2. Fixed wrong system prompt transformation [PR](https://github.com/BerriAI/litellm/pull/10120) +- **VertexAI / Google AI Studio** +1. Allow setting `budget_tokens=0` for `gemini-2.5-flash` [Get Started](https://docs.litellm.ai/docs/providers/gemini#usage---thinking--reasoning_content), [PR](https://github.com/BerriAI/litellm/pull/10198) +2. Ensure returned `usage` includes thinking token usage [PR](https://github.com/BerriAI/litellm/pull/10198) +3. Added cost tracking for `gemini-2.5-pro-preview-03-25` [PR](https://github.com/BerriAI/litellm/pull/10178) +- **Cohere** +1. Added support for cohere command-a-03-2025 [Get Started](https://docs.litellm.ai/docs/providers/cohere), [PR](https://github.com/BerriAI/litellm/pull/10295) +- **SageMaker** +1. Added support for max\_completion\_tokens parameter [Get Started](https://docs.litellm.ai/docs/providers/sagemaker), [PR](https://github.com/BerriAI/litellm/pull/10300) +- **Responses API** +1. Added support for GET and DELETE operations - `/v1/responses/{response_id}` [Get Started](https://docs.litellm.ai/docs/response_api) +2. Added session management support for non-OpenAI models [PR](https://github.com/BerriAI/litellm/pull/10321) +3. Added routing affinity to maintain model consistency within sessions [Get Started](https://docs.litellm.ai/docs/response_api#load-balancing-with-routing-affinity), [PR](https://github.com/BerriAI/litellm/pull/10193) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/session-management\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- **Bug Fix**: Fixed spend tracking bug, ensuring default litellm params aren't modified in memory [PR](https://github.com/BerriAI/litellm/pull/10167) +- **Deprecation Dates**: Added deprecation dates for Azure, VertexAI models [PR](https://github.com/BerriAI/litellm/pull/10308) + +## Management Endpoints / UI [​](https://docs.litellm.ai/release_notes/tags/session-management\#management-endpoints--ui "Direct link to Management Endpoints / UI") + +#### Users [​](https://docs.litellm.ai/release_notes/tags/session-management\#users "Direct link to Users") + +- **Filtering and Searching**: + + + - Filter users by user\_id, role, team, sso\_id + - Search users by email + +![](https://docs.litellm.ai/assets/ideal-img/user_filters.e2b4a8c.1920.png) + +- **User Info Panel**: Added a new user information pane [PR](https://github.com/BerriAI/litellm/pull/10213) + + - View teams, keys, models associated with User + - Edit user role, model permissions + +#### Teams [​](https://docs.litellm.ai/release_notes/tags/session-management\#teams "Direct link to Teams") + +- **Filtering and Searching**: + + + - Filter teams by Organization, Team ID [PR](https://github.com/BerriAI/litellm/pull/10324) + - Search teams by Team Name [PR](https://github.com/BerriAI/litellm/pull/10324) + +![](https://docs.litellm.ai/assets/ideal-img/team_filters.c9c085b.1920.png) + +#### Keys [​](https://docs.litellm.ai/release_notes/tags/session-management\#keys "Direct link to Keys") + +- **Key Management**: + - Support for cross-filtering and filtering by key hash [PR](https://github.com/BerriAI/litellm/pull/10322) + - Fixed key alias reset when resetting filters [PR](https://github.com/BerriAI/litellm/pull/10099) + - Fixed table rendering on key creation [PR](https://github.com/BerriAI/litellm/pull/10224) + +#### UI Logs Page [​](https://docs.litellm.ai/release_notes/tags/session-management\#ui-logs-page "Direct link to UI Logs Page") + +- **Session Logs**: Added UI Session Logs [Get Started](https://docs.litellm.ai/docs/proxy/ui_logs_sessions) + +#### UI Authentication & Security [​](https://docs.litellm.ai/release_notes/tags/session-management\#ui-authentication--security "Direct link to UI Authentication & Security") + +- **Required Authentication**: Authentication now required for all dashboard pages [PR](https://github.com/BerriAI/litellm/pull/10229) +- **SSO Fixes**: Fixed SSO user login invalid token error [PR](https://github.com/BerriAI/litellm/pull/10298) +- \[BETA\] **Encrypted Tokens**: Moved UI to encrypted token usage [PR](https://github.com/BerriAI/litellm/pull/10302) +- **Token Expiry**: Support token refresh by re-routing to login page (fixes issue where expired token would show a blank page) [PR](https://github.com/BerriAI/litellm/pull/10250) + +#### UI General fixes [​](https://docs.litellm.ai/release_notes/tags/session-management\#ui-general-fixes "Direct link to UI General fixes") + +- **Fixed UI Flicker**: Addressed UI flickering issues in Dashboard [PR](https://github.com/BerriAI/litellm/pull/10261) +- **Improved Terminology**: Better loading and no-data states on Keys and Tools pages [PR](https://github.com/BerriAI/litellm/pull/10253) +- **Azure Model Support**: Fixed editing Azure public model names and changing model names after creation [PR](https://github.com/BerriAI/litellm/pull/10249) +- **Team Model Selector**: Bug fix for team model selection [PR](https://github.com/BerriAI/litellm/pull/10171) + +## Logging / Guardrail Integrations [​](https://docs.litellm.ai/release_notes/tags/session-management\#logging--guardrail-integrations "Direct link to Logging / Guardrail Integrations") + +- **Datadog**: +1. Fixed Datadog LLM observability logging [Get Started](https://docs.litellm.ai/docs/proxy/logging#datadog), [PR](https://github.com/BerriAI/litellm/pull/10206) +- **Prometheus / Grafana**: +1. Enable datasource selection on LiteLLM Grafana Template [Get Started](https://docs.litellm.ai/docs/proxy/prometheus#-litellm-maintained-grafana-dashboards-), [PR](https://github.com/BerriAI/litellm/pull/10257) +- **AgentOps**: +1. Added AgentOps Integration [Get Started](https://docs.litellm.ai/docs/observability/agentops_integration), [PR](https://github.com/BerriAI/litellm/pull/9685) +- **Arize**: +1. Added missing attributes for Arize & Phoenix Integration [Get Started](https://docs.litellm.ai/docs/observability/arize_integration), [PR](https://github.com/BerriAI/litellm/pull/10215) + +## General Proxy Improvements [​](https://docs.litellm.ai/release_notes/tags/session-management\#general-proxy-improvements "Direct link to General Proxy Improvements") + +- **Caching**: Fixed caching to account for `thinking` or `reasoning_effort` when calculating cache key [PR](https://github.com/BerriAI/litellm/pull/10140) +- **Model Groups**: Fixed handling for cases where user sets model\_group inside model\_info [PR](https://github.com/BerriAI/litellm/pull/10191) +- **Passthrough Endpoints**: Ensured `PassthroughStandardLoggingPayload` is logged with method, URL, request/response body [PR](https://github.com/BerriAI/litellm/pull/10194) +- **Fix SQL Injection**: Fixed potential SQL injection vulnerability in spend\_management\_endpoints.py [PR](https://github.com/BerriAI/litellm/pull/9878) + +## Helm [​](https://docs.litellm.ai/release_notes/tags/session-management\#helm "Direct link to Helm") + +- Fixed serviceAccountName on migration job [PR](https://github.com/BerriAI/litellm/pull/10258) + +## Full Changelog [​](https://docs.litellm.ai/release_notes/tags/session-management\#full-changelog "Direct link to Full Changelog") + +The complete list of changes can be found in the [GitHub release notes](https://github.com/BerriAI/litellm/compare/v1.67.0-stable...v1.67.4-stable). + +## LiteLLM Release Notes +[Skip to main content](https://docs.litellm.ai/release_notes/tags/snowflake#__docusaurus_skipToContent_fallback) + +These are the changes since `v1.63.11-stable`. + +This release brings: + +- LLM Translation Improvements (MCP Support and Bedrock Application Profiles) +- Perf improvements for Usage-based Routing +- Streaming guardrail support via websockets +- Azure OpenAI client perf fix (from previous release) + +## Docker Run LiteLLM Proxy [​](https://docs.litellm.ai/release_notes/tags/snowflake\#docker-run-litellm-proxy "Direct link to Docker Run LiteLLM Proxy") + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.63.14-stable.patch1 + +``` + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/snowflake\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/snowflake\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Azure gpt-4o - fixed pricing to latest global pricing - [PR](https://github.com/BerriAI/litellm/pull/9361) +- O1-Pro - add pricing + model information - [PR](https://github.com/BerriAI/litellm/pull/9397) +- Azure AI - mistral 3.1 small pricing added - [PR](https://github.com/BerriAI/litellm/pull/9453) +- Azure - gpt-4.5-preview pricing added - [PR](https://github.com/BerriAI/litellm/pull/9453) + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/snowflake\#llm-translation "Direct link to LLM Translation") + +1. **New LLM Features** + +- Bedrock: Support bedrock application inference profiles [Docs](https://docs.litellm.ai/docs/providers/bedrock#bedrock-application-inference-profile) + - Infer aws region from bedrock application profile id - ( `arn:aws:bedrock:us-east-1:...`) +- Ollama - support calling via `/v1/completions` [Get Started](https://docs.litellm.ai/docs/providers/ollama#using-ollama-fim-on-v1completions) +- Bedrock - support `us.deepseek.r1-v1:0` model name [Docs](https://docs.litellm.ai/docs/providers/bedrock#supported-aws-bedrock-models) +- OpenRouter - `OPENROUTER_API_BASE` env var support [Docs](https://docs.litellm.ai/docs/providers/openrouter.md) +- Azure - add audio model parameter support - [Docs](https://docs.litellm.ai/docs/providers/azure#azure-audio-model) +- OpenAI - PDF File support [Docs](https://docs.litellm.ai/docs/completion/document_understanding#openai-file-message-type) +- OpenAI - o1-pro Responses API streaming support [Docs](https://docs.litellm.ai/docs/response_api.md#streaming) +- \[BETA\] MCP - Use MCP Tools with LiteLLM SDK [Docs](https://docs.litellm.ai/docs/mcp) + +2. **Bug Fixes** + +- Voyage: prompt token on embedding tracking fix - [PR](https://github.com/BerriAI/litellm/commit/56d3e75b330c3c3862dc6e1c51c1210e48f1068e) +- Sagemaker - Fix ‘Too little data for declared Content-Length’ error - [PR](https://github.com/BerriAI/litellm/pull/9326) +- OpenAI-compatible models - fix issue when calling openai-compatible models w/ custom\_llm\_provider set - [PR](https://github.com/BerriAI/litellm/pull/9355) +- VertexAI - Embedding ‘outputDimensionality’ support - [PR](https://github.com/BerriAI/litellm/commit/437dbe724620675295f298164a076cbd8019d304) +- Anthropic - return consistent json response format on streaming/non-streaming - [PR](https://github.com/BerriAI/litellm/pull/9437) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/snowflake\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +- `litellm_proxy/` \- support reading litellm response cost header from proxy, when using client sdk +- Reset Budget Job - fix budget reset error on keys/teams/users [PR](https://github.com/BerriAI/litellm/pull/9329) +- Streaming - Prevents final chunk w/ usage from being ignored (impacted bedrock streaming + cost tracking) [PR](https://github.com/BerriAI/litellm/pull/9314) + +## UI [​](https://docs.litellm.ai/release_notes/tags/snowflake\#ui "Direct link to UI") + +1. Users Page + - Feature: Control default internal user settings [PR](https://github.com/BerriAI/litellm/pull/9328) +2. Icons: + - Feature: Replace external "artificialanalysis.ai" icons by local svg [PR](https://github.com/BerriAI/litellm/pull/9374) +3. Sign In/Sign Out + - Fix: Default login when `default_user_id` user does not exist in DB [PR](https://github.com/BerriAI/litellm/pull/9395) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes/tags/snowflake\#logging-integrations "Direct link to Logging Integrations") + +- Support post-call guardrails for streaming responses [Get Started](https://docs.litellm.ai/docs/proxy/guardrails/custom_guardrail#1-write-a-customguardrail-class) +- Arize [Get Started](https://docs.litellm.ai/docs/observability/arize_integration) + - fix invalid package import [PR](https://github.com/BerriAI/litellm/pull/9338) + - migrate to using standardloggingpayload for metadata, ensures spans land successfully [PR](https://github.com/BerriAI/litellm/pull/9338) + - fix logging to just log the LLM I/O [PR](https://github.com/BerriAI/litellm/pull/9353) + - Dynamic API Key/Space param support [Get Started](https://docs.litellm.ai/docs/observability/arize_integration#pass-arize-spacekey-per-request) +- StandardLoggingPayload - Log litellm\_model\_name in payload. Allows knowing what the model sent to API provider was [Get Started](https://docs.litellm.ai/docs/proxy/logging_spec#standardlogginghiddenparams) +- Prompt Management - Allow building custom prompt management integration [Get Started](https://docs.litellm.ai/docs/proxy/custom_prompt_management.md) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/snowflake\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +- Redis Caching - add 5s default timeout, prevents hanging redis connection from impacting llm calls [PR](https://github.com/BerriAI/litellm/commit/db92956ae33ed4c4e3233d7e1b0c7229817159bf) +- Allow disabling all spend updates / writes to DB - patch to allow disabling all spend updates to DB with a flag [PR](https://github.com/BerriAI/litellm/pull/9331) +- Azure OpenAI - correctly re-use azure openai client, fixes perf issue from previous Stable release [PR](https://github.com/BerriAI/litellm/commit/f2026ef907c06d94440930917add71314b901413) +- Azure OpenAI - uses litellm.ssl\_verify on Azure/OpenAI clients [PR](https://github.com/BerriAI/litellm/commit/f2026ef907c06d94440930917add71314b901413) +- Usage-based routing - Wildcard model support [Get Started](https://docs.litellm.ai/docs/proxy/usage_based_routing#wildcard-model-support) +- Usage-based routing - Support batch writing increments to redis - reduces latency to same as ‘simple-shuffle’ [PR](https://github.com/BerriAI/litellm/pull/9357) +- Router - show reason for model cooldown on ‘no healthy deployments available error’ [PR](https://github.com/BerriAI/litellm/pull/9438) +- Caching - add max value limit to an item in in-memory cache (1MB) - prevents OOM errors on large image url’s being sent through proxy [PR](https://github.com/BerriAI/litellm/pull/9448) + +## General Improvements [​](https://docs.litellm.ai/release_notes/tags/snowflake\#general-improvements "Direct link to General Improvements") + +- Passthrough Endpoints - support returning api-base on pass-through endpoints Response Headers [Docs](https://docs.litellm.ai/docs/proxy/response_headers#litellm-specific-headers) +- SSL - support reading ssl security level from env var - Allows user to specify lower security settings [Get Started](https://docs.litellm.ai/docs/guides/security_settings) +- Credentials - only poll Credentials table when `STORE_MODEL_IN_DB` is True [PR](https://github.com/BerriAI/litellm/pull/9376) +- Image URL Handling - new architecture doc on image url handling [Docs](https://docs.litellm.ai/docs/proxy/image_handling) +- OpenAI - bump to pip install "openai==1.68.2" [PR](https://github.com/BerriAI/litellm/commit/e85e3bc52a9de86ad85c3dbb12d87664ee567a5a) +- Gunicorn - security fix - bump gunicorn==23.0.0 [PR](https://github.com/BerriAI/litellm/commit/7e9fc92f5c7fea1e7294171cd3859d55384166eb) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/snowflake\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.11-stable...v1.63.14.rc) + +These are the changes since `v1.63.2-stable`. + +This release is primarily focused on: + +- \[Beta\] Responses API Support +- Snowflake Cortex Support, Amazon Nova Image Generation +- UI - Credential Management, re-use credentials when adding new models +- UI - Test Connection to LLM Provider before adding a model + +## Known Issues [​](https://docs.litellm.ai/release_notes/tags/snowflake\#known-issues "Direct link to Known Issues") + +- 🚨 Known issue on Azure OpenAI - We don't recommend upgrading if you use Azure OpenAI. This version failed our Azure OpenAI load test + +## Docker Run LiteLLM Proxy [​](https://docs.litellm.ai/release_notes/tags/snowflake\#docker-run-litellm-proxy "Direct link to Docker Run LiteLLM Proxy") + +```codeBlockLines_e6Vv +docker run +-e STORE_MODEL_IN_DB=True +-p 4000:4000 +ghcr.io/berriai/litellm:main-v1.63.11-stable + +``` + +## Demo Instance [​](https://docs.litellm.ai/release_notes/tags/snowflake\#demo-instance "Direct link to Demo Instance") + +Here's a Demo Instance to test changes: + +- Instance: [https://demo.litellm.ai/](https://demo.litellm.ai/) +- Login Credentials: + - Username: admin + - Password: sk-1234 + +## New Models / Updated Models [​](https://docs.litellm.ai/release_notes/tags/snowflake\#new-models--updated-models "Direct link to New Models / Updated Models") + +- Image Generation support for Amazon Nova Canvas [Getting Started](https://docs.litellm.ai/docs/providers/bedrock#image-generation) +- Add pricing for Jamba new models [PR](https://github.com/BerriAI/litellm/pull/9032/files) +- Add pricing for Amazon EU models [PR](https://github.com/BerriAI/litellm/pull/9056/files) +- Add Bedrock Deepseek R1 model pricing [PR](https://github.com/BerriAI/litellm/pull/9108/files) +- Update Gemini pricing: Gemma 3, Flash 2 thinking update, LearnLM [PR](https://github.com/BerriAI/litellm/pull/9190/files) +- Mark Cohere Embedding 3 models as Multimodal [PR](https://github.com/BerriAI/litellm/pull/9176/commits/c9a576ce4221fc6e50dc47cdf64ab62736c9da41) +- Add Azure Data Zone pricing [PR](https://github.com/BerriAI/litellm/pull/9185/files#diff-19ad91c53996e178c1921cbacadf6f3bae20cfe062bd03ee6bfffb72f847ee37) + - LiteLLM Tracks cost for `azure/eu` and `azure/us` models + +## LLM Translation [​](https://docs.litellm.ai/release_notes/tags/snowflake\#llm-translation "Direct link to LLM Translation") + +![](https://docs.litellm.ai/assets/ideal-img/responses_api.01dd45d.1200.png) + +1. **New Endpoints** + +- \[Beta\] POST `/responses` API. [Getting Started](https://docs.litellm.ai/docs/response_api) + +2. **New LLM Providers** + +- Snowflake Cortex [Getting Started](https://docs.litellm.ai/docs/providers/snowflake) + +3. **New LLM Features** + +- Support OpenRouter `reasoning_content` on streaming [Getting Started](https://docs.litellm.ai/docs/reasoning_content) + +4. **Bug Fixes** + +- OpenAI: Return `code`, `param` and `type` on bad request error [More information on litellm exceptions](https://docs.litellm.ai/docs/exception_mapping) +- Bedrock: Fix converse chunk parsing to only return empty dict on tool use [PR](https://github.com/BerriAI/litellm/pull/9166) +- Bedrock: Support extra\_headers [PR](https://github.com/BerriAI/litellm/pull/9113) +- Azure: Fix Function Calling Bug & Update Default API Version to `2025-02-01-preview` [PR](https://github.com/BerriAI/litellm/pull/9191) +- Azure: Fix AI services URL [PR](https://github.com/BerriAI/litellm/pull/9185) +- Vertex AI: Handle HTTP 201 status code in response [PR](https://github.com/BerriAI/litellm/pull/9193) +- Perplexity: Fix incorrect streaming response [PR](https://github.com/BerriAI/litellm/pull/9081) +- Triton: Fix streaming completions bug [PR](https://github.com/BerriAI/litellm/pull/8386) +- Deepgram: Support bytes.IO when handling audio files for transcription [PR](https://github.com/BerriAI/litellm/pull/9071) +- Ollama: Fix "system" role has become unacceptable [PR](https://github.com/BerriAI/litellm/pull/9261) +- All Providers (Streaming): Fix String `data:` stripped from entire content in streamed responses [PR](https://github.com/BerriAI/litellm/pull/9070) + +## Spend Tracking Improvements [​](https://docs.litellm.ai/release_notes/tags/snowflake\#spend-tracking-improvements "Direct link to Spend Tracking Improvements") + +1. Support Bedrock converse cache token tracking [Getting Started](https://docs.litellm.ai/docs/completion/prompt_caching) +2. Cost Tracking for Responses API [Getting Started](https://docs.litellm.ai/docs/response_api) +3. Fix Azure Whisper cost tracking [Getting Started](https://docs.litellm.ai/docs/audio_transcription) + +## UI [​](https://docs.litellm.ai/release_notes/tags/snowflake\#ui "Direct link to UI") + +### Re-Use Credentials on UI [​](https://docs.litellm.ai/release_notes/tags/snowflake\#re-use-credentials-on-ui "Direct link to Re-Use Credentials on UI") + +You can now onboard LLM provider credentials on LiteLLM UI. Once these credentials are added you can re-use them when adding new models [Getting Started](https://docs.litellm.ai/docs/proxy/ui_credentials) + +### Test Connections before adding models [​](https://docs.litellm.ai/release_notes/tags/snowflake\#test-connections-before-adding-models "Direct link to Test Connections before adding models") + +Before adding a model you can test the connection to the LLM provider to verify you have setup your API Base + API Key correctly + +![](https://docs.litellm.ai/assets/images/litellm_test_connection-029765a2de4dcabccfe3be9a8d33dbdd.gif) + +### General UI Improvements [​](https://docs.litellm.ai/release_notes/tags/snowflake\#general-ui-improvements "Direct link to General UI Improvements") + +1. Add Models Page + - Allow adding Cerebras, Sambanova, Perplexity, Fireworks, Openrouter, TogetherAI Models, Text-Completion OpenAI on Admin UI + - Allow adding EU OpenAI models + - Fix: Instantly show edit + deletes to models +2. Keys Page + - Fix: Instantly show newly created keys on Admin UI (don't require refresh) + - Fix: Allow clicking into Top Keys when showing users Top API Key + - Fix: Allow Filter Keys by Team Alias, Key Alias and Org + - UI Improvements: Show 100 Keys Per Page, Use full height, increase width of key alias +3. Users Page + - Fix: Show correct count of internal user keys on Users Page + - Fix: Metadata not updating in Team UI +4. Logs Page + - UI Improvements: Keep expanded log in focus on LiteLLM UI + - UI Improvements: Minor improvements to logs page + - Fix: Allow internal user to query their own logs + - Allow switching off storing Error Logs in DB [Getting Started](https://docs.litellm.ai/docs/proxy/ui_logs) +5. Sign In/Sign Out + - Fix: Correctly use `PROXY_LOGOUT_URL` when set [Getting Started](https://docs.litellm.ai/docs/proxy/self_serve#setting-custom-logout-urls) + +## Security [​](https://docs.litellm.ai/release_notes/tags/snowflake\#security "Direct link to Security") + +1. Support for Rotating Master Keys [Getting Started](https://docs.litellm.ai/docs/proxy/master_key_rotations) +2. Fix: Internal User Viewer Permissions, don't allow `internal_user_viewer` role to see `Test Key Page` or `Create Key Button` [More information on role based access controls](https://docs.litellm.ai/docs/proxy/access_control) +3. Emit audit logs on All user + model Create/Update/Delete endpoints [Getting Started](https://docs.litellm.ai/docs/proxy/multiple_admins) +4. JWT + - Support multiple JWT OIDC providers [Getting Started](https://docs.litellm.ai/docs/proxy/token_auth) + - Fix JWT access with Groups not working when team is assigned All Proxy Models access +5. Using K/V pairs in 1 AWS Secret [Getting Started](https://docs.litellm.ai/docs/secret#using-kv-pairs-in-1-aws-secret) + +## Logging Integrations [​](https://docs.litellm.ai/release_notes/tags/snowflake\#logging-integrations "Direct link to Logging Integrations") + +1. Prometheus: Track Azure LLM API latency metric [Getting Started](https://docs.litellm.ai/docs/proxy/prometheus#request-latency-metrics) +2. Athina: Added tags, user\_feedback and model\_options to additional\_keys which can be sent to Athina [Getting Started](https://docs.litellm.ai/docs/observability/athina_integration) + +## Performance / Reliability improvements [​](https://docs.litellm.ai/release_notes/tags/snowflake\#performance--reliability-improvements "Direct link to Performance / Reliability improvements") + +1. Redis + litellm router - Fix Redis cluster mode for litellm router [PR](https://github.com/BerriAI/litellm/pull/9010) + +## General Improvements [​](https://docs.litellm.ai/release_notes/tags/snowflake\#general-improvements "Direct link to General Improvements") + +1. OpenWebUI Integration - display `thinking` tokens + +- Guide on getting started with LiteLLM x OpenWebUI. [Getting Started](https://docs.litellm.ai/docs/tutorials/openweb_ui) +- Display `thinking` tokens on OpenWebUI (Bedrock, Anthropic, Deepseek) [Getting Started](https://docs.litellm.ai/docs/tutorials/openweb_ui#render-thinking-content-on-openweb-ui) + +![](https://docs.litellm.ai/assets/images/litellm_thinking_openweb-5ec7dddb7e7b6a10252694c27cfc177d.gif) + +## Complete Git Diff [​](https://docs.litellm.ai/release_notes/tags/snowflake\#complete-git-diff "Direct link to Complete Git Diff") + +[Here's the complete git diff](https://github.com/BerriAI/litellm/compare/v1.63.2-stable...v1.63.11-stable) + diff --git a/docs/my-website/static/llms.txt b/docs/my-website/static/llms.txt new file mode 100644 index 00000000000..a0fa82d2ec6 --- /dev/null +++ b/docs/my-website/static/llms.txt @@ -0,0 +1,52 @@ +# https://docs.litellm.ai/ llms.txt + +- [LiteLLM Overview](https://docs.litellm.ai/): Access and manage 100+ LLMs with LiteLLM tools. +- [Completion Function Guide](https://docs.litellm.ai/completion/input): Guide for using completion function with various models. +- [Litellm Completion Function](https://docs.litellm.ai/completion/output): Learn about the litellm completion function and its output. +- [AI Completion Models](https://docs.litellm.ai/completion/supported): Explore various AI completion models and their requirements. +- [Contact Litellm](https://docs.litellm.ai/contact): Get in touch with Litellm for support and inquiries. +- [Contributing to Documentation](https://docs.litellm.ai/contributing): Guide for contributing to Litellm documentation and setup. +- [Supported Embedding Models](https://docs.litellm.ai/embedding/supported_embedding): Overview of supported embedding models and their requirements. +- [Docusaurus Setup Guide](https://docs.litellm.ai/intro): Quickly learn to set up a Docusaurus site. +- [Callbacks for Data Output](https://docs.litellm.ai/observability/callbacks): Learn to use callbacks for data output integration. +- [Helicone Integration Guide](https://docs.litellm.ai/observability/helicone_integration): Integrate Helicone for logging and proxying LLM requests. +- [Supabase Integration Guide](https://docs.litellm.ai/observability/supabase_integration): Learn to integrate Supabase for logging LLM requests. +- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes): Explore the latest features and improvements in LiteLLM releases. +- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/archive): Comprehensive release notes for LiteLLM updates and features. +- [LiteLLM Release Tags](https://docs.litellm.ai/release_notes/tags): Explore various tags related to LiteLLM release notes. +- [LiteLLM Admin UI Updates](https://docs.litellm.ai/release_notes/tags/admin-ui): Explore LiteLLM's admin UI updates and new features. +- [Alerting Features Updates](https://docs.litellm.ai/release_notes/tags/alerting): Latest updates on alerting features and improvements. +- [LiteLLM Azure Storage Updates](https://docs.litellm.ai/release_notes/tags/azure-storage): Updates on LiteLLM Stable release and Azure Storage support. +- [Batch Processing Updates](https://docs.litellm.ai/release_notes/tags/batch): Updates on models, improvements, and integrations for batch processing. +- [Batches API Features](https://docs.litellm.ai/release_notes/tags/batches): Explore cost tracking, guardrails, and team management features. +- [Budgets and Rate Limits](https://docs.litellm.ai/release_notes/tags/budgets-rate-limits): Manage budgets and rate limits for LiteLLM keys effectively. +- [Claude 3.7 Sonnet Release](https://docs.litellm.ai/release_notes/tags/claude-3-7-sonnet): Release notes for Claude 3.7 Sonnet with updates. +- [Cost Tracking Features](https://docs.litellm.ai/release_notes/tags/cost-tracking): Explore cost tracking features, SCIM integration, and API updates. +- [Credential Management Updates](https://docs.litellm.ai/release_notes/tags/credential-management): Latest updates on credential management and LLM features. +- [Custom Auth Features](https://docs.litellm.ai/release_notes/tags/custom-auth): Explore custom authentication features for team management and cost tracking. +- [LiteLLM v1.65.0 Release](https://docs.litellm.ai/release_notes/tags/custom-prompt-management): New features and improvements in LiteLLM v1.65.0 release. +- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/tags/db-schema): Explore LiteLLM's latest updates and improvements in models. +- [Deepgram Release Notes](https://docs.litellm.ai/release_notes/tags/deepgram): Deepgram integration with speech, vision, and admin features. +- [Dependency Upgrades](https://docs.litellm.ai/release_notes/tags/dependency-upgrades): Dependency upgrades and new model support for LiteLLM. +- [Docker Image Release Notes](https://docs.litellm.ai/release_notes/tags/docker-image): LiteLLM Docker image updates for security and migration. +- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/tags/fallbacks): Updates on LiteLLM Stable release and new features. +- [Finetuning Updates and Improvements](https://docs.litellm.ai/release_notes/tags/finetuning): Explore finetuning updates, model improvements, and integrations. +- [Fireworks AI Updates](https://docs.litellm.ai/release_notes/tags/fireworks-ai): New features and updates for Fireworks AI models and tools. +- [Guardrails and Logging Updates](https://docs.litellm.ai/release_notes/tags/guardrails): Explore new guardrail features, logging, and model updates. +- [LLM Features and Updates](https://docs.litellm.ai/release_notes/tags/humanloop): Updates on models, integrations, and improvements in LLM features. +- [Key Management Overview](https://docs.litellm.ai/release_notes/tags/key-management): Manage keys, budgets, logging, and guardrails effectively. +- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/tags/langfuse): Explore new models, improvements, and integrations in LiteLLM. +- [LLM Translation Updates](https://docs.litellm.ai/release_notes/tags/llm-translation): Latest LLM translation updates and UI improvements released. +- [LiteLLM Logging Updates](https://docs.litellm.ai/release_notes/tags/logging): Explore LiteLLM logging updates, features, and improvements. +- [Management Endpoints Updates](https://docs.litellm.ai/release_notes/tags/management-endpoints): Updates on management endpoints for team model handling. +- [MCP Support Updates](https://docs.litellm.ai/release_notes/tags/mcp): MCP support and usage analytics enhancements in LiteLLM. +- [LiteLLM New Features](https://docs.litellm.ai/release_notes/tags/new-models): Explore new features, models, and updates for LiteLLM. +- [Prometheus Integration Updates](https://docs.litellm.ai/release_notes/tags/prometheus): Explore new features and improvements in Prometheus integration. +- [Prompt Management Updates](https://docs.litellm.ai/release_notes/tags/prompt-management): Explore prompt management updates, model improvements, and integrations. +- [LLM Translation Updates](https://docs.litellm.ai/release_notes/tags/reasoning-content): Release notes detailing LLM translation and UI improvements. +- [Release Notes Overview](https://docs.litellm.ai/release_notes/tags/rerank): Latest release notes on LLM translation and UI improvements. +- [Responses API Release Notes](https://docs.litellm.ai/release_notes/tags/responses-api): Explore the latest updates and features of the Responses API. +- [Secret Management Updates](https://docs.litellm.ai/release_notes/tags/secret-management): Enhancements in secret management, alerting, and model updates. +- [LiteLLM Security Updates](https://docs.litellm.ai/release_notes/tags/security): Security updates and features for LiteLLM deployment and management. +- [Session Management Updates](https://docs.litellm.ai/release_notes/tags/session-management): Enhancements in session management and user handling features. +- [LiteLLM Release Notes](https://docs.litellm.ai/release_notes/tags/snowflake): Latest updates on LiteLLM features and improvements. diff --git a/enterprise/dist/litellm_enterprise-0.1.1-py3-none-any.whl b/enterprise/dist/litellm_enterprise-0.1.1-py3-none-any.whl new file mode 100644 index 00000000000..d9a8ef41e62 Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.1-py3-none-any.whl differ diff --git a/enterprise/dist/litellm_enterprise-0.1.1.tar.gz b/enterprise/dist/litellm_enterprise-0.1.1.tar.gz new file mode 100644 index 00000000000..98cf132b213 Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.1.tar.gz differ diff --git a/enterprise/dist/litellm_enterprise-0.1.2-py3-none-any.whl b/enterprise/dist/litellm_enterprise-0.1.2-py3-none-any.whl new file mode 100644 index 00000000000..1f75e0f1b5c Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.2-py3-none-any.whl differ diff --git a/enterprise/dist/litellm_enterprise-0.1.2.tar.gz b/enterprise/dist/litellm_enterprise-0.1.2.tar.gz new file mode 100644 index 00000000000..b6fa4dd5f7b Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.2.tar.gz differ diff --git a/enterprise/dist/litellm_enterprise-0.1.3-py3-none-any.whl b/enterprise/dist/litellm_enterprise-0.1.3-py3-none-any.whl new file mode 100644 index 00000000000..7b5cb856566 Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.3-py3-none-any.whl differ diff --git a/enterprise/dist/litellm_enterprise-0.1.3.tar.gz b/enterprise/dist/litellm_enterprise-0.1.3.tar.gz new file mode 100644 index 00000000000..d5ac9f26a47 Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.3.tar.gz differ diff --git a/enterprise/dist/litellm_enterprise-0.1.4-py3-none-any.whl b/enterprise/dist/litellm_enterprise-0.1.4-py3-none-any.whl new file mode 100644 index 00000000000..f862a55b18b Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.4-py3-none-any.whl differ diff --git a/enterprise/dist/litellm_enterprise-0.1.4.tar.gz b/enterprise/dist/litellm_enterprise-0.1.4.tar.gz new file mode 100644 index 00000000000..bf1b3ec57c1 Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.4.tar.gz differ diff --git a/enterprise/dist/litellm_enterprise-0.1.5-py3-none-any.whl b/enterprise/dist/litellm_enterprise-0.1.5-py3-none-any.whl new file mode 100644 index 00000000000..661638db3f9 Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.5-py3-none-any.whl differ diff --git a/enterprise/dist/litellm_enterprise-0.1.5.tar.gz b/enterprise/dist/litellm_enterprise-0.1.5.tar.gz new file mode 100644 index 00000000000..2808574ddac Binary files /dev/null and b/enterprise/dist/litellm_enterprise-0.1.5.tar.gz differ diff --git a/enterprise/enterprise_callbacks/generic_api_callback.py b/enterprise/enterprise_callbacks/generic_api_callback.py deleted file mode 100644 index 2f39ce856b7..00000000000 --- a/enterprise/enterprise_callbacks/generic_api_callback.py +++ /dev/null @@ -1,126 +0,0 @@ -# callback to make a request to an API endpoint - -#### What this does #### -# On success, logs events to Promptlayer -import os - - -from typing import Optional - -import traceback - - -#### What this does #### -# On success + failure, log events to Supabase - -import litellm -import uuid -from litellm._logging import print_verbose, verbose_logger - - -class GenericAPILogger: - # Class variables or attributes - def __init__(self, endpoint: Optional[str] = None, headers: Optional[dict] = None): - try: - if endpoint is None: - # check env for "GENERIC_LOGGER_ENDPOINT" - if os.getenv("GENERIC_LOGGER_ENDPOINT"): - # Do something with the endpoint - endpoint = os.getenv("GENERIC_LOGGER_ENDPOINT") - else: - # Handle the case when the endpoint is not found in the environment variables - raise ValueError( - "endpoint not set for GenericAPILogger, GENERIC_LOGGER_ENDPOINT not found in environment variables" - ) - headers = headers or litellm.generic_logger_headers - - if endpoint is None: - raise ValueError("endpoint not set for GenericAPILogger") - if headers is None: - raise ValueError("headers not set for GenericAPILogger") - - self.endpoint = endpoint - self.headers = headers - - verbose_logger.debug( - f"in init GenericAPILogger, endpoint {self.endpoint}, headers {self.headers}" - ) - - pass - - except Exception as e: - print_verbose(f"Got exception on init GenericAPILogger client {str(e)}") - raise e - - # This is sync, because we run this in a separate thread. Running in a sepearate thread ensures it will never block an LLM API call - # Experience with s3, Langfuse shows that async logging events are complicated and can block LLM calls - def log_event( - self, kwargs, response_obj, start_time, end_time, user_id, print_verbose - ): - try: - verbose_logger.debug( - f"GenericAPILogger Logging - Enters logging function for model {kwargs}" - ) - - # construct payload to send custom logger - # follows the same params as langfuse.py - litellm_params = kwargs.get("litellm_params", {}) - metadata = ( - litellm_params.get("metadata", {}) or {} - ) # if litellm_params['metadata'] == None - messages = kwargs.get("messages") - cost = kwargs.get("response_cost", 0.0) - optional_params = kwargs.get("optional_params", {}) - call_type = kwargs.get("call_type", "litellm.completion") - cache_hit = kwargs.get("cache_hit", False) - usage = response_obj["usage"] - id = response_obj.get("id", str(uuid.uuid4())) - - # Build the initial payload - payload = { - "id": id, - "call_type": call_type, - "cache_hit": cache_hit, - "startTime": start_time, - "endTime": end_time, - "model": kwargs.get("model", ""), - "user": kwargs.get("user", ""), - "modelParameters": optional_params, - "messages": messages, - "response": response_obj, - "usage": usage, - "metadata": metadata, - "cost": cost, - } - - # Ensure everything in the payload is converted to str - for key, value in payload.items(): - try: - payload[key] = str(value) - except Exception: - # non blocking if it can't cast to a str - pass - - import json - - data = { - "data": payload, - } - data = json.dumps(data) - print_verbose(f"\nGeneric Logger - Logging payload = {data}") - - # make request to endpoint with payload - response = litellm.module_level_client.post( - self.endpoint, json=data, headers=self.headers - ) - - response_status = response.status_code - response_text = response.text - - print_verbose( - f"Generic Logger - final response status = {response_status}, response text = {response_text}" - ) - return response - except Exception as e: - verbose_logger.error(f"Generic - {str(e)}\n{traceback.format_exc()}") - pass diff --git a/enterprise/enterprise_hooks/__init__.py b/enterprise/enterprise_hooks/__init__.py new file mode 100644 index 00000000000..830d97886a6 --- /dev/null +++ b/enterprise/enterprise_hooks/__init__.py @@ -0,0 +1,29 @@ +import os +from typing import Dict, Literal, Type, Union + +from litellm.integrations.custom_logger import CustomLogger + +from .managed_files import _PROXY_LiteLLMManagedFiles + +ENTERPRISE_PROXY_HOOKS: Dict[str, Type[CustomLogger]] = { + "managed_files": _PROXY_LiteLLMManagedFiles, +} + + +def get_enterprise_proxy_hook( + hook_name: Union[ + Literal[ + "managed_files", + "max_parallel_requests", + ], + str, + ] +): + """ + Factory method to get a enterprise hook instance by name + """ + if hook_name not in ENTERPRISE_PROXY_HOOKS: + raise ValueError( + f"Unknown hook: {hook_name}. Available hooks: {list(ENTERPRISE_PROXY_HOOKS.keys())}" + ) + return ENTERPRISE_PROXY_HOOKS[hook_name] diff --git a/litellm/proxy/hooks/managed_files.py b/enterprise/enterprise_hooks/managed_files.py similarity index 66% rename from litellm/proxy/hooks/managed_files.py rename to enterprise/enterprise_hooks/managed_files.py index 9ac6cc580b7..0dc86294d36 100644 --- a/litellm/proxy/hooks/managed_files.py +++ b/enterprise/enterprise_hooks/managed_files.py @@ -4,14 +4,18 @@ import base64 import json import uuid -from abc import ABC, abstractmethod from typing import TYPE_CHECKING, Any, Dict, List, Literal, Optional, Union, cast from litellm import Router, verbose_logger from litellm.caching.caching import DualCache from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.prompt_templates.common_utils import extract_file_data +from litellm.llms.base_llm.files.transformation import BaseFileEndpoints from litellm.proxy._types import CallTypes, LiteLLM_ManagedFileTable, UserAPIKeyAuth +from litellm.proxy.openai_files_endpoints.common_utils import ( + _is_base64_encoded_unified_file_id, + convert_b64_uid_to_unified_uid, +) from litellm.types.llms.openai import ( AllMessageValues, ChatCompletionFileObject, @@ -19,7 +23,7 @@ from litellm.types.llms.openai import ( OpenAIFileObject, OpenAIFilesPurpose, ) -from litellm.types.utils import SpecialEnums +from litellm.types.utils import LiteLLMBatch, LLMResponseTypes, SpecialEnums if TYPE_CHECKING: from opentelemetry.trace import Span as _Span @@ -36,29 +40,7 @@ else: PrismaClient = Any -class BaseFileEndpoints(ABC): - @abstractmethod - async def afile_retrieve( - self, - file_id: str, - litellm_parent_otel_span: Optional[Span], - ) -> OpenAIFileObject: - pass - - @abstractmethod - async def afile_list( - self, custom_llm_provider: str, **data: dict - ) -> List[OpenAIFileObject]: - pass - - @abstractmethod - async def afile_delete( - self, custom_llm_provider: str, file_id: str, **data: dict - ) -> OpenAIFileObject: - pass - - -class _PROXY_LiteLLMManagedFiles(CustomLogger): +class _PROXY_LiteLLMManagedFiles(CustomLogger, BaseFileEndpoints): # Class variables or attributes def __init__( self, internal_usage_cache: InternalUsageCache, prisma_client: PrismaClient @@ -153,6 +135,9 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger): "audio_transcription", "pass_through_endpoint", "rerank", + "acreate_batch", + "aretrieve_batch", + "afile_content", ], ) -> Union[Exception, str, Dict, None]: """ @@ -169,9 +154,72 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger): ) data["model_file_id_mapping"] = model_file_id_mapping + elif call_type == CallTypes.afile_content.value: + retrieve_file_id = cast(Optional[str], data.get("file_id")) + potential_file_id = ( + _is_base64_encoded_unified_file_id(retrieve_file_id) + if retrieve_file_id + else False + ) + if potential_file_id: + model_id = self.get_model_id_from_unified_file_id(potential_file_id) + if model_id: + data["model"] = model_id + data["file_id"] = self.get_output_file_id_from_unified_file_id( + potential_file_id + ) + elif call_type == CallTypes.acreate_batch.value: + input_file_id = cast(Optional[str], data.get("input_file_id")) + if input_file_id: + model_file_id_mapping = await self.get_model_file_id_mapping( + [input_file_id], user_api_key_dict.parent_otel_span + ) + + data["model_file_id_mapping"] = model_file_id_mapping + elif call_type == CallTypes.aretrieve_batch.value: + retrieve_batch_id = cast(Optional[str], data.get("batch_id")) + potential_batch_id = ( + _is_base64_encoded_unified_file_id(retrieve_batch_id) + if retrieve_batch_id + else False + ) + if potential_batch_id: + ## for managed batch id - get the model id + potential_model_id = self.get_model_id_from_unified_batch_id( + potential_batch_id + ) + if potential_model_id is None: + raise Exception( + f"LiteLLM Managed Batch ID with id={retrieve_batch_id} is invalid - does not contain encoded model_id." + ) + data["model"] = potential_model_id + data["batch_id"] = self.get_batch_id_from_unified_batch_id( + potential_batch_id + ) return data + async def async_pre_call_deployment_hook( + self, kwargs: Dict[str, Any], call_type: Optional[CallTypes] + ) -> Optional[dict]: + """ + Allow modifying the request just before it's sent to the deployment. + """ + if call_type and call_type == CallTypes.acreate_batch: + input_file_id = cast(Optional[str], kwargs.get("input_file_id")) + model_file_id_mapping = cast( + Optional[Dict[str, Dict[str, str]]], kwargs.get("model_file_id_mapping") + ) + model_id = cast(Optional[str], kwargs.get("model_info", {}).get("id", None)) + mapped_file_id: Optional[str] = None + if input_file_id and model_file_id_mapping and model_id: + mapped_file_id = model_file_id_mapping.get(input_file_id, {}).get( + model_id, None + ) + if mapped_file_id: + kwargs["input_file_id"] = mapped_file_id + return kwargs + def get_file_ids_from_messages(self, messages: List[AllMessageValues]) -> List[str]: """ Gets file ids from messages @@ -192,37 +240,6 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger): file_ids.append(file_id) return file_ids - @staticmethod - def _convert_b64_uid_to_unified_uid(b64_uid: str) -> str: - is_base64_unified_file_id = ( - _PROXY_LiteLLMManagedFiles._is_base64_encoded_unified_file_id(b64_uid) - ) - if is_base64_unified_file_id: - return is_base64_unified_file_id - else: - return b64_uid - - @staticmethod - def _is_base64_encoded_unified_file_id(b64_uid: str) -> Union[str, Literal[False]]: - # Add padding back if needed - padded = b64_uid + "=" * (-len(b64_uid) % 4) - # Decode from base64 - try: - decoded = base64.urlsafe_b64decode(padded).decode() - if decoded.startswith(SpecialEnums.LITELM_MANAGED_FILE_ID_PREFIX.value): - return decoded - else: - return False - except Exception: - return False - - def convert_b64_uid_to_unified_uid(self, b64_uid: str) -> str: - is_base64_unified_file_id = self._is_base64_encoded_unified_file_id(b64_uid) - if is_base64_unified_file_id: - return is_base64_unified_file_id - else: - return b64_uid - async def get_model_file_id_mapping( self, file_ids: List[str], litellm_parent_otel_span: Span ) -> dict: @@ -247,7 +264,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger): for file_id in file_ids: ## CHECK IF FILE ID IS MANAGED BY LITELM - is_base64_unified_file_id = self._is_base64_encoded_unified_file_id(file_id) + is_base64_unified_file_id = _is_base64_encoded_unified_file_id(file_id) if is_base64_unified_file_id: litellm_managed_file_ids.append(file_id) @@ -300,6 +317,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger): create_file_request=create_file_request, internal_usage_cache=self.internal_usage_cache, litellm_parent_otel_span=litellm_parent_otel_span, + target_model_names_list=target_model_names_list, ) ## STORE MODEL MAPPINGS IN DB @@ -328,14 +346,22 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger): create_file_request: CreateFileRequest, internal_usage_cache: InternalUsageCache, litellm_parent_otel_span: Span, + target_model_names_list: List[str], ) -> OpenAIFileObject: ## GET THE FILE TYPE FROM THE CREATE FILE REQUEST file_data = extract_file_data(create_file_request["file"]) file_type = file_data["content_type"] + output_file_id = file_objects[0].id + model_id = file_objects[0]._hidden_params.get("model_id") + unified_file_id = SpecialEnums.LITELLM_MANAGED_FILE_COMPLETE_STR.value.format( - file_type, str(uuid.uuid4()) + file_type, + str(uuid.uuid4()), + ",".join(target_model_names_list), + output_file_id, + model_id, ) # Convert to URL-safe base64 and strip padding @@ -357,6 +383,83 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger): return response + def get_unified_batch_id(self, batch_id: str, model_id: str) -> str: + unified_batch_id = SpecialEnums.LITELLM_MANAGED_BATCH_COMPLETE_STR.value.format( + model_id, batch_id + ) + return base64.urlsafe_b64encode(unified_batch_id.encode()).decode().rstrip("=") + + def get_unified_output_file_id( + self, output_file_id: str, model_id: str, model_name: str + ) -> str: + unified_output_file_id = ( + SpecialEnums.LITELLM_MANAGED_FILE_COMPLETE_STR.value.format( + "application/json", + str(uuid.uuid4()), + model_name, + output_file_id, + model_id, + ) + ) + return ( + base64.urlsafe_b64encode(unified_output_file_id.encode()) + .decode() + .rstrip("=") + ) + + def get_model_id_from_unified_file_id(self, file_id: str) -> str: + return file_id.split("llm_output_file_model_id,")[1].split(";")[0] + + def get_output_file_id_from_unified_file_id(self, file_id: str) -> str: + return file_id.split("llm_output_file_id,")[1].split(";")[0] + + def get_model_id_from_unified_batch_id(self, file_id: str) -> Optional[str]: + """ + Get the model_id from the file_id + + Expected format: litellm_proxy;model_id:{};llm_batch_id:{};llm_output_file_id:{} + """ + ## use regex to get the model_id from the file_id + try: + return file_id.split("model_id:")[1].split(";")[0] + except Exception: + return None + + def get_batch_id_from_unified_batch_id(self, file_id: str) -> str: + ## use regex to get the batch_id from the file_id + return file_id.split("llm_batch_id:")[1].split(",")[0] + + async def async_post_call_success_hook( + self, data: Dict, user_api_key_dict: UserAPIKeyAuth, response: LLMResponseTypes + ) -> Any: + if isinstance(response, LiteLLMBatch): + ## Check if unified_file_id is in the response + unified_file_id = response._hidden_params.get( + "unified_file_id" + ) # managed file id + unified_batch_id = response._hidden_params.get( + "unified_batch_id" + ) # managed batch id + model_id = cast(Optional[str], response._hidden_params.get("model_id")) + model_name = cast(Optional[str], response._hidden_params.get("model_name")) + if (unified_batch_id or unified_file_id) and model_id: + response.id = self.get_unified_batch_id( + batch_id=response.id, model_id=model_id + ) + + if ( + response.output_file_id and model_name and model_id + ): # return a file id with the model_id and output_file_id + response.output_file_id = self.get_unified_output_file_id( + output_file_id=response.output_file_id, + model_id=model_id, + model_name=model_name, + ) + + return await super().async_post_call_success_hook( + data, user_api_key_dict, response + ) + async def afile_retrieve( self, file_id: str, litellm_parent_otel_span: Optional[Span] ) -> OpenAIFileObject: @@ -383,7 +486,7 @@ class _PROXY_LiteLLMManagedFiles(CustomLogger): llm_router: Router, **data: Dict, ) -> OpenAIFileObject: - file_id = self.convert_b64_uid_to_unified_uid(file_id) + file_id = convert_b64_uid_to_unified_uid(file_id) model_file_id_mapping = await self.get_model_file_id_mapping( [file_id], litellm_parent_otel_span ) diff --git a/enterprise/enterprise_hooks/secrets_plugins/__init__.py b/enterprise/litellm_enterprise/__init__.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/__init__.py rename to enterprise/litellm_enterprise/__init__.py diff --git a/enterprise/enterprise_callbacks/example_logging_api.py b/enterprise/litellm_enterprise/enterprise_callbacks/example_logging_api.py similarity index 83% rename from enterprise/enterprise_callbacks/example_logging_api.py rename to enterprise/litellm_enterprise/enterprise_callbacks/example_logging_api.py index 2084ffb548e..14d34f5d1e8 100644 --- a/enterprise/enterprise_callbacks/example_logging_api.py +++ b/enterprise/litellm_enterprise/enterprise_callbacks/example_logging_api.py @@ -7,11 +7,11 @@ app = FastAPI() @app.post("/log-event") async def log_event(request: Request): try: - print("Received /log-event request") # noqa + print("Received /log-event request") # noqa # Assuming the incoming request has JSON data data = await request.json() - print("Received request data:") # noqa - print(data) # noqa + print("Received request data:") # noqa + print(data) # noqa # Your additional logic can go here # For now, just printing the received data diff --git a/enterprise/litellm_enterprise/enterprise_callbacks/generic_api_callback.py b/enterprise/litellm_enterprise/enterprise_callbacks/generic_api_callback.py new file mode 100644 index 00000000000..d239be41257 --- /dev/null +++ b/enterprise/litellm_enterprise/enterprise_callbacks/generic_api_callback.py @@ -0,0 +1,266 @@ +""" +Callback to log events to a Generic API Endpoint + +- Creates a StandardLoggingPayload +- Adds to batch queue +- Flushes based on CustomBatchLogger settings +""" + +import asyncio +import os +import traceback +import uuid +from typing import Dict, List, Optional, Union + +import litellm +from litellm._logging import verbose_logger +from litellm.integrations.custom_batch_logger import CustomBatchLogger +from litellm.litellm_core_utils.safe_json_dumps import safe_dumps +from litellm.llms.custom_httpx.http_handler import ( + get_async_httpx_client, + httpxSpecialProvider, +) +from litellm.types.utils import StandardLoggingPayload + + +class GenericAPILogger(CustomBatchLogger): + def __init__( + self, + endpoint: Optional[str] = None, + headers: Optional[dict] = None, + **kwargs, + ): + """ + Initialize the GenericAPILogger + + Args: + endpoint: Optional[str] = None, + headers: Optional[dict] = None, + """ + ######################################################### + # Init httpx client + ######################################################### + self.async_httpx_client = get_async_httpx_client( + llm_provider=httpxSpecialProvider.LoggingCallback + ) + endpoint = endpoint or os.getenv("GENERIC_LOGGER_ENDPOINT") + if endpoint is None: + raise ValueError( + "endpoint not set for GenericAPILogger, GENERIC_LOGGER_ENDPOINT not found in environment variables" + ) + + self.headers: Dict = self._get_headers(headers) + self.endpoint: str = endpoint + verbose_logger.debug( + f"in init GenericAPILogger, endpoint {self.endpoint}, headers {self.headers}" + ) + + ######################################################### + # Init variables for batch flushing logs + ######################################################### + self.flush_lock = asyncio.Lock() + super().__init__(**kwargs, flush_lock=self.flush_lock) + asyncio.create_task(self.periodic_flush()) + self.log_queue: List[Union[Dict, StandardLoggingPayload]] = [] + + def _get_headers(self, headers: Optional[dict] = None): + """ + Get headers for the Generic API Logger + + Returns: + Dict: Headers for the Generic API Logger + + Args: + headers: Optional[dict] = None + """ + # Process headers from different sources + headers_dict = { + "Content-Type": "application/json", + } + + # 1. First check for headers from env var + env_headers = os.getenv("GENERIC_LOGGER_HEADERS") + if env_headers: + try: + # Parse headers in format "key1=value1,key2=value2" or "key1=value1" + header_items = env_headers.split(",") + for item in header_items: + if "=" in item: + key, value = item.split("=", 1) + headers_dict[key.strip()] = value.strip() + except Exception as e: + verbose_logger.warning( + f"Error parsing headers from environment variables: {str(e)}" + ) + + # 2. Update with litellm generic headers if available + if litellm.generic_logger_headers: + headers_dict.update(litellm.generic_logger_headers) + + # 3. Override with directly provided headers if any + if headers: + headers_dict.update(headers) + + return headers_dict + + async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): + """ + Async Log success events to Generic API Endpoint + + - Creates a StandardLoggingPayload + - Adds to batch queue + - Flushes based on CustomBatchLogger settings + + Raises: + Raises a NON Blocking verbose_logger.exception if an error occurs + """ + from litellm.proxy.utils import _premium_user_check + + _premium_user_check() + + try: + verbose_logger.debug( + "Generic API Logger - Enters logging function for model %s", kwargs + ) + standard_logging_payload = kwargs.get("standard_logging_object", None) + + # Backwards compatibility with old logging payload + if litellm.generic_api_use_v1 is True: + payload = self._get_v1_logging_payload( + kwargs=kwargs, + response_obj=response_obj, + start_time=start_time, + end_time=end_time, + ) + self.log_queue.append(payload) + else: + # New logging payload, StandardLoggingPayload + self.log_queue.append(standard_logging_payload) + + if len(self.log_queue) >= self.batch_size: + await self.async_send_batch() + + except Exception as e: + verbose_logger.exception( + f"Generic API Logger Error - {str(e)}\n{traceback.format_exc()}" + ) + pass + + async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): + """ + Async Log failure events to Generic API Endpoint + + - Creates a StandardLoggingPayload + - Adds to batch queue + """ + from litellm.proxy.utils import _premium_user_check + + _premium_user_check() + + try: + verbose_logger.debug( + "Generic API Logger - Enters logging function for model %s", kwargs + ) + standard_logging_payload = kwargs.get("standard_logging_object", None) + + if litellm.generic_api_use_v1 is True: + payload = self._get_v1_logging_payload( + kwargs=kwargs, + response_obj=response_obj, + start_time=start_time, + end_time=end_time, + ) + self.log_queue.append(payload) + else: + self.log_queue.append(standard_logging_payload) + + if len(self.log_queue) >= self.batch_size: + await self.async_send_batch() + + except Exception as e: + verbose_logger.exception( + f"Generic API Logger Error - {str(e)}\n{traceback.format_exc()}" + ) + + async def async_send_batch(self): + """ + Sends the batch of messages to Generic API Endpoint + """ + try: + if not self.log_queue: + return + + verbose_logger.debug( + f"Generic API Logger - about to flush {len(self.log_queue)} events" + ) + + # make POST request to Generic API Endpoint + response = await self.async_httpx_client.post( + url=self.endpoint, + headers=self.headers, + data=safe_dumps(self.log_queue), + ) + + verbose_logger.debug( + f"Generic API Logger - sent batch to {self.endpoint}, status code {response.status_code}" + ) + + except Exception as e: + verbose_logger.exception( + f"Generic API Logger Error sending batch - {str(e)}\n{traceback.format_exc()}" + ) + finally: + self.log_queue.clear() + + def _get_v1_logging_payload( + self, kwargs, response_obj, start_time, end_time + ) -> dict: + """ + Maintained for backwards compatibility with old logging payload + + Returns a dict of the payload to send to the Generic API Endpoint + """ + verbose_logger.debug( + f"GenericAPILogger Logging - Enters logging function for model {kwargs}" + ) + + # construct payload to send custom logger + # follows the same params as langfuse.py + litellm_params = kwargs.get("litellm_params", {}) + metadata = ( + litellm_params.get("metadata", {}) or {} + ) # if litellm_params['metadata'] == None + messages = kwargs.get("messages") + cost = kwargs.get("response_cost", 0.0) + optional_params = kwargs.get("optional_params", {}) + call_type = kwargs.get("call_type", "litellm.completion") + cache_hit = kwargs.get("cache_hit", False) + usage = response_obj["usage"] + id = response_obj.get("id", str(uuid.uuid4())) + + # Build the initial payload + payload = { + "id": id, + "call_type": call_type, + "cache_hit": cache_hit, + "startTime": start_time, + "endTime": end_time, + "model": kwargs.get("model", ""), + "user": kwargs.get("user", ""), + "modelParameters": optional_params, + "messages": messages, + "response": response_obj, + "usage": usage, + "metadata": metadata, + "cost": cost, + } + + # Ensure everything in the payload is converted to str + for key, value in payload.items(): + try: + payload[key] = str(value) + except Exception: + # non blocking if it can't cast to a str + pass + + return payload diff --git a/enterprise/enterprise_hooks/llama_guard.py b/enterprise/litellm_enterprise/enterprise_callbacks/llama_guard.py similarity index 97% rename from enterprise/enterprise_hooks/llama_guard.py rename to enterprise/litellm_enterprise/enterprise_callbacks/llama_guard.py index 2c53fafa5b6..a2d77f51a49 100644 --- a/enterprise/enterprise_hooks/llama_guard.py +++ b/enterprise/litellm_enterprise/enterprise_callbacks/llama_guard.py @@ -7,24 +7,23 @@ # +-------------------------------------------------------------+ # Thank you users! We ❤️ you! - Krrish & Ishaan -import sys import os +import sys from collections.abc import Iterable sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path -from typing import Optional, Literal -import litellm import sys -from litellm.proxy._types import UserAPIKeyAuth -from litellm.integrations.custom_logger import CustomLogger +from typing import Literal, Optional + from fastapi import HTTPException + +import litellm from litellm._logging import verbose_proxy_logger -from litellm.types.utils import ( - ModelResponse, - Choices, -) +from litellm.integrations.custom_logger import CustomLogger +from litellm.proxy._types import UserAPIKeyAuth +from litellm.types.utils import Choices, ModelResponse litellm.set_verbose = True diff --git a/enterprise/enterprise_hooks/llm_guard.py b/enterprise/litellm_enterprise/enterprise_callbacks/llm_guard.py similarity index 99% rename from enterprise/enterprise_hooks/llm_guard.py rename to enterprise/litellm_enterprise/enterprise_callbacks/llm_guard.py index 934646acb0e..59981154aa5 100644 --- a/enterprise/enterprise_hooks/llm_guard.py +++ b/enterprise/litellm_enterprise/enterprise_callbacks/llm_guard.py @@ -7,15 +7,17 @@ # Thank you users! We ❤️ you! - Krrish & Ishaan ## This provides an LLM Guard Integration for content moderation on the proxy -from typing import Optional, Literal -import litellm -from litellm.proxy._types import UserAPIKeyAuth -from litellm.integrations.custom_logger import CustomLogger -from fastapi import HTTPException -from litellm._logging import verbose_proxy_logger +from typing import Literal, Optional + import aiohttp -from litellm.utils import get_formatted_prompt +from fastapi import HTTPException + +import litellm +from litellm._logging import verbose_proxy_logger +from litellm.integrations.custom_logger import CustomLogger +from litellm.proxy._types import UserAPIKeyAuth from litellm.secret_managers.main import get_secret_str +from litellm.utils import get_formatted_prompt litellm.set_verbose = True diff --git a/litellm/integrations/pagerduty/pagerduty.py b/enterprise/litellm_enterprise/enterprise_callbacks/pagerduty/pagerduty.py similarity index 93% rename from litellm/integrations/pagerduty/pagerduty.py rename to enterprise/litellm_enterprise/enterprise_callbacks/pagerduty/pagerduty.py index 6085bc237ae..773c34401df 100644 --- a/litellm/integrations/pagerduty/pagerduty.py +++ b/enterprise/litellm_enterprise/enterprise_callbacks/pagerduty/pagerduty.py @@ -4,6 +4,10 @@ PagerDuty Alerting Integration Handles two types of alerts: - High LLM API Failure Rate. Configure X fails in Y seconds to trigger an alert. - High Number of Hanging LLM Requests. Configure X hangs in Y seconds to trigger an alert. + +Note: This is a Free feature on the regular litellm docker image. + +However, this is under the enterprise license """ import asyncio @@ -46,8 +50,6 @@ class PagerDutyAlerting(SlackAlerting): def __init__( self, alerting_args: Optional[Union[AlertingConfig, dict]] = None, **kwargs ): - from litellm.proxy.proxy_server import CommonProxyErrors, premium_user - super().__init__() _api_key = os.getenv("PAGERDUTY_API_KEY") if not _api_key: @@ -55,7 +57,7 @@ class PagerDutyAlerting(SlackAlerting): self.api_key: str = _api_key alerting_args = alerting_args or {} - self.alerting_args: AlertingConfig = AlertingConfig( + self.pagerduty_alerting_args: AlertingConfig = AlertingConfig( failure_threshold=alerting_args.get( "failure_threshold", PAGERDUTY_DEFAULT_FAILURE_THRESHOLD ), @@ -76,12 +78,6 @@ class PagerDutyAlerting(SlackAlerting): self._failure_events: List[PagerDutyInternalEvent] = [] self._hanging_events: List[PagerDutyInternalEvent] = [] - # premium user check - if premium_user is not True: - raise ValueError( - f"PagerDutyAlerting is only available for LiteLLM Enterprise users. {CommonProxyErrors.not_premium_user.value}" - ) - # ------------------ MAIN LOGIC ------------------ # async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): @@ -123,8 +119,10 @@ class PagerDutyAlerting(SlackAlerting): ) # Prune + Possibly alert - window_seconds = self.alerting_args.get("failure_threshold_window_seconds", 60) - threshold = self.alerting_args.get("failure_threshold", 1) + window_seconds = self.pagerduty_alerting_args.get( + "failure_threshold_window_seconds", 60 + ) + threshold = self.pagerduty_alerting_args.get("failure_threshold", 1) # If threshold is crossed, send PD alert for failures await self._send_alert_if_thresholds_crossed( @@ -170,10 +168,10 @@ class PagerDutyAlerting(SlackAlerting): If not, we classify it as a hanging request. """ verbose_logger.debug( - f"Inside Hanging Response Handler!..sleeping for {self.alerting_args.get('hanging_threshold_seconds', PAGERDUTY_DEFAULT_HANGING_THRESHOLD_SECONDS)} seconds" + f"Inside Hanging Response Handler!..sleeping for {self.pagerduty_alerting_args.get('hanging_threshold_seconds', PAGERDUTY_DEFAULT_HANGING_THRESHOLD_SECONDS)} seconds" ) await asyncio.sleep( - self.alerting_args.get( + self.pagerduty_alerting_args.get( "hanging_threshold_seconds", PAGERDUTY_DEFAULT_HANGING_THRESHOLD_SECONDS ) ) @@ -201,11 +199,11 @@ class PagerDutyAlerting(SlackAlerting): ) # Prune + Possibly alert - window_seconds = self.alerting_args.get( + window_seconds = self.pagerduty_alerting_args.get( "hanging_threshold_window_seconds", PAGERDUTY_DEFAULT_HANGING_THRESHOLD_WINDOW_SECONDS, ) - threshold: int = self.alerting_args.get( + threshold: int = self.pagerduty_alerting_args.get( "hanging_threshold_fails", PAGERDUTY_DEFAULT_HANGING_THRESHOLD_SECONDS ) diff --git a/enterprise/enterprise_hooks/secret_detection.py b/enterprise/litellm_enterprise/enterprise_callbacks/secret_detection.py similarity index 99% rename from enterprise/enterprise_hooks/secret_detection.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secret_detection.py index 158f26efa30..8a7a82df686 100644 --- a/enterprise/enterprise_hooks/secret_detection.py +++ b/enterprise/litellm_enterprise/enterprise_callbacks/secret_detection.py @@ -5,18 +5,19 @@ # +-------------------------------------------------------------+ # Thank you users! We ❤️ you! - Krrish & Ishaan -import sys import os +import sys sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path -from typing import Optional -from litellm.caching.caching import DualCache -from litellm.proxy._types import UserAPIKeyAuth -from litellm._logging import verbose_proxy_logger import tempfile +from typing import Optional + +from litellm._logging import verbose_proxy_logger +from litellm.caching.caching import DualCache from litellm.integrations.custom_guardrail import CustomGuardrail +from litellm.proxy._types import UserAPIKeyAuth GUARDRAIL_NAME = "hide_secrets" diff --git a/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/__init__.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/enterprise/enterprise_hooks/secrets_plugins/adafruit.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/adafruit.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/adafruit.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/adafruit.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/adobe.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/adobe.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/adobe.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/adobe.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/age_secret_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/age_secret_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/age_secret_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/age_secret_key.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/airtable_api_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/airtable_api_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/airtable_api_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/airtable_api_key.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/algolia_api_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/algolia_api_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/algolia_api_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/algolia_api_key.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/alibaba.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/alibaba.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/alibaba.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/alibaba.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/asana.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/asana.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/asana.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/asana.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/atlassian_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/atlassian_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/atlassian_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/atlassian_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/authress_access_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/authress_access_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/authress_access_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/authress_access_key.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/beamer_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/beamer_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/beamer_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/beamer_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/bitbucket.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/bitbucket.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/bitbucket.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/bitbucket.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/bittrex.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/bittrex.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/bittrex.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/bittrex.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/clojars_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/clojars_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/clojars_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/clojars_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/codecov_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/codecov_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/codecov_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/codecov_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/coinbase_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/coinbase_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/coinbase_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/coinbase_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/confluent.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/confluent.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/confluent.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/confluent.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/contentful_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/contentful_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/contentful_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/contentful_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/databricks_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/databricks_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/databricks_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/databricks_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/datadog_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/datadog_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/datadog_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/datadog_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/defined_networking_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/defined_networking_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/defined_networking_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/defined_networking_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/digitalocean.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/digitalocean.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/digitalocean.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/digitalocean.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/discord.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/discord.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/discord.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/discord.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/doppler_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/doppler_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/doppler_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/doppler_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/droneci_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/droneci_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/droneci_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/droneci_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/dropbox.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/dropbox.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/dropbox.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/dropbox.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/duffel_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/duffel_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/duffel_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/duffel_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/dynatrace_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/dynatrace_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/dynatrace_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/dynatrace_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/easypost.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/easypost.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/easypost.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/easypost.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/etsy_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/etsy_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/etsy_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/etsy_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/facebook_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/facebook_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/facebook_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/facebook_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/fastly_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/fastly_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/fastly_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/fastly_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/finicity.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/finicity.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/finicity.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/finicity.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/finnhub_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/finnhub_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/finnhub_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/finnhub_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/flickr_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/flickr_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/flickr_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/flickr_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/flutterwave.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/flutterwave.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/flutterwave.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/flutterwave.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/frameio_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/frameio_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/frameio_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/frameio_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/freshbooks_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/freshbooks_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/freshbooks_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/freshbooks_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/gcp_api_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/gcp_api_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/gcp_api_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/gcp_api_key.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/github_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/github_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/github_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/github_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/gitlab.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/gitlab.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/gitlab.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/gitlab.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/gitter_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/gitter_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/gitter_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/gitter_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/gocardless_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/gocardless_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/gocardless_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/gocardless_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/grafana.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/grafana.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/grafana.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/grafana.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/hashicorp_tf_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/hashicorp_tf_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/hashicorp_tf_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/hashicorp_tf_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/heroku_api_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/heroku_api_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/heroku_api_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/heroku_api_key.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/hubspot_api_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/hubspot_api_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/hubspot_api_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/hubspot_api_key.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/huggingface.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/huggingface.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/huggingface.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/huggingface.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/intercom_api_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/intercom_api_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/intercom_api_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/intercom_api_key.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/jfrog.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/jfrog.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/jfrog.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/jfrog.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/jwt.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/jwt.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/jwt.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/jwt.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/kraken_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/kraken_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/kraken_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/kraken_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/kucoin.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/kucoin.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/kucoin.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/kucoin.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/launchdarkly_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/launchdarkly_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/launchdarkly_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/launchdarkly_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/linear.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/linear.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/linear.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/linear.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/linkedin.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/linkedin.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/linkedin.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/linkedin.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/lob.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/lob.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/lob.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/lob.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/mailgun.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/mailgun.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/mailgun.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/mailgun.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/mapbox_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/mapbox_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/mapbox_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/mapbox_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/mattermost_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/mattermost_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/mattermost_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/mattermost_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/messagebird.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/messagebird.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/messagebird.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/messagebird.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/microsoft_teams_webhook.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/microsoft_teams_webhook.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/microsoft_teams_webhook.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/microsoft_teams_webhook.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/netlify_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/netlify_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/netlify_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/netlify_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/new_relic.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/new_relic.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/new_relic.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/new_relic.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/nytimes_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/nytimes_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/nytimes_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/nytimes_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/okta_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/okta_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/okta_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/okta_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/openai_api_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/openai_api_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/openai_api_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/openai_api_key.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/planetscale.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/planetscale.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/planetscale.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/planetscale.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/postman_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/postman_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/postman_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/postman_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/prefect_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/prefect_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/prefect_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/prefect_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/pulumi_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/pulumi_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/pulumi_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/pulumi_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/pypi_upload_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/pypi_upload_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/pypi_upload_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/pypi_upload_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/rapidapi_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/rapidapi_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/rapidapi_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/rapidapi_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/readme_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/readme_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/readme_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/readme_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/rubygems_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/rubygems_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/rubygems_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/rubygems_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/scalingo_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/scalingo_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/scalingo_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/scalingo_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/sendbird.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sendbird.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/sendbird.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sendbird.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/sendgrid_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sendgrid_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/sendgrid_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sendgrid_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/sendinblue_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sendinblue_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/sendinblue_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sendinblue_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/sentry_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sentry_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/sentry_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sentry_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/shippo_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/shippo_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/shippo_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/shippo_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/shopify.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/shopify.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/shopify.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/shopify.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/slack.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/slack.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/slack.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/slack.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/snyk_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/snyk_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/snyk_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/snyk_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/squarespace_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/squarespace_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/squarespace_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/squarespace_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/sumologic.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sumologic.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/sumologic.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/sumologic.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/telegram_bot_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/telegram_bot_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/telegram_bot_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/telegram_bot_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/travisci_access_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/travisci_access_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/travisci_access_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/travisci_access_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/twitch_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/twitch_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/twitch_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/twitch_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/twitter.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/twitter.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/twitter.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/twitter.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/typeform_api_token.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/typeform_api_token.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/typeform_api_token.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/typeform_api_token.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/vault.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/vault.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/vault.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/vault.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/yandex.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/yandex.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/yandex.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/yandex.py diff --git a/enterprise/enterprise_hooks/secrets_plugins/zendesk_secret_key.py b/enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/zendesk_secret_key.py similarity index 100% rename from enterprise/enterprise_hooks/secrets_plugins/zendesk_secret_key.py rename to enterprise/litellm_enterprise/enterprise_callbacks/secrets_plugins/zendesk_secret_key.py diff --git a/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/base_email.py b/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/base_email.py new file mode 100644 index 00000000000..9ea22074c01 --- /dev/null +++ b/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/base_email.py @@ -0,0 +1,217 @@ +""" +Base class for sending emails to user after creating keys or invite links + +""" + +import json +import os +from typing import List, Optional + +from litellm._logging import verbose_proxy_logger +from litellm.integrations.custom_logger import CustomLogger +from litellm.integrations.email_templates.email_footer import EMAIL_FOOTER +from litellm.integrations.email_templates.key_created_email import ( + KEY_CREATED_EMAIL_TEMPLATE, +) +from litellm.integrations.email_templates.user_invitation_email import ( + USER_INVITATION_EMAIL_TEMPLATE, +) +from litellm.proxy._types import WebhookEvent +from litellm.types.enterprise.enterprise_callbacks.send_emails import ( + EmailEvent, + EmailParams, + SendKeyCreatedEmailEvent, +) +from litellm.types.integrations.slack_alerting import LITELLM_LOGO_URL + + +class BaseEmailLogger(CustomLogger): + DEFAULT_LITELLM_EMAIL = "notifications@alerts.litellm.ai" + DEFAULT_SUPPORT_EMAIL = "support@berri.ai" + + async def send_user_invitation_email(self, event: WebhookEvent): + """ + Send email to user after inviting them to the team + """ + email_params = await self._get_email_params( + email_event=EmailEvent.new_user_invitation, + user_id=event.user_id, + user_email=getattr(event, "user_email", None), + ) + # Implement invitation email logic using email_params + + verbose_proxy_logger.debug( + f"send_user_invitation_email_event: {json.dumps(event, indent=4, default=str)}" + ) + + email_html_content = USER_INVITATION_EMAIL_TEMPLATE.format( + email_logo_url=email_params.logo_url, + recipient_email=email_params.recipient_email, + base_url=email_params.base_url, + email_support_contact=email_params.support_contact, + email_footer=EMAIL_FOOTER, + ) + + await self.send_email( + from_email=self.DEFAULT_LITELLM_EMAIL, + to_email=[email_params.recipient_email], + subject=f"LiteLLM: {event.event_message}", + html_body=email_html_content, + ) + + pass + + async def send_key_created_email( + self, send_key_created_email_event: SendKeyCreatedEmailEvent + ): + """ + Send email to user after creating key for the user + """ + + email_params = await self._get_email_params( + user_id=send_key_created_email_event.user_id, + user_email=send_key_created_email_event.user_email, + email_event=EmailEvent.virtual_key_created, + ) + + verbose_proxy_logger.debug( + f"send_key_created_email_event: {json.dumps(send_key_created_email_event, indent=4, default=str)}" + ) + + email_html_content = KEY_CREATED_EMAIL_TEMPLATE.format( + email_logo_url=email_params.logo_url, + recipient_email=email_params.recipient_email, + key_budget=self._format_key_budget(send_key_created_email_event.max_budget), + key_token=send_key_created_email_event.virtual_key, + base_url=email_params.base_url, + email_support_contact=email_params.support_contact, + email_footer=EMAIL_FOOTER, + ) + + await self.send_email( + from_email=self.DEFAULT_LITELLM_EMAIL, + to_email=[email_params.recipient_email], + subject=f"LiteLLM: {send_key_created_email_event.event_message}", + html_body=email_html_content, + ) + pass + + async def _get_email_params( + self, + email_event: EmailEvent, + user_id: Optional[str] = None, + user_email: Optional[str] = None, + ) -> EmailParams: + """ + Get common email parameters used across different email sending methods + + Returns: + EmailParams object containing logo_url, support_contact, base_url, and recipient_email + """ + logo_url = os.getenv("EMAIL_LOGO_URL", None) or LITELLM_LOGO_URL + support_contact = os.getenv("EMAIL_SUPPORT_CONTACT", self.DEFAULT_SUPPORT_EMAIL) + base_url = os.getenv("PROXY_BASE_URL", "http://0.0.0.0:4000") + + recipient_email: Optional[ + str + ] = user_email or await self._lookup_user_email_from_db(user_id=user_id) + if recipient_email is None: + raise ValueError( + f"User email not found for user_id: {user_id}. User email is required to send email." + ) + + # if user invited event then send invitation link + if email_event == EmailEvent.new_user_invitation: + base_url = await self._get_invitation_link( + user_id=user_id, base_url=base_url + ) + + return EmailParams( + logo_url=logo_url, + support_contact=support_contact, + base_url=base_url, + recipient_email=recipient_email, + ) + + def _format_key_budget(self, max_budget: Optional[float]) -> str: + """ + Format the key budget to be displayed in the email + """ + if max_budget is None: + return "No budget" + return f"${max_budget}" + + async def _lookup_user_email_from_db(self, user_id: Optional[str]) -> Optional[str]: + """ + Lookup user email from user_id + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + verbose_proxy_logger.debug( + f"Prisma client not found. Unable to lookup user email for user_id: {user_id}" + ) + return None + + user_row = await prisma_client.db.litellm_usertable.find_unique( + where={"user_id": user_id} + ) + + if user_row is not None: + return user_row.user_email + return None + + async def _get_invitation_link(self, user_id: Optional[str], base_url: str) -> str: + """ + Get invitation link for the user + """ + import asyncio + + from litellm.proxy.proxy_server import prisma_client + + ################################################################################ + ########## Sleep for 10 seconds to wait for the invitation link to be created ### + ################################################################################ + # The UI, calls /invitation/new to generate the invitation link + # We wait 10 seconds to ensure the link is created + ################################################################################ + await asyncio.sleep(10) + + if prisma_client is None: + verbose_proxy_logger.debug( + f"Prisma client not found. Unable to lookup user email for user_id: {user_id}" + ) + return base_url + + if user_id is None: + return base_url + + # get the latest invitation link for the user + invitation_rows = await prisma_client.db.litellm_invitationlink.find_many( + where={"user_id": user_id}, + orderBy={"created_at": "desc"}, + ) + if len(invitation_rows) > 0: + invitation_row = invitation_rows[0] + return self._construct_invitation_link( + invitation_id=invitation_row.id, base_url=base_url + ) + + return base_url + + def _construct_invitation_link(self, invitation_id: str, base_url: str) -> str: + """ + Construct invitation link for the user + + # http://localhost:4000/ui?invitation_id=7a096b3a-37c6-440f-9dd1-ba22e8043f6b + """ + return f"{base_url}/ui?invitation_id={invitation_id}" + + async def send_email( + self, + from_email: str, + to_email: List[str], + subject: str, + html_body: str, + ): + pass diff --git a/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/endpoints.py b/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/endpoints.py new file mode 100644 index 00000000000..cc6f0be80f9 --- /dev/null +++ b/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/endpoints.py @@ -0,0 +1,202 @@ +""" +Endpoints for managing email alerts on litellm +""" + +import json +from typing import Dict + +from fastapi import APIRouter, Depends, HTTPException + +from litellm._logging import verbose_proxy_logger +from litellm.proxy._types import UserAPIKeyAuth +from litellm.proxy.auth.user_api_key_auth import user_api_key_auth +from litellm.types.enterprise.enterprise_callbacks.send_emails import ( + DefaultEmailSettings, + EmailEvent, + EmailEventSettings, + EmailEventSettingsResponse, + EmailEventSettingsUpdateRequest, +) + +router = APIRouter() + + +async def _get_email_settings(prisma_client) -> Dict[str, bool]: + """Helper function to get email settings from general_settings in db""" + try: + # Get general settings from db + general_settings_entry = await prisma_client.db.litellm_config.find_unique( + where={"param_name": "general_settings"} + ) + + # Initialize with default email settings + settings_dict = DefaultEmailSettings.get_defaults() + + if ( + general_settings_entry is not None + and general_settings_entry.param_value is not None + ): + # Get general settings value + if isinstance(general_settings_entry.param_value, str): + general_settings = json.loads(general_settings_entry.param_value) + else: + general_settings = general_settings_entry.param_value + + # Extract email_settings from general settings if it exists + if general_settings and "email_settings" in general_settings: + email_settings = general_settings["email_settings"] + # Update settings_dict with values from general_settings + for event_name, enabled in email_settings.items(): + settings_dict[event_name] = enabled + + return settings_dict + except Exception as e: + verbose_proxy_logger.error( + f"Error getting email settings from general_settings: {str(e)}" + ) + # Return default settings in case of error + return DefaultEmailSettings.get_defaults() + + +async def _save_email_settings(prisma_client, settings: Dict[str, bool]): + """Helper function to save email settings to general_settings in db""" + try: + verbose_proxy_logger.debug( + f"Saving email settings to general_settings: {settings}" + ) + + # Get current general settings + general_settings_entry = await prisma_client.db.litellm_config.find_unique( + where={"param_name": "general_settings"} + ) + + # Initialize general settings dict + if ( + general_settings_entry is not None + and general_settings_entry.param_value is not None + ): + if isinstance(general_settings_entry.param_value, str): + general_settings = json.loads(general_settings_entry.param_value) + else: + general_settings = dict(general_settings_entry.param_value) + else: + general_settings = {} + + # Update email_settings in general_settings + general_settings["email_settings"] = settings + + # Convert to JSON for storage + json_settings = json.dumps(general_settings, default=str) + + # Save updated general settings + await prisma_client.db.litellm_config.upsert( + where={"param_name": "general_settings"}, + data={ + "create": { + "param_name": "general_settings", + "param_value": json_settings, + }, + "update": {"param_value": json_settings}, + }, + ) + except Exception as e: + raise HTTPException( + status_code=500, + detail=f"Error saving email settings to general_settings: {str(e)}", + ) + + +@router.get( + "/email/event_settings", + response_model=EmailEventSettingsResponse, + tags=["email management"], + dependencies=[Depends(user_api_key_auth)], +) +async def get_email_event_settings( + user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), +): + """ + Get all email event settings + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + raise HTTPException(status_code=500, detail="Database not connected") + + try: + # Get existing settings + settings_dict = await _get_email_settings(prisma_client) + + # Create a response with all events (enabled or disabled) + response_settings = [] + for event in EmailEvent: + enabled = settings_dict.get(event.value, False) + response_settings.append(EmailEventSettings(event=event, enabled=enabled)) + + return EmailEventSettingsResponse(settings=response_settings) + except Exception as e: + verbose_proxy_logger.exception(f"Error getting email settings: {str(e)}") + raise HTTPException(status_code=500, detail=str(e)) + + +@router.patch( + "/email/event_settings", + tags=["email management"], + dependencies=[Depends(user_api_key_auth)], +) +async def update_event_settings( + request: EmailEventSettingsUpdateRequest, + user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), +): + """ + Update the settings for email events + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + raise HTTPException(status_code=500, detail="Database not connected") + + try: + # Get existing settings + settings_dict = await _get_email_settings(prisma_client) + + # Update with new settings + for setting in request.settings: + settings_dict[setting.event.value] = setting.enabled + + # Save updated settings + await _save_email_settings(prisma_client, settings_dict) + + return {"message": "Email event settings updated successfully"} + except Exception as e: + verbose_proxy_logger.exception(f"Error updating email settings: {str(e)}") + raise HTTPException(status_code=500, detail=str(e)) + + +@router.post( + "/email/event_settings/reset", + tags=["email management"], + dependencies=[Depends(user_api_key_auth)], +) +async def reset_event_settings( + user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), +): + """ + Reset all email event settings to default (new user invitations on, virtual key creation off) + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + raise HTTPException(status_code=500, detail="Database not connected") + + try: + # Reset to default settings using the Pydantic model + default_settings = DefaultEmailSettings.get_defaults() + + # Save default settings + await _save_email_settings(prisma_client, default_settings) + + return {"message": "Email event settings reset to defaults"} + except Exception as e: + verbose_proxy_logger.exception(f"Error resetting email settings: {str(e)}") + raise HTTPException(status_code=500, detail=str(e)) diff --git a/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/resend_email.py b/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/resend_email.py new file mode 100644 index 00000000000..8119e4a7ef5 --- /dev/null +++ b/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/resend_email.py @@ -0,0 +1,51 @@ +""" +This is the litellm x resend email integration + +https://resend.com/docs/api-reference/emails/send-email +""" + +import os +from typing import List + +from litellm._logging import verbose_logger +from litellm.llms.custom_httpx.http_handler import ( + get_async_httpx_client, + httpxSpecialProvider, +) + +from .base_email import BaseEmailLogger + +RESEND_API_ENDPOINT = "https://api.resend.com/emails" + + +class ResendEmailLogger(BaseEmailLogger): + def __init__(self): + self.async_httpx_client = get_async_httpx_client( + llm_provider=httpxSpecialProvider.LoggingCallback + ) + self.resend_api_key = os.getenv("RESEND_API_KEY") + + async def send_email( + self, + from_email: str, + to_email: List[str], + subject: str, + html_body: str, + ): + verbose_logger.debug( + f"Sending email from {from_email} to {to_email} with subject {subject}" + ) + response = await self.async_httpx_client.post( + url=RESEND_API_ENDPOINT, + json={ + "from": from_email, + "to": to_email, + "subject": subject, + "html": html_body, + }, + headers={"Authorization": f"Bearer {self.resend_api_key}"}, + ) + verbose_logger.debug( + f"Email sent with status code {response.status_code}. Got response: {response.json()}" + ) + return diff --git a/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/smtp_email.py b/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/smtp_email.py new file mode 100644 index 00000000000..4ede8ee59fe --- /dev/null +++ b/enterprise/litellm_enterprise/enterprise_callbacks/send_emails/smtp_email.py @@ -0,0 +1,47 @@ +""" +This is the litellm SMTP email integration +""" +import asyncio +from typing import List + +from litellm._logging import verbose_logger + +from .base_email import BaseEmailLogger + + +class SMTPEmailLogger(BaseEmailLogger): + """ + This is the litellm SMTP email integration + + Required SMTP environment variables: + - SMTP_HOST + - SMTP_PORT + - SMTP_USERNAME + - SMTP_PASSWORD + - SMTP_SENDER_EMAIL + """ + + def __init__(self): + verbose_logger.debug("SMTP Email Logger initialized....") + + async def send_email( + self, + from_email: str, + to_email: List[str], + subject: str, + html_body: str, + ): + from litellm.proxy.utils import send_email as send_smtp_email + + verbose_logger.debug( + f"Sending email from {from_email} to {to_email} with subject {subject}" + ) + for receiver_email in to_email: + asyncio.create_task( + send_smtp_email( + receiver_email=receiver_email, + subject=subject, + html=html_body, + ) + ) + return diff --git a/enterprise/litellm_enterprise/litellm_core_utils/litellm_logging.py b/enterprise/litellm_enterprise/litellm_core_utils/litellm_logging.py new file mode 100644 index 00000000000..44ba0063ffe --- /dev/null +++ b/enterprise/litellm_enterprise/litellm_core_utils/litellm_logging.py @@ -0,0 +1,28 @@ +""" +Enterprise specific logging utils +""" +from litellm.litellm_core_utils.litellm_logging import StandardLoggingMetadata + + +class StandardLoggingPayloadSetup: + @staticmethod + def apply_enterprise_specific_metadata( + standard_logging_metadata: StandardLoggingMetadata, + proxy_server_request: dict, + ) -> StandardLoggingMetadata: + """ + Adds enterprise-only metadata to the standard logging metadata. + """ + + _request_headers = proxy_server_request.get("headers", {}) + + if _request_headers: + custom_headers = { + k: v + for k, v in _request_headers.items() + if k.startswith("x-") and v is not None and isinstance(v, str) + } + + standard_logging_metadata["requester_custom_headers"] = custom_headers + + return standard_logging_metadata diff --git a/enterprise/proxy/enterprise_routes.py b/enterprise/litellm_enterprise/proxy/enterprise_routes.py similarity index 71% rename from enterprise/proxy/enterprise_routes.py rename to enterprise/litellm_enterprise/proxy/enterprise_routes.py index 26183874c6d..2420ad2e055 100644 --- a/enterprise/proxy/enterprise_routes.py +++ b/enterprise/litellm_enterprise/proxy/enterprise_routes.py @@ -1,10 +1,17 @@ from fastapi import APIRouter from fastapi.responses import Response +from litellm_enterprise.enterprise_callbacks.send_emails.endpoints import ( + router as email_events_router, +) + +from .guardrails.endpoints import router as guardrails_router from .utils import _should_block_robots from .vector_stores.endpoints import router as vector_stores_router router = APIRouter() router.include_router(vector_stores_router) +router.include_router(guardrails_router) +router.include_router(email_events_router) @router.get("/robots.txt") diff --git a/enterprise/litellm_enterprise/proxy/guardrails/endpoints.py b/enterprise/litellm_enterprise/proxy/guardrails/endpoints.py new file mode 100644 index 00000000000..cdf86dcea67 --- /dev/null +++ b/enterprise/litellm_enterprise/proxy/guardrails/endpoints.py @@ -0,0 +1,41 @@ +""" +Enterprise Guardrail Routes on LiteLLM Proxy + +To see all free guardrails see litellm/proxy/guardrails/* + + +Exposed Routes: +- /mask_pii +""" +from typing import Optional + +from fastapi import APIRouter, Depends + +from litellm.integrations.custom_guardrail import CustomGuardrail +from litellm.proxy._types import UserAPIKeyAuth +from litellm.proxy.auth.user_api_key_auth import user_api_key_auth +from litellm.proxy.guardrails.guardrail_endpoints import GUARDRAIL_REGISTRY +from litellm.types.guardrails import ApplyGuardrailRequest, ApplyGuardrailResponse + +router = APIRouter(tags=["guardrails"], prefix="/guardrails") + + +@router.post("/apply_guardrail", response_model=ApplyGuardrailResponse) +async def apply_guardrail( + request: ApplyGuardrailRequest, + user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), +): + """ + Mask PII from a given text, requires a guardrail to be added to litellm. + """ + active_guardrail: Optional[ + CustomGuardrail + ] = GUARDRAIL_REGISTRY.get_initialized_guardrail_callback( + guardrail_name=request.guardrail_name + ) + if active_guardrail is None: + raise Exception(f"Guardrail {request.guardrail_name} not found") + + return await active_guardrail.apply_guardrail( + text=request.text, language=request.language, entities=request.entities + ) diff --git a/enterprise/proxy/readme.md b/enterprise/litellm_enterprise/proxy/readme.md similarity index 100% rename from enterprise/proxy/readme.md rename to enterprise/litellm_enterprise/proxy/readme.md diff --git a/enterprise/proxy/utils.py b/enterprise/litellm_enterprise/proxy/utils.py similarity index 65% rename from enterprise/proxy/utils.py rename to enterprise/litellm_enterprise/proxy/utils.py index c3396964ee6..227ea0a9ff0 100644 --- a/enterprise/proxy/utils.py +++ b/enterprise/litellm_enterprise/proxy/utils.py @@ -1,4 +1,5 @@ -from typing import Union, Optional +from typing import Optional, Union + from litellm.secret_managers.main import str_to_bool @@ -6,14 +7,19 @@ def _should_block_robots(): """ Returns True if the robots.txt file should block web crawlers - Controlled by - + Controlled by + ```yaml general_settings: block_robots: true ``` """ - from litellm.proxy.proxy_server import general_settings, premium_user, CommonProxyErrors + from litellm.proxy.proxy_server import ( + CommonProxyErrors, + general_settings, + premium_user, + ) + _block_robots: Union[bool, str] = general_settings.get("block_robots", False) block_robots: Optional[bool] = None if isinstance(_block_robots, bool): @@ -22,6 +28,8 @@ def _should_block_robots(): block_robots = str_to_bool(_block_robots) if block_robots is True: if premium_user is not True: - raise ValueError(f"Blocking web crawlers is an enterprise feature. {CommonProxyErrors.not_premium_user.value}") + raise ValueError( + f"Blocking web crawlers is an enterprise feature. {CommonProxyErrors.not_premium_user.value}" + ) return True return False diff --git a/enterprise/proxy/vector_stores/endpoints.py b/enterprise/litellm_enterprise/proxy/vector_stores/endpoints.py similarity index 97% rename from enterprise/proxy/vector_stores/endpoints.py rename to enterprise/litellm_enterprise/proxy/vector_stores/endpoints.py index c4198f6a551..77286a648f1 100644 --- a/enterprise/proxy/vector_stores/endpoints.py +++ b/enterprise/litellm_enterprise/proxy/vector_stores/endpoints.py @@ -70,12 +70,16 @@ async def new_vector_store( vector_store.get("vector_store_metadata") ) - new_vector_store = ( + _new_vector_store = ( await prisma_client.db.litellm_managedvectorstorestable.create( data=vector_store ) ) + new_vector_store: LiteLLM_ManagedVectorStore = LiteLLM_ManagedVectorStore( + **_new_vector_store.model_dump() + ) + # Add vector store to registry if litellm.vector_store_registry is not None: litellm.vector_store_registry.add_vector_store_to_registry( diff --git a/enterprise/poetry.lock b/enterprise/poetry.lock new file mode 100644 index 00000000000..bb436a168cd --- /dev/null +++ b/enterprise/poetry.lock @@ -0,0 +1,7 @@ +# This file is automatically @generated by Poetry 2.1.2 and should not be changed by hand. +package = [] + +[metadata] +lock-version = "2.1" +python-versions = ">=3.8.1,<4.0, !=3.9.7" +content-hash = "2cf39473e67ff0615f0a61c9d2ac9f02b38cc08cbb1bdb893d89bee002646623" diff --git a/enterprise/pyproject.toml b/enterprise/pyproject.toml new file mode 100644 index 00000000000..e8b5f1dfcee --- /dev/null +++ b/enterprise/pyproject.toml @@ -0,0 +1,30 @@ +[tool.poetry] +name = "litellm-enterprise" +version = "0.1.5" +description = "Package for LiteLLM Enterprise features" +authors = ["BerriAI"] +readme = "README.md" + + +[tool.poetry.urls] +homepage = "https://litellm.ai" +Homepage = "https://litellm.ai" +repository = "https://github.com/BerriAI/litellm" +Repository = "https://github.com/BerriAI/litellm" +documentation = "https://docs.litellm.ai" +Documentation = "https://docs.litellm.ai" + +[tool.poetry.dependencies] +python = ">=3.8.1,<4.0, !=3.9.7" + +[build-system] +requires = ["poetry-core"] +build-backend = "poetry.core.masonry.api" + +[tool.commitizen] +version = "0.1.5" +version_files = [ + "pyproject.toml:version", + "../requirements.txt:litellm-enterprise==", + "../pyproject.toml:litellm-enterprise = {version = \"" +] \ No newline at end of file diff --git a/litellm-proxy-extras/README.md b/litellm-proxy-extras/README.md index 29453f65ba9..d6d00a62d42 100644 --- a/litellm-proxy-extras/README.md +++ b/litellm-proxy-extras/README.md @@ -10,7 +10,7 @@ pip install litellm-proxy-extras OR ```bash -pip install litellm[proxy] # installs litellm-proxy-extras and other proxy dependencies. +pip install litellm[proxy] # installs litellm-proxy-extras and other proxy dependencies ``` To use the migrations, run: diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.17-py3-none-any.whl b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.17-py3-none-any.whl new file mode 100644 index 00000000000..5e64ad7733a Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.17-py3-none-any.whl differ diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.17.tar.gz b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.17.tar.gz new file mode 100644 index 00000000000..49183ba24f2 Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.17.tar.gz differ diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.18-py3-none-any.whl b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.18-py3-none-any.whl new file mode 100644 index 00000000000..42621943a0e Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.18-py3-none-any.whl differ diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.18.tar.gz b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.18.tar.gz new file mode 100644 index 00000000000..9b83e532a84 Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.18.tar.gz differ diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.19-py3-none-any.whl b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.19-py3-none-any.whl new file mode 100644 index 00000000000..1506322a02d Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.19-py3-none-any.whl differ diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.19.tar.gz b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.19.tar.gz new file mode 100644 index 00000000000..3deaee390f8 Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.19.tar.gz differ diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.20-py3-none-any.whl b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.20-py3-none-any.whl new file mode 100644 index 00000000000..60d9d8130aa Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.20-py3-none-any.whl differ diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.20.tar.gz b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.20.tar.gz new file mode 100644 index 00000000000..1f01d5067e9 Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.20.tar.gz differ diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.21-py3-none-any.whl b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.21-py3-none-any.whl new file mode 100644 index 00000000000..8602cd14ed6 Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.21-py3-none-any.whl differ diff --git a/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.21.tar.gz b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.21.tar.gz new file mode 100644 index 00000000000..2074d2256fa Binary files /dev/null and b/litellm-proxy-extras/dist/litellm_proxy_extras-0.1.21.tar.gz differ diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20250507161526_add_mcp_table_to_db/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250507161526_add_mcp_table_to_db/migration.sql new file mode 100644 index 00000000000..6b8adc6e7e8 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250507161526_add_mcp_table_to_db/migration.sql @@ -0,0 +1,17 @@ +-- CreateTable +CREATE TABLE "LiteLLM_MCPServerTable" ( + "server_id" TEXT NOT NULL, + "alias" TEXT, + "description" TEXT, + "url" TEXT NOT NULL, + "transport" TEXT NOT NULL DEFAULT 'sse', + "spec_version" TEXT NOT NULL DEFAULT '2025-03-26', + "auth_type" TEXT, + "created_at" TIMESTAMP(3) DEFAULT CURRENT_TIMESTAMP, + "created_by" TEXT, + "updated_at" TIMESTAMP(3) DEFAULT CURRENT_TIMESTAMP, + "updated_by" TEXT, + + CONSTRAINT "LiteLLM_MCPServerTable_pkey" PRIMARY KEY ("server_id") +); + diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20250507184818_add_mcp_key_team_permission_mgmt/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250507184818_add_mcp_key_team_permission_mgmt/migration.sql new file mode 100644 index 00000000000..dcfce07a487 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250507184818_add_mcp_key_team_permission_mgmt/migration.sql @@ -0,0 +1,32 @@ +-- AlterTable +ALTER TABLE "LiteLLM_OrganizationTable" ADD COLUMN "object_permission_id" TEXT; + +-- AlterTable +ALTER TABLE "LiteLLM_TeamTable" ADD COLUMN "object_permission_id" TEXT; + +-- AlterTable +ALTER TABLE "LiteLLM_UserTable" ADD COLUMN "object_permission_id" TEXT; + +-- AlterTable +ALTER TABLE "LiteLLM_VerificationToken" ADD COLUMN "object_permission_id" TEXT; + +-- CreateTable +CREATE TABLE "LiteLLM_ObjectPermissionTable" ( + "object_permission_id" TEXT NOT NULL, + "mcp_servers" TEXT[] DEFAULT ARRAY[]::TEXT[], + + CONSTRAINT "LiteLLM_ObjectPermissionTable_pkey" PRIMARY KEY ("object_permission_id") +); + +-- AddForeignKey +ALTER TABLE "LiteLLM_OrganizationTable" ADD CONSTRAINT "LiteLLM_OrganizationTable_object_permission_id_fkey" FOREIGN KEY ("object_permission_id") REFERENCES "LiteLLM_ObjectPermissionTable"("object_permission_id") ON DELETE SET NULL ON UPDATE CASCADE; + +-- AddForeignKey +ALTER TABLE "LiteLLM_TeamTable" ADD CONSTRAINT "LiteLLM_TeamTable_object_permission_id_fkey" FOREIGN KEY ("object_permission_id") REFERENCES "LiteLLM_ObjectPermissionTable"("object_permission_id") ON DELETE SET NULL ON UPDATE CASCADE; + +-- AddForeignKey +ALTER TABLE "LiteLLM_UserTable" ADD CONSTRAINT "LiteLLM_UserTable_object_permission_id_fkey" FOREIGN KEY ("object_permission_id") REFERENCES "LiteLLM_ObjectPermissionTable"("object_permission_id") ON DELETE SET NULL ON UPDATE CASCADE; + +-- AddForeignKey +ALTER TABLE "LiteLLM_VerificationToken" ADD CONSTRAINT "LiteLLM_VerificationToken_object_permission_id_fkey" FOREIGN KEY ("object_permission_id") REFERENCES "LiteLLM_ObjectPermissionTable"("object_permission_id") ON DELETE SET NULL ON UPDATE CASCADE; + diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20250508072103_add_status_to_spendlogs/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250508072103_add_status_to_spendlogs/migration.sql new file mode 100644 index 00000000000..8f6c68aa67e --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250508072103_add_status_to_spendlogs/migration.sql @@ -0,0 +1,3 @@ +-- AlterTable +ALTER TABLE "LiteLLM_SpendLogs" ADD COLUMN "status" TEXT; + diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20250509141545_use_big_int_for_daily_spend_tables/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250509141545_use_big_int_for_daily_spend_tables/migration.sql new file mode 100644 index 00000000000..582b7947a37 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250509141545_use_big_int_for_daily_spend_tables/migration.sql @@ -0,0 +1,27 @@ +-- AlterTable +ALTER TABLE "LiteLLM_DailyTagSpend" ALTER COLUMN "prompt_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "completion_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "cache_read_input_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "cache_creation_input_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "api_requests" SET DATA TYPE BIGINT, +ALTER COLUMN "successful_requests" SET DATA TYPE BIGINT, +ALTER COLUMN "failed_requests" SET DATA TYPE BIGINT; + +-- AlterTable +ALTER TABLE "LiteLLM_DailyTeamSpend" ALTER COLUMN "prompt_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "completion_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "api_requests" SET DATA TYPE BIGINT, +ALTER COLUMN "successful_requests" SET DATA TYPE BIGINT, +ALTER COLUMN "failed_requests" SET DATA TYPE BIGINT, +ALTER COLUMN "cache_creation_input_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "cache_read_input_tokens" SET DATA TYPE BIGINT; + +-- AlterTable +ALTER TABLE "LiteLLM_DailyUserSpend" ALTER COLUMN "prompt_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "completion_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "api_requests" SET DATA TYPE BIGINT, +ALTER COLUMN "failed_requests" SET DATA TYPE BIGINT, +ALTER COLUMN "successful_requests" SET DATA TYPE BIGINT, +ALTER COLUMN "cache_creation_input_tokens" SET DATA TYPE BIGINT, +ALTER COLUMN "cache_read_input_tokens" SET DATA TYPE BIGINT; + diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20250510142544_add_session_id_index_spend_logs/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250510142544_add_session_id_index_spend_logs/migration.sql new file mode 100644 index 00000000000..eda055d6e56 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250510142544_add_session_id_index_spend_logs/migration.sql @@ -0,0 +1,3 @@ +-- CreateIndex +CREATE INDEX "LiteLLM_SpendLogs_session_id_idx" ON "LiteLLM_SpendLogs"("session_id"); + diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20250514142245_add_guardrails_table/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250514142245_add_guardrails_table/migration.sql new file mode 100644 index 00000000000..fa99e3be637 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20250514142245_add_guardrails_table/migration.sql @@ -0,0 +1,15 @@ +-- CreateTable +CREATE TABLE "LiteLLM_GuardrailsTable" ( + "guardrail_id" TEXT NOT NULL, + "guardrail_name" TEXT NOT NULL, + "litellm_params" JSONB NOT NULL, + "guardrail_info" JSONB, + "created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP, + "updated_at" TIMESTAMP(3) NOT NULL, + + CONSTRAINT "LiteLLM_GuardrailsTable_pkey" PRIMARY KEY ("guardrail_id") +); + +-- CreateIndex +CREATE UNIQUE INDEX "LiteLLM_GuardrailsTable_guardrail_name_key" ON "LiteLLM_GuardrailsTable"("guardrail_name"); + diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index 4c9856909ce..1d6f3b52118 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -61,6 +61,7 @@ model LiteLLM_OrganizationTable { models String[] spend Float @default(0.0) model_spend Json @default("{}") + object_permission_id String? created_at DateTime @default(now()) @map("created_at") created_by String updated_at DateTime @default(now()) @updatedAt @map("updated_at") @@ -70,6 +71,7 @@ model LiteLLM_OrganizationTable { users LiteLLM_UserTable[] keys LiteLLM_VerificationToken[] members LiteLLM_OrganizationMembership[] @relation("OrganizationToMembership") + object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id]) } // Model info for teams, just has model aliases for now. @@ -89,6 +91,7 @@ model LiteLLM_TeamTable { team_id String @id @default(uuid()) team_alias String? organization_id String? + object_permission_id String? admins String[] members String[] members_with_roles Json @default("{}") @@ -110,6 +113,7 @@ model LiteLLM_TeamTable { model_id Int? @unique // id for LiteLLM_ModelTable -> stores team-level model aliases litellm_organization_table LiteLLM_OrganizationTable? @relation(fields: [organization_id], references: [organization_id]) litellm_model_table LiteLLM_ModelTable? @relation(fields: [model_id], references: [id]) + object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id]) } // Track spend, rate limit, budget Users @@ -119,6 +123,7 @@ model LiteLLM_UserTable { team_id String? sso_user_id String? @unique organization_id String? + object_permission_id String? password String? teams String[] @default([]) user_role String? @@ -144,6 +149,32 @@ model LiteLLM_UserTable { invitations_created LiteLLM_InvitationLink[] @relation("CreatedBy") invitations_updated LiteLLM_InvitationLink[] @relation("UpdatedBy") invitations_user LiteLLM_InvitationLink[] @relation("UserId") + object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id]) +} + +model LiteLLM_ObjectPermissionTable { + object_permission_id String @id @default(uuid()) + mcp_servers String[] @default([]) + + teams LiteLLM_TeamTable[] + verification_tokens LiteLLM_VerificationToken[] + organizations LiteLLM_OrganizationTable[] + users LiteLLM_UserTable[] +} + +// Holds the MCP server configuration +model LiteLLM_MCPServerTable { + server_id String @id @default(uuid()) + alias String? + description String? + url String + transport String @default("sse") + spec_version String @default("2025-03-26") + auth_type String? + created_at DateTime? @default(now()) @map("created_at") + created_by String? + updated_at DateTime? @default(now()) @updatedAt @map("updated_at") + updated_by String? } // Generate Tokens for Proxy @@ -174,12 +205,14 @@ model LiteLLM_VerificationToken { model_max_budget Json @default("{}") budget_id String? organization_id String? + object_permission_id String? created_at DateTime? @default(now()) @map("created_at") created_by String? updated_at DateTime? @default(now()) @updatedAt @map("updated_at") updated_by String? litellm_budget_table LiteLLM_BudgetTable? @relation(fields: [budget_id], references: [budget_id]) litellm_organization_table LiteLLM_OrganizationTable? @relation(fields: [organization_id], references: [organization_id]) + object_permission LiteLLM_ObjectPermissionTable? @relation(fields: [object_permission_id], references: [object_permission_id]) } model LiteLLM_EndUserTable { @@ -227,9 +260,11 @@ model LiteLLM_SpendLogs { messages Json? @default("{}") response Json? @default("{}") session_id String? + status String? proxy_server_request Json? @default("{}") @@index([startTime]) @@index([end_user]) + @@index([session_id]) } // View spend, model, api_key per request @@ -327,14 +362,14 @@ model LiteLLM_DailyUserSpend { model String model_group String? custom_llm_provider String? - prompt_tokens Int @default(0) - completion_tokens Int @default(0) - cache_read_input_tokens Int @default(0) - cache_creation_input_tokens Int @default(0) + prompt_tokens BigInt @default(0) + completion_tokens BigInt @default(0) + cache_read_input_tokens BigInt @default(0) + cache_creation_input_tokens BigInt @default(0) spend Float @default(0.0) - api_requests Int @default(0) - successful_requests Int @default(0) - failed_requests Int @default(0) + api_requests BigInt @default(0) + successful_requests BigInt @default(0) + failed_requests BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -354,14 +389,14 @@ model LiteLLM_DailyTeamSpend { model String model_group String? custom_llm_provider String? - prompt_tokens Int @default(0) - completion_tokens Int @default(0) - cache_read_input_tokens Int @default(0) - cache_creation_input_tokens Int @default(0) + prompt_tokens BigInt @default(0) + completion_tokens BigInt @default(0) + cache_read_input_tokens BigInt @default(0) + cache_creation_input_tokens BigInt @default(0) spend Float @default(0.0) - api_requests Int @default(0) - successful_requests Int @default(0) - failed_requests Int @default(0) + api_requests BigInt @default(0) + successful_requests BigInt @default(0) + failed_requests BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -381,14 +416,14 @@ model LiteLLM_DailyTagSpend { model String model_group String? custom_llm_provider String? - prompt_tokens Int @default(0) - completion_tokens Int @default(0) - cache_read_input_tokens Int @default(0) - cache_creation_input_tokens Int @default(0) + prompt_tokens BigInt @default(0) + completion_tokens BigInt @default(0) + cache_read_input_tokens BigInt @default(0) + cache_creation_input_tokens BigInt @default(0) spend Float @default(0.0) - api_requests Int @default(0) - successful_requests Int @default(0) - failed_requests Int @default(0) + api_requests BigInt @default(0) + successful_requests BigInt @default(0) + failed_requests BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -435,4 +470,14 @@ model LiteLLM_ManagedVectorStoresTable { created_at DateTime @default(now()) updated_at DateTime @updatedAt litellm_credential_name String? +} + +// Guardrails table for storing guardrail configurations +model LiteLLM_GuardrailsTable { + guardrail_id String @id @default(uuid()) + guardrail_name String @unique + litellm_params Json + guardrail_info Json? + created_at DateTime @default(now()) + updated_at DateTime @updatedAt } \ No newline at end of file diff --git a/litellm-proxy-extras/poetry.lock b/litellm-proxy-extras/poetry.lock index f526fec8da0..bb436a168cd 100644 --- a/litellm-proxy-extras/poetry.lock +++ b/litellm-proxy-extras/poetry.lock @@ -1,7 +1,7 @@ -# This file is automatically @generated by Poetry 1.8.3 and should not be changed by hand. +# This file is automatically @generated by Poetry 2.1.2 and should not be changed by hand. package = [] [metadata] -lock-version = "2.0" +lock-version = "2.1" python-versions = ">=3.8.1,<4.0, !=3.9.7" content-hash = "2cf39473e67ff0615f0a61c9d2ac9f02b38cc08cbb1bdb893d89bee002646623" diff --git a/litellm-proxy-extras/pyproject.toml b/litellm-proxy-extras/pyproject.toml index c001a4397da..1246a9233cc 100644 --- a/litellm-proxy-extras/pyproject.toml +++ b/litellm-proxy-extras/pyproject.toml @@ -1,6 +1,6 @@ [tool.poetry] name = "litellm-proxy-extras" -version = "0.1.15" +version = "0.1.21" description = "Additional files for the LiteLLM Proxy. Reduces the size of the main litellm package." authors = ["BerriAI"] readme = "README.md" @@ -22,7 +22,7 @@ requires = ["poetry-core"] build-backend = "poetry.core.masonry.api" [tool.commitizen] -version = "0.1.15" +version = "0.1.21" version_files = [ "pyproject.toml:version", "../requirements.txt:litellm-proxy-extras==", diff --git a/litellm/__init__.py b/litellm/__init__.py index b507e52bdbe..96c1552c36a 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -115,6 +115,9 @@ _custom_logger_compatible_callbacks_literal = Literal[ "agentops", "anthropic_cache_control_hook", "bedrock_vector_store", + "generic_api", + "resend_email", + "smtp_email", ] logged_real_time_event_types: Optional[Union[List[str], Literal["*"]]] = None _known_custom_logger_compatible_callbacks: List = list( @@ -132,6 +135,9 @@ datadog_use_v1: Optional[bool] = False # if you want to use v1 datadog logged p gcs_pub_sub_use_v1: Optional[bool] = ( False # if you want to use v1 gcs pubsub logged payload ) +generic_api_use_v1: Optional[bool] = ( + False # if you want to use v1 generic api logged payload +) argilla_transformation_object: Optional[Dict[str, Any]] = None _async_input_callback: List[Union[str, Callable, CustomLogger]] = ( [] @@ -190,13 +196,18 @@ predibase_tenant_id: Optional[str] = None togetherai_api_key: Optional[str] = None cloudflare_api_key: Optional[str] = None baseten_key: Optional[str] = None +llama_api_key: Optional[str] = None aleph_alpha_key: Optional[str] = None nlp_cloud_key: Optional[str] = None +novita_api_key: Optional[str] = None snowflake_key: Optional[str] = None common_cloud_provider_auth_params: dict = { "params": ["project", "region_name", "token"], "providers": ["vertex_ai", "bedrock", "watsonx", "azure", "vertex_ai_beta"], } +use_litellm_proxy: bool = ( + False # when True, requests will be sent to the specified litellm proxy endpoint +) use_client: bool = False ssl_verify: Union[str, bool] = True ssl_certificate: Optional[str] = None @@ -422,6 +433,7 @@ databricks_models: List = [] cloudflare_models: List = [] codestral_models: List = [] friendliai_models: List = [] +featherless_ai_models: List = [] palm_models: List = [] groq_models: List = [] azure_models: List = [] @@ -430,9 +442,11 @@ anyscale_models: List = [] cerebras_models: List = [] galadriel_models: List = [] sambanova_models: List = [] +novita_models: List = [] assemblyai_models: List = [] snowflake_models: List = [] - +llama_models: List = [] +nscale_models: List = [] def is_bedrock_pricing_only_model(key: str) -> bool: """ @@ -555,6 +569,10 @@ def add_known_models(): xai_models.append(key) elif value.get("litellm_provider") == "deepseek": deepseek_models.append(key) + elif value.get("litellm_provider") == "meta_llama": + llama_models.append(key) + elif value.get("litellm_provider") == "nscale": + nscale_models.append(key) elif value.get("litellm_provider") == "azure_ai": azure_ai_models.append(key) elif value.get("litellm_provider") == "voyage": @@ -581,14 +599,18 @@ def add_known_models(): cerebras_models.append(key) elif value.get("litellm_provider") == "galadriel": galadriel_models.append(key) - elif value.get("litellm_provider") == "sambanova_models": + elif value.get("litellm_provider") == "sambanova": sambanova_models.append(key) + elif value.get("litellm_provider") == "novita": + novita_models.append(key) elif value.get("litellm_provider") == "assemblyai": assemblyai_models.append(key) elif value.get("litellm_provider") == "jina_ai": jina_ai_models.append(key) elif value.get("litellm_provider") == "snowflake": snowflake_models.append(key) + elif value.get("litellm_provider") == "featherless_ai": + featherless_ai_models.append(key) add_known_models() @@ -662,9 +684,13 @@ model_list = ( + galadriel_models + sambanova_models + azure_text_models + + novita_models + assemblyai_models + jina_ai_models + snowflake_models + + llama_models + + featherless_ai_models + + nscale_models ) model_list_set = set(model_list) @@ -719,9 +745,13 @@ models_by_provider: dict = { "cerebras": cerebras_models, "galadriel": galadriel_models, "sambanova": sambanova_models, + "novita": novita_models, "assemblyai": assemblyai_models, "jina_ai": jina_ai_models, "snowflake": snowflake_models, + "meta_llama": llama_models, + "nscale": nscale_models, + "featherless_ai": featherless_ai_models, } # mapping for those models which have larger equivalents @@ -848,12 +878,17 @@ from .llms.infinity.rerank.transformation import InfinityRerankConfig from .llms.jina_ai.rerank.transformation import JinaAIRerankConfig from .llms.clarifai.chat.transformation import ClarifaiConfig from .llms.ai21.chat.transformation import AI21ChatConfig, AI21ChatConfig as AI21Config +from .llms.meta_llama.chat.transformation import LlamaAPIConfig from .llms.anthropic.experimental_pass_through.messages.transformation import ( AnthropicMessagesConfig, ) +from .llms.bedrock.messages.invoke_transformations.anthropic_claude3_transformation import ( + AmazonAnthropicClaude3MessagesConfig, +) from .llms.together_ai.chat import TogetherAIConfig from .llms.together_ai.completion.transformation import TogetherAITextCompletionConfig from .llms.cloudflare.chat.transformation import CloudflareChatConfig +from .llms.novita.chat.transformation import NovitaConfig from .llms.deprecated_providers.palm import ( PalmConfig, ) # here to prevent breaking changes @@ -983,12 +1018,13 @@ from .llms.openai.chat.gpt_audio_transformation import ( openAIGPTAudioConfig = OpenAIGPTAudioConfig() -from .llms.nvidia_nim.chat import NvidiaNimConfig +from .llms.nvidia_nim.chat.transformation import NvidiaNimConfig from .llms.nvidia_nim.embed import NvidiaNimEmbeddingConfig nvidiaNimConfig = NvidiaNimConfig() nvidiaNimEmbeddingConfig = NvidiaNimEmbeddingConfig() +from .llms.featherless_ai.chat.transformation import FeatherlessAIConfig from .llms.cerebras.chat import CerebrasConfig from .llms.sambanova.chat import SambanovaConfig from .llms.ai21.chat.transformation import AI21ChatConfig @@ -1020,6 +1056,7 @@ from .llms.vllm.completion.transformation import VLLMConfig from .llms.deepseek.chat.transformation import DeepSeekChatConfig from .llms.lm_studio.chat.transformation import LMStudioChatConfig from .llms.lm_studio.embed.transformation import LmStudioEmbeddingConfig +from .llms.nscale.chat.transformation import NscaleConfig from .llms.perplexity.chat.transformation import PerplexityChatConfig from .llms.azure.chat.o_series_transformation import AzureOpenAIO1Config from .llms.watsonx.completion.transformation import IBMWatsonXAIConfig diff --git a/litellm/_logging.py b/litellm/_logging.py index d7e2c9e7783..356bb3dcaf7 100644 --- a/litellm/_logging.py +++ b/litellm/_logging.py @@ -108,26 +108,36 @@ verbose_router_logger.addHandler(handler) verbose_proxy_logger.addHandler(handler) verbose_logger.addHandler(handler) +ALL_LOGGERS = [ + logging.getLogger(), + verbose_logger, + verbose_router_logger, + verbose_proxy_logger, +] + + +def _initialize_loggers_with_handler(handler: logging.Handler): + """ + Initialize all loggers with a handler + + - Adds a handler to each logger + - Prevents bubbling to parent/root (critical to prevent duplicate JSON logs) + """ + for lg in ALL_LOGGERS: + lg.handlers.clear() # remove any existing handlers + lg.addHandler(handler) # add JSON formatter handler + lg.propagate = False # prevent bubbling to parent/root + def _turn_on_json(): + """ + Turn on JSON logging + + - Adds a JSON formatter to all loggers + """ handler = logging.StreamHandler() handler.setFormatter(JsonFormatter()) - - # Define all loggers to update, including root logger - loggers = [logging.getLogger()] + [ - verbose_router_logger, - verbose_proxy_logger, - verbose_logger, - ] - - # Iterate through each logger and update its handlers - for logger in loggers: - # Remove all existing handlers - for h in logger.handlers[:]: - logger.removeHandler(h) - # Add the new handler - logger.addHandler(handler) - + _initialize_loggers_with_handler(handler) # Set up exception handlers _setup_json_exception_handlers(JsonFormatter()) diff --git a/litellm/_service_logger.py b/litellm/_service_logger.py index 7a60359d544..969a9ef1483 100644 --- a/litellm/_service_logger.py +++ b/litellm/_service_logger.py @@ -276,6 +276,7 @@ class ServiceLogging(CustomLogger): request_data: dict, original_exception: Exception, user_api_key_dict: UserAPIKeyAuth, + traceback_str: Optional[str] = None, ): """ Hook to track failed litellm-service calls diff --git a/litellm/anthropic_interface/messages/__init__.py b/litellm/anthropic_interface/messages/__init__.py index f3249f981b1..15becd43af0 100644 --- a/litellm/anthropic_interface/messages/__init__.py +++ b/litellm/anthropic_interface/messages/__init__.py @@ -28,7 +28,7 @@ async def acreate( stop_sequences: Optional[List[str]] = None, stream: Optional[bool] = False, system: Optional[str] = None, - temperature: Optional[float] = 1.0, + temperature: Optional[float] = None, thinking: Optional[Dict] = None, tool_choice: Optional[Dict] = None, tools: Optional[List[Dict]] = None, @@ -84,7 +84,7 @@ async def create( stop_sequences: Optional[List[str]] = None, stream: Optional[bool] = False, system: Optional[str] = None, - temperature: Optional[float] = 1.0, + temperature: Optional[float] = None, thinking: Optional[Dict] = None, tool_choice: Optional[Dict] = None, tools: Optional[List[Dict]] = None, diff --git a/litellm/batches/main.py b/litellm/batches/main.py index 0be96677905..98527556226 100644 --- a/litellm/batches/main.py +++ b/litellm/batches/main.py @@ -51,7 +51,7 @@ async def acreate_batch( extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, **kwargs, -) -> Batch: +) -> LiteLLMBatch: """ Async: Creates and executes a batch from an uploaded file of request diff --git a/litellm/caching/caching_handler.py b/litellm/caching/caching_handler.py index afe96c7f468..6b41c1ff40a 100644 --- a/litellm/caching/caching_handler.py +++ b/litellm/caching/caching_handler.py @@ -515,9 +515,11 @@ class LLMCachingHandler: ) ) cached_result: Optional[Any] = None - if call_type == CallTypes.aembedding.value and isinstance( - new_kwargs["input"], list - ): + if call_type == CallTypes.aembedding.value: + if isinstance(new_kwargs["input"], str): + new_kwargs["input"] = [new_kwargs["input"]] + elif not isinstance(new_kwargs["input"], list): + raise ValueError("input must be a string or a list") tasks = [] for idx, i in enumerate(new_kwargs["input"]): preset_cache_key = litellm.cache.get_cache_key( @@ -700,6 +702,7 @@ class LLMCachingHandler: Raises: None """ + if litellm.cache is None: return diff --git a/litellm/constants.py b/litellm/constants.py index 9526ae141a9..e224c0dd69a 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1,30 +1,51 @@ +import os from typing import List, Literal -ROUTER_MAX_FALLBACKS = 5 -DEFAULT_BATCH_SIZE = 512 -DEFAULT_FLUSH_INTERVAL_SECONDS = 5 -DEFAULT_MAX_RETRIES = 2 -DEFAULT_MAX_RECURSE_DEPTH = 10 -DEFAULT_FAILURE_THRESHOLD_PERCENT = ( - 0.5 # default cooldown a deployment if 50% of requests fail in a given minute +ROUTER_MAX_FALLBACKS = int(os.getenv("ROUTER_MAX_FALLBACKS", 5)) +DEFAULT_BATCH_SIZE = int(os.getenv("DEFAULT_BATCH_SIZE", 512)) +DEFAULT_FLUSH_INTERVAL_SECONDS = int(os.getenv("DEFAULT_FLUSH_INTERVAL_SECONDS", 5)) +DEFAULT_MAX_RETRIES = int(os.getenv("DEFAULT_MAX_RETRIES", 2)) +DEFAULT_MAX_RECURSE_DEPTH = int(os.getenv("DEFAULT_MAX_RECURSE_DEPTH", 100)) +DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER = int( + os.getenv("DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER", 10) ) -DEFAULT_MAX_TOKENS = 4096 -DEFAULT_ALLOWED_FAILS = 3 -DEFAULT_REDIS_SYNC_INTERVAL = 1 -DEFAULT_COOLDOWN_TIME_SECONDS = 5 -DEFAULT_REPLICATE_POLLING_RETRIES = 5 -DEFAULT_REPLICATE_POLLING_DELAY_SECONDS = 1 -DEFAULT_IMAGE_TOKEN_COUNT = 250 -DEFAULT_IMAGE_WIDTH = 300 -DEFAULT_IMAGE_HEIGHT = 300 -DEFAULT_MAX_TOKENS = 256 # used when providers need a default -MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = 1024 # 1MB = 1024KB -SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD = 1000 # Minimum number of requests to consider "reasonable traffic". Used for single-deployment cooldown logic. +DEFAULT_FAILURE_THRESHOLD_PERCENT = float( + os.getenv("DEFAULT_FAILURE_THRESHOLD_PERCENT", 0.5) +) # default cooldown a deployment if 50% of requests fail in a given minute +DEFAULT_MAX_TOKENS = int(os.getenv("DEFAULT_MAX_TOKENS", 4096)) +DEFAULT_ALLOWED_FAILS = int(os.getenv("DEFAULT_ALLOWED_FAILS", 3)) +DEFAULT_REDIS_SYNC_INTERVAL = int(os.getenv("DEFAULT_REDIS_SYNC_INTERVAL", 1)) +DEFAULT_COOLDOWN_TIME_SECONDS = int(os.getenv("DEFAULT_COOLDOWN_TIME_SECONDS", 5)) +DEFAULT_REPLICATE_POLLING_RETRIES = int( + os.getenv("DEFAULT_REPLICATE_POLLING_RETRIES", 5) +) +DEFAULT_REPLICATE_POLLING_DELAY_SECONDS = int( + os.getenv("DEFAULT_REPLICATE_POLLING_DELAY_SECONDS", 1) +) +DEFAULT_IMAGE_TOKEN_COUNT = int(os.getenv("DEFAULT_IMAGE_TOKEN_COUNT", 250)) +DEFAULT_IMAGE_WIDTH = int(os.getenv("DEFAULT_IMAGE_WIDTH", 300)) +DEFAULT_IMAGE_HEIGHT = int(os.getenv("DEFAULT_IMAGE_HEIGHT", 300)) +MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = int( + os.getenv("MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB", 1024) +) # 1MB = 1024KB +SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD = int( + os.getenv("SINGLE_DEPLOYMENT_TRAFFIC_FAILURE_THRESHOLD", 1000) +) # Minimum number of requests to consider "reasonable traffic". Used for single-deployment cooldown logic. + +DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET = int( + os.getenv("DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET", 1024) +) +DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET = int( + os.getenv("DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET", 2048) +) +DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET = int( + os.getenv("DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET", 4096) +) +MAX_TOKEN_TRIMMING_ATTEMPTS = int( + os.getenv("MAX_TOKEN_TRIMMING_ATTEMPTS", 10) +) # Maximum number of attempts to trim the message + -DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET = 1024 -DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET = 2048 -DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET = 4096 -MAX_TOKEN_TRIMMING_ATTEMPTS = 10 # Maximum number of attempts to trim the message ########## Networking constants ############################################################## _DEFAULT_TTL_FOR_HTTPX_CLIENTS = 3600 # 1 hour, re-use the same httpx client for 1 hour @@ -33,73 +54,112 @@ REDIS_UPDATE_BUFFER_KEY = "litellm_spend_update_buffer" REDIS_DAILY_SPEND_UPDATE_BUFFER_KEY = "litellm_daily_spend_update_buffer" REDIS_DAILY_TEAM_SPEND_UPDATE_BUFFER_KEY = "litellm_daily_team_spend_update_buffer" REDIS_DAILY_TAG_SPEND_UPDATE_BUFFER_KEY = "litellm_daily_tag_spend_update_buffer" -MAX_REDIS_BUFFER_DEQUEUE_COUNT = 100 -MAX_SIZE_IN_MEMORY_QUEUE = 10000 -MAX_IN_MEMORY_QUEUE_FLUSH_COUNT = 1000 -############################################################################################### -MINIMUM_PROMPT_CACHE_TOKEN_COUNT = ( - 1024 # minimum number of tokens to cache a prompt by Anthropic +MAX_REDIS_BUFFER_DEQUEUE_COUNT = int(os.getenv("MAX_REDIS_BUFFER_DEQUEUE_COUNT", 100)) +MAX_SIZE_IN_MEMORY_QUEUE = int(os.getenv("MAX_SIZE_IN_MEMORY_QUEUE", 10000)) +MAX_IN_MEMORY_QUEUE_FLUSH_COUNT = int( + os.getenv("MAX_IN_MEMORY_QUEUE_FLUSH_COUNT", 1000) +) +############################################################################################### +MINIMUM_PROMPT_CACHE_TOKEN_COUNT = int( + os.getenv("MINIMUM_PROMPT_CACHE_TOKEN_COUNT", 1024) +) # minimum number of tokens to cache a prompt by Anthropic +DEFAULT_TRIM_RATIO = float( + os.getenv("DEFAULT_TRIM_RATIO", 0.75) +) # default ratio of tokens to trim from the end of a prompt +HOURS_IN_A_DAY = int(os.getenv("HOURS_IN_A_DAY", 24)) +DAYS_IN_A_WEEK = int(os.getenv("DAYS_IN_A_WEEK", 7)) +DAYS_IN_A_MONTH = int(os.getenv("DAYS_IN_A_MONTH", 28)) +DAYS_IN_A_YEAR = int(os.getenv("DAYS_IN_A_YEAR", 365)) +REPLICATE_MODEL_NAME_WITH_ID_LENGTH = int( + os.getenv("REPLICATE_MODEL_NAME_WITH_ID_LENGTH", 64) ) -DEFAULT_TRIM_RATIO = 0.75 # default ratio of tokens to trim from the end of a prompt -HOURS_IN_A_DAY = 24 -DAYS_IN_A_WEEK = 7 -DAYS_IN_A_MONTH = 28 -DAYS_IN_A_YEAR = 365 -REPLICATE_MODEL_NAME_WITH_ID_LENGTH = 64 #### TOKEN COUNTING #### -FUNCTION_DEFINITION_TOKEN_COUNT = 9 -SYSTEM_MESSAGE_TOKEN_COUNT = 4 -TOOL_CHOICE_OBJECT_TOKEN_COUNT = 4 -DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT = 10 -DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT = 20 -MAX_SHORT_SIDE_FOR_IMAGE_HIGH_RES = 768 -MAX_LONG_SIDE_FOR_IMAGE_HIGH_RES = 2000 -MAX_TILE_WIDTH = 512 -MAX_TILE_HEIGHT = 512 -OPENAI_FILE_SEARCH_COST_PER_1K_CALLS = 2.5 / 1000 -MIN_NON_ZERO_TEMPERATURE = 0.0001 +FUNCTION_DEFINITION_TOKEN_COUNT = int(os.getenv("FUNCTION_DEFINITION_TOKEN_COUNT", 9)) +SYSTEM_MESSAGE_TOKEN_COUNT = int(os.getenv("SYSTEM_MESSAGE_TOKEN_COUNT", 4)) +TOOL_CHOICE_OBJECT_TOKEN_COUNT = int(os.getenv("TOOL_CHOICE_OBJECT_TOKEN_COUNT", 4)) +DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT = int( + os.getenv("DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT", 10) +) +DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT = int( + os.getenv("DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT", 20) +) +MAX_SHORT_SIDE_FOR_IMAGE_HIGH_RES = int( + os.getenv("MAX_SHORT_SIDE_FOR_IMAGE_HIGH_RES", 768) +) +MAX_LONG_SIDE_FOR_IMAGE_HIGH_RES = int( + os.getenv("MAX_LONG_SIDE_FOR_IMAGE_HIGH_RES", 2000) +) +MAX_TILE_WIDTH = int(os.getenv("MAX_TILE_WIDTH", 512)) +MAX_TILE_HEIGHT = int(os.getenv("MAX_TILE_HEIGHT", 512)) +OPENAI_FILE_SEARCH_COST_PER_1K_CALLS = float( + os.getenv("OPENAI_FILE_SEARCH_COST_PER_1K_CALLS", 2.5 / 1000) +) +MIN_NON_ZERO_TEMPERATURE = float(os.getenv("MIN_NON_ZERO_TEMPERATURE", 0.0001)) #### RELIABILITY #### -REPEATED_STREAMING_CHUNK_LIMIT = 100 # catch if model starts looping the same chunk while streaming. Uses high default to prevent false positives. -DEFAULT_MAX_LRU_CACHE_SIZE = 16 -INITIAL_RETRY_DELAY = 0.5 -MAX_RETRY_DELAY = 8.0 -JITTER = 0.75 -DEFAULT_IN_MEMORY_TTL = 5 # default time to live for the in-memory cache -DEFAULT_POLLING_INTERVAL = 0.03 # default polling interval for the scheduler -AZURE_OPERATION_POLLING_TIMEOUT = 120 -REDIS_SOCKET_TIMEOUT = 0.1 -REDIS_CONNECTION_POOL_TIMEOUT = 5 -NON_LLM_CONNECTION_TIMEOUT = 15 # timeout for adjacent services (e.g. jwt auth) -MAX_EXCEPTION_MESSAGE_LENGTH = 2000 -BEDROCK_MAX_POLICY_SIZE = 75 -REPLICATE_POLLING_DELAY_SECONDS = 0.5 -DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS = 4096 -TOGETHER_AI_4_B = 4 -TOGETHER_AI_8_B = 8 -TOGETHER_AI_21_B = 21 -TOGETHER_AI_41_B = 41 -TOGETHER_AI_80_B = 80 -TOGETHER_AI_110_B = 110 -TOGETHER_AI_EMBEDDING_150_M = 150 -TOGETHER_AI_EMBEDDING_350_M = 350 -QDRANT_SCALAR_QUANTILE = 0.99 -QDRANT_VECTOR_SIZE = 1536 -CACHED_STREAMING_CHUNK_DELAY = 0.02 -MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = 512 -DEFAULT_MAX_TOKENS_FOR_TRITON = 2000 +REPEATED_STREAMING_CHUNK_LIMIT = int( + os.getenv("REPEATED_STREAMING_CHUNK_LIMIT", 100) +) # catch if model starts looping the same chunk while streaming. Uses high default to prevent false positives. +DEFAULT_MAX_LRU_CACHE_SIZE = int(os.getenv("DEFAULT_MAX_LRU_CACHE_SIZE", 16)) +INITIAL_RETRY_DELAY = float(os.getenv("INITIAL_RETRY_DELAY", 0.5)) +MAX_RETRY_DELAY = float(os.getenv("MAX_RETRY_DELAY", 8.0)) +JITTER = float(os.getenv("JITTER", 0.75)) +DEFAULT_IN_MEMORY_TTL = int( + os.getenv("DEFAULT_IN_MEMORY_TTL", 5) +) # default time to live for the in-memory cache +DEFAULT_POLLING_INTERVAL = float( + os.getenv("DEFAULT_POLLING_INTERVAL", 0.03) +) # default polling interval for the scheduler +AZURE_OPERATION_POLLING_TIMEOUT = int(os.getenv("AZURE_OPERATION_POLLING_TIMEOUT", 120)) +REDIS_SOCKET_TIMEOUT = float(os.getenv("REDIS_SOCKET_TIMEOUT", 0.1)) +REDIS_CONNECTION_POOL_TIMEOUT = int(os.getenv("REDIS_CONNECTION_POOL_TIMEOUT", 5)) +NON_LLM_CONNECTION_TIMEOUT = int( + os.getenv("NON_LLM_CONNECTION_TIMEOUT", 15) +) # timeout for adjacent services (e.g. jwt auth) +MAX_EXCEPTION_MESSAGE_LENGTH = int(os.getenv("MAX_EXCEPTION_MESSAGE_LENGTH", 2000)) +BEDROCK_MAX_POLICY_SIZE = int(os.getenv("BEDROCK_MAX_POLICY_SIZE", 75)) +REPLICATE_POLLING_DELAY_SECONDS = float( + os.getenv("REPLICATE_POLLING_DELAY_SECONDS", 0.5) +) +DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS = int( + os.getenv("DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS", 4096) +) +TOGETHER_AI_4_B = int(os.getenv("TOGETHER_AI_4_B", 4)) +TOGETHER_AI_8_B = int(os.getenv("TOGETHER_AI_8_B", 8)) +TOGETHER_AI_21_B = int(os.getenv("TOGETHER_AI_21_B", 21)) +TOGETHER_AI_41_B = int(os.getenv("TOGETHER_AI_41_B", 41)) +TOGETHER_AI_80_B = int(os.getenv("TOGETHER_AI_80_B", 80)) +TOGETHER_AI_110_B = int(os.getenv("TOGETHER_AI_110_B", 110)) +TOGETHER_AI_EMBEDDING_150_M = int(os.getenv("TOGETHER_AI_EMBEDDING_150_M", 150)) +TOGETHER_AI_EMBEDDING_350_M = int(os.getenv("TOGETHER_AI_EMBEDDING_350_M", 350)) +QDRANT_SCALAR_QUANTILE = float(os.getenv("QDRANT_SCALAR_QUANTILE", 0.99)) +QDRANT_VECTOR_SIZE = int(os.getenv("QDRANT_VECTOR_SIZE", 1536)) +CACHED_STREAMING_CHUNK_DELAY = float(os.getenv("CACHED_STREAMING_CHUNK_DELAY", 0.02)) +MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB = int( + os.getenv("MAX_SIZE_PER_ITEM_IN_MEMORY_CACHE_IN_KB", 512) +) +DEFAULT_MAX_TOKENS_FOR_TRITON = int(os.getenv("DEFAULT_MAX_TOKENS_FOR_TRITON", 2000)) #### Networking settings #### -request_timeout: float = 6000 # time in seconds +request_timeout: float = float(os.getenv("REQUEST_TIMEOUT", 6000)) # time in seconds STREAM_SSE_DONE_STRING: str = "[DONE]" ### SPEND TRACKING ### -DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND = 0.001400 # price per second for a100 80GB -FIREWORKS_AI_56_B_MOE = 56 -FIREWORKS_AI_176_B_MOE = 176 -FIREWORKS_AI_4_B = 4 -FIREWORKS_AI_16_B = 16 -FIREWORKS_AI_80_B = 80 +DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND = float( + os.getenv("DEFAULT_REPLICATE_GPU_PRICE_PER_SECOND", 0.001400) +) # price per second for a100 80GB +FIREWORKS_AI_56_B_MOE = int(os.getenv("FIREWORKS_AI_56_B_MOE", 56)) +FIREWORKS_AI_176_B_MOE = int(os.getenv("FIREWORKS_AI_176_B_MOE", 176)) +FIREWORKS_AI_4_B = int(os.getenv("FIREWORKS_AI_4_B", 4)) +FIREWORKS_AI_16_B = int(os.getenv("FIREWORKS_AI_16_B", 16)) +FIREWORKS_AI_80_B = int(os.getenv("FIREWORKS_AI_80_B", 80)) #### Logging callback constants #### REDACTED_BY_LITELM_STRING = "REDACTED_BY_LITELM" +### ANTHROPIC CONSTANTS ### +ANTHROPIC_WEB_SEARCH_TOOL_MAX_USES = { + "low": 1, + "medium": 5, + "high": 10, +} + LITELLM_CHAT_PROVIDERS = [ "openai", "openai_like", @@ -161,6 +221,16 @@ LITELLM_CHAT_PROVIDERS = [ "llamafile", "lm_studio", "galadriel", + "novita", + "meta_llama", + "featherless_ai", + "nscale", +] + +LITELLM_EMBEDDING_PROVIDERS_SUPPORTING_INPUT_ARRAY_OF_TOKENS = [ + "openai", + "azure", + "hosted_vllm", ] @@ -203,6 +273,7 @@ OPENAI_CHAT_COMPLETION_PARAMS = [ "reasoning_effort", "extra_headers", "thinking", + "web_search_options", ] openai_compatible_endpoints: List = [ @@ -221,6 +292,9 @@ openai_compatible_endpoints: List = [ "api.sambanova.ai/v1", "api.x.ai/v1", "api.galadriel.ai/v1", + "api.llama.com/compat/v1/", + "api.featherless.ai/v1", + "inference.api.nscale.com/v1", ] @@ -251,13 +325,19 @@ openai_compatible_providers: List = [ "llamafile", "lm_studio", "galadriel", + "novita", + "meta_llama", + "featherless_ai", + "nscale", ] openai_text_completion_compatible_providers: List = ( [ # providers that support `/v1/completions` "together_ai", "fireworks_ai", "hosted_vllm", + "meta_llama", "llamafile", + "featherless_ai", ] ) _openai_like_providers: List = [ @@ -404,6 +484,18 @@ baseten_models: List = [ "31dxrj3", ] # FALCON 7B # WizardLM # Mosaic ML +featherless_ai_models: List = [ + "featherless-ai/Qwerky-72B", + "featherless-ai/Qwerky-QwQ-32B", + "Qwen/Qwen2.5-72B-Instruct", + "all-hands/openhands-lm-32b-v0.1", + "Qwen/Qwen2.5-Coder-32B-Instruct", + "deepseek-ai/DeepSeek-V3-0324", + "mistralai/Mistral-Small-24B-Instruct-2501", + "mistralai/Mistral-Nemo-Instruct-2407", + "ProdeusUnity/Stellar-Odyssey-12b-v0.0", +] + BEDROCK_INVOKE_PROVIDERS_LITERAL = Literal[ "cohere", "anthropic", @@ -490,22 +582,27 @@ known_tokenizer_config = { OPENAI_FINISH_REASONS = ["stop", "length", "function_call", "content_filter", "null"] -HUMANLOOP_PROMPT_CACHE_TTL_SECONDS = 60 # 1 minute +HUMANLOOP_PROMPT_CACHE_TTL_SECONDS = int( + os.getenv("HUMANLOOP_PROMPT_CACHE_TTL_SECONDS", 60) +) # 1 minute RESPONSE_FORMAT_TOOL_NAME = "json_tool_call" # default tool name used when converting response format to tool call ########################### Logging Callback Constants ########################### AZURE_STORAGE_MSFT_VERSION = "2019-07-07" -PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES = 5 +PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES = int( + os.getenv("PROMETHEUS_BUDGET_METRICS_REFRESH_INTERVAL_MINUTES", 5) +) MCP_TOOL_NAME_PREFIX = "mcp_tool" +MAXIMUM_TRACEBACK_LINES_TO_LOG = int(os.getenv("MAXIMUM_TRACEBACK_LINES_TO_LOG", 100)) ########################### LiteLLM Proxy Specific Constants ########################### ######################################################################################## -MAX_SPENDLOG_ROWS_TO_QUERY = ( - 1_000_000 # if spendLogs has more than 1M rows, do not query the DB -) -DEFAULT_SOFT_BUDGET = ( - 50.0 # by default all litellm proxy keys have a soft budget of 50.0 -) +MAX_SPENDLOG_ROWS_TO_QUERY = int( + os.getenv("MAX_SPENDLOG_ROWS_TO_QUERY", 1_000_000) +) # if spendLogs has more than 1M rows, do not query the DB +DEFAULT_SOFT_BUDGET = float( + os.getenv("DEFAULT_SOFT_BUDGET", 50.0) +) # by default all litellm proxy keys have a soft budget of 50.0 # makes it clear this is a rate limit error for a litellm virtual key RATE_LIMIT_ERROR_MESSAGE_FOR_VIRTUAL_KEY = "LiteLLM Virtual Key user_api_key_hash" @@ -520,26 +617,52 @@ BEDROCK_AGENT_RUNTIME_PASS_THROUGH_ROUTES = [ "optimize-prompt/", ] -BATCH_STATUS_POLL_INTERVAL_SECONDS = 3600 # 1 hour -BATCH_STATUS_POLL_MAX_ATTEMPTS = 24 # for 24 hours +BATCH_STATUS_POLL_INTERVAL_SECONDS = int( + os.getenv("BATCH_STATUS_POLL_INTERVAL_SECONDS", 3600) +) # 1 hour +BATCH_STATUS_POLL_MAX_ATTEMPTS = int( + os.getenv("BATCH_STATUS_POLL_MAX_ATTEMPTS", 24) +) # for 24 hours -HEALTH_CHECK_TIMEOUT_SECONDS = 60 # 60 seconds +HEALTH_CHECK_TIMEOUT_SECONDS = int( + os.getenv("HEALTH_CHECK_TIMEOUT_SECONDS", 60) +) # 60 seconds UI_SESSION_TOKEN_TEAM_ID = "litellm-dashboard" LITELLM_PROXY_ADMIN_NAME = "default_user_id" ########################### DB CRON JOB NAMES ########################### DB_SPEND_UPDATE_JOB_NAME = "db_spend_update_job" -PROMETHEUS_EMIT_BUDGET_METRICS_JOB_NAME = "prometheus_emit_budget_metrics_job" -DEFAULT_CRON_JOB_LOCK_TTL_SECONDS = 60 # 1 minute -PROXY_BUDGET_RESCHEDULER_MIN_TIME = 597 -PROXY_BUDGET_RESCHEDULER_MAX_TIME = 605 -PROXY_BATCH_WRITE_AT = 10 # in seconds -DEFAULT_HEALTH_CHECK_INTERVAL = 300 # 5 minutes -PROMETHEUS_FALLBACK_STATS_SEND_TIME_HOURS = 9 -DEFAULT_MODEL_CREATED_AT_TIME = 1677610602 # returns on `/models` endpoint -DEFAULT_SLACK_ALERTING_THRESHOLD = 300 -MAX_TEAM_LIST_LIMIT = 20 -DEFAULT_PROMPT_INJECTION_SIMILARITY_THRESHOLD = 0.7 -LENGTH_OF_LITELLM_GENERATED_KEY = 16 -SECRET_MANAGER_REFRESH_INTERVAL = 86400 +PROMETHEUS_EMIT_BUDGET_METRICS_JOB_NAME = "prometheus_emit_budget_metrics" +SPEND_LOG_CLEANUP_JOB_NAME = "spend_log_cleanup" +SPEND_LOG_RUN_LOOPS = int(os.getenv("SPEND_LOG_RUN_LOOPS", 500)) +DEFAULT_CRON_JOB_LOCK_TTL_SECONDS = int( + os.getenv("DEFAULT_CRON_JOB_LOCK_TTL_SECONDS", 60) +) # 1 minute +PROXY_BUDGET_RESCHEDULER_MIN_TIME = int( + os.getenv("PROXY_BUDGET_RESCHEDULER_MIN_TIME", 597) +) +PROXY_BUDGET_RESCHEDULER_MAX_TIME = int( + os.getenv("PROXY_BUDGET_RESCHEDULER_MAX_TIME", 605) +) +PROXY_BATCH_WRITE_AT = int(os.getenv("PROXY_BATCH_WRITE_AT", 10)) # in seconds +DEFAULT_HEALTH_CHECK_INTERVAL = int( + os.getenv("DEFAULT_HEALTH_CHECK_INTERVAL", 300) +) # 5 minutes +PROMETHEUS_FALLBACK_STATS_SEND_TIME_HOURS = int( + os.getenv("PROMETHEUS_FALLBACK_STATS_SEND_TIME_HOURS", 9) +) +DEFAULT_MODEL_CREATED_AT_TIME = int( + os.getenv("DEFAULT_MODEL_CREATED_AT_TIME", 1677610602) +) # returns on `/models` endpoint +DEFAULT_SLACK_ALERTING_THRESHOLD = int( + os.getenv("DEFAULT_SLACK_ALERTING_THRESHOLD", 300) +) +MAX_TEAM_LIST_LIMIT = int(os.getenv("MAX_TEAM_LIST_LIMIT", 20)) +DEFAULT_PROMPT_INJECTION_SIMILARITY_THRESHOLD = float( + os.getenv("DEFAULT_PROMPT_INJECTION_SIMILARITY_THRESHOLD", 0.7) +) +LENGTH_OF_LITELLM_GENERATED_KEY = int(os.getenv("LENGTH_OF_LITELLM_GENERATED_KEY", 16)) +SECRET_MANAGER_REFRESH_INTERVAL = int( + os.getenv("SECRET_MANAGER_REFRESH_INTERVAL", 86400) +) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index f7c13827d6b..041e8b4c388 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -909,6 +909,7 @@ def completion_cost( # noqa: PLR0915 StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( model=model, response_object=completion_response, + usage=cost_per_token_usage_object, standard_built_in_tools_params=standard_built_in_tools_params, custom_llm_provider=custom_llm_provider, ) diff --git a/litellm/exceptions.py b/litellm/exceptions.py index 3cdc70b08aa..9f3411143a6 100644 --- a/litellm/exceptions.py +++ b/litellm/exceptions.py @@ -807,3 +807,25 @@ class LiteLLMUnknownProvider(BadRequestError): def __str__(self): return self.message + + +class GuardrailRaisedException(Exception): + def __init__(self, guardrail_name: Optional[str] = None, message: str = ""): + self.guardrail_name = guardrail_name + self.message = f"Guardrail raised an exception, Guardrail: {guardrail_name}, Message: {message}" + super().__init__(self.message) + + +class BlockedPiiEntityError(Exception): + def __init__( + self, + entity_type: str, + guardrail_name: Optional[str] = None, + ): + """ + Raised when a blocked entity is detected by a guardrail. + """ + self.entity_type = entity_type + self.guardrail_name = guardrail_name + self.message = f"Blocked entity detected: {entity_type} by Guardrail: {guardrail_name}. This entity is not allowed to be used in this request." + super().__init__(self.message) diff --git a/litellm/files/main.py b/litellm/files/main.py index ded74cc6533..5d0dc05771a 100644 --- a/litellm/files/main.py +++ b/litellm/files/main.py @@ -15,6 +15,7 @@ import httpx import litellm from litellm import get_secret_str +from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.azure.files.handler import AzureOpenAIFilesAPI from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler @@ -743,11 +744,13 @@ async def afile_content( try: loop = asyncio.get_event_loop() kwargs["afile_content"] = True + model = kwargs.pop("model", None) # Use a partial function to pass your keyword arguments func = partial( file_content, file_id, + model, custom_llm_provider, extra_headers, extra_body, @@ -770,7 +773,10 @@ async def afile_content( def file_content( file_id: str, - custom_llm_provider: Literal["openai", "azure"] = "openai", + model: Optional[str] = None, + custom_llm_provider: Optional[ + Union[Literal["openai", "azure", "vertex_ai"], str] + ] = None, extra_headers: Optional[Dict[str, str]] = None, extra_body: Optional[Dict[str, str]] = None, **kwargs, @@ -788,10 +794,18 @@ def file_content( client = kwargs.get("client") # set timeout for 10 minutes by default + try: + if model is not None: + _, custom_llm_provider, _, _ = get_llm_provider( + model, custom_llm_provider + ) + except Exception: + pass + if ( timeout is not None and isinstance(timeout, httpx.Timeout) - and supports_httpx_timeout(custom_llm_provider) is False + and supports_httpx_timeout(cast(str, custom_llm_provider)) is False ): read_timeout = timeout.read or 600 timeout = read_timeout # default 10 min timeout diff --git a/litellm/integrations/SlackAlerting/slack_alerting.py b/litellm/integrations/SlackAlerting/slack_alerting.py index 7e7aa4d370e..16305061ec8 100644 --- a/litellm/integrations/SlackAlerting/slack_alerting.py +++ b/litellm/integrations/SlackAlerting/slack_alerting.py @@ -85,6 +85,7 @@ class SlackAlerting(CustomBatchLogger): self.alerting_args = SlackAlertingArgs(**alerting_args) self.default_webhook_url = default_webhook_url self.flush_lock = asyncio.Lock() + self.periodic_started = False super().__init__(**kwargs, flush_lock=self.flush_lock) def update_values( @@ -99,12 +100,17 @@ class SlackAlerting(CustomBatchLogger): if alerting is not None: self.alerting = alerting asyncio.create_task(self.periodic_flush()) + self.periodic_started = True if alerting_threshold is not None: self.alerting_threshold = alerting_threshold if alert_types is not None: self.alert_types = alert_types if alerting_args is not None: self.alerting_args = SlackAlertingArgs(**alerting_args) + if not self.periodic_started: + asyncio.create_task(self.periodic_flush()) + self.periodic_started = True + if alert_to_webhook_url is not None: # update the dict if self.alert_to_webhook_url is None: diff --git a/litellm/integrations/arize/arize_phoenix.py b/litellm/integrations/arize/arize_phoenix.py index 2b4909885a3..044486fcd27 100644 --- a/litellm/integrations/arize/arize_phoenix.py +++ b/litellm/integrations/arize/arize_phoenix.py @@ -1,4 +1,5 @@ import os +import urllib.parse from typing import TYPE_CHECKING, Any, Union from litellm._logging import verbose_logger @@ -69,7 +70,7 @@ class ArizePhoenixLogger: otlp_auth_headers = f"api_key={api_key}" elif api_key is not None: # api_key/auth is optional for self hosted phoenix - otlp_auth_headers = f"Authorization=Bearer {api_key}" + otlp_auth_headers = f"Authorization={urllib.parse.quote(f'Bearer {api_key}')}" return ArizePhoenixConfig( otlp_auth_headers=otlp_auth_headers, protocol=protocol, endpoint=endpoint diff --git a/litellm/integrations/custom_guardrail.py b/litellm/integrations/custom_guardrail.py index 41a3800116e..a82eed8eb8f 100644 --- a/litellm/integrations/custom_guardrail.py +++ b/litellm/integrations/custom_guardrail.py @@ -1,8 +1,14 @@ +from datetime import datetime from typing import Dict, List, Literal, Optional, Union from litellm._logging import verbose_logger from litellm.integrations.custom_logger import CustomLogger -from litellm.types.guardrails import DynamicGuardrailParams, GuardrailEventHooks +from litellm.types.guardrails import ( + DynamicGuardrailParams, + GuardrailEventHooks, + LitellmParams, + PiiEntityType, +) from litellm.types.utils import StandardLoggingGuardrailInformation @@ -15,6 +21,8 @@ class CustomGuardrail(CustomLogger): Union[GuardrailEventHooks, List[GuardrailEventHooks]] ] = None, default_on: bool = False, + mask_request_content: bool = False, + mask_response_content: bool = False, **kwargs, ): """ @@ -25,6 +33,8 @@ class CustomGuardrail(CustomLogger): supported_event_hooks: The event hooks that the guardrail supports event_hook: The event hook to run the guardrail on default_on: If True, the guardrail will be run by default on all requests + mask_request_content: If True, the guardrail will mask the request content + mask_response_content: If True, the guardrail will mask the response content """ self.guardrail_name = guardrail_name self.supported_event_hooks = supported_event_hooks @@ -32,6 +42,8 @@ class CustomGuardrail(CustomLogger): Union[GuardrailEventHooks, List[GuardrailEventHooks]] ] = event_hook self.default_on: bool = default_on + self.mask_request_content: bool = mask_request_content + self.mask_response_content: bool = mask_response_content if supported_event_hooks: ## validate event_hook is in supported_event_hooks @@ -176,20 +188,17 @@ class CustomGuardrail(CustomLogger): def add_standard_logging_guardrail_information_to_request_data( self, - guardrail_json_response: Union[Exception, str, dict], + guardrail_json_response: Union[Exception, str, dict, List[dict]], request_data: dict, guardrail_status: Literal["success", "failure"], + start_time: Optional[float] = None, + end_time: Optional[float] = None, + duration: Optional[float] = None, + masked_entity_count: Optional[Dict[str, int]] = None, ) -> None: """ Builds `StandardLoggingGuardrailInformation` and adds it to the request metadata so it can be used for logging to DataDog, Langfuse, etc. """ - from litellm.proxy.proxy_server import premium_user - - if premium_user is not True: - verbose_logger.warning( - f"Guardrail Tracing is only available for premium users. Skipping guardrail logging for guardrail={self.guardrail_name} event_hook={self.event_hook}" - ) - return if isinstance(guardrail_json_response, Exception): guardrail_json_response = str(guardrail_json_response) slg = StandardLoggingGuardrailInformation( @@ -197,8 +206,14 @@ class CustomGuardrail(CustomLogger): guardrail_mode=self.event_hook, guardrail_response=guardrail_json_response, guardrail_status=guardrail_status, + start_time=start_time, + end_time=end_time, + duration=duration, + masked_entity_count=masked_entity_count, ) if "metadata" in request_data: + if request_data["metadata"] is None: + request_data["metadata"] = {} request_data["metadata"]["standard_logging_guardrail_information"] = slg elif "litellm_metadata" in request_data: request_data["litellm_metadata"][ @@ -209,6 +224,103 @@ class CustomGuardrail(CustomLogger): "unable to log guardrail information. No metadata found in request_data" ) + async def apply_guardrail( + self, + text: str, + language: Optional[str] = None, + entities: Optional[List[PiiEntityType]] = None, + ) -> str: + """ + Apply your guardrail logic to the given text + + Args: + text: The text to apply the guardrail to + language: The language of the text + entities: The entities to mask, optional + + Any of the custom guardrails can override this method to provide custom guardrail logic + + Returns the text with the guardrail applied + + Raises: + Exception: + - If the guardrail raises an exception + + """ + return text + + def _process_response( + self, + response: Optional[Dict], + request_data: dict, + start_time: Optional[float] = None, + end_time: Optional[float] = None, + duration: Optional[float] = None, + ): + """ + Add StandardLoggingGuardrailInformation to the request data + + This gets logged on downsteam Langfuse, DataDog, etc. + """ + # Convert None to empty dict to satisfy type requirements + guardrail_response = {} if response is None else response + self.add_standard_logging_guardrail_information_to_request_data( + guardrail_json_response=guardrail_response, + request_data=request_data, + guardrail_status="success", + duration=duration, + start_time=start_time, + end_time=end_time, + ) + return response + + def _process_error( + self, + e: Exception, + request_data: dict, + start_time: Optional[float] = None, + end_time: Optional[float] = None, + duration: Optional[float] = None, + ): + """ + Add StandardLoggingGuardrailInformation to the request data + + This gets logged on downsteam Langfuse, DataDog, etc. + """ + self.add_standard_logging_guardrail_information_to_request_data( + guardrail_json_response=e, + request_data=request_data, + guardrail_status="failure", + duration=duration, + start_time=start_time, + end_time=end_time, + ) + raise e + + def mask_content_in_string( + self, + content_string: str, + mask_string: str, + start_index: int, + end_index: int, + ) -> str: + """ + Mask the content in the string between the start and end indices. + """ + + # Do nothing if the start or end are not valid + if not (0 <= start_index < end_index <= len(content_string)): + return content_string + + # Mask the content + return content_string[:start_index] + mask_string + content_string[end_index:] + + def update_in_memory_litellm_params(self, litellm_params: LitellmParams) -> None: + """ + Update the guardrails litellm params in memory + """ + pass + def log_guardrail_information(func): """ @@ -224,21 +336,7 @@ def log_guardrail_information(func): import asyncio import functools - def process_response(self, response, request_data): - self.add_standard_logging_guardrail_information_to_request_data( - guardrail_json_response=response, - request_data=request_data, - guardrail_status="success", - ) - return response - - def process_error(self, e, request_data): - self.add_standard_logging_guardrail_information_to_request_data( - guardrail_json_response=e, - request_data=request_data, - guardrail_status="failure", - ) - raise e + start_time = datetime.now() @functools.wraps(func) async def async_wrapper(*args, **kwargs): @@ -248,9 +346,21 @@ def log_guardrail_information(func): ) try: response = await func(*args, **kwargs) - return process_response(self, response, request_data) + return self._process_response( + response=response, + request_data=request_data, + start_time=start_time.timestamp(), + end_time=datetime.now().timestamp(), + duration=(datetime.now() - start_time).total_seconds(), + ) except Exception as e: - return process_error(self, e, request_data) + return self._process_error( + e=e, + request_data=request_data, + start_time=start_time.timestamp(), + end_time=datetime.now().timestamp(), + duration=(datetime.now() - start_time).total_seconds(), + ) @functools.wraps(func) def sync_wrapper(*args, **kwargs): @@ -260,9 +370,17 @@ def log_guardrail_information(func): ) try: response = func(*args, **kwargs) - return process_response(self, response, request_data) + return self._process_response( + response=response, + request_data=request_data, + duration=(datetime.now() - start_time).total_seconds(), + ) except Exception as e: - return process_error(self, e, request_data) + return self._process_error( + e=e, + request_data=request_data, + duration=(datetime.now() - start_time).total_seconds(), + ) @functools.wraps(func) def wrapper(*args, **kwargs): diff --git a/litellm/integrations/custom_logger.py b/litellm/integrations/custom_logger.py index 17441ba1d0a..960dc715e7e 100644 --- a/litellm/integrations/custom_logger.py +++ b/litellm/integrations/custom_logger.py @@ -21,6 +21,7 @@ from litellm.types.integrations.argilla import ArgillaItem from litellm.types.llms.openai import AllMessageValues, ChatCompletionRequest from litellm.types.utils import ( AdapterCompletionStreamWrapper, + CallTypes, LLMResponseTypes, ModelResponse, ModelResponseStream, @@ -41,7 +42,7 @@ else: class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callback#callback-class # Class variables or attributes - def __init__(self, message_logging: bool = True) -> None: + def __init__(self, message_logging: bool = True, **kwargs) -> None: self.message_logging = message_logging pass @@ -127,6 +128,18 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac ) -> List[dict]: return healthy_deployments + async def async_pre_call_deployment_hook( + self, kwargs: Dict[str, Any], call_type: Optional[CallTypes] + ) -> Optional[dict]: + """ + Allow modifying the request just before it's sent to the deployment. + + Use this instead of 'async_pre_call_hook' when you need to modify the request AFTER a deployment is selected, but BEFORE the request is sent. + + Used in managed_files.py + """ + pass + async def async_pre_call_check( self, deployment: dict, parent_otel_span: Optional[Span] ) -> Optional[dict]: @@ -221,6 +234,7 @@ class CustomLogger: # https://docs.litellm.ai/docs/observability/custom_callbac request_data: dict, original_exception: Exception, user_api_key_dict: UserAPIKeyAuth, + traceback_str: Optional[str] = None, ): pass diff --git a/litellm/integrations/email_templates/email_footer.py b/litellm/integrations/email_templates/email_footer.py new file mode 100644 index 00000000000..feb692354a0 --- /dev/null +++ b/litellm/integrations/email_templates/email_footer.py @@ -0,0 +1,10 @@ +EMAIL_FOOTER = """ + +""" diff --git a/litellm/integrations/email_templates/key_created_email.py b/litellm/integrations/email_templates/key_created_email.py new file mode 100644 index 00000000000..3cf18d827c1 --- /dev/null +++ b/litellm/integrations/email_templates/key_created_email.py @@ -0,0 +1,212 @@ +""" +Modern Email Templates for LiteLLM Email Service with professional styling +""" + +KEY_CREATED_EMAIL_TEMPLATE = """ + + + + + + Your API Key is Ready + + + +
+
+ LiteLLM Logo +
+
+
+

Hi {recipient_email},

+
+ +
+

Great news! Your LiteLLM API key is ready to use.

+
+ +
+

Monthly Budget: {key_budget}

+
+ +
+
Your API Key
+
{key_token}
+
+ +

Quick Start Guide

+

Here's how to use your key with the OpenAI SDK:

+ +
+import openai
+
+client = openai.OpenAI(
+  api_key="{key_token}",
+  base_url="{base_url}"
+)
+
+response = client.chat.completions.create(
+  model="gpt-3.5-turbo", # model to send to the proxy
+  messages = [
+    {{
+      "role": "user",
+      "content": "this is a test request, write a short poem"
+    }}
+  ]
+) +
+ + View Documentation + +
+ +

Need Help?

+

If you have any questions or need assistance, please contact us at {email_support_contact}.

+
+ {email_footer} +
+ + +""" diff --git a/litellm/integrations/email_templates/user_invitation_email.py b/litellm/integrations/email_templates/user_invitation_email.py new file mode 100644 index 00000000000..68cda56a92b --- /dev/null +++ b/litellm/integrations/email_templates/user_invitation_email.py @@ -0,0 +1,175 @@ +""" +Modern Email Templates for LiteLLM Email Service with professional styling +""" + +USER_INVITATION_EMAIL_TEMPLATE = """ + + + + + + Welcome to LiteLLM + + + +
+ +
+

Welcome to LiteLLM

+ +
+

Hi {recipient_email},

+
+ +
+

LiteLLM allows you to call 100+ LLM providers in the OpenAI API format. Get started by accepting your invitation.

+
+ + + +
+

Here's a quickstart guide to get you started:

+
+ +
+ + + Make your first LLM request → + + + +

Making LLM requests with OpenAI SDK, Langchain, LlamaIndex, and more.

+ +
+ + + Supported Endpoints → + + + +

View all supported LLM endpoints on LiteLLM (/chat/completions, /embeddings, /responses etc.)

+ +
+ + + Passthrough Endpoints → + + + +

We support calling VertexAI, Anthropic, and other providers in their native API format.

+ +
+ +

Thanks for signing up. We're here to help you and your team. If you have any questions, contact us at {email_support_contact}

+ +
+ {email_footer} +
+ + +""" diff --git a/litellm/integrations/langfuse/langfuse.py b/litellm/integrations/langfuse/langfuse.py index ccc072149cc..f4d1a1f79c4 100644 --- a/litellm/integrations/langfuse/langfuse.py +++ b/litellm/integrations/langfuse/langfuse.py @@ -27,9 +27,12 @@ from litellm.types.utils import ( ) if TYPE_CHECKING: + from langfuse.client import StatefulTraceClient + from litellm.litellm_core_utils.litellm_logging import DynamicLoggingCache else: DynamicLoggingCache = Any + StatefulTraceClient = Any class LangFuseLogger: @@ -626,16 +629,17 @@ class LangFuseLogger: if key.lower() not in ["authorization", "cookie", "referer"]: clean_headers[key] = value - # clean_metadata["request"] = { - # "method": method, - # "url": url, - # "headers": clean_headers, - # } - trace = self.Langfuse.trace(**trace_params) + trace: StatefulTraceClient = self.Langfuse.trace(**trace_params) # Log provider specific information as a span log_provider_specific_information_as_span(trace, clean_metadata) + # Log guardrail information as a span + self._log_guardrail_information_as_span( + trace=trace, + standard_logging_object=standard_logging_object, + ) + generation_id = None usage = None usage_details = None @@ -818,6 +822,47 @@ class LangFuseLogger: """ return int(os.getenv("LANGFUSE_FLUSH_INTERVAL") or flush_interval) + def _log_guardrail_information_as_span( + self, + trace: StatefulTraceClient, + standard_logging_object: Optional[StandardLoggingPayload], + ): + """ + Log guardrail information as a span + """ + if standard_logging_object is None: + verbose_logger.debug( + "Not logging guardrail information as span because standard_logging_object is None" + ) + return + + guardrail_information = standard_logging_object.get( + "guardrail_information", None + ) + if guardrail_information is None: + verbose_logger.debug( + "Not logging guardrail information as span because guardrail_information is None" + ) + return + + span = trace.span( + name="guardrail", + input=guardrail_information.get("guardrail_request", None), + output=guardrail_information.get("guardrail_response", None), + metadata={ + "guardrail_name": guardrail_information.get("guardrail_name", None), + "guardrail_mode": guardrail_information.get("guardrail_mode", None), + "guardrail_masked_entity_count": guardrail_information.get( + "masked_entity_count", None + ), + }, + start_time=guardrail_information.get("start_time", None), # type: ignore + end_time=guardrail_information.get("end_time", None), # type: ignore + ) + + verbose_logger.debug(f"Logged guardrail information as span: {span}") + span.end() + def _add_prompt_to_generation_params( generation_params: dict, diff --git a/litellm/integrations/opentelemetry.py b/litellm/integrations/opentelemetry.py index f4fe40738ba..304fa827a7a 100644 --- a/litellm/integrations/opentelemetry.py +++ b/litellm/integrations/opentelemetry.py @@ -6,6 +6,7 @@ from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union, cast import litellm from litellm._logging import verbose_logger from litellm.integrations.custom_logger import CustomLogger +from litellm.litellm_core_utils.safe_json_dumps import safe_dumps from litellm.types.services import ServiceLoggerPayload from litellm.types.utils import ( ChatCompletionMessageToolCall, @@ -16,6 +17,7 @@ from litellm.types.utils import ( if TYPE_CHECKING: from opentelemetry.sdk.trace.export import SpanExporter as _SpanExporter + from opentelemetry.trace import Context as _Context from opentelemetry.trace import Span as _Span from litellm.proxy._types import ( @@ -24,6 +26,7 @@ if TYPE_CHECKING: from litellm.proxy.proxy_server import UserAPIKeyAuth as _UserAPIKeyAuth Span = Union[_Span, Any] + Context = Union[_Context, Any] SpanExporter = Union[_SpanExporter, Any] UserAPIKeyAuth = Union[_UserAPIKeyAuth, Any] ManagementEndpointLoggingPayload = Union[_ManagementEndpointLoggingPayload, Any] @@ -32,7 +35,7 @@ else: SpanExporter = Any UserAPIKeyAuth = Any ManagementEndpointLoggingPayload = Any - + Context = Any LITELLM_TRACER_NAME = os.getenv("OTEL_TRACER_NAME", "litellm") LITELLM_RESOURCE: Dict[Any, Any] = { @@ -63,14 +66,20 @@ class OpenTelemetryConfig: InMemorySpanExporter, ) - if os.getenv("OTEL_EXPORTER") == "in_memory": + exporter = os.getenv( + "OTEL_EXPORTER_OTLP_PROTOCOL", os.getenv("OTEL_EXPORTER", "console") + ) + endpoint = os.getenv("OTEL_EXPORTER_OTLP_ENDPOINT", os.getenv("OTEL_ENDPOINT")) + headers = os.getenv( + "OTEL_EXPORTER_OTLP_HEADERS", os.getenv("OTEL_HEADERS") + ) # example: OTEL_HEADERS=x-honeycomb-team=B85YgLm96***" + + if exporter == "in_memory": return cls(exporter=InMemorySpanExporter()) return cls( - exporter=os.getenv("OTEL_EXPORTER", "console"), - endpoint=os.getenv("OTEL_ENDPOINT"), - headers=os.getenv( - "OTEL_HEADERS" - ), # example: OTEL_HEADERS=x-honeycomb-team=B85YgLm96***" + exporter=exporter, + endpoint=endpoint, + headers=headers, # example: OTEL_HEADERS=x-honeycomb-team=B85YgLm96***" ) @@ -273,6 +282,7 @@ class OpenTelemetry(CustomLogger): request_data: dict, original_exception: Exception, user_api_key_dict: UserAPIKeyAuth, + traceback_str: Optional[str] = None, ): from opentelemetry import trace from opentelemetry.trace import Status, StatusCode @@ -338,9 +348,72 @@ class OpenTelemetry(CustomLogger): span.end(end_time=self._to_ns(end_time)) + # Create span for guardrail information + self._create_guardrail_span(kwargs=kwargs, context=_parent_context) + if parent_otel_span is not None: parent_otel_span.end(end_time=self._to_ns(datetime.now())) + def _create_guardrail_span( + self, kwargs: Optional[dict], context: Optional[Context] + ): + """ + Creates a span for Guardrail, if any guardrail information is present in standard_logging_object + """ + # Create span for guardrail information + kwargs = kwargs or {} + standard_logging_payload: Optional[StandardLoggingPayload] = kwargs.get( + "standard_logging_object" + ) + if standard_logging_payload is None: + return + + guardrail_information = standard_logging_payload.get("guardrail_information") + if guardrail_information is None: + return + + start_time_float = guardrail_information.get("start_time") + end_time_float = guardrail_information.get("end_time") + start_time_datetime = datetime.now() + if start_time_float is not None: + start_time_datetime = datetime.fromtimestamp(start_time_float) + end_time_datetime = datetime.now() + if end_time_float is not None: + end_time_datetime = datetime.fromtimestamp(end_time_float) + + guardrail_span = self.tracer.start_span( + name="guardrail", + start_time=self._to_ns(start_time_datetime), + context=context, + ) + + self.safe_set_attribute( + span=guardrail_span, + key="guardrail_name", + value=guardrail_information.get("guardrail_name"), + ) + + self.safe_set_attribute( + span=guardrail_span, + key="guardrail_mode", + value=guardrail_information.get("guardrail_mode"), + ) + + # Set masked_entity_count directly without conversion + masked_entity_count = guardrail_information.get("masked_entity_count") + if masked_entity_count is not None: + guardrail_span.set_attribute( + "masked_entity_count", safe_dumps(masked_entity_count) + ) + + self.safe_set_attribute( + span=guardrail_span, + key="guardrail_response", + value=guardrail_information.get("guardrail_response"), + ) + + guardrail_span.end(end_time=self._to_ns(end_time_datetime)) + def _add_dynamic_span_processor_if_needed(self, kwargs): """ Helper method to add a span processor with dynamic headers if needed. @@ -400,6 +473,9 @@ class OpenTelemetry(CustomLogger): self.set_attributes(span, kwargs, response_obj) span.end(end_time=self._to_ns(end_time)) + # Create span for guardrail information + self._create_guardrail_span(kwargs=kwargs, context=_parent_context) + if parent_otel_span is not None: parent_otel_span.end(end_time=self._to_ns(datetime.now())) @@ -417,7 +493,7 @@ class OpenTelemetry(CustomLogger): if not function: continue - prefix = f"{SpanAttributes.LLM_REQUEST_FUNCTIONS}.{i}" + prefix = f"{SpanAttributes.LLM_REQUEST_FUNCTIONS.value}.{i}" self.safe_set_attribute( span=span, key=f"{prefix}.name", @@ -473,7 +549,7 @@ class OpenTelemetry(CustomLogger): _value = _function.get(key) if _value: kv_pairs[ - f"{SpanAttributes.LLM_COMPLETIONS}.{idx}.function_call.{key}" + f"{SpanAttributes.LLM_COMPLETIONS.value}.{idx}.function_call.{key}" ] = _value return kv_pairs @@ -525,21 +601,21 @@ class OpenTelemetry(CustomLogger): if kwargs.get("model"): self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_REQUEST_MODEL, + key=SpanAttributes.LLM_REQUEST_MODEL.value, value=kwargs.get("model"), ) # The LLM request type self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_REQUEST_TYPE, + key=SpanAttributes.LLM_REQUEST_TYPE.value, value=standard_logging_payload["call_type"], ) # The Generative AI Provider: Azure, OpenAI, etc. self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_SYSTEM, + key=SpanAttributes.LLM_SYSTEM.value, value=litellm_params.get("custom_llm_provider", "Unknown"), ) @@ -547,7 +623,7 @@ class OpenTelemetry(CustomLogger): if optional_params.get("max_tokens"): self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_REQUEST_MAX_TOKENS, + key=SpanAttributes.LLM_REQUEST_MAX_TOKENS.value, value=optional_params.get("max_tokens"), ) @@ -555,7 +631,7 @@ class OpenTelemetry(CustomLogger): if optional_params.get("temperature"): self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_REQUEST_TEMPERATURE, + key=SpanAttributes.LLM_REQUEST_TEMPERATURE.value, value=optional_params.get("temperature"), ) @@ -563,20 +639,20 @@ class OpenTelemetry(CustomLogger): if optional_params.get("top_p"): self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_REQUEST_TOP_P, + key=SpanAttributes.LLM_REQUEST_TOP_P.value, value=optional_params.get("top_p"), ) self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_IS_STREAMING, + key=SpanAttributes.LLM_IS_STREAMING.value, value=str(optional_params.get("stream", False)), ) if optional_params.get("user"): self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_USER, + key=SpanAttributes.LLM_USER.value, value=optional_params.get("user"), ) @@ -590,7 +666,7 @@ class OpenTelemetry(CustomLogger): if response_obj and response_obj.get("model"): self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_RESPONSE_MODEL, + key=SpanAttributes.LLM_RESPONSE_MODEL.value, value=response_obj.get("model"), ) @@ -598,21 +674,21 @@ class OpenTelemetry(CustomLogger): if usage: self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_USAGE_TOTAL_TOKENS, + key=SpanAttributes.LLM_USAGE_TOTAL_TOKENS.value, value=usage.get("total_tokens"), ) # The number of tokens used in the LLM response (completion). self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_USAGE_COMPLETION_TOKENS, + key=SpanAttributes.LLM_USAGE_COMPLETION_TOKENS.value, value=usage.get("completion_tokens"), ) # The number of tokens used in the LLM prompt. self.safe_set_attribute( span=span, - key=SpanAttributes.LLM_USAGE_PROMPT_TOKENS, + key=SpanAttributes.LLM_USAGE_PROMPT_TOKENS.value, value=usage.get("prompt_tokens"), ) @@ -634,7 +710,7 @@ class OpenTelemetry(CustomLogger): if prompt.get("role"): self.safe_set_attribute( span=span, - key=f"{SpanAttributes.LLM_PROMPTS}.{idx}.role", + key=f"{SpanAttributes.LLM_PROMPTS.value}.{idx}.role", value=prompt.get("role"), ) @@ -643,7 +719,7 @@ class OpenTelemetry(CustomLogger): prompt["content"] = str(prompt.get("content")) self.safe_set_attribute( span=span, - key=f"{SpanAttributes.LLM_PROMPTS}.{idx}.content", + key=f"{SpanAttributes.LLM_PROMPTS.value}.{idx}.content", value=prompt.get("content"), ) ############################################# @@ -655,14 +731,14 @@ class OpenTelemetry(CustomLogger): if choice.get("finish_reason"): self.safe_set_attribute( span=span, - key=f"{SpanAttributes.LLM_COMPLETIONS}.{idx}.finish_reason", + key=f"{SpanAttributes.LLM_COMPLETIONS.value}.{idx}.finish_reason", value=choice.get("finish_reason"), ) if choice.get("message"): if choice.get("message").get("role"): self.safe_set_attribute( span=span, - key=f"{SpanAttributes.LLM_COMPLETIONS}.{idx}.role", + key=f"{SpanAttributes.LLM_COMPLETIONS.value}.{idx}.role", value=choice.get("message").get("role"), ) if choice.get("message").get("content"): @@ -674,7 +750,7 @@ class OpenTelemetry(CustomLogger): ) self.safe_set_attribute( span=span, - key=f"{SpanAttributes.LLM_COMPLETIONS}.{idx}.content", + key=f"{SpanAttributes.LLM_COMPLETIONS.value}.{idx}.content", value=choice.get("message").get("content"), ) @@ -854,7 +930,11 @@ class OpenTelemetry(CustomLogger): self.OTEL_EXPORTER, ) return BatchSpanProcessor(ConsoleSpanExporter()) - elif self.OTEL_EXPORTER == "otlp_http": + elif ( + self.OTEL_EXPORTER == "otlp_http" + or self.OTEL_EXPORTER == "http/protobuf" + or self.OTEL_EXPORTER == "http/json" + ): verbose_logger.debug( "OpenTelemetry: intiializing http exporter. Value of OTEL_EXPORTER: %s", self.OTEL_EXPORTER, @@ -864,7 +944,7 @@ class OpenTelemetry(CustomLogger): endpoint=self.OTEL_ENDPOINT, headers=_split_otel_headers ), ) - elif self.OTEL_EXPORTER == "otlp_grpc": + elif self.OTEL_EXPORTER == "otlp_grpc" or self.OTEL_EXPORTER == "grpc": verbose_logger.debug( "OpenTelemetry: intiializing grpc exporter. Value of OTEL_EXPORTER: %s", self.OTEL_EXPORTER, diff --git a/litellm/integrations/prometheus.py b/litellm/integrations/prometheus.py index 03bf1cd29e8..aa543ee4891 100644 --- a/litellm/integrations/prometheus.py +++ b/litellm/integrations/prometheus.py @@ -802,6 +802,7 @@ class PrometheusLogger(CustomLogger): request_data: dict, original_exception: Exception, user_api_key_dict: UserAPIKeyAuth, + traceback_str: Optional[str] = None, ): """ Track client side failures diff --git a/litellm/integrations/vector_stores/bedrock_vector_store.py b/litellm/integrations/vector_stores/bedrock_vector_store.py index e52a90bcdad..e0af1a66364 100644 --- a/litellm/integrations/vector_stores/bedrock_vector_store.py +++ b/litellm/integrations/vector_stores/bedrock_vector_store.py @@ -31,8 +31,8 @@ from litellm.types.llms.openai import AllMessageValues, ChatCompletionUserMessag from litellm.types.utils import StandardLoggingVectorStoreRequest from litellm.types.vector_stores import ( VectorStoreResultContent, + VectorStoreSearchResponse, VectorStoreSearchResult, - VectorStorSearchResponse, ) from litellm.utils import load_credentials_from_list @@ -110,7 +110,7 @@ class BedrockVectorStore(BaseVectorStore, BaseAWSLLM): ################################################################################################# ########## LOGGING for Standard Logging Payload, Langfuse, s3, LiteLLM DB etc. ################## ################################################################################################# - vector_store_search_response: VectorStorSearchResponse = ( + vector_store_search_response: VectorStoreSearchResponse = ( self.transform_bedrock_kb_response_to_vector_store_search_response( bedrock_kb_response=bedrock_kb_response, query=query ) @@ -136,15 +136,15 @@ class BedrockVectorStore(BaseVectorStore, BaseAWSLLM): self, bedrock_kb_response: BedrockKBResponse, query: str, - ) -> VectorStorSearchResponse: + ) -> VectorStoreSearchResponse: """ - Transform a BedrockKBResponse to a VectorStorSearchResponse + Transform a BedrockKBResponse to a VectorStoreSearchResponse """ retrieval_results: Optional[List[BedrockKBRetrievalResult]] = ( bedrock_kb_response.get("retrievalResults", None) ) - vector_store_search_response: VectorStorSearchResponse = ( - VectorStorSearchResponse(search_query=query, data=[]) + vector_store_search_response: VectorStoreSearchResponse = ( + VectorStoreSearchResponse(search_query=query, data=[]) ) if retrieval_results is None: return vector_store_search_response diff --git a/litellm/litellm_core_utils/core_helpers.py b/litellm/litellm_core_utils/core_helpers.py index 275c53ad308..28a0097c30d 100644 --- a/litellm/litellm_core_utils/core_helpers.py +++ b/litellm/litellm_core_utils/core_helpers.py @@ -70,6 +70,22 @@ def remove_index_from_tool_calls( return +def add_missing_spend_metadata_to_litellm_metadata( + litellm_metadata: dict, metadata: dict +) -> dict: + """ + Helper to get litellm metadata for spend tracking + + PATCH for issue where both `litellm_metadata` and `metadata` are present in the kwargs + and user_api_key values are in 'metadata'. + """ + potential_spend_tracking_metadata_substring = "user_api_key" + for key, value in metadata.items(): + if potential_spend_tracking_metadata_substring in key: + litellm_metadata[key] = value + return litellm_metadata + + def get_litellm_metadata_from_kwargs(kwargs: dict): """ Helper to get litellm metadata from all litellm request kwargs @@ -80,6 +96,10 @@ def get_litellm_metadata_from_kwargs(kwargs: dict): if litellm_params: metadata = litellm_params.get("metadata", {}) litellm_metadata = litellm_params.get("litellm_metadata", {}) + if litellm_metadata and metadata: + litellm_metadata = add_missing_spend_metadata_to_litellm_metadata( + litellm_metadata, metadata + ) if litellm_metadata: return litellm_metadata elif metadata: diff --git a/litellm/litellm_core_utils/duration_parser.py b/litellm/litellm_core_utils/duration_parser.py index 41d8218ff6b..08f1d4c82d0 100644 --- a/litellm/litellm_core_utils/duration_parser.py +++ b/litellm/litellm_core_utils/duration_parser.py @@ -138,6 +138,8 @@ def get_next_standardized_reset_time( return _handle_minute_reset(current_time, base_midnight, value) elif unit == "s": return _handle_second_reset(current_time, base_midnight, value) + elif unit == "mo": + return _handle_month_reset(current_time, base_midnight, value) else: # Unrecognized unit, default to next midnight return base_midnight + timedelta(days=1) @@ -343,3 +345,40 @@ def _handle_second_reset( return current_time.replace( hour=next_hour, minute=next_minute, second=next_second, microsecond=0 ) + + +def _handle_month_reset( + current_time: datetime, base_midnight: datetime, value: int +) -> datetime: + """ + Handle monthly reset times. For monthly resets, we always reset at the start of the next month. + + Args: + current_time: Current datetime + base_midnight: Midnight of current day + value: Number of months (currently only supports 1 month resets) + + Returns: + datetime: First day of next month at midnight + """ + if value != 1: + raise ValueError("Monthly resets currently only support 1 month intervals") + + # Get the first day of next month + if current_time.month == 12: + next_month = 1 + next_year = current_time.year + 1 + else: + next_month = current_time.month + 1 + next_year = current_time.year + + return datetime( + year=next_year, + month=next_month, + day=1, + hour=0, + minute=0, + second=0, + microsecond=0, + tzinfo=current_time.tzinfo, + ) diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index e567c7daad8..e96c73e4272 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -274,7 +274,15 @@ def exception_type( # type: ignore # noqa: PLR0915 + "Exception" ) - if ( + if "429" in error_str: + exception_mapping_worked = True + raise RateLimitError( + message=f"RateLimitError: {exception_provider} - {message}", + model=model, + llm_provider=custom_llm_provider, + response=getattr(original_exception, "response", None), + ) + elif ( "This model's maximum context length is" in error_str or "string too long. Expected a string with maximum length" in error_str diff --git a/litellm/litellm_core_utils/get_litellm_params.py b/litellm/litellm_core_utils/get_litellm_params.py index 7e8f60bd4fd..19c8ec8d808 100644 --- a/litellm/litellm_core_utils/get_litellm_params.py +++ b/litellm/litellm_core_utils/get_litellm_params.py @@ -59,6 +59,7 @@ def get_litellm_params( async_call: Optional[bool] = None, ssl_verify: Optional[bool] = None, merge_reasoning_content_in_choices: Optional[bool] = None, + use_litellm_proxy: Optional[bool] = None, api_version: Optional[str] = None, max_retries: Optional[int] = None, **kwargs, @@ -115,5 +116,6 @@ def get_litellm_params( "bucket_name": kwargs.get("bucket_name"), "vertex_credentials": kwargs.get("vertex_credentials"), "vertex_project": kwargs.get("vertex_project"), + "use_litellm_proxy": use_litellm_proxy, } return litellm_params diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 331087e02ef..f792d249b3f 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -102,8 +102,15 @@ def get_llm_provider( # noqa: PLR0915 Return model, custom_llm_provider, dynamic_api_key, api_base """ try: + if litellm.LiteLLMProxyChatConfig._should_use_litellm_proxy_by_default( + litellm_params=litellm_params + ): + return litellm.LiteLLMProxyChatConfig.litellm_proxy_get_custom_llm_provider_info( + model=model, api_base=api_base, api_key=api_key + ) + ## IF LITELLM PARAMS GIVEN ## - if litellm_params is not None: + if litellm_params: assert ( custom_llm_provider is None and api_base is None and api_key is None ), "Either pass in litellm_params or the custom_llm_provider/api_base/api_key. Otherwise, these values will be overriden." @@ -215,6 +222,15 @@ def get_llm_provider( # noqa: PLR0915 elif endpoint == "api.galadriel.com/v1": custom_llm_provider = "galadriel" dynamic_api_key = get_secret_str("GALADRIEL_API_KEY") + elif endpoint == "https://api.llama.com/compat/v1": + custom_llm_provider = "meta_llama" + dynamic_api_key = api_key or get_secret_str("LLAMA_API_KEY") + elif endpoint == "https://api.featherless.ai/v1": + custom_llm_provider = "featherless_ai" + dynamic_api_key = get_secret_str("FEATHERLESS_AI_API_KEY") + elif endpoint == litellm.NscaleConfig.API_BASE_URL: + custom_llm_provider = "nscale" + dynamic_api_key = litellm.NscaleConfig.get_api_key() if api_base is not None and not isinstance(api_base, str): raise Exception( @@ -444,6 +460,13 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 or "https://api.sambanova.ai/v1" ) # type: ignore dynamic_api_key = api_key or get_secret_str("SAMBANOVA_API_KEY") + elif custom_llm_provider == "meta_llama": + api_base = ( + api_base + or get_secret("LLAMA_API_BASE") + or "https://api.llama.com/compat/v1" + ) # type: ignore + dynamic_api_key = api_key or get_secret_str("LLAMA_API_KEY") elif (custom_llm_provider == "ai21_chat") or ( custom_llm_provider == "ai21" and model in litellm.ai21_chat_models ): @@ -478,10 +501,12 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 ) elif custom_llm_provider == "llamafile": # llamafile is OpenAI compatible. - (api_base, dynamic_api_key) = litellm.LlamafileChatConfig()._get_openai_compatible_provider_info( - api_base, - api_key - ) + ( + api_base, + dynamic_api_key, + ) = litellm.LlamafileChatConfig()._get_openai_compatible_provider_info( + api_base, api_key + ) elif custom_llm_provider == "lm_studio": # lm_studio is openai compatible, we just need to set this to custom_openai ( @@ -523,8 +548,12 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 ) dynamic_api_key = api_key or get_secret_str("GITHUB_API_KEY") elif custom_llm_provider == "litellm_proxy": - api_base = api_base or get_secret_str("LITELLM_PROXY_API_BASE") - dynamic_api_key = api_key or get_secret_str("LITELLM_PROXY_API_KEY") + ( + api_base, + dynamic_api_key, + ) = litellm.LiteLLMProxyChatConfig()._get_openai_compatible_provider_info( + api_base=api_base, api_key=api_key + ) elif custom_llm_provider == "mistral": ( @@ -578,6 +607,13 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 or "https://api.galadriel.com/v1" ) # type: ignore dynamic_api_key = api_key or get_secret_str("GALADRIEL_API_KEY") + elif custom_llm_provider == "novita": + api_base = ( + api_base + or get_secret("NOVITA_API_BASE") + or "https://api.novita.ai/v3/openai" + ) # type: ignore + dynamic_api_key = api_key or get_secret_str("NOVITA_API_KEY") elif custom_llm_provider == "snowflake": api_base = ( api_base @@ -585,6 +621,20 @@ def _get_openai_compatible_provider_info( # noqa: PLR0915 or f"https://{get_secret('SNOWFLAKE_ACCOUNT_ID')}.snowflakecomputing.com/api/v2/cortex/inference:complete" ) # type: ignore dynamic_api_key = api_key or get_secret_str("SNOWFLAKE_JWT") + elif custom_llm_provider == "featherless_ai": + ( + api_base, + dynamic_api_key, + ) = litellm.FeatherlessAIConfig()._get_openai_compatible_provider_info( + api_base, api_key + ) + elif custom_llm_provider == "nscale": + ( + api_base, + dynamic_api_key, + ) = litellm.NscaleConfig()._get_openai_compatible_provider_info( + api_base=api_base, api_key=api_key + ) if api_base is not None and not isinstance(api_base, str): raise Exception("api base needs to be a string. api_base={}".format(api_base)) diff --git a/litellm/litellm_core_utils/get_supported_openai_params.py b/litellm/litellm_core_utils/get_supported_openai_params.py index c0f638ddc24..043444cfc6a 100644 --- a/litellm/litellm_core_utils/get_supported_openai_params.py +++ b/litellm/litellm_core_utils/get_supported_openai_params.py @@ -46,6 +46,12 @@ def get_supported_openai_params( # noqa: PLR0915 if custom_llm_provider == "bedrock": return litellm.AmazonConverseConfig().get_supported_openai_params(model=model) + elif custom_llm_provider == "meta_llama": + provider_config = litellm.ProviderConfigManager.get_provider_chat_config( + model=model, provider=LlmProviders.LLAMA + ) + if provider_config: + return provider_config.get_supported_openai_params(model=model) elif custom_llm_provider == "ollama": return litellm.OllamaConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "ollama_chat": @@ -149,6 +155,8 @@ def get_supported_openai_params( # noqa: PLR0915 return litellm.GoogleAIStudioGeminiConfig().get_supported_openai_params( model=model ) + elif custom_llm_provider == "novita": + return litellm.NovitaConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "vertex_ai" or custom_llm_provider == "vertex_ai_beta": if request_type == "chat_completion": if model.startswith("mistral"): @@ -196,6 +204,8 @@ def get_supported_openai_params( # noqa: PLR0915 return litellm.DeepInfraConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "perplexity": return litellm.PerplexityChatConfig().get_supported_openai_params(model=model) + elif custom_llm_provider == "nscale": + return litellm.NscaleConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "anyscale": return [ "temperature", @@ -222,7 +232,9 @@ def get_supported_openai_params( # noqa: PLR0915 elif custom_llm_provider == "voyage": return litellm.VoyageEmbeddingConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "infinity": - return litellm.InfinityEmbeddingConfig().get_supported_openai_params(model=model) + return litellm.InfinityEmbeddingConfig().get_supported_openai_params( + model=model + ) elif custom_llm_provider == "triton": if request_type == "embeddings": return litellm.TritonEmbeddingConfig().get_supported_openai_params( diff --git a/litellm/litellm_core_utils/json_validation_rule.py b/litellm/litellm_core_utils/json_validation_rule.py index 0f37e673729..53e1479783b 100644 --- a/litellm/litellm_core_utils/json_validation_rule.py +++ b/litellm/litellm_core_utils/json_validation_rule.py @@ -17,7 +17,7 @@ def validate_schema(schema: dict, response: str): response_dict = json.loads(response) except json.JSONDecodeError: raise JSONSchemaValidationError( - model="", llm_provider="", raw_response=response, schema=response + model="", llm_provider="", raw_response=response, schema=json.dumps(schema) ) try: diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index c6957a4e5d6..88ce34245a6 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -13,7 +13,18 @@ import traceback import uuid from datetime import datetime as dt_object from functools import lru_cache -from typing import Any, Callable, Dict, List, Literal, Optional, Tuple, Union, cast +from typing import ( + Any, + Callable, + Dict, + List, + Literal, + Optional, + Tuple, + Type, + Union, + cast, +) from pydantic import BaseModel @@ -42,7 +53,6 @@ from litellm.integrations.arize.arize import ArizeLogger from litellm.integrations.custom_guardrail import CustomGuardrail from litellm.integrations.custom_logger import CustomLogger from litellm.integrations.mlflow import MlflowLogger -from litellm.integrations.pagerduty.pagerduty import PagerDutyAlerting from litellm.integrations.vector_stores.bedrock_vector_store import BedrockVectorStore from litellm.litellm_core_utils.get_litellm_params import get_litellm_params from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import ( @@ -134,14 +144,34 @@ from .initialize_dynamic_callback_params import ( from .specialty_caches.dynamic_logging_cache import DynamicLoggingCache try: - from ..proxy.enterprise.enterprise_callbacks.generic_api_callback import ( + from litellm_enterprise.enterprise_callbacks.generic_api_callback import ( GenericAPILogger, ) + from litellm_enterprise.enterprise_callbacks.pagerduty.pagerduty import ( + PagerDutyAlerting, + ) + from litellm_enterprise.enterprise_callbacks.send_emails.resend_email import ( + ResendEmailLogger, + ) + from litellm_enterprise.enterprise_callbacks.send_emails.smtp_email import ( + SMTPEmailLogger, + ) + from litellm_enterprise.litellm_core_utils.litellm_logging import ( + StandardLoggingPayloadSetup as EnterpriseStandardLoggingPayloadSetup, + ) + + EnterpriseStandardLoggingPayloadSetupVAR: Optional[ + Type[EnterpriseStandardLoggingPayloadSetup] + ] = EnterpriseStandardLoggingPayloadSetup except Exception as e: verbose_logger.debug( f"[Non-Blocking] Unable to import GenericAPILogger - LiteLLM Enterprise Feature - {str(e)}" ) - + GenericAPILogger = CustomLogger # type: ignore + ResendEmailLogger = CustomLogger # type: ignore + SMTPEmailLogger = CustomLogger # type: ignore + PagerDutyAlerting = CustomLogger # type: ignore + EnterpriseStandardLoggingPayloadSetupVAR = None _in_memory_loggers: List[Any] = [] ### GLOBAL VARIABLES ### @@ -165,7 +195,6 @@ dataDogLogger = None prometheusLogger = None dynamoLogger = None s3Logger = None -genericAPILogger = None greenscaleLogger = None lunaryLogger = None supabaseClient = None @@ -255,9 +284,9 @@ class Logging(LiteLLMLoggingBaseClass): self.litellm_trace_id: str = litellm_trace_id or str(uuid.uuid4()) self.function_id = function_id self.streaming_chunks: List[Any] = [] # for generating complete stream response - self.sync_streaming_chunks: List[Any] = ( - [] - ) # for generating complete stream response + self.sync_streaming_chunks: List[ + Any + ] = [] # for generating complete stream response self.log_raw_request_response = log_raw_request_response # Initialize dynamic callbacks @@ -608,9 +637,9 @@ class Logging(LiteLLMLoggingBaseClass): if anthropic_cache_control_logger := AnthropicCacheControlHook.get_custom_logger_for_anthropic_cache_control_hook( non_default_params ): - self.model_call_details["prompt_integration"] = ( - anthropic_cache_control_logger.__class__.__name__ - ) + self.model_call_details[ + "prompt_integration" + ] = anthropic_cache_control_logger.__class__.__name__ return anthropic_cache_control_logger ######################################################### @@ -628,9 +657,9 @@ class Logging(LiteLLMLoggingBaseClass): ), ) ) - self.model_call_details["prompt_integration"] = ( - vector_store_custom_logger.__class__.__name__ - ) + self.model_call_details[ + "prompt_integration" + ] = vector_store_custom_logger.__class__.__name__ return vector_store_custom_logger return None @@ -682,9 +711,9 @@ class Logging(LiteLLMLoggingBaseClass): model ): # if model name was changes pre-call, overwrite the initial model call name with the new one self.model_call_details["model"] = model - self.model_call_details["litellm_params"]["api_base"] = ( - self._get_masked_api_base(additional_args.get("api_base", "")) - ) + self.model_call_details["litellm_params"][ + "api_base" + ] = self._get_masked_api_base(additional_args.get("api_base", "")) def pre_call(self, input, api_key, model=None, additional_args={}): # noqa: PLR0915 # Log the exact input to the LLM API @@ -713,10 +742,10 @@ class Logging(LiteLLMLoggingBaseClass): try: # [Non-blocking Extra Debug Information in metadata] if turn_off_message_logging is True: - _metadata["raw_request"] = ( - "redacted by litellm. \ + _metadata[ + "raw_request" + ] = "redacted by litellm. \ 'litellm.turn_off_message_logging=True'" - ) else: curl_command = self._get_request_curl_command( api_base=additional_args.get("api_base", ""), @@ -727,32 +756,32 @@ class Logging(LiteLLMLoggingBaseClass): _metadata["raw_request"] = str(curl_command) # split up, so it's easier to parse in the UI - self.model_call_details["raw_request_typed_dict"] = ( - RawRequestTypedDict( - raw_request_api_base=str( - additional_args.get("api_base") or "" - ), - raw_request_body=self._get_raw_request_body( - additional_args.get("complete_input_dict", {}) - ), - raw_request_headers=self._get_masked_headers( - additional_args.get("headers", {}) or {}, - ignore_sensitive_headers=True, - ), - error=None, - ) + self.model_call_details[ + "raw_request_typed_dict" + ] = RawRequestTypedDict( + raw_request_api_base=str( + additional_args.get("api_base") or "" + ), + raw_request_body=self._get_raw_request_body( + additional_args.get("complete_input_dict", {}) + ), + raw_request_headers=self._get_masked_headers( + additional_args.get("headers", {}) or {}, + ignore_sensitive_headers=True, + ), + error=None, ) except Exception as e: - self.model_call_details["raw_request_typed_dict"] = ( - RawRequestTypedDict( - error=str(e), - ) + self.model_call_details[ + "raw_request_typed_dict" + ] = RawRequestTypedDict( + error=str(e), ) - _metadata["raw_request"] = ( - "Unable to Log \ + _metadata[ + "raw_request" + ] = "Unable to Log \ raw request: {}".format( - str(e) - ) + str(e) ) if self.logger_fn and callable(self.logger_fn): try: @@ -1083,9 +1112,9 @@ class Logging(LiteLLMLoggingBaseClass): verbose_logger.debug( f"response_cost_failure_debug_information: {debug_info}" ) - self.model_call_details["response_cost_failure_debug_information"] = ( - debug_info - ) + self.model_call_details[ + "response_cost_failure_debug_information" + ] = debug_info return None try: @@ -1110,9 +1139,9 @@ class Logging(LiteLLMLoggingBaseClass): verbose_logger.debug( f"response_cost_failure_debug_information: {debug_info}" ) - self.model_call_details["response_cost_failure_debug_information"] = ( - debug_info - ) + self.model_call_details[ + "response_cost_failure_debug_information" + ] = debug_info return None @@ -1172,9 +1201,9 @@ class Logging(LiteLLMLoggingBaseClass): end_time = datetime.datetime.now() if self.completion_start_time is None: self.completion_start_time = end_time - self.model_call_details["completion_start_time"] = ( - self.completion_start_time - ) + self.model_call_details[ + "completion_start_time" + ] = self.completion_start_time self.model_call_details["log_event_type"] = "successful_api_call" self.model_call_details["end_time"] = end_time self.model_call_details["cache_hit"] = cache_hit @@ -1254,39 +1283,39 @@ class Logging(LiteLLMLoggingBaseClass): "response_cost" ] else: - self.model_call_details["response_cost"] = ( - self._response_cost_calculator(result=logging_result) - ) + self.model_call_details[ + "response_cost" + ] = self._response_cost_calculator(result=logging_result) ## STANDARDIZED LOGGING PAYLOAD - self.model_call_details["standard_logging_object"] = ( - get_standard_logging_object_payload( - kwargs=self.model_call_details, - init_response_obj=logging_result, - start_time=start_time, - end_time=end_time, - logging_obj=self, - status="success", - standard_built_in_tools_params=self.standard_built_in_tools_params, - ) + self.model_call_details[ + "standard_logging_object" + ] = get_standard_logging_object_payload( + kwargs=self.model_call_details, + init_response_obj=logging_result, + start_time=start_time, + end_time=end_time, + logging_obj=self, + status="success", + standard_built_in_tools_params=self.standard_built_in_tools_params, ) elif isinstance(result, dict) or isinstance(result, list): ## STANDARDIZED LOGGING PAYLOAD - self.model_call_details["standard_logging_object"] = ( - get_standard_logging_object_payload( - kwargs=self.model_call_details, - init_response_obj=result, - start_time=start_time, - end_time=end_time, - logging_obj=self, - status="success", - standard_built_in_tools_params=self.standard_built_in_tools_params, - ) + self.model_call_details[ + "standard_logging_object" + ] = get_standard_logging_object_payload( + kwargs=self.model_call_details, + init_response_obj=result, + start_time=start_time, + end_time=end_time, + logging_obj=self, + status="success", + standard_built_in_tools_params=self.standard_built_in_tools_params, ) elif standard_logging_object is not None: - self.model_call_details["standard_logging_object"] = ( - standard_logging_object - ) + self.model_call_details[ + "standard_logging_object" + ] = standard_logging_object else: # streaming chunks + image gen. self.model_call_details["response_cost"] = None @@ -1342,23 +1371,23 @@ class Logging(LiteLLMLoggingBaseClass): verbose_logger.debug( "Logging Details LiteLLM-Success Call streaming complete" ) - self.model_call_details["complete_streaming_response"] = ( - complete_streaming_response - ) - self.model_call_details["response_cost"] = ( - self._response_cost_calculator(result=complete_streaming_response) - ) + self.model_call_details[ + "complete_streaming_response" + ] = complete_streaming_response + self.model_call_details[ + "response_cost" + ] = self._response_cost_calculator(result=complete_streaming_response) ## STANDARDIZED LOGGING PAYLOAD - self.model_call_details["standard_logging_object"] = ( - get_standard_logging_object_payload( - kwargs=self.model_call_details, - init_response_obj=complete_streaming_response, - start_time=start_time, - end_time=end_time, - logging_obj=self, - status="success", - standard_built_in_tools_params=self.standard_built_in_tools_params, - ) + self.model_call_details[ + "standard_logging_object" + ] = get_standard_logging_object_payload( + kwargs=self.model_call_details, + init_response_obj=complete_streaming_response, + start_time=start_time, + end_time=end_time, + logging_obj=self, + status="success", + standard_built_in_tools_params=self.standard_built_in_tools_params, ) callbacks = self.get_combined_callback_list( dynamic_success_callbacks=self.dynamic_success_callbacks, @@ -1564,35 +1593,6 @@ class Logging(LiteLLMLoggingBaseClass): service_name="langfuse", trace_id=_trace_id, ) - if callback == "generic": - global genericAPILogger - verbose_logger.debug("reaches langfuse for success logging!") - kwargs = {} - for k, v in self.model_call_details.items(): - if ( - k != "original_response" - ): # copy.deepcopy raises errors as this could be a coroutine - kwargs[k] = v - # this only logs streaming once, complete_streaming_response exists i.e when stream ends - if self.stream: - verbose_logger.debug( - f"is complete_streaming_response in kwargs: {kwargs.get('complete_streaming_response', None)}" - ) - if complete_streaming_response is None: - continue - else: - print_verbose("reaches langfuse for streaming logging!") - result = kwargs["complete_streaming_response"] - if genericAPILogger is None: - genericAPILogger = GenericAPILogger() # type: ignore - genericAPILogger.log_event( - kwargs=kwargs, - response_obj=result, - start_time=start_time, - end_time=end_time, - user_id=kwargs.get("user", None), - print_verbose=print_verbose, - ) if callback == "greenscale" and greenscaleLogger is not None: kwargs = {} for k, v in self.model_call_details.items(): @@ -1707,10 +1707,10 @@ class Logging(LiteLLMLoggingBaseClass): ) else: if self.stream and complete_streaming_response: - self.model_call_details["complete_response"] = ( - self.model_call_details.get( - "complete_streaming_response", {} - ) + self.model_call_details[ + "complete_response" + ] = self.model_call_details.get( + "complete_streaming_response", {} ) result = self.model_call_details["complete_response"] openMeterLogger.log_success_event( @@ -1750,10 +1750,10 @@ class Logging(LiteLLMLoggingBaseClass): ) else: if self.stream and complete_streaming_response: - self.model_call_details["complete_response"] = ( - self.model_call_details.get( - "complete_streaming_response", {} - ) + self.model_call_details[ + "complete_response" + ] = self.model_call_details.get( + "complete_streaming_response", {} ) result = self.model_call_details["complete_response"] @@ -1860,9 +1860,9 @@ class Logging(LiteLLMLoggingBaseClass): if complete_streaming_response is not None: print_verbose("Async success callbacks: Got a complete streaming response") - self.model_call_details["async_complete_streaming_response"] = ( - complete_streaming_response - ) + self.model_call_details[ + "async_complete_streaming_response" + ] = complete_streaming_response try: if self.model_call_details.get("cache_hit", False) is True: self.model_call_details["response_cost"] = 0.0 @@ -1872,10 +1872,10 @@ class Logging(LiteLLMLoggingBaseClass): model_call_details=self.model_call_details ) # base_model defaults to None if not set on model_info - self.model_call_details["response_cost"] = ( - self._response_cost_calculator( - result=complete_streaming_response - ) + self.model_call_details[ + "response_cost" + ] = self._response_cost_calculator( + result=complete_streaming_response ) verbose_logger.debug( @@ -1888,16 +1888,16 @@ class Logging(LiteLLMLoggingBaseClass): self.model_call_details["response_cost"] = None ## STANDARDIZED LOGGING PAYLOAD - self.model_call_details["standard_logging_object"] = ( - get_standard_logging_object_payload( - kwargs=self.model_call_details, - init_response_obj=complete_streaming_response, - start_time=start_time, - end_time=end_time, - logging_obj=self, - status="success", - standard_built_in_tools_params=self.standard_built_in_tools_params, - ) + self.model_call_details[ + "standard_logging_object" + ] = get_standard_logging_object_payload( + kwargs=self.model_call_details, + init_response_obj=complete_streaming_response, + start_time=start_time, + end_time=end_time, + logging_obj=self, + status="success", + standard_built_in_tools_params=self.standard_built_in_tools_params, ) callbacks = self.get_combined_callback_list( dynamic_success_callbacks=self.dynamic_async_success_callbacks, @@ -2103,18 +2103,18 @@ class Logging(LiteLLMLoggingBaseClass): ## STANDARDIZED LOGGING PAYLOAD - self.model_call_details["standard_logging_object"] = ( - get_standard_logging_object_payload( - kwargs=self.model_call_details, - init_response_obj={}, - start_time=start_time, - end_time=end_time, - logging_obj=self, - status="failure", - error_str=str(exception), - original_exception=exception, - standard_built_in_tools_params=self.standard_built_in_tools_params, - ) + self.model_call_details[ + "standard_logging_object" + ] = get_standard_logging_object_payload( + kwargs=self.model_call_details, + init_response_obj={}, + start_time=start_time, + end_time=end_time, + logging_obj=self, + status="failure", + error_str=str(exception), + original_exception=exception, + standard_built_in_tools_params=self.standard_built_in_tools_params, ) return start_time, end_time @@ -2888,9 +2888,9 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915 endpoint=arize_config.endpoint, ) - os.environ["OTEL_EXPORTER_OTLP_TRACES_HEADERS"] = ( - f"space_key={arize_config.space_key},api_key={arize_config.api_key}" - ) + os.environ[ + "OTEL_EXPORTER_OTLP_TRACES_HEADERS" + ] = f"space_key={arize_config.space_key},api_key={arize_config.api_key}" for callback in _in_memory_loggers: if ( isinstance(callback, ArizeLogger) @@ -2914,9 +2914,9 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915 # auth can be disabled on local deployments of arize phoenix if arize_phoenix_config.otlp_auth_headers is not None: - os.environ["OTEL_EXPORTER_OTLP_TRACES_HEADERS"] = ( - arize_phoenix_config.otlp_auth_headers - ) + os.environ[ + "OTEL_EXPORTER_OTLP_TRACES_HEADERS" + ] = arize_phoenix_config.otlp_auth_headers for callback in _in_memory_loggers: if ( @@ -3007,9 +3007,9 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915 exporter="otlp_http", endpoint="https://langtrace.ai/api/trace", ) - os.environ["OTEL_EXPORTER_OTLP_TRACES_HEADERS"] = ( - f"api_key={os.getenv('LANGTRACE_API_KEY')}" - ) + os.environ[ + "OTEL_EXPORTER_OTLP_TRACES_HEADERS" + ] = f"api_key={os.getenv('LANGTRACE_API_KEY')}" for callback in _in_memory_loggers: if ( isinstance(callback, OpenTelemetry) @@ -3064,6 +3064,27 @@ def _init_custom_logger_compatible_class( # noqa: PLR0915 _gcs_pubsub_logger = GcsPubSubLogger() _in_memory_loggers.append(_gcs_pubsub_logger) return _gcs_pubsub_logger # type: ignore + elif logging_integration == "generic_api": + for callback in _in_memory_loggers: + if isinstance(callback, GenericAPILogger): + return callback + generic_api_logger = GenericAPILogger() + _in_memory_loggers.append(generic_api_logger) + return generic_api_logger # type: ignore + elif logging_integration == "resend_email": + for callback in _in_memory_loggers: + if isinstance(callback, ResendEmailLogger): + return callback + resend_email_logger = ResendEmailLogger() + _in_memory_loggers.append(resend_email_logger) + return resend_email_logger # type: ignore + elif logging_integration == "smtp_email": + for callback in _in_memory_loggers: + if isinstance(callback, SMTPEmailLogger): + return callback + smtp_email_logger = SMTPEmailLogger() + _in_memory_loggers.append(smtp_email_logger) + return smtp_email_logger # type: ignore elif logging_integration == "humanloop": for callback in _in_memory_loggers: if isinstance(callback, HumanloopLogger): @@ -3207,8 +3228,20 @@ def get_custom_logger_compatible_class( # noqa: PLR0915 for callback in _in_memory_loggers: if isinstance(callback, GcsPubSubLogger): return callback - + elif logging_integration == "generic_api": + for callback in _in_memory_loggers: + if isinstance(callback, GenericAPILogger): + return callback + elif logging_integration == "resend_email": + for callback in _in_memory_loggers: + if isinstance(callback, ResendEmailLogger): + return callback + elif logging_integration == "smtp_email": + for callback in _in_memory_loggers: + if isinstance(callback, SMTPEmailLogger): + return callback return None + except Exception as e: verbose_logger.exception( f"[Non-Blocking Error] Error getting custom logger: {e}" @@ -3317,6 +3350,7 @@ class StandardLoggingPayloadSetup: List[StandardLoggingVectorStoreRequest] ] = None, usage_object: Optional[dict] = None, + proxy_server_request: Optional[dict] = None, ) -> StandardLoggingMetadata: """ Clean and filter the metadata dictionary to include only the specified keys in StandardLoggingMetadata. @@ -3366,6 +3400,7 @@ class StandardLoggingPayloadSetup: mcp_tool_call_metadata=mcp_tool_call_metadata, vector_store_request_metadata=vector_store_request_metadata, usage_object=usage_object, + requester_custom_headers=None, ) if isinstance(metadata, dict): # Filter the metadata dictionary to include only the specified keys @@ -3388,6 +3423,16 @@ class StandardLoggingPayloadSetup: and isinstance(_potential_requester_metadata, dict) ): clean_metadata["requester_metadata"] = _potential_requester_metadata + + if ( + EnterpriseStandardLoggingPayloadSetupVAR + and proxy_server_request is not None + ): + clean_metadata = EnterpriseStandardLoggingPayloadSetupVAR.apply_enterprise_specific_metadata( + standard_logging_metadata=clean_metadata, + proxy_server_request=proxy_server_request, + ) + return clean_metadata @staticmethod @@ -3531,10 +3576,10 @@ class StandardLoggingPayloadSetup: for key in StandardLoggingHiddenParams.__annotations__.keys(): if key in hidden_params: if key == "additional_headers": - clean_hidden_params["additional_headers"] = ( - StandardLoggingPayloadSetup.get_additional_headers( - hidden_params[key] - ) + clean_hidden_params[ + "additional_headers" + ] = StandardLoggingPayloadSetup.get_additional_headers( + hidden_params[key] ) else: clean_hidden_params[key] = hidden_params[key] # type: ignore @@ -3549,7 +3594,10 @@ class StandardLoggingPayloadSetup: @staticmethod def get_error_information( original_exception: Optional[Exception], + traceback_str: Optional[str] = None, ) -> StandardLoggingPayloadErrorInformation: + from litellm.constants import MAXIMUM_TRACEBACK_LINES_TO_LOG + error_status: str = str(getattr(original_exception, "status_code", "")) error_class: str = ( str(original_exception.__class__.__name__) if original_exception else "" @@ -3557,14 +3605,14 @@ class StandardLoggingPayloadSetup: _llm_provider_in_exception = getattr(original_exception, "llm_provider", "") # Get traceback information (first 100 lines) - traceback_info = "" + traceback_info = traceback_str or "" if original_exception: tb = getattr(original_exception, "__traceback__", None) if tb: - import traceback - tb_lines = traceback.format_tb(tb) - traceback_info = "".join(tb_lines[:100]) # Limit to first 100 lines + traceback_info += "".join( + tb_lines[:MAXIMUM_TRACEBACK_LINES_TO_LOG] + ) # Limit to first 100 lines # Get additional error details error_message = str(original_exception) @@ -3677,6 +3725,7 @@ def get_standard_logging_object_payload( or litellm_params.get("metadata", None) or {} ) + completion_start_time = kwargs.get("completion_start_time", end_time) call_type = kwargs.get("call_type") cache_hit = kwargs.get("cache_hit", False) @@ -3718,6 +3767,7 @@ def get_standard_logging_object_payload( clean_hidden_params = StandardLoggingPayloadSetup.get_hidden_params( hidden_params ) + # clean up litellm metadata clean_metadata = StandardLoggingPayloadSetup.get_standard_logging_metadata( metadata=metadata, @@ -3729,6 +3779,7 @@ def get_standard_logging_object_payload( "vector_store_request_metadata", None ), usage_object=usage.model_dump(), + proxy_server_request=proxy_server_request, ) _request_body = proxy_server_request.get("body", {}) @@ -3874,6 +3925,7 @@ def get_standard_logging_metadata( mcp_tool_call_metadata=None, vector_store_request_metadata=None, usage_object=None, + requester_custom_headers=None, ) if isinstance(metadata, dict): # Filter the metadata dictionary to include only the specified keys @@ -3906,9 +3958,9 @@ def scrub_sensitive_keys_in_metadata(litellm_params: Optional[dict]): ): for k, v in metadata["user_api_key_metadata"].items(): if k == "logging": # prevent logging user logging keys - cleaned_user_api_key_metadata[k] = ( - "scrubbed_by_litellm_for_sensitive_keys" - ) + cleaned_user_api_key_metadata[ + k + ] = "scrubbed_by_litellm_for_sensitive_keys" else: cleaned_user_api_key_metadata[k] = v diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index 53d658c5c34..0c534534323 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -17,6 +17,7 @@ from litellm.types.utils import ( ModelResponse, SearchContextCostPerQuery, StandardBuiltInToolsParams, + Usage, ) @@ -27,10 +28,46 @@ class StandardBuiltInToolCostTracking: Example: Web Search """ + @staticmethod + def get_cost_for_anthropic_web_search( + model_info: Optional[ModelInfo] = None, + usage: Optional[Usage] = None, + ) -> float: + """ + Get the cost of using a web search tool for Anthropic. + """ + ## Check if web search requests are in the usage object + if model_info is None: + return 0.0 + + if ( + usage is None + or usage.server_tool_use is None + or usage.server_tool_use.web_search_requests is None + ): + return 0.0 + + ## Get the cost per web search request + search_context_pricing: SearchContextCostPerQuery = ( + model_info.get("search_context_cost_per_query", {}) or {} + ) + cost_per_web_search_request = search_context_pricing.get( + "search_context_size_medium", 0.0 + ) + if cost_per_web_search_request is None or cost_per_web_search_request == 0.0: + return 0.0 + + ## Calculate the total cost + total_cost = ( + cost_per_web_search_request * usage.server_tool_use.web_search_requests + ) + return total_cost + @staticmethod def get_cost_for_built_in_tools( model: str, response_object: Any, + usage: Optional[Usage] = None, custom_llm_provider: Optional[str] = None, standard_built_in_tools_params: Optional[StandardBuiltInToolsParams] = None, ) -> float: @@ -46,17 +83,26 @@ class StandardBuiltInToolCostTracking: # Web Search ######################################################### if StandardBuiltInToolCostTracking.response_object_includes_web_search_call( - response_object=response_object + response_object=response_object, + usage=usage, ): model_info = StandardBuiltInToolCostTracking._safe_get_model_info( model=model, custom_llm_provider=custom_llm_provider ) - return StandardBuiltInToolCostTracking.get_cost_for_web_search( - web_search_options=standard_built_in_tools_params.get( - "web_search_options", None - ), - model_info=model_info, - ) + if custom_llm_provider == "anthropic": + return ( + StandardBuiltInToolCostTracking.get_cost_for_anthropic_web_search( + model_info=model_info, + usage=usage, + ) + ) + else: + return StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=standard_built_in_tools_params.get( + "web_search_options", None + ), + model_info=model_info, + ) ######################################################### # File Search @@ -72,7 +118,7 @@ class StandardBuiltInToolCostTracking: @staticmethod def response_object_includes_web_search_call( - response_object: Any, + response_object: Any, usage: Optional[Usage] = None ) -> bool: """ Check if the response object includes a web search call. @@ -91,6 +137,13 @@ class StandardBuiltInToolCostTracking: return StandardBuiltInToolCostTracking.response_includes_output_type( response_object=response_object, output_type="web_search_call" ) + elif ( + usage is not None + and hasattr(usage, "server_tool_use") + and usage.server_tool_use is not None + and usage.server_tool_use.web_search_requests is not None + ): + return True return False @staticmethod diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index ca355fe0ed3..5055b5db5a8 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -255,7 +255,9 @@ def _parse_content_for_reasoning( if not message_text: return None, message_text - reasoning_match = re.match(r"(.*?)(.*)", message_text, re.DOTALL) + reasoning_match = re.match( + r"<(?:think|thinking)>(.*?)(.*)", message_text, re.DOTALL + ) if reasoning_match: return reasoning_match.group(1), reasoning_match.group(2) diff --git a/litellm/litellm_core_utils/llm_response_utils/get_api_base.py b/litellm/litellm_core_utils/llm_response_utils/get_api_base.py index ddac7ac3244..6f9fa36591f 100644 --- a/litellm/litellm_core_utils/llm_response_utils/get_api_base.py +++ b/litellm/litellm_core_utils/llm_response_utils/get_api_base.py @@ -37,8 +37,7 @@ def get_api_base( _optional_params = LiteLLM_Params( model=model, **optional_params ) # convert to pydantic object - except Exception as e: - verbose_logger.debug("Error occurred in getting api base - {}".format(str(e))) + except Exception: return None # get llm provider diff --git a/litellm/litellm_core_utils/logging_callback_manager.py b/litellm/litellm_core_utils/logging_callback_manager.py index c57a2401b7b..dec3add4e1b 100644 --- a/litellm/litellm_core_utils/logging_callback_manager.py +++ b/litellm/litellm_core_utils/logging_callback_manager.py @@ -260,4 +260,9 @@ class LoggingCallbackManager: """ Get all custom loggers that are instances of the given class type """ - return [c for c in self._get_all_callbacks() if isinstance(c, callback_type)] + # ensure we don't have duplicate instances + all_callbacks = [] + for callback in self._get_all_callbacks(): + if isinstance(callback, callback_type) and callback not in all_callbacks: + all_callbacks.append(callback) + return all_callbacks diff --git a/litellm/litellm_core_utils/prompt_templates/common_utils.py b/litellm/litellm_core_utils/prompt_templates/common_utils.py index b6af4a710ad..387c072ffd7 100644 --- a/litellm/litellm_core_utils/prompt_templates/common_utils.py +++ b/litellm/litellm_core_utils/prompt_templates/common_utils.py @@ -346,14 +346,14 @@ def get_format_from_file_id(file_id: Optional[str]) -> Optional[str]: unified_file_id = litellm_proxy:{};unified_id,{} If not a unified file id, returns 'file' as default format """ - from litellm.proxy.hooks.managed_files import _PROXY_LiteLLMManagedFiles + from litellm.proxy.openai_files_endpoints.common_utils import ( + convert_b64_uid_to_unified_uid, + ) if not file_id: return None try: - transformed_file_id = ( - _PROXY_LiteLLMManagedFiles._convert_b64_uid_to_unified_uid(file_id) - ) + transformed_file_id = convert_b64_uid_to_unified_uid(file_id) if transformed_file_id.startswith( SpecialEnums.LITELM_MANAGED_FILE_ID_PREFIX.value ): diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 5b11b224bb0..2386e82d4a5 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -55,6 +55,11 @@ DEFAULT_USER_CONTINUE_MESSAGE = { "content": "Please continue.", } # similar to autogen. Only used if `litellm.modify_params=True`. +DEFAULT_USER_CONTINUE_MESSAGE_TYPED = ChatCompletionUserMessage( + role="user", + content="Please continue.", +) + # used to interweave assistant messages, to ensure user/assistant alternating DEFAULT_ASSISTANT_CONTINUE_MESSAGE = ChatCompletionAssistantMessage( role="assistant", @@ -1139,7 +1144,7 @@ def convert_to_gemini_tool_call_result( def convert_to_anthropic_tool_result( - message: Union[ChatCompletionToolMessage, ChatCompletionFunctionMessage] + message: Union[ChatCompletionToolMessage, ChatCompletionFunctionMessage], ) -> AnthropicMessagesToolResultParam: """ OpenAI message with a tool result looks like: @@ -1408,6 +1413,17 @@ def anthropic_messages_pt( # noqa: PLR0915 AnthopicMessagesAssistantMessageParam, ] ] = [] + + if len(messages) == 0: + if not litellm.modify_params: + raise litellm.BadRequestError( + message=f"Anthropic requires at least one non-system message. Either provide one, or set `litellm.modify_params = True` // `litellm_settings::modify_params: True` to add the dummy user message - {DEFAULT_USER_CONTINUE_MESSAGE_TYPED}.", + model=model, + llm_provider=llm_provider, + ) + else: + messages.append(DEFAULT_USER_CONTINUE_MESSAGE_TYPED) + msg_i = 0 while msg_i < len(messages): user_content: List[AnthropicMessagesUserMessageValues] = [] @@ -1613,7 +1629,7 @@ def anthropic_messages_pt( # noqa: PLR0915 llm_provider=llm_provider, ) - if new_messages[-1]["role"] == "assistant": + if len(new_messages) > 0 and new_messages[-1]["role"] == "assistant": if isinstance(new_messages[-1]["content"], str): new_messages[-1]["content"] = new_messages[-1]["content"].rstrip() elif isinstance(new_messages[-1]["content"], list): @@ -2244,6 +2260,7 @@ from litellm.types.llms.bedrock import ToolBlock as BedrockToolBlock from litellm.types.llms.bedrock import ( ToolInputSchemaBlock as BedrockToolInputSchemaBlock, ) +from litellm.types.llms.bedrock import ToolJsonSchemaBlock as BedrockToolJsonSchemaBlock from litellm.types.llms.bedrock import ToolResultBlock as BedrockToolResultBlock from litellm.types.llms.bedrock import ( ToolResultContentBlock as BedrockToolResultContentBlock, @@ -2499,7 +2516,7 @@ def _convert_to_bedrock_tool_call_invoke( def _convert_to_bedrock_tool_call_result( - message: Union[ChatCompletionToolMessage, ChatCompletionFunctionMessage] + message: Union[ChatCompletionToolMessage, ChatCompletionFunctionMessage], ) -> BedrockContentBlock: """ OpenAI message with a tool result looks like: @@ -2672,7 +2689,7 @@ def get_user_message_block_or_continue_message( def return_assistant_continue_message( assistant_continue_message: Optional[ Union[str, ChatCompletionAssistantMessage] - ] = None + ] = None, ) -> ChatCompletionAssistantMessage: if assistant_continue_message and isinstance(assistant_continue_message, str): return ChatCompletionAssistantMessage( @@ -3024,6 +3041,19 @@ class BedrockConverseMessagesProcessor: ) ) _assistant_content = assistant_message_block.get("content", None) + thinking_blocks = cast( + Optional[List[ChatCompletionThinkingBlock]], + assistant_message_block.get("thinking_blocks"), + ) + + if thinking_blocks is not None: + converted_thinking_blocks = BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks + ) + assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=converted_thinking_blocks, + assistant_parts=assistant_content, + ) if _assistant_content is not None and isinstance( _assistant_content, list @@ -3037,7 +3067,10 @@ class BedrockConverseMessagesProcessor: cast(ChatCompletionThinkingBlock, element) ] ) - assistants_parts.extend(thinking_block) + assistants_parts = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=thinking_block, + assistant_parts=assistants_parts, + ) elif element["type"] == "text": assistants_part = BedrockContentBlock( text=element["text"] @@ -3142,6 +3175,37 @@ class BedrockConverseMessagesProcessor: image_url=cast(str, file_id or file_data), format=format ) + @staticmethod + def add_thinking_blocks_to_assistant_content( + thinking_blocks: List[BedrockContentBlock], + assistant_parts: List[BedrockContentBlock], + ) -> List[BedrockContentBlock]: + """ + If contains 'signature', it is a thinking block. + If missing 'signature', it is a text block - e.g. when using a non-anthropic model. + + Handle error raised by bedrock if thinking blocks are provided for a non-thinking model (e.g. nova with tool use) + + Relevant Issue: https://github.com/BerriAI/litellm/issues/9063 + """ + filtered_thinking_blocks = [] + for block in thinking_blocks: + reasoning_content = block.get("reasoningContent", None) + reasoning_text = ( + reasoning_content.get("reasoningText", None) + if reasoning_content is not None + else None + ) + if reasoning_text and not reasoning_text.get("signature"): + reasoning_text_text = reasoning_text["text"] + assistants_part = BedrockContentBlock(text=reasoning_text_text) + assistant_parts.append(assistants_part) + else: + filtered_thinking_blocks.append(block) + if len(filtered_thinking_blocks) > 0: + assistant_parts.extend(filtered_thinking_blocks) + return assistant_parts + def _bedrock_converse_messages_pt( # noqa: PLR0915 messages: List, @@ -3309,10 +3373,12 @@ def _bedrock_converse_messages_pt( # noqa: PLR0915 ) if thinking_blocks is not None: - assistant_content.extend( - BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( - thinking_blocks - ) + converted_thinking_blocks = BedrockConverseMessagesProcessor.translate_thinking_blocks_to_reasoning_content_blocks( + thinking_blocks + ) + assistant_content = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=converted_thinking_blocks, + assistant_parts=assistant_content, ) if _assistant_content is not None and isinstance(_assistant_content, list): @@ -3325,7 +3391,10 @@ def _bedrock_converse_messages_pt( # noqa: PLR0915 cast(ChatCompletionThinkingBlock, element) ] ) - assistants_parts.extend(thinking_block) + assistants_parts = BedrockConverseMessagesProcessor.add_thinking_blocks_to_assistant_content( + thinking_blocks=thinking_block, + assistant_parts=assistants_parts, + ) elif element["type"] == "text": assistants_part = BedrockContentBlock(text=element["text"]) assistants_parts.append(assistants_part) @@ -3399,6 +3468,15 @@ def make_valid_bedrock_tool_name(input_tool_name: str) -> str: return valid_string +def add_cache_point_tool_block(tool: dict) -> Optional[BedrockToolBlock]: + cache_control = tool.get("cache_control", None) + if cache_control is not None: + cache_point = cache_control.get("type", "ephemeral") + if cache_point == "ephemeral": + return {"cachePoint": {"type": "default"}} + return None + + def _bedrock_tools_pt(tools: List) -> List[BedrockToolBlock]: """ OpenAI tools looks like: @@ -3470,13 +3548,24 @@ def _bedrock_tools_pt(tools: List) -> List[BedrockToolBlock]: for _, value in defs_copy.items(): unpack_defs(value, defs_copy) unpack_defs(parameters, defs_copy) - tool_input_schema = BedrockToolInputSchemaBlock(json=parameters) + tool_input_schema = BedrockToolInputSchemaBlock( + json=BedrockToolJsonSchemaBlock( + type=parameters.get("type", ""), + properties=parameters.get("properties", {}), + required=parameters.get("required", []), + ) + ) tool_spec = BedrockToolSpecBlock( inputSchema=tool_input_schema, name=name, description=description ) tool_block = BedrockToolBlock(toolSpec=tool_spec) tool_block_list.append(tool_block) + ## ADD CACHE POINT TOOL BLOCK ## + cache_point_tool_block = add_cache_point_tool_block(tool) + if cache_point_tool_block is not None: + tool_block_list.append(cache_point_tool_block) + return tool_block_list @@ -3633,7 +3722,7 @@ def prompt_factory( return mistral_instruct_pt(messages=messages) elif "llama2" in model and "chat" in model: return llama_2_chat_pt(messages=messages) - elif "llama3" in model and "instruct" in model: + elif ("llama3" in model or "llama4" in model) and "instruct" in model: return hf_chat_template( model="meta-llama/Meta-Llama-3-8B-Instruct", messages=messages, diff --git a/litellm/litellm_core_utils/realtime_streaming.py b/litellm/litellm_core_utils/realtime_streaming.py index 5dcabe2dd35..329f2b63c20 100644 --- a/litellm/litellm_core_utils/realtime_streaming.py +++ b/litellm/litellm_core_utils/realtime_streaming.py @@ -1,43 +1,29 @@ -""" -async with websockets.connect( # type: ignore - url, - extra_headers={ - "api-key": api_key, # type: ignore - }, - ) as backend_ws: - forward_task = asyncio.create_task( - forward_messages(websocket, backend_ws) - ) - - try: - while True: - message = await websocket.receive_text() - await backend_ws.send(message) - except websockets.exceptions.ConnectionClosed: # type: ignore - forward_task.cancel() - finally: - if not forward_task.done(): - forward_task.cancel() - try: - await forward_task - except asyncio.CancelledError: - pass -""" - import asyncio import concurrent.futures import json -from typing import Any, Dict, List, Optional, Union +from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union import litellm from litellm._logging import verbose_logger +from litellm.llms.base_llm.realtime.transformation import BaseRealtimeConfig from litellm.types.llms.openai import ( + OpenAIRealtimeEvents, + OpenAIRealtimeOutputItemDone, + OpenAIRealtimeResponseDelta, OpenAIRealtimeStreamResponseBaseObject, OpenAIRealtimeStreamSessionEvents, ) +from litellm.types.realtime import ALL_DELTA_TYPES from .litellm_logging import Logging as LiteLLMLogging +if TYPE_CHECKING: + from websockets.asyncio.client import ClientConnection + + CLIENT_CONNECTION_CLASS = ClientConnection +else: + CLIENT_CONNECTION_CLASS = Any + # Create a thread pool with a maximum of 10 threads executor = concurrent.futures.ThreadPoolExecutor(max_workers=10) @@ -52,18 +38,15 @@ class RealTimeStreaming: def __init__( self, websocket: Any, - backend_ws: Any, - logging_obj: Optional[LiteLLMLogging] = None, + backend_ws: CLIENT_CONNECTION_CLASS, + logging_obj: LiteLLMLogging, + provider_config: Optional[BaseRealtimeConfig] = None, + model: str = "", ): self.websocket = websocket self.backend_ws = backend_ws self.logging_obj = logging_obj - self.messages: List[ - Union[ - OpenAIRealtimeStreamResponseBaseObject, - OpenAIRealtimeStreamSessionEvents, - ] - ] = [] + self.messages: List[OpenAIRealtimeEvents] = [] self.input_message: Dict = {} _logged_real_time_event_types = litellm.logged_real_time_event_types @@ -71,34 +54,43 @@ class RealTimeStreaming: if _logged_real_time_event_types is None: _logged_real_time_event_types = DefaultLoggedRealTimeEventTypes self.logged_real_time_event_types = _logged_real_time_event_types + self.provider_config = provider_config + self.model = model + self.current_delta_chunks: Optional[List[OpenAIRealtimeResponseDelta]] = None + self.current_output_item_id: Optional[str] = None + self.current_response_id: Optional[str] = None + self.current_conversation_id: Optional[str] = None + self.current_item_chunks: Optional[List[OpenAIRealtimeOutputItemDone]] = None + self.current_delta_type: Optional[ALL_DELTA_TYPES] = None + self.session_configuration_request: Optional[str] = None def _should_store_message( self, - message_obj: Union[ - dict, - OpenAIRealtimeStreamSessionEvents, - OpenAIRealtimeStreamResponseBaseObject, - ], + message_obj: Union[dict, OpenAIRealtimeEvents], ) -> bool: - _msg_type = message_obj["type"] + _msg_type = message_obj["type"] if "type" in message_obj else None if self.logged_real_time_event_types == "*": return True - if _msg_type in self.logged_real_time_event_types: + if _msg_type and _msg_type in self.logged_real_time_event_types: return True return False - def store_message(self, message: Union[str, bytes]): + def store_message(self, message: Union[str, bytes, OpenAIRealtimeEvents]): """Store message in list""" if isinstance(message, bytes): message = message.decode("utf-8") - message_obj = json.loads(message) + if isinstance(message, dict): + message_obj = message + else: + message_obj = json.loads(message) try: if ( - message_obj.get("type") == "session.created" + not isinstance(message, dict) + or message_obj.get("type") == "session.created" or message_obj.get("type") == "session.updated" ): message_obj = OpenAIRealtimeStreamSessionEvents(**message_obj) # type: ignore - else: + elif not isinstance(message, dict): message_obj = OpenAIRealtimeStreamResponseBaseObject(**message_obj) # type: ignore except Exception as e: verbose_logger.debug(f"Error parsing message for logging: {e}") @@ -126,15 +118,66 @@ class RealTimeStreaming: try: while True: - message = await self.backend_ws.recv() - await self.websocket.send_text(message) + try: + raw_response = await self.backend_ws.recv( + decode=False + ) # improves performance + except TypeError: + raw_response = await self.backend_ws.recv() # type: ignore[assignment] - ## LOGGING - self.store_message(message) - except websockets.exceptions.ConnectionClosed: # type: ignore - pass - except Exception: - pass + if self.provider_config: + returned_object = self.provider_config.transform_realtime_response( + raw_response, + self.model, + self.logging_obj, + realtime_response_transform_input={ + "session_configuration_request": self.session_configuration_request, + "current_output_item_id": self.current_output_item_id, + "current_response_id": self.current_response_id, + "current_delta_chunks": self.current_delta_chunks, + "current_conversation_id": self.current_conversation_id, + "current_item_chunks": self.current_item_chunks, + "current_delta_type": self.current_delta_type, + }, + ) + + transformed_response = returned_object["response"] + self.current_output_item_id = returned_object[ + "current_output_item_id" + ] + self.current_response_id = returned_object["current_response_id"] + self.current_delta_chunks = returned_object["current_delta_chunks"] + self.current_conversation_id = returned_object[ + "current_conversation_id" + ] + self.current_item_chunks = returned_object["current_item_chunks"] + self.current_delta_type = returned_object["current_delta_type"] + self.session_configuration_request = returned_object[ + "session_configuration_request" + ] + if isinstance(transformed_response, list): + for event in transformed_response: + event_str = json.dumps(event) + ## LOGGING + self.store_message(event_str) + await self.websocket.send_text(event_str) + else: + event_str = json.dumps(transformed_response) + ## LOGGING + self.store_message(event_str) + await self.websocket.send_text(event_str) + + else: + ## LOGGING + self.store_message(raw_response) + await self.websocket.send_text(raw_response) + + except websockets.exceptions.ConnectionClosed as e: # type: ignore + verbose_logger.exception( + f"Connection closed in backend to client send messages - {e}" + ) + except Exception as e: + verbose_logger.exception(f"Error in backend to client send messages: {e}") finally: await self.log_messages() @@ -142,18 +185,29 @@ class RealTimeStreaming: try: while True: message = await self.websocket.receive_text() + ## LOGGING self.store_input(message=message) ## FORWARD TO BACKEND - await self.backend_ws.send(message) - except self.websockets.exceptions.ConnectionClosed: # type: ignore - pass + if self.provider_config: + message = self.provider_config.transform_realtime_request( + message, self.model + ) + + for msg in message: + await self.backend_ws.send(msg) + else: + await self.backend_ws.send(message) + + except Exception as e: + verbose_logger.debug(f"Error in client ack messages: {e}") async def bidirectional_forward(self): forward_task = asyncio.create_task(self.backend_to_client_send_messages()) try: await self.client_ack_messages() - except self.websockets.exceptions.ConnectionClosed: # type: ignore + except self.websocket.exceptions.ConnectionClosed: # type: ignore + verbose_logger.debug("Connection closed") forward_task.cancel() finally: if not forward_task.done(): diff --git a/litellm/litellm_core_utils/safe_json_loads.py b/litellm/litellm_core_utils/safe_json_loads.py new file mode 100644 index 00000000000..a7ab0d3e3b5 --- /dev/null +++ b/litellm/litellm_core_utils/safe_json_loads.py @@ -0,0 +1,14 @@ +""" +Helper for safe JSON loading in LiteLLM. +""" +from typing import Any +import json + +def safe_json_loads(data: str, default: Any = None) -> Any: + """ + Safely parse a JSON string. If parsing fails, return the default value (None by default). + """ + try: + return json.loads(data) + except Exception: + return default \ No newline at end of file diff --git a/litellm/litellm_core_utils/sensitive_data_masker.py b/litellm/litellm_core_utils/sensitive_data_masker.py index 23b9ec32fc7..900239602df 100644 --- a/litellm/litellm_core_utils/sensitive_data_masker.py +++ b/litellm/litellm_core_utils/sensitive_data_masker.py @@ -1,6 +1,6 @@ from typing import Any, Dict, Optional, Set -from litellm.constants import DEFAULT_MAX_RECURSE_DEPTH +from litellm.constants import DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER class SensitiveDataMasker: @@ -44,7 +44,7 @@ class SensitiveDataMasker: self, data: Dict[str, Any], depth: int = 0, - max_depth: int = DEFAULT_MAX_RECURSE_DEPTH, + max_depth: int = DEFAULT_MAX_RECURSE_DEPTH_SENSITIVE_DATA_MASKER, ) -> Dict[str, Any]: if depth >= max_depth: return data diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index ec20a1ad4cf..785186b8ab8 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -322,7 +322,7 @@ class CustomStreamWrapper: is_finished = False finish_reason = "" try: - if "dolphin" in self.model: + if self.model and "dolphin" in self.model: chunk = self.process_chunk(chunk=chunk) else: data_json = json.loads(chunk) @@ -978,8 +978,10 @@ class CustomStreamWrapper: ] if anthropic_response_obj["usage"] is not None: - model_response.usage = litellm.Usage( - **anthropic_response_obj["usage"] + setattr( + model_response, + "usage", + litellm.Usage(**anthropic_response_obj["usage"]), ) if ( @@ -1046,19 +1048,21 @@ class CustomStreamWrapper: if self.sent_first_chunk is False: raise Exception("An unknown error occurred with the stream") self.received_finish_reason = "stop" - elif self.custom_llm_provider == "vertex_ai": + elif self.custom_llm_provider == "vertex_ai" and not isinstance( + chunk, ModelResponseStream + ): import proto # type: ignore if hasattr(chunk, "candidates") is True: try: try: - completion_obj["content"] = chunk.text + completion_obj["content"] = chunk.text # type: ignore except Exception as e: original_exception = e if "Part has no text." in str(e): ## check for function calling function_call = ( - chunk.candidates[0].content.parts[0].function_call + chunk.candidates[0].content.parts[0].function_call # type: ignore ) args_dict = {} @@ -1067,7 +1071,7 @@ class CustomStreamWrapper: for key, val in function_call.args.items(): if isinstance( val, - proto.marshal.collections.repeated.RepeatedComposite, + proto.marshal.collections.repeated.RepeatedComposite, # type: ignore ): # If so, convert to list args_dict[key] = [v for v in val] @@ -1098,15 +1102,15 @@ class CustomStreamWrapper: else: raise original_exception if ( - hasattr(chunk.candidates[0], "finish_reason") - and chunk.candidates[0].finish_reason.name + hasattr(chunk.candidates[0], "finish_reason") # type: ignore + and chunk.candidates[0].finish_reason.name # type: ignore != "FINISH_REASON_UNSPECIFIED" ): # every non-final chunk in vertex ai has this - self.received_finish_reason = chunk.candidates[ + self.received_finish_reason = chunk.candidates[ # type: ignore 0 ].finish_reason.name except Exception: - if chunk.candidates[0].finish_reason.name == "SAFETY": + if chunk.candidates[0].finish_reason.name == "SAFETY": # type: ignore raise Exception( f"The response was blocked by VertexAI. {str(chunk)}" ) @@ -1153,12 +1157,18 @@ class CustomStreamWrapper: if response_obj["is_finished"]: self.received_finish_reason = response_obj["finish_reason"] if response_obj["usage"] is not None: - model_response.usage = litellm.Usage( - prompt_tokens=response_obj["usage"].prompt_tokens, - completion_tokens=response_obj["usage"].completion_tokens, - total_tokens=response_obj["usage"].total_tokens, + setattr( + model_response, + "usage", + litellm.Usage( + prompt_tokens=response_obj["usage"].prompt_tokens, + completion_tokens=response_obj["usage"].completion_tokens, + total_tokens=response_obj["usage"].total_tokens, + ), ) elif self.custom_llm_provider == "text-completion-codestral": + if not isinstance(chunk, str): + raise ValueError(f"chunk is not a string: {chunk}") response_obj = cast( Dict[str, Any], litellm.CodestralTextCompletionConfig()._chunk_parser(chunk), @@ -1168,10 +1178,14 @@ class CustomStreamWrapper: if response_obj["is_finished"]: self.received_finish_reason = response_obj["finish_reason"] if "usage" in response_obj is not None: - model_response.usage = litellm.Usage( - prompt_tokens=response_obj["usage"].prompt_tokens, - completion_tokens=response_obj["usage"].completion_tokens, - total_tokens=response_obj["usage"].total_tokens, + setattr( + model_response, + "usage", + litellm.Usage( + prompt_tokens=response_obj["usage"].prompt_tokens, + completion_tokens=response_obj["usage"].completion_tokens, + total_tokens=response_obj["usage"].total_tokens, + ), ) elif self.custom_llm_provider == "azure_text": response_obj = self.handle_azure_text_completion_chunk(chunk) @@ -1669,7 +1683,7 @@ class CustomStreamWrapper: processed_chunk: Optional[ModelResponseStream] = self.chunk_creator( chunk=chunk ) - print_verbose( + verbose_logger.debug( f"PROCESSED ASYNC CHUNK POST CHUNK CREATOR: {processed_chunk}" ) if processed_chunk is None: diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 06e0553f8d5..9052cec97cf 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -6,6 +6,7 @@ import httpx import litellm from litellm.constants import ( + ANTHROPIC_WEB_SEARCH_TOOL_MAX_USES, DEFAULT_ANTHROPIC_CHAT_MAX_TOKENS, DEFAULT_REASONING_EFFORT_HIGH_THINKING_BUDGET, DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, @@ -25,6 +26,8 @@ from litellm.types.llms.anthropic import ( AnthropicMessagesToolChoice, AnthropicSystemMessageContent, AnthropicThinkingParam, + AnthropicWebSearchTool, + AnthropicWebSearchUserLocation, ) from litellm.types.llms.openai import ( REASONING_EFFORT, @@ -36,10 +39,11 @@ from litellm.types.llms.openai import ( ChatCompletionToolCallChunk, ChatCompletionToolCallFunctionChunk, ChatCompletionToolParam, + OpenAIWebSearchOptions, ) from litellm.types.utils import CompletionTokensDetailsWrapper from litellm.types.utils import Message as LitellmMessage -from litellm.types.utils import PromptTokensDetailsWrapper +from litellm.types.utils import PromptTokensDetailsWrapper, ServerToolUse from litellm.utils import ( ModelResponse, Usage, @@ -58,6 +62,9 @@ else: LoggingClass = Any +ANTHROPIC_HOSTED_TOOLS = ["web_search", "bash", "text_editor"] + + class AnthropicConfig(AnthropicModelInfo, BaseConfig): """ Reference: https://docs.anthropic.com/claude/reference/messages_post @@ -111,6 +118,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): "response_format", "user", "reasoning_effort", + "web_search_options", ] if "claude-3-7-sonnet" in model: @@ -212,16 +220,18 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): _computer_tool["display_number"] = _display_number returned_tool = _computer_tool - elif tool["type"].startswith("bash_") or tool["type"].startswith( - "text_editor_" - ): - function_name = tool["function"].get("name") - if function_name is None: + elif any(tool["type"].startswith(t) for t in ANTHROPIC_HOSTED_TOOLS): + function_name = tool.get("name", tool.get("function", {}).get("name")) + if function_name is None or not isinstance(function_name, str): raise ValueError("Missing required parameter: name") + additional_tool_params = {} + for k, v in tool.items(): + if k != "type" and k != "name": + additional_tool_params[k] = v + returned_tool = AnthropicHostedTools( - type=tool["type"], - name=function_name, + type=tool["type"], name=function_name, **additional_tool_params # type: ignore ) if returned_tool is None: raise ValueError(f"Unsupported tool type: {tool['type']}") @@ -324,6 +334,37 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): return _tool + def map_web_search_tool( + self, + value: OpenAIWebSearchOptions, + ) -> AnthropicWebSearchTool: + value_typed = cast(OpenAIWebSearchOptions, value) + hosted_web_search_tool = AnthropicWebSearchTool( + type="web_search_20250305", + name="web_search", + ) + user_location = value_typed.get("user_location") + if user_location is not None: + anthropic_user_location = AnthropicWebSearchUserLocation(type="approximate") + anthropic_user_location_keys = ( + AnthropicWebSearchUserLocation.__annotations__.keys() + ) + user_location_approximate = user_location.get("approximate") + if user_location_approximate is not None: + for key, user_location_value in user_location_approximate.items(): + if key in anthropic_user_location_keys and key != "type": + anthropic_user_location[key] = user_location_value # type: ignore + hosted_web_search_tool["user_location"] = anthropic_user_location + + ## MAP SEARCH CONTEXT SIZE + search_context_size = value_typed.get("search_context_size") + if search_context_size is not None: + hosted_web_search_tool["max_uses"] = ANTHROPIC_WEB_SEARCH_TOOL_MAX_USES[ + search_context_size + ] + + return hosted_web_search_tool + def map_openai_params( self, non_default_params: dict, @@ -387,11 +428,19 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): optional_params["thinking"] = AnthropicConfig._map_reasoning_effort( value ) + elif param == "web_search_options" and isinstance(value, dict): + hosted_web_search_tool = self.map_web_search_tool( + cast(OpenAIWebSearchOptions, value) + ) + self._add_tools_to_optional_params( + optional_params=optional_params, tools=[hosted_web_search_tool] + ) ## handle thinking tokens self.update_optional_params_with_thinking_tokens( non_default_params=non_default_params, optional_params=optional_params ) + return optional_params def _create_json_tool_call_for_response_format( @@ -643,15 +692,20 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): _usage = usage_object cache_creation_input_tokens: int = 0 cache_read_input_tokens: int = 0 - + web_search_requests: Optional[int] = None if "cache_creation_input_tokens" in _usage: cache_creation_input_tokens = _usage["cache_creation_input_tokens"] if "cache_read_input_tokens" in _usage: cache_read_input_tokens = _usage["cache_read_input_tokens"] prompt_tokens += cache_read_input_tokens + if "server_tool_use" in _usage: + if "web_search_requests" in _usage["server_tool_use"]: + web_search_requests = cast( + int, _usage["server_tool_use"]["web_search_requests"] + ) prompt_tokens_details = PromptTokensDetailsWrapper( - cached_tokens=cache_read_input_tokens + cached_tokens=cache_read_input_tokens, ) completion_token_details = ( CompletionTokensDetailsWrapper( @@ -663,6 +717,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): else None ) total_tokens = prompt_tokens + completion_tokens + usage = Usage( prompt_tokens=prompt_tokens, completion_tokens=completion_tokens, @@ -671,6 +726,9 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): cache_creation_input_tokens=cache_creation_input_tokens, cache_read_input_tokens=cache_read_input_tokens, completion_tokens_details=completion_token_details, + server_tool_use=ServerToolUse(web_search_requests=web_search_requests) + if web_search_requests is not None + else None, ) return usage diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index ab335ca7c16..b7c8fb56502 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -5,62 +5,30 @@ """ -import json -from typing import AsyncIterator, Dict, List, Optional, Union, cast - -import httpx +import asyncio +import contextvars +from functools import partial +from typing import Any, AsyncIterator, Coroutine, Dict, List, Optional, Union import litellm from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.base_llm.anthropic_messages.transformation import ( BaseAnthropicMessagesConfig, ) -from litellm.llms.custom_httpx.http_handler import ( - AsyncHTTPHandler, - get_async_httpx_client, -) +from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler +from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler from litellm.types.llms.anthropic_messages.anthropic_response import ( AnthropicMessagesResponse, ) from litellm.types.router import GenericLiteLLMParams -from litellm.types.utils import ProviderSpecificHeader from litellm.utils import ProviderConfigManager, client +from .utils import AnthropicMessagesRequestUtils -class AnthropicMessagesHandler: - @staticmethod - async def _handle_anthropic_streaming( - response: httpx.Response, - request_body: dict, - litellm_logging_obj: LiteLLMLoggingObj, - ) -> AsyncIterator: - """Helper function to handle Anthropic streaming responses using the existing logging handlers""" - from datetime import datetime - - from litellm.proxy.pass_through_endpoints.streaming_handler import ( - PassThroughStreamingHandler, - ) - from litellm.proxy.pass_through_endpoints.success_handler import ( - PassThroughEndpointLogging, - ) - from litellm.types.passthrough_endpoints.pass_through_endpoints import ( - EndpointType, - ) - - # Create success handler object - passthrough_success_handler_obj = PassThroughEndpointLogging() - - # Use the existing streaming handler for Anthropic - start_time = datetime.now() - return PassThroughStreamingHandler.chunk_processor( - response=response, - request_body=request_body, - litellm_logging_obj=litellm_logging_obj, - endpoint_type=EndpointType.ANTHROPIC, - start_time=start_time, - passthrough_success_handler_obj=passthrough_success_handler_obj, - url_route="/v1/messages", - ) +####### ENVIRONMENT VARIABLES ################### +# Initialize any necessary instances or variables here +base_llm_http_handler = BaseLLMHTTPHandler() +################################################# @client @@ -84,114 +52,121 @@ async def anthropic_messages( custom_llm_provider: Optional[str] = None, **kwargs, ) -> Union[AnthropicMessagesResponse, AsyncIterator]: + """ + Async: Make llm api request in Anthropic /messages API spec + """ + local_vars = locals() + loop = asyncio.get_event_loop() + kwargs["anthropic_messages"] = True + + func = partial( + anthropic_messages_handler, + max_tokens=max_tokens, + messages=messages, + model=model, + metadata=metadata, + stop_sequences=stop_sequences, + stream=stream, + system=system, + temperature=temperature, + thinking=thinking, + tool_choice=tool_choice, + tools=tools, + top_k=top_k, + top_p=top_p, + api_key=api_key, + api_base=api_base, + client=client, + custom_llm_provider=custom_llm_provider, + **kwargs, + ) + ctx = contextvars.copy_context() + func_with_context = partial(ctx.run, func) + init_response = await loop.run_in_executor(None, func_with_context) + + if asyncio.iscoroutine(init_response): + response = await init_response + else: + response = init_response + return response + + +def anthropic_messages_handler( + max_tokens: int, + messages: List[Dict], + model: str, + metadata: Optional[Dict] = None, + stop_sequences: Optional[List[str]] = None, + stream: Optional[bool] = False, + system: Optional[str] = None, + temperature: Optional[float] = None, + thinking: Optional[Dict] = None, + tool_choice: Optional[Dict] = None, + tools: Optional[List[Dict]] = None, + top_k: Optional[int] = None, + top_p: Optional[float] = None, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + client: Optional[AsyncHTTPHandler] = None, + custom_llm_provider: Optional[str] = None, + **kwargs, +) -> Union[ + AnthropicMessagesResponse, + Coroutine[Any, Any, Union[AnthropicMessagesResponse, AsyncIterator]], +]: """ Makes Anthropic `/v1/messages` API calls In the Anthropic API Spec """ + local_vars = locals() # Use provided client or create a new one - optional_params = GenericLiteLLMParams(**kwargs) + litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj") # type: ignore + litellm_params = GenericLiteLLMParams(**kwargs) ( model, - _custom_llm_provider, + custom_llm_provider, dynamic_api_key, dynamic_api_base, ) = litellm.get_llm_provider( model=model, custom_llm_provider=custom_llm_provider, - api_base=optional_params.api_base, - api_key=optional_params.api_key, + api_base=litellm_params.api_base, + api_key=litellm_params.api_key, ) - anthropic_messages_provider_config: Optional[BaseAnthropicMessagesConfig] = ( - ProviderConfigManager.get_provider_anthropic_messages_config( - model=model, - provider=litellm.LlmProviders(_custom_llm_provider), - ) + anthropic_messages_provider_config: Optional[ + BaseAnthropicMessagesConfig + ] = ProviderConfigManager.get_provider_anthropic_messages_config( + model=model, + provider=litellm.LlmProviders(custom_llm_provider), ) if anthropic_messages_provider_config is None: raise ValueError( f"Anthropic messages provider config not found for model: {model}" ) - if client is None or not isinstance(client, AsyncHTTPHandler): - async_httpx_client = get_async_httpx_client( - llm_provider=litellm.LlmProviders.ANTHROPIC + if custom_llm_provider is None: + raise ValueError( + f"custom_llm_provider is required for Anthropic messages, passed in model={model}, custom_llm_provider={custom_llm_provider}" ) - else: - async_httpx_client = client - litellm_logging_obj: LiteLLMLoggingObj = kwargs.get("litellm_logging_obj", None) - - # Prepare headers - provider_specific_header = cast( - Optional[ProviderSpecificHeader], kwargs.get("provider_specific_header", None) + local_vars.update(kwargs) + anthropic_messages_optional_request_params = ( + AnthropicMessagesRequestUtils.get_requested_anthropic_messages_optional_param( + params=local_vars + ) ) - extra_headers = ( - provider_specific_header.get("extra_headers", {}) - if provider_specific_header - else {} - ) - headers = anthropic_messages_provider_config.validate_environment( - headers=extra_headers or {}, + return base_llm_http_handler.anthropic_messages_handler( model=model, + messages=messages, + anthropic_messages_provider_config=anthropic_messages_provider_config, + anthropic_messages_optional_request_params=dict( + anthropic_messages_optional_request_params + ), + _is_async=True, + client=client, + custom_llm_provider=custom_llm_provider, + litellm_params=litellm_params, + logging_obj=litellm_logging_obj, api_key=api_key, + api_base=api_base, + stream=stream, + kwargs=kwargs, ) - - litellm_logging_obj.update_environment_variables( - model=model, - optional_params=dict(optional_params), - litellm_params={ - "metadata": kwargs.get("metadata", {}), - "preset_cache_key": None, - "stream_response": {}, - **optional_params.model_dump(exclude_unset=True), - }, - custom_llm_provider=_custom_llm_provider, - ) - # Prepare request body - request_body = locals().copy() - request_body = { - k: v - for k, v in request_body.items() - if k - in anthropic_messages_provider_config.get_supported_anthropic_messages_params( - model=model - ) - and v is not None - } - request_body["stream"] = stream - request_body["model"] = model - litellm_logging_obj.stream = stream - litellm_logging_obj.model_call_details.update(request_body) - - # Make the request - request_url = anthropic_messages_provider_config.get_complete_url( - api_base=api_base, model=model - ) - - litellm_logging_obj.pre_call( - input=[{"role": "user", "content": json.dumps(request_body)}], - api_key="", - additional_args={ - "complete_input_dict": request_body, - "api_base": str(request_url), - "headers": headers, - }, - ) - - response = await async_httpx_client.post( - url=request_url, - headers=headers, - data=json.dumps(request_body), - stream=stream or False, - ) - response.raise_for_status() - - # used for logging + cost tracking - litellm_logging_obj.model_call_details["httpx_response"] = response - - if stream: - return await AnthropicMessagesHandler._handle_anthropic_streaming( - response=response, - request_body=request_body, - litellm_logging_obj=litellm_logging_obj, - ) - else: - return response.json() diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index e9b598f18da..5b5e2e6f36d 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -1,8 +1,18 @@ -from typing import Optional +from typing import Any, AsyncIterator, Dict, List, Optional +import httpx + +from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.base_llm.anthropic_messages.transformation import ( BaseAnthropicMessagesConfig, ) +from litellm.types.llms.anthropic import AnthropicMessagesRequest +from litellm.types.llms.anthropic_messages.anthropic_response import ( + AnthropicMessagesResponse, +) +from litellm.types.router import GenericLiteLLMParams + +from ...common_utils import AnthropicError DEFAULT_ANTHROPIC_API_BASE = "https://api.anthropic.com" DEFAULT_ANTHROPIC_API_VERSION = "2023-06-01" @@ -26,7 +36,15 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): # "metadata", ] - def get_complete_url(self, api_base: Optional[str], model: str) -> str: + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: api_base = api_base or DEFAULT_ANTHROPIC_API_BASE if not api_base.endswith("/v1/messages"): api_base = f"{api_base}/v1/messages" @@ -36,7 +54,11 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): self, headers: dict, model: str, + messages: List[Any], + optional_params: dict, + litellm_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: if "x-api-key" not in headers: headers["x-api-key"] = api_key @@ -45,3 +67,84 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): if "content-type" not in headers: headers["content-type"] = "application/json" return headers + + def transform_anthropic_messages_request( + self, + model: str, + messages: List[Dict], + anthropic_messages_optional_request_params: Dict, + litellm_params: GenericLiteLLMParams, + headers: dict, + ) -> Dict: + """ + No transformation is needed for Anthropic messages + + + This takes in a request in the Anthropic /v1/messages API spec -> transforms it to /v1/messages API spec (i.e) no transformation is needed + """ + max_tokens = anthropic_messages_optional_request_params.pop("max_tokens", None) + if max_tokens is None: + raise AnthropicError( + message="max_tokens is required for Anthropic /v1/messages API", + status_code=400, + ) + ####### get required params for all anthropic messages requests ###### + anthropic_messages_request: AnthropicMessagesRequest = AnthropicMessagesRequest( + messages=messages, + max_tokens=max_tokens, + model=model, + **anthropic_messages_optional_request_params, + ) + return dict(anthropic_messages_request) + + def transform_anthropic_messages_response( + self, + model: str, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + ) -> AnthropicMessagesResponse: + """ + No transformation is needed for Anthropic messages, since we want the response in the Anthropic /v1/messages API spec + """ + try: + raw_response_json = raw_response.json() + except Exception: + raise AnthropicError( + message=raw_response.text, status_code=raw_response.status_code + ) + return AnthropicMessagesResponse(**raw_response_json) + + def get_async_streaming_response_iterator( + self, + model: str, + httpx_response: httpx.Response, + request_body: dict, + litellm_logging_obj: LiteLLMLoggingObj, + ) -> AsyncIterator: + """Helper function to handle Anthropic streaming responses using the existing logging handlers""" + from datetime import datetime + + from litellm.proxy.pass_through_endpoints.streaming_handler import ( + PassThroughStreamingHandler, + ) + from litellm.proxy.pass_through_endpoints.success_handler import ( + PassThroughEndpointLogging, + ) + from litellm.types.passthrough_endpoints.pass_through_endpoints import ( + EndpointType, + ) + + # Create success handler object + passthrough_success_handler_obj = PassThroughEndpointLogging() + + # Use the existing streaming handler for Anthropic + start_time = datetime.now() + return PassThroughStreamingHandler.chunk_processor( + response=httpx_response, + request_body=request_body, + litellm_logging_obj=litellm_logging_obj, + endpoint_type=EndpointType.ANTHROPIC, + start_time=start_time, + passthrough_success_handler_obj=passthrough_success_handler_obj, + url_route="/v1/messages", + ) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/utils.py b/litellm/llms/anthropic/experimental_pass_through/messages/utils.py new file mode 100644 index 00000000000..29d00cd04cc --- /dev/null +++ b/litellm/llms/anthropic/experimental_pass_through/messages/utils.py @@ -0,0 +1,24 @@ +from typing import Any, Dict, cast, get_type_hints + +from litellm.types.llms.anthropic import AnthropicMessagesRequestOptionalParams + + +class AnthropicMessagesRequestUtils: + @staticmethod + def get_requested_anthropic_messages_optional_param( + params: Dict[str, Any], + ) -> AnthropicMessagesRequestOptionalParams: + """ + Filter parameters to only include those defined in AnthropicMessagesRequestOptionalParams. + + Args: + params: Dictionary of parameters to filter + + Returns: + AnthropicMessagesRequestOptionalParams instance with only the valid parameters + """ + valid_keys = get_type_hints(AnthropicMessagesRequestOptionalParams).keys() + filtered_params = { + k: v for k, v in params.items() if k in valid_keys and v is not None + } + return cast(AnthropicMessagesRequestOptionalParams, filtered_params) diff --git a/litellm/llms/azure/chat/gpt_transformation.py b/litellm/llms/azure/chat/gpt_transformation.py index 238566faf73..2ae684ddaeb 100644 --- a/litellm/llms/azure/chat/gpt_transformation.py +++ b/litellm/llms/azure/chat/gpt_transformation.py @@ -105,6 +105,7 @@ class AzureOpenAIConfig(BaseConfig): "prediction", "modalities", "audio", + "web_search_options", ] def _is_response_format_supported_model(self, model: str) -> bool: diff --git a/litellm/llms/azure/common_utils.py b/litellm/llms/azure/common_utils.py index 4ebd54e8fcb..3238b8e862e 100644 --- a/litellm/llms/azure/common_utils.py +++ b/litellm/llms/azure/common_utils.py @@ -272,6 +272,7 @@ class BaseAzureLLM(BaseOpenAILLM): ) -> Optional[Union[AzureOpenAI, AsyncAzureOpenAI]]: openai_client: Optional[Union[AzureOpenAI, AsyncAzureOpenAI]] = None client_initialization_params: dict = locals() + client_initialization_params["is_async"] = _is_async if client is None: cached_client = self.get_cached_openai_client( client_initialization_params=client_initialization_params, @@ -320,7 +321,7 @@ class BaseAzureLLM(BaseOpenAILLM): api_version: Optional[str], is_async: bool, ) -> dict: - azure_ad_token_provider: Optional[Callable[[], str]] = None + azure_ad_token_provider = litellm_params.get("azure_ad_token_provider") # If we have api_key, then we have higher priority azure_ad_token = litellm_params.get("azure_ad_token") tenant_id = litellm_params.get("tenant_id", os.getenv("AZURE_TENANT_ID")) @@ -336,7 +337,11 @@ class BaseAzureLLM(BaseOpenAILLM): ) max_retries = litellm_params.get("max_retries") timeout = litellm_params.get("timeout") - if not api_key and tenant_id and client_id and client_secret: + if ( + not api_key + and azure_ad_token_provider is None + and tenant_id and client_id and client_secret + ): verbose_logger.debug( "Using Azure AD Token Provider from Entra ID for Azure Auth" ) @@ -345,7 +350,7 @@ class BaseAzureLLM(BaseOpenAILLM): client_id=client_id, client_secret=client_secret, ) - if azure_username and azure_password and client_id: + if azure_ad_token_provider is None and azure_username and azure_password and client_id: verbose_logger.debug("Using Azure Username and Password for Azure Auth") azure_ad_token_provider = get_azure_ad_token_from_username_password( azure_username=azure_username, diff --git a/litellm/llms/azure/image_generation/__init__.py b/litellm/llms/azure/image_generation/__init__.py new file mode 100644 index 00000000000..fcdf49f2916 --- /dev/null +++ b/litellm/llms/azure/image_generation/__init__.py @@ -0,0 +1,29 @@ +from litellm._logging import verbose_logger +from litellm.llms.base_llm.image_generation.transformation import ( + BaseImageGenerationConfig, +) + +from .dall_e_2_transformation import AzureDallE2ImageGenerationConfig +from .dall_e_3_transformation import AzureDallE3ImageGenerationConfig +from .gpt_transformation import AzureGPTImageGenerationConfig + +__all__ = [ + "AzureDallE2ImageGenerationConfig", + "AzureDallE3ImageGenerationConfig", + "AzureGPTImageGenerationConfig", +] + + +def get_azure_image_generation_config(model: str) -> BaseImageGenerationConfig: + model = model.lower() + model = model.replace("-", "") + model = model.replace("_", "") + if model == "" or "dalle2" in model: # empty model is dall-e-2 + return AzureDallE2ImageGenerationConfig() + elif "dalle3" in model: + return AzureDallE3ImageGenerationConfig() + else: + verbose_logger.debug( + f"Using AzureGPTImageGenerationConfig for model: {model}. This follows the gpt-image-1 model format." + ) + return AzureGPTImageGenerationConfig() diff --git a/litellm/llms/azure/image_generation/dall_e_2_transformation.py b/litellm/llms/azure/image_generation/dall_e_2_transformation.py new file mode 100644 index 00000000000..3fe702f57f0 --- /dev/null +++ b/litellm/llms/azure/image_generation/dall_e_2_transformation.py @@ -0,0 +1,9 @@ +from litellm.llms.openai.image_generation import DallE2ImageGenerationConfig + + +class AzureDallE2ImageGenerationConfig(DallE2ImageGenerationConfig): + """ + Azure dall-e-2 image generation config + """ + + pass diff --git a/litellm/llms/azure/image_generation/dall_e_3_transformation.py b/litellm/llms/azure/image_generation/dall_e_3_transformation.py new file mode 100644 index 00000000000..5e0bfcd108f --- /dev/null +++ b/litellm/llms/azure/image_generation/dall_e_3_transformation.py @@ -0,0 +1,9 @@ +from litellm.llms.openai.image_generation import DallE3ImageGenerationConfig + + +class AzureDallE3ImageGenerationConfig(DallE3ImageGenerationConfig): + """ + Azure dall-e-3 image generation config + """ + + pass diff --git a/litellm/llms/azure/image_generation/gpt_transformation.py b/litellm/llms/azure/image_generation/gpt_transformation.py new file mode 100644 index 00000000000..1f5f65f693a --- /dev/null +++ b/litellm/llms/azure/image_generation/gpt_transformation.py @@ -0,0 +1,9 @@ +from litellm.llms.openai.image_generation import GPTImageGenerationConfig + + +class AzureGPTImageGenerationConfig(GPTImageGenerationConfig): + """ + Azure gpt-image-1 image generation config + """ + + pass diff --git a/litellm/llms/azure/realtime/handler.py b/litellm/llms/azure/realtime/handler.py index 5a4865e7d73..c5447b4ccd9 100644 --- a/litellm/llms/azure/realtime/handler.py +++ b/litellm/llms/azure/realtime/handler.py @@ -4,7 +4,7 @@ This file contains the calling Azure OpenAI's `/openai/realtime` endpoint. This requires websockets, and is currently only supported on LiteLLM Proxy. """ -from typing import Any, Optional +from typing import Any, Optional, cast from ....litellm_core_utils.litellm_logging import Logging as LiteLLMLogging from ....litellm_core_utils.realtime_streaming import RealTimeStreaming @@ -40,15 +40,16 @@ class AzureOpenAIRealtime(AzureChatCompletion): self, model: str, websocket: Any, + logging_obj: LiteLLMLogging, api_base: Optional[str] = None, api_key: Optional[str] = None, api_version: Optional[str] = None, azure_ad_token: Optional[str] = None, client: Optional[Any] = None, - logging_obj: Optional[LiteLLMLogging] = None, timeout: Optional[float] = None, ): import websockets + from websockets.asyncio.client import ClientConnection if api_base is None: raise ValueError("api_base is required for Azure OpenAI calls") @@ -65,7 +66,7 @@ class AzureOpenAIRealtime(AzureChatCompletion): }, ) as backend_ws: realtime_streaming = RealTimeStreaming( - websocket, backend_ws, logging_obj + websocket, cast(ClientConnection, backend_ws), logging_obj ) await realtime_streaming.bidirectional_forward() diff --git a/litellm/llms/base_llm/__init__.py b/litellm/llms/base_llm/__init__.py new file mode 100644 index 00000000000..cd682a0dbea --- /dev/null +++ b/litellm/llms/base_llm/__init__.py @@ -0,0 +1,13 @@ +from .anthropic_messages.transformation import BaseAnthropicMessagesConfig +from .audio_transcription.transformation import BaseAudioTranscriptionConfig +from .chat.transformation import BaseConfig +from .embedding.transformation import BaseEmbeddingConfig +from .image_generation.transformation import BaseImageGenerationConfig + +__all__ = [ + "BaseImageGenerationConfig", + "BaseConfig", + "BaseAudioTranscriptionConfig", + "BaseAnthropicMessagesConfig", + "BaseEmbeddingConfig", +] diff --git a/litellm/llms/base_llm/anthropic_messages/transformation.py b/litellm/llms/base_llm/anthropic_messages/transformation.py index 7619ffbbf6c..710a1076887 100644 --- a/litellm/llms/base_llm/anthropic_messages/transformation.py +++ b/litellm/llms/base_llm/anthropic_messages/transformation.py @@ -1,5 +1,12 @@ from abc import ABC, abstractmethod -from typing import TYPE_CHECKING, Any, Optional +from typing import TYPE_CHECKING, Any, AsyncIterator, Dict, List, Optional, Tuple + +import httpx + +from litellm.types.llms.anthropic_messages.anthropic_response import ( + AnthropicMessagesResponse, +) +from litellm.types.router import GenericLiteLLMParams if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj @@ -15,12 +22,29 @@ class BaseAnthropicMessagesConfig(ABC): self, headers: dict, model: str, + messages: List[Any], + optional_params: dict, + litellm_params: dict, api_key: Optional[str] = None, + api_base: Optional[str] = None, ) -> dict: - pass + """ + OPTIONAL + + Validate the environment for the request + """ + return headers @abstractmethod - def get_complete_url(self, api_base: Optional[str], model: str) -> str: + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: """ OPTIONAL @@ -33,3 +57,51 @@ class BaseAnthropicMessagesConfig(ABC): @abstractmethod def get_supported_anthropic_messages_params(self, model: str) -> list: pass + + @abstractmethod + def transform_anthropic_messages_request( + self, + model: str, + messages: List[Dict], + anthropic_messages_optional_request_params: Dict, + litellm_params: GenericLiteLLMParams, + headers: dict, + ) -> Dict: + pass + + @abstractmethod + def transform_anthropic_messages_response( + self, + model: str, + raw_response: httpx.Response, + logging_obj: LiteLLMLoggingObj, + ) -> AnthropicMessagesResponse: + pass + + def sign_request( + self, + headers: dict, + optional_params: dict, + request_data: dict, + api_base: str, + model: Optional[str] = None, + stream: Optional[bool] = None, + fake_stream: Optional[bool] = None, + ) -> Tuple[dict, Optional[bytes]]: + """ + OPTIONAL + + Sign the request, providers like Bedrock need to sign the request before sending it to the API + + For all other providers, this is a no-op and we just return the headers + """ + return headers, None + + def get_async_streaming_response_iterator( + self, + model: str, + httpx_response: httpx.Response, + request_body: dict, + litellm_logging_obj: LiteLLMLoggingObj, + ) -> AsyncIterator: + raise NotImplementedError("Subclasses must implement this method") diff --git a/litellm/llms/base_llm/base_model_iterator.py b/litellm/llms/base_llm/base_model_iterator.py index 4cf757d6cd8..9f293905d72 100644 --- a/litellm/llms/base_llm/base_model_iterator.py +++ b/litellm/llms/base_llm/base_model_iterator.py @@ -41,13 +41,13 @@ class BaseModelResponseIterator: self, str_line: str ) -> Union[GenericStreamingChunk, ModelResponseStream]: # chunk is a str at this point - + stripped_json_chunk: Optional[dict] = None stripped_chunk = litellm.CustomStreamWrapper._strip_sse_data_from_chunk( str_line ) try: if stripped_chunk is not None: - stripped_json_chunk: Optional[dict] = json.loads(stripped_chunk) + stripped_json_chunk = json.loads(stripped_chunk) else: stripped_json_chunk = None except json.JSONDecodeError: diff --git a/litellm/llms/base_llm/chat/transformation.py b/litellm/llms/base_llm/chat/transformation.py index fa278c805eb..26faa4a5b89 100644 --- a/litellm/llms/base_llm/chat/transformation.py +++ b/litellm/llms/base_llm/chat/transformation.py @@ -11,6 +11,7 @@ from typing import ( Iterator, List, Optional, + Tuple, Type, Union, cast, @@ -277,7 +278,7 @@ class BaseConfig(ABC): model: Optional[str] = None, stream: Optional[bool] = None, fake_stream: Optional[bool] = None, - ) -> dict: + ) -> Tuple[dict, Optional[bytes]]: """ Some providers like Bedrock require signing the request. The sign request funtion needs access to `request_data` and `complete_url` Args: @@ -290,7 +291,7 @@ class BaseConfig(ABC): Update the headers with the signed headers in this function. The return values will be sent as headers in the http request. """ - return headers + return headers, None def get_complete_url( self, @@ -323,6 +324,27 @@ class BaseConfig(ABC): ) -> dict: pass + async def async_transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + """ + Override to allow for http requests on async calls - e.g. converting url to base64 + + Currently only used by openai.py + """ + return self.transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + headers=headers, + ) + @abstractmethod def transform_response( self, @@ -354,7 +376,7 @@ class BaseConfig(ABC): ) -> Any: pass - def get_async_custom_stream_wrapper( + async def get_async_custom_stream_wrapper( self, model: str, custom_llm_provider: str, @@ -365,6 +387,7 @@ class BaseConfig(ABC): messages: list, client: Optional[AsyncHTTPHandler] = None, json_mode: Optional[bool] = None, + signed_json_body: Optional[bytes] = None, ) -> CustomStreamWrapper: raise NotImplementedError @@ -379,6 +402,7 @@ class BaseConfig(ABC): messages: list, client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, json_mode: Optional[bool] = None, + signed_json_body: Optional[bytes] = None, ) -> CustomStreamWrapper: raise NotImplementedError diff --git a/litellm/llms/base_llm/files/transformation.py b/litellm/llms/base_llm/files/transformation.py index 9925004c896..4d749af21e1 100644 --- a/litellm/llms/base_llm/files/transformation.py +++ b/litellm/llms/base_llm/files/transformation.py @@ -1,5 +1,5 @@ -from abc import abstractmethod -from typing import TYPE_CHECKING, Any, List, Optional, Union +from abc import ABC, abstractmethod +from typing import TYPE_CHECKING, Any, Dict, List, Optional, Union import httpx @@ -8,6 +8,7 @@ from litellm.types.llms.openai import ( CreateFileRequest, OpenAICreateFileRequestOptionalParams, OpenAIFileObject, + OpenAIFilesPurpose, ) from litellm.types.utils import LlmProviders, ModelResponse @@ -15,10 +16,15 @@ from ..chat.transformation import BaseConfig if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + from litellm.router import Router as _Router LiteLLMLoggingObj = _LiteLLMLoggingObj + Span = Any + Router = _Router else: LiteLLMLoggingObj = Any + Span = Any + Router = Any class BaseFilesConfig(BaseConfig): @@ -99,3 +105,52 @@ class BaseFilesConfig(BaseConfig): raise NotImplementedError( "AudioTranscriptionConfig does not need a response transformation for audio transcription models" ) + + +class BaseFileEndpoints(ABC): + @abstractmethod + async def acreate_file( + self, + create_file_request: CreateFileRequest, + llm_router: Router, + target_model_names_list: List[str], + litellm_parent_otel_span: Span, + ) -> OpenAIFileObject: + pass + + @abstractmethod + async def afile_retrieve( + self, + file_id: str, + litellm_parent_otel_span: Optional[Span], + ) -> OpenAIFileObject: + pass + + @abstractmethod + async def afile_list( + self, + purpose: Optional[OpenAIFilesPurpose], + litellm_parent_otel_span: Optional[Span], + **data: Dict, + ) -> List[OpenAIFileObject]: + pass + + @abstractmethod + async def afile_delete( + self, + file_id: str, + litellm_parent_otel_span: Optional[Span], + llm_router: Router, + **data: Dict, + ) -> OpenAIFileObject: + pass + + @abstractmethod + async def afile_content( + self, + file_id: str, + litellm_parent_otel_span: Optional[Span], + llm_router: Router, + **data: Dict, + ) -> str: + pass diff --git a/litellm/llms/base_llm/image_generation/transformation.py b/litellm/llms/base_llm/image_generation/transformation.py new file mode 100644 index 00000000000..134c95b1c8e --- /dev/null +++ b/litellm/llms/base_llm/image_generation/transformation.py @@ -0,0 +1,95 @@ +from abc import ABC, abstractmethod +from typing import TYPE_CHECKING, Any, List, Optional, Union + +import httpx + +from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException +from litellm.types.llms.openai import ( + AllMessageValues, + OpenAIImageGenerationOptionalParams, +) +from litellm.types.utils import ModelResponse + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + + LiteLLMLoggingObj = _LiteLLMLoggingObj +else: + LiteLLMLoggingObj = Any + + +class BaseImageGenerationConfig(BaseConfig, ABC): + @abstractmethod + def get_supported_openai_params( + self, model: str + ) -> List[OpenAIImageGenerationOptionalParams]: + pass + + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: + """ + OPTIONAL + + Get the complete url for the request + + Some providers need `model` in `api_base` + """ + return api_base or "" + + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + return {} + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] + ) -> BaseLLMException: + raise BaseLLMException( + status_code=status_code, + message=error_message, + headers=headers, + ) + + def transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + raise NotImplementedError( + "ImageVariationConfig implementa 'transform_request_image_variation' for image variation models" + ) + + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + json_mode: Optional[bool] = None, + ) -> ModelResponse: + raise NotImplementedError( + "ImageVariationConfig implements 'transform_response_image_variation' for image variation models" + ) diff --git a/litellm/llms/base_llm/realtime/transformation.py b/litellm/llms/base_llm/realtime/transformation.py new file mode 100644 index 00000000000..d5531a532b9 --- /dev/null +++ b/litellm/llms/base_llm/realtime/transformation.py @@ -0,0 +1,83 @@ +from abc import ABC, abstractmethod +from typing import TYPE_CHECKING, Any, List, Optional, Union + +import httpx + +from litellm.types.realtime import ( + RealtimeResponseTransformInput, + RealtimeResponseTypedDict, +) + +from ..chat.transformation import BaseLLMException + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + + LiteLLMLoggingObj = _LiteLLMLoggingObj +else: + LiteLLMLoggingObj = Any + + +class BaseRealtimeConfig(ABC): + @abstractmethod + def validate_environment( + self, + headers: dict, + model: str, + api_key: Optional[str] = None, + ) -> dict: + pass + + @abstractmethod + def get_complete_url( + self, api_base: Optional[str], model: str, api_key: Optional[str] = None + ) -> str: + """ + OPTIONAL + + Get the complete url for the request + + Some providers need `model` in `api_base` + """ + return api_base or "" + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] + ) -> BaseLLMException: + raise BaseLLMException( + status_code=status_code, + message=error_message, + headers=headers, + ) + + @abstractmethod + def transform_realtime_request( + self, + message: str, + model: str, + session_configuration_request: Optional[str] = None, + ) -> List[str]: + pass + + def requires_session_configuration( + self, + ) -> bool: # initial configuration message sent to setup the realtime session + return False + + def session_configuration_request( + self, model: str + ) -> Optional[str]: # message sent to setup the realtime session + return None + + @abstractmethod + def transform_realtime_response( + self, + message: Union[str, bytes], + model: str, + logging_obj: LiteLLMLoggingObj, + realtime_response_transform_input: RealtimeResponseTransformInput, + ) -> RealtimeResponseTypedDict: # message sent to setup the realtime session + """ + Keep this state less - leave the state management (e.g. tracking current_output_item_id, current_response_id, current_conversation_id, current_delta_chunks) to the caller. + """ + pass diff --git a/litellm/llms/bedrock/base_aws_llm.py b/litellm/llms/bedrock/base_aws_llm.py index 133ef6a9524..a2832e69eb5 100644 --- a/litellm/llms/bedrock/base_aws_llm.py +++ b/litellm/llms/bedrock/base_aws_llm.py @@ -2,7 +2,17 @@ import hashlib import json import os from datetime import datetime -from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, cast, get_args +from typing import ( + TYPE_CHECKING, + Any, + Dict, + List, + Literal, + Optional, + Tuple, + cast, + get_args, +) import httpx from pydantic import BaseModel @@ -625,3 +635,74 @@ class BaseAWSLLM: prepped = request.prepare() return prepped + + def _sign_request( + self, + service_name: Literal["bedrock", "sagemaker"], + headers: dict, + optional_params: dict, + request_data: dict, + api_base: str, + model: Optional[str] = None, + stream: Optional[bool] = None, + fake_stream: Optional[bool] = None, + ) -> Tuple[dict, Optional[bytes]]: + """ + Sign a request for Bedrock or Sagemaker + + Returns: + Tuple[dict, Optional[str]]: A tuple containing the headers and the json str body of the request + """ + try: + from botocore.auth import SigV4Auth + from botocore.awsrequest import AWSRequest + from botocore.credentials import Credentials + except ImportError: + raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.") + + ## CREDENTIALS ## + # pop aws_secret_access_key, aws_access_key_id, aws_session_token, aws_region_name from kwargs, since completion calls fail with them + aws_secret_access_key = optional_params.get("aws_secret_access_key", None) + aws_access_key_id = optional_params.get("aws_access_key_id", None) + aws_session_token = optional_params.get("aws_session_token", None) + aws_role_name = optional_params.get("aws_role_name", None) + aws_session_name = optional_params.get("aws_session_name", None) + aws_profile_name = optional_params.get("aws_profile_name", None) + aws_web_identity_token = optional_params.get("aws_web_identity_token", None) + aws_sts_endpoint = optional_params.get("aws_sts_endpoint", None) + aws_region_name = self._get_aws_region_name( + optional_params=optional_params, model=model + ) + + credentials: Credentials = self.get_credentials( + aws_access_key_id=aws_access_key_id, + aws_secret_access_key=aws_secret_access_key, + aws_session_token=aws_session_token, + aws_region_name=aws_region_name, + aws_session_name=aws_session_name, + aws_profile_name=aws_profile_name, + aws_role_name=aws_role_name, + aws_web_identity_token=aws_web_identity_token, + aws_sts_endpoint=aws_sts_endpoint, + ) + + sigv4 = SigV4Auth(credentials, service_name, aws_region_name) + if headers is not None: + headers = {"Content-Type": "application/json", **headers} + else: + headers = {"Content-Type": "application/json"} + + request = AWSRequest( + method="POST", + url=api_base, + data=json.dumps(request_data), + headers=headers, + ) + sigv4.add_auth(request) + + request_headers_dict = dict(request.headers) + if ( + headers is not None and "Authorization" in headers + ): # prevent sigv4 from overwriting the auth header + request_headers_dict["Authorization"] = headers["Authorization"] + return request_headers_dict, request.body diff --git a/litellm/llms/bedrock/chat/converse_transformation.py b/litellm/llms/bedrock/chat/converse_transformation.py index 8332463c5c8..7dad73d871d 100644 --- a/litellm/llms/bedrock/chat/converse_transformation.py +++ b/litellm/llms/bedrock/chat/converse_transformation.py @@ -12,6 +12,9 @@ import httpx import litellm from litellm.litellm_core_utils.core_helpers import map_finish_reason from litellm.litellm_core_utils.litellm_logging import Logging +from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import ( + _parse_content_for_reasoning, +) from litellm.litellm_core_utils.prompt_templates.factory import ( BedrockConverseMessagesProcessor, _bedrock_converse_messages_pt, @@ -34,7 +37,14 @@ from litellm.types.llms.openai import ( OpenAIChatCompletionToolParam, OpenAIMessageContentListBlock, ) -from litellm.types.utils import ModelResponse, PromptTokensDetailsWrapper, Usage +from litellm.types.utils import ( + ChatCompletionMessageToolCall, + Function, + Message, + ModelResponse, + PromptTokensDetailsWrapper, + Usage, +) from litellm.utils import add_dummy_tool, has_tool_call_blocks from ..common_utils import BedrockError, BedrockModelInfo, get_bedrock_tool_name @@ -690,6 +700,132 @@ class AmazonConverseConfig(BaseConfig): ) return openai_usage + def get_tool_call_names( + self, + tools: Optional[ + Union[List[ToolBlock], List[OpenAIChatCompletionToolParam]] + ] = None, + ) -> List[str]: + if tools is None: + return [] + tool_set: set[str] = set() + for tool in tools: + tool_spec = tool.get("toolSpec") + function = tool.get("function") + if tool_spec is not None: + _name = cast(dict, tool_spec).get("name") + if _name is not None and isinstance(_name, str): + tool_set.add(_name) + if function is not None: + _name = cast(dict, function).get("name") + if _name is not None and isinstance(_name, str): + tool_set.add(_name) + return list(tool_set) + + def apply_tool_call_transformation_if_needed( + self, + message: Message, + tools: Optional[List[ToolBlock]] = None, + initial_finish_reason: Optional[str] = None, + ) -> Tuple[Message, Optional[str]]: + """ + Apply tool call transformation to a message. + + LLM providers (e.g. Bedrock, Vertex AI) sometimes return tool call in the response content. + + If the response content is a JSON object, we can parse it and return the tool call in the tool_calls field. + """ + returned_finish_reason = initial_finish_reason + if tools is None: + return message, returned_finish_reason + + if message.content is not None: + try: + tool_call_names = self.get_tool_call_names(tools) + json_content = json.loads(message.content) + if ( + json_content.get("type") == "function" + and json_content.get("name") in tool_call_names + ): + tool_calls = [ + ChatCompletionMessageToolCall(function=Function(**json_content)) + ] + + message.tool_calls = tool_calls + message.content = None + returned_finish_reason = "tool_calls" + except Exception: + pass + + return message, returned_finish_reason + + def _translate_message_content( + self, content_blocks: List[ContentBlock] + ) -> Tuple[ + str, + List[ChatCompletionToolCallChunk], + Optional[List[BedrockConverseReasoningContentBlock]], + ]: + """ + Translate the message content to a string and a list of tool calls and reasoning content blocks + + Returns: + content_str: str + tools: List[ChatCompletionToolCallChunk] + reasoningContentBlocks: Optional[List[BedrockConverseReasoningContentBlock]] + """ + content_str = "" + tools: List[ChatCompletionToolCallChunk] = [] + reasoningContentBlocks: Optional[ + List[BedrockConverseReasoningContentBlock] + ] = None + for idx, content in enumerate(content_blocks): + """ + - Content is either a tool response or text + """ + extracted_reasoning_content_str: Optional[str] = None + if "text" in content: + ( + extracted_reasoning_content_str, + _content_str, + ) = _parse_content_for_reasoning(content["text"]) + if _content_str is not None: + content_str += _content_str + if "toolUse" in content: + ## check tool name was formatted by litellm + _response_tool_name = content["toolUse"]["name"] + response_tool_name = get_bedrock_tool_name( + response_tool_name=_response_tool_name + ) + _function_chunk = ChatCompletionToolCallFunctionChunk( + name=response_tool_name, + arguments=json.dumps(content["toolUse"]["input"]), + ) + + _tool_response_chunk = ChatCompletionToolCallChunk( + id=content["toolUse"]["toolUseId"], + type="function", + function=_function_chunk, + index=idx, + ) + tools.append(_tool_response_chunk) + if extracted_reasoning_content_str is not None: + if reasoningContentBlocks is None: + reasoningContentBlocks = [] + reasoningContentBlocks.append( + BedrockConverseReasoningContentBlock( + reasoningText=BedrockConverseReasoningTextBlock( + text=extracted_reasoning_content_str, + ) + ) + ) + if "reasoningContent" in content: + if reasoningContentBlocks is None: + reasoningContentBlocks = [] + reasoningContentBlocks.append(content["reasoningContent"]) + + return content_str, tools, reasoningContentBlocks + def _transform_response( self, model: str, @@ -768,34 +904,11 @@ class AmazonConverseConfig(BaseConfig): ] = None if message is not None: - for idx, content in enumerate(message["content"]): - """ - - Content is either a tool response or text - """ - if "text" in content: - content_str += content["text"] - if "toolUse" in content: - ## check tool name was formatted by litellm - _response_tool_name = content["toolUse"]["name"] - response_tool_name = get_bedrock_tool_name( - response_tool_name=_response_tool_name - ) - _function_chunk = ChatCompletionToolCallFunctionChunk( - name=response_tool_name, - arguments=json.dumps(content["toolUse"]["input"]), - ) - - _tool_response_chunk = ChatCompletionToolCallChunk( - id=content["toolUse"]["toolUseId"], - type="function", - function=_function_chunk, - index=idx, - ) - tools.append(_tool_response_chunk) - if "reasoningContent" in content: - if reasoningContentBlocks is None: - reasoningContentBlocks = [] - reasoningContentBlocks.append(content["reasoningContent"]) + ( + content_str, + tools, + reasoningContentBlocks, + ) = self._translate_message_content(message["content"]) if reasoningContentBlocks is not None: chat_completion_message["provider_specific_fields"] = { @@ -819,11 +932,23 @@ class AmazonConverseConfig(BaseConfig): ## CALCULATING USAGE - bedrock returns usage in the headers usage = self._transform_usage(completion_response["usage"]) + ## HANDLE TOOL CALLS + _message = Message(**chat_completion_message) + initial_finish_reason = map_finish_reason(completion_response["stopReason"]) + + ( + returned_message, + returned_finish_reason, + ) = self.apply_tool_call_transformation_if_needed( + message=_message, + tools=optional_params.get("tools"), + initial_finish_reason=initial_finish_reason, + ) model_response.choices = [ litellm.Choices( - finish_reason=map_finish_reason(completion_response["stopReason"]), + finish_reason=returned_finish_reason, index=0, - message=litellm.Message(**chat_completion_message), + message=returned_message, ) ] model_response.created = int(time.time()) diff --git a/litellm/llms/bedrock/chat/invoke_handler.py b/litellm/llms/bedrock/chat/invoke_handler.py index dfd16585434..2c3cf59585c 100644 --- a/litellm/llms/bedrock/chat/invoke_handler.py +++ b/litellm/llms/bedrock/chat/invoke_handler.py @@ -272,6 +272,7 @@ def make_sync_call( api_base: str, headers: dict, data: str, + signed_json_body: Optional[bytes], model: str, messages: list, logging_obj: Logging, @@ -286,7 +287,7 @@ def make_sync_call( response = client.post( api_base, headers=headers, - data=data, + data=signed_json_body if signed_json_body is not None else data, stream=not fake_stream, logging_obj=logging_obj, ) @@ -1413,7 +1414,9 @@ class AWSEventStreamDecoder: except Exception as e: raise Exception("Received streaming error - {}".format(str(e))) - def _chunk_parser(self, chunk_data: dict) -> Union[GChunk, ModelResponseStream]: + def _chunk_parser( + self, chunk_data: dict + ) -> Union[GChunk, ModelResponseStream, dict]: text = "" is_finished = False finish_reason = "" @@ -1473,7 +1476,7 @@ class AWSEventStreamDecoder: def iter_bytes( self, iterator: Iterator[bytes] - ) -> Iterator[Union[GChunk, ModelResponseStream]]: + ) -> Iterator[Union[GChunk, ModelResponseStream, dict]]: """Given an iterator that yields lines, iterate over it & yield every event encountered""" from botocore.eventstream import EventStreamBuffer @@ -1489,7 +1492,7 @@ class AWSEventStreamDecoder: async def aiter_bytes( self, iterator: AsyncIterator[bytes] - ) -> AsyncIterator[Union[GChunk, ModelResponseStream]]: + ) -> AsyncIterator[Union[GChunk, ModelResponseStream, dict]]: """Given an async iterator that yields lines, iterate over it & yield every event encountered""" from botocore.eventstream import EventStreamBuffer @@ -1576,7 +1579,9 @@ class AmazonDeepSeekR1StreamDecoder(AWSEventStreamDecoder): sync_stream=sync_stream, ) - def _chunk_parser(self, chunk_data: dict) -> Union[GChunk, ModelResponseStream]: + def _chunk_parser( + self, chunk_data: dict + ) -> Union[GChunk, ModelResponseStream, dict]: return self.deepseek_model_response_iterator.chunk_parser(chunk=chunk_data) diff --git a/litellm/llms/bedrock/chat/invoke_transformations/amazon_mistral_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/amazon_mistral_transformation.py index ef3c237f9d0..58dfa17a722 100644 --- a/litellm/llms/bedrock/chat/invoke_transformations/amazon_mistral_transformation.py +++ b/litellm/llms/bedrock/chat/invoke_transformations/amazon_mistral_transformation.py @@ -1,10 +1,14 @@ import types -from typing import List, Optional +from typing import List, Optional, TYPE_CHECKING from litellm.llms.base_llm.chat.transformation import BaseConfig from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation import ( AmazonInvokeConfig, ) +from litellm.llms.bedrock.common_utils import BedrockError + +if TYPE_CHECKING: + from litellm.types.utils import ModelResponse class AmazonMistralConfig(AmazonInvokeConfig, BaseConfig): @@ -81,3 +85,27 @@ class AmazonMistralConfig(AmazonInvokeConfig, BaseConfig): if k == "stream": optional_params["stream"] = v return optional_params + + @staticmethod + def get_outputText(completion_response: dict, model_response: "ModelResponse") -> str: + """This function extracts the output text from a bedrock mistral completion. + As a side effect, it updates the finish reason for a model response. + + Args: + completion_response: JSON from the completion. + model_response: ModelResponse + + Returns: + A string with the response of the LLM + + """ + if "choices" in completion_response: + outputText = completion_response["choices"][0]["message"]["content"] + model_response.choices[0].finish_reason = completion_response["choices"][0]["finish_reason"] + elif "outputs" in completion_response: + outputText = completion_response["outputs"][0]["text"] + model_response.choices[0].finish_reason = completion_response["outputs"][0]["stop_reason"] + else: + raise BedrockError(message="Unexpected mistral completion response", status_code=400) + + return outputText diff --git a/litellm/llms/bedrock/chat/invoke_transformations/base_invoke_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/base_invoke_transformation.py index 67194e83e74..4c977af2fd3 100644 --- a/litellm/llms/bedrock/chat/invoke_transformations/base_invoke_transformation.py +++ b/litellm/llms/bedrock/chat/invoke_transformations/base_invoke_transformation.py @@ -121,60 +121,17 @@ class AmazonInvokeConfig(BaseConfig, BaseAWSLLM): model: Optional[str] = None, stream: Optional[bool] = None, fake_stream: Optional[bool] = None, - ) -> dict: - try: - from botocore.auth import SigV4Auth - from botocore.awsrequest import AWSRequest - from botocore.credentials import Credentials - except ImportError: - raise ImportError("Missing boto3 to call bedrock. Run 'pip install boto3'.") - - ## CREDENTIALS ## - # pop aws_secret_access_key, aws_access_key_id, aws_session_token, aws_region_name from kwargs, since completion calls fail with them - aws_secret_access_key = optional_params.get("aws_secret_access_key", None) - aws_access_key_id = optional_params.get("aws_access_key_id", None) - aws_session_token = optional_params.get("aws_session_token", None) - aws_role_name = optional_params.get("aws_role_name", None) - aws_session_name = optional_params.get("aws_session_name", None) - aws_profile_name = optional_params.get("aws_profile_name", None) - aws_web_identity_token = optional_params.get("aws_web_identity_token", None) - aws_sts_endpoint = optional_params.get("aws_sts_endpoint", None) - aws_region_name = self._get_aws_region_name( - optional_params=optional_params, model=model - ) - - credentials: Credentials = self.get_credentials( - aws_access_key_id=aws_access_key_id, - aws_secret_access_key=aws_secret_access_key, - aws_session_token=aws_session_token, - aws_region_name=aws_region_name, - aws_session_name=aws_session_name, - aws_profile_name=aws_profile_name, - aws_role_name=aws_role_name, - aws_web_identity_token=aws_web_identity_token, - aws_sts_endpoint=aws_sts_endpoint, - ) - - sigv4 = SigV4Auth(credentials, "bedrock", aws_region_name) - if headers is not None: - headers = {"Content-Type": "application/json", **headers} - else: - headers = {"Content-Type": "application/json"} - - request = AWSRequest( - method="POST", - url=api_base, - data=json.dumps(request_data), + ) -> Tuple[dict, Optional[bytes]]: + return self._sign_request( + service_name="bedrock", headers=headers, + optional_params=optional_params, + request_data=request_data, + api_base=api_base, + model=model, + stream=stream, + fake_stream=fake_stream, ) - sigv4.add_auth(request) - - request_headers_dict = dict(request.headers) - if ( - headers is not None and "Authorization" in headers - ): # prevent sigv4 from overwriting the auth header - request_headers_dict["Authorization"] = headers["Authorization"] - return request_headers_dict def transform_request( self, @@ -366,10 +323,7 @@ class AmazonInvokeConfig(BaseConfig, BaseAWSLLM): elif provider == "meta" or provider == "llama" or provider == "deepseek_r1": outputText = completion_response["generation"] elif provider == "mistral": - outputText = completion_response["outputs"][0]["text"] - model_response.choices[0].finish_reason = completion_response[ - "outputs" - ][0]["stop_reason"] + outputText = litellm.AmazonMistralConfig.get_outputText(completion_response, model_response) else: # amazon titan outputText = completion_response.get("results")[0].get("outputText") except Exception as e: @@ -454,7 +408,7 @@ class AmazonInvokeConfig(BaseConfig, BaseAWSLLM): return BedrockError(status_code=status_code, message=error_message) @track_llm_api_timing() - def get_async_custom_stream_wrapper( + async def get_async_custom_stream_wrapper( self, model: str, custom_llm_provider: str, @@ -465,6 +419,7 @@ class AmazonInvokeConfig(BaseConfig, BaseAWSLLM): messages: list, client: Optional[AsyncHTTPHandler] = None, json_mode: Optional[bool] = None, + signed_json_body: Optional[bytes] = None, ) -> CustomStreamWrapper: streaming_response = CustomStreamWrapper( completion_stream=None, @@ -499,6 +454,7 @@ class AmazonInvokeConfig(BaseConfig, BaseAWSLLM): messages: list, client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, json_mode: Optional[bool] = None, + signed_json_body: Optional[bytes] = None, ) -> CustomStreamWrapper: if client is None or isinstance(client, AsyncHTTPHandler): client = _get_httpx_client(params={}) @@ -510,6 +466,7 @@ class AmazonInvokeConfig(BaseConfig, BaseAWSLLM): api_base=api_base, headers=headers, data=json.dumps(data), + signed_json_body=signed_json_body, model=model, messages=messages, logging_obj=logging_obj, diff --git a/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py new file mode 100644 index 00000000000..ff475a95db0 --- /dev/null +++ b/litellm/llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py @@ -0,0 +1,166 @@ +from typing import TYPE_CHECKING, Any, AsyncIterator, Dict, List, Optional, Tuple, Union + +import httpx + +from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( + AnthropicMessagesConfig, +) +from litellm.llms.base_llm.anthropic_messages.transformation import ( + BaseAnthropicMessagesConfig, +) +from litellm.llms.bedrock.chat.invoke_handler import AWSEventStreamDecoder +from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation import ( + AmazonInvokeConfig, +) +from litellm.types.router import GenericLiteLLMParams +from litellm.types.utils import GenericStreamingChunk as GChunk +from litellm.types.utils import ModelResponseStream + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + + LiteLLMLoggingObj = _LiteLLMLoggingObj +else: + LiteLLMLoggingObj = Any + + +class AmazonAnthropicClaude3MessagesConfig( + AnthropicMessagesConfig, + AmazonInvokeConfig, +): + """ + Call Claude model family in the /v1/messages API spec + """ + + DEFAULT_BEDROCK_ANTHROPIC_API_VERSION = "bedrock-2023-05-31" + + def __init__(self, **kwargs): + BaseAnthropicMessagesConfig.__init__(self, **kwargs) + AmazonInvokeConfig.__init__(self, **kwargs) + + def sign_request( + self, + headers: dict, + optional_params: dict, + request_data: dict, + api_base: str, + model: Optional[str] = None, + stream: Optional[bool] = None, + fake_stream: Optional[bool] = None, + ) -> Tuple[dict, Optional[bytes]]: + return AmazonInvokeConfig.sign_request( + self=self, + headers=headers, + optional_params=optional_params, + request_data=request_data, + api_base=api_base, + model=model, + stream=stream, + fake_stream=fake_stream, + ) + + def validate_environment( + self, + headers: dict, + model: str, + messages: List[Any], + optional_params: dict, + litellm_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + return headers + + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: + return AmazonInvokeConfig.get_complete_url( + self=self, + api_base=api_base, + api_key=api_key, + model=model, + optional_params=optional_params, + litellm_params=litellm_params, + stream=stream, + ) + + def transform_anthropic_messages_request( + self, + model: str, + messages: List[Dict], + anthropic_messages_optional_request_params: Dict, + litellm_params: GenericLiteLLMParams, + headers: dict, + ) -> Dict: + anthropic_messages_request = AnthropicMessagesConfig.transform_anthropic_messages_request( + self=self, + model=model, + messages=messages, + anthropic_messages_optional_request_params=anthropic_messages_optional_request_params, + litellm_params=litellm_params, + headers=headers, + ) + + ######################################################### + ############## BEDROCK Invoke SPECIFIC TRANSFORMATION ### + ######################################################### + + # 1. anthropic_version is required for all claude models + if "anthropic_version" not in anthropic_messages_request: + anthropic_messages_request[ + "anthropic_version" + ] = self.DEFAULT_BEDROCK_ANTHROPIC_API_VERSION + + # 2. `stream` is not allowed in request body for bedrock invoke + if "stream" in anthropic_messages_request: + anthropic_messages_request.pop("stream", None) + + # 3. `model` is not allowed in request body for bedrock invoke + if "model" in anthropic_messages_request: + anthropic_messages_request.pop("model", None) + return anthropic_messages_request + + def get_async_streaming_response_iterator( + self, + model: str, + httpx_response: httpx.Response, + request_body: dict, + litellm_logging_obj: LiteLLMLoggingObj, + ) -> AsyncIterator: + aws_decoder = AmazonAnthropicClaudeMessagesStreamDecoder( + model=model, + ) + completion_stream = aws_decoder.aiter_bytes( + httpx_response.aiter_bytes(chunk_size=aws_decoder.DEFAULT_CHUNK_SIZE) + ) + return completion_stream + + +class AmazonAnthropicClaudeMessagesStreamDecoder(AWSEventStreamDecoder): + def __init__( + self, + model: str, + ) -> None: + """ + Iterator to return Bedrock invoke response in anthropic /messages format + """ + super().__init__(model=model) + self.DEFAULT_CHUNK_SIZE = 1024 + + def _chunk_parser( + self, chunk_data: dict + ) -> Union[GChunk, ModelResponseStream, dict]: + """ + Parse the chunk data into anthropic /messages format + + No transformation is needed for anthropic /messages format + + since bedrock invoke returns the response in the correct format + """ + return chunk_data diff --git a/litellm/llms/bedrock/messages/readme.md b/litellm/llms/bedrock/messages/readme.md new file mode 100644 index 00000000000..5d8d386accb --- /dev/null +++ b/litellm/llms/bedrock/messages/readme.md @@ -0,0 +1,3 @@ +# /v1/messages + +This folder contains transformation logic for calling bedrock models in the Anthropic /v1/messages API spec. \ No newline at end of file diff --git a/litellm/llms/cohere/embed/handler.py b/litellm/llms/cohere/embed/handler.py index 7a25bf7e541..41b81279723 100644 --- a/litellm/llms/cohere/embed/handler.py +++ b/litellm/llms/cohere/embed/handler.py @@ -1,3 +1,7 @@ +""" +Legacy /v1/embedding handler for Bedrock Cohere. +""" + import json from typing import Any, Callable, Optional, Union @@ -13,7 +17,7 @@ from litellm.llms.custom_httpx.http_handler import ( from litellm.types.llms.bedrock import CohereEmbeddingRequest from litellm.types.utils import EmbeddingResponse -from .transformation import CohereEmbeddingConfig +from .v1_transformation import CohereEmbeddingConfig def validate_environment(api_key, headers: dict): diff --git a/litellm/llms/cohere/embed/transformation.py b/litellm/llms/cohere/embed/transformation.py index 837dd5e006e..03d7edd1262 100644 --- a/litellm/llms/cohere/embed/transformation.py +++ b/litellm/llms/cohere/embed/transformation.py @@ -10,21 +10,27 @@ Convers Docs - https://docs.cohere.com/v2/reference/embed """ -from typing import Any, List, Optional, Union +from typing import Any, List, Optional, Union, cast import httpx +import litellm from litellm import COHERE_DEFAULT_EMBEDDING_INPUT_TYPE from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj +from litellm.llms.base_llm import BaseEmbeddingConfig +from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.types.llms.bedrock import ( CohereEmbeddingRequest, CohereEmbeddingRequestWithModel, ) +from litellm.types.llms.openai import AllEmbeddingInputValues, AllMessageValues from litellm.types.utils import EmbeddingResponse, PromptTokensDetailsWrapper, Usage from litellm.utils import is_base64_encoded +from ..common_utils import CohereError -class CohereEmbeddingConfig: + +class CohereEmbeddingConfig(BaseEmbeddingConfig): """ Reference: https://docs.cohere.com/v2/reference/embed """ @@ -32,20 +38,55 @@ class CohereEmbeddingConfig: def __init__(self) -> None: pass - def get_supported_openai_params(self) -> List[str]: - return ["encoding_format"] + def get_supported_openai_params(self, model: str) -> List[str]: + return ["encoding_format", "dimensions"] def map_openai_params( - self, non_default_params: dict, optional_params: dict + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool = False, ) -> dict: for k, v in non_default_params.items(): if k == "encoding_format": optional_params["embedding_types"] = v + elif k == "dimensions": + optional_params["output_dimension"] = v return optional_params + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + default_headers = { + "Content-Type": "application/json", + } + if api_key: + default_headers["Authorization"] = f"Bearer {api_key}" + headers = {**default_headers, **headers} + return headers + def _is_v3_model(self, model: str) -> bool: return "3" in model + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: + return api_base or "https://api.cohere.ai/v2/embed" + def _transform_request( self, model: str, input: List[str], inference_params: dict ) -> CohereEmbeddingRequestWithModel: @@ -71,6 +112,26 @@ class CohereEmbeddingConfig: return transformed_request + def transform_embedding_request( + self, + model: str, + input: AllEmbeddingInputValues, + optional_params: dict, + headers: dict, + ) -> dict: + if isinstance(input, list) and ( + isinstance(input[0], list) or isinstance(input[0], int) + ): + raise ValueError("Input must be a list of strings") + return cast( + dict, + self._transform_request( + model=model, + input=cast(List[str], input) if isinstance(input, List) else [input], + inference_params=optional_params, + ), + ) + def _calculate_usage(self, input: List[str], encoding: Any, meta: dict) -> Usage: input_tokens = 0 @@ -131,10 +192,11 @@ class CohereEmbeddingConfig: """ embeddings = response_json["embeddings"] output_data = [] - for idx, embedding in enumerate(embeddings): - output_data.append( - {"object": "embedding", "index": idx, "embedding": embedding} - ) + for k, embedding_list in embeddings.items(): + for idx, embedding in enumerate(embedding_list): + output_data.append( + {"object": "embedding", "index": idx, "embedding": embedding} + ) model_response.object = "list" model_response.data = output_data model_response.model = model @@ -149,3 +211,33 @@ class CohereEmbeddingConfig: ) return model_response + + def transform_embedding_response( + self, + model: str, + raw_response: httpx.Response, + model_response: EmbeddingResponse, + logging_obj: LiteLLMLoggingObj, + api_key: Optional[str], + request_data: dict, + optional_params: dict, + litellm_params: dict, + ) -> EmbeddingResponse: + return self._transform_response( + response=raw_response, + api_key=api_key, + logging_obj=logging_obj, + data=request_data, + model_response=model_response, + model=model, + encoding=litellm.encoding, + input=logging_obj.model_call_details["input"], + ) + + def get_error_class( + self, error_message: str, status_code: int, headers: Union[dict, httpx.Headers] + ) -> BaseLLMException: + return CohereError( + status_code=status_code, + message=error_message, + ) diff --git a/litellm/llms/cohere/embed/v1_transformation.py b/litellm/llms/cohere/embed/v1_transformation.py new file mode 100644 index 00000000000..e55899a4afa --- /dev/null +++ b/litellm/llms/cohere/embed/v1_transformation.py @@ -0,0 +1,143 @@ +""" +Legacy /v1/embedding transformation logic for Bedrock Cohere. +""" + +from typing import Any, List, Optional, Union + +import httpx + +from litellm import COHERE_DEFAULT_EMBEDDING_INPUT_TYPE +from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj +from litellm.types.llms.bedrock import ( + CohereEmbeddingRequest, + CohereEmbeddingRequestWithModel, +) +from litellm.types.utils import EmbeddingResponse, PromptTokensDetailsWrapper, Usage +from litellm.utils import is_base64_encoded + + +class CohereEmbeddingConfig: + """ + Reference: https://docs.cohere.com/v2/reference/embed + """ + + def __init__(self) -> None: + pass + + def get_supported_openai_params(self) -> List[str]: + return ["encoding_format"] + + def map_openai_params( + self, non_default_params: dict, optional_params: dict + ) -> dict: + for k, v in non_default_params.items(): + if k == "encoding_format": + optional_params["embedding_types"] = v + return optional_params + + def _is_v3_model(self, model: str) -> bool: + return "3" in model + + def _transform_request( + self, model: str, input: List[str], inference_params: dict + ) -> CohereEmbeddingRequestWithModel: + is_encoded = False + for input_str in input: + is_encoded = is_base64_encoded(input_str) + + if is_encoded: # check if string is b64 encoded image or not + transformed_request = CohereEmbeddingRequestWithModel( + model=model, + images=input, + input_type="image", + ) + else: + transformed_request = CohereEmbeddingRequestWithModel( + model=model, + texts=input, + input_type=COHERE_DEFAULT_EMBEDDING_INPUT_TYPE, + ) + + for k, v in inference_params.items(): + transformed_request[k] = v # type: ignore + + return transformed_request + + def _calculate_usage(self, input: List[str], encoding: Any, meta: dict) -> Usage: + input_tokens = 0 + + text_tokens: Optional[int] = meta.get("billed_units", {}).get("input_tokens") + + image_tokens: Optional[int] = meta.get("billed_units", {}).get("images") + + prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None + if image_tokens is None and text_tokens is None: + for text in input: + input_tokens += len(encoding.encode(text)) + else: + prompt_tokens_details = PromptTokensDetailsWrapper( + image_tokens=image_tokens, + text_tokens=text_tokens, + ) + if image_tokens: + input_tokens += image_tokens + if text_tokens: + input_tokens += text_tokens + + return Usage( + prompt_tokens=input_tokens, + completion_tokens=0, + total_tokens=input_tokens, + prompt_tokens_details=prompt_tokens_details, + ) + + def _transform_response( + self, + response: httpx.Response, + api_key: Optional[str], + logging_obj: LiteLLMLoggingObj, + data: Union[dict, CohereEmbeddingRequest], + model_response: EmbeddingResponse, + model: str, + encoding: Any, + input: list, + ) -> EmbeddingResponse: + response_json = response.json() + ## LOGGING + logging_obj.post_call( + input=input, + api_key=api_key, + additional_args={"complete_input_dict": data}, + original_response=response_json, + ) + """ + response + { + 'object': "list", + 'data': [ + + ] + 'model', + 'usage' + } + """ + embeddings = response_json["embeddings"] + output_data = [] + for idx, embedding in enumerate(embeddings): + output_data.append( + {"object": "embedding", "index": idx, "embedding": embedding} + ) + model_response.object = "list" + model_response.data = output_data + model_response.model = model + input_tokens = 0 + for text in input: + input_tokens += len(encoding.encode(text)) + + setattr( + model_response, + "usage", + self._calculate_usage(input, encoding, response_json.get("meta", {})), + ) + + return model_response diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index abbbc2e5959..4cf89accfce 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -1,5 +1,16 @@ import json -from typing import TYPE_CHECKING, Any, Coroutine, Dict, Optional, Tuple, Union +from typing import ( + TYPE_CHECKING, + Any, + AsyncIterator, + Coroutine, + Dict, + List, + Optional, + Tuple, + Union, + cast, +) import httpx # type: ignore @@ -8,6 +19,10 @@ import litellm.litellm_core_utils import litellm.types import litellm.types.utils from litellm._logging import verbose_logger +from litellm.litellm_core_utils.realtime_streaming import RealTimeStreaming +from litellm.llms.base_llm.anthropic_messages.transformation import ( + BaseAnthropicMessagesConfig, +) from litellm.llms.base_llm.audio_transcription.transformation import ( BaseAudioTranscriptionConfig, ) @@ -15,6 +30,7 @@ from litellm.llms.base_llm.base_model_iterator import MockResponseIterator from litellm.llms.base_llm.chat.transformation import BaseConfig from litellm.llms.base_llm.embedding.transformation import BaseEmbeddingConfig from litellm.llms.base_llm.files.transformation import BaseFilesConfig +from litellm.llms.base_llm.realtime.transformation import BaseRealtimeConfig from litellm.llms.base_llm.rerank.transformation import BaseRerankConfig from litellm.llms.base_llm.responses.transformation import BaseResponsesAPIConfig from litellm.llms.custom_httpx.http_handler import ( @@ -29,6 +45,9 @@ from litellm.responses.streaming_iterator import ( ResponsesAPIStreamingIterator, SyncResponsesAPIStreamingIterator, ) +from litellm.types.llms.anthropic_messages.anthropic_response import ( + AnthropicMessagesResponse, +) from litellm.types.llms.openai import ( CreateFileRequest, OpenAIFileObject, @@ -61,6 +80,7 @@ class BaseLLMHTTPHandler: litellm_params: dict, logging_obj: LiteLLMLoggingObj, stream: bool = False, + signed_json_body: Optional[bytes] = None, ) -> httpx.Response: """Common implementation across stream + non-stream calls. Meant to ensure consistent error-handling.""" max_retry_on_unprocessable_entity_error = ( @@ -73,7 +93,11 @@ class BaseLLMHTTPHandler: response = await async_httpx_client.post( url=api_base, headers=headers, - data=json.dumps(data), + data=( + signed_json_body + if signed_json_body is not None + else json.dumps(data) + ), timeout=timeout, stream=stream, logging_obj=logging_obj, @@ -116,6 +140,7 @@ class BaseLLMHTTPHandler: litellm_params: dict, logging_obj: LiteLLMLoggingObj, stream: bool = False, + signed_json_body: Optional[bytes] = None, ) -> httpx.Response: max_retry_on_unprocessable_entity_error = ( provider_config.max_retry_on_unprocessable_entity_error @@ -128,7 +153,11 @@ class BaseLLMHTTPHandler: response = sync_httpx_client.post( url=api_base, headers=headers, - data=json.dumps(data), + data=( + signed_json_body + if signed_json_body is not None + else json.dumps(data) + ), timeout=timeout, stream=stream, logging_obj=logging_obj, @@ -178,6 +207,7 @@ class BaseLLMHTTPHandler: api_key: Optional[str] = None, client: Optional[AsyncHTTPHandler] = None, json_mode: bool = False, + signed_json_body: Optional[bytes] = None, ): if client is None: async_httpx_client = get_async_httpx_client( @@ -197,6 +227,7 @@ class BaseLLMHTTPHandler: litellm_params=litellm_params, stream=False, logging_obj=logging_obj, + signed_json_body=signed_json_body, ) return provider_config.transform_response( model=model, @@ -228,7 +259,7 @@ class BaseLLMHTTPHandler: stream: Optional[bool] = False, fake_stream: bool = False, api_key: Optional[str] = None, - headers: Optional[dict] = {}, + headers: Optional[Dict[str, Any]] = None, client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, provider_config: Optional[BaseConfig] = None, ): @@ -278,7 +309,7 @@ class BaseLLMHTTPHandler: if extra_body is not None: data = {**data, **extra_body} - headers = provider_config.sign_request( + headers, signed_json_body = provider_config.sign_request( headers=headers, optional_params=optional_params, request_data=data, @@ -325,6 +356,7 @@ class BaseLLMHTTPHandler: litellm_params=litellm_params, json_mode=json_mode, optional_params=optional_params, + signed_json_body=signed_json_body, ) else: @@ -349,6 +381,7 @@ class BaseLLMHTTPHandler: else None ), json_mode=json_mode, + signed_json_body=signed_json_body, ) if stream is True: @@ -365,6 +398,7 @@ class BaseLLMHTTPHandler: api_base=api_base, headers=headers, data=data, + signed_json_body=signed_json_body, messages=messages, client=client, json_mode=json_mode, @@ -374,6 +408,8 @@ class BaseLLMHTTPHandler: api_base=api_base, headers=headers, # type: ignore data=data, + signed_json_body=signed_json_body, + original_data=data, model=model, messages=messages, logging_obj=logging_obj, @@ -408,6 +444,7 @@ class BaseLLMHTTPHandler: api_base=api_base, headers=headers, data=data, + signed_json_body=signed_json_body, timeout=timeout, litellm_params=litellm_params, logging_obj=logging_obj, @@ -432,6 +469,8 @@ class BaseLLMHTTPHandler: api_base: str, headers: dict, data: dict, + signed_json_body: Optional[bytes], + original_data: dict, model: str, messages: list, logging_obj, @@ -460,6 +499,7 @@ class BaseLLMHTTPHandler: api_base=api_base, headers=headers, data=data, + signed_json_body=signed_json_body, timeout=timeout, litellm_params=litellm_params, stream=stream, @@ -472,7 +512,7 @@ class BaseLLMHTTPHandler: raw_response=response, model_response=litellm.ModelResponse(), logging_obj=logging_obj, - request_data=data, + request_data=original_data, messages=messages, optional_params=optional_params, litellm_params=litellm_params, @@ -516,9 +556,10 @@ class BaseLLMHTTPHandler: fake_stream: bool = False, client: Optional[AsyncHTTPHandler] = None, json_mode: Optional[bool] = None, + signed_json_body: Optional[bytes] = None, ): if provider_config.has_custom_stream_wrapper is True: - return provider_config.get_async_custom_stream_wrapper( + return await provider_config.get_async_custom_stream_wrapper( model=model, custom_llm_provider=custom_llm_provider, logging_obj=logging_obj, @@ -528,6 +569,7 @@ class BaseLLMHTTPHandler: messages=messages, client=client, json_mode=json_mode, + signed_json_body=signed_json_body, ) completion_stream, _response_headers = await self.make_async_call_stream_helper( @@ -545,6 +587,7 @@ class BaseLLMHTTPHandler: litellm_params=litellm_params, optional_params=optional_params, json_mode=json_mode, + signed_json_body=signed_json_body, ) streamwrapper = CustomStreamWrapper( completion_stream=completion_stream, @@ -570,6 +613,7 @@ class BaseLLMHTTPHandler: fake_stream: bool = False, client: Optional[AsyncHTTPHandler] = None, json_mode: Optional[bool] = None, + signed_json_body: Optional[bytes] = None, ) -> Tuple[Any, httpx.Headers]: """ Helper function for making an async call with stream. @@ -593,6 +637,7 @@ class BaseLLMHTTPHandler: api_base=api_base, headers=headers, data=data, + signed_json_body=signed_json_body, timeout=timeout, litellm_params=litellm_params, stream=stream, @@ -663,15 +708,19 @@ class BaseLLMHTTPHandler: api_key: Optional[str] = None, client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, aembedding: bool = False, - headers={}, + headers: Optional[Dict[str, Any]] = None, ) -> EmbeddingResponse: provider_config = ProviderConfigManager.get_provider_embedding_config( model=model, provider=litellm.LlmProviders(custom_llm_provider) ) + if provider_config is None: + raise ValueError( + f"Provider {custom_llm_provider} does not support embedding" + ) # get config from model, custom llm provider headers = provider_config.validate_environment( api_key=api_key, - headers=headers, + headers=headers or {}, model=model, messages=[], optional_params=optional_params, @@ -804,7 +853,7 @@ class BaseLLMHTTPHandler: timeout: Optional[Union[float, httpx.Timeout]], model_response: RerankResponse, _is_async: bool = False, - headers: dict = {}, + headers: Optional[Dict[str, Any]] = None, api_key: Optional[str] = None, api_base: Optional[str] = None, client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, @@ -812,7 +861,7 @@ class BaseLLMHTTPHandler: # get config from model, custom llm provider headers = provider_config.validate_environment( api_key=api_key, - headers=headers, + headers=headers or {}, model=model, ) @@ -934,7 +983,7 @@ class BaseLLMHTTPHandler: custom_llm_provider: str, client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, atranscription: bool = False, - headers: dict = {}, + headers: Optional[Dict[str, Any]] = None, provider_config: Optional[BaseAudioTranscriptionConfig] = None, ) -> TranscriptionResponse: if provider_config is None: @@ -943,7 +992,7 @@ class BaseLLMHTTPHandler: ) headers = provider_config.validate_environment( api_key=api_key, - headers=headers, + headers=headers or {}, model=model, messages=[], optional_params=optional_params, @@ -1001,6 +1050,173 @@ class BaseLLMHTTPHandler: return returned_response return model_response + async def async_anthropic_messages_handler( + self, + model: str, + messages: List[Dict], + anthropic_messages_provider_config: BaseAnthropicMessagesConfig, + anthropic_messages_optional_request_params: Dict, + custom_llm_provider: str, + litellm_params: GenericLiteLLMParams, + logging_obj: LiteLLMLoggingObj, + client: Optional[AsyncHTTPHandler] = None, + extra_headers: Optional[Dict[str, Any]] = None, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + stream: Optional[bool] = False, + kwargs: Optional[Dict[str, Any]] = None, + ) -> Union[AnthropicMessagesResponse, AsyncIterator]: + if client is None or not isinstance(client, AsyncHTTPHandler): + async_httpx_client = get_async_httpx_client( + llm_provider=litellm.LlmProviders.ANTHROPIC + ) + else: + async_httpx_client = client + + # Prepare headers + kwargs = kwargs or {} + provider_specific_header = cast( + Optional[litellm.types.utils.ProviderSpecificHeader], + kwargs.get("provider_specific_header", None), + ) + extra_headers = ( + provider_specific_header.get("extra_headers", {}) + if provider_specific_header + else {} + ) + headers = anthropic_messages_provider_config.validate_environment( + headers=extra_headers or {}, + model=model, + messages=messages, + optional_params=anthropic_messages_optional_request_params, + litellm_params=dict(litellm_params), + api_key=api_key, + api_base=api_base, + ) + + logging_obj.update_environment_variables( + model=model, + optional_params=dict(anthropic_messages_optional_request_params), + litellm_params={ + "metadata": kwargs.get("metadata", {}), + "preset_cache_key": None, + "stream_response": {}, + **anthropic_messages_optional_request_params, + }, + custom_llm_provider=custom_llm_provider, + ) + # Prepare request body + request_body = anthropic_messages_provider_config.transform_anthropic_messages_request( + model=model, + messages=messages, + anthropic_messages_optional_request_params=anthropic_messages_optional_request_params, + litellm_params=litellm_params, + headers=headers, + ) + logging_obj.stream = stream + logging_obj.model_call_details.update(request_body) + + # Make the request + request_url = anthropic_messages_provider_config.get_complete_url( + api_base=api_base, + api_key=api_key, + model=model, + optional_params=dict( + litellm_params + ), # this uses the invoke config, which expects aws_* params in optional_params + litellm_params=dict(litellm_params), + stream=stream, + ) + + headers, signed_json_body = anthropic_messages_provider_config.sign_request( + headers=headers, + optional_params=dict( + litellm_params + ), # dynamic aws_* params are passed under litellm_params + request_data=request_body, + api_base=request_url, + stream=stream, + fake_stream=False, + model=model, + ) + + logging_obj.pre_call( + input=[{"role": "user", "content": json.dumps(request_body)}], + api_key="", + additional_args={ + "complete_input_dict": request_body, + "api_base": str(request_url), + "headers": headers, + }, + ) + + response = await async_httpx_client.post( + url=request_url, + headers=headers, + data=signed_json_body or json.dumps(request_body), + stream=stream or False, + logging_obj=logging_obj, + ) + response.raise_for_status() + + # used for logging + cost tracking + logging_obj.model_call_details["httpx_response"] = response + + if stream: + completion_stream = anthropic_messages_provider_config.get_async_streaming_response_iterator( + model=model, + httpx_response=response, + request_body=request_body, + litellm_logging_obj=logging_obj, + ) + return completion_stream + else: + return anthropic_messages_provider_config.transform_anthropic_messages_response( + model=model, + raw_response=response, + logging_obj=logging_obj, + ) + + def anthropic_messages_handler( + self, + model: str, + messages: List[Dict], + anthropic_messages_provider_config: BaseAnthropicMessagesConfig, + anthropic_messages_optional_request_params: Dict, + custom_llm_provider: str, + _is_async: bool, + litellm_params: GenericLiteLLMParams, + logging_obj: LiteLLMLoggingObj, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + stream: Optional[bool] = False, + kwargs: Optional[Dict[str, Any]] = None, + ) -> Union[ + AnthropicMessagesResponse, + Coroutine[Any, Any, Union[AnthropicMessagesResponse, AsyncIterator]], + ]: + """ + LLM HTTP Handler for Anthropic Messages + """ + if _is_async: + # Return the async coroutine if called with _is_async=True + return self.async_anthropic_messages_handler( + model=model, + messages=messages, + anthropic_messages_provider_config=anthropic_messages_provider_config, + anthropic_messages_optional_request_params=anthropic_messages_optional_request_params, + client=client if isinstance(client, AsyncHTTPHandler) else None, + custom_llm_provider=custom_llm_provider, + litellm_params=litellm_params, + logging_obj=logging_obj, + api_key=api_key, + api_base=api_base, + stream=stream, + kwargs=kwargs, + ) + raise ValueError("anthropic_messages_handler is not implemented for sync calls") + def response_api_handler( self, model: str, @@ -1455,7 +1671,7 @@ class BaseLLMHTTPHandler: timeout=timeout, client=client, ) - + if client is None or not isinstance(client, HTTPHandler): sync_httpx_client = _get_httpx_client( params={"ssl_verify": litellm_params.get("ssl_verify", None)} @@ -1496,9 +1712,7 @@ class BaseLLMHTTPHandler: ) try: - response = sync_httpx_client.get( - url=url, headers=headers, params=data - ) + response = sync_httpx_client.get(url=url, headers=headers, params=data) except Exception as e: raise self._handle_error( e=e, @@ -1808,7 +2022,11 @@ class BaseLLMHTTPHandler: ): status_code = getattr(e, "status_code", 500) error_headers = getattr(e, "headers", None) - error_text = getattr(e, "text", str(e)) + if isinstance(e, httpx.HTTPStatusError): + error_text = e.response.text + status_code = e.response.status_code + else: + error_text = getattr(e, "text", str(e)) error_response = getattr(e, "response", None) if error_headers is None and error_response: error_headers = getattr(error_response, "headers", None) @@ -1818,8 +2036,65 @@ class BaseLLMHTTPHandler: error_headers = dict(error_headers) else: error_headers = {} + raise provider_config.get_error_class( error_message=error_text, status_code=status_code, headers=error_headers, ) + + async def async_realtime( + self, + model: str, + websocket: Any, + logging_obj: LiteLLMLoggingObj, + provider_config: BaseRealtimeConfig, + headers: dict, + api_base: Optional[str] = None, + api_key: Optional[str] = None, + client: Optional[Any] = None, + timeout: Optional[float] = None, + ): + import websockets + from websockets.asyncio.client import ClientConnection + + url = provider_config.get_complete_url(api_base, model, api_key) + headers = provider_config.validate_environment( + headers=headers, + model=model, + api_key=api_key, + ) + + try: + async with websockets.connect( # type: ignore + url, extra_headers=headers + ) as backend_ws: + realtime_streaming = RealTimeStreaming( + websocket, + cast(ClientConnection, backend_ws), + logging_obj, + provider_config, + model, + ) + await realtime_streaming.bidirectional_forward() + + except websockets.exceptions.InvalidStatusCode as e: # type: ignore + verbose_logger.exception(f"Error connecting to backend: {e}") + await websocket.close(code=e.status_code, reason=str(e)) + except Exception as e: + verbose_logger.exception(f"Error connecting to backend: {e}") + try: + await websocket.close( + code=1011, reason=f"Internal server error: {str(e)}" + ) + except RuntimeError as close_error: + if "already completed" in str(close_error) or "websocket.close" in str( + close_error + ): + # The WebSocket is already closed or the response is completed, so we can ignore this error + pass + else: + # If it's a different RuntimeError, we might want to log it or handle it differently + raise Exception( + f"Unexpected error while closing WebSocket: {close_error}" + ) diff --git a/litellm/llms/databricks/chat/transformation.py b/litellm/llms/databricks/chat/transformation.py index 06f525719cf..ba22f7ac443 100644 --- a/litellm/llms/databricks/chat/transformation.py +++ b/litellm/llms/databricks/chat/transformation.py @@ -6,12 +6,15 @@ from typing import ( TYPE_CHECKING, Any, AsyncIterator, + Coroutine, Iterator, List, + Literal, Optional, Tuple, Union, cast, + overload, ) import httpx @@ -276,9 +279,24 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig): return False + @overload def _transform_messages( - self, messages: List[AllMessageValues], model: str + self, messages: List[AllMessageValues], model: str, is_async: Literal[True] + ) -> Coroutine[Any, Any, List[AllMessageValues]]: + ... + + @overload + def _transform_messages( + self, + messages: List[AllMessageValues], + model: str, + is_async: Literal[False] = False, ) -> List[AllMessageValues]: + ... + + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: bool = False + ) -> Union[List[AllMessageValues], Coroutine[Any, Any, List[AllMessageValues]]]: """ Databricks does not support: - content in list format. @@ -293,7 +311,15 @@ class DatabricksConfig(DatabricksBase, OpenAILikeChatConfig, AnthropicConfig): new_messages.append(_message) new_messages = handle_messages_with_content_list_to_str_conversion(new_messages) new_messages = strip_name_from_messages(new_messages) - return super()._transform_messages(messages=new_messages, model=model) + + if is_async: + return super()._transform_messages( + messages=new_messages, model=model, is_async=cast(Literal[True], True) + ) + else: + return super()._transform_messages( + messages=new_messages, model=model, is_async=cast(Literal[False], False) + ) @staticmethod def extract_content_str( @@ -543,7 +569,7 @@ class DatabricksChatResponseIterator(BaseModelResponseIterator): reasoning_content, thinking_blocks, ) = DatabricksConfig.extract_reasoning_content( - choice["delta"]["content"] + choice["delta"].get("content") ) choice["delta"]["content"] = content_str diff --git a/litellm/llms/deepseek/chat/transformation.py b/litellm/llms/deepseek/chat/transformation.py index f429f46331f..a7defa886b5 100644 --- a/litellm/llms/deepseek/chat/transformation.py +++ b/litellm/llms/deepseek/chat/transformation.py @@ -2,7 +2,7 @@ Translates from OpenAI's `/v1/chat/completions` to DeepSeek's `/v1/chat/completions` """ -from typing import List, Optional, Tuple +from typing import Any, Coroutine, List, Literal, Optional, Tuple, Union, overload from litellm.litellm_core_utils.prompt_templates.common_utils import ( handle_messages_with_content_list_to_str_conversion, @@ -14,14 +14,36 @@ from ...openai.chat.gpt_transformation import OpenAIGPTConfig class DeepSeekChatConfig(OpenAIGPTConfig): + @overload def _transform_messages( - self, messages: List[AllMessageValues], model: str + self, messages: List[AllMessageValues], model: str, is_async: Literal[True] + ) -> Coroutine[Any, Any, List[AllMessageValues]]: + ... + + @overload + def _transform_messages( + self, + messages: List[AllMessageValues], + model: str, + is_async: Literal[False] = False, ) -> List[AllMessageValues]: + ... + + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: bool = False + ) -> Union[List[AllMessageValues], Coroutine[Any, Any, List[AllMessageValues]]]: """ DeepSeek does not support content in list format. """ messages = handle_messages_with_content_list_to_str_conversion(messages) - return super()._transform_messages(messages=messages, model=model) + if is_async: + return super()._transform_messages( + messages=messages, model=model, is_async=True + ) + else: + return super()._transform_messages( + messages=messages, model=model, is_async=False + ) def _get_openai_compatible_provider_info( self, api_base: Optional[str], api_key: Optional[str] diff --git a/litellm/llms/featherless_ai/chat/transformation.py b/litellm/llms/featherless_ai/chat/transformation.py new file mode 100644 index 00000000000..96702cf886e --- /dev/null +++ b/litellm/llms/featherless_ai/chat/transformation.py @@ -0,0 +1,128 @@ +from typing import Optional, Tuple, Union + +import litellm +from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig +from litellm.secret_managers.main import get_secret_str + + +class FeatherlessAIConfig(OpenAIGPTConfig): + """ + Reference: https://featherless.ai/docs/completions + + The class `FeatherlessAI` provides configuration for the FeatherlessAI's Chat Completions API interface. Below are the parameters: + """ + + frequency_penalty: Optional[int] = None + function_call: Optional[Union[str, dict]] = None + functions: Optional[list] = None + logit_bias: Optional[dict] = None + max_tokens: Optional[int] = None + n: Optional[int] = None + presence_penalty: Optional[int] = None + stop: Optional[Union[str, list]] = None + temperature: Optional[int] = None + top_p: Optional[int] = None + response_format: Optional[dict] = None + tool_choice: Optional[str] = None + tools: Optional[list] = None + + def __init__( + self, + frequency_penalty: Optional[int] = None, + function_call: Optional[Union[str, dict]] = None, + functions: Optional[list] = None, + logit_bias: Optional[dict] = None, + max_tokens: Optional[int] = None, + n: Optional[int] = None, + presence_penalty: Optional[int] = None, + stop: Optional[Union[str, list]] = None, + temperature: Optional[int] = None, + top_p: Optional[int] = None, + response_format: Optional[dict] = None, + tool_choice: Optional[str] = None, + tools: Optional[list] = None, + ) -> None: + locals_ = locals().copy() + for key, value in locals_.items(): + if key != "self" and value is not None: + setattr(self.__class__, key, value) + + @classmethod + def get_config(cls): + return super().get_config() + + def get_supported_openai_params(self, model: str): + return [ + "stream", + "frequency_penalty", + "function_call", + "functions", + "logit_bias", + "max_tokens", + "max_completion_tokens", + "n", + "presence_penalty", + "stop", + "temperature", + "top_p", + ] + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + supported_openai_params = self.get_supported_openai_params(model=model) + for param, value in non_default_params.items(): + if param == "tool_choice" or param == "tools": + if param == "tool_choice" and (value == "auto" or value == "none"): + # These values are supported, so add them to optional_params + optional_params[param] = value + else: # https://featherless.ai/docs/completions + ## UNSUPPORTED TOOL CHOICE VALUE + if litellm.drop_params is True or drop_params is True: + value = None + else: + error_message = f"Featherless AI doesn't support {param}={value}. To drop unsupported openai params from the call, set `litellm.drop_params = True`" + raise litellm.utils.UnsupportedParamsError( + message=error_message, + status_code=400, + ) + elif param == "max_completion_tokens": + optional_params["max_tokens"] = value + elif param in supported_openai_params: + if value is not None: + optional_params[param] = value + return optional_params + + def _get_openai_compatible_provider_info( + self, api_base: Optional[str], api_key: Optional[str] + ) -> Tuple[Optional[str], Optional[str]]: + # FeatherlessAI is openai compatible, set to custom_openai and use FeatherlessAI's endpoint + api_base = ( + api_base + or get_secret_str("FEATHERLESS_API_BASE") + or "https://api.featherless.ai/v1" + ) + dynamic_api_key = api_key or get_secret_str("FEATHERLESS_API_KEY") + return api_base, dynamic_api_key + + def validate_environment( + self, + headers: dict, + model: str, + messages: list, + optional_params: dict, + litellm_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + if not api_key: + raise ValueError("Missing Featherless AI API Key") + + headers["Authorization"] = f"Bearer {api_key}" + headers["Content-Type"] = "application/json" + + return headers diff --git a/litellm/llms/gemini/common_utils.py b/litellm/llms/gemini/common_utils.py index fef41f7d584..3331f584b51 100644 --- a/litellm/llms/gemini/common_utils.py +++ b/litellm/llms/gemini/common_utils.py @@ -1,8 +1,11 @@ -from typing import List, Optional, Union +import base64 +import datetime +from typing import Dict, List, Optional, Union import httpx import litellm +from litellm.constants import DEFAULT_MAX_RECURSE_DEPTH from litellm.llms.base_llm.base_utils import BaseLLMModelInfo from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.secret_managers.main import get_secret_str @@ -82,3 +85,47 @@ class GeminiModelInfo(BaseLLMModelInfo): return GeminiError( status_code=status_code, message=error_message, headers=headers ) + + +def encode_unserializable_types( + data: Dict[str, object], depth: int = 0 +) -> Dict[str, object]: + """Converts unserializable types in dict to json.dumps() compatible types. + + This function is called in models.py after calling convert_to_dict(). The + convert_to_dict() can convert pydantic object to dict. However, the input to + convert_to_dict() is dict mixed of pydantic object and nested dict(the output + of converters). So they may be bytes in the dict and they are out of + `ser_json_bytes` control in model_dump(mode='json') called in + `convert_to_dict`, as well as datetime deserialization in Pydantic json mode. + + Returns: + A dictionary with json.dumps() incompatible type (e.g. bytes datetime) + to compatible type (e.g. base64 encoded string, isoformat date string). + """ + if depth > DEFAULT_MAX_RECURSE_DEPTH: + return data + processed_data: dict[str, object] = {} + if not isinstance(data, dict): + return data + for key, value in data.items(): + if isinstance(value, bytes): + processed_data[key] = base64.urlsafe_b64encode(value).decode("ascii") + elif isinstance(value, datetime.datetime): + processed_data[key] = value.isoformat() + elif isinstance(value, dict): + processed_data[key] = encode_unserializable_types(value, depth + 1) + elif isinstance(value, list): + if all(isinstance(v, bytes) for v in value): + processed_data[key] = [ + base64.urlsafe_b64encode(v).decode("ascii") for v in value + ] + if all(isinstance(v, datetime.datetime) for v in value): + processed_data[key] = [v.isoformat() for v in value] + else: + processed_data[key] = [ + encode_unserializable_types(v, depth + 1) for v in value + ] + else: + processed_data[key] = value + return processed_data diff --git a/litellm/llms/gemini/realtime/transformation.py b/litellm/llms/gemini/realtime/transformation.py new file mode 100644 index 00000000000..01fc6b86e39 --- /dev/null +++ b/litellm/llms/gemini/realtime/transformation.py @@ -0,0 +1,950 @@ +""" +This file contains the transformation logic for the Gemini realtime API. +""" + +import json +import os +import uuid +from typing import Any, Dict, List, Optional, Union, cast + +from litellm import verbose_logger +from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj +from litellm.llms.base_llm.realtime.transformation import BaseRealtimeConfig +from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( + VertexGeminiConfig, +) +from litellm.responses.litellm_completion_transformation.transformation import ( + LiteLLMCompletionResponsesConfig, +) +from litellm.types.llms.gemini import ( + AutomaticActivityDetection, + BidiGenerateContentRealtimeInput, + BidiGenerateContentRealtimeInputConfig, + BidiGenerateContentServerContent, + BidiGenerateContentServerMessage, + BidiGenerateContentSetup, +) +from litellm.types.llms.openai import ( + OpenAIRealtimeContentPartDone, + OpenAIRealtimeConversationItemCreated, + OpenAIRealtimeDoneEvent, + OpenAIRealtimeEvents, + OpenAIRealtimeEventTypes, + OpenAIRealtimeOutputItemDone, + OpenAIRealtimeResponseAudioDone, + OpenAIRealtimeResponseContentPartAdded, + OpenAIRealtimeResponseDelta, + OpenAIRealtimeResponseDoneObject, + OpenAIRealtimeResponseTextDone, + OpenAIRealtimeStreamResponseBaseObject, + OpenAIRealtimeStreamResponseOutputItemAdded, + OpenAIRealtimeStreamSession, + OpenAIRealtimeStreamSessionEvents, + OpenAIRealtimeTurnDetection, +) +from litellm.types.llms.vertex_ai import ( + GeminiResponseModalities, + HttpxBlobType, + HttpxContentType, +) +from litellm.types.realtime import ( + ALL_DELTA_TYPES, + RealtimeModalityResponseTransformOutput, + RealtimeResponseTransformInput, + RealtimeResponseTypedDict, +) +from litellm.utils import get_empty_usage + +from ..common_utils import encode_unserializable_types + +MAP_GEMINI_FIELD_TO_OPENAI_EVENT: Dict[str, OpenAIRealtimeEventTypes] = { + "setupComplete": OpenAIRealtimeEventTypes.SESSION_CREATED, + "serverContent.generationComplete": OpenAIRealtimeEventTypes.RESPONSE_TEXT_DONE, + "serverContent.turnComplete": OpenAIRealtimeEventTypes.RESPONSE_DONE, + "serverContent.interrupted": OpenAIRealtimeEventTypes.RESPONSE_DONE, +} + + +class GeminiRealtimeConfig(BaseRealtimeConfig): + def validate_environment( + self, headers: dict, model: str, api_key: Optional[str] = None + ) -> dict: + return headers + + def get_complete_url( + self, api_base: Optional[str], model: str, api_key: Optional[str] = None + ) -> str: + """ + Example output: + "BACKEND_WS_URL = "wss://generativelanguage.googleapis.com/ws/google.ai.generativelanguage.v1beta.GenerativeService.BidiGenerateContent""; + """ + if api_base is None: + api_base = "wss://generativelanguage.googleapis.com" + if api_key is None: + api_key = os.environ.get("GEMINI_API_KEY") + if api_key is None: + raise ValueError("api_key is required for Gemini API calls") + api_base = api_base.replace("https://", "wss://") + api_base = api_base.replace("http://", "ws://") + return f"{api_base}/ws/google.ai.generativelanguage.v1beta.GenerativeService.BidiGenerateContent?key={api_key}" + + def map_model_turn_event( + self, model_turn: HttpxContentType + ) -> OpenAIRealtimeEventTypes: + """ + Map the model turn event to the OpenAI realtime events. + + Returns either: + - response.text.delta - model_turn: {"parts": [{"text": "..."}]} + - response.audio.delta - model_turn: {"parts": [{"inlineData": {"mimeType": "audio/pcm", "data": "..."}}]} + + Assumes parts is a single element list. + """ + if "parts" in model_turn: + parts = model_turn["parts"] + if len(parts) != 1: + verbose_logger.warning( + f"Realtime: Expected 1 part, got {len(parts)} for Gemini model turn event." + ) + part = parts[0] + if "text" in part: + return OpenAIRealtimeEventTypes.RESPONSE_TEXT_DELTA + elif "inlineData" in part: + return OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DELTA + else: + raise ValueError(f"Unexpected part type: {part}") + raise ValueError(f"Unexpected model turn event, no 'parts' key: {model_turn}") + + def map_generation_complete_event( + self, delta_type: Optional[ALL_DELTA_TYPES] + ) -> OpenAIRealtimeEventTypes: + if delta_type == "text": + return OpenAIRealtimeEventTypes.RESPONSE_TEXT_DONE + elif delta_type == "audio": + return OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DONE + else: + raise ValueError(f"Unexpected delta type: {delta_type}") + + def get_audio_mime_type(self, input_audio_format: str = "pcm16"): + mime_types = { + "pcm16": "audio/pcm", + "g711_ulaw": "audio/pcmu", + "g711_alaw": "audio/pcma", + } + + return mime_types.get(input_audio_format, "application/octet-stream") + + def map_automatic_turn_detection( + self, value: OpenAIRealtimeTurnDetection + ) -> AutomaticActivityDetection: + automatic_activity_dection = AutomaticActivityDetection() + if "create_response" in value and isinstance(value["create_response"], bool): + automatic_activity_dection["disabled"] = not value["create_response"] + else: + automatic_activity_dection["disabled"] = True + if "prefix_padding_ms" in value and isinstance(value["prefix_padding_ms"], int): + automatic_activity_dection["prefixPaddingMs"] = value["prefix_padding_ms"] + if "silence_duration_ms" in value and isinstance( + value["silence_duration_ms"], int + ): + automatic_activity_dection["silenceDurationMs"] = value[ + "silence_duration_ms" + ] + return automatic_activity_dection + + def get_supported_openai_params(self, model: str) -> List[str]: + return [ + "instructions", + "temperature", + "max_response_output_tokens", + "modalities", + "tools", + "input_audio_transcription", + "turn_detection", + ] + + def map_openai_params( + self, optional_params: dict, non_default_params: dict + ) -> dict: + if "generationConfig" not in optional_params: + optional_params["generationConfig"] = {} + for key, value in non_default_params.items(): + if key == "instructions": + optional_params["systemInstruction"] = HttpxContentType( + role="user", parts=[{"text": value}] + ) + elif key == "temperature": + optional_params["generationConfig"]["temperature"] = value + elif key == "max_response_output_tokens": + optional_params["generationConfig"]["maxOutputTokens"] = value + elif key == "modalities": + optional_params["generationConfig"]["responseModalities"] = [ + modality.upper() for modality in cast(List[str], value) + ] + elif key == "tools": + from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( + VertexGeminiConfig, + ) + + vertex_gemini_config = VertexGeminiConfig() + vertex_gemini_config._map_function(value) + optional_params["generationConfig"][ + "tools" + ] = vertex_gemini_config._map_function(value) + elif key == "input_audio_transcription" and value is not None: + optional_params["inputAudioTranscription"] = {} + elif key == "turn_detection": + value_typed = cast(OpenAIRealtimeTurnDetection, value) + transformed_audio_activity_config = self.map_automatic_turn_detection( + value_typed + ) + if ( + len(transformed_audio_activity_config) > 0 + ): # if the config is not empty, add it to the optional params + optional_params[ + "realtimeInputConfig" + ] = BidiGenerateContentRealtimeInputConfig( + automaticActivityDetection=transformed_audio_activity_config + ) + if len(optional_params["generationConfig"]) == 0: + optional_params.pop("generationConfig") + return optional_params + + def transform_realtime_request( + self, + message: str, + model: str, + session_configuration_request: Optional[str] = None, + ) -> List[str]: + realtime_input_dict: BidiGenerateContentRealtimeInput = {} + try: + json_message = json.loads(message) + except json.JSONDecodeError: + if isinstance(message, bytes): + message_str = message.decode("utf-8", errors="replace") + else: + message_str = str(message) + raise ValueError(f"Invalid JSON message: {message_str}") + + ## HANDLE SESSION UPDATE ## + messages: List[str] = [] + if "type" in json_message and json_message["type"] == "session.update": + client_session_configuration_request = self.map_openai_params( + optional_params={}, non_default_params=json_message["session"] + ) + client_session_configuration_request["model"] = f"models/{model}" + + messages.append( + json.dumps( + { + "setup": client_session_configuration_request, + } + ) + ) + # elif session_configuration_request is None: + # default_session_configuration_request = self.session_configuration_request(model) + # messages.append(default_session_configuration_request) + + ## HANDLE INPUT AUDIO BUFFER ## + if ( + "type" in json_message + and json_message["type"] == "input_audio_buffer.append" + ): + realtime_input_dict["audio"] = HttpxBlobType( + mimeType=self.get_audio_mime_type(), data=json_message["audio"] + ) + else: + realtime_input_dict["text"] = message + + if len(realtime_input_dict) != 1: + raise ValueError( + f"Only one argument can be set, got {len(realtime_input_dict)}:" + f" {list(realtime_input_dict.keys())}" + ) + + realtime_input_dict = cast( + BidiGenerateContentRealtimeInput, + encode_unserializable_types(cast(Dict[str, object], realtime_input_dict)), + ) + + messages.append(json.dumps({"realtime_input": realtime_input_dict})) + return messages + + def transform_session_created_event( + self, + model: str, + logging_session_id: str, + session_configuration_request: Optional[str] = None, + ) -> OpenAIRealtimeStreamSessionEvents: + if session_configuration_request: + session_configuration_request_dict: BidiGenerateContentSetup = json.loads( + session_configuration_request + ).get("setup", {}) + else: + session_configuration_request_dict = {} + + _model = session_configuration_request_dict.get("model") or model + generation_config = ( + session_configuration_request_dict.get("generationConfig", {}) or {} + ) + gemini_modalities = generation_config.get("responseModalities", ["TEXT"]) + _modalities = [ + modality.lower() for modality in cast(List[str], gemini_modalities) + ] + _system_instruction = session_configuration_request_dict.get( + "systemInstruction" + ) + session = OpenAIRealtimeStreamSession( + id=logging_session_id, + modalities=_modalities, + ) + if _system_instruction is not None and isinstance(_system_instruction, str): + session["instructions"] = _system_instruction + if _model is not None and isinstance(_model, str): + session["model"] = _model.strip( + "models/" + ) # keep it consistent with how openai returns the model name + + return OpenAIRealtimeStreamSessionEvents( + type="session.created", + session=session, + event_id=str(uuid.uuid4()), + ) + + def _is_new_content_delta( + self, + previous_messages: Optional[List[OpenAIRealtimeEvents]] = None, + ) -> bool: + if previous_messages is None or len(previous_messages) == 0: + return True + if "type" in previous_messages[-1] and previous_messages[-1]["type"].endswith( + "delta" + ): + return False + return True + + def return_new_content_delta_events( + self, + response_id: str, + output_item_id: str, + conversation_id: str, + delta_type: ALL_DELTA_TYPES, + session_configuration_request: Optional[str] = None, + ) -> List[OpenAIRealtimeEvents]: + if session_configuration_request is None: + raise ValueError( + "session_configuration_request is required for Gemini API calls" + ) + + session_configuration_request_dict: BidiGenerateContentSetup = json.loads( + session_configuration_request + ).get("setup", {}) + generation_config = session_configuration_request_dict.get( + "generationConfig", {} + ) + gemini_modalities = generation_config.get("responseModalities", ["TEXT"]) + _modalities = [ + modality.lower() for modality in cast(List[str], gemini_modalities) + ] + + _temperature = generation_config.get("temperature") + _max_output_tokens = generation_config.get("maxOutputTokens") + + response_items: List[OpenAIRealtimeEvents] = [] + + ## - return response.created + response_created = OpenAIRealtimeStreamResponseBaseObject( + type="response.created", + event_id="event_{}".format(uuid.uuid4()), + response={ + "object": "realtime.response", + "id": response_id, + "status": "in_progress", + "output": [], + "conversation_id": conversation_id, + "modalities": _modalities, + "temperature": _temperature, + "max_output_tokens": _max_output_tokens, + }, + ) + response_items.append(response_created) + + ## - return response.output_item.added ← adds ‘item_id’ same for all subsequent events + response_output_item_added = OpenAIRealtimeStreamResponseOutputItemAdded( + type="response.output_item.added", + response_id=response_id, + output_index=0, + item={ + "id": output_item_id, + "object": "realtime.item", + "type": "message", + "status": "in_progress", + "role": "assistant", + "content": [], + }, + ) + response_items.append(response_output_item_added) + ## - return conversation.item.created + conversation_item_created = OpenAIRealtimeConversationItemCreated( + type="conversation.item.created", + event_id="event_{}".format(uuid.uuid4()), + item={ + "id": output_item_id, + "object": "realtime.item", + "type": "message", + "status": "in_progress", + "role": "assistant", + "content": [], + }, + ) + response_items.append(conversation_item_created) + ## - return response.content_part.added + response_content_part_added = OpenAIRealtimeResponseContentPartAdded( + type="response.content_part.added", + content_index=0, + output_index=0, + event_id="event_{}".format(uuid.uuid4()), + item_id=output_item_id, + part={ + "type": "text", + "text": "", + } + if delta_type == "text" + else { + "type": "audio", + "transcript": "", + }, + response_id=response_id, + ) + response_items.append(response_content_part_added) + return response_items + + def transform_content_delta_events( + self, + message: BidiGenerateContentServerContent, + output_item_id: str, + response_id: str, + delta_type: ALL_DELTA_TYPES, + ) -> OpenAIRealtimeResponseDelta: + delta = "" + try: + if "modelTurn" in message and "parts" in message["modelTurn"]: + for part in message["modelTurn"]["parts"]: + if "text" in part: + delta += part["text"] + elif "inlineData" in part: + delta += part["inlineData"]["data"] + except Exception as e: + raise ValueError( + f"Error transforming content delta events: {e}, got message: {message}" + ) + + return OpenAIRealtimeResponseDelta( + type="response.text.delta" + if delta_type == "text" + else "response.audio.delta", + content_index=0, + event_id="event_{}".format(uuid.uuid4()), + item_id=output_item_id, + output_index=0, + response_id=response_id, + delta=delta, + ) + + def transform_content_done_event( + self, + delta_chunks: Optional[List[OpenAIRealtimeResponseDelta]], + current_output_item_id: Optional[str], + current_response_id: Optional[str], + delta_type: ALL_DELTA_TYPES, + ) -> Union[OpenAIRealtimeResponseTextDone, OpenAIRealtimeResponseAudioDone]: + if delta_chunks: + delta = "".join([delta_chunk["delta"] for delta_chunk in delta_chunks]) + else: + delta = "" + if current_output_item_id is None or current_response_id is None: + raise ValueError( + "current_output_item_id and current_response_id cannot be None for a 'done' event." + ) + if delta_type == "text": + return OpenAIRealtimeResponseTextDone( + type="response.text.done", + content_index=0, + event_id="event_{}".format(uuid.uuid4()), + item_id=current_output_item_id, + output_index=0, + response_id=current_response_id, + text=delta, + ) + elif delta_type == "audio": + return OpenAIRealtimeResponseAudioDone( + type="response.audio.done", + content_index=0, + event_id="event_{}".format(uuid.uuid4()), + item_id=current_output_item_id, + output_index=0, + response_id=current_response_id, + ) + + def return_additional_content_done_events( + self, + current_output_item_id: Optional[str], + current_response_id: Optional[str], + delta_done_event: Union[ + OpenAIRealtimeResponseTextDone, OpenAIRealtimeResponseAudioDone + ], + delta_type: ALL_DELTA_TYPES, + ) -> List[OpenAIRealtimeEvents]: + """ + - return response.content_part.done + - return response.output_item.done + """ + if current_output_item_id is None or current_response_id is None: + raise ValueError( + "current_output_item_id and current_response_id cannot be None for a 'done' event." + ) + returned_items: List[OpenAIRealtimeEvents] = [] + + delta_done_event_text = cast(Optional[str], delta_done_event.get("text")) + # response.content_part.done + response_content_part_done = OpenAIRealtimeContentPartDone( + type="response.content_part.done", + content_index=0, + event_id="event_{}".format(uuid.uuid4()), + item_id=current_output_item_id, + output_index=0, + part={"type": "text", "text": delta_done_event_text} + if delta_done_event_text and delta_type == "text" + else { + "type": "audio", + "transcript": "", # gemini doesn't return transcript for audio + }, + response_id=current_response_id, + ) + returned_items.append(response_content_part_done) + # response.output_item.done + response_output_item_done = OpenAIRealtimeOutputItemDone( + type="response.output_item.done", + event_id="event_{}".format(uuid.uuid4()), + output_index=0, + response_id=current_response_id, + item={ + "id": current_output_item_id, + "object": "realtime.item", + "type": "message", + "status": "completed", + "role": "assistant", + "content": [ + {"type": "text", "text": delta_done_event_text} + if delta_done_event_text and delta_type == "text" + else { + "type": "audio", + "transcript": "", + } + ], + }, + ) + returned_items.append(response_output_item_done) + return returned_items + + @staticmethod + def get_nested_value(obj: dict, path: str) -> Any: + keys = path.split(".") + current = obj + for key in keys: + if isinstance(current, dict) and key in current: + current = current[key] + else: + return None + return current + + def update_current_delta_chunks( + self, + transformed_message: Union[OpenAIRealtimeEvents, List[OpenAIRealtimeEvents]], + current_delta_chunks: Optional[List[OpenAIRealtimeResponseDelta]], + ) -> Optional[List[OpenAIRealtimeResponseDelta]]: + try: + if isinstance(transformed_message, list): + current_delta_chunks = [] + any_delta_chunk = False + for event in transformed_message: + if event["type"] == "response.text.delta": + current_delta_chunks.append( + cast(OpenAIRealtimeResponseDelta, event) + ) + any_delta_chunk = True + if not any_delta_chunk: + current_delta_chunks = ( + None # reset current_delta_chunks if no delta chunks + ) + else: + if ( + transformed_message["type"] == "response.text.delta" + ): # ONLY ACCUMULATE TEXT DELTA CHUNKS - AUDIO WILL CAUSE SERVER MEMORY ISSUES + if current_delta_chunks is None: + current_delta_chunks = [] + current_delta_chunks.append( + cast(OpenAIRealtimeResponseDelta, transformed_message) + ) + else: + current_delta_chunks = None + return current_delta_chunks + except Exception as e: + raise ValueError( + f"Error updating current delta chunks: {e}, got transformed_message: {transformed_message}" + ) + + def update_current_item_chunks( + self, + transformed_message: Union[OpenAIRealtimeEvents, List[OpenAIRealtimeEvents]], + current_item_chunks: Optional[List[OpenAIRealtimeOutputItemDone]], + ) -> Optional[List[OpenAIRealtimeOutputItemDone]]: + try: + if isinstance(transformed_message, list): + current_item_chunks = [] + any_item_chunk = False + for event in transformed_message: + if event["type"] == "response.output_item.done": + current_item_chunks.append( + cast(OpenAIRealtimeOutputItemDone, event) + ) + any_item_chunk = True + if not any_item_chunk: + current_item_chunks = ( + None # reset current_item_chunks if no item chunks + ) + else: + if transformed_message["type"] == "response.output_item.done": + if current_item_chunks is None: + current_item_chunks = [] + current_item_chunks.append( + cast(OpenAIRealtimeOutputItemDone, transformed_message) + ) + else: + current_item_chunks = None + return current_item_chunks + except Exception as e: + raise ValueError( + f"Error updating current item chunks: {e}, got transformed_message: {transformed_message}" + ) + + def transform_response_done_event( + self, + message: BidiGenerateContentServerMessage, + current_response_id: Optional[str], + current_conversation_id: Optional[str], + output_items: Optional[List[OpenAIRealtimeOutputItemDone]], + session_configuration_request: Optional[str] = None, + ) -> OpenAIRealtimeDoneEvent: + if current_conversation_id is None or current_response_id is None: + raise ValueError( + f"current_conversation_id and current_response_id must all be set for a 'done' event. Got=current_conversation_id: {current_conversation_id}, current_response_id: {current_response_id}" + ) + + if session_configuration_request: + session_configuration_request_dict: BidiGenerateContentSetup = json.loads( + session_configuration_request + ).get("setup", {}) + else: + session_configuration_request_dict = {} + + generation_config = session_configuration_request_dict.get( + "generationConfig", {} + ) + temperature = generation_config.get("temperature") + max_output_tokens = generation_config.get("max_output_tokens") + gemini_modalities = generation_config.get("responseModalities", ["TEXT"]) + _modalities = [ + modality.lower() for modality in cast(List[str], gemini_modalities) + ] + if "usageMetadata" in message: + _chat_completion_usage = VertexGeminiConfig()._calculate_usage( + completion_response=message, + ) + else: + _chat_completion_usage = get_empty_usage() + + responses_api_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage( + _chat_completion_usage, + ) + response_done_event = OpenAIRealtimeDoneEvent( + type="response.done", + event_id="event_{}".format(uuid.uuid4()), + response=OpenAIRealtimeResponseDoneObject( + object="realtime.response", + id=current_response_id, + status="completed", + output=[output_item["item"] for output_item in output_items] + if output_items + else [], + conversation_id=current_conversation_id, + modalities=_modalities, + usage=responses_api_usage.model_dump(), + ), + ) + if temperature is not None: + response_done_event["response"]["temperature"] = temperature + if max_output_tokens is not None: + response_done_event["response"]["max_output_tokens"] = max_output_tokens + + return response_done_event + + def handle_openai_modality_event( + self, + openai_event: OpenAIRealtimeEventTypes, + json_message: dict, + realtime_response_transform_input: RealtimeResponseTransformInput, + delta_type: ALL_DELTA_TYPES, + ) -> RealtimeModalityResponseTransformOutput: + current_output_item_id = realtime_response_transform_input[ + "current_output_item_id" + ] + current_response_id = realtime_response_transform_input["current_response_id"] + current_conversation_id = realtime_response_transform_input[ + "current_conversation_id" + ] + current_delta_chunks = realtime_response_transform_input["current_delta_chunks"] + session_configuration_request = realtime_response_transform_input[ + "session_configuration_request" + ] + + returned_message: List[OpenAIRealtimeEvents] = [] + if ( + openai_event == OpenAIRealtimeEventTypes.RESPONSE_TEXT_DELTA + or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DELTA + ): + current_response_id = current_response_id or "resp_{}".format(uuid.uuid4()) + if not current_output_item_id: + # send the list of standard 'new' content.delta events + current_output_item_id = "item_{}".format(uuid.uuid4()) + current_conversation_id = current_conversation_id or "conv_{}".format( + uuid.uuid4() + ) + returned_message = self.return_new_content_delta_events( + session_configuration_request=session_configuration_request, + response_id=current_response_id, + output_item_id=current_output_item_id, + conversation_id=current_conversation_id, + delta_type=delta_type, + ) + + # send the list of standard 'new' content.delta events + transformed_message = self.transform_content_delta_events( + BidiGenerateContentServerContent(**json_message["serverContent"]), + current_output_item_id, + current_response_id, + delta_type=delta_type, + ) + returned_message.append(transformed_message) + elif ( + openai_event == OpenAIRealtimeEventTypes.RESPONSE_TEXT_DONE + or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DONE + ): + transformed_content_done_event = self.transform_content_done_event( + current_output_item_id=current_output_item_id, + current_response_id=current_response_id, + delta_chunks=current_delta_chunks, + delta_type=delta_type, + ) + returned_message = [transformed_content_done_event] + + additional_items = self.return_additional_content_done_events( + current_output_item_id=current_output_item_id, + current_response_id=current_response_id, + delta_done_event=transformed_content_done_event, + delta_type=delta_type, + ) + returned_message.extend(additional_items) + + return { + "returned_message": returned_message, + "current_output_item_id": current_output_item_id, + "current_response_id": current_response_id, + "current_conversation_id": current_conversation_id, + "current_delta_chunks": current_delta_chunks, + "current_delta_type": delta_type, + } + + def map_openai_event( + self, + key: str, + value: dict, + current_delta_type: Optional[ALL_DELTA_TYPES], + json_message: dict, + ) -> OpenAIRealtimeEventTypes: + model_turn_event = value.get("modelTurn") + generation_complete_event = value.get("generationComplete") + openai_event: Optional[OpenAIRealtimeEventTypes] = None + if model_turn_event: # check if model turn event + openai_event = self.map_model_turn_event(model_turn_event) + elif generation_complete_event: + openai_event = self.map_generation_complete_event( + delta_type=current_delta_type + ) + else: + # Check if this key or any nested key matches our mapping + for map_key, openai_event in MAP_GEMINI_FIELD_TO_OPENAI_EVENT.items(): + if map_key == key or ( + "." in map_key + and GeminiRealtimeConfig.get_nested_value(json_message, map_key) + is not None + ): + openai_event = openai_event + break + if openai_event is None: + raise ValueError(f"Unknown openai event: {key}, value: {value}") + return openai_event + + def transform_realtime_response( + self, + message: Union[str, bytes], + model: str, + logging_obj: LiteLLMLoggingObj, + realtime_response_transform_input: RealtimeResponseTransformInput, + ) -> RealtimeResponseTypedDict: + """ + Keep this state less - leave the state management (e.g. tracking current_output_item_id, current_response_id, current_conversation_id, current_delta_chunks) to the caller. + """ + try: + json_message = json.loads(message) + except json.JSONDecodeError: + if isinstance(message, bytes): + message_str = message.decode("utf-8", errors="replace") + else: + message_str = str(message) + raise ValueError(f"Invalid JSON message: {message_str}") + + logging_session_id = logging_obj.litellm_trace_id + + current_output_item_id = realtime_response_transform_input[ + "current_output_item_id" + ] + current_response_id = realtime_response_transform_input["current_response_id"] + current_conversation_id = realtime_response_transform_input[ + "current_conversation_id" + ] + current_delta_chunks = realtime_response_transform_input["current_delta_chunks"] + session_configuration_request = realtime_response_transform_input[ + "session_configuration_request" + ] + current_item_chunks = realtime_response_transform_input["current_item_chunks"] + current_delta_type: Optional[ + ALL_DELTA_TYPES + ] = realtime_response_transform_input["current_delta_type"] + returned_message: List[OpenAIRealtimeEvents] = [] + + for key, value in json_message.items(): + # Check if this key or any nested key matches our mapping + openai_event = self.map_openai_event( + key=key, + value=value, + current_delta_type=current_delta_type, + json_message=json_message, + ) + + if openai_event == OpenAIRealtimeEventTypes.SESSION_CREATED: + transformed_message = self.transform_session_created_event( + model, + logging_session_id, + realtime_response_transform_input["session_configuration_request"], + ) + session_configuration_request = json.dumps(transformed_message) + returned_message.append(transformed_message) + elif openai_event == OpenAIRealtimeEventTypes.RESPONSE_DONE: + transformed_response_done_event = self.transform_response_done_event( + message=BidiGenerateContentServerMessage(**json_message), # type: ignore + current_response_id=current_response_id, + current_conversation_id=current_conversation_id, + session_configuration_request=session_configuration_request, + output_items=None, + ) + returned_message.append(transformed_response_done_event) + elif ( + openai_event == OpenAIRealtimeEventTypes.RESPONSE_TEXT_DELTA + or openai_event == OpenAIRealtimeEventTypes.RESPONSE_TEXT_DONE + or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DELTA + or openai_event == OpenAIRealtimeEventTypes.RESPONSE_AUDIO_DONE + ): + _returned_message = self.handle_openai_modality_event( + openai_event, + json_message, + realtime_response_transform_input, + delta_type="text" if "text" in openai_event.value else "audio", + ) + returned_message.extend(_returned_message["returned_message"]) + current_output_item_id = _returned_message["current_output_item_id"] + current_response_id = _returned_message["current_response_id"] + current_conversation_id = _returned_message["current_conversation_id"] + current_delta_chunks = _returned_message["current_delta_chunks"] + current_delta_type = _returned_message["current_delta_type"] + else: + raise ValueError(f"Unknown openai event: {openai_event}") + if len(returned_message) == 0: + if isinstance(message, bytes): + message_str = message.decode("utf-8", errors="replace") + else: + message_str = str(message) + raise ValueError(f"Unknown message type: {message_str}") + + current_delta_chunks = self.update_current_delta_chunks( + transformed_message=returned_message, + current_delta_chunks=current_delta_chunks, + ) + current_item_chunks = self.update_current_item_chunks( + transformed_message=returned_message, + current_item_chunks=current_item_chunks, + ) + return { + "response": returned_message, + "current_output_item_id": current_output_item_id, + "current_response_id": current_response_id, + "current_delta_chunks": current_delta_chunks, + "current_conversation_id": current_conversation_id, + "current_item_chunks": current_item_chunks, + "current_delta_type": current_delta_type, + "session_configuration_request": session_configuration_request, + } + + def requires_session_configuration(self) -> bool: + return True + + def session_configuration_request(self, model: str) -> str: + """ + + ``` + { + "model": string, + "generationConfig": { + "candidateCount": integer, + "maxOutputTokens": integer, + "temperature": number, + "topP": number, + "topK": integer, + "presencePenalty": number, + "frequencyPenalty": number, + "responseModalities": [string], + "speechConfig": object, + "mediaResolution": object + }, + "systemInstruction": string, + "tools": [object] + } + ``` + """ + + response_modalities: List[GeminiResponseModalities] = ["AUDIO"] + output_audio_transcription = False + # if "audio" in model: ## UNCOMMENT THIS WHEN AUDIO IS SUPPORTED + # output_audio_transcription = True + + setup_config: BidiGenerateContentSetup = { + "model": f"models/{model}", + "generationConfig": {"responseModalities": response_modalities}, + } + if output_audio_transcription: + setup_config["outputAudioTranscription"] = {} + return json.dumps( + { + "setup": setup_config, + } + ) diff --git a/litellm/llms/groq/chat/transformation.py b/litellm/llms/groq/chat/transformation.py index 4befdc504e8..877d9a6edbd 100644 --- a/litellm/llms/groq/chat/transformation.py +++ b/litellm/llms/groq/chat/transformation.py @@ -2,7 +2,7 @@ Translate from OpenAI's `/v1/chat/completions` to Groq's `/v1/chat/completions` """ -from typing import List, Optional, Tuple, Union +from typing import Any, Coroutine, List, Literal, Optional, Tuple, Union, overload from pydantic import BaseModel @@ -65,7 +65,24 @@ class GroqChatConfig(OpenAILikeChatConfig): pass return base_params - def _transform_messages(self, messages: List[AllMessageValues], model: str) -> List: + @overload + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: Literal[True] + ) -> Coroutine[Any, Any, List[AllMessageValues]]: + ... + + @overload + def _transform_messages( + self, + messages: List[AllMessageValues], + model: str, + is_async: Literal[False] = False, + ) -> List[AllMessageValues]: + ... + + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: bool = False + ) -> Union[List[AllMessageValues], Coroutine[Any, Any, List[AllMessageValues]]]: for idx, message in enumerate(messages): """ 1. Don't pass 'null' function_call assistant message to groq - https://github.com/BerriAI/litellm/issues/5839 @@ -82,7 +99,14 @@ class GroqChatConfig(OpenAILikeChatConfig): new_message[k] = v # type: ignore messages[idx] = new_message - return messages + if is_async: + return super()._transform_messages( + messages=messages, model=model, is_async=True + ) + else: + return super()._transform_messages( + messages=messages, model=model, is_async=False + ) def _get_openai_compatible_provider_info( self, api_base: Optional[str], api_key: Optional[str] diff --git a/litellm/llms/hosted_vllm/chat/transformation.py b/litellm/llms/hosted_vllm/chat/transformation.py index e328bf2881c..529354f80eb 100644 --- a/litellm/llms/hosted_vllm/chat/transformation.py +++ b/litellm/llms/hosted_vllm/chat/transformation.py @@ -2,7 +2,7 @@ Translate from OpenAI's `/v1/chat/completions` to VLLM's `/v1/chat/completions` """ -from typing import List, Optional, Tuple, cast +from typing import Any, Coroutine, List, Literal, Optional, Tuple, Union, cast, overload from litellm.litellm_core_utils.prompt_templates.common_utils import ( _get_image_mime_type_from_url, @@ -92,9 +92,24 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): ) raise ValueError("file_id or file_data is required") + @overload def _transform_messages( - self, messages: List[AllMessageValues], model: str + self, messages: List[AllMessageValues], model: str, is_async: Literal[True] + ) -> Coroutine[Any, Any, List[AllMessageValues]]: + ... + + @overload + def _transform_messages( + self, + messages: List[AllMessageValues], + model: str, + is_async: Literal[False] = False, ) -> List[AllMessageValues]: + ... + + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: bool = False + ) -> Union[List[AllMessageValues], Coroutine[Any, Any, List[AllMessageValues]]]: """ Support translating video files from file_id or file_data to video_url """ @@ -114,5 +129,11 @@ class HostedVLLMChatConfig(OpenAIGPTConfig): message_content[idx] = self._convert_file_to_video_url( content_item ) - transformed_messages = super()._transform_messages(messages, model) - return transformed_messages + if is_async: + return super()._transform_messages( + messages, model, is_async=cast(Literal[True], True) + ) + else: + return super()._transform_messages( + messages, model, is_async=cast(Literal[False], False) + ) diff --git a/litellm/llms/litellm_proxy/chat/transformation.py b/litellm/llms/litellm_proxy/chat/transformation.py index 22013198ba6..6896b37e61d 100644 --- a/litellm/llms/litellm_proxy/chat/transformation.py +++ b/litellm/llms/litellm_proxy/chat/transformation.py @@ -4,17 +4,18 @@ Translate from OpenAI's `/v1/chat/completions` to VLLM's `/v1/chat/completions` from typing import List, Optional, Tuple -from litellm.secret_managers.main import get_secret_str +from litellm.secret_managers.main import get_secret_bool, get_secret_str +from litellm.types.router import LiteLLM_Params from ...openai.chat.gpt_transformation import OpenAIGPTConfig class LiteLLMProxyChatConfig(OpenAIGPTConfig): def get_supported_openai_params(self, model: str) -> List: - list = super().get_supported_openai_params(model) - list.append("thinking") - list.append("reasoning_effort") - return list + params_list = super().get_supported_openai_params(model) + params_list.append("thinking") + params_list.append("reasoning_effort") + return params_list def _map_openai_params( self, @@ -52,3 +53,63 @@ class LiteLLMProxyChatConfig(OpenAIGPTConfig): @staticmethod def get_api_key(api_key: Optional[str] = None) -> Optional[str]: return api_key or get_secret_str("LITELLM_PROXY_API_KEY") + + @staticmethod + def _should_use_litellm_proxy_by_default( + litellm_params: Optional[LiteLLM_Params] = None, + ): + """ + Returns True if litellm proxy should be used by default for a given request + + Issue: https://github.com/BerriAI/litellm/issues/10559 + + Use case: + - When using Google ADK, users want a flag to dynamically enable sending the request to litellm proxy or not + - Allow the model name to be passed in original format and still use litellm proxy: + "gemini/gemini-1.5-pro", "openai/gpt-4", "mistral/llama-2-70b-chat" etc. + """ + import litellm + + if get_secret_bool("USE_LITELLM_PROXY") is True: + return True + if litellm_params and litellm_params.use_litellm_proxy is True: + return True + if litellm.use_litellm_proxy is True: + return True + return False + + @staticmethod + def litellm_proxy_get_custom_llm_provider_info( + model: str, api_base: Optional[str] = None, api_key: Optional[str] = None + ) -> Tuple[str, str, Optional[str], Optional[str]]: + """ + Force use litellm proxy for all models + + Issue: https://github.com/BerriAI/litellm/issues/10559 + + Expected behavior: + - custom_llm_provider will be 'litellm_proxy' + - api_base = api_base OR LITELLM_PROXY_API_BASE + - api_key = api_key OR LITELLM_PROXY_API_KEY + + Use case: + - When using Google ADK, users want a flag to dynamically enable sending the request to litellm proxy or not + - Allow the model name to be passed in original format and still use litellm proxy: + "gemini/gemini-1.5-pro", "openai/gpt-4", "mistral/llama-2-70b-chat" etc. + + Return model, custom_llm_provider, dynamic_api_key, api_base + """ + import litellm + + custom_llm_provider = "litellm_proxy" + if model.startswith("litellm_proxy/"): + model = model.split("/", 1)[1] + + ( + api_base, + api_key, + ) = litellm.LiteLLMProxyChatConfig()._get_openai_compatible_provider_info( + api_base=api_base, api_key=api_key + ) + + return model, custom_llm_provider, api_key, api_base diff --git a/litellm/llms/lm_studio/chat/transformation.py b/litellm/llms/lm_studio/chat/transformation.py index 147e8e923f2..f7a2cc0f28a 100644 --- a/litellm/llms/lm_studio/chat/transformation.py +++ b/litellm/llms/lm_studio/chat/transformation.py @@ -18,3 +18,32 @@ class LMStudioChatConfig(OpenAIGPTConfig): api_key or get_secret_str("LM_STUDIO_API_KEY") or " " ) # vllm does not require an api key return api_base, dynamic_api_key + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + for param, value in list(non_default_params.items()): + if param == "response_format" and isinstance(value, dict): + if value.get("type") == "json_schema": + if "json_schema" not in value and "schema" in value: + optional_params["response_format"] = { + "type": "json_schema", + "json_schema": {"schema": value.get("schema")}, + } + else: + optional_params["response_format"] = value + non_default_params.pop(param, None) + elif value.get("type") == "json_object": + optional_params["response_format"] = value + non_default_params.pop(param, None) + + return super().map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model=model, + drop_params=drop_params, + ) \ No newline at end of file diff --git a/litellm/llms/meta_llama/chat/transformation.py b/litellm/llms/meta_llama/chat/transformation.py new file mode 100644 index 00000000000..aa09e330918 --- /dev/null +++ b/litellm/llms/meta_llama/chat/transformation.py @@ -0,0 +1,60 @@ +""" +Support for Llama API's `https://api.llama.com/compat/v1` endpoint. + +Calls done in OpenAI/openai.py as Llama API is openai-compatible. + +Docs: https://llama.developer.meta.com/docs/features/compatibility/ +""" + +from typing import Optional + +from litellm import get_model_info, verbose_logger +from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig + + +class LlamaAPIConfig(OpenAIGPTConfig): + def get_supported_openai_params(self, model: str) -> list: + """ + Llama API has limited support for OpenAI parameters + + Tool calling, Functional Calling, tool choice are not working right now + response_format: only json_schema is working + """ + supports_function_calling: Optional[bool] = None + supports_tool_choice: Optional[bool] = None + try: + model_info = get_model_info(model, custom_llm_provider="meta_llama") + supports_function_calling = model_info.get( + "supports_function_calling", False + ) + supports_tool_choice = model_info.get("supports_tool_choice", False) + except Exception as e: + verbose_logger.debug(f"Error getting supported openai params: {e}") + pass + + optional_params = super().get_supported_openai_params(model) + if not supports_function_calling: + optional_params.remove("function_call") + if not supports_tool_choice: + optional_params.remove("tools") + optional_params.remove("tool_choice") + return optional_params + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + mapped_openai_params = super().map_openai_params( + non_default_params, optional_params, model, drop_params + ) + + # Only json_schema is working for response_format + if ( + "response_format" in mapped_openai_params + and mapped_openai_params["response_format"].get("type") != "json_schema" + ): + mapped_openai_params.pop("response_format") + return mapped_openai_params diff --git a/litellm/llms/mistral/mistral_chat_transformation.py b/litellm/llms/mistral/mistral_chat_transformation.py index 67d88868d35..a675beebbda 100644 --- a/litellm/llms/mistral/mistral_chat_transformation.py +++ b/litellm/llms/mistral/mistral_chat_transformation.py @@ -6,7 +6,7 @@ Why separate file? Make it easy to see how transformation works Docs - https://docs.mistral.ai/api/ """ -from typing import List, Literal, Optional, Tuple, Union +from typing import Any, Coroutine, List, Literal, Optional, Tuple, Union, overload from litellm.litellm_core_utils.prompt_templates.common_utils import ( handle_messages_with_content_list_to_str_conversion, @@ -152,9 +152,24 @@ class MistralConfig(OpenAIGPTConfig): ) return api_base, dynamic_api_key + @overload def _transform_messages( - self, messages: List[AllMessageValues], model: str + self, messages: List[AllMessageValues], model: str, is_async: Literal[True] + ) -> Coroutine[Any, Any, List[AllMessageValues]]: + ... + + @overload + def _transform_messages( + self, + messages: List[AllMessageValues], + model: str, + is_async: Literal[False] = False, ) -> List[AllMessageValues]: + ... + + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: bool = False + ) -> Union[List[AllMessageValues], Coroutine[Any, Any, List[AllMessageValues]]]: """ - handles scenario where content is list and not string - content list is just text, and no images @@ -169,7 +184,10 @@ class MistralConfig(OpenAIGPTConfig): if _content_block and isinstance(_content_block, list): for c in _content_block: if c.get("type") == "image_url": - return messages + if is_async: + return super()._transform_messages(messages, model, True) + else: + return super()._transform_messages(messages, model, False) ## 2. If content is list, then convert to string messages = handle_messages_with_content_list_to_str_conversion(messages) @@ -182,7 +200,10 @@ class MistralConfig(OpenAIGPTConfig): m = strip_none_values_from_message(m) # prevents 'extra_forbidden' error new_messages.append(m) - return new_messages + if is_async: + return super()._transform_messages(new_messages, model, True) + else: + return super()._transform_messages(new_messages, model, False) @classmethod def _handle_name_in_message(cls, message: AllMessageValues) -> AllMessageValues: diff --git a/litellm/llms/novita/chat/transformation.py b/litellm/llms/novita/chat/transformation.py new file mode 100644 index 00000000000..c05d2d7b2c5 --- /dev/null +++ b/litellm/llms/novita/chat/transformation.py @@ -0,0 +1,33 @@ +""" +Support for OpenAI's `/v1/chat/completions` endpoint. + +Calls done in OpenAI/openai.py as Novita AI is openai-compatible. + +Docs: https://novita.ai/docs/guides/llm-api +""" + +from typing import List, Optional + +from ....types.llms.openai import AllMessageValues +from ...openai.chat.gpt_transformation import OpenAIGPTConfig + + +class NovitaConfig(OpenAIGPTConfig): + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + if api_key is None: + raise ValueError( + "Missing Novita AI API Key - A call is being made to novita but no key is set either in the environment variables or via params" + ) + headers["Authorization"] = f"Bearer {api_key}" + headers["Content-Type"] = "application/json" + headers["X-Novita-Source"] = "litellm" + return headers diff --git a/litellm/llms/nscale/chat/transformation.py b/litellm/llms/nscale/chat/transformation.py new file mode 100644 index 00000000000..6103b8e3c49 --- /dev/null +++ b/litellm/llms/nscale/chat/transformation.py @@ -0,0 +1,52 @@ +from typing import Optional, Tuple + +from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig +from litellm.secret_managers.main import get_secret_str + + +class NscaleConfig(OpenAIGPTConfig): + """ + Reference: Nscale is OpenAI compatible. + API Key: NSCALE_API_KEY + Default API Base: https://inference.api.nscale.com/v1 + """ + + API_BASE_URL = "https://inference.api.nscale.com/v1" + + @property + def custom_llm_provider(self) -> Optional[str]: + return "nscale" + + @staticmethod + def get_api_key(api_key: Optional[str] = None) -> Optional[str]: + return api_key or get_secret_str("NSCALE_API_KEY") + + @staticmethod + def get_api_base(api_base: Optional[str] = None) -> Optional[str]: + return ( + api_base or get_secret_str("NSCALE_API_BASE") or NscaleConfig.API_BASE_URL + ) + + def _get_openai_compatible_provider_info( + self, api_base: Optional[str], api_key: Optional[str] + ) -> Tuple[Optional[str], Optional[str]]: + # This method is called by get_llm_provider to resolve api_base and api_key + resolved_api_base = NscaleConfig.get_api_base(api_base) + resolved_api_key = NscaleConfig.get_api_key(api_key) + return resolved_api_base, resolved_api_key + + def get_supported_openai_params(self, model: str) -> list: + return [ + "max_tokens", + "n", + "temperature", + "top_p", + "stream", + "logprobs", + "top_logprobs", + "frequency_penalty", + "presence_penalty", + "response_format", + "stop", + "logit_bias", + ] diff --git a/litellm/llms/nvidia_nim/chat.py b/litellm/llms/nvidia_nim/chat/transformation.py similarity index 81% rename from litellm/llms/nvidia_nim/chat.py rename to litellm/llms/nvidia_nim/chat/transformation.py index eedac6e38fe..20478afb59f 100644 --- a/litellm/llms/nvidia_nim/chat.py +++ b/litellm/llms/nvidia_nim/chat/transformation.py @@ -7,9 +7,6 @@ This file only contains param mapping logic API calling is done using the OpenAI SDK with an api_base """ - -from typing import Optional, Union - from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig @@ -20,31 +17,6 @@ class NvidiaNimConfig(OpenAIGPTConfig): The class `NvidiaNimConfig` provides configuration for the Nvidia NIM's Chat Completions API interface. Below are the parameters: """ - temperature: Optional[int] = None - top_p: Optional[int] = None - frequency_penalty: Optional[int] = None - presence_penalty: Optional[int] = None - max_tokens: Optional[int] = None - stop: Optional[Union[str, list]] = None - - def __init__( - self, - temperature: Optional[int] = None, - top_p: Optional[int] = None, - frequency_penalty: Optional[int] = None, - presence_penalty: Optional[int] = None, - max_tokens: Optional[int] = None, - stop: Optional[Union[str, list]] = None, - ) -> None: - locals_ = locals().copy() - for key, value in locals_.items(): - if key != "self" and value is not None: - setattr(self.__class__, key, value) - - @classmethod - def get_config(cls): - return super().get_config() - def get_supported_openai_params(self, model: str) -> list: """ Get the supported OpenAI params for the given model @@ -116,6 +88,9 @@ class NvidiaNimConfig(OpenAIGPTConfig): "max_completion_tokens", "stop", "seed", + "tools", + "tool_choice", + "parallel_tool_calls", ] def map_openai_params( diff --git a/litellm/llms/ollama/completion/transformation.py b/litellm/llms/ollama/completion/transformation.py index 789b728337f..133554befeb 100644 --- a/litellm/llms/ollama/completion/transformation.py +++ b/litellm/llms/ollama/completion/transformation.py @@ -22,6 +22,7 @@ from litellm.types.utils import ( GenericStreamingChunk, ModelInfoBase, ModelResponse, + ModelResponseStream, ProviderField, ) @@ -150,6 +151,7 @@ class OllamaConfig(BaseConfig): "frequency_penalty", "stop", "response_format", + "max_completion_tokens", ] def map_openai_params( @@ -160,7 +162,7 @@ class OllamaConfig(BaseConfig): drop_params: bool, ) -> dict: for param, value in non_default_params.items(): - if param == "max_tokens": + if param == "max_tokens" or param == "max_completion_tokens": optional_params["num_predict"] = value if param == "stream": optional_params["stream"] = value @@ -256,22 +258,38 @@ class OllamaConfig(BaseConfig): ## RESPONSE OBJECT model_response.choices[0].finish_reason = "stop" if request_data.get("format", "") == "json": - function_call = json.loads(response_json["response"]) - message = litellm.Message( - content=None, - tool_calls=[ - { - "id": f"call_{str(uuid.uuid4())}", - "function": { - "name": function_call["name"], - "arguments": json.dumps(function_call["arguments"]), - }, - "type": "function", - } - ], - ) - model_response.choices[0].message = message # type: ignore - model_response.choices[0].finish_reason = "tool_calls" + response_content = json.loads(response_json["response"]) + + # Check if this is a function call format with name/arguments structure + if ( + isinstance(response_content, dict) + and "name" in response_content + and "arguments" in response_content + ): + # Handle as function call (original behavior) + function_call = response_content + message = litellm.Message( + content=None, + tool_calls=[ + { + "id": f"call_{str(uuid.uuid4())}", + "function": { + "name": function_call["name"], + "arguments": json.dumps(function_call["arguments"]), + }, + "type": "function", + } + ], + ) + model_response.choices[0].message = message # type: ignore + model_response.choices[0].finish_reason = "tool_calls" + else: + # Handle as regular JSON (new behavior) + message = litellm.Message( + content=json.dumps(response_content), + ) + model_response.choices[0].message = message # type: ignore + model_response.choices[0].finish_reason = "stop" else: model_response.choices[0].message.content = response_json["response"] # type: ignore model_response.created = int(time.time()) @@ -398,7 +416,9 @@ class OllamaConfig(BaseConfig): class OllamaTextCompletionResponseIterator(BaseModelResponseIterator): - def _handle_string_chunk(self, str_line: str) -> GenericStreamingChunk: + def _handle_string_chunk( + self, str_line: str + ) -> Union[GenericStreamingChunk, ModelResponseStream]: return self.chunk_parser(json.loads(str_line)) def chunk_parser(self, chunk: dict) -> GenericStreamingChunk: diff --git a/litellm/llms/ollama_chat.py b/litellm/llms/ollama_chat.py index 6f421680b40..22438eca082 100644 --- a/litellm/llms/ollama_chat.py +++ b/litellm/llms/ollama_chat.py @@ -156,13 +156,21 @@ class OllamaChatConfig(OpenAIGPTConfig): optional_params["repeat_penalty"] = value if param == "stop": optional_params["stop"] = value - if param == "response_format" and value["type"] == "json_object": + if ( + param == "response_format" + and isinstance(value, dict) + and value.get("type") == "json_object" + ): optional_params["format"] = "json" - if param == "response_format" and value["type"] == "json_schema": - optional_params["format"] = value["json_schema"]["schema"] + if ( + param == "response_format" + and isinstance(value, dict) + and value.get("type") == "json_schema" + ): + if value.get("json_schema") and value["json_schema"].get("schema"): + optional_params["format"] = value["json_schema"]["schema"] ### FUNCTION CALLING LOGIC ### if param == "tools": - # ollama actually supports json output ## CHECK IF MODEL SUPPORTS TOOL CALLING ## try: model_info = litellm.get_model_info( @@ -185,14 +193,23 @@ class OllamaChatConfig(OpenAIGPTConfig): ][0]["function"]["name"] if param == "functions": - # ollama actually supports json output - optional_params["format"] = "json" - litellm.add_function_to_prompt = ( - True # so that main.py adds the function call to the prompt - ) - optional_params["functions_unsupported_model"] = non_default_params.get( - "functions" - ) + ## CHECK IF MODEL SUPPORTS TOOL CALLING ## + try: + model_info = litellm.get_model_info( + model=model, custom_llm_provider="ollama" + ) + if model_info.get("supports_function_calling") is True: + optional_params["tools"] = value + else: + raise Exception + except Exception: + optional_params["format"] = "json" + litellm.add_function_to_prompt = ( + True # so that main.py adds the function call to the prompt + ) + optional_params["functions_unsupported_model"] = ( + non_default_params.get("functions") + ) non_default_params.pop("tool_choice", None) # causes ollama requests to hang non_default_params.pop("functions", None) # causes ollama requests to hang return optional_params diff --git a/litellm/llms/openai/chat/gpt_transformation.py b/litellm/llms/openai/chat/gpt_transformation.py index 9e3d9e5fc91..907da5002f0 100644 --- a/litellm/llms/openai/chat/gpt_transformation.py +++ b/litellm/llms/openai/chat/gpt_transformation.py @@ -6,11 +6,14 @@ from typing import ( TYPE_CHECKING, Any, AsyncIterator, + Coroutine, Iterator, List, + Literal, Optional, Union, cast, + overload, ) import httpx @@ -22,6 +25,10 @@ from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response impo _should_convert_tool_call_to_json_mode, ) from litellm.litellm_core_utils.prompt_templates.common_utils import get_tool_call_names +from litellm.litellm_core_utils.prompt_templates.image_handling import ( + async_convert_url_to_base64, + convert_url_to_base64, +) from litellm.llms.base_llm.base_model_iterator import BaseModelResponseIterator from litellm.llms.base_llm.base_utils import BaseLLMModelInfo from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException @@ -33,6 +40,7 @@ from litellm.types.llms.openai import ( ChatCompletionImageObject, ChatCompletionImageUrlObject, OpenAIChatCompletionChoices, + OpenAIMessageContentListBlock, ) from litellm.types.utils import ( ChatCompletionMessageToolCall, @@ -142,6 +150,7 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): "extra_headers", "parallel_tool_calls", "audio", + "web_search_options", ] # works across all models model_specific_params = [] @@ -196,42 +205,173 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): drop_params=drop_params, ) + def contains_pdf_url(self, content_item: ChatCompletionFileObjectFile) -> bool: + potential_pdf_url_starts = ["https://", "http://", "www."] + file_id = content_item.get("file_id") + if file_id and any( + file_id.startswith(start) for start in potential_pdf_url_starts + ): + return True + return False + + def _handle_pdf_url( + self, content_item: ChatCompletionFileObjectFile + ) -> ChatCompletionFileObjectFile: + content_copy = content_item.copy() + file_id = content_copy.get("file_id") + if file_id is not None: + base64_data = convert_url_to_base64(file_id) + content_copy["file_data"] = base64_data + content_copy["filename"] = "my_file.pdf" + content_copy.pop("file_id") + return content_copy + + async def _async_handle_pdf_url( + self, content_item: ChatCompletionFileObjectFile + ) -> ChatCompletionFileObjectFile: + file_id = content_item.get("file_id") + if file_id is not None: # check for file id being url done in _handle_pdf_url + base64_data = await async_convert_url_to_base64(file_id) + content_item["file_data"] = base64_data + content_item["filename"] = "my_file.pdf" + content_item.pop("file_id") + return content_item + + def _common_file_data_check( + self, content_item: ChatCompletionFileObjectFile + ) -> ChatCompletionFileObjectFile: + file_data = content_item.get("file_data") + filename = content_item.get("filename") + if file_data is not None and filename is None: + content_item["filename"] = "my_file.pdf" + return content_item + + def _apply_common_transform_content_item( + self, + content_item: OpenAIMessageContentListBlock, + ) -> OpenAIMessageContentListBlock: + litellm_specific_params = {"format"} + if content_item.get("type") == "image_url": + content_item = cast(ChatCompletionImageObject, content_item) + if isinstance(content_item["image_url"], str): + content_item["image_url"] = { + "url": content_item["image_url"], + } + elif isinstance(content_item["image_url"], dict): + new_image_url_obj = ChatCompletionImageUrlObject( + **{ # type: ignore + k: v + for k, v in content_item["image_url"].items() + if k not in litellm_specific_params + } + ) + content_item["image_url"] = new_image_url_obj + elif content_item.get("type") == "file": + content_item = cast(ChatCompletionFileObject, content_item) + file_obj = content_item["file"] + new_file_obj = ChatCompletionFileObjectFile( + **{ # type: ignore + k: v + for k, v in file_obj.items() + if k not in litellm_specific_params + } + ) + content_item["file"] = new_file_obj + + return content_item + + def _transform_content_item( + self, + content_item: OpenAIMessageContentListBlock, + ) -> OpenAIMessageContentListBlock: + content_item = self._apply_common_transform_content_item(content_item) + content_item_type = content_item.get("type") + potential_file_obj = content_item.get("file") + if content_item_type == "file" and potential_file_obj: + file_obj = cast(ChatCompletionFileObjectFile, potential_file_obj) + content_item_typed = cast(ChatCompletionFileObject, content_item) + if self.contains_pdf_url(file_obj): + file_obj = self._handle_pdf_url(file_obj) + file_obj = self._common_file_data_check(file_obj) + content_item_typed["file"] = file_obj + content_item = content_item_typed + return content_item + + async def _async_transform_content_item( + self, content_item: OpenAIMessageContentListBlock, is_async: bool = False + ) -> OpenAIMessageContentListBlock: + content_item = self._apply_common_transform_content_item(content_item) + content_item_type = content_item.get("type") + potential_file_obj = content_item.get("file") + if content_item_type == "file" and potential_file_obj: + file_obj = cast(ChatCompletionFileObjectFile, potential_file_obj) + content_item_typed = cast(ChatCompletionFileObject, content_item) + if self.contains_pdf_url(file_obj): + file_obj = await self._async_handle_pdf_url(file_obj) + file_obj = self._common_file_data_check(file_obj) + content_item_typed["file"] = file_obj + content_item = content_item_typed + return content_item + + @overload def _transform_messages( - self, messages: List[AllMessageValues], model: str + self, messages: List[AllMessageValues], model: str, is_async: Literal[True] + ) -> Coroutine[Any, Any, List[AllMessageValues]]: + ... + + @overload + def _transform_messages( + self, + messages: List[AllMessageValues], + model: str, + is_async: Literal[False] = False, ) -> List[AllMessageValues]: + ... + + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: bool = False + ) -> Union[List[AllMessageValues], Coroutine[Any, Any, List[AllMessageValues]]]: """OpenAI no longer supports image_url as a string, so we need to convert it to a dict""" - for message in messages: - message_content = message.get("content") - if message_content and isinstance(message_content, list): - for content_item in message_content: - litellm_specific_params = {"format"} - if content_item.get("type") == "image_url": - content_item = cast(ChatCompletionImageObject, content_item) - if isinstance(content_item["image_url"], str): - content_item["image_url"] = { - "url": content_item["image_url"], - } - elif isinstance(content_item["image_url"], dict): - new_image_url_obj = ChatCompletionImageUrlObject( - **{ # type: ignore - k: v - for k, v in content_item["image_url"].items() - if k not in litellm_specific_params - } - ) - content_item["image_url"] = new_image_url_obj - elif content_item.get("type") == "file": - content_item = cast(ChatCompletionFileObject, content_item) - file_obj = content_item["file"] - new_file_obj = ChatCompletionFileObjectFile( - **{ # type: ignore - k: v - for k, v in file_obj.items() - if k not in litellm_specific_params - } + + async def _async_transform(): + for message in messages: + message_content = message.get("content") + message_role = message.get("role") + if ( + message_role == "user" + and message_content + and isinstance(message_content, list) + ): + message_content_types = cast( + List[OpenAIMessageContentListBlock], message_content + ) + for i, content_item in enumerate(message_content_types): + message_content_types[ + i + ] = await self._async_transform_content_item( + cast(OpenAIMessageContentListBlock, content_item), ) - content_item["file"] = new_file_obj - return messages + return messages + + if is_async: + return _async_transform() + else: + for message in messages: + message_content = message.get("content") + message_role = message.get("role") + if ( + message_role == "user" + and message_content + and isinstance(message_content, list) + ): + message_content_types = cast( + List[OpenAIMessageContentListBlock], message_content + ) + for i, content_item in enumerate(message_content): + message_content_types[i] = self._transform_content_item( + cast(OpenAIMessageContentListBlock, content_item) + ) + return messages def transform_request( self, @@ -254,6 +394,24 @@ class OpenAIGPTConfig(BaseLLMModelInfo, BaseConfig): **optional_params, } + async def async_transform_request( + self, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + transformed_messages = await self._transform_messages( + messages=messages, model=model, is_async=True + ) + + return { + "model": model, + "messages": transformed_messages, + **optional_params, + } + def _passed_in_tools(self, optional_params: dict) -> bool: return optional_params.get("tools", None) is not None diff --git a/litellm/llms/openai/chat/o_series_transformation.py b/litellm/llms/openai/chat/o_series_transformation.py index c9a700facef..30647f58687 100644 --- a/litellm/llms/openai/chat/o_series_transformation.py +++ b/litellm/llms/openai/chat/o_series_transformation.py @@ -11,7 +11,7 @@ Translations handled by LiteLLM: - Logprobs => drop param (if user opts in to dropping param) """ -from typing import List, Optional +from typing import Any, Coroutine, List, Literal, Optional, Union, cast, overload import litellm from litellm import verbose_logger @@ -130,18 +130,29 @@ class OpenAIOSeriesConfig(OpenAIGPTConfig): ) def is_model_o_series_model(self, model: str) -> bool: - if model in litellm.open_ai_chat_completion_models and ( - "o1" in model - or "o3" in model - or "o4" - in model # [TODO] make this a more generic check (e.g. using `openai-o-series` as provider like gemini) - ): - return True - return False + model = model.split("/")[-1] # could be "openai/o3" or "o3" + return model in litellm.open_ai_chat_completion_models and any( + model.startswith(pfx) for pfx in ("o1", "o3", "o4") + ) + + @overload + def _transform_messages( + self, messages: List[AllMessageValues], model: str, is_async: Literal[True] + ) -> Coroutine[Any, Any, List[AllMessageValues]]: + ... + + @overload + def _transform_messages( + self, + messages: List[AllMessageValues], + model: str, + is_async: Literal[False] = False, + ) -> List[AllMessageValues]: + ... def _transform_messages( - self, messages: List[AllMessageValues], model: str - ) -> List[AllMessageValues]: + self, messages: List[AllMessageValues], model: str, is_async: bool = False + ) -> Union[List[AllMessageValues], Coroutine[Any, Any, List[AllMessageValues]]]: """ Handles limitations of O-1 model family. - modalities: image => drop param (if user opts in to dropping param) @@ -155,5 +166,11 @@ class OpenAIOSeriesConfig(OpenAIGPTConfig): ) messages[i] = new_message # Replace the old message with the new one - messages = super()._transform_messages(messages, model) - return messages + if is_async: + return super()._transform_messages( + messages, model, is_async=cast(Literal[True], True) + ) + else: + return super()._transform_messages( + messages, model, is_async=cast(Literal[False], False) + ) diff --git a/litellm/llms/openai/image_generation/__init__.py b/litellm/llms/openai/image_generation/__init__.py new file mode 100644 index 00000000000..eb2a0576b66 --- /dev/null +++ b/litellm/llms/openai/image_generation/__init__.py @@ -0,0 +1,22 @@ +from litellm.llms.base_llm.image_generation.transformation import ( + BaseImageGenerationConfig, +) + +from .dall_e_2_transformation import DallE2ImageGenerationConfig +from .dall_e_3_transformation import DallE3ImageGenerationConfig +from .gpt_transformation import GPTImageGenerationConfig + +__all__ = [ + "DallE2ImageGenerationConfig", + "DallE3ImageGenerationConfig", + "GPTImageGenerationConfig", +] + + +def get_openai_image_generation_config(model: str) -> BaseImageGenerationConfig: + if model.startswith("dall-e-2") or model == "": # empty model is dall-e-2 + return DallE2ImageGenerationConfig() + elif model.startswith("dall-e-3"): + return DallE3ImageGenerationConfig() + else: + return GPTImageGenerationConfig() diff --git a/litellm/llms/openai/image_generation/dall_e_2_transformation.py b/litellm/llms/openai/image_generation/dall_e_2_transformation.py new file mode 100644 index 00000000000..8e306a83375 --- /dev/null +++ b/litellm/llms/openai/image_generation/dall_e_2_transformation.py @@ -0,0 +1,38 @@ +from typing import List + +from litellm.llms.base_llm.image_generation.transformation import ( + BaseImageGenerationConfig, +) +from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams + + +class DallE2ImageGenerationConfig(BaseImageGenerationConfig): + """ + OpenAI dall-e-2 image generation config + """ + + def get_supported_openai_params( + self, model: str + ) -> List[OpenAIImageGenerationOptionalParams]: + return ["n", "response_format", "quality", "size", "user"] + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + supported_params = self.get_supported_openai_params(model) + for k in non_default_params.keys(): + if k not in optional_params.keys(): + if k in supported_params: + optional_params[k] = non_default_params[k] + elif drop_params: + pass + else: + raise ValueError( + f"Parameter {k} is not supported for model {model}. Supported parameters are {supported_params}. Set drop_params=True to drop unsupported parameters." + ) + + return optional_params diff --git a/litellm/llms/openai/image_generation/dall_e_3_transformation.py b/litellm/llms/openai/image_generation/dall_e_3_transformation.py new file mode 100644 index 00000000000..c4b0b66e112 --- /dev/null +++ b/litellm/llms/openai/image_generation/dall_e_3_transformation.py @@ -0,0 +1,38 @@ +from typing import List + +from litellm.llms.base_llm.image_generation.transformation import ( + BaseImageGenerationConfig, +) +from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams + + +class DallE3ImageGenerationConfig(BaseImageGenerationConfig): + """ + OpenAI dall-e-3 image generation config + """ + + def get_supported_openai_params( + self, model: str + ) -> List[OpenAIImageGenerationOptionalParams]: + return ["n", "response_format", "quality", "size", "user", "style"] + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + supported_params = self.get_supported_openai_params(model) + for k in non_default_params.keys(): + if k not in optional_params.keys(): + if k in supported_params: + optional_params[k] = non_default_params[k] + elif drop_params: + pass + else: + raise ValueError( + f"Parameter {k} is not supported for model {model}. Supported parameters are {supported_params}. Set drop_params=True to drop unsupported parameters." + ) + + return optional_params diff --git a/litellm/llms/openai/image_generation/gpt_transformation.py b/litellm/llms/openai/image_generation/gpt_transformation.py new file mode 100644 index 00000000000..1cee13784e7 --- /dev/null +++ b/litellm/llms/openai/image_generation/gpt_transformation.py @@ -0,0 +1,47 @@ +from typing import List + +from litellm.llms.base_llm.image_generation.transformation import ( + BaseImageGenerationConfig, +) +from litellm.types.llms.openai import OpenAIImageGenerationOptionalParams + + +class GPTImageGenerationConfig(BaseImageGenerationConfig): + """ + OpenAI gpt-image-1 image generation config + """ + + def get_supported_openai_params( + self, model: str + ) -> List[OpenAIImageGenerationOptionalParams]: + return [ + "background", + "moderation", + "n", + "output_compression", + "output_format", + "quality", + "size", + "user", + ] + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + supported_params = self.get_supported_openai_params(model) + for k in non_default_params.keys(): + if k not in optional_params.keys(): + if k in supported_params: + optional_params[k] = non_default_params[k] + elif drop_params: + pass + else: + raise ValueError( + f"Parameter {k} is not supported for model {model}. Supported parameters are {supported_params}. Set drop_params=True to drop unsupported parameters." + ) + + return optional_params diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index 13412ef96ab..e9bed019a91 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -527,6 +527,9 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): model=model, provider=LlmProviders(custom_llm_provider) ) + if provider_config is None: + provider_config = OpenAIConfig() + if provider_config: fake_stream = provider_config.should_fake_stream( model=model, custom_llm_provider=custom_llm_provider, stream=stream @@ -551,30 +554,17 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): for _ in range( 2 ): # if call fails due to alternating messages, retry with reformatted message - if provider_config is not None: - data = provider_config.transform_request( - model=model, - messages=messages, - optional_params=inference_params, - litellm_params=litellm_params, - headers=headers or {}, - ) - else: - data = OpenAIConfig().transform_request( - model=model, - messages=messages, - optional_params=inference_params, - litellm_params=litellm_params, - headers=headers or {}, - ) try: - max_retries = data.pop("max_retries", 2) + max_retries = inference_params.pop("max_retries", 2) if acompletion is True: if stream is True and fake_stream is False: return self.async_streaming( logging_obj=logging_obj, headers=headers, - data=data, + messages=messages, + optional_params=inference_params, + litellm_params=litellm_params, + provider_config=provider_config, model=model, api_base=api_base, api_key=api_key, @@ -588,7 +578,10 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): ) else: return self.acompletion( - data=data, + messages=messages, + optional_params=inference_params, + litellm_params=litellm_params, + provider_config=provider_config, headers=headers, model=model, logging_obj=logging_obj, @@ -603,7 +596,15 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): drop_params=drop_params, fake_stream=fake_stream, ) - elif stream is True and fake_stream is False: + + data = provider_config.transform_request( + model=model, + messages=messages, + optional_params=inference_params, + litellm_params=litellm_params, + headers=headers or {}, + ) + if stream is True and fake_stream is False: return self.streaming( logging_obj=logging_obj, headers=headers, @@ -741,7 +742,10 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): async def acompletion( self, - data: dict, + messages: list, + optional_params: dict, + litellm_params: dict, + provider_config: BaseConfig, model: str, model_response: ModelResponse, logging_obj: LiteLLMLoggingObj, @@ -758,6 +762,13 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): fake_stream: bool = False, ): response = None + data = await provider_config.async_transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + headers=headers or {}, + ) for _ in range( 2 ): # if call fails due to alternating messages, retry with reformatted message @@ -903,7 +914,10 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): async def async_streaming( self, timeout: Union[float, httpx.Timeout], - data: dict, + messages: list, + optional_params: dict, + litellm_params: dict, + provider_config: BaseConfig, model: str, logging_obj: LiteLLMLoggingObj, api_key: Optional[str] = None, @@ -917,6 +931,13 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): stream_options: Optional[dict] = None, ): response = None + data = provider_config.transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + headers=headers or {}, + ) data["stream"] = True data.update( self.get_stream_options(stream_options=stream_options, api_base=api_base) diff --git a/litellm/llms/openai/realtime/handler.py b/litellm/llms/openai/realtime/handler.py index 83398ad11a6..099eeab7e52 100644 --- a/litellm/llms/openai/realtime/handler.py +++ b/litellm/llms/openai/realtime/handler.py @@ -4,7 +4,7 @@ This file contains the calling Azure OpenAI's `/openai/realtime` endpoint. This requires websockets, and is currently only supported on LiteLLM Proxy. """ -from typing import Any, Optional +from typing import Any, Optional, cast from ....litellm_core_utils.litellm_logging import Logging as LiteLLMLogging from ....litellm_core_utils.realtime_streaming import RealTimeStreaming @@ -32,6 +32,7 @@ class OpenAIRealtime(OpenAIChatCompletion): timeout: Optional[float] = None, ): import websockets + from websockets.asyncio.client import ClientConnection if api_base is None: raise ValueError("api_base is required for Azure OpenAI calls") @@ -49,7 +50,7 @@ class OpenAIRealtime(OpenAIChatCompletion): }, ) as backend_ws: realtime_streaming = RealTimeStreaming( - websocket, backend_ws, logging_obj + websocket, cast(ClientConnection, backend_ws), logging_obj ) await realtime_streaming.bidirectional_forward() diff --git a/litellm/llms/sagemaker/chat/transformation.py b/litellm/llms/sagemaker/chat/transformation.py index 42c7e0d5fcf..14dde144af1 100644 --- a/litellm/llms/sagemaker/chat/transformation.py +++ b/litellm/llms/sagemaker/chat/transformation.py @@ -7,20 +7,209 @@ LiteLLM Docs: https://docs.litellm.ai/docs/providers/aws_sagemaker#sagemaker-mes Huggingface Docs: https://huggingface.co/docs/text-generation-inference/en/messages_api """ -from typing import Union +from typing import TYPE_CHECKING, Any, List, Optional, Tuple, Union, cast +import httpx from httpx._models import Headers +from litellm.litellm_core_utils.logging_utils import track_llm_api_timing +from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM +from litellm.llms.custom_httpx.http_handler import ( + AsyncHTTPHandler, + HTTPHandler, + _get_httpx_client, + get_async_httpx_client, +) +from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import LlmProviders from ...openai.chat.gpt_transformation import OpenAIGPTConfig -from ..common_utils import SagemakerError +from ..common_utils import AWSEventStreamDecoder, SagemakerError + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + + LiteLLMLoggingObj = _LiteLLMLoggingObj +else: + LiteLLMLoggingObj = Any -class SagemakerChatConfig(OpenAIGPTConfig): +class SagemakerChatConfig(OpenAIGPTConfig, BaseAWSLLM): + def __init__(self, **kwargs): + OpenAIGPTConfig.__init__(self, **kwargs) + BaseAWSLLM.__init__(self, **kwargs) + def get_error_class( self, error_message: str, status_code: int, headers: Union[dict, Headers] ) -> BaseLLMException: return SagemakerError( status_code=status_code, message=error_message, headers=headers ) + + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + return headers + + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: + aws_region_name = self._get_aws_region_name( + optional_params=optional_params, + model=model, + model_id=None, + ) + if stream is True: + api_base = f"https://runtime.sagemaker.{aws_region_name}.amazonaws.com/endpoints/{model}/invocations-response-stream" + else: + api_base = f"https://runtime.sagemaker.{aws_region_name}.amazonaws.com/endpoints/{model}/invocations" + + sagemaker_base_url = cast( + Optional[str], optional_params.get("sagemaker_base_url") + ) + if sagemaker_base_url is not None: + api_base = sagemaker_base_url + + return api_base + + def sign_request( + self, + headers: dict, + optional_params: dict, + request_data: dict, + api_base: str, + model: Optional[str] = None, + stream: Optional[bool] = None, + fake_stream: Optional[bool] = None, + ) -> Tuple[dict, Optional[bytes]]: + return self._sign_request( + service_name="sagemaker", + headers=headers, + optional_params=optional_params, + request_data=request_data, + api_base=api_base, + model=model, + stream=stream, + fake_stream=fake_stream, + ) + + @property + def has_custom_stream_wrapper(self) -> bool: + return True + + @property + def supports_stream_param_in_request_body(self) -> bool: + return False + + @track_llm_api_timing() + def get_sync_custom_stream_wrapper( + self, + model: str, + custom_llm_provider: str, + logging_obj: LiteLLMLoggingObj, + api_base: str, + headers: dict, + data: dict, + messages: list, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + json_mode: Optional[bool] = None, + signed_json_body: Optional[bytes] = None, + ) -> CustomStreamWrapper: + if client is None or isinstance(client, AsyncHTTPHandler): + client = _get_httpx_client(params={}) + + try: + response = client.post( + api_base, + headers=headers, + data=signed_json_body if signed_json_body is not None else data, + stream=True, + logging_obj=logging_obj, + ) + except httpx.HTTPStatusError as e: + raise SagemakerError( + status_code=e.response.status_code, message=e.response.text + ) + + if response.status_code != 200: + raise SagemakerError( + status_code=response.status_code, message=response.text + ) + + custom_stream_decoder = AWSEventStreamDecoder(model="", is_messages_api=True) + completion_stream = custom_stream_decoder.iter_bytes( + response.iter_bytes(chunk_size=1024) + ) + + streaming_response = CustomStreamWrapper( + completion_stream=completion_stream, + model=model, + custom_llm_provider="sagemaker_chat", + logging_obj=logging_obj, + ) + return streaming_response + + @track_llm_api_timing() + async def get_async_custom_stream_wrapper( + self, + model: str, + custom_llm_provider: str, + logging_obj: LiteLLMLoggingObj, + api_base: str, + headers: dict, + data: dict, + messages: list, + client: Optional[Union[HTTPHandler, AsyncHTTPHandler]] = None, + json_mode: Optional[bool] = None, + signed_json_body: Optional[bytes] = None, + ) -> CustomStreamWrapper: + if client is None or isinstance(client, HTTPHandler): + client = get_async_httpx_client( + llm_provider=LlmProviders.SAGEMAKER_CHAT, params={} + ) + + try: + response = await client.post( + api_base, + headers=headers, + data=signed_json_body if signed_json_body is not None else data, + stream=True, + logging_obj=logging_obj, + ) + except httpx.HTTPStatusError as e: + raise SagemakerError( + status_code=e.response.status_code, message=e.response.text + ) + + if response.status_code != 200: + raise SagemakerError( + status_code=response.status_code, message=response.text + ) + + custom_stream_decoder = AWSEventStreamDecoder(model="", is_messages_api=True) + completion_stream = custom_stream_decoder.aiter_bytes( + response.aiter_bytes(chunk_size=1024) + ) + + streaming_response = CustomStreamWrapper( + completion_stream=completion_stream, + model=model, + custom_llm_provider="sagemaker_chat", + logging_obj=logging_obj, + ) + return streaming_response diff --git a/litellm/llms/sagemaker/common_utils.py b/litellm/llms/sagemaker/common_utils.py index 031a0c7f051..ad6b24d85a3 100644 --- a/litellm/llms/sagemaker/common_utils.py +++ b/litellm/llms/sagemaker/common_utils.py @@ -34,7 +34,9 @@ class AWSEventStreamDecoder: def _chunk_parser_messages_api( self, chunk_data: dict ) -> StreamingChatCompletionChunk: - openai_chunk = StreamingChatCompletionChunk(**chunk_data) + openai_chunk = StreamingChatCompletionChunk( + **{"model": self.model, **chunk_data} + ) return openai_chunk diff --git a/litellm/llms/sambanova/chat.py b/litellm/llms/sambanova/chat.py index abf55d44fbb..57a39ec8bbc 100644 --- a/litellm/llms/sambanova/chat.py +++ b/litellm/llms/sambanova/chat.py @@ -4,7 +4,7 @@ Sambanova Chat Completions API this is OpenAI compatible - no translation needed / occurs """ -from typing import Optional +from typing import Optional, Union from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig @@ -17,26 +17,28 @@ class SambanovaConfig(OpenAIGPTConfig): """ max_tokens: Optional[int] = None - response_format: Optional[dict] = None - seed: Optional[int] = None - stream: Optional[bool] = None + temperature: Optional[int] = None top_p: Optional[int] = None + top_k: Optional[int] = None + stop: Optional[Union[str, list]] = None + stream: Optional[bool] = None + stream_options: Optional[dict] = None tool_choice: Optional[str] = None + response_format: Optional[dict] = None tools: Optional[list] = None - user: Optional[str] = None def __init__( self, max_tokens: Optional[int] = None, response_format: Optional[dict] = None, - seed: Optional[int] = None, stop: Optional[str] = None, stream: Optional[bool] = None, + stream_options: Optional[dict] = None, temperature: Optional[float] = None, - top_p: Optional[int] = None, + top_p: Optional[float] = None, + top_k: Optional[int] = None, tool_choice: Optional[str] = None, tools: Optional[list] = None, - user: Optional[str] = None, ) -> None: locals_ = locals().copy() for key, value in locals_.items(): @@ -52,16 +54,41 @@ class SambanovaConfig(OpenAIGPTConfig): Get the supported OpenAI params for the given model """ + from litellm.utils import supports_function_calling - return [ + params = [ + "max_completion_tokens", "max_tokens", "response_format", - "seed", "stop", "stream", + "stream_options", "temperature", "top_p", - "tool_choice", - "tools", - "user", + "top_k", ] + + if supports_function_calling(model, custom_llm_provider="sambanova"): + params.append("tools") + params.append("tool_choice") + params.append("parallel_tool_calls") + + return params + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + """ + map max_completion_tokens param to max_tokens + """ + supported_openai_params = self.get_supported_openai_params(model=model) + for param, value in non_default_params.items(): + if param == "max_completion_tokens": + optional_params["max_tokens"] = value + elif param in supported_openai_params: + optional_params[param] = value + return optional_params diff --git a/litellm/llms/vertex_ai/gemini/transformation.py b/litellm/llms/vertex_ai/gemini/transformation.py index e50954b8f96..39edb9642e2 100644 --- a/litellm/llms/vertex_ai/gemini/transformation.py +++ b/litellm/llms/vertex_ai/gemini/transformation.py @@ -16,6 +16,7 @@ from litellm.litellm_core_utils.prompt_templates.common_utils import ( _get_image_mime_type_from_url, ) from litellm.litellm_core_utils.prompt_templates.factory import ( + convert_generic_image_chunk_to_openai_image_obj, convert_to_anthropic_image_obj, convert_to_gemini_tool_call_invoke, convert_to_gemini_tool_call_result, @@ -45,6 +46,7 @@ from litellm.types.llms.vertex_ai import ( ToolConfig, Tools, ) +from litellm.types.utils import GenericImageParsingChunk from ..common_utils import ( _check_text_in_content, @@ -154,10 +156,26 @@ def _gemini_convert_messages_with_history( # noqa: PLR0915 _parts.append(_part) elif element["type"] == "input_audio": audio_element = cast(ChatCompletionAudioObject, element) - if audio_element["input_audio"].get("data") is not None: + audio_data = audio_element["input_audio"].get("data") + audio_format = audio_element["input_audio"].get("format") + if audio_data is not None and audio_format is not None: + audio_format_modified = ( + "audio/" + audio_format + if audio_format.startswith("audio/") is False + else audio_format + ) # Gemini expects audio/wav, audio/mp3, etc. + openai_image_str = ( + convert_generic_image_chunk_to_openai_image_obj( + image_chunk=GenericImageParsingChunk( + type="base64", + media_type=audio_format_modified, + data=audio_data, + ) + ) + ) _part = _process_gemini_image( - image_url=audio_element["input_audio"]["data"], - format=audio_element["input_audio"].get("format"), + image_url=openai_image_str, + format=audio_format_modified, ) _parts.append(_part) elif element["type"] == "file": diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index 82d06538962..cd67be3545a 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -29,7 +29,6 @@ from litellm.constants import ( DEFAULT_REASONING_EFFORT_LOW_THINKING_BUDGET, DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET, ) -from litellm.litellm_core_utils.core_helpers import map_finish_reason from litellm.llms.base_llm.chat.transformation import BaseConfig, BaseLLMException from litellm.llms.custom_httpx.http_handler import ( AsyncHTTPHandler, @@ -37,6 +36,7 @@ from litellm.llms.custom_httpx.http_handler import ( get_async_httpx_client, ) from litellm.types.llms.anthropic import AnthropicThinkingParam +from litellm.types.llms.gemini import BidiGenerateContentServerMessage from litellm.types.llms.openai import ( AllMessageValues, ChatCompletionResponseMessage, @@ -63,6 +63,7 @@ from litellm.types.llms.vertex_ai import ( from litellm.types.utils import ( ChatCompletionTokenLogprob, ChoiceLogprobs, + CompletionTokensDetailsWrapper, GenericStreamingChunk, PromptTokensDetailsWrapper, TopLogprob, @@ -337,21 +338,22 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): return old_schema def apply_response_schema_transformation(self, value: dict, optional_params: dict): + new_value = deepcopy(value) # remove 'additionalProperties' from json schema - value = _remove_additional_properties(value) + new_value = _remove_additional_properties(new_value) # remove 'strict' from json schema - value = _remove_strict_from_schema(value) - if value["type"] == "json_object": + new_value = _remove_strict_from_schema(new_value) + if new_value["type"] == "json_object": optional_params["response_mime_type"] = "application/json" - elif value["type"] == "text": + elif new_value["type"] == "text": optional_params["response_mime_type"] = "text/plain" - if "response_schema" in value: + if "response_schema" in new_value: optional_params["response_mime_type"] = "application/json" - optional_params["response_schema"] = value["response_schema"] - elif value["type"] == "json_schema": # type: ignore - if "json_schema" in value and "schema" in value["json_schema"]: # type: ignore + optional_params["response_schema"] = new_value["response_schema"] + elif new_value["type"] == "json_schema": # type: ignore + if "json_schema" in new_value and "schema" in new_value["json_schema"]: # type: ignore optional_params["response_mime_type"] = "application/json" - optional_params["response_schema"] = value["json_schema"]["schema"] # type: ignore + optional_params["response_schema"] = new_value["json_schema"]["schema"] # type: ignore if "response_schema" in optional_params and isinstance( optional_params["response_schema"], dict @@ -397,6 +399,19 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): return params + def map_response_modalities(self, value: list) -> list: + response_modalities = [] + for modality in value: + if modality == "text": + response_modalities.append("TEXT") + elif modality == "image": + response_modalities.append("IMAGE") + elif modality == "audio": + response_modalities.append("AUDIO") + else: + response_modalities.append("MODALITY_UNSPECIFIED") + return response_modalities + def map_openai_params( self, non_default_params: Dict, @@ -464,14 +479,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): cast(AnthropicThinkingParam, value) ) elif param == "modalities" and isinstance(value, list): - response_modalities = [] - for modality in value: - if modality == "text": - response_modalities.append("TEXT") - elif modality == "image": - response_modalities.append("IMAGE") - else: - response_modalities.append("MODALITY_UNSPECIFIED") + response_modalities = self.map_response_modalities(value) optional_params["responseModalities"] = response_modalities if litellm.vertex_ai_safety_settings is not None: @@ -564,6 +572,28 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): "BLOCKLIST": "The token generation was stopped as the response was flagged for the terms which are included from the terminology blocklist.", "PROHIBITED_CONTENT": "The token generation was stopped as the response was flagged for the prohibited contents.", "SPII": "The token generation was stopped as the response was flagged for Sensitive Personally Identifiable Information (SPII) contents.", + "IMAGE_SAFETY": "The token generation was stopped as the response was flagged for image safety reasons.", + } + + def get_finish_reason_mapping(self) -> Dict[str, OpenAIChatCompletionFinishReason]: + """ + Return Dictionary of finish reasons which indicate response was flagged + + and what it means + """ + return { + "FINISH_REASON_UNSPECIFIED": "stop", # openai doesn't have a way of representing this + "STOP": "stop", + "MAX_TOKENS": "length", + "SAFETY": "content_filter", + "RECITATION": "content_filter", + "LANGUAGE": "content_filter", + "OTHER": "content_filter", + "BLOCKLIST": "content_filter", + "PROHIBITED_CONTENT": "content_filter", + "SPII": "content_filter", + "MALFORMED_FUNCTION_CALL": "stop", # openai doesn't have a way of representing this + "IMAGE_SAFETY": "content_filter", } def translate_exception_str(self, exception_string: str): @@ -761,17 +791,38 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): def _calculate_usage( self, - completion_response: GenerateContentResponseBody, + completion_response: Union[ + GenerateContentResponseBody, BidiGenerateContentServerMessage + ], ) -> Usage: + if "usageMetadata" not in completion_response: + raise ValueError( + f"usageMetadata not found in completion_response. Got={completion_response}" + ) cached_tokens: Optional[int] = None audio_tokens: Optional[int] = None text_tokens: Optional[int] = None prompt_tokens_details: Optional[PromptTokensDetailsWrapper] = None reasoning_tokens: Optional[int] = None + response_tokens: Optional[int] = None + response_tokens_details: Optional[CompletionTokensDetailsWrapper] = None if "cachedContentTokenCount" in completion_response["usageMetadata"]: cached_tokens = completion_response["usageMetadata"][ "cachedContentTokenCount" ] + + ## GEMINI LIVE API ONLY PARAMS ## + if "responseTokenCount" in completion_response["usageMetadata"]: + response_tokens = completion_response["usageMetadata"]["responseTokenCount"] + if "responseTokensDetails" in completion_response["usageMetadata"]: + response_tokens_details = CompletionTokensDetailsWrapper() + for detail in completion_response["usageMetadata"]["responseTokensDetails"]: + if detail["modality"] == "TEXT": + response_tokens_details.text_tokens = detail["tokenCount"] + elif detail["modality"] == "AUDIO": + response_tokens_details.audio_tokens = detail["tokenCount"] + ######################################################### + if "promptTokensDetails" in completion_response["usageMetadata"]: for detail in completion_response["usageMetadata"]["promptTokensDetails"]: if detail["modality"] == "AUDIO": @@ -788,7 +839,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): text_tokens=text_tokens, ) - completion_tokens = completion_response["usageMetadata"].get( + completion_tokens = response_tokens or completion_response["usageMetadata"].get( "candidatesTokenCount", 0 ) if ( @@ -807,23 +858,25 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): total_tokens=completion_response["usageMetadata"].get("totalTokenCount", 0), prompt_tokens_details=prompt_tokens_details, reasoning_tokens=reasoning_tokens, + completion_tokens_details=response_tokens_details, ) return usage def _check_finish_reason( self, - chat_completion_message: ChatCompletionResponseMessage, + chat_completion_message: Optional[ChatCompletionResponseMessage], finish_reason: Optional[str], ) -> OpenAIChatCompletionFinishReason: - if chat_completion_message.get("function_call"): + mapped_finish_reason = self.get_finish_reason_mapping() + if chat_completion_message and chat_completion_message.get("function_call"): return "function_call" - elif chat_completion_message.get("tool_calls"): + elif chat_completion_message and chat_completion_message.get("tool_calls"): return "tool_calls" - elif finish_reason and ( - finish_reason == "SAFETY" or finish_reason == "RECITATION" + elif ( + finish_reason and finish_reason in mapped_finish_reason.keys() ): # vertex ai - return "content_filter" + return mapped_finish_reason[finish_reason] else: return "stop" @@ -1579,8 +1632,9 @@ class ModelResponseIterator: ) if gemini_chunk and "finishReason" in gemini_chunk: - finish_reason = map_finish_reason( - finish_reason=gemini_chunk["finishReason"] + finish_reason = VertexGeminiConfig()._check_finish_reason( + chat_completion_message=None, + finish_reason=gemini_chunk["finishReason"], ) ## DO NOT SET 'is_finished' = True ## GEMINI SETS FINISHREASON ON EVERY CHUNK! @@ -1596,6 +1650,11 @@ class ModelResponseIterator: total_tokens=processed_chunk["usageMetadata"].get( "totalTokenCount", 0 ), + completion_tokens_details={ + "reasoning_tokens": processed_chunk["usageMetadata"].get( + "thoughtsTokenCount", 0 + ) + }, ) returned_chunk = GenericStreamingChunk( diff --git a/litellm/llms/vertex_ai/vertex_llm_base.py b/litellm/llms/vertex_ai/vertex_llm_base.py index 8f3037c7911..9349fb56da9 100644 --- a/litellm/llms/vertex_ai/vertex_llm_base.py +++ b/litellm/llms/vertex_ai/vertex_llm_base.py @@ -40,15 +40,7 @@ class VertexBase: def load_auth( self, credentials: Optional[VERTEX_CREDENTIALS_TYPES], project_id: Optional[str] ) -> Tuple[Any, str]: - import google.auth as google_auth - from google.auth import identity_pool - from google.auth.transport.requests import ( - Request, # type: ignore[import-untyped] - ) - if credentials is not None: - import google.oauth2.service_account - if isinstance(credentials, str): verbose_logger.debug( "Vertex: Loading vertex credentials from %s", credentials @@ -80,26 +72,33 @@ class VertexBase: # Check if the JSON object contains Workload Identity Federation configuration if "type" in json_obj and json_obj["type"] == "external_account": - creds = identity_pool.Credentials.from_info(json_obj) + creds = self._credentials_from_identity_pool(json_obj) + # Check if the JSON object contains Authorized User configuration (via gcloud auth application-default login) + elif "type" in json_obj and json_obj["type"] == "authorized_user": + creds = self._credentials_from_authorized_user( + json_obj, + scopes=["https://www.googleapis.com/auth/cloud-platform"], + ) + if project_id is None: + project_id = ( + creds.quota_project_id + ) # authorized user credentials don't have a project_id, only quota_project_id else: - creds = ( - google.oauth2.service_account.Credentials.from_service_account_info( - json_obj, - scopes=["https://www.googleapis.com/auth/cloud-platform"], - ) + creds = self._credentials_from_service_account( + json_obj, + scopes=["https://www.googleapis.com/auth/cloud-platform"], ) if project_id is None: project_id = getattr(creds, "project_id", None) else: - creds, creds_project_id = google_auth.default( - quota_project_id=project_id, - scopes=["https://www.googleapis.com/auth/cloud-platform"], + creds, creds_project_id = self._credentials_from_default_auth( + scopes=["https://www.googleapis.com/auth/cloud-platform"] ) if project_id is None: project_id = creds_project_id - creds.refresh(Request()) # type: ignore + self.refresh_auth(creds) if not project_id: raise ValueError("Could not resolve project_id") @@ -111,6 +110,31 @@ class VertexBase: return creds, project_id + # Google Auth Helpers -- extracted for mocking purposes in tests + def _credentials_from_identity_pool(self, json_obj): + from google.auth import identity_pool + + return identity_pool.Credentials.from_info(json_obj) + + def _credentials_from_authorized_user(self, json_obj, scopes): + import google.oauth2.credentials + + return google.oauth2.credentials.Credentials.from_authorized_user_info( + json_obj, scopes=scopes + ) + + def _credentials_from_service_account(self, json_obj, scopes): + import google.oauth2.service_account + + return google.oauth2.service_account.Credentials.from_service_account_info( + json_obj, scopes=scopes + ) + + def _credentials_from_default_auth(self, scopes): + import google.auth as google_auth + + return google_auth.default(scopes=scopes) + def refresh_auth(self, credentials: Any) -> None: from google.auth.transport.requests import ( Request, # type: ignore[import-untyped] @@ -288,7 +312,7 @@ class VertexBase: ) except Exception as e: verbose_logger.exception( - "Failed to load vertex credentials. Check to see if credentials containing partial/invalid information." + f"Failed to load vertex credentials. Check to see if credentials containing partial/invalid information. Error: {str(e)}" ) raise e @@ -304,16 +328,6 @@ class VertexBase: ## VALIDATE CREDENTIALS verbose_logger.debug(f"Validating credentials for project_id: {project_id}") if ( - project_id is not None - and credential_project_id - and credential_project_id != project_id - ): - raise ValueError( - "Could not resolve project_id. Credential project_id: {} does not match requested project_id: {}".format( - _credentials.quota_project_id, project_id - ) - ) - elif ( project_id is None and credential_project_id is not None and isinstance(credential_project_id, str) diff --git a/litellm/main.py b/litellm/main.py index 9487de5043e..68589d7127a 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -34,6 +34,7 @@ from typing import ( Type, Union, cast, + get_args, ) import dotenv @@ -73,7 +74,7 @@ from litellm.litellm_core_utils.mock_functions import ( from litellm.litellm_core_utils.prompt_templates.common_utils import ( get_content_from_model_response, ) -from litellm.llms.base_llm.chat.transformation import BaseConfig +from litellm.llms.base_llm import BaseConfig, BaseImageGenerationConfig from litellm.llms.bedrock.common_utils import BedrockModelInfo from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.realtime_api.main import _realtime_health_check @@ -184,6 +185,7 @@ from .types.llms.openai import ( HttpxBinaryResponseContent, ImageGenerationRequestQuality, OpenAIModerationResponse, + OpenAIWebSearchOptions, ) from .types.utils import ( LITELLM_IMAGE_VARIATION_PROVIDERS, @@ -352,6 +354,7 @@ async def acompletion( extra_headers: Optional[dict] = None, # Optional liteLLM function params thinking: Optional[AnthropicThinkingParam] = None, + web_search_options: Optional[OpenAIWebSearchOptions] = None, **kwargs, ) -> Union[ModelResponse, CustomStreamWrapper]: """ @@ -471,6 +474,7 @@ async def acompletion( "extra_headers": extra_headers, "acompletion": True, # assuming this is a required parameter "thinking": thinking, + "web_search_options": web_search_options, } if custom_llm_provider is None: _, custom_llm_provider, _, _ = get_llm_provider( @@ -834,6 +838,7 @@ def completion( # type: ignore # noqa: PLR0915 logprobs: Optional[bool] = None, top_logprobs: Optional[int] = None, parallel_tool_calls: Optional[bool] = None, + web_search_options: Optional[OpenAIWebSearchOptions] = None, deployment_id=None, extra_headers: Optional[dict] = None, # soon to be deprecated params by OpenAI @@ -1167,6 +1172,7 @@ def completion( # type: ignore # noqa: PLR0915 messages=messages, reasoning_effort=reasoning_effort, thinking=thinking, + web_search_options=web_search_options, allowed_openai_params=kwargs.get("allowed_openai_params"), **non_default_params, ) @@ -1220,6 +1226,7 @@ def completion( # type: ignore # noqa: PLR0915 merge_reasoning_content_in_choices=kwargs.get( "merge_reasoning_content_in_choices", None ), + use_litellm_proxy=kwargs.get("use_litellm_proxy", False), api_version=api_version, azure_ad_token=kwargs.get("azure_ad_token"), tenant_id=kwargs.get("tenant_id"), @@ -2663,19 +2670,21 @@ def completion( # type: ignore # noqa: PLR0915 response = _model_response elif custom_llm_provider == "sagemaker_chat": # boto3 reads keys from .env - model_response = sagemaker_chat_completion.completion( + model_response = base_llm_http_handler.completion( model=model, + stream=stream, messages=messages, + acompletion=acompletion, + api_base=api_base, model_response=model_response, - print_verbose=print_verbose, optional_params=optional_params, litellm_params=litellm_params, + custom_llm_provider="sagemaker_chat", timeout=timeout, - custom_prompt_dict=custom_prompt_dict, - logger_fn=logger_fn, + headers=headers, encoding=encoding, - logging_obj=logging, - acompletion=acompletion, + api_key=api_key, + logging_obj=logging, # model call logging done inside the class as we make need to modify I/O to fit aleph alpha's requirements client=client, ) @@ -2722,9 +2731,9 @@ def completion( # type: ignore # noqa: PLR0915 "aws_region_name" not in optional_params or optional_params["aws_region_name"] is None ): - optional_params["aws_region_name"] = ( - aws_bedrock_client.meta.region_name - ) + optional_params[ + "aws_region_name" + ] = aws_bedrock_client.meta.region_name bedrock_route = BedrockModelInfo.get_bedrock_route(model) if bedrock_route == "converse": @@ -3324,7 +3333,6 @@ async def aembedding(*args, **kwargs) -> EmbeddingResponse: response = init_response elif asyncio.iscoroutine(init_response): response = await init_response # type: ignore - if ( response is not None and isinstance(response, EmbeddingResponse) @@ -3646,8 +3654,8 @@ def embedding( # noqa: PLR0915 cohere_key = ( api_key or litellm.cohere_key - or get_secret("COHERE_API_KEY") - or get_secret("CO_API_KEY") + or get_secret_str("COHERE_API_KEY") + or get_secret_str("CO_API_KEY") or litellm.api_key ) @@ -3655,18 +3663,21 @@ def embedding( # noqa: PLR0915 headers = extra_headers else: headers = {} - response = cohere_embed.embedding( + + response = base_llm_http_handler.embedding( model=model, input=input, - optional_params=optional_params, - encoding=encoding, - api_key=cohere_key, # type: ignore - headers=headers, + custom_llm_provider=custom_llm_provider, + api_base=api_base, + api_key=cohere_key, logging_obj=logging, - model_response=EmbeddingResponse(), - aembedding=aembedding, timeout=timeout, + model_response=EmbeddingResponse(), + optional_params=optional_params, client=client, + aembedding=aembedding, + litellm_params=litellm_params_dict, + headers=headers, ) elif custom_llm_provider == "huggingface": api_key = ( @@ -4448,9 +4459,9 @@ def adapter_completion( new_kwargs = translation_obj.translate_completion_input_params(kwargs=kwargs) response: Union[ModelResponse, CustomStreamWrapper] = completion(**new_kwargs) # type: ignore - translated_response: Optional[Union[BaseModel, AdapterCompletionStreamWrapper]] = ( - None - ) + translated_response: Optional[ + Union[BaseModel, AdapterCompletionStreamWrapper] + ] = None if isinstance(response, ModelResponse): translated_response = translation_obj.translate_completion_output_params( response=response @@ -4658,9 +4669,11 @@ def image_generation( # noqa: PLR0915 client = kwargs.get("client", None) extra_headers = kwargs.get("extra_headers", None) headers: dict = kwargs.get("headers", None) or {} + base_model = kwargs.get("base_model", None) if extra_headers is not None: headers.update(extra_headers) model_response: ImageResponse = litellm.utils.ImageResponse() + dynamic_api_key: Optional[str] = None if model is not None or custom_llm_provider is not None: model, custom_llm_provider, dynamic_api_key, api_base = get_llm_provider( model=model, # type: ignore @@ -4694,8 +4707,20 @@ def image_generation( # noqa: PLR0915 k: v for k, v in kwargs.items() if k not in default_params } # model-specific params - pass them straight to the model/provider + image_generation_config: Optional[BaseImageGenerationConfig] = None + if ( + custom_llm_provider is not None + and custom_llm_provider in LlmProviders._member_map_.values() + ): + image_generation_config = ( + ProviderConfigManager.get_provider_image_generation_config( + model=base_model or model, + provider=LlmProviders(custom_llm_provider), + ) + ) + optional_params = get_optional_params_image_gen( - model=model, + model=base_model or model, n=n, quality=quality, response_format=response_format, @@ -4703,6 +4728,7 @@ def image_generation( # noqa: PLR0915 style=style, user=user, custom_llm_provider=custom_llm_provider, + provider_config=image_generation_config, **non_default_params, ) @@ -4788,7 +4814,7 @@ def image_generation( # noqa: PLR0915 model=model, prompt=prompt, timeout=timeout, - api_key=api_key, + api_key=api_key or dynamic_api_key, api_base=api_base, logging_obj=litellm_logging_obj, optional_params=optional_params, @@ -5372,6 +5398,7 @@ def speech( # noqa: PLR0915 timeout: Optional[Union[float, httpx.Timeout]] = None, response_format: Optional[str] = None, speed: Optional[int] = None, + instructions: Optional[str] = None, client=None, headers: Optional[dict] = None, custom_llm_provider: Optional[str] = None, @@ -5393,7 +5420,8 @@ def speech( # noqa: PLR0915 optional_params["response_format"] = response_format if speed is not None: optional_params["speed"] = speed # type: ignore - + if instructions is not None: + optional_params["instructions"] = instructions if timeout is None: timeout = litellm.request_timeout @@ -5901,9 +5929,9 @@ def stream_chunk_builder( # noqa: PLR0915 ] if len(content_chunks) > 0: - response["choices"][0]["message"]["content"] = ( - processor.get_combined_content(content_chunks) - ) + response["choices"][0]["message"][ + "content" + ] = processor.get_combined_content(content_chunks) reasoning_chunks = [ chunk @@ -5914,9 +5942,9 @@ def stream_chunk_builder( # noqa: PLR0915 ] if len(reasoning_chunks) > 0: - response["choices"][0]["message"]["reasoning_content"] = ( - processor.get_combined_reasoning_content(reasoning_chunks) - ) + response["choices"][0]["message"][ + "reasoning_content" + ] = processor.get_combined_reasoning_content(reasoning_chunks) audio_chunks = [ chunk diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c148f04e336..a1fd7f7366b 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -79,6 +79,7 @@ "supported_endpoints": ["/v1/chat/completions", "/v1/batch", "/v1/responses"], "supported_modalities": ["text", "image"], "supported_output_modalities": ["text"], + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -108,6 +109,7 @@ "supported_endpoints": ["/v1/chat/completions", "/v1/batch", "/v1/responses"], "supported_modalities": ["text", "image"], "supported_output_modalities": ["text"], + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -137,6 +139,7 @@ "supported_endpoints": ["/v1/chat/completions", "/v1/batch", "/v1/responses"], "supported_modalities": ["text", "image"], "supported_output_modalities": ["text"], + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -166,6 +169,7 @@ "supported_endpoints": ["/v1/chat/completions", "/v1/batch", "/v1/responses"], "supported_modalities": ["text", "image"], "supported_output_modalities": ["text"], + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -195,6 +199,7 @@ "supported_endpoints": ["/v1/chat/completions", "/v1/batch", "/v1/responses"], "supported_modalities": ["text", "image"], "supported_output_modalities": ["text"], + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -218,6 +223,7 @@ "supported_endpoints": ["/v1/chat/completions", "/v1/batch", "/v1/responses"], "supported_modalities": ["text", "image"], "supported_output_modalities": ["text"], + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -238,6 +244,7 @@ "cache_read_input_token_cost": 0.00000125, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -281,6 +288,7 @@ "cache_read_input_token_cost": 0.00000125, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -306,6 +314,7 @@ "cache_read_input_token_cost": 0.00000125, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -331,6 +340,7 @@ "cache_read_input_token_cost": 0.0000375, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -350,6 +360,7 @@ "cache_read_input_token_cost": 0.0000375, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -438,6 +449,7 @@ "cache_read_input_token_cost": 0.000000075, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -463,6 +475,7 @@ "cache_read_input_token_cost": 0.000000075, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -488,6 +501,7 @@ "cache_read_input_token_cost": 0.000000075, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -513,6 +527,7 @@ "cache_read_input_token_cost": 0.000000075, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -536,6 +551,7 @@ "output_cost_per_token_batches": 0.0003, "litellm_provider": "openai", "mode": "responses", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_vision": true, @@ -559,6 +575,7 @@ "output_cost_per_token_batches": 0.0003, "litellm_provider": "openai", "mode": "responses", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_vision": true, @@ -584,6 +601,7 @@ "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_vision": true, + "supports_pdf_input": true, "supports_prompt_caching": true, "supports_system_messages": true, "supports_response_schema": true, @@ -600,6 +618,7 @@ "litellm_provider": "openai", "mode": "chat", "supports_vision": true, + "supports_pdf_input": true, "supports_prompt_caching": true }, "computer-use-preview": { @@ -634,6 +653,7 @@ "supports_function_calling": true, "supports_parallel_function_calling": false, "supports_vision": true, + "supports_pdf_input": true, "supports_prompt_caching": true, "supports_response_schema": true, "supports_reasoning": true, @@ -651,6 +671,7 @@ "supports_function_calling": true, "supports_parallel_function_calling": false, "supports_vision": true, + "supports_pdf_input": true, "supports_prompt_caching": true, "supports_response_schema": true, "supports_reasoning": true, @@ -699,6 +720,7 @@ "cache_read_input_token_cost": 2.75e-7, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": false, "supports_vision": true, @@ -716,6 +738,7 @@ "cache_read_input_token_cost": 2.75e-7, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": false, "supports_vision": true, @@ -733,6 +756,7 @@ "cache_read_input_token_cost": 0.0000015, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_vision": true, "supports_reasoning": true, "supports_prompt_caching": true @@ -746,6 +770,7 @@ "cache_read_input_token_cost": 0.0000075, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_vision": true, "supports_reasoning": true, "supports_prompt_caching": true @@ -759,6 +784,7 @@ "cache_read_input_token_cost": 0.0000075, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_vision": true, "supports_reasoning": true, "supports_prompt_caching": true @@ -772,6 +798,7 @@ "cache_read_input_token_cost": 0.0000075, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_vision": true, @@ -789,6 +816,7 @@ "output_cost_per_token": 0.000015, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_vision": true, @@ -806,6 +834,7 @@ "output_cost_per_token_batches": 0.0000075, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_vision": true, @@ -824,6 +853,7 @@ "cache_read_input_token_cost": 0.00000125, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -849,6 +879,7 @@ "cache_read_input_token_cost": 0.00000125, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -958,6 +989,7 @@ "output_cost_per_token": 0.00003, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_prompt_caching": true, @@ -1034,6 +1066,7 @@ "output_cost_per_token": 0.00003, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_vision": true, @@ -1049,6 +1082,7 @@ "output_cost_per_token": 0.00003, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_vision": true, @@ -1093,6 +1127,7 @@ "litellm_provider": "openai", "mode": "chat", "supports_vision": true, + "supports_pdf_input": true, "supports_prompt_caching": true, "supports_system_messages": true, "deprecation_date": "2024-12-06", @@ -1107,6 +1142,7 @@ "litellm_provider": "openai", "mode": "chat", "supports_vision": true, + "supports_pdf_input": true, "supports_prompt_caching": true, "supports_system_messages": true, "deprecation_date": "2024-12-06", @@ -1271,6 +1307,7 @@ "output_cost_per_token_batches": 0.000007500, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -1287,6 +1324,7 @@ "output_cost_per_token": 0.000015, "litellm_provider": "openai", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_response_schema": true, @@ -1310,6 +1348,7 @@ "supports_parallel_function_calling": true, "supports_response_schema": true, "supports_vision": true, + "supports_pdf_input": true, "supports_prompt_caching": true, "supports_system_messages": true, "supports_tool_choice": true @@ -1578,6 +1617,17 @@ "supported_output_modalities": ["audio"], "supported_endpoints": ["/v1/audio/speech"] }, + "azure/gpt-4o-mini-tts": { + "mode": "audio_speech", + "input_cost_per_token": 2.5e-6, + "output_cost_per_token": 10e-6, + "output_cost_per_audio_token": 12e-6, + "output_cost_per_second": 0.00025, + "litellm_provider": "azure", + "supported_modalities": ["text", "audio"], + "supported_output_modalities": ["audio"], + "supported_endpoints": ["/v1/audio/speech"] + }, "azure/computer-use-preview": { "max_tokens": 1024, "max_input_tokens": 8192, @@ -2284,6 +2334,7 @@ "cache_read_input_token_cost": 0.0000075, "litellm_provider": "azure", "mode": "chat", + "supports_pdf_input": true, "supports_function_calling": true, "supports_parallel_function_calling": true, "supports_vision": false, @@ -3033,6 +3084,18 @@ "supports_tool_choice": true, "source": "https://techcommunity.microsoft.com/blog/machinelearningblog/announcing-deepseek-v3-on-azure-ai-foundry-and-github/4390438" }, + "azure_ai/deepseek-v3-0324": { + "max_tokens": 8192, + "max_input_tokens": 128000, + "max_output_tokens": 8192, + "input_cost_per_token": 0.00000114, + "output_cost_per_token": 0.00000456, + "litellm_provider": "azure_ai", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "source": "https://techcommunity.microsoft.com/blog/machinelearningblog/announcing-deepseek-v3-on-azure-ai-foundry-and-github/4390438" + }, "azure_ai/jamba-instruct": { "max_tokens": 4096, "max_input_tokens": 70000, @@ -3149,6 +3212,32 @@ "source": "https://azuremarketplace.microsoft.com/en/marketplace/apps/metagenai.llama-3-3-70b-instruct-offer?tab=Overview", "supports_tool_choice": true }, + "azure_ai/Llama-4-Scout-17B-16E-Instruct": { + "max_tokens": 16384, + "max_input_tokens": 10000000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.0000002, + "output_cost_per_token": 0.00000078, + "litellm_provider": "azure_ai", + "supports_function_calling": true, + "supports_vision": true, + "mode": "chat", + "source": "https://azure.microsoft.com/en-us/blog/introducing-the-llama-4-herd-in-azure-ai-foundry-and-azure-databricks/", + "supports_tool_choice": true + }, + "azure_ai/Llama-4-Maverick-17B-128E-Instruct-FP8": { + "max_tokens": 16384, + "max_input_tokens": 1000000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.00000141, + "output_cost_per_token": 0.00000035, + "litellm_provider": "azure_ai", + "supports_function_calling": true, + "supports_vision": true, + "mode": "chat", + "source": "https://azure.microsoft.com/en-us/blog/introducing-the-llama-4-herd-in-azure-ai-foundry-and-azure-databricks/", + "supports_tool_choice": true + }, "azure_ai/Llama-3.2-90B-Vision-Instruct": { "max_tokens": 2048, "max_input_tokens": 128000, @@ -3395,6 +3484,19 @@ "supports_embedding_image_input": true, "source":"https://azuremarketplace.microsoft.com/en-us/marketplace/apps/cohere.cohere-embed-v3-english-offer?tab=PlansAndPrice" }, + "azure_ai/embed-v-4-0": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "output_vector_size": 3072, + "input_cost_per_token": 0.00000012, + "output_cost_per_token": 0.0, + "litellm_provider": "azure_ai", + "mode": "embedding", + "supports_embedding_image_input": true, + "supported_endpoints": ["/v1/embeddings"], + "supported_modalities": ["text", "image"], + "source":"https://azuremarketplace.microsoft.com/pt-br/marketplace/apps/cohere.cohere-embed-4-offer?tab=PlansAndPrice" + }, "babbage-002": { "max_tokens": 16384, "max_input_tokens": 16384, @@ -3976,25 +4078,24 @@ "supports_prompt_caching": true }, "groq/deepseek-r1-distill-llama-70b": { - "max_tokens": 131072, - "max_input_tokens": 131072, - "max_output_tokens": 131072, - "input_cost_per_token": 0.00000075, - "output_cost_per_token": 0.00000099, + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 7.5e-07, + "output_cost_per_token": 9.9e-07, "litellm_provider": "groq", "mode": "chat", - "supports_system_messages": false, - "supports_function_calling": false, + "supports_function_calling": true, + "supports_response_schema": true, "supports_reasoning": true, - "supports_response_schema": false, "supports_tool_choice": true }, "groq/llama-3.3-70b-versatile": { - "max_tokens": 8192, + "max_tokens": 32768, "max_input_tokens": 128000, - "max_output_tokens": 8192, - "input_cost_per_token": 0.00000059, - "output_cost_per_token": 0.00000079, + "max_output_tokens": 32768, + "input_cost_per_token": 5.9e-07, + "output_cost_per_token": 7.9e-07, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, @@ -4005,11 +4106,21 @@ "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000059, - "output_cost_per_token": 0.00000099, + "input_cost_per_token": 5.9e-07, + "output_cost_per_token": 9.9e-07, "litellm_provider": "groq", "mode": "chat", - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2025-04-14" + }, + "groq/llama-guard-3-8b": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 2e-07, + "litellm_provider": "groq", + "mode": "chat" }, "groq/llama2-70b-4096": { "max_tokens": 4096, @@ -4027,106 +4138,109 @@ "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000005, - "output_cost_per_token": 0.00000008, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 8e-08, "litellm_provider": "groq", "mode": "chat", - "supports_function_calling": true, - "supports_response_schema": true, "supports_tool_choice": true }, "groq/llama-3.2-1b-preview": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000004, - "output_cost_per_token": 0.00000004, + "input_cost_per_token": 4e-08, + "output_cost_per_token": 4e-08, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2025-04-14" }, "groq/llama-3.2-3b-preview": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000006, - "output_cost_per_token": 0.00000006, + "input_cost_per_token": 6e-08, + "output_cost_per_token": 6e-08, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2025-04-14" }, "groq/llama-3.2-11b-text-preview": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000018, - "output_cost_per_token": 0.00000018, + "input_cost_per_token": 1.8e-07, + "output_cost_per_token": 1.8e-07, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2024-10-28" }, "groq/llama-3.2-11b-vision-preview": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000018, - "output_cost_per_token": 0.00000018, + "input_cost_per_token": 1.8e-07, + "output_cost_per_token": 1.8e-07, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, "supports_vision": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2025-04-14" }, "groq/llama-3.2-90b-text-preview": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.0000009, - "output_cost_per_token": 0.0000009, + "input_cost_per_token": 9e-07, + "output_cost_per_token": 9e-07, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2024-11-25" }, "groq/llama-3.2-90b-vision-preview": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.0000009, - "output_cost_per_token": 0.0000009, + "input_cost_per_token": 9e-07, + "output_cost_per_token": 9e-07, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, "supports_vision": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2025-04-14" }, "groq/llama3-70b-8192": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000059, - "output_cost_per_token": 0.00000079, + "input_cost_per_token": 5.9e-07, + "output_cost_per_token": 7.9e-07, "litellm_provider": "groq", "mode": "chat", - "supports_function_calling": true, "supports_response_schema": true, "supports_tool_choice": true }, "groq/llama-3.1-8b-instant": { "max_tokens": 8192, - "max_input_tokens": 8192, + "max_input_tokens": 128000, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000005, - "output_cost_per_token": 0.00000008, + "input_cost_per_token": 5e-08, + "output_cost_per_token": 8e-08, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, @@ -4137,13 +4251,14 @@ "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000059, - "output_cost_per_token": 0.00000079, + "input_cost_per_token": 5.9e-07, + "output_cost_per_token": 7.9e-07, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2025-01-24" }, "groq/llama-3.1-405b-reasoning": { "max_tokens": 8192, @@ -4157,83 +4272,141 @@ "supports_response_schema": true, "supports_tool_choice": true }, - "groq/mixtral-8x7b-32768": { - "max_tokens": 32768, - "max_input_tokens": 32768, - "max_output_tokens": 32768, - "input_cost_per_token": 0.00000024, - "output_cost_per_token": 0.00000024, + "groq/meta-llama/llama-4-scout-17b-16e-instruct": { + "max_tokens": 8192, + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "input_cost_per_token": 1.1e-07, + "output_cost_per_token": 3.4e-07, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, "supports_tool_choice": true }, + "groq/meta-llama/llama-4-maverick-17b-128e-instruct": { + "max_tokens": 8192, + "max_input_tokens": 131072, + "max_output_tokens": 8192, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 6e-07, + "litellm_provider": "groq", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true + }, + "groq/mistral-saba-24b": { + "max_tokens": 32000, + "max_input_tokens": 32000, + "max_output_tokens": 32000, + "input_cost_per_token": 7.9e-07, + "output_cost_per_token": 7.9e-07, + "litellm_provider": "groq", + "mode": "chat" + }, + "groq/mixtral-8x7b-32768": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 2.4e-07, + "output_cost_per_token": 2.4e-07, + "litellm_provider": "groq", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "deprecation_date": "2025-03-20" + }, "groq/gemma-7b-it": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000007, - "output_cost_per_token": 0.00000007, + "input_cost_per_token": 7e-08, + "output_cost_per_token": 7e-08, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2024-12-18" }, "groq/gemma2-9b-it": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000020, - "output_cost_per_token": 0.00000020, + "input_cost_per_token": 2e-07, + "output_cost_per_token": 2e-07, "litellm_provider": "groq", "mode": "chat", - "supports_function_calling": true, + "supports_function_calling": false, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": false }, "groq/llama3-groq-70b-8192-tool-use-preview": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000089, - "output_cost_per_token": 0.00000089, + "input_cost_per_token": 8.9e-07, + "output_cost_per_token": 8.9e-07, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "deprecation_date": "2025-1-6" }, "groq/llama3-groq-8b-8192-tool-use-preview": { "max_tokens": 8192, "max_input_tokens": 8192, "max_output_tokens": 8192, - "input_cost_per_token": 0.00000019, - "output_cost_per_token": 0.00000019, + "input_cost_per_token": 1.9e-07, + "output_cost_per_token": 1.9e-07, "litellm_provider": "groq", "mode": "chat", "supports_function_calling": true, "supports_response_schema": true, + "supports_tool_choice": true, + "deprecation_date": "2025-1-6" + }, + "groq/qwen-qwq-32b": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 128000, + "input_cost_per_token": 2.9e-07, + "output_cost_per_token": 3.9e-07, + "litellm_provider": "groq", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_reasoning": true, "supports_tool_choice": true }, + "groq/playai-tts": { + "max_tokens": 10000, + "max_input_tokens": 10000, + "max_output_tokens": 10000, + "input_cost_per_character": 5e-05, + "litellm_provider": "groq", + "mode": "audio_speech" + }, "groq/whisper-large-v3": { - "mode": "audio_transcription", - "input_cost_per_second": 0.00003083, - "output_cost_per_second": 0, - "litellm_provider": "groq" + "input_cost_per_second": 3.083e-05, + "output_cost_per_second": 0.0, + "litellm_provider": "groq", + "mode": "audio_transcription" }, "groq/whisper-large-v3-turbo": { - "mode": "audio_transcription", - "input_cost_per_second": 0.00001111, - "output_cost_per_second": 0, - "litellm_provider": "groq" + "input_cost_per_second": 1.111e-05, + "output_cost_per_second": 0.0, + "litellm_provider": "groq", + "mode": "audio_transcription" }, "groq/distil-whisper-large-v3-en": { - "mode": "audio_transcription", - "input_cost_per_second": 0.00000556, - "output_cost_per_second": 0, - "litellm_provider": "groq" + "input_cost_per_second": 5.56e-06, + "output_cost_per_second": 0.0, + "litellm_provider": "groq", + "mode": "audio_transcription" }, "cerebras/llama3.1-8b": { "max_tokens": 128000, @@ -4257,7 +4430,7 @@ "supports_function_calling": true, "supports_tool_choice": true }, - "cerebras/llama3.3-70b": { + "cerebras/llama-3.3-70b": { "max_tokens": 128000, "max_input_tokens": 128000, "max_output_tokens": 128000, @@ -4352,6 +4525,11 @@ "output_cost_per_token": 0.000004, "cache_creation_input_token_cost": 0.000001, "cache_read_input_token_cost": 0.00000008, + "search_context_cost_per_query": { + "search_context_size_low": 1e-2, + "search_context_size_medium": 1e-2, + "search_context_size_high": 1e-2 + }, "litellm_provider": "anthropic", "mode": "chat", "supports_function_calling": true, @@ -4362,7 +4540,8 @@ "supports_prompt_caching": true, "supports_response_schema": true, "deprecation_date": "2025-10-01", - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true }, "claude-3-5-haiku-latest": { "max_tokens": 8192, @@ -4372,6 +4551,11 @@ "output_cost_per_token": 0.000005, "cache_creation_input_token_cost": 0.00000125, "cache_read_input_token_cost": 0.0000001, + "search_context_cost_per_query": { + "search_context_size_low": 1e-2, + "search_context_size_medium": 1e-2, + "search_context_size_high": 1e-2 + }, "litellm_provider": "anthropic", "mode": "chat", "supports_function_calling": true, @@ -4382,7 +4566,8 @@ "supports_prompt_caching": true, "supports_response_schema": true, "deprecation_date": "2025-10-01", - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true }, "claude-3-opus-latest": { "max_tokens": 4096, @@ -4440,6 +4625,7 @@ "supports_tool_choice": true }, "claude-3-5-sonnet-latest": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -4447,6 +4633,11 @@ "output_cost_per_token": 0.000015, "cache_creation_input_token_cost": 0.00000375, "cache_read_input_token_cost": 0.0000003, + "search_context_cost_per_query": { + "search_context_size_low": 1e-2, + "search_context_size_medium": 1e-2, + "search_context_size_high": 1e-2 + }, "litellm_provider": "anthropic", "mode": "chat", "supports_function_calling": true, @@ -4457,7 +4648,8 @@ "supports_prompt_caching": true, "supports_response_schema": true, "deprecation_date": "2025-06-01", - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true }, "claude-3-5-sonnet-20240620": { "max_tokens": 8192, @@ -4480,11 +4672,17 @@ "supports_tool_choice": true }, "claude-3-7-sonnet-latest": { + "supports_computer_use": true, "max_tokens": 128000, "max_input_tokens": 200000, "max_output_tokens": 128000, "input_cost_per_token": 0.000003, "output_cost_per_token": 0.000015, + "search_context_cost_per_query": { + "search_context_size_low": 1e-2, + "search_context_size_medium": 1e-2, + "search_context_size_high": 1e-2 + }, "cache_creation_input_token_cost": 0.00000375, "cache_read_input_token_cost": 0.0000003, "litellm_provider": "anthropic", @@ -4501,6 +4699,7 @@ "supports_reasoning": true }, "claude-3-7-sonnet-20250219": { + "supports_computer_use": true, "max_tokens": 128000, "max_input_tokens": 200000, "max_output_tokens": 128000, @@ -4508,6 +4707,11 @@ "output_cost_per_token": 0.000015, "cache_creation_input_token_cost": 0.00000375, "cache_read_input_token_cost": 0.0000003, + "search_context_cost_per_query": { + "search_context_size_low": 1e-2, + "search_context_size_medium": 1e-2, + "search_context_size_high": 1e-2 + }, "litellm_provider": "anthropic", "mode": "chat", "supports_function_calling": true, @@ -4519,9 +4723,11 @@ "supports_response_schema": true, "deprecation_date": "2026-02-01", "supports_tool_choice": true, - "supports_reasoning": true + "supports_reasoning": true, + "supports_web_search": true }, "claude-3-5-sonnet-20241022": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -4529,6 +4735,11 @@ "output_cost_per_token": 0.000015, "cache_creation_input_token_cost": 0.00000375, "cache_read_input_token_cost": 0.0000003, + "search_context_cost_per_query": { + "search_context_size_low": 1e-2, + "search_context_size_medium": 1e-2, + "search_context_size_high": 1e-2 + }, "litellm_provider": "anthropic", "mode": "chat", "supports_function_calling": true, @@ -4539,7 +4750,8 @@ "supports_prompt_caching": true, "supports_response_schema": true, "deprecation_date": "2025-10-01", - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true }, "text-bison": { "max_tokens": 2048, @@ -4860,6 +5072,54 @@ "source": "https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models", "supports_tool_choice": true }, + "meta_llama/Llama-4-Scout-17B-16E-Instruct-FP8": { + "max_tokens": 128000, + "max_input_tokens": 10000000, + "max_output_tokens": 4028, + "litellm_provider": "meta_llama", + "mode": "chat", + "supports_function_calling": false, + "source": "https://llama.developer.meta.com/docs/models", + "supports_tool_choice": false, + "supported_modalities": ["text", "image"], + "supported_output_modalities": ["text"] + }, + "meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8": { + "max_tokens": 128000, + "max_input_tokens": 1000000, + "max_output_tokens": 4028, + "litellm_provider": "meta_llama", + "mode": "chat", + "supports_function_calling": false, + "source": "https://llama.developer.meta.com/docs/models", + "supports_tool_choice": false, + "supported_modalities": ["text", "image"], + "supported_output_modalities": ["text"] + }, + "meta_llama/Llama-3.3-70B-Instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 4028, + "litellm_provider": "meta_llama", + "mode": "chat", + "supports_function_calling": false, + "source": "https://llama.developer.meta.com/docs/models", + "supports_tool_choice": false, + "supported_modalities": ["text"], + "supported_output_modalities": ["text"] + }, + "meta_llama/Llama-3.3-8B-Instruct": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "max_output_tokens": 4028, + "litellm_provider": "meta_llama", + "mode": "chat", + "supports_function_calling": false, + "source": "https://llama.developer.meta.com/docs/models", + "supports_tool_choice": false, + "supported_modalities": ["text"], + "supported_output_modalities": ["text"] + }, "gemini-pro": { "max_tokens": 8192, "max_input_tokens": 32760, @@ -5388,9 +5648,9 @@ "supports_tool_choice": true }, "gemini-2.5-pro-exp-03-25": { - "max_tokens": 65536, + "max_tokens": 65535, "max_input_tokens": 1048576, - "max_output_tokens": 65536, + "max_output_tokens": 65535, "max_images_per_prompt": 3000, "max_videos_per_prompt": 10, "max_video_length": 1, @@ -5580,9 +5840,9 @@ "supports_tool_choice": true }, "gemini/gemini-2.5-pro-exp-03-25": { - "max_tokens": 65536, + "max_tokens": 65535, "max_input_tokens": 1048576, - "max_output_tokens": 65536, + "max_output_tokens": 65535, "max_images_per_prompt": 3000, "max_videos_per_prompt": 10, "max_video_length": 1, @@ -5611,9 +5871,9 @@ "source": "https://cloud.google.com/vertex-ai/generative-ai/pricing" }, "gemini/gemini-2.5-flash-preview-04-17": { - "max_tokens": 65536, + "max_tokens": 65535, "max_input_tokens": 1048576, - "max_output_tokens": 65536, + "max_output_tokens": 65535, "max_images_per_prompt": 3000, "max_videos_per_prompt": 10, "max_video_length": 1, @@ -5641,9 +5901,9 @@ "source": "https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview" }, "gemini-2.5-flash-preview-04-17": { - "max_tokens": 65536, + "max_tokens": 65535, "max_input_tokens": 1048576, - "max_output_tokens": 65536, + "max_output_tokens": 65535, "max_images_per_prompt": 3000, "max_videos_per_prompt": 10, "max_video_length": 1, @@ -5743,10 +6003,39 @@ "supports_tool_choice": true, "deprecation_date": "2026-02-25" }, - "gemini-2.5-pro-preview-03-25": { - "max_tokens": 65536, + "gemini-2.5-pro-preview-05-06": { + "max_tokens": 65535, "max_input_tokens": 1048576, - "max_output_tokens": 65536, + "max_output_tokens": 65535, + "max_images_per_prompt": 3000, + "max_videos_per_prompt": 10, + "max_video_length": 1, + "max_audio_length_hours": 8.4, + "max_audio_per_prompt": 1, + "max_pdf_size_mb": 30, + "input_cost_per_audio_token": 0.00000125, + "input_cost_per_token": 0.00000125, + "input_cost_per_token_above_200k_tokens": 0.0000025, + "output_cost_per_token": 0.00001, + "output_cost_per_token_above_200k_tokens": 0.000015, + "litellm_provider": "vertex_ai-language-models", + "mode": "chat", + "supports_reasoning": true, + "supports_system_messages": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_response_schema": true, + "supports_audio_output": false, + "supports_tool_choice": true, + "supported_endpoints": ["/v1/chat/completions", "/v1/completions", "/v1/batch"], + "supported_modalities": ["text", "image", "audio", "video"], + "supported_output_modalities": ["text"], + "source": "https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview" + }, + "gemini-2.5-pro-preview-03-25": { + "max_tokens": 65535, + "max_input_tokens": 1048576, + "max_output_tokens": 65535, "max_images_per_prompt": 3000, "max_videos_per_prompt": 10, "max_video_length": 1, @@ -5891,10 +6180,39 @@ "supported_output_modalities": ["text", "image"], "source": "https://ai.google.dev/pricing#2_0flash" }, - "gemini/gemini-2.5-pro-preview-03-25": { - "max_tokens": 65536, + "gemini/gemini-2.5-pro-preview-05-06": { + "max_tokens": 65535, "max_input_tokens": 1048576, - "max_output_tokens": 65536, + "max_output_tokens": 65535, + "max_images_per_prompt": 3000, + "max_videos_per_prompt": 10, + "max_video_length": 1, + "max_audio_length_hours": 8.4, + "max_audio_per_prompt": 1, + "max_pdf_size_mb": 30, + "input_cost_per_audio_token": 0.0000007, + "input_cost_per_token": 0.00000125, + "input_cost_per_token_above_200k_tokens": 0.0000025, + "output_cost_per_token": 0.00001, + "output_cost_per_token_above_200k_tokens": 0.000015, + "litellm_provider": "gemini", + "mode": "chat", + "rpm": 10000, + "tpm": 10000000, + "supports_system_messages": true, + "supports_function_calling": true, + "supports_vision": true, + "supports_response_schema": true, + "supports_audio_output": false, + "supports_tool_choice": true, + "supported_modalities": ["text", "image", "audio", "video"], + "supported_output_modalities": ["text"], + "source": "https://ai.google.dev/gemini-api/docs/pricing#gemini-2.5-pro-preview" + }, + "gemini/gemini-2.5-pro-preview-03-25": { + "max_tokens": 65535, + "max_input_tokens": 1048576, + "max_output_tokens": 65535, "max_images_per_prompt": 3000, "max_videos_per_prompt": 10, "max_video_length": 1, @@ -6144,6 +6462,7 @@ "supports_tool_choice": true }, "vertex_ai/claude-3-5-sonnet": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -6172,6 +6491,7 @@ "supports_tool_choice": true }, "vertex_ai/claude-3-5-sonnet-v2": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -6186,6 +6506,7 @@ "supports_tool_choice": true }, "vertex_ai/claude-3-5-sonnet-v2@20241022": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -6200,6 +6521,7 @@ "supports_tool_choice": true }, "vertex_ai/claude-3-7-sonnet@20250219": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -6323,7 +6645,7 @@ "supported_modalities": ["text", "image"], "supported_output_modalities": ["text", "code"] }, - "vertex_ai/meta/llama-4-scout-128b-16e-instruct-maas": { + "vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas": { "max_tokens": 10e6, "max_input_tokens": 10e6, "max_output_tokens": 10e6, @@ -7741,6 +8063,7 @@ "supports_tool_choice": true }, "openrouter/anthropic/claude-3.5-sonnet": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -7755,6 +8078,7 @@ "supports_tool_choice": true }, "openrouter/anthropic/claude-3.5-sonnet:beta": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -7768,6 +8092,7 @@ "supports_tool_choice": true }, "openrouter/anthropic/claude-3.7-sonnet": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -7784,6 +8109,7 @@ "supports_tool_choice": true }, "openrouter/anthropic/claude-3.7-sonnet:beta": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -8796,11 +9122,14 @@ "supports_tool_choice": true }, "anthropic.claude-3-7-sonnet-20250219-v1:0": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, "input_cost_per_token": 0.000003, "output_cost_per_token": 0.000015, + "cache_creation_input_token_cost": 0.00000375, + "cache_read_input_token_cost": 0.0000003, "litellm_provider": "bedrock_converse", "mode": "chat", "supports_function_calling": true, @@ -8813,11 +9142,14 @@ "supports_tool_choice": true }, "anthropic.claude-3-5-sonnet-20241022-v2:0": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, "input_cost_per_token": 0.000003, "output_cost_per_token": 0.000015, + "cache_creation_input_token_cost": 0.00000375, + "cache_read_input_token_cost": 0.0000003, "litellm_provider": "bedrock", "mode": "chat", "supports_function_calling": true, @@ -8848,6 +9180,8 @@ "max_output_tokens": 8192, "input_cost_per_token": 0.0000008, "output_cost_per_token": 0.000004, + "cache_creation_input_token_cost": 0.000001, + "cache_read_input_token_cost": 0.00000008, "litellm_provider": "bedrock", "mode": "chat", "supports_assistant_prefill": true, @@ -8899,11 +9233,14 @@ "supports_tool_choice": true }, "us.anthropic.claude-3-5-sonnet-20241022-v2:0": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, "input_cost_per_token": 0.000003, "output_cost_per_token": 0.000015, + "cache_creation_input_token_cost": 0.00000375, + "cache_read_input_token_cost": 0.0000003, "litellm_provider": "bedrock", "mode": "chat", "supports_function_calling": true, @@ -8915,11 +9252,14 @@ "supports_tool_choice": true }, "us.anthropic.claude-3-7-sonnet-20250219-v1:0": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, "input_cost_per_token": 0.000003, "output_cost_per_token": 0.000015, + "cache_creation_input_token_cost": 0.00000375, + "cache_read_input_token_cost": 0.0000003, "litellm_provider": "bedrock_converse", "mode": "chat", "supports_function_calling": true, @@ -8951,6 +9291,8 @@ "max_output_tokens": 8192, "input_cost_per_token": 0.0000008, "output_cost_per_token": 0.000004, + "cache_creation_input_token_cost": 0.000001, + "cache_read_input_token_cost": 0.00000008, "litellm_provider": "bedrock", "mode": "chat", "supports_assistant_prefill": true, @@ -9002,6 +9344,7 @@ "supports_tool_choice": true }, "eu.anthropic.claude-3-5-sonnet-20241022-v2:0": { + "supports_computer_use": true, "max_tokens": 8192, "max_input_tokens": 200000, "max_output_tokens": 8192, @@ -9017,6 +9360,24 @@ "supports_response_schema": true, "supports_tool_choice": true }, + "eu.anthropic.claude-3-7-sonnet-20250219-v1:0": { + "supports_computer_use": true, + "max_tokens": 8192, + "max_input_tokens": 200000, + "max_output_tokens": 8192, + "input_cost_per_token": 0.000003, + "output_cost_per_token": 0.000015, + "litellm_provider": "bedrock", + "mode": "chat", + "supports_function_calling": true, + "supports_vision": true, + "supports_assistant_prefill": true, + "supports_prompt_caching": true, + "supports_response_schema": true, + "supports_pdf_input": true, + "supports_tool_choice": true, + "supports_reasoning": true + }, "eu.anthropic.claude-3-haiku-20240307-v1:0": { "max_tokens": 4096, "max_input_tokens": 200000, @@ -10057,6 +10418,66 @@ "supports_function_calling": true, "supports_tool_choice": false }, + "meta.llama4-maverick-17b-instruct-v1:0": { + "max_tokens": 4096, + "max_input_tokens": 128000, + "max_output_tokens": 4096, + "input_cost_per_token": 0.00024e-3, + "input_cost_per_token_batches": 0.00012e-3, + "output_cost_per_token": 0.00097e-3, + "output_cost_per_token_batches": 0.000485e-3, + "litellm_provider": "bedrock_converse", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": false, + "supported_modalities": ["text", "image"], + "supported_output_modalities": ["text", "code"] + }, + "us.meta.llama4-maverick-17b-instruct-v1:0": { + "max_tokens": 4096, + "max_input_tokens": 128000, + "max_output_tokens": 4096, + "input_cost_per_token": 0.00024e-3, + "input_cost_per_token_batches": 0.00012e-3, + "output_cost_per_token": 0.00097e-3, + "output_cost_per_token_batches": 0.000485e-3, + "litellm_provider": "bedrock_converse", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": false, + "supported_modalities": ["text", "image"], + "supported_output_modalities": ["text", "code"] + }, + "meta.llama4-scout-17b-instruct-v1:0": { + "max_tokens": 4096, + "max_input_tokens": 128000, + "max_output_tokens": 4096, + "input_cost_per_token": 0.00017e-3, + "input_cost_per_token_batches": 0.000085e-3, + "output_cost_per_token": 0.00066e-3, + "output_cost_per_token_batches": 0.00033e-3, + "litellm_provider": "bedrock_converse", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": false, + "supported_modalities": ["text", "image"], + "supported_output_modalities": ["text", "code"] + }, + "us.meta.llama4-scout-17b-instruct-v1:0": { + "max_tokens": 4096, + "max_input_tokens": 128000, + "max_output_tokens": 4096, + "input_cost_per_token": 0.00017e-3, + "input_cost_per_token_batches": 0.000085e-3, + "output_cost_per_token": 0.00066e-3, + "output_cost_per_token_batches": 0.00033e-3, + "litellm_provider": "bedrock_converse", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": false, + "supported_modalities": ["text", "image"], + "supported_output_modalities": ["text", "code"] + }, "512-x-512/50-steps/stability.stable-diffusion-xl-v0": { "max_tokens": 77, "max_input_tokens": 77, @@ -10525,7 +10946,8 @@ "input_cost_per_token": 0.0, "output_cost_per_token": 0.0, "litellm_provider": "ollama", - "mode": "chat" + "mode": "chat", + "supports_function_calling": true }, "ollama/mistral": { "max_tokens": 8192, @@ -10534,7 +10956,8 @@ "input_cost_per_token": 0.0, "output_cost_per_token": 0.0, "litellm_provider": "ollama", - "mode": "completion" + "mode": "completion", + "supports_function_calling": true }, "ollama/mistral-7B-Instruct-v0.1": { "max_tokens": 8192, @@ -10543,7 +10966,8 @@ "input_cost_per_token": 0.0, "output_cost_per_token": 0.0, "litellm_provider": "ollama", - "mode": "chat" + "mode": "chat", + "supports_function_calling": true }, "ollama/mistral-7B-Instruct-v0.2": { "max_tokens": 32768, @@ -10552,7 +10976,8 @@ "input_cost_per_token": 0.0, "output_cost_per_token": 0.0, "litellm_provider": "ollama", - "mode": "chat" + "mode": "chat", + "supports_function_calling": true }, "ollama/mixtral-8x7B-Instruct-v0.1": { "max_tokens": 32768, @@ -10561,7 +10986,8 @@ "input_cost_per_token": 0.0, "output_cost_per_token": 0.0, "litellm_provider": "ollama", - "mode": "chat" + "mode": "chat", + "supports_function_calling": true }, "ollama/mixtral-8x22B-Instruct-v0.1": { "max_tokens": 65536, @@ -10570,7 +10996,8 @@ "input_cost_per_token": 0.0, "output_cost_per_token": 0.0, "litellm_provider": "ollama", - "mode": "chat" + "mode": "chat", + "supports_function_calling": true }, "ollama/codellama": { "max_tokens": 4096, @@ -10894,42 +11321,6 @@ "mode": "chat" , "deprecation_date": "2025-02-22" }, - "perplexity/sonar": { - "max_tokens": 127072, - "max_input_tokens": 127072, - "max_output_tokens": 127072, - "input_cost_per_token": 0.000001, - "output_cost_per_token": 0.000001, - "litellm_provider": "perplexity", - "mode": "chat" - }, - "perplexity/sonar-pro": { - "max_tokens": 200000, - "max_input_tokens": 200000, - "max_output_tokens": 8096, - "input_cost_per_token": 0.000003, - "output_cost_per_token": 0.000015, - "litellm_provider": "perplexity", - "mode": "chat" - }, - "perplexity/sonar": { - "max_tokens": 127072, - "max_input_tokens": 127072, - "max_output_tokens": 127072, - "input_cost_per_token": 0.000001, - "output_cost_per_token": 0.000001, - "litellm_provider": "perplexity", - "mode": "chat" - }, - "perplexity/sonar-pro": { - "max_tokens": 200000, - "max_input_tokens": 200000, - "max_output_tokens": 8096, - "input_cost_per_token": 0.000003, - "output_cost_per_token": 0.000015, - "litellm_provider": "perplexity", - "mode": "chat" - }, "perplexity/pplx-7b-chat": { "max_tokens": 8192, "max_input_tokens": 8192, @@ -11033,6 +11424,81 @@ "litellm_provider": "perplexity", "mode": "chat" }, + "perplexity/sonar": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "input_cost_per_token": 1e-6, + "output_cost_per_token": 1e-6, + "litellm_provider": "perplexity", + "mode": "chat", + "search_context_cost_per_query": { + "search_context_size_low": 5e-3, + "search_context_size_medium": 8e-3, + "search_context_size_high": 12e-3 + }, + "supports_web_search": true + }, + "perplexity/sonar-pro": { + "max_tokens": 8000, + "max_input_tokens": 200000, + "max_output_tokens": 8000, + "input_cost_per_token": 3e-6, + "output_cost_per_token": 15e-6, + "litellm_provider": "perplexity", + "mode": "chat", + "search_context_cost_per_query": { + "search_context_size_low": 6e-3, + "search_context_size_medium": 10e-3, + "search_context_size_high": 14e-3 + }, + "supports_web_search": true + }, + "perplexity/sonar-reasoning": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "input_cost_per_token": 1e-6, + "output_cost_per_token": 5e-6, + "litellm_provider": "perplexity", + "mode": "chat", + "search_context_cost_per_query": { + "search_context_size_low": 5e-3, + "search_context_size_medium": 8e-3, + "search_context_size_high": 14e-3 + }, + "supports_web_search": true, + "supports_reasoning": true + }, + "perplexity/sonar-reasoning-pro": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "input_cost_per_token": 2e-6, + "output_cost_per_token": 8e-6, + "litellm_provider": "perplexity", + "mode": "chat", + "search_context_cost_per_query": { + "search_context_size_low": 6e-3, + "search_context_size_medium": 10e-3, + "search_context_size_high": 14e-3 + }, + "supports_web_search": true, + "supports_reasoning": true + }, + "perplexity/sonar-deep-research": { + "max_tokens": 128000, + "max_input_tokens": 128000, + "input_cost_per_token": 2e-6, + "output_cost_per_token": 8e-6, + "output_cost_per_reasoning_token": 3e-6, + "litellm_provider": "perplexity", + "mode": "chat", + "search_context_cost_per_query": { + "search_context_size_low": 5e-3, + "search_context_size_medium": 5e-3, + "search_context_size_high": 5e-3 + }, + "supports_reasoning": true, + "supports_web_search": true + }, "fireworks_ai/accounts/fireworks/models/llama-v3p2-1b-instruct": { "max_tokens": 16384, "max_input_tokens": 16384, @@ -11784,81 +12250,169 @@ "metadata": {"notes": "Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."} }, "sambanova/Meta-Llama-3.1-8B-Instruct": { - "max_tokens": 16000, - "max_input_tokens": 16000, - "max_output_tokens": 16000, + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, "input_cost_per_token": 0.0000001, "output_cost_per_token": 0.0000002, "litellm_provider": "sambanova", - "supports_function_calling": true, "mode": "chat", - "supports_tool_choice": true - }, - "sambanova/Meta-Llama-3.1-70B-Instruct": { - "max_tokens": 128000, - "max_input_tokens": 128000, - "max_output_tokens": 128000, - "input_cost_per_token": 0.0000006, - "output_cost_per_token": 0.0000012, - "litellm_provider": "sambanova", "supports_function_calling": true, - "mode": "chat", - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_response_schema": true, + "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/Meta-Llama-3.1-405B-Instruct": { - "max_tokens": 16000, - "max_input_tokens": 16000, - "max_output_tokens": 16000, + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, "input_cost_per_token": 0.000005, "output_cost_per_token": 0.000010, "litellm_provider": "sambanova", - "supports_function_calling": true, "mode": "chat", - "supports_tool_choice": true + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "source": "https://cloud.sambanova.ai/plans/pricing" }, "sambanova/Meta-Llama-3.2-1B-Instruct": { - "max_tokens": 16000, - "max_input_tokens": 16000, - "max_output_tokens": 16000, + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 0.00000004, + "output_cost_per_token": 0.00000008, + "litellm_provider": "sambanova", + "mode": "chat", + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "sambanova/Meta-Llama-3.2-3B-Instruct": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 0.00000008, + "output_cost_per_token": 0.00000016, + "litellm_provider": "sambanova", + "mode": "chat", + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "sambanova/Llama-4-Maverick-17B-128E-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.00000063, + "output_cost_per_token": 0.0000018, + "litellm_provider": "sambanova", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "supports_vision": true, + "source": "https://cloud.sambanova.ai/plans/pricing", + "metadata": {"notes": "For vision models, images are converted to 6432 input tokens and are billed at that amount"} + }, + "sambanova/Llama-4-Scout-17B-16E-Instruct": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, + "input_cost_per_token": 0.0000004, + "output_cost_per_token": 0.0000007, + "litellm_provider": "sambanova", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_response_schema": true, + "source": "https://cloud.sambanova.ai/plans/pricing", + "metadata": {"notes": "For vision models, images are converted to 6432 input tokens and are billed at that amount"} + }, + "sambanova/Meta-Llama-3.3-70B-Instruct": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.0000006, + "output_cost_per_token": 0.0000012, + "litellm_provider": "sambanova", + "mode": "chat", + "supports_function_calling": true, + "supports_response_schema": true, + "supports_tool_choice": true, + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "sambanova/Meta-Llama-Guard-3-8B": { + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 0.0000003, + "output_cost_per_token": 0.0000003, + "litellm_provider": "sambanova", + "mode": "chat", + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "sambanova/Qwen3-32B": { + "max_tokens": 8192, + "max_input_tokens": 8192, + "max_output_tokens": 8192, "input_cost_per_token": 0.0000004, "output_cost_per_token": 0.0000008, "litellm_provider": "sambanova", "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, "mode": "chat", - "supports_tool_choice": true + "source": "https://cloud.sambanova.ai/plans/pricing" }, - "sambanova/Meta-Llama-3.2-3B-Instruct": { - "max_tokens": 4000, - "max_input_tokens": 4000, - "max_output_tokens": 4000, - "input_cost_per_token": 0.0000008, - "output_cost_per_token": 0.0000016, + "sambanova/QwQ-32B": { + "max_tokens": 16384, + "max_input_tokens": 16384, + "max_output_tokens": 16384, + "input_cost_per_token": 0.0000005, + "output_cost_per_token": 0.0000010, "litellm_provider": "sambanova", - "supports_function_calling": true, "mode": "chat", - "supports_tool_choice": true + "source": "https://cloud.sambanova.ai/plans/pricing" }, - "sambanova/Qwen2.5-Coder-32B-Instruct": { - "max_tokens": 8000, - "max_input_tokens": 8000, - "max_output_tokens": 8000, - "input_cost_per_token": 0.0000015, - "output_cost_per_token": 0.000003, + "sambanova/Qwen2-Audio-7B-Instruct": { + "max_tokens": 4096, + "max_input_tokens": 4096, + "max_output_tokens": 4096, + "input_cost_per_token": 0.0000005, + "output_cost_per_token": 0.0001, "litellm_provider": "sambanova", - "supports_function_calling": true, "mode": "chat", - "supports_tool_choice": true + "supports_audio_input": true, + "source": "https://cloud.sambanova.ai/plans/pricing" }, - "sambanova/Qwen2.5-72B-Instruct": { - "max_tokens": 8000, - "max_input_tokens": 8000, - "max_output_tokens": 8000, - "input_cost_per_token": 0.000002, - "output_cost_per_token": 0.000004, + "sambanova/DeepSeek-R1-Distill-Llama-70B": { + "max_tokens": 131072, + "max_input_tokens": 131072, + "max_output_tokens": 131072, + "input_cost_per_token": 0.0000007, + "output_cost_per_token": 0.0000014, "litellm_provider": "sambanova", - "supports_function_calling": true, "mode": "chat", - "supports_tool_choice": true + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "sambanova/DeepSeek-R1": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 0.000005, + "output_cost_per_token": 0.000007, + "litellm_provider": "sambanova", + "mode": "chat", + "source": "https://cloud.sambanova.ai/plans/pricing" + }, + "sambanova/DeepSeek-V3-0324": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 32768, + "input_cost_per_token": 0.0000030, + "output_cost_per_token": 0.0000045, + "litellm_provider": "sambanova", + "mode": "chat", + "supports_function_calling": true, + "supports_tool_choice": true, + "supports_reasoning": true, + "source": "https://cloud.sambanova.ai/plans/pricing" }, "assemblyai/nano": { "mode": "audio_transcription", @@ -11898,6 +12452,7 @@ "mode": "chat" }, "snowflake/claude-3-5-sonnet": { + "supports_computer_use": true, "max_tokens": 18000, "max_input_tokens": 18000, "max_output_tokens": 8192, @@ -12050,5 +12605,164 @@ "max_output_tokens": 8192, "litellm_provider": "snowflake", "mode": "chat" + }, + "nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct": { + "input_cost_per_token": 9e-8, + "output_cost_per_token": 2.9e-7, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models" + }, + "nscale/Qwen/Qwen2.5-Coder-3B-Instruct": { + "input_cost_per_token": 1e-8, + "output_cost_per_token": 3e-8, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models" + }, + "nscale/Qwen/Qwen2.5-Coder-7B-Instruct": { + "input_cost_per_token": 1e-8, + "output_cost_per_token": 3e-8, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models" + }, + "nscale/Qwen/Qwen2.5-Coder-32B-Instruct": { + "input_cost_per_token": 6e-8, + "output_cost_per_token": 2e-7, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models" + }, + "nscale/Qwen/QwQ-32B": { + "input_cost_per_token": 1.8e-7, + "output_cost_per_token": 2e-7, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models" + }, + "nscale/deepseek-ai/DeepSeek-R1-Distill-Llama-70B": { + "input_cost_per_token": 3.75e-7, + "output_cost_per_token": 3.75e-7, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models", + "metadata": { + "notes": "Pricing listed as $0.75/1M tokens total. Assumed 50/50 split for input/output." + } + }, + "nscale/deepseek-ai/DeepSeek-R1-Distill-Llama-8B": { + "input_cost_per_token": 2.5e-8, + "output_cost_per_token": 2.5e-8, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models", + "metadata": { + "notes": "Pricing listed as $0.05/1M tokens total. Assumed 50/50 split for input/output." + } + }, + "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B": { + "input_cost_per_token": 9e-8, + "output_cost_per_token": 9e-8, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models", + "metadata": { + "notes": "Pricing listed as $0.18/1M tokens total. Assumed 50/50 split for input/output." + } + }, + "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-7B": { + "input_cost_per_token": 2e-7, + "output_cost_per_token": 2e-7, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models", + "metadata": { + "notes": "Pricing listed as $0.40/1M tokens total. Assumed 50/50 split for input/output." + } + }, + "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B": { + "input_cost_per_token": 7e-8, + "output_cost_per_token": 7e-8, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models", + "metadata": { + "notes": "Pricing listed as $0.14/1M tokens total. Assumed 50/50 split for input/output." + } + }, + "nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B": { + "input_cost_per_token": 1.5e-7, + "output_cost_per_token": 1.5e-7, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models", + "metadata": { + "notes": "Pricing listed as $0.30/1M tokens total. Assumed 50/50 split for input/output." + } + }, + "nscale/mistralai/mixtral-8x22b-instruct-v0.1": { + "input_cost_per_token": 6e-7, + "output_cost_per_token": 6e-7, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models", + "metadata": { + "notes": "Pricing listed as $1.20/1M tokens total. Assumed 50/50 split for input/output." + } + }, + "nscale/meta-llama/Llama-3.1-8B-Instruct": { + "input_cost_per_token": 3e-8, + "output_cost_per_token": 3e-8, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models", + "metadata": { + "notes": "Pricing listed as $0.06/1M tokens total. Assumed 50/50 split for input/output." + } + }, + "nscale/meta-llama/Llama-3.3-70B-Instruct": { + "input_cost_per_token": 2e-7, + "output_cost_per_token": 2e-7, + "litellm_provider": "nscale", + "mode": "chat", + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#chat-models", + "metadata": { + "notes": "Pricing listed as $0.40/1M tokens total. Assumed 50/50 split for input/output." + } + }, + "nscale/black-forest-labs/FLUX.1-schnell": { + "mode": "image_generation", + "input_cost_per_pixel": 1.3e-9, + "output_cost_per_pixel": 0.0, + "litellm_provider": "nscale", + "supported_endpoints": [ + "/v1/images/generations" + ], + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#image-models" + }, + "nscale/stabilityai/stable-diffusion-xl-base-1.0": { + "mode": "image_generation", + "input_cost_per_pixel": 3e-9, + "output_cost_per_pixel": 0.0, + "litellm_provider": "nscale", + "supported_endpoints": [ + "/v1/images/generations" + ], + "source": "https://docs.nscale.com/docs/inference/serverless-models/current#image-models" + }, + "featherless_ai/featherless-ai/Qwerky-72B": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 4096, + "litellm_provider": "featherless_ai", + "mode": "chat" + }, + "featherless_ai/featherless-ai/Qwerky-QwQ-32B": { + "max_tokens": 32768, + "max_input_tokens": 32768, + "max_output_tokens": 4096, + "litellm_provider": "featherless_ai", + "mode": "chat" } -} +} \ No newline at end of file diff --git a/litellm/mypy.ini b/litellm/mypy.ini index df3c6ed5c7b..c084de7c563 100644 --- a/litellm/mypy.ini +++ b/litellm/mypy.ini @@ -9,3 +9,6 @@ disable_error_code = [mypy-google.*] ignore_missing_imports = True + +[mypy-cryptography.hazmat.bindings._rust.x509] +ignore_errors = True \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/_next/static/chunks/250-035458b72061498b.js b/litellm/proxy/_experimental/out/_next/static/chunks/250-035458b72061498b.js deleted file mode 100644 index 084668d39e2..00000000000 --- a/litellm/proxy/_experimental/out/_next/static/chunks/250-035458b72061498b.js +++ /dev/null @@ -1 +0,0 @@ -"use strict";(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[250],{19250:function(e,t,o){o.d(t,{$D:function(){return eP},$I:function(){return K},AZ:function(){return H},Au:function(){return em},BL:function(){return eM},Br:function(){return b},E9:function(){return eH},EB:function(){return to},EG:function(){return e$},EY:function(){return eQ},Eb:function(){return C},FC:function(){return el},Gh:function(){return eB},H1:function(){return A},H2:function(){return n},Hx:function(){return e_},I1:function(){return E},It:function(){return x},J$:function(){return ea},JO:function(){return v},K8:function(){return d},K_:function(){return eK},LY:function(){return ez},Lp:function(){return eJ},Mx:function(){return ts},N3:function(){return eb},N8:function(){return et},NL:function(){return e2},NV:function(){return f},Nc:function(){return eO},O3:function(){return eL},OD:function(){return ej},OU:function(){return ep},Of:function(){return N},Og:function(){return y},Ou:function(){return ti},Ov:function(){return j},PT:function(){return Y},Pv:function(){return tl},Qg:function(){return eF},RQ:function(){return _},Rg:function(){return W},Sb:function(){return eR},So:function(){return eo},TF:function(){return tn},Tj:function(){return eW},UM:function(){return te},VA:function(){return G},Vt:function(){return eq},W_:function(){return V},X:function(){return en},XB:function(){return tc},XO:function(){return k},Xd:function(){return eE},Xm:function(){return F},YU:function(){return eZ},Yo:function(){return I},Z9:function(){return z},Zr:function(){return g},a6:function(){return O},aC:function(){return ta},ao:function(){return eY},b1:function(){return eh},cq:function(){return J},cu:function(){return eA},e2:function(){return ek},eH:function(){return $},eZ:function(){return ex},fE:function(){return tt},fP:function(){return ee},g:function(){return e0},gX:function(){return ev},h3:function(){return ei},hT:function(){return eS},hy:function(){return u},ix:function(){return q},j2:function(){return ec},jA:function(){return eX},jE:function(){return eV},kK:function(){return w},kn:function(){return X},lP:function(){return h},lU:function(){return e5},lg:function(){return eC},mC:function(){return e7},mR:function(){return er},mY:function(){return e9},m_:function(){return L},mp:function(){return eD},n$:function(){return ef},n9:function(){return e8},nd:function(){return e4},o6:function(){return Q},oC:function(){return eN},ol:function(){return U},pf:function(){return eU},qI:function(){return m},qk:function(){return e1},qm:function(){return p},r1:function(){return tr},r6:function(){return B},rs:function(){return S},s0:function(){return M},sN:function(){return eG},t$:function(){return P},t0:function(){return eT},t3:function(){return e3},tB:function(){return e6},tN:function(){return ed},u5:function(){return es},v9:function(){return eg},vh:function(){return eI},wX:function(){return T},wd:function(){return ew},xA:function(){return ey},xX:function(){return R},zg:function(){return eu}});var r=o(20347),a=o(41021);let n=null;console.log=function(){};let c=0,s=e=>new Promise(t=>setTimeout(t,e)),i=async e=>{let t=Date.now();t-c>6e4?(e.includes("Authentication Error - Expired Key")&&(a.ZP.info("UI Session Expired. Logging out."),c=t,await s(3e3),document.cookie="token=; expires=Thu, 01 Jan 1970 00:00:00 UTC; path=/;",window.location.href="/"),c=t):console.log("Error suppressed to prevent spam:",e)},l="Authorization";function d(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"Authorization";console.log("setGlobalLitellmHeaderName: ".concat(e)),l=e}let h=async()=>{let e=n?"".concat(n,"/openapi.json"):"/openapi.json",t=await fetch(e);return await t.json()},p=async e=>{try{let t=n?"".concat(n,"/get/litellm_model_cost_map"):"/get/litellm_model_cost_map",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}}),r=await o.json();return console.log("received litellm model cost data: ".concat(r)),r}catch(e){throw console.error("Failed to get model cost map:",e),e}},w=async(e,t)=>{try{let o=n?"".concat(n,"/model/new"):"/model/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text()||"Network response was not ok";throw a.ZP.error(e),Error(e)}let c=await r.json();return console.log("API Response:",c),a.ZP.destroy(),a.ZP.success("Model ".concat(t.model_name," created successfully"),2),c}catch(e){throw console.error("Failed to create key:",e),e}},u=async e=>{try{let t=n?"".concat(n,"/model/settings"):"/model/settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){console.error("Failed to get model settings:",e)}},y=async(e,t)=>{console.log("model_id in model delete call: ".concat(t));try{let o=n?"".concat(n,"/model/delete"):"/model/delete",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({id:t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},f=async(e,t)=>{if(console.log("budget_id in budget delete call: ".concat(t)),null!=e)try{let o=n?"".concat(n,"/budget/delete"):"/budget/delete",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({id:t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},g=async(e,t)=>{try{console.log("Form Values in budgetCreateCall:",t),console.log("Form Values after check:",t);let o=n?"".concat(n,"/budget/new"):"/budget/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},m=async(e,t)=>{try{console.log("Form Values in budgetUpdateCall:",t),console.log("Form Values after check:",t);let o=n?"".concat(n,"/budget/update"):"/budget/update",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},k=async(e,t)=>{try{let o=n?"".concat(n,"/invitation/new"):"/invitation/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},_=async e=>{try{let t=n?"".concat(n,"/alerting/settings"):"/alerting/settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},T=async(e,t,o)=>{try{if(console.log("Form Values in keyCreateCall:",o),o.description&&(o.metadata||(o.metadata={}),o.metadata.description=o.description,delete o.description,o.metadata=JSON.stringify(o.metadata)),o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}console.log("Form Values after check:",o);let r=n?"".concat(n,"/key/generate"):"/key/generate",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t,...o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error(e)}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},j=async(e,t,o)=>{try{if(console.log("Form Values in keyCreateCall:",o),o.description&&(o.metadata||(o.metadata={}),o.metadata.description=o.description,delete o.description,o.metadata=JSON.stringify(o.metadata)),o.auto_create_key=!1,o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}console.log("Form Values after check:",o);let r=n?"".concat(n,"/user/new"):"/user/new",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t,...o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error(e)}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},E=async(e,t)=>{try{let o=n?"".concat(n,"/key/delete"):"/key/delete";console.log("in keyDeleteCall:",t);let r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({keys:[t]})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to create key:",e),e}},C=async(e,t)=>{try{let o=n?"".concat(n,"/user/delete"):"/user/delete";console.log("in userDeleteCall:",t);let r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_ids:t})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to delete user(s):",e),e}},S=async(e,t)=>{try{let o=n?"".concat(n,"/team/delete"):"/team/delete";console.log("in teamDeleteCall:",t);let r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_ids:[t]})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to delete key:",e),e}},N=async function(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:null,o=arguments.length>2&&void 0!==arguments[2]?arguments[2]:null,r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:null,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null,c=arguments.length>5&&void 0!==arguments[5]?arguments[5]:null,s=arguments.length>6&&void 0!==arguments[6]?arguments[6]:null,d=arguments.length>7&&void 0!==arguments[7]?arguments[7]:null,h=arguments.length>8&&void 0!==arguments[8]?arguments[8]:null,p=arguments.length>9&&void 0!==arguments[9]?arguments[9]:null;try{let w=n?"".concat(n,"/user/list"):"/user/list";console.log("in userListCall");let u=new URLSearchParams;if(t&&t.length>0){let e=t.join(",");u.append("user_ids",e)}o&&u.append("page",o.toString()),r&&u.append("page_size",r.toString()),a&&u.append("user_email",a),c&&u.append("role",c),s&&u.append("team",s),d&&u.append("sso_user_ids",d),h&&u.append("sort_by",h),p&&u.append("sort_order",p);let y=u.toString();y&&(w+="?".concat(y));let f=await fetch(w,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!f.ok){let e=await f.text();throw i(e),Error("Network response was not ok")}let g=await f.json();return console.log("/user/list API Response:",g),g}catch(e){throw console.error("Failed to create key:",e),e}},b=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]&&arguments[3],a=arguments.length>4?arguments[4]:void 0,c=arguments.length>5?arguments[5]:void 0,s=arguments.length>6&&void 0!==arguments[6]&&arguments[6];console.log("userInfoCall: ".concat(t,", ").concat(o,", ").concat(r,", ").concat(a,", ").concat(c,", ").concat(s));try{let d;if(r){d=n?"".concat(n,"/user/list"):"/user/list";let e=new URLSearchParams;null!=a&&e.append("page",a.toString()),null!=c&&e.append("page_size",c.toString()),d+="?".concat(e.toString())}else d=n?"".concat(n,"/user/info"):"/user/info",("Admin"!==o&&"Admin Viewer"!==o||s)&&t&&(d+="?user_id=".concat(t));console.log("Requesting user data from:",d);let h=await fetch(d,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}let p=await h.json();return console.log("API Response:",p),p}catch(e){throw console.error("Failed to fetch user data:",e),e}},F=async(e,t)=>{try{let o=n?"".concat(n,"/team/info"):"/team/info";t&&(o="".concat(o,"?team_id=").concat(t)),console.log("in teamInfoCall");let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},v=async function(e,t){let o=arguments.length>2&&void 0!==arguments[2]?arguments[2]:null,r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:null,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null;arguments.length>5&&void 0!==arguments[5]&&arguments[5],arguments.length>6&&void 0!==arguments[6]&&arguments[6],arguments.length>7&&void 0!==arguments[7]&&arguments[7],arguments.length>8&&void 0!==arguments[8]&&arguments[8];try{let c=n?"".concat(n,"/v2/team/list"):"/v2/team/list";console.log("in teamInfoCall");let s=new URLSearchParams;o&&s.append("user_id",o.toString()),t&&s.append("organization_id",t.toString()),r&&s.append("team_id",r.toString()),a&&s.append("team_alias",a.toString());let d=s.toString();d&&(c+="?".concat(d));let h=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}let p=await h.json();return console.log("/v2/team/list API Response:",p),p}catch(e){throw console.error("Failed to create key:",e),e}},x=async function(e,t){let o=arguments.length>2&&void 0!==arguments[2]?arguments[2]:null,r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:null,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null;try{let c=n?"".concat(n,"/team/list"):"/team/list";console.log("in teamInfoCall");let s=new URLSearchParams;o&&s.append("user_id",o.toString()),t&&s.append("organization_id",t.toString()),r&&s.append("team_id",r.toString()),a&&s.append("team_alias",a.toString());let d=s.toString();d&&(c+="?".concat(d));let h=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}let p=await h.json();return console.log("/team/list API Response:",p),p}catch(e){throw console.error("Failed to create key:",e),e}},O=async e=>{try{let t=n?"".concat(n,"/team/available"):"/team/available";console.log("in availableTeamListCall");let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("/team/available_teams API Response:",r),r}catch(e){throw e}},B=async e=>{try{let t=n?"".concat(n,"/organization/list"):"/organization/list",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},P=async(e,t)=>{try{let o=n?"".concat(n,"/organization/info"):"/organization/info";t&&(o="".concat(o,"?organization_id=").concat(t)),console.log("in teamInfoCall");let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},A=async(e,t)=>{try{if(console.log("Form Values in organizationCreateCall:",t),t.metadata){console.log("formValues.metadata:",t.metadata);try{t.metadata=JSON.parse(t.metadata)}catch(e){throw console.error("Failed to parse metadata:",e),Error("Failed to parse metadata: "+e)}}let o=n?"".concat(n,"/organization/new"):"/organization/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},G=async(e,t)=>{try{console.log("Form Values in organizationUpdateCall:",t);let o=n?"".concat(n,"/organization/update"):"/organization/update",r=await fetch(o,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("Update Team Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},J=async(e,t)=>{try{let o=n?"".concat(n,"/organization/delete"):"/organization/delete",r=await fetch(o,{method:"DELETE",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({organization_ids:[t]})});if(!r.ok){let e=await r.text();throw i(e),Error("Error deleting organization: ".concat(e))}return await r.json()}catch(e){throw console.error("Failed to delete organization:",e),e}},I=async(e,t)=>{try{let o=n?"".concat(n,"/utils/transform_request"):"/utils/transform_request",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to create key:",e),e}},R=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:1;try{let a=n?"".concat(n,"/user/daily/activity"):"/user/daily/activity",c=new URLSearchParams;c.append("start_date",t.toISOString()),c.append("end_date",o.toISOString()),c.append("page_size","1000"),c.append("page",r.toString());let s=c.toString();s&&(a+="?".concat(s));let d=await fetch(a,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!d.ok){let e=await d.text();throw i(e),Error("Network response was not ok")}return await d.json()}catch(e){throw console.error("Failed to create key:",e),e}},z=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:1,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null;try{let c=n?"".concat(n,"/tag/daily/activity"):"/tag/daily/activity",s=new URLSearchParams;s.append("start_date",t.toISOString()),s.append("end_date",o.toISOString()),s.append("page_size","1000"),s.append("page",r.toString()),a&&s.append("tags",a.join(","));let d=s.toString();d&&(c+="?".concat(d));let h=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}return await h.json()}catch(e){throw console.error("Failed to create key:",e),e}},U=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:1,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null;try{let c=n?"".concat(n,"/team/daily/activity"):"/team/daily/activity",s=new URLSearchParams;s.append("start_date",t.toISOString()),s.append("end_date",o.toISOString()),s.append("page_size","1000"),s.append("page",r.toString()),a&&s.append("team_ids",a.join(",")),s.append("exclude_team_ids","litellm-dashboard");let d=s.toString();d&&(c+="?".concat(d));let h=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}return await h.json()}catch(e){throw console.error("Failed to create key:",e),e}},V=async e=>{try{let t=n?"".concat(n,"/onboarding/get_token"):"/onboarding/get_token";t+="?invite_link=".concat(e);let o=await fetch(t,{method:"GET",headers:{"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},L=async(e,t,o,r)=>{let a=n?"".concat(n,"/onboarding/claim_token"):"/onboarding/claim_token";try{let n=await fetch(a,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({invitation_link:t,user_id:o,password:r})});if(!n.ok){let e=await n.text();throw i(e),Error("Network response was not ok")}let c=await n.json();return console.log(c),c}catch(e){throw console.error("Failed to delete key:",e),e}},M=async(e,t,o)=>{try{let r=n?"".concat(n,"/key/").concat(t,"/regenerate"):"/key/".concat(t,"/regenerate"),a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(o)});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("Regenerate key Response:",c),c}catch(e){throw console.error("Failed to regenerate key:",e),e}},Z=!1,D=null,H=async(e,t,o)=>{try{console.log("modelInfoCall:",e,t,o);let c=n?"".concat(n,"/v2/model/info"):"/v2/model/info";r.ZL.includes(o)||(c+="?user_models_only=true");let s=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!s.ok){let e=await s.text();throw e+="error shown=".concat(Z),Z||(e.includes("No model list passed")&&(e="No Models Exist. Click Add Model to get started."),a.ZP.info(e,10),Z=!0,D&&clearTimeout(D),D=setTimeout(()=>{Z=!1},1e4)),Error("Network response was not ok")}let i=await s.json();return console.log("modelInfoCall:",i),i}catch(e){throw console.error("Failed to create key:",e),e}},q=async(e,t)=>{try{let o=n?"".concat(n,"/v1/model/info"):"/v1/model/info";o+="?litellm_model_id=".concat(t);let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok)throw await r.text(),Error("Network response was not ok");let a=await r.json();return console.log("modelInfoV1Call:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},X=async e=>{try{let t=n?"".concat(n,"/model_group/info"):"/model_group/info",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log("modelHubCall:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},Y=async e=>{try{let t=n?"".concat(n,"/get/allowed_ips"):"/get/allowed_ips",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw Error("Network response was not ok: ".concat(e))}let r=await o.json();return console.log("getAllowedIPs:",r),r.data}catch(e){throw console.error("Failed to get allowed IPs:",e),e}},$=async(e,t)=>{try{let o=n?"".concat(n,"/add/allowed_ip"):"/add/allowed_ip",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({ip:t})});if(!r.ok){let e=await r.text();throw Error("Network response was not ok: ".concat(e))}let a=await r.json();return console.log("addAllowedIP:",a),a}catch(e){throw console.error("Failed to add allowed IP:",e),e}},K=async(e,t)=>{try{let o=n?"".concat(n,"/delete/allowed_ip"):"/delete/allowed_ip",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({ip:t})});if(!r.ok){let e=await r.text();throw Error("Network response was not ok: ".concat(e))}let a=await r.json();return console.log("deleteAllowedIP:",a),a}catch(e){throw console.error("Failed to delete allowed IP:",e),e}},Q=async(e,t,o,r,a,c,s,d)=>{try{let t=n?"".concat(n,"/model/metrics"):"/model/metrics";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(c,"&api_key=").concat(s,"&customer=").concat(d));let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},W=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/model/streaming_metrics"):"/model/streaming_metrics";t&&(a="".concat(a,"?_selected_model_group=").concat(t,"&startTime=").concat(o,"&endTime=").concat(r));let c=await fetch(a,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!c.ok){let e=await c.text();throw i(e),Error("Network response was not ok")}return await c.json()}catch(e){throw console.error("Failed to create key:",e),e}},ee=async(e,t,o,r,a,c,s,d)=>{try{let t=n?"".concat(n,"/model/metrics/slow_responses"):"/model/metrics/slow_responses";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(c,"&api_key=").concat(s,"&customer=").concat(d));let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},et=async(e,t,o,r,a,c,s,d)=>{try{let t=n?"".concat(n,"/model/metrics/exceptions"):"/model/metrics/exceptions";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(c,"&api_key=").concat(s,"&customer=").concat(d));let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},eo=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]&&arguments[3],a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null;console.log("in /models calls, globalLitellmHeaderName",l);try{let t=n?"".concat(n,"/models"):"/models",o=new URLSearchParams;!0===r&&o.append("return_wildcard_routes","True"),a&&o.append("team_id",a.toString()),o.toString()&&(t+="?".concat(o.toString()));let c=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!c.ok){let e=await c.text();throw i(e),Error("Network response was not ok")}return await c.json()}catch(e){throw console.error("Failed to create key:",e),e}},er=async e=>{try{let t=n?"".concat(n,"/global/spend/teams"):"/global/spend/teams";console.log("in teamSpendLogsCall:",t);let o=await fetch("".concat(t),{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},ea=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/spend/tags"):"/global/spend/tags";t&&o&&(a="".concat(a,"?start_date=").concat(t,"&end_date=").concat(o)),r&&(a+="".concat(a,"&tags=").concat(r.join(","))),console.log("in tagsSpendLogsCall:",a);let c=await fetch("".concat(a),{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!c.ok)throw await c.text(),Error("Network response was not ok");let s=await c.json();return console.log(s),s}catch(e){throw console.error("Failed to create key:",e),e}},en=async e=>{try{let t=n?"".concat(n,"/global/spend/all_tag_names"):"/global/spend/all_tag_names";console.log("in global/spend/all_tag_names call",t);let o=await fetch("".concat(t),{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},ec=async e=>{try{let t=n?"".concat(n,"/global/all_end_users"):"/global/all_end_users";console.log("in global/all_end_users call",t);let o=await fetch("".concat(t),{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},es=async(e,t)=>{try{let o=n?"".concat(n,"/user/filter/ui"):"/user/filter/ui";t.get("user_email")&&(o+="?user_email=".concat(t.get("user_email"))),t.get("user_id")&&(o+="?user_id=".concat(t.get("user_id")));let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to create key:",e),e}},ei=async(e,t,o,r,a,c,s,d,h)=>{try{let p=n?"".concat(n,"/spend/logs/ui"):"/spend/logs/ui",w=new URLSearchParams;t&&w.append("api_key",t),o&&w.append("team_id",o),r&&w.append("request_id",r),a&&w.append("start_date",a),c&&w.append("end_date",c),s&&w.append("page",s.toString()),d&&w.append("page_size",d.toString()),h&&w.append("user_id",h);let u=w.toString();u&&(p+="?".concat(u));let y=await fetch(p,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!y.ok){let e=await y.text();throw i(e),Error("Network response was not ok")}let f=await y.json();return console.log("Spend Logs Response:",f),f}catch(e){throw console.error("Failed to fetch spend logs:",e),e}},el=async e=>{try{let t=n?"".concat(n,"/global/spend/logs"):"/global/spend/logs",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},ed=async e=>{try{let t=n?"".concat(n,"/global/spend/keys?limit=5"):"/global/spend/keys?limit=5",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},eh=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/spend/end_users"):"/global/spend/end_users",c="";c=t?JSON.stringify({api_key:t,startTime:o,endTime:r}):JSON.stringify({startTime:o,endTime:r});let s={method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:c},d=await fetch(a,s);if(!d.ok){let e=await d.text();throw i(e),Error("Network response was not ok")}let h=await d.json();return console.log(h),h}catch(e){throw console.error("Failed to create key:",e),e}},ep=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/spend/provider"):"/global/spend/provider";o&&r&&(a+="?start_date=".concat(o,"&end_date=").concat(r)),t&&(a+="&api_key=".concat(t));let c={method:"GET",headers:{[l]:"Bearer ".concat(e)}},s=await fetch(a,c);if(!s.ok){let e=await s.text();throw i(e),Error("Network response was not ok")}let d=await s.json();return console.log(d),d}catch(e){throw console.error("Failed to fetch spend data:",e),e}},ew=async(e,t,o)=>{try{let r=n?"".concat(n,"/global/activity"):"/global/activity";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[l]:"Bearer ".concat(e)}},c=await fetch(r,a);if(!c.ok)throw await c.text(),Error("Network response was not ok");let s=await c.json();return console.log(s),s}catch(e){throw console.error("Failed to fetch spend data:",e),e}},eu=async(e,t,o)=>{try{let r=n?"".concat(n,"/global/activity/cache_hits"):"/global/activity/cache_hits";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[l]:"Bearer ".concat(e)}},c=await fetch(r,a);if(!c.ok)throw await c.text(),Error("Network response was not ok");let s=await c.json();return console.log(s),s}catch(e){throw console.error("Failed to fetch spend data:",e),e}},ey=async(e,t,o)=>{try{let r=n?"".concat(n,"/global/activity/model"):"/global/activity/model";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[l]:"Bearer ".concat(e)}},c=await fetch(r,a);if(!c.ok)throw await c.text(),Error("Network response was not ok");let s=await c.json();return console.log(s),s}catch(e){throw console.error("Failed to fetch spend data:",e),e}},ef=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/activity/exceptions"):"/global/activity/exceptions";t&&o&&(a+="?start_date=".concat(t,"&end_date=").concat(o)),r&&(a+="&model_group=".concat(r));let c={method:"GET",headers:{[l]:"Bearer ".concat(e)}},s=await fetch(a,c);if(!s.ok)throw await s.text(),Error("Network response was not ok");let i=await s.json();return console.log(i),i}catch(e){throw console.error("Failed to fetch spend data:",e),e}},eg=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/activity/exceptions/deployment"):"/global/activity/exceptions/deployment";t&&o&&(a+="?start_date=".concat(t,"&end_date=").concat(o)),r&&(a+="&model_group=".concat(r));let c={method:"GET",headers:{[l]:"Bearer ".concat(e)}},s=await fetch(a,c);if(!s.ok)throw await s.text(),Error("Network response was not ok");let i=await s.json();return console.log(i),i}catch(e){throw console.error("Failed to fetch spend data:",e),e}},em=async e=>{try{let t=n?"".concat(n,"/global/spend/models?limit=5"):"/global/spend/models?limit=5",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},ek=async(e,t)=>{try{let o=n?"".concat(n,"/v2/key/info"):"/v2/key/info",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({keys:t})});if(!r.ok){let e=await r.text();if(e.includes("Invalid proxy server token passed"))throw Error("Invalid proxy server token passed");throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to create key:",e),e}},e_=async(e,t,o)=>{try{console.log("Sending model connection test request:",JSON.stringify(t));let a=n?"".concat(n,"/health/test_connection"):"/health/test_connection",c=await fetch(a,{method:"POST",headers:{"Content-Type":"application/json",[l]:"Bearer ".concat(e)},body:JSON.stringify({litellm_params:t,mode:o})}),s=c.headers.get("content-type");if(!s||!s.includes("application/json")){let e=await c.text();throw console.error("Received non-JSON response:",e),Error("Received non-JSON response (".concat(c.status,": ").concat(c.statusText,"). Check network tab for details."))}let i=await c.json();if(!c.ok||"error"===i.status){if("error"===i.status);else{var r;return{status:"error",message:(null===(r=i.error)||void 0===r?void 0:r.message)||"Connection test failed: ".concat(c.status," ").concat(c.statusText)}}}return i}catch(e){throw console.error("Model connection test error:",e),e}},eT=async(e,t)=>{try{console.log("entering keyInfoV1Call");let o=n?"".concat(n,"/key/info"):"/key/info";o="".concat(o,"?key=").concat(t);let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(console.log("response",r),!r.ok){let e=await r.text();i(e),a.ZP.error("Failed to fetch key info - "+e)}let c=await r.json();return console.log("data",c),c}catch(e){throw console.error("Failed to fetch key info:",e),e}},ej=async function(e,t,o,r,a,c,s,d){let h=arguments.length>8&&void 0!==arguments[8]?arguments[8]:null,p=arguments.length>9&&void 0!==arguments[9]?arguments[9]:null;try{let w=n?"".concat(n,"/key/list"):"/key/list";console.log("in keyListCall");let u=new URLSearchParams;o&&u.append("team_id",o.toString()),t&&u.append("organization_id",t.toString()),r&&u.append("key_alias",r),c&&u.append("key_hash",c),a&&u.append("user_id",a.toString()),s&&u.append("page",s.toString()),d&&u.append("size",d.toString()),h&&u.append("sort_by",h),p&&u.append("sort_order",p),u.append("return_full_object","true"),u.append("include_team_keys","true");let y=u.toString();y&&(w+="?".concat(y));let f=await fetch(w,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!f.ok){let e=await f.text();throw i(e),Error("Network response was not ok")}let g=await f.json();return console.log("/team/list API Response:",g),g}catch(e){throw console.error("Failed to create key:",e),e}},eE=async(e,t)=>{try{let o=n?"".concat(n,"/user/get_users?role=").concat(t):"/user/get_users?role=".concat(t);console.log("in userGetAllUsersCall:",o);let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to get requested models:",e),e}},eC=async e=>{try{let t=n?"".concat(n,"/user/available_roles"):"/user/available_roles",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log("response from user/available_role",r),r}catch(e){throw e}},eS=async(e,t)=>{try{if(console.log("Form Values in teamCreateCall:",t),t.metadata){console.log("formValues.metadata:",t.metadata);try{t.metadata=JSON.parse(t.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}let o=n?"".concat(n,"/team/new"):"/team/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},eN=async(e,t)=>{try{if(console.log("Form Values in credentialCreateCall:",t),t.metadata){console.log("formValues.metadata:",t.metadata);try{t.metadata=JSON.parse(t.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}let o=n?"".concat(n,"/credentials"):"/credentials",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},eb=async e=>{try{let t=n?"".concat(n,"/credentials"):"/credentials";console.log("in credentialListCall");let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("/credentials API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},eF=async(e,t,o)=>{try{let r=n?"".concat(n,"/credentials"):"/credentials";t?r+="/by_name/".concat(t):o&&(r+="/by_model/".concat(o)),console.log("in credentialListCall");let a=await fetch(r,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("/credentials API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},ev=async(e,t)=>{try{let o=n?"".concat(n,"/credentials/").concat(t):"/credentials/".concat(t);console.log("in credentialDeleteCall:",t);let r=await fetch(o,{method:"DELETE",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to delete key:",e),e}},ex=async(e,t,o)=>{try{if(console.log("Form Values in credentialUpdateCall:",o),o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}let r=n?"".concat(n,"/credentials/").concat(t):"/credentials/".concat(t),a=await fetch(r,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},eO=async(e,t)=>{try{if(console.log("Form Values in keyUpdateCall:",t),t.model_tpm_limit){console.log("formValues.model_tpm_limit:",t.model_tpm_limit);try{t.model_tpm_limit=JSON.parse(t.model_tpm_limit)}catch(e){throw Error("Failed to parse model_tpm_limit: "+e)}}if(t.model_rpm_limit){console.log("formValues.model_rpm_limit:",t.model_rpm_limit);try{t.model_rpm_limit=JSON.parse(t.model_rpm_limit)}catch(e){throw Error("Failed to parse model_rpm_limit: "+e)}}let o=n?"".concat(n,"/key/update"):"/key/update",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("Update key Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},eB=async(e,t)=>{try{console.log("Form Values in teamUpateCall:",t);let o=n?"".concat(n,"/team/update"):"/team/update",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),a.ZP.error("Failed to update team settings: "+e),Error(e)}let c=await r.json();return console.log("Update Team Response:",c),c}catch(e){throw console.error("Failed to update team:",e),e}},eP=async(e,t,o)=>{try{console.log("Form Values in modelUpateCall:",t);let r=n?"".concat(n,"/model/").concat(o,"/update"):"/model/".concat(o,"/update"),a=await fetch(r,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error update from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("Update model Response:",c),c}catch(e){throw console.error("Failed to update model:",e),e}},eA=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=n?"".concat(n,"/team/member_add"):"/team/member_add",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_id:t,member:o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},eG=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=n?"".concat(n,"/team/member_update"):"/team/member_update",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_id:t,role:o.role,user_id:o.user_id})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},eJ=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=n?"".concat(n,"/team/member_delete"):"/team/member_delete",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_id:t,...void 0!==o.user_email&&{user_email:o.user_email},...void 0!==o.user_id&&{user_id:o.user_id}})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},eI=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=n?"".concat(n,"/organization/member_add"):"/organization/member_add",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({organization_id:t,member:o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error(e)}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create organization member:",e),e}},eR=async(e,t,o)=>{try{console.log("Form Values in organizationMemberDeleteCall:",o);let r=n?"".concat(n,"/organization/member_delete"):"/organization/member_delete",a=await fetch(r,{method:"DELETE",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({organization_id:t,user_id:o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to delete organization member:",e),e}},ez=async(e,t,o)=>{try{console.log("Form Values in organizationMemberUpdateCall:",o);let r=n?"".concat(n,"/organization/member_update"):"/organization/member_update",a=await fetch(r,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({organization_id:t,...o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to update organization member:",e),e}},eU=async(e,t,o)=>{try{console.log("Form Values in userUpdateUserCall:",t);let r=n?"".concat(n,"/user/update"):"/user/update",a={...t};null!==o&&(a.user_role=o),a=JSON.stringify(a);let c=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:a});if(!c.ok){let e=await c.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let s=await c.json();return console.log("API Response:",s),s}catch(e){throw console.error("Failed to create key:",e),e}},eV=async(e,t)=>{try{let o=n?"".concat(n,"/health/services?service=").concat(t):"/health/services?service=".concat(t);console.log("Checking Slack Budget Alerts service health");let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error(e)}let c=await r.json();return a.ZP.success("Test request to ".concat(t," made - check logs/alerts on ").concat(t," to verify")),c}catch(e){throw console.error("Failed to perform health check:",e),e}},eL=async e=>{try{let t=n?"".concat(n,"/budget/list"):"/budget/list",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eM=async(e,t,o)=>{try{let t=n?"".concat(n,"/get/config/callbacks"):"/get/config/callbacks",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eZ=async e=>{try{let t=n?"".concat(n,"/config/list?config_type=general_settings"):"/config/list?config_type=general_settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eD=async e=>{try{let t=n?"".concat(n,"/config/pass_through_endpoint"):"/config/pass_through_endpoint",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eH=async(e,t)=>{try{let o=n?"".concat(n,"/config/field/info?field_name=").concat(t):"/config/field/info?field_name=".concat(t),r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok)throw await r.text(),Error("Network response was not ok");return await r.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eq=async(e,t)=>{try{let o=n?"".concat(n,"/config/pass_through_endpoint"):"/config/pass_through_endpoint",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eX=async(e,t,o)=>{try{let r=n?"".concat(n,"/config/field/update"):"/config/field/update",c=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({field_name:t,field_value:o,config_type:"general_settings"})});if(!c.ok){let e=await c.text();throw i(e),Error("Network response was not ok")}let s=await c.json();return a.ZP.success("Successfully updated value!"),s}catch(e){throw console.error("Failed to set callbacks:",e),e}},eY=async(e,t)=>{try{let o=n?"".concat(n,"/config/field/delete"):"/config/field/delete",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({field_name:t,config_type:"general_settings"})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let c=await r.json();return a.ZP.success("Field reset on proxy"),c}catch(e){throw console.error("Failed to get callbacks:",e),e}},e$=async(e,t)=>{try{let o=n?"".concat(n,"/config/pass_through_endpoint?endpoint_id=").concat(t):"/config/pass_through_endpoint".concat(t),r=await fetch(o,{method:"DELETE",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eK=async(e,t)=>{try{let o=n?"".concat(n,"/config/update"):"/config/update",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eQ=async e=>{try{let t=n?"".concat(n,"/health"):"/health",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to call /health:",e),e}},eW=async e=>{try{let t=n?"".concat(n,"/cache/ping"):"/cache/ping",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error(e)}return await o.json()}catch(e){throw console.error("Failed to call /cache/ping:",e),e}},e0=async e=>{try{let t=n?"".concat(n,"/sso/get/ui_settings"):"/sso/get/ui_settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},e3=async e=>{try{let t=n?"".concat(n,"/guardrails/list"):"/guardrails/list",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("Guardrails list response:",r),r}catch(e){throw console.error("Failed to fetch guardrails list:",e),e}},e1=async(e,t,o)=>{try{let r=n?"".concat(n,"/spend/logs/ui/").concat(t,"?start_date=").concat(encodeURIComponent(o)):"/spend/logs/ui/".concat(t,"?start_date=").concat(encodeURIComponent(o));console.log("Fetching log details from:",r);let a=await fetch(r,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("Fetched log details:",c),c}catch(e){throw console.error("Failed to fetch log details:",e),e}},e2=async e=>{try{let t=n?"".concat(n,"/get/internal_user_settings"):"/get/internal_user_settings";console.log("Fetching SSO settings from:",t);let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("Fetched SSO settings:",r),r}catch(e){throw console.error("Failed to fetch SSO settings:",e),e}},e4=async(e,t)=>{try{let o=n?"".concat(n,"/update/internal_user_settings"):"/update/internal_user_settings";console.log("Updating internal user settings:",t);let r=await fetch(o,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let c=await r.json();return console.log("Updated internal user settings:",c),a.ZP.success("Internal user settings updated successfully"),c}catch(e){throw console.error("Failed to update internal user settings:",e),e}},e5=async e=>{try{let t=n?"".concat(n,"/mcp/tools/list"):"/mcp/tools/list";console.log("Fetching MCP tools from:",t);let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("Fetched MCP tools:",r),r}catch(e){throw console.error("Failed to fetch MCP tools:",e),e}},e6=async(e,t,o)=>{try{let r=n?"".concat(n,"/mcp/tools/call"):"/mcp/tools/call";console.log("Calling MCP tool:",t,"with arguments:",o);let a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({name:t,arguments:o})});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("MCP tool call response:",c),c}catch(e){throw console.error("Failed to call MCP tool:",e),e}},e9=async(e,t)=>{try{let o=n?"".concat(n,"/tag/new"):"/tag/new",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();await i(e);return}return await r.json()}catch(e){throw console.error("Error creating tag:",e),e}},e8=async(e,t)=>{try{let o=n?"".concat(n,"/tag/update"):"/tag/update",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();await i(e);return}return await r.json()}catch(e){throw console.error("Error updating tag:",e),e}},e7=async(e,t)=>{try{let o=n?"".concat(n,"/tag/info"):"/tag/info",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify({names:t})});if(!r.ok){let e=await r.text();return await i(e),{}}return await r.json()}catch(e){throw console.error("Error getting tag info:",e),e}},te=async e=>{try{let t=n?"".concat(n,"/tag/list"):"/tag/list",o=await fetch(t,{method:"GET",headers:{Authorization:"Bearer ".concat(e)}});if(!o.ok){let e=await o.text();return await i(e),{}}return await o.json()}catch(e){throw console.error("Error listing tags:",e),e}},tt=async(e,t)=>{try{let o=n?"".concat(n,"/tag/delete"):"/tag/delete",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify({name:t})});if(!r.ok){let e=await r.text();await i(e);return}return await r.json()}catch(e){throw console.error("Error deleting tag:",e),e}},to=async e=>{try{let t=n?"".concat(n,"/get/default_team_settings"):"/get/default_team_settings";console.log("Fetching default team settings from:",t);let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("Fetched default team settings:",r),r}catch(e){throw console.error("Failed to fetch default team settings:",e),e}},tr=async(e,t)=>{try{let o=n?"".concat(n,"/update/default_team_settings"):"/update/default_team_settings";console.log("Updating default team settings:",t);let r=await fetch(o,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let c=await r.json();return console.log("Updated default team settings:",c),a.ZP.success("Default team settings updated successfully"),c}catch(e){throw console.error("Failed to update default team settings:",e),e}},ta=async(e,t)=>{try{let o=n?"".concat(n,"/team/permissions_list?team_id=").concat(t):"/team/permissions_list?team_id=".concat(t),r=await fetch(o,{method:"GET",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log("Team permissions response:",a),a}catch(e){throw console.error("Failed to get team permissions:",e),e}},tn=async(e,t,o)=>{try{let r=n?"".concat(n,"/team/permissions_update"):"/team/permissions_update",a=await fetch(r,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify({team_id:t,team_member_permissions:o})});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("Team permissions response:",c),c}catch(e){throw console.error("Failed to update team permissions:",e),e}},tc=async(e,t)=>{try{let o=n?"".concat(n,"/spend/logs/session/ui?session_id=").concat(encodeURIComponent(t)):"/spend/logs/session/ui?session_id=".concat(encodeURIComponent(t)),r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to fetch session logs:",e),e}},ts=async(e,t)=>{try{let o=n?"".concat(n,"/vector_store/new"):"/vector_store/new",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify(t)});if(!r.ok){let e=await r.json();throw Error(e.detail||"Failed to create vector store")}return await r.json()}catch(e){throw console.error("Error creating vector store:",e),e}},ti=async function(e){arguments.length>1&&void 0!==arguments[1]&&arguments[1],arguments.length>2&&void 0!==arguments[2]&&arguments[2];try{let t=n?"".concat(n,"/vector_store/list"):"/vector_store/list",o=await fetch(t,{method:"GET",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)}});if(!o.ok){let e=await o.json();throw Error(e.detail||"Failed to list vector stores")}return await o.json()}catch(e){throw console.error("Error listing vector stores:",e),e}},tl=async(e,t)=>{try{let o=n?"".concat(n,"/vector_store/delete"):"/vector_store/delete",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify({vector_store_id:t})});if(!r.ok){let e=await r.json();throw Error(e.detail||"Failed to delete vector store")}return await r.json()}catch(e){throw console.error("Error deleting vector store:",e),e}}},20347:function(e,t,o){o.d(t,{LQ:function(){return n},ZL:function(){return r},lo:function(){return a},tY:function(){return c}});let r=["Admin","Admin Viewer","proxy_admin","proxy_admin_viewer","org_admin"],a=["Internal User","Internal Viewer"],n=["Internal User","Admin"],c=e=>r.includes(e)}}]); \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/_next/static/chunks/250-ca9d788a8d1c6d58.js b/litellm/proxy/_experimental/out/_next/static/chunks/250-ca9d788a8d1c6d58.js new file mode 100644 index 00000000000..e3a28966fd1 --- /dev/null +++ b/litellm/proxy/_experimental/out/_next/static/chunks/250-ca9d788a8d1c6d58.js @@ -0,0 +1 @@ +"use strict";(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[250],{19250:function(e,t,o){o.d(t,{$D:function(){return eP},$I:function(){return $},AZ:function(){return H},Au:function(){return em},BL:function(){return eM},Br:function(){return F},E9:function(){return eH},EB:function(){return tr},EG:function(){return eK},EY:function(){return eQ},Eb:function(){return C},FC:function(){return el},Gh:function(){return eO},H1:function(){return G},H2:function(){return n},Hx:function(){return e_},I1:function(){return E},It:function(){return x},J$:function(){return ea},JO:function(){return v},K8:function(){return d},K_:function(){return e$},Ko:function(){return tm},LY:function(){return ez},Lp:function(){return eJ},Mx:function(){return ti},N3:function(){return eF},N8:function(){return et},NL:function(){return e4},NV:function(){return f},Nc:function(){return eB},O3:function(){return eL},OD:function(){return ej},OU:function(){return ep},Of:function(){return N},Og:function(){return g},Ou:function(){return tl},Ov:function(){return j},PC:function(){return e1},PT:function(){return Y},Pv:function(){return td},Qg:function(){return eb},RQ:function(){return _},Rg:function(){return W},Sb:function(){return eR},So:function(){return eo},TF:function(){return tc},Tj:function(){return eW},UM:function(){return tt},VA:function(){return A},Vt:function(){return eq},W_:function(){return V},X:function(){return en},XB:function(){return ts},XO:function(){return k},Xd:function(){return eE},Xm:function(){return b},YU:function(){return eZ},Yi:function(){return tu},Yo:function(){return I},Z9:function(){return z},Zr:function(){return y},a6:function(){return B},aC:function(){return tn},ao:function(){return eY},b1:function(){return eh},cq:function(){return J},cu:function(){return eG},e2:function(){return ek},eH:function(){return K},eW:function(){return tf},eZ:function(){return ex},fE:function(){return to},fP:function(){return ee},fk:function(){return ty},g:function(){return e0},gX:function(){return ev},h3:function(){return ei},hT:function(){return eS},hy:function(){return u},ix:function(){return q},j2:function(){return ec},jA:function(){return eX},jE:function(){return eV},jr:function(){return tw},kK:function(){return w},kn:function(){return X},lP:function(){return h},lU:function(){return e6},lg:function(){return eC},mC:function(){return te},mR:function(){return er},mY:function(){return e8},m_:function(){return L},mp:function(){return eD},n$:function(){return ef},n9:function(){return e7},nd:function(){return e5},o6:function(){return Q},oC:function(){return eN},ol:function(){return U},pf:function(){return eU},pu:function(){return th},qI:function(){return m},qW:function(){return tp},qd:function(){return tg},qk:function(){return e2},qm:function(){return p},r1:function(){return ta},r6:function(){return O},rs:function(){return S},s0:function(){return M},sN:function(){return eA},t$:function(){return P},t0:function(){return eT},t3:function(){return e3},tB:function(){return e9},tN:function(){return ed},u5:function(){return es},v9:function(){return ey},vh:function(){return eI},wX:function(){return T},wd:function(){return ew},xA:function(){return eg},xX:function(){return R},zg:function(){return eu}});var r=o(20347),a=o(41021);let n=null;console.log=function(){};let c=0,s=e=>new Promise(t=>setTimeout(t,e)),i=async e=>{let t=Date.now();t-c>6e4?(e.includes("Authentication Error - Expired Key")&&(a.ZP.info("UI Session Expired. Logging out."),c=t,await s(3e3),document.cookie="token=; expires=Thu, 01 Jan 1970 00:00:00 UTC; path=/;",window.location.href="/"),c=t):console.log("Error suppressed to prevent spam:",e)},l="Authorization";function d(){let e=arguments.length>0&&void 0!==arguments[0]?arguments[0]:"Authorization";console.log("setGlobalLitellmHeaderName: ".concat(e)),l=e}let h=async()=>{let e=n?"".concat(n,"/openapi.json"):"/openapi.json",t=await fetch(e);return await t.json()},p=async e=>{try{let t=n?"".concat(n,"/get/litellm_model_cost_map"):"/get/litellm_model_cost_map",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}}),r=await o.json();return console.log("received litellm model cost data: ".concat(r)),r}catch(e){throw console.error("Failed to get model cost map:",e),e}},w=async(e,t)=>{try{let o=n?"".concat(n,"/model/new"):"/model/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text()||"Network response was not ok";throw a.ZP.error(e),Error(e)}let c=await r.json();return console.log("API Response:",c),a.ZP.destroy(),a.ZP.success("Model ".concat(t.model_name," created successfully"),2),c}catch(e){throw console.error("Failed to create key:",e),e}},u=async e=>{try{let t=n?"".concat(n,"/model/settings"):"/model/settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){console.error("Failed to get model settings:",e)}},g=async(e,t)=>{console.log("model_id in model delete call: ".concat(t));try{let o=n?"".concat(n,"/model/delete"):"/model/delete",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({id:t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},f=async(e,t)=>{if(console.log("budget_id in budget delete call: ".concat(t)),null!=e)try{let o=n?"".concat(n,"/budget/delete"):"/budget/delete",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({id:t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},y=async(e,t)=>{try{console.log("Form Values in budgetCreateCall:",t),console.log("Form Values after check:",t);let o=n?"".concat(n,"/budget/new"):"/budget/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},m=async(e,t)=>{try{console.log("Form Values in budgetUpdateCall:",t),console.log("Form Values after check:",t);let o=n?"".concat(n,"/budget/update"):"/budget/update",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},k=async(e,t)=>{try{let o=n?"".concat(n,"/invitation/new"):"/invitation/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},_=async e=>{try{let t=n?"".concat(n,"/alerting/settings"):"/alerting/settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},T=async(e,t,o)=>{try{if(console.log("Form Values in keyCreateCall:",o),o.description&&(o.metadata||(o.metadata={}),o.metadata.description=o.description,delete o.description,o.metadata=JSON.stringify(o.metadata)),o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}console.log("Form Values after check:",o);let r=n?"".concat(n,"/key/generate"):"/key/generate",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t,...o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error(e)}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},j=async(e,t,o)=>{try{if(console.log("Form Values in keyCreateCall:",o),o.description&&(o.metadata||(o.metadata={}),o.metadata.description=o.description,delete o.description,o.metadata=JSON.stringify(o.metadata)),o.auto_create_key=!1,o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}console.log("Form Values after check:",o);let r=n?"".concat(n,"/user/new"):"/user/new",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_id:t,...o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error(e)}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},E=async(e,t)=>{try{let o=n?"".concat(n,"/key/delete"):"/key/delete";console.log("in keyDeleteCall:",t);let r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({keys:[t]})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to create key:",e),e}},C=async(e,t)=>{try{let o=n?"".concat(n,"/user/delete"):"/user/delete";console.log("in userDeleteCall:",t);let r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({user_ids:t})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to delete user(s):",e),e}},S=async(e,t)=>{try{let o=n?"".concat(n,"/team/delete"):"/team/delete";console.log("in teamDeleteCall:",t);let r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_ids:[t]})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to delete key:",e),e}},N=async function(e){let t=arguments.length>1&&void 0!==arguments[1]?arguments[1]:null,o=arguments.length>2&&void 0!==arguments[2]?arguments[2]:null,r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:null,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null,c=arguments.length>5&&void 0!==arguments[5]?arguments[5]:null,s=arguments.length>6&&void 0!==arguments[6]?arguments[6]:null,d=arguments.length>7&&void 0!==arguments[7]?arguments[7]:null,h=arguments.length>8&&void 0!==arguments[8]?arguments[8]:null,p=arguments.length>9&&void 0!==arguments[9]?arguments[9]:null;try{let w=n?"".concat(n,"/user/list"):"/user/list";console.log("in userListCall");let u=new URLSearchParams;if(t&&t.length>0){let e=t.join(",");u.append("user_ids",e)}o&&u.append("page",o.toString()),r&&u.append("page_size",r.toString()),a&&u.append("user_email",a),c&&u.append("role",c),s&&u.append("team",s),d&&u.append("sso_user_ids",d),h&&u.append("sort_by",h),p&&u.append("sort_order",p);let g=u.toString();g&&(w+="?".concat(g));let f=await fetch(w,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!f.ok){let e=await f.text();throw i(e),Error("Network response was not ok")}let y=await f.json();return console.log("/user/list API Response:",y),y}catch(e){throw console.error("Failed to create key:",e),e}},F=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]&&arguments[3],a=arguments.length>4?arguments[4]:void 0,c=arguments.length>5?arguments[5]:void 0,s=arguments.length>6&&void 0!==arguments[6]&&arguments[6];console.log("userInfoCall: ".concat(t,", ").concat(o,", ").concat(r,", ").concat(a,", ").concat(c,", ").concat(s));try{let d;if(r){d=n?"".concat(n,"/user/list"):"/user/list";let e=new URLSearchParams;null!=a&&e.append("page",a.toString()),null!=c&&e.append("page_size",c.toString()),d+="?".concat(e.toString())}else d=n?"".concat(n,"/user/info"):"/user/info",("Admin"!==o&&"Admin Viewer"!==o||s)&&t&&(d+="?user_id=".concat(t));console.log("Requesting user data from:",d);let h=await fetch(d,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}let p=await h.json();return console.log("API Response:",p),p}catch(e){throw console.error("Failed to fetch user data:",e),e}},b=async(e,t)=>{try{let o=n?"".concat(n,"/team/info"):"/team/info";t&&(o="".concat(o,"?team_id=").concat(t)),console.log("in teamInfoCall");let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},v=async function(e,t){let o=arguments.length>2&&void 0!==arguments[2]?arguments[2]:null,r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:null,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null;arguments.length>5&&void 0!==arguments[5]&&arguments[5],arguments.length>6&&void 0!==arguments[6]&&arguments[6],arguments.length>7&&void 0!==arguments[7]&&arguments[7],arguments.length>8&&void 0!==arguments[8]&&arguments[8];try{let c=n?"".concat(n,"/v2/team/list"):"/v2/team/list";console.log("in teamInfoCall");let s=new URLSearchParams;o&&s.append("user_id",o.toString()),t&&s.append("organization_id",t.toString()),r&&s.append("team_id",r.toString()),a&&s.append("team_alias",a.toString());let d=s.toString();d&&(c+="?".concat(d));let h=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}let p=await h.json();return console.log("/v2/team/list API Response:",p),p}catch(e){throw console.error("Failed to create key:",e),e}},x=async function(e,t){let o=arguments.length>2&&void 0!==arguments[2]?arguments[2]:null,r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:null,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null;try{let c=n?"".concat(n,"/team/list"):"/team/list";console.log("in teamInfoCall");let s=new URLSearchParams;o&&s.append("user_id",o.toString()),t&&s.append("organization_id",t.toString()),r&&s.append("team_id",r.toString()),a&&s.append("team_alias",a.toString());let d=s.toString();d&&(c+="?".concat(d));let h=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}let p=await h.json();return console.log("/team/list API Response:",p),p}catch(e){throw console.error("Failed to create key:",e),e}},B=async e=>{try{let t=n?"".concat(n,"/team/available"):"/team/available";console.log("in availableTeamListCall");let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("/team/available_teams API Response:",r),r}catch(e){throw e}},O=async e=>{try{let t=n?"".concat(n,"/organization/list"):"/organization/list",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},P=async(e,t)=>{try{let o=n?"".concat(n,"/organization/info"):"/organization/info";t&&(o="".concat(o,"?organization_id=").concat(t)),console.log("in teamInfoCall");let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},G=async(e,t)=>{try{if(console.log("Form Values in organizationCreateCall:",t),t.metadata){console.log("formValues.metadata:",t.metadata);try{t.metadata=JSON.parse(t.metadata)}catch(e){throw console.error("Failed to parse metadata:",e),Error("Failed to parse metadata: "+e)}}let o=n?"".concat(n,"/organization/new"):"/organization/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},A=async(e,t)=>{try{console.log("Form Values in organizationUpdateCall:",t);let o=n?"".concat(n,"/organization/update"):"/organization/update",r=await fetch(o,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("Update Team Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},J=async(e,t)=>{try{let o=n?"".concat(n,"/organization/delete"):"/organization/delete",r=await fetch(o,{method:"DELETE",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({organization_ids:[t]})});if(!r.ok){let e=await r.text();throw i(e),Error("Error deleting organization: ".concat(e))}return await r.json()}catch(e){throw console.error("Failed to delete organization:",e),e}},I=async(e,t)=>{try{let o=n?"".concat(n,"/utils/transform_request"):"/utils/transform_request",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to create key:",e),e}},R=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:1;try{let a=n?"".concat(n,"/user/daily/activity"):"/user/daily/activity",c=new URLSearchParams;c.append("start_date",t.toISOString()),c.append("end_date",o.toISOString()),c.append("page_size","1000"),c.append("page",r.toString());let s=c.toString();s&&(a+="?".concat(s));let d=await fetch(a,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!d.ok){let e=await d.text();throw i(e),Error("Network response was not ok")}return await d.json()}catch(e){throw console.error("Failed to create key:",e),e}},z=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:1,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null;try{let c=n?"".concat(n,"/tag/daily/activity"):"/tag/daily/activity",s=new URLSearchParams;s.append("start_date",t.toISOString()),s.append("end_date",o.toISOString()),s.append("page_size","1000"),s.append("page",r.toString()),a&&s.append("tags",a.join(","));let d=s.toString();d&&(c+="?".concat(d));let h=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}return await h.json()}catch(e){throw console.error("Failed to create key:",e),e}},U=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]?arguments[3]:1,a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null;try{let c=n?"".concat(n,"/team/daily/activity"):"/team/daily/activity",s=new URLSearchParams;s.append("start_date",t.toISOString()),s.append("end_date",o.toISOString()),s.append("page_size","1000"),s.append("page",r.toString()),a&&s.append("team_ids",a.join(",")),s.append("exclude_team_ids","litellm-dashboard");let d=s.toString();d&&(c+="?".concat(d));let h=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!h.ok){let e=await h.text();throw i(e),Error("Network response was not ok")}return await h.json()}catch(e){throw console.error("Failed to create key:",e),e}},V=async e=>{try{let t=n?"".concat(n,"/onboarding/get_token"):"/onboarding/get_token";t+="?invite_link=".concat(e);let o=await fetch(t,{method:"GET",headers:{"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},L=async(e,t,o,r)=>{let a=n?"".concat(n,"/onboarding/claim_token"):"/onboarding/claim_token";try{let n=await fetch(a,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({invitation_link:t,user_id:o,password:r})});if(!n.ok){let e=await n.text();throw i(e),Error("Network response was not ok")}let c=await n.json();return console.log(c),c}catch(e){throw console.error("Failed to delete key:",e),e}},M=async(e,t,o)=>{try{let r=n?"".concat(n,"/key/").concat(t,"/regenerate"):"/key/".concat(t,"/regenerate"),a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(o)});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("Regenerate key Response:",c),c}catch(e){throw console.error("Failed to regenerate key:",e),e}},Z=!1,D=null,H=async(e,t,o)=>{try{console.log("modelInfoCall:",e,t,o);let c=n?"".concat(n,"/v2/model/info"):"/v2/model/info";r.ZL.includes(o)||(c+="?user_models_only=true");let s=await fetch(c,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!s.ok){let e=await s.text();throw e+="error shown=".concat(Z),Z||(e.includes("No model list passed")&&(e="No Models Exist. Click Add Model to get started."),a.ZP.info(e,10),Z=!0,D&&clearTimeout(D),D=setTimeout(()=>{Z=!1},1e4)),Error("Network response was not ok")}let i=await s.json();return console.log("modelInfoCall:",i),i}catch(e){throw console.error("Failed to create key:",e),e}},q=async(e,t)=>{try{let o=n?"".concat(n,"/v1/model/info"):"/v1/model/info";o+="?litellm_model_id=".concat(t);let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok)throw await r.text(),Error("Network response was not ok");let a=await r.json();return console.log("modelInfoV1Call:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},X=async e=>{try{let t=n?"".concat(n,"/model_group/info"):"/model_group/info",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log("modelHubCall:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},Y=async e=>{try{let t=n?"".concat(n,"/get/allowed_ips"):"/get/allowed_ips",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw Error("Network response was not ok: ".concat(e))}let r=await o.json();return console.log("getAllowedIPs:",r),r.data}catch(e){throw console.error("Failed to get allowed IPs:",e),e}},K=async(e,t)=>{try{let o=n?"".concat(n,"/add/allowed_ip"):"/add/allowed_ip",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({ip:t})});if(!r.ok){let e=await r.text();throw Error("Network response was not ok: ".concat(e))}let a=await r.json();return console.log("addAllowedIP:",a),a}catch(e){throw console.error("Failed to add allowed IP:",e),e}},$=async(e,t)=>{try{let o=n?"".concat(n,"/delete/allowed_ip"):"/delete/allowed_ip",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({ip:t})});if(!r.ok){let e=await r.text();throw Error("Network response was not ok: ".concat(e))}let a=await r.json();return console.log("deleteAllowedIP:",a),a}catch(e){throw console.error("Failed to delete allowed IP:",e),e}},Q=async(e,t,o,r,a,c,s,d)=>{try{let t=n?"".concat(n,"/model/metrics"):"/model/metrics";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(c,"&api_key=").concat(s,"&customer=").concat(d));let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},W=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/model/streaming_metrics"):"/model/streaming_metrics";t&&(a="".concat(a,"?_selected_model_group=").concat(t,"&startTime=").concat(o,"&endTime=").concat(r));let c=await fetch(a,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!c.ok){let e=await c.text();throw i(e),Error("Network response was not ok")}return await c.json()}catch(e){throw console.error("Failed to create key:",e),e}},ee=async(e,t,o,r,a,c,s,d)=>{try{let t=n?"".concat(n,"/model/metrics/slow_responses"):"/model/metrics/slow_responses";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(c,"&api_key=").concat(s,"&customer=").concat(d));let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},et=async(e,t,o,r,a,c,s,d)=>{try{let t=n?"".concat(n,"/model/metrics/exceptions"):"/model/metrics/exceptions";r&&(t="".concat(t,"?_selected_model_group=").concat(r,"&startTime=").concat(a,"&endTime=").concat(c,"&api_key=").concat(s,"&customer=").concat(d));let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to create key:",e),e}},eo=async function(e,t,o){let r=arguments.length>3&&void 0!==arguments[3]&&arguments[3],a=arguments.length>4&&void 0!==arguments[4]?arguments[4]:null,c=arguments.length>5&&void 0!==arguments[5]&&arguments[5];console.log("in /models calls, globalLitellmHeaderName",l);try{let t=n?"".concat(n,"/models"):"/models",o=new URLSearchParams;!0===r&&o.append("return_wildcard_routes","True"),!0===c&&o.append("include_model_access_groups","True"),a&&o.append("team_id",a.toString()),o.toString()&&(t+="?".concat(o.toString()));let s=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!s.ok){let e=await s.text();throw i(e),Error("Network response was not ok")}return await s.json()}catch(e){throw console.error("Failed to create key:",e),e}},er=async e=>{try{let t=n?"".concat(n,"/global/spend/teams"):"/global/spend/teams";console.log("in teamSpendLogsCall:",t);let o=await fetch("".concat(t),{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},ea=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/spend/tags"):"/global/spend/tags";t&&o&&(a="".concat(a,"?start_date=").concat(t,"&end_date=").concat(o)),r&&(a+="".concat(a,"&tags=").concat(r.join(","))),console.log("in tagsSpendLogsCall:",a);let c=await fetch("".concat(a),{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!c.ok)throw await c.text(),Error("Network response was not ok");let s=await c.json();return console.log(s),s}catch(e){throw console.error("Failed to create key:",e),e}},en=async e=>{try{let t=n?"".concat(n,"/global/spend/all_tag_names"):"/global/spend/all_tag_names";console.log("in global/spend/all_tag_names call",t);let o=await fetch("".concat(t),{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},ec=async e=>{try{let t=n?"".concat(n,"/global/all_end_users"):"/global/all_end_users";console.log("in global/all_end_users call",t);let o=await fetch("".concat(t),{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},es=async(e,t)=>{try{let o=n?"".concat(n,"/user/filter/ui"):"/user/filter/ui";t.get("user_email")&&(o+="?user_email=".concat(t.get("user_email"))),t.get("user_id")&&(o+="?user_id=".concat(t.get("user_id")));let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to create key:",e),e}},ei=async(e,t,o,r,a,c,s,d,h,p,w)=>{try{let u=n?"".concat(n,"/spend/logs/ui"):"/spend/logs/ui",g=new URLSearchParams;t&&g.append("api_key",t),o&&g.append("team_id",o),r&&g.append("request_id",r),a&&g.append("start_date",a),c&&g.append("end_date",c),s&&g.append("page",s.toString()),d&&g.append("page_size",d.toString()),h&&g.append("user_id",h),p&&g.append("status_filter",p),w&&g.append("model",w);let f=g.toString();f&&(u+="?".concat(f));let y=await fetch(u,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!y.ok){let e=await y.text();throw i(e),Error("Network response was not ok")}let m=await y.json();return console.log("Spend Logs Response:",m),m}catch(e){throw console.error("Failed to fetch spend logs:",e),e}},el=async e=>{try{let t=n?"".concat(n,"/global/spend/logs"):"/global/spend/logs",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},ed=async e=>{try{let t=n?"".concat(n,"/global/spend/keys?limit=5"):"/global/spend/keys?limit=5",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},eh=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/spend/end_users"):"/global/spend/end_users",c="";c=t?JSON.stringify({api_key:t,startTime:o,endTime:r}):JSON.stringify({startTime:o,endTime:r});let s={method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:c},d=await fetch(a,s);if(!d.ok){let e=await d.text();throw i(e),Error("Network response was not ok")}let h=await d.json();return console.log(h),h}catch(e){throw console.error("Failed to create key:",e),e}},ep=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/spend/provider"):"/global/spend/provider";o&&r&&(a+="?start_date=".concat(o,"&end_date=").concat(r)),t&&(a+="&api_key=".concat(t));let c={method:"GET",headers:{[l]:"Bearer ".concat(e)}},s=await fetch(a,c);if(!s.ok){let e=await s.text();throw i(e),Error("Network response was not ok")}let d=await s.json();return console.log(d),d}catch(e){throw console.error("Failed to fetch spend data:",e),e}},ew=async(e,t,o)=>{try{let r=n?"".concat(n,"/global/activity"):"/global/activity";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[l]:"Bearer ".concat(e)}},c=await fetch(r,a);if(!c.ok)throw await c.text(),Error("Network response was not ok");let s=await c.json();return console.log(s),s}catch(e){throw console.error("Failed to fetch spend data:",e),e}},eu=async(e,t,o)=>{try{let r=n?"".concat(n,"/global/activity/cache_hits"):"/global/activity/cache_hits";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[l]:"Bearer ".concat(e)}},c=await fetch(r,a);if(!c.ok)throw await c.text(),Error("Network response was not ok");let s=await c.json();return console.log(s),s}catch(e){throw console.error("Failed to fetch spend data:",e),e}},eg=async(e,t,o)=>{try{let r=n?"".concat(n,"/global/activity/model"):"/global/activity/model";t&&o&&(r+="?start_date=".concat(t,"&end_date=").concat(o));let a={method:"GET",headers:{[l]:"Bearer ".concat(e)}},c=await fetch(r,a);if(!c.ok)throw await c.text(),Error("Network response was not ok");let s=await c.json();return console.log(s),s}catch(e){throw console.error("Failed to fetch spend data:",e),e}},ef=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/activity/exceptions"):"/global/activity/exceptions";t&&o&&(a+="?start_date=".concat(t,"&end_date=").concat(o)),r&&(a+="&model_group=".concat(r));let c={method:"GET",headers:{[l]:"Bearer ".concat(e)}},s=await fetch(a,c);if(!s.ok)throw await s.text(),Error("Network response was not ok");let i=await s.json();return console.log(i),i}catch(e){throw console.error("Failed to fetch spend data:",e),e}},ey=async(e,t,o,r)=>{try{let a=n?"".concat(n,"/global/activity/exceptions/deployment"):"/global/activity/exceptions/deployment";t&&o&&(a+="?start_date=".concat(t,"&end_date=").concat(o)),r&&(a+="&model_group=".concat(r));let c={method:"GET",headers:{[l]:"Bearer ".concat(e)}},s=await fetch(a,c);if(!s.ok)throw await s.text(),Error("Network response was not ok");let i=await s.json();return console.log(i),i}catch(e){throw console.error("Failed to fetch spend data:",e),e}},em=async e=>{try{let t=n?"".concat(n,"/global/spend/models?limit=5"):"/global/spend/models?limit=5",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log(r),r}catch(e){throw console.error("Failed to create key:",e),e}},ek=async(e,t)=>{try{let o=n?"".concat(n,"/v2/key/info"):"/v2/key/info",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({keys:t})});if(!r.ok){let e=await r.text();if(e.includes("Invalid proxy server token passed"))throw Error("Invalid proxy server token passed");throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to create key:",e),e}},e_=async(e,t,o)=>{try{console.log("Sending model connection test request:",JSON.stringify(t));let a=n?"".concat(n,"/health/test_connection"):"/health/test_connection",c=await fetch(a,{method:"POST",headers:{"Content-Type":"application/json",[l]:"Bearer ".concat(e)},body:JSON.stringify({litellm_params:t,mode:o})}),s=c.headers.get("content-type");if(!s||!s.includes("application/json")){let e=await c.text();throw console.error("Received non-JSON response:",e),Error("Received non-JSON response (".concat(c.status,": ").concat(c.statusText,"). Check network tab for details."))}let i=await c.json();if(!c.ok||"error"===i.status){if("error"===i.status);else{var r;return{status:"error",message:(null===(r=i.error)||void 0===r?void 0:r.message)||"Connection test failed: ".concat(c.status," ").concat(c.statusText)}}}return i}catch(e){throw console.error("Model connection test error:",e),e}},eT=async(e,t)=>{try{console.log("entering keyInfoV1Call");let o=n?"".concat(n,"/key/info"):"/key/info";o="".concat(o,"?key=").concat(t);let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(console.log("response",r),!r.ok){let e=await r.text();i(e),a.ZP.error("Failed to fetch key info - "+e)}let c=await r.json();return console.log("data",c),c}catch(e){throw console.error("Failed to fetch key info:",e),e}},ej=async function(e,t,o,r,a,c,s,d){let h=arguments.length>8&&void 0!==arguments[8]?arguments[8]:null,p=arguments.length>9&&void 0!==arguments[9]?arguments[9]:null;try{let w=n?"".concat(n,"/key/list"):"/key/list";console.log("in keyListCall");let u=new URLSearchParams;o&&u.append("team_id",o.toString()),t&&u.append("organization_id",t.toString()),r&&u.append("key_alias",r),c&&u.append("key_hash",c),a&&u.append("user_id",a.toString()),s&&u.append("page",s.toString()),d&&u.append("size",d.toString()),h&&u.append("sort_by",h),p&&u.append("sort_order",p),u.append("return_full_object","true"),u.append("include_team_keys","true");let g=u.toString();g&&(w+="?".concat(g));let f=await fetch(w,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!f.ok){let e=await f.text();throw i(e),Error("Network response was not ok")}let y=await f.json();return console.log("/team/list API Response:",y),y}catch(e){throw console.error("Failed to create key:",e),e}},eE=async(e,t)=>{try{let o=n?"".concat(n,"/user/get_users?role=").concat(t):"/user/get_users?role=".concat(t);console.log("in userGetAllUsersCall:",o);let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to get requested models:",e),e}},eC=async e=>{try{let t=n?"".concat(n,"/user/available_roles"):"/user/available_roles",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");let r=await o.json();return console.log("response from user/available_role",r),r}catch(e){throw e}},eS=async(e,t)=>{try{if(console.log("Form Values in teamCreateCall:",t),t.metadata){console.log("formValues.metadata:",t.metadata);try{t.metadata=JSON.parse(t.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}let o=n?"".concat(n,"/team/new"):"/team/new",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},eN=async(e,t)=>{try{if(console.log("Form Values in credentialCreateCall:",t),t.metadata){console.log("formValues.metadata:",t.metadata);try{t.metadata=JSON.parse(t.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}let o=n?"".concat(n,"/credentials"):"/credentials",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("API Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},eF=async e=>{try{let t=n?"".concat(n,"/credentials"):"/credentials";console.log("in credentialListCall");let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("/credentials API Response:",r),r}catch(e){throw console.error("Failed to create key:",e),e}},eb=async(e,t,o)=>{try{let r=n?"".concat(n,"/credentials"):"/credentials";t?r+="/by_name/".concat(t):o&&(r+="/by_model/".concat(o)),console.log("in credentialListCall");let a=await fetch(r,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("/credentials API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},ev=async(e,t)=>{try{let o=n?"".concat(n,"/credentials/").concat(t):"/credentials/".concat(t);console.log("in credentialDeleteCall:",t);let r=await fetch(o,{method:"DELETE",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log(a),a}catch(e){throw console.error("Failed to delete key:",e),e}},ex=async(e,t,o)=>{try{if(console.log("Form Values in credentialUpdateCall:",o),o.metadata){console.log("formValues.metadata:",o.metadata);try{o.metadata=JSON.parse(o.metadata)}catch(e){throw Error("Failed to parse metadata: "+e)}}let r=n?"".concat(n,"/credentials/").concat(t):"/credentials/".concat(t),a=await fetch(r,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},eB=async(e,t)=>{try{if(console.log("Form Values in keyUpdateCall:",t),t.model_tpm_limit){console.log("formValues.model_tpm_limit:",t.model_tpm_limit);try{t.model_tpm_limit=JSON.parse(t.model_tpm_limit)}catch(e){throw Error("Failed to parse model_tpm_limit: "+e)}}if(t.model_rpm_limit){console.log("formValues.model_rpm_limit:",t.model_rpm_limit);try{t.model_rpm_limit=JSON.parse(t.model_rpm_limit)}catch(e){throw Error("Failed to parse model_rpm_limit: "+e)}}let o=n?"".concat(n,"/key/update"):"/key/update",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let a=await r.json();return console.log("Update key Response:",a),a}catch(e){throw console.error("Failed to create key:",e),e}},eO=async(e,t)=>{try{console.log("Form Values in teamUpateCall:",t);let o=n?"".concat(n,"/team/update"):"/team/update",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),console.error("Error response from the server:",e),a.ZP.error("Failed to update team settings: "+e),Error(e)}let c=await r.json();return console.log("Update Team Response:",c),c}catch(e){throw console.error("Failed to update team:",e),e}},eP=async(e,t,o)=>{try{console.log("Form Values in modelUpateCall:",t);let r=n?"".concat(n,"/model/").concat(o,"/update"):"/model/".concat(o,"/update"),a=await fetch(r,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error update from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("Update model Response:",c),c}catch(e){throw console.error("Failed to update model:",e),e}},eG=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=n?"".concat(n,"/team/member_add"):"/team/member_add",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_id:t,member:o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},eA=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=n?"".concat(n,"/team/member_update"):"/team/member_update",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_id:t,role:o.role,user_id:o.user_id})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},eJ=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=n?"".concat(n,"/team/member_delete"):"/team/member_delete",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({team_id:t,...void 0!==o.user_email&&{user_email:o.user_email},...void 0!==o.user_id&&{user_id:o.user_id}})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create key:",e),e}},eI=async(e,t,o)=>{try{console.log("Form Values in teamMemberAddCall:",o);let r=n?"".concat(n,"/organization/member_add"):"/organization/member_add",a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({organization_id:t,member:o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error(e)}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to create organization member:",e),e}},eR=async(e,t,o)=>{try{console.log("Form Values in organizationMemberDeleteCall:",o);let r=n?"".concat(n,"/organization/member_delete"):"/organization/member_delete",a=await fetch(r,{method:"DELETE",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({organization_id:t,user_id:o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to delete organization member:",e),e}},ez=async(e,t,o)=>{try{console.log("Form Values in organizationMemberUpdateCall:",o);let r=n?"".concat(n,"/organization/member_update"):"/organization/member_update",a=await fetch(r,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({organization_id:t,...o})});if(!a.ok){let e=await a.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let c=await a.json();return console.log("API Response:",c),c}catch(e){throw console.error("Failed to update organization member:",e),e}},eU=async(e,t,o)=>{try{console.log("Form Values in userUpdateUserCall:",t);let r=n?"".concat(n,"/user/update"):"/user/update",a={...t};null!==o&&(a.user_role=o),a=JSON.stringify(a);let c=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:a});if(!c.ok){let e=await c.text();throw i(e),console.error("Error response from the server:",e),Error("Network response was not ok")}let s=await c.json();return console.log("API Response:",s),s}catch(e){throw console.error("Failed to create key:",e),e}},eV=async(e,t)=>{try{let o=n?"".concat(n,"/health/services?service=").concat(t):"/health/services?service=".concat(t);console.log("Checking Slack Budget Alerts service health");let r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error(e)}let c=await r.json();return a.ZP.success("Test request to ".concat(t," made - check logs/alerts on ").concat(t," to verify")),c}catch(e){throw console.error("Failed to perform health check:",e),e}},eL=async e=>{try{let t=n?"".concat(n,"/budget/list"):"/budget/list",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eM=async(e,t,o)=>{try{let t=n?"".concat(n,"/get/config/callbacks"):"/get/config/callbacks",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eZ=async e=>{try{let t=n?"".concat(n,"/config/list?config_type=general_settings"):"/config/list?config_type=general_settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eD=async e=>{try{let t=n?"".concat(n,"/config/pass_through_endpoint"):"/config/pass_through_endpoint",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},eH=async(e,t)=>{try{let o=n?"".concat(n,"/config/field/info?field_name=").concat(t):"/config/field/info?field_name=".concat(t),r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok)throw await r.text(),Error("Network response was not ok");return await r.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eq=async(e,t)=>{try{let o=n?"".concat(n,"/config/pass_through_endpoint"):"/config/pass_through_endpoint",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eX=async(e,t,o)=>{try{let r=n?"".concat(n,"/config/field/update"):"/config/field/update",c=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({field_name:t,field_value:o,config_type:"general_settings"})});if(!c.ok){let e=await c.text();throw i(e),Error("Network response was not ok")}let s=await c.json();return a.ZP.success("Successfully updated value!"),s}catch(e){throw console.error("Failed to set callbacks:",e),e}},eY=async(e,t)=>{try{let o=n?"".concat(n,"/config/field/delete"):"/config/field/delete",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({field_name:t,config_type:"general_settings"})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let c=await r.json();return a.ZP.success("Field reset on proxy"),c}catch(e){throw console.error("Failed to get callbacks:",e),e}},eK=async(e,t)=>{try{let o=n?"".concat(n,"/config/pass_through_endpoint?endpoint_id=").concat(t):"/config/pass_through_endpoint".concat(t),r=await fetch(o,{method:"DELETE",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},e$=async(e,t)=>{try{let o=n?"".concat(n,"/config/update"):"/config/update",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({...t})});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to set callbacks:",e),e}},eQ=async e=>{try{let t=n?"".concat(n,"/health"):"/health",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to call /health:",e),e}},eW=async e=>{try{let t=n?"".concat(n,"/cache/ping"):"/cache/ping",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error(e)}return await o.json()}catch(e){throw console.error("Failed to call /cache/ping:",e),e}},e0=async e=>{try{let t=n?"".concat(n,"/sso/get/ui_settings"):"/sso/get/ui_settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok)throw await o.text(),Error("Network response was not ok");return await o.json()}catch(e){throw console.error("Failed to get callbacks:",e),e}},e3=async e=>{try{let t=n?"".concat(n,"/v2/guardrails/list"):"/v2/guardrails/list",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}return await o.json()}catch(e){throw console.error("Failed to get guardrails list:",e),e}},e1=async(e,t)=>{try{let o=n?"".concat(n,"/guardrails"):"/guardrails",r=await fetch(o,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({guardrail:t})});if(!r.ok){let e=await r.text();throw i(e),Error(e)}let a=await r.json();return console.log("Create guardrail response:",a),a}catch(e){throw console.error("Failed to create guardrail:",e),e}},e2=async(e,t,o)=>{try{let r=n?"".concat(n,"/spend/logs/ui/").concat(t,"?start_date=").concat(encodeURIComponent(o)):"/spend/logs/ui/".concat(t,"?start_date=").concat(encodeURIComponent(o));console.log("Fetching log details from:",r);let a=await fetch(r,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("Fetched log details:",c),c}catch(e){throw console.error("Failed to fetch log details:",e),e}},e4=async e=>{try{let t=n?"".concat(n,"/get/internal_user_settings"):"/get/internal_user_settings";console.log("Fetching SSO settings from:",t);let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("Fetched SSO settings:",r),r}catch(e){throw console.error("Failed to fetch SSO settings:",e),e}},e5=async(e,t)=>{try{let o=n?"".concat(n,"/update/internal_user_settings"):"/update/internal_user_settings";console.log("Updating internal user settings:",t);let r=await fetch(o,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let c=await r.json();return console.log("Updated internal user settings:",c),a.ZP.success("Internal user settings updated successfully"),c}catch(e){throw console.error("Failed to update internal user settings:",e),e}},e6=async e=>{try{let t=n?"".concat(n,"/mcp/tools/list"):"/mcp/tools/list";console.log("Fetching MCP tools from:",t);let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("Fetched MCP tools:",r),r}catch(e){throw console.error("Failed to fetch MCP tools:",e),e}},e9=async(e,t,o)=>{try{let r=n?"".concat(n,"/mcp/tools/call"):"/mcp/tools/call";console.log("Calling MCP tool:",t,"with arguments:",o);let a=await fetch(r,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify({name:t,arguments:o})});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("MCP tool call response:",c),c}catch(e){throw console.error("Failed to call MCP tool:",e),e}},e8=async(e,t)=>{try{let o=n?"".concat(n,"/tag/new"):"/tag/new",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();await i(e);return}return await r.json()}catch(e){throw console.error("Error creating tag:",e),e}},e7=async(e,t)=>{try{let o=n?"".concat(n,"/tag/update"):"/tag/update",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();await i(e);return}return await r.json()}catch(e){throw console.error("Error updating tag:",e),e}},te=async(e,t)=>{try{let o=n?"".concat(n,"/tag/info"):"/tag/info",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify({names:t})});if(!r.ok){let e=await r.text();return await i(e),{}}return await r.json()}catch(e){throw console.error("Error getting tag info:",e),e}},tt=async e=>{try{let t=n?"".concat(n,"/tag/list"):"/tag/list",o=await fetch(t,{method:"GET",headers:{Authorization:"Bearer ".concat(e)}});if(!o.ok){let e=await o.text();return await i(e),{}}return await o.json()}catch(e){throw console.error("Error listing tags:",e),e}},to=async(e,t)=>{try{let o=n?"".concat(n,"/tag/delete"):"/tag/delete",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify({name:t})});if(!r.ok){let e=await r.text();await i(e);return}return await r.json()}catch(e){throw console.error("Error deleting tag:",e),e}},tr=async e=>{try{let t=n?"".concat(n,"/get/default_team_settings"):"/get/default_team_settings";console.log("Fetching default team settings from:",t);let o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Network response was not ok")}let r=await o.json();return console.log("Fetched default team settings:",r),r}catch(e){throw console.error("Failed to fetch default team settings:",e),e}},ta=async(e,t)=>{try{let o=n?"".concat(n,"/update/default_team_settings"):"/update/default_team_settings";console.log("Updating default team settings:",t);let r=await fetch(o,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let c=await r.json();return console.log("Updated default team settings:",c),a.ZP.success("Default team settings updated successfully"),c}catch(e){throw console.error("Failed to update default team settings:",e),e}},tn=async(e,t)=>{try{let o=n?"".concat(n,"/team/permissions_list?team_id=").concat(t):"/team/permissions_list?team_id=".concat(t),r=await fetch(o,{method:"GET",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}let a=await r.json();return console.log("Team permissions response:",a),a}catch(e){throw console.error("Failed to get team permissions:",e),e}},tc=async(e,t,o)=>{try{let r=n?"".concat(n,"/team/permissions_update"):"/team/permissions_update",a=await fetch(r,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify({team_id:t,team_member_permissions:o})});if(!a.ok){let e=await a.text();throw i(e),Error("Network response was not ok")}let c=await a.json();return console.log("Team permissions response:",c),c}catch(e){throw console.error("Failed to update team permissions:",e),e}},ts=async(e,t)=>{try{let o=n?"".concat(n,"/spend/logs/session/ui?session_id=").concat(encodeURIComponent(t)):"/spend/logs/session/ui?session_id=".concat(encodeURIComponent(t)),r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Network response was not ok")}return await r.json()}catch(e){throw console.error("Failed to fetch session logs:",e),e}},ti=async(e,t)=>{try{let o=n?"".concat(n,"/vector_store/new"):"/vector_store/new",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify(t)});if(!r.ok){let e=await r.json();throw Error(e.detail||"Failed to create vector store")}return await r.json()}catch(e){throw console.error("Error creating vector store:",e),e}},tl=async function(e){arguments.length>1&&void 0!==arguments[1]&&arguments[1],arguments.length>2&&void 0!==arguments[2]&&arguments[2];try{let t=n?"".concat(n,"/vector_store/list"):"/vector_store/list",o=await fetch(t,{method:"GET",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)}});if(!o.ok){let e=await o.json();throw Error(e.detail||"Failed to list vector stores")}return await o.json()}catch(e){throw console.error("Error listing vector stores:",e),e}},td=async(e,t)=>{try{let o=n?"".concat(n,"/vector_store/delete"):"/vector_store/delete",r=await fetch(o,{method:"POST",headers:{"Content-Type":"application/json",Authorization:"Bearer ".concat(e)},body:JSON.stringify({vector_store_id:t})});if(!r.ok){let e=await r.json();throw Error(e.detail||"Failed to delete vector store")}return await r.json()}catch(e){throw console.error("Error deleting vector store:",e),e}},th=async e=>{try{let t=n?"".concat(n,"/email/event_settings"):"/email/event_settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Failed to get email event settings")}let r=await o.json();return console.log("Email event settings response:",r),r}catch(e){throw console.error("Failed to get email event settings:",e),e}},tp=async(e,t)=>{try{let o=n?"".concat(n,"/email/event_settings"):"/email/event_settings",r=await fetch(o,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(t)});if(!r.ok){let e=await r.text();throw i(e),Error("Failed to update email event settings")}let a=await r.json();return console.log("Update email event settings response:",a),a}catch(e){throw console.error("Failed to update email event settings:",e),e}},tw=async e=>{try{let t=n?"".concat(n,"/email/event_settings/reset"):"/email/event_settings/reset",o=await fetch(t,{method:"POST",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Failed to reset email event settings")}let r=await o.json();return console.log("Reset email event settings response:",r),r}catch(e){throw console.error("Failed to reset email event settings:",e),e}},tu=async(e,t)=>{try{let o=n?"".concat(n,"/guardrails/").concat(t):"/guardrails/".concat(t),r=await fetch(o,{method:"DELETE",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error(e)}let a=await r.json();return console.log("Delete guardrail response:",a),a}catch(e){throw console.error("Failed to delete guardrail:",e),e}},tg=async e=>{try{let t=n?"".concat(n,"/guardrails/ui/add_guardrail_settings"):"/guardrails/ui/add_guardrail_settings",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Failed to get guardrail UI settings")}let r=await o.json();return console.log("Guardrail UI settings response:",r),r}catch(e){throw console.error("Failed to get guardrail UI settings:",e),e}},tf=async e=>{try{let t=n?"".concat(n,"/guardrails/ui/provider_specific_params"):"/guardrails/ui/provider_specific_params",o=await fetch(t,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!o.ok){let e=await o.text();throw i(e),Error("Failed to get guardrail provider specific parameters")}let r=await o.json();return console.log("Guardrail provider specific params response:",r),r}catch(e){throw console.error("Failed to get guardrail provider specific parameters:",e),e}},ty=async(e,t)=>{try{let o=n?"".concat(n,"/guardrails/").concat(t,"/info"):"/guardrails/".concat(t,"/info"),r=await fetch(o,{method:"GET",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"}});if(!r.ok){let e=await r.text();throw i(e),Error("Failed to get guardrail info")}let a=await r.json();return console.log("Guardrail info response:",a),a}catch(e){throw console.error("Failed to get guardrail info:",e),e}},tm=async(e,t,o)=>{try{let r=n?"".concat(n,"/guardrails/").concat(t):"/guardrails/".concat(t),a=await fetch(r,{method:"PATCH",headers:{[l]:"Bearer ".concat(e),"Content-Type":"application/json"},body:JSON.stringify(o)});if(!a.ok){let e=await a.text();throw i(e),Error("Failed to update guardrail")}let c=await a.json();return console.log("Update guardrail response:",c),c}catch(e){throw console.error("Failed to update guardrail:",e),e}}},20347:function(e,t,o){o.d(t,{LQ:function(){return n},ZL:function(){return r},lo:function(){return a},tY:function(){return c}});let r=["Admin","Admin Viewer","proxy_admin","proxy_admin_viewer","org_admin"],a=["Internal User","Internal Viewer"],n=["Internal User","Admin"],c=e=>r.includes(e)}}]); \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/_next/static/chunks/261-ee7f0f1f1c8c22a0.js b/litellm/proxy/_experimental/out/_next/static/chunks/261-ee7f0f1f1c8c22a0.js deleted file mode 100644 index 78658f0a2ec..00000000000 --- a/litellm/proxy/_experimental/out/_next/static/chunks/261-ee7f0f1f1c8c22a0.js +++ /dev/null @@ -1 +0,0 @@ -(self.webpackChunk_N_E=self.webpackChunk_N_E||[]).push([[261],{23639:function(e,t,n){"use strict";n.d(t,{Z:function(){return s}});var a=n(1119),r=n(2265),i={icon:{tag:"svg",attrs:{viewBox:"64 64 896 896",focusable:"false"},children:[{tag:"path",attrs:{d:"M832 64H296c-4.4 0-8 3.6-8 8v56c0 4.4 3.6 8 8 8h496v688c0 4.4 3.6 8 8 8h56c4.4 0 8-3.6 8-8V96c0-17.7-14.3-32-32-32zM704 192H192c-17.7 0-32 14.3-32 32v530.7c0 8.5 3.4 16.6 9.4 22.6l173.3 173.3c2.2 2.2 4.7 4 7.4 5.5v1.9h4.2c3.5 1.3 7.2 2 11 2H704c17.7 0 32-14.3 32-32V224c0-17.7-14.3-32-32-32zM350 856.2L263.9 770H350v86.2zM664 888H414V746c0-22.1-17.9-40-40-40H232V264h432v624z"}}]},name:"copy",theme:"outlined"},o=n(55015),s=r.forwardRef(function(e,t){return r.createElement(o.Z,(0,a.Z)({},e,{ref:t,icon:i}))})},77565:function(e,t,n){"use strict";n.d(t,{Z:function(){return s}});var a=n(1119),r=n(2265),i={icon:{tag:"svg",attrs:{viewBox:"64 64 896 896",focusable:"false"},children:[{tag:"path",attrs:{d:"M765.7 486.8L314.9 134.7A7.97 7.97 0 00302 141v77.3c0 4.9 2.3 9.6 6.1 12.6l360 281.1-360 281.1c-3.9 3-6.1 7.7-6.1 12.6V883c0 6.7 7.7 10.4 12.9 6.3l450.8-352.1a31.96 31.96 0 000-50.4z"}}]},name:"right",theme:"outlined"},o=n(55015),s=r.forwardRef(function(e,t){return r.createElement(o.Z,(0,a.Z)({},e,{ref:t,icon:i}))})},12485:function(e,t,n){"use strict";n.d(t,{Z:function(){return p}});var a=n(5853),r=n(31492),i=n(26898),o=n(97324),s=n(1153),l=n(2265),c=n(35242),u=n(42698);n(64016),n(8710),n(33232);let d=(0,s.fn)("Tab"),p=l.forwardRef((e,t)=>{let{icon:n,className:p,children:g}=e,m=(0,a._T)(e,["icon","className","children"]),b=(0,l.useContext)(c.O),f=(0,l.useContext)(u.Z);return l.createElement(r.O,Object.assign({ref:t,className:(0,o.q)(d("root"),"flex whitespace-nowrap truncate max-w-xs outline-none focus:ring-0 text-tremor-default transition duration-100",f?(0,s.bM)(f,i.K.text).selectTextColor:"solid"===b?"ui-selected:text-tremor-content-emphasis dark:ui-selected:text-dark-tremor-content-emphasis":"ui-selected:text-tremor-brand dark:ui-selected:text-dark-tremor-brand",function(e,t){switch(e){case"line":return(0,o.q)("ui-selected:border-b-2 hover:border-b-2 border-transparent transition duration-100 -mb-px px-2 py-2","hover:border-tremor-content hover:text-tremor-content-emphasis text-tremor-content","dark:hover:border-dark-tremor-content-emphasis dark:hover:text-dark-tremor-content-emphasis dark:text-dark-tremor-content",t?(0,s.bM)(t,i.K.border).selectBorderColor:"ui-selected:border-tremor-brand dark:ui-selected:border-dark-tremor-brand");case"solid":return(0,o.q)("border-transparent border rounded-tremor-small px-2.5 py-1","ui-selected:border-tremor-border ui-selected:bg-tremor-background ui-selected:shadow-tremor-input hover:text-tremor-content-emphasis ui-selected:text-tremor-brand","dark:ui-selected:border-dark-tremor-border dark:ui-selected:bg-dark-tremor-background dark:ui-selected:shadow-dark-tremor-input dark:hover:text-dark-tremor-content-emphasis dark:ui-selected:text-dark-tremor-brand",t?(0,s.bM)(t,i.K.text).selectTextColor:"text-tremor-content dark:text-dark-tremor-content")}}(b,f),p)},m),n?l.createElement(n,{className:(0,o.q)(d("icon"),"flex-none h-5 w-5",g?"mr-2":"")}):null,g?l.createElement("span",null,g):null)});p.displayName="Tab"},18135:function(e,t,n){"use strict";n.d(t,{Z:function(){return c}});var a=n(5853),r=n(31492),i=n(97324),o=n(1153),s=n(2265);let l=(0,o.fn)("TabGroup"),c=s.forwardRef((e,t)=>{let{defaultIndex:n,index:o,onIndexChange:c,children:u,className:d}=e,p=(0,a._T)(e,["defaultIndex","index","onIndexChange","children","className"]);return s.createElement(r.O.Group,Object.assign({as:"div",ref:t,defaultIndex:n,selectedIndex:o,onChange:c,className:(0,i.q)(l("root"),"w-full",d)},p),u)});c.displayName="TabGroup"},35242:function(e,t,n){"use strict";n.d(t,{O:function(){return c},Z:function(){return d}});var a=n(5853),r=n(2265),i=n(42698);n(64016),n(8710),n(33232);var o=n(31492),s=n(97324);let l=(0,n(1153).fn)("TabList"),c=(0,r.createContext)("line"),u={line:(0,s.q)("flex border-b space-x-4","border-tremor-border","dark:border-dark-tremor-border"),solid:(0,s.q)("inline-flex p-0.5 rounded-tremor-default space-x-1.5","bg-tremor-background-subtle","dark:bg-dark-tremor-background-subtle")},d=r.forwardRef((e,t)=>{let{color:n,variant:d="line",children:p,className:g}=e,m=(0,a._T)(e,["color","variant","children","className"]);return r.createElement(o.O.List,Object.assign({ref:t,className:(0,s.q)(l("root"),"justify-start overflow-x-clip",u[d],g)},m),r.createElement(c.Provider,{value:d},r.createElement(i.Z.Provider,{value:n},p)))});d.displayName="TabList"},29706:function(e,t,n){"use strict";n.d(t,{Z:function(){return u}});var a=n(5853);n(42698);var r=n(64016);n(8710);var i=n(33232),o=n(97324),s=n(1153),l=n(2265);let c=(0,s.fn)("TabPanel"),u=l.forwardRef((e,t)=>{let{children:n,className:s}=e,u=(0,a._T)(e,["children","className"]),{selectedValue:d}=(0,l.useContext)(i.Z),p=d===(0,l.useContext)(r.Z);return l.createElement("div",Object.assign({ref:t,className:(0,o.q)(c("root"),"w-full mt-2",p?"":"hidden",s),"aria-selected":p?"true":"false"},u),n)});u.displayName="TabPanel"},77991:function(e,t,n){"use strict";n.d(t,{Z:function(){return d}});var a=n(5853),r=n(31492);n(42698);var i=n(64016);n(8710);var o=n(33232),s=n(97324),l=n(1153),c=n(2265);let u=(0,l.fn)("TabPanels"),d=c.forwardRef((e,t)=>{let{children:n,className:l}=e,d=(0,a._T)(e,["children","className"]);return c.createElement(r.O.Panels,Object.assign({as:"div",ref:t,className:(0,s.q)(u("root"),"w-full",l)},d),e=>{let{selectedIndex:t}=e;return c.createElement(o.Z.Provider,{value:{selectedValue:t}},c.Children.map(n,(e,t)=>c.createElement(i.Z.Provider,{value:t},e)))})});d.displayName="TabPanels"},42698:function(e,t,n){"use strict";n.d(t,{Z:function(){return i}});var a=n(2265),r=n(7084);n(97324);let i=(0,a.createContext)(r.fr.Blue)},64016:function(e,t,n){"use strict";n.d(t,{Z:function(){return a}});let a=(0,n(2265).createContext)(0)},8710:function(e,t,n){"use strict";n.d(t,{Z:function(){return a}});let a=(0,n(2265).createContext)(void 0)},33232:function(e,t,n){"use strict";n.d(t,{Z:function(){return a}});let a=(0,n(2265).createContext)({selectedValue:void 0,handleValueChange:void 0})},93942:function(e,t,n){"use strict";n.d(t,{i:function(){return s}});var a=n(2265),r=n(50506),i=n(13959),o=n(71744);function s(e){return t=>a.createElement(i.ZP,{theme:{token:{motion:!1,zIndexPopupBase:0}}},a.createElement(e,Object.assign({},t)))}t.Z=(e,t,n,i)=>s(s=>{let{prefixCls:l,style:c}=s,u=a.useRef(null),[d,p]=a.useState(0),[g,m]=a.useState(0),[b,f]=(0,r.Z)(!1,{value:s.open}),{getPrefixCls:E}=a.useContext(o.E_),h=E(t||"select",l);a.useEffect(()=>{if(f(!0),"undefined"!=typeof ResizeObserver){let e=new ResizeObserver(e=>{let t=e[0].target;p(t.offsetHeight+8),m(t.offsetWidth)}),t=setInterval(()=>{var a;let r=n?".".concat(n(h)):".".concat(h,"-dropdown"),i=null===(a=u.current)||void 0===a?void 0:a.querySelector(r);i&&(clearInterval(t),e.observe(i))},10);return()=>{clearInterval(t),e.disconnect()}}},[]);let S=Object.assign(Object.assign({},s),{style:Object.assign(Object.assign({},c),{margin:0}),open:b,visible:b,getPopupContainer:()=>u.current});return i&&(S=i(S)),a.createElement("div",{ref:u,style:{paddingBottom:d,position:"relative",minWidth:g}},a.createElement(e,Object.assign({},S)))})},51369:function(e,t,n){"use strict";let a;n.d(t,{Z:function(){return eY}});var r=n(83145),i=n(2265),o=n(18404),s=n(71744),l=n(13959),c=n(8900),u=n(39725),d=n(54537),p=n(55726),g=n(36760),m=n.n(g),b=n(62236),f=n(68710),E=n(55274),h=n(29961),S=n(69819),y=n(73002),T=n(51248),A=e=>{let{type:t,children:n,prefixCls:a,buttonProps:r,close:o,autoFocus:s,emitEvent:l,isSilent:c,quitOnNullishReturnValue:u,actionFn:d}=e,p=i.useRef(!1),g=i.useRef(null),[m,b]=(0,S.Z)(!1),f=function(){null==o||o.apply(void 0,arguments)};i.useEffect(()=>{let e=null;return s&&(e=setTimeout(()=>{var e;null===(e=g.current)||void 0===e||e.focus()})),()=>{e&&clearTimeout(e)}},[]);let E=e=>{e&&e.then&&(b(!0),e.then(function(){b(!1,!0),f.apply(void 0,arguments),p.current=!1},e=>{if(b(!1,!0),p.current=!1,null==c||!c())return Promise.reject(e)}))};return i.createElement(y.ZP,Object.assign({},(0,T.nx)(t),{onClick:e=>{let t;if(!p.current){if(p.current=!0,!d){f();return}if(l){var n;if(t=d(e),u&&!((n=t)&&n.then)){p.current=!1,f(e);return}}else if(d.length)t=d(o),p.current=!1;else if(!(t=d())){f();return}E(t)}},loading:m,prefixCls:a},r,{ref:g}),n)};let R=i.createContext({}),{Provider:I}=R;var N=()=>{let{autoFocusButton:e,cancelButtonProps:t,cancelTextLocale:n,isSilent:a,mergedOkCancel:r,rootPrefixCls:o,close:s,onCancel:l,onConfirm:c}=(0,i.useContext)(R);return r?i.createElement(A,{isSilent:a,actionFn:l,close:function(){null==s||s.apply(void 0,arguments),null==c||c(!1)},autoFocus:"cancel"===e,buttonProps:t,prefixCls:"".concat(o,"-btn")},n):null},_=()=>{let{autoFocusButton:e,close:t,isSilent:n,okButtonProps:a,rootPrefixCls:r,okTextLocale:o,okType:s,onConfirm:l,onOk:c}=(0,i.useContext)(R);return i.createElement(A,{isSilent:n,type:s||"primary",actionFn:c,close:function(){null==t||t.apply(void 0,arguments),null==l||l(!0)},autoFocus:"ok"===e,buttonProps:a,prefixCls:"".concat(r,"-btn")},o)},v=n(49638),w=n(1119),k=n(26365),C=n(28036),O=i.createContext({}),x=n(31686),L=n(2161),D=n(92491),P=n(95814),M=n(18242);function F(e,t,n){var a=t;return!a&&n&&(a="".concat(e,"-").concat(n)),a}function U(e,t){var n=e["page".concat(t?"Y":"X","Offset")],a="scroll".concat(t?"Top":"Left");if("number"!=typeof n){var r=e.document;"number"!=typeof(n=r.documentElement[a])&&(n=r.body[a])}return n}var B=n(47970),G=n(28791),$=i.memo(function(e){return e.children},function(e,t){return!t.shouldUpdate}),H={width:0,height:0,overflow:"hidden",outline:"none"},z=i.forwardRef(function(e,t){var n,a,r,o=e.prefixCls,s=e.className,l=e.style,c=e.title,u=e.ariaId,d=e.footer,p=e.closable,g=e.closeIcon,b=e.onClose,f=e.children,E=e.bodyStyle,h=e.bodyProps,S=e.modalRender,y=e.onMouseDown,T=e.onMouseUp,A=e.holderRef,R=e.visible,I=e.forceRender,N=e.width,_=e.height,v=e.classNames,k=e.styles,C=i.useContext(O).panel,L=(0,G.x1)(A,C),D=(0,i.useRef)(),P=(0,i.useRef)();i.useImperativeHandle(t,function(){return{focus:function(){var e;null===(e=D.current)||void 0===e||e.focus()},changeActive:function(e){var t=document.activeElement;e&&t===P.current?D.current.focus():e||t!==D.current||P.current.focus()}}});var M={};void 0!==N&&(M.width=N),void 0!==_&&(M.height=_),d&&(n=i.createElement("div",{className:m()("".concat(o,"-footer"),null==v?void 0:v.footer),style:(0,x.Z)({},null==k?void 0:k.footer)},d)),c&&(a=i.createElement("div",{className:m()("".concat(o,"-header"),null==v?void 0:v.header),style:(0,x.Z)({},null==k?void 0:k.header)},i.createElement("div",{className:"".concat(o,"-title"),id:u},c))),p&&(r=i.createElement("button",{type:"button",onClick:b,"aria-label":"Close",className:"".concat(o,"-close")},g||i.createElement("span",{className:"".concat(o,"-close-x")})));var F=i.createElement("div",{className:m()("".concat(o,"-content"),null==v?void 0:v.content),style:null==k?void 0:k.content},r,a,i.createElement("div",(0,w.Z)({className:m()("".concat(o,"-body"),null==v?void 0:v.body),style:(0,x.Z)((0,x.Z)({},E),null==k?void 0:k.body)},h),f),n);return i.createElement("div",{key:"dialog-element",role:"dialog","aria-labelledby":c?u:null,"aria-modal":"true",ref:L,style:(0,x.Z)((0,x.Z)({},l),M),className:m()(o,s),onMouseDown:y,onMouseUp:T},i.createElement("div",{tabIndex:0,ref:D,style:H,"aria-hidden":"true"}),i.createElement($,{shouldUpdate:R||I},S?S(F):F),i.createElement("div",{tabIndex:0,ref:P,style:H,"aria-hidden":"true"}))}),j=i.forwardRef(function(e,t){var n=e.prefixCls,a=e.title,r=e.style,o=e.className,s=e.visible,l=e.forceRender,c=e.destroyOnClose,u=e.motionName,d=e.ariaId,p=e.onVisibleChanged,g=e.mousePosition,b=(0,i.useRef)(),f=i.useState(),E=(0,k.Z)(f,2),h=E[0],S=E[1],y={};function T(){var e,t,n,a,r,i=(n={left:(t=(e=b.current).getBoundingClientRect()).left,top:t.top},r=(a=e.ownerDocument).defaultView||a.parentWindow,n.left+=U(r),n.top+=U(r,!0),n);S(g?"".concat(g.x-i.left,"px ").concat(g.y-i.top,"px"):"")}return h&&(y.transformOrigin=h),i.createElement(B.ZP,{visible:s,onVisibleChanged:p,onAppearPrepare:T,onEnterPrepare:T,forceRender:l,motionName:u,removeOnLeave:c,ref:b},function(s,l){var c=s.className,u=s.style;return i.createElement(z,(0,w.Z)({},e,{ref:t,title:a,ariaId:d,prefixCls:n,holderRef:l,style:(0,x.Z)((0,x.Z)((0,x.Z)({},u),r),y),className:m()(o,c)}))})});function V(e){var t=e.prefixCls,n=e.style,a=e.visible,r=e.maskProps,o=e.motionName,s=e.className;return i.createElement(B.ZP,{key:"mask",visible:a,motionName:o,leavedClassName:"".concat(t,"-mask-hidden")},function(e,a){var o=e.className,l=e.style;return i.createElement("div",(0,w.Z)({ref:a,style:(0,x.Z)((0,x.Z)({},l),n),className:m()("".concat(t,"-mask"),o,s)},r))})}function W(e){var t=e.prefixCls,n=void 0===t?"rc-dialog":t,a=e.zIndex,r=e.visible,o=void 0!==r&&r,s=e.keyboard,l=void 0===s||s,c=e.focusTriggerAfterClose,u=void 0===c||c,d=e.wrapStyle,p=e.wrapClassName,g=e.wrapProps,b=e.onClose,f=e.afterOpenChange,E=e.afterClose,h=e.transitionName,S=e.animation,y=e.closable,T=e.mask,A=void 0===T||T,R=e.maskTransitionName,I=e.maskAnimation,N=e.maskClosable,_=e.maskStyle,v=e.maskProps,C=e.rootClassName,O=e.classNames,U=e.styles,B=(0,i.useRef)(),G=(0,i.useRef)(),$=(0,i.useRef)(),H=i.useState(o),z=(0,k.Z)(H,2),W=z[0],q=z[1],Y=(0,D.Z)();function K(e){null==b||b(e)}var Z=(0,i.useRef)(!1),X=(0,i.useRef)(),Q=null;return(void 0===N||N)&&(Q=function(e){Z.current?Z.current=!1:G.current===e.target&&K(e)}),(0,i.useEffect)(function(){o&&(q(!0),(0,L.Z)(G.current,document.activeElement)||(B.current=document.activeElement))},[o]),(0,i.useEffect)(function(){return function(){clearTimeout(X.current)}},[]),i.createElement("div",(0,w.Z)({className:m()("".concat(n,"-root"),C)},(0,M.Z)(e,{data:!0})),i.createElement(V,{prefixCls:n,visible:A&&o,motionName:F(n,R,I),style:(0,x.Z)((0,x.Z)({zIndex:a},_),null==U?void 0:U.mask),maskProps:v,className:null==O?void 0:O.mask}),i.createElement("div",(0,w.Z)({tabIndex:-1,onKeyDown:function(e){if(l&&e.keyCode===P.Z.ESC){e.stopPropagation(),K(e);return}o&&e.keyCode===P.Z.TAB&&$.current.changeActive(!e.shiftKey)},className:m()("".concat(n,"-wrap"),p,null==O?void 0:O.wrapper),ref:G,onClick:Q,style:(0,x.Z)((0,x.Z)((0,x.Z)({zIndex:a},d),null==U?void 0:U.wrapper),{},{display:W?null:"none"})},g),i.createElement(j,(0,w.Z)({},e,{onMouseDown:function(){clearTimeout(X.current),Z.current=!0},onMouseUp:function(){X.current=setTimeout(function(){Z.current=!1})},ref:$,closable:void 0===y||y,ariaId:Y,prefixCls:n,visible:o&&W,onClose:K,onVisibleChanged:function(e){if(e)!function(){if(!(0,L.Z)(G.current,document.activeElement)){var e;null===(e=$.current)||void 0===e||e.focus()}}();else{if(q(!1),A&&B.current&&u){try{B.current.focus({preventScroll:!0})}catch(e){}B.current=null}W&&(null==E||E())}null==f||f(e)},motionName:F(n,h,S)}))))}j.displayName="Content",n(32559);var q=function(e){var t=e.visible,n=e.getContainer,a=e.forceRender,r=e.destroyOnClose,o=void 0!==r&&r,s=e.afterClose,l=e.panelRef,c=i.useState(t),u=(0,k.Z)(c,2),d=u[0],p=u[1],g=i.useMemo(function(){return{panel:l}},[l]);return(i.useEffect(function(){t&&p(!0)},[t]),a||!o||d)?i.createElement(O.Provider,{value:g},i.createElement(C.Z,{open:t||a||d,autoDestroy:!1,getContainer:n,autoLock:t||d},i.createElement(W,(0,w.Z)({},e,{destroyOnClose:o,afterClose:function(){null==s||s(),p(!1)}})))):null};q.displayName="Dialog";var Y=function(e,t,n){let a=arguments.length>3&&void 0!==arguments[3]?arguments[3]:i.createElement(v.Z,null),r=arguments.length>4&&void 0!==arguments[4]&&arguments[4];if("boolean"==typeof e?!e:void 0===t?!r:!1===t||null===t)return[!1,null];let o="boolean"==typeof t||null==t?a:t;return[!0,n?n(o):o]},K=n(94981),Z=n(95140),X=n(39109),Q=n(65658),J=n(74126);function ee(){}let et=i.createContext({add:ee,remove:ee});var en=n(86586),ea=()=>{let{cancelButtonProps:e,cancelTextLocale:t,onCancel:n}=(0,i.useContext)(R);return i.createElement(y.ZP,Object.assign({onClick:n},e),t)},er=()=>{let{confirmLoading:e,okButtonProps:t,okType:n,okTextLocale:a,onOk:r}=(0,i.useContext)(R);return i.createElement(y.ZP,Object.assign({},(0,T.nx)(n),{loading:e,onClick:r},t),a)},ei=n(92246);function eo(e,t){return i.createElement("span",{className:"".concat(e,"-close-x")},t||i.createElement(v.Z,{className:"".concat(e,"-close-icon")}))}let es=e=>{let t;let{okText:n,okType:a="primary",cancelText:o,confirmLoading:s,onOk:l,onCancel:c,okButtonProps:u,cancelButtonProps:d,footer:p}=e,[g]=(0,E.Z)("Modal",(0,ei.A)()),m={confirmLoading:s,okButtonProps:u,cancelButtonProps:d,okTextLocale:n||(null==g?void 0:g.okText),cancelTextLocale:o||(null==g?void 0:g.cancelText),okType:a,onOk:l,onCancel:c},b=i.useMemo(()=>m,(0,r.Z)(Object.values(m)));return"function"==typeof p||void 0===p?(t=i.createElement(i.Fragment,null,i.createElement(ea,null),i.createElement(er,null)),"function"==typeof p&&(t=p(t,{OkBtn:er,CancelBtn:ea})),t=i.createElement(I,{value:b},t)):t=p,i.createElement(en.n,{disabled:!1},t)};var el=n(12918),ec=n(11699),eu=n(691),ed=n(3104),ep=n(80669),eg=n(352);function em(e){return{position:e,inset:0}}let eb=e=>{let{componentCls:t,antCls:n}=e;return[{["".concat(t,"-root")]:{["".concat(t).concat(n,"-zoom-enter, ").concat(t).concat(n,"-zoom-appear")]:{transform:"none",opacity:0,animationDuration:e.motionDurationSlow,userSelect:"none"},["".concat(t).concat(n,"-zoom-leave ").concat(t,"-content")]:{pointerEvents:"none"},["".concat(t,"-mask")]:Object.assign(Object.assign({},em("fixed")),{zIndex:e.zIndexPopupBase,height:"100%",backgroundColor:e.colorBgMask,pointerEvents:"none",["".concat(t,"-hidden")]:{display:"none"}}),["".concat(t,"-wrap")]:Object.assign(Object.assign({},em("fixed")),{zIndex:e.zIndexPopupBase,overflow:"auto",outline:0,WebkitOverflowScrolling:"touch",["&:has(".concat(t).concat(n,"-zoom-enter), &:has(").concat(t).concat(n,"-zoom-appear)")]:{pointerEvents:"none"}})}},{["".concat(t,"-root")]:(0,ec.J$)(e)}]},ef=e=>{let{componentCls:t}=e;return[{["".concat(t,"-root")]:{["".concat(t,"-wrap-rtl")]:{direction:"rtl"},["".concat(t,"-centered")]:{textAlign:"center","&::before":{display:"inline-block",width:0,height:"100%",verticalAlign:"middle",content:'""'},[t]:{top:0,display:"inline-block",paddingBottom:0,textAlign:"start",verticalAlign:"middle"}},["@media (max-width: ".concat(e.screenSMMax,"px)")]:{[t]:{maxWidth:"calc(100vw - 16px)",margin:"".concat((0,eg.bf)(e.marginXS)," auto")},["".concat(t,"-centered")]:{[t]:{flex:1}}}}},{[t]:Object.assign(Object.assign({},(0,el.Wf)(e)),{pointerEvents:"none",position:"relative",top:100,width:"auto",maxWidth:"calc(100vw - ".concat((0,eg.bf)(e.calc(e.margin).mul(2).equal()),")"),margin:"0 auto",paddingBottom:e.paddingLG,["".concat(t,"-title")]:{margin:0,color:e.titleColor,fontWeight:e.fontWeightStrong,fontSize:e.titleFontSize,lineHeight:e.titleLineHeight,wordWrap:"break-word"},["".concat(t,"-content")]:{position:"relative",backgroundColor:e.contentBg,backgroundClip:"padding-box",border:0,borderRadius:e.borderRadiusLG,boxShadow:e.boxShadow,pointerEvents:"auto",padding:e.contentPadding},["".concat(t,"-close")]:Object.assign({position:"absolute",top:e.calc(e.modalHeaderHeight).sub(e.modalCloseBtnSize).div(2).equal(),insetInlineEnd:e.calc(e.modalHeaderHeight).sub(e.modalCloseBtnSize).div(2).equal(),zIndex:e.calc(e.zIndexPopupBase).add(10).equal(),padding:0,color:e.modalCloseIconColor,fontWeight:e.fontWeightStrong,lineHeight:1,textDecoration:"none",background:"transparent",borderRadius:e.borderRadiusSM,width:e.modalCloseBtnSize,height:e.modalCloseBtnSize,border:0,outline:0,cursor:"pointer",transition:"color ".concat(e.motionDurationMid,", background-color ").concat(e.motionDurationMid),"&-x":{display:"flex",fontSize:e.fontSizeLG,fontStyle:"normal",lineHeight:"".concat((0,eg.bf)(e.modalCloseBtnSize)),justifyContent:"center",textTransform:"none",textRendering:"auto"},"&:hover":{color:e.modalIconHoverColor,backgroundColor:e.closeBtnHoverBg,textDecoration:"none"},"&:active":{backgroundColor:e.closeBtnActiveBg}},(0,el.Qy)(e)),["".concat(t,"-header")]:{color:e.colorText,background:e.headerBg,borderRadius:"".concat((0,eg.bf)(e.borderRadiusLG)," ").concat((0,eg.bf)(e.borderRadiusLG)," 0 0"),marginBottom:e.headerMarginBottom,padding:e.headerPadding,borderBottom:e.headerBorderBottom},["".concat(t,"-body")]:{fontSize:e.fontSize,lineHeight:e.lineHeight,wordWrap:"break-word",padding:e.bodyPadding},["".concat(t,"-footer")]:{textAlign:"end",background:e.footerBg,marginTop:e.footerMarginTop,padding:e.footerPadding,borderTop:e.footerBorderTop,borderRadius:e.footerBorderRadius,["> ".concat(e.antCls,"-btn + ").concat(e.antCls,"-btn")]:{marginInlineStart:e.marginXS}},["".concat(t,"-open")]:{overflow:"hidden"}})},{["".concat(t,"-pure-panel")]:{top:"auto",padding:0,display:"flex",flexDirection:"column",["".concat(t,"-content,\n ").concat(t,"-body,\n ").concat(t,"-confirm-body-wrapper")]:{display:"flex",flexDirection:"column",flex:"auto"},["".concat(t,"-confirm-body")]:{marginBottom:"auto"}}}]},eE=e=>{let{componentCls:t}=e;return{["".concat(t,"-root")]:{["".concat(t,"-wrap-rtl")]:{direction:"rtl",["".concat(t,"-confirm-body")]:{direction:"rtl"}}}}},eh=e=>{let t=e.padding,n=e.fontSizeHeading5,a=e.lineHeightHeading5;return(0,ed.TS)(e,{modalHeaderHeight:e.calc(e.calc(a).mul(n).equal()).add(e.calc(t).mul(2).equal()).equal(),modalFooterBorderColorSplit:e.colorSplit,modalFooterBorderStyle:e.lineType,modalFooterBorderWidth:e.lineWidth,modalIconHoverColor:e.colorIconHover,modalCloseIconColor:e.colorIcon,modalCloseBtnSize:e.fontHeight,modalConfirmIconSize:e.fontHeight,modalTitleHeight:e.calc(e.titleFontSize).mul(e.titleLineHeight).equal()})},eS=e=>({footerBg:"transparent",headerBg:e.colorBgElevated,titleLineHeight:e.lineHeightHeading5,titleFontSize:e.fontSizeHeading5,contentBg:e.colorBgElevated,titleColor:e.colorTextHeading,closeBtnHoverBg:e.wireframe?"transparent":e.colorFillContent,closeBtnActiveBg:e.wireframe?"transparent":e.colorFillContentHover,contentPadding:e.wireframe?0:"".concat((0,eg.bf)(e.paddingMD)," ").concat((0,eg.bf)(e.paddingContentHorizontalLG)),headerPadding:e.wireframe?"".concat((0,eg.bf)(e.padding)," ").concat((0,eg.bf)(e.paddingLG)):0,headerBorderBottom:e.wireframe?"".concat((0,eg.bf)(e.lineWidth)," ").concat(e.lineType," ").concat(e.colorSplit):"none",headerMarginBottom:e.wireframe?0:e.marginXS,bodyPadding:e.wireframe?e.paddingLG:0,footerPadding:e.wireframe?"".concat((0,eg.bf)(e.paddingXS)," ").concat((0,eg.bf)(e.padding)):0,footerBorderTop:e.wireframe?"".concat((0,eg.bf)(e.lineWidth)," ").concat(e.lineType," ").concat(e.colorSplit):"none",footerBorderRadius:e.wireframe?"0 0 ".concat((0,eg.bf)(e.borderRadiusLG)," ").concat((0,eg.bf)(e.borderRadiusLG)):0,footerMarginTop:e.wireframe?0:e.marginSM,confirmBodyPadding:e.wireframe?"".concat((0,eg.bf)(2*e.padding)," ").concat((0,eg.bf)(2*e.padding)," ").concat((0,eg.bf)(e.paddingLG)):0,confirmIconMarginInlineEnd:e.wireframe?e.margin:e.marginSM,confirmBtnsMarginTop:e.wireframe?e.marginLG:e.marginSM});var ey=(0,ep.I$)("Modal",e=>{let t=eh(e);return[ef(t),eE(t),eb(t),(0,eu._y)(t,"zoom")]},eS,{unitless:{titleLineHeight:!0}}),eT=n(64024),eA=function(e,t){var n={};for(var a in e)Object.prototype.hasOwnProperty.call(e,a)&&0>t.indexOf(a)&&(n[a]=e[a]);if(null!=e&&"function"==typeof Object.getOwnPropertySymbols)for(var r=0,a=Object.getOwnPropertySymbols(e);rt.indexOf(a[r])&&Object.prototype.propertyIsEnumerable.call(e,a[r])&&(n[a[r]]=e[a[r]]);return n};(0,K.Z)()&&window.document.documentElement&&document.documentElement.addEventListener("click",e=>{a={x:e.pageX,y:e.pageY},setTimeout(()=>{a=null},100)},!0);var eR=e=>{var t;let{getPopupContainer:n,getPrefixCls:r,direction:o,modal:l}=i.useContext(s.E_),c=t=>{let{onCancel:n}=e;null==n||n(t)},{prefixCls:u,className:d,rootClassName:p,open:g,wrapClassName:E,centered:h,getContainer:S,closeIcon:y,closable:T,focusTriggerAfterClose:A=!0,style:R,visible:I,width:N=520,footer:_,classNames:w,styles:k}=e,C=eA(e,["prefixCls","className","rootClassName","open","wrapClassName","centered","getContainer","closeIcon","closable","focusTriggerAfterClose","style","visible","width","footer","classNames","styles"]),O=r("modal",u),x=r(),L=(0,eT.Z)(O),[D,P,M]=ey(O,L),F=m()(E,{["".concat(O,"-centered")]:!!h,["".concat(O,"-wrap-rtl")]:"rtl"===o}),U=null!==_&&i.createElement(es,Object.assign({},e,{onOk:t=>{let{onOk:n}=e;null==n||n(t)},onCancel:c})),[B,G]=Y(T,y,e=>eo(O,e),i.createElement(v.Z,{className:"".concat(O,"-close-icon")}),!0),$=function(e){let t=i.useContext(et),n=i.useRef();return(0,J.zX)(a=>{if(a){let r=e?a.querySelector(e):a;t.add(r),n.current=r}else t.remove(n.current)})}(".".concat(O,"-content")),[H,z]=(0,b.Cn)("Modal",C.zIndex);return D(i.createElement(Q.BR,null,i.createElement(X.Ux,{status:!0,override:!0},i.createElement(Z.Z.Provider,{value:z},i.createElement(q,Object.assign({width:N},C,{zIndex:H,getContainer:void 0===S?n:S,prefixCls:O,rootClassName:m()(P,p,M,L),footer:U,visible:null!=g?g:I,mousePosition:null!==(t=C.mousePosition)&&void 0!==t?t:a,onClose:c,closable:B,closeIcon:G,focusTriggerAfterClose:A,transitionName:(0,f.m)(x,"zoom",e.transitionName),maskTransitionName:(0,f.m)(x,"fade",e.maskTransitionName),className:m()(P,d,null==l?void 0:l.className),style:Object.assign(Object.assign({},null==l?void 0:l.style),R),classNames:Object.assign(Object.assign({wrapper:F},null==l?void 0:l.classNames),w),styles:Object.assign(Object.assign({},null==l?void 0:l.styles),k),panelRef:$}))))))};let eI=e=>{let{componentCls:t,titleFontSize:n,titleLineHeight:a,modalConfirmIconSize:r,fontSize:i,lineHeight:o,modalTitleHeight:s,fontHeight:l,confirmBodyPadding:c}=e,u="".concat(t,"-confirm");return{[u]:{"&-rtl":{direction:"rtl"},["".concat(e.antCls,"-modal-header")]:{display:"none"},["".concat(u,"-body-wrapper")]:Object.assign({},(0,el.dF)()),["&".concat(t," ").concat(t,"-body")]:{padding:c},["".concat(u,"-body")]:{display:"flex",flexWrap:"nowrap",alignItems:"start",["> ".concat(e.iconCls)]:{flex:"none",fontSize:r,marginInlineEnd:e.confirmIconMarginInlineEnd,marginTop:e.calc(e.calc(l).sub(r).equal()).div(2).equal()},["&-has-title > ".concat(e.iconCls)]:{marginTop:e.calc(e.calc(s).sub(r).equal()).div(2).equal()}},["".concat(u,"-paragraph")]:{display:"flex",flexDirection:"column",flex:"auto",rowGap:e.marginXS,maxWidth:"calc(100% - ".concat((0,eg.bf)(e.calc(e.modalConfirmIconSize).add(e.marginSM).equal()),")")},["".concat(u,"-title")]:{color:e.colorTextHeading,fontWeight:e.fontWeightStrong,fontSize:n,lineHeight:a},["".concat(u,"-content")]:{color:e.colorText,fontSize:i,lineHeight:o},["".concat(u,"-btns")]:{textAlign:"end",marginTop:e.confirmBtnsMarginTop,["".concat(e.antCls,"-btn + ").concat(e.antCls,"-btn")]:{marginBottom:0,marginInlineStart:e.marginXS}}},["".concat(u,"-error ").concat(u,"-body > ").concat(e.iconCls)]:{color:e.colorError},["".concat(u,"-warning ").concat(u,"-body > ").concat(e.iconCls,",\n ").concat(u,"-confirm ").concat(u,"-body > ").concat(e.iconCls)]:{color:e.colorWarning},["".concat(u,"-info ").concat(u,"-body > ").concat(e.iconCls)]:{color:e.colorInfo},["".concat(u,"-success ").concat(u,"-body > ").concat(e.iconCls)]:{color:e.colorSuccess}}};var eN=(0,ep.bk)(["Modal","confirm"],e=>[eI(eh(e))],eS,{order:-1e3}),e_=function(e,t){var n={};for(var a in e)Object.prototype.hasOwnProperty.call(e,a)&&0>t.indexOf(a)&&(n[a]=e[a]);if(null!=e&&"function"==typeof Object.getOwnPropertySymbols)for(var r=0,a=Object.getOwnPropertySymbols(e);rt.indexOf(a[r])&&Object.prototype.propertyIsEnumerable.call(e,a[r])&&(n[a[r]]=e[a[r]]);return n};function ev(e){let{prefixCls:t,icon:n,okText:a,cancelText:o,confirmPrefixCls:s,type:l,okCancel:g,footer:b,locale:f}=e,h=e_(e,["prefixCls","icon","okText","cancelText","confirmPrefixCls","type","okCancel","footer","locale"]),S=n;if(!n&&null!==n)switch(l){case"info":S=i.createElement(p.Z,null);break;case"success":S=i.createElement(c.Z,null);break;case"error":S=i.createElement(u.Z,null);break;default:S=i.createElement(d.Z,null)}let y=null!=g?g:"confirm"===l,T=null!==e.autoFocusButton&&(e.autoFocusButton||"ok"),[A]=(0,E.Z)("Modal"),R=f||A,v=a||(y?null==R?void 0:R.okText:null==R?void 0:R.justOkText),w=Object.assign({autoFocusButton:T,cancelTextLocale:o||(null==R?void 0:R.cancelText),okTextLocale:v,mergedOkCancel:y},h),k=i.useMemo(()=>w,(0,r.Z)(Object.values(w))),C=i.createElement(i.Fragment,null,i.createElement(N,null),i.createElement(_,null)),O=void 0!==e.title&&null!==e.title,x="".concat(s,"-body");return i.createElement("div",{className:"".concat(s,"-body-wrapper")},i.createElement("div",{className:m()(x,{["".concat(x,"-has-title")]:O})},S,i.createElement("div",{className:"".concat(s,"-paragraph")},O&&i.createElement("span",{className:"".concat(s,"-title")},e.title),i.createElement("div",{className:"".concat(s,"-content")},e.content))),void 0===b||"function"==typeof b?i.createElement(I,{value:k},i.createElement("div",{className:"".concat(s,"-btns")},"function"==typeof b?b(C,{OkBtn:_,CancelBtn:N}):C)):b,i.createElement(eN,{prefixCls:t}))}let ew=e=>{let{close:t,zIndex:n,afterClose:a,open:r,keyboard:o,centered:s,getContainer:l,maskStyle:c,direction:u,prefixCls:d,wrapClassName:p,rootPrefixCls:g,bodyStyle:E,closable:S=!1,closeIcon:y,modalRender:T,focusTriggerAfterClose:A,onConfirm:R,styles:I}=e,N="".concat(d,"-confirm"),_=e.width||416,v=e.style||{},w=void 0===e.mask||e.mask,k=void 0!==e.maskClosable&&e.maskClosable,C=m()(N,"".concat(N,"-").concat(e.type),{["".concat(N,"-rtl")]:"rtl"===u},e.className),[,O]=(0,h.ZP)(),x=i.useMemo(()=>void 0!==n?n:O.zIndexPopupBase+b.u6,[n,O]);return i.createElement(eR,{prefixCls:d,className:C,wrapClassName:m()({["".concat(N,"-centered")]:!!e.centered},p),onCancel:()=>{null==t||t({triggerCancel:!0}),null==R||R(!1)},open:r,title:"",footer:null,transitionName:(0,f.m)(g||"","zoom",e.transitionName),maskTransitionName:(0,f.m)(g||"","fade",e.maskTransitionName),mask:w,maskClosable:k,style:v,styles:Object.assign({body:E,mask:c},I),width:_,zIndex:x,afterClose:a,keyboard:o,centered:s,getContainer:l,closable:S,closeIcon:y,modalRender:T,focusTriggerAfterClose:A},i.createElement(ev,Object.assign({},e,{confirmPrefixCls:N})))};var ek=e=>{let{rootPrefixCls:t,iconPrefixCls:n,direction:a,theme:r}=e;return i.createElement(l.ZP,{prefixCls:t,iconPrefixCls:n,direction:a,theme:r},i.createElement(ew,Object.assign({},e)))},eC=[];let eO="",ex=e=>{var t,n;let{prefixCls:a,getContainer:r,direction:o}=e,l=(0,ei.A)(),c=(0,i.useContext)(s.E_),u=eO||c.getPrefixCls(),d=a||"".concat(u,"-modal"),p=r;return!1===p&&(p=void 0),i.createElement(ek,Object.assign({},e,{rootPrefixCls:u,prefixCls:d,iconPrefixCls:c.iconPrefixCls,theme:c.theme,direction:null!=o?o:c.direction,locale:null!==(n=null===(t=c.locale)||void 0===t?void 0:t.Modal)&&void 0!==n?n:l,getContainer:p}))};function eL(e){let t;let n=(0,l.w6)(),a=document.createDocumentFragment(),s=Object.assign(Object.assign({},e),{close:d,open:!0});function c(){for(var t=arguments.length,n=Array(t),i=0;ie&&e.triggerCancel);e.onCancel&&s&&e.onCancel.apply(e,[()=>{}].concat((0,r.Z)(n.slice(1))));for(let e=0;e{let t=n.getPrefixCls(void 0,eO),r=n.getIconPrefixCls(),s=n.getTheme(),c=i.createElement(ex,Object.assign({},e));(0,o.s)(i.createElement(l.ZP,{prefixCls:t,iconPrefixCls:r,theme:s},n.holderRender?n.holderRender(c):c),a)})}function d(){for(var t=arguments.length,n=Array(t),a=0;a{"function"==typeof e.afterClose&&e.afterClose(),c.apply(this,n)}})).visible&&delete s.visible,u(s)}return u(s),eC.push(d),{destroy:d,update:function(e){u(s="function"==typeof e?e(s):Object.assign(Object.assign({},s),e))}}}function eD(e){return Object.assign(Object.assign({},e),{type:"warning"})}function eP(e){return Object.assign(Object.assign({},e),{type:"info"})}function eM(e){return Object.assign(Object.assign({},e),{type:"success"})}function eF(e){return Object.assign(Object.assign({},e),{type:"error"})}function eU(e){return Object.assign(Object.assign({},e),{type:"confirm"})}var eB=n(93942),eG=function(e,t){var n={};for(var a in e)Object.prototype.hasOwnProperty.call(e,a)&&0>t.indexOf(a)&&(n[a]=e[a]);if(null!=e&&"function"==typeof Object.getOwnPropertySymbols)for(var r=0,a=Object.getOwnPropertySymbols(e);rt.indexOf(a[r])&&Object.prototype.propertyIsEnumerable.call(e,a[r])&&(n[a[r]]=e[a[r]]);return n},e$=(0,eB.i)(e=>{let{prefixCls:t,className:n,closeIcon:a,closable:r,type:o,title:l,children:c,footer:u}=e,d=eG(e,["prefixCls","className","closeIcon","closable","type","title","children","footer"]),{getPrefixCls:p}=i.useContext(s.E_),g=p(),b=t||p("modal"),f=(0,eT.Z)(g),[E,h,S]=ey(b,f),y="".concat(b,"-confirm"),T={};return T=o?{closable:null!=r&&r,title:"",footer:"",children:i.createElement(ev,Object.assign({},e,{prefixCls:b,confirmPrefixCls:y,rootPrefixCls:g,content:c}))}:{closable:null==r||r,title:l,footer:null!==u&&i.createElement(es,Object.assign({},e)),children:c},E(i.createElement(z,Object.assign({prefixCls:b,className:m()(h,"".concat(b,"-pure-panel"),o&&y,o&&"".concat(y,"-").concat(o),n,S,f)},d,{closeIcon:eo(b,a),closable:r},T)))}),eH=n(13823),ez=function(e,t){var n={};for(var a in e)Object.prototype.hasOwnProperty.call(e,a)&&0>t.indexOf(a)&&(n[a]=e[a]);if(null!=e&&"function"==typeof Object.getOwnPropertySymbols)for(var r=0,a=Object.getOwnPropertySymbols(e);rt.indexOf(a[r])&&Object.prototype.propertyIsEnumerable.call(e,a[r])&&(n[a[r]]=e[a[r]]);return n},ej=i.forwardRef((e,t)=>{var n,{afterClose:a,config:o}=e,l=ez(e,["afterClose","config"]);let[c,u]=i.useState(!0),[d,p]=i.useState(o),{direction:g,getPrefixCls:m}=i.useContext(s.E_),b=m("modal"),f=m(),h=function(){u(!1);for(var e=arguments.length,t=Array(e),n=0;ne&&e.triggerCancel);d.onCancel&&a&&d.onCancel.apply(d,[()=>{}].concat((0,r.Z)(t.slice(1))))};i.useImperativeHandle(t,()=>({destroy:h,update:e=>{p(t=>Object.assign(Object.assign({},t),e))}}));let S=null!==(n=d.okCancel)&&void 0!==n?n:"confirm"===d.type,[y]=(0,E.Z)("Modal",eH.Z.Modal);return i.createElement(ek,Object.assign({prefixCls:b,rootPrefixCls:f},d,{close:h,open:c,afterClose:()=>{var e;a(),null===(e=d.afterClose)||void 0===e||e.call(d)},okText:d.okText||(S?null==y?void 0:y.okText:null==y?void 0:y.justOkText),direction:d.direction||g,cancelText:d.cancelText||(null==y?void 0:y.cancelText)},l))});let eV=0,eW=i.memo(i.forwardRef((e,t)=>{let[n,a]=function(){let[e,t]=i.useState([]);return[e,i.useCallback(e=>(t(t=>[].concat((0,r.Z)(t),[e])),()=>{t(t=>t.filter(t=>t!==e))}),[])]}();return i.useImperativeHandle(t,()=>({patchElement:a}),[]),i.createElement(i.Fragment,null,n)}));function eq(e){return eL(eD(e))}eR.useModal=function(){let e=i.useRef(null),[t,n]=i.useState([]);i.useEffect(()=>{t.length&&((0,r.Z)(t).forEach(e=>{e()}),n([]))},[t]);let a=i.useCallback(t=>function(a){var o;let s,l;eV+=1;let c=i.createRef(),u=new Promise(e=>{s=e}),d=!1,p=i.createElement(ej,{key:"modal-".concat(eV),config:t(a),ref:c,afterClose:()=>{null==l||l()},isSilent:()=>d,onConfirm:e=>{s(e)}});return(l=null===(o=e.current)||void 0===o?void 0:o.patchElement(p))&&eC.push(l),{destroy:()=>{function e(){var e;null===(e=c.current)||void 0===e||e.destroy()}c.current?e():n(t=>[].concat((0,r.Z)(t),[e]))},update:e=>{function t(){var t;null===(t=c.current)||void 0===t||t.update(e)}c.current?t():n(e=>[].concat((0,r.Z)(e),[t]))},then:e=>(d=!0,u.then(e))}},[]);return[i.useMemo(()=>({info:a(eP),success:a(eM),error:a(eF),warning:a(eD),confirm:a(eU)}),[]),i.createElement(eW,{key:"modal-holder",ref:e})]},eR.info=function(e){return eL(eP(e))},eR.success=function(e){return eL(eM(e))},eR.error=function(e){return eL(eF(e))},eR.warning=eq,eR.warn=eq,eR.confirm=function(e){return eL(eU(e))},eR.destroyAll=function(){for(;eC.length;){let e=eC.pop();e&&e()}},eR.config=function(e){let{rootPrefixCls:t}=e;eO=t},eR._InternalPanelDoNotUseOrYouWillBeFired=e$;var eY=eR},11699:function(e,t,n){"use strict";n.d(t,{J$:function(){return s}});var a=n(352),r=n(37133);let i=new a.E4("antFadeIn",{"0%":{opacity:0},"100%":{opacity:1}}),o=new a.E4("antFadeOut",{"0%":{opacity:1},"100%":{opacity:0}}),s=function(e){let t=arguments.length>1&&void 0!==arguments[1]&&arguments[1],{antCls:n}=e,a="".concat(n,"-fade"),s=t?"&":"";return[(0,r.R)(a,i,o,e.motionDurationMid,t),{["\n ".concat(s).concat(a,"-enter,\n ").concat(s).concat(a,"-appear\n ")]:{opacity:0,animationTimingFunction:"linear"},["".concat(s).concat(a,"-leave")]:{animationTimingFunction:"linear"}}]}},26035:function(e){"use strict";e.exports=function(e,n){for(var a,r,i,o=e||"",s=n||"div",l={},c=0;c4&&m.slice(0,4)===o&&s.test(t)&&("-"===t.charAt(4)?b=o+(n=t.slice(5).replace(l,d)).charAt(0).toUpperCase()+n.slice(1):(g=(p=t).slice(4),t=l.test(g)?p:("-"!==(g=g.replace(c,u)).charAt(0)&&(g="-"+g),o+g)),f=r),new f(b,t))};var s=/^data[-\w.:]+$/i,l=/-[a-z]/g,c=/[A-Z]/g;function u(e){return"-"+e.toLowerCase()}function d(e){return e.charAt(1).toUpperCase()}},30466:function(e,t,n){"use strict";var a=n(82855),r=n(64541),i=n(80808),o=n(44987),s=n(72731),l=n(98946);e.exports=a([i,r,o,s,l])},72731:function(e,t,n){"use strict";var a=n(20321),r=n(41757),i=a.booleanish,o=a.number,s=a.spaceSeparated;e.exports=r({transform:function(e,t){return"role"===t?t:"aria-"+t.slice(4).toLowerCase()},properties:{ariaActiveDescendant:null,ariaAtomic:i,ariaAutoComplete:null,ariaBusy:i,ariaChecked:i,ariaColCount:o,ariaColIndex:o,ariaColSpan:o,ariaControls:s,ariaCurrent:null,ariaDescribedBy:s,ariaDetails:null,ariaDisabled:i,ariaDropEffect:s,ariaErrorMessage:null,ariaExpanded:i,ariaFlowTo:s,ariaGrabbed:i,ariaHasPopup:null,ariaHidden:i,ariaInvalid:null,ariaKeyShortcuts:null,ariaLabel:null,ariaLabelledBy:s,ariaLevel:o,ariaLive:null,ariaModal:i,ariaMultiLine:i,ariaMultiSelectable:i,ariaOrientation:null,ariaOwns:s,ariaPlaceholder:null,ariaPosInSet:o,ariaPressed:i,ariaReadOnly:i,ariaRelevant:null,ariaRequired:i,ariaRoleDescription:s,ariaRowCount:o,ariaRowIndex:o,ariaRowSpan:o,ariaSelected:i,ariaSetSize:o,ariaSort:null,ariaValueMax:o,ariaValueMin:o,ariaValueNow:o,ariaValueText:null,role:null}})},98946:function(e,t,n){"use strict";var a=n(20321),r=n(41757),i=n(53296),o=a.boolean,s=a.overloadedBoolean,l=a.booleanish,c=a.number,u=a.spaceSeparated,d=a.commaSeparated;e.exports=r({space:"html",attributes:{acceptcharset:"accept-charset",classname:"class",htmlfor:"for",httpequiv:"http-equiv"},transform:i,mustUseProperty:["checked","multiple","muted","selected"],properties:{abbr:null,accept:d,acceptCharset:u,accessKey:u,action:null,allow:null,allowFullScreen:o,allowPaymentRequest:o,allowUserMedia:o,alt:null,as:null,async:o,autoCapitalize:null,autoComplete:u,autoFocus:o,autoPlay:o,capture:o,charSet:null,checked:o,cite:null,className:u,cols:c,colSpan:null,content:null,contentEditable:l,controls:o,controlsList:u,coords:c|d,crossOrigin:null,data:null,dateTime:null,decoding:null,default:o,defer:o,dir:null,dirName:null,disabled:o,download:s,draggable:l,encType:null,enterKeyHint:null,form:null,formAction:null,formEncType:null,formMethod:null,formNoValidate:o,formTarget:null,headers:u,height:c,hidden:o,high:c,href:null,hrefLang:null,htmlFor:u,httpEquiv:u,id:null,imageSizes:null,imageSrcSet:d,inputMode:null,integrity:null,is:null,isMap:o,itemId:null,itemProp:u,itemRef:u,itemScope:o,itemType:u,kind:null,label:null,lang:null,language:null,list:null,loading:null,loop:o,low:c,manifest:null,max:null,maxLength:c,media:null,method:null,min:null,minLength:c,multiple:o,muted:o,name:null,nonce:null,noModule:o,noValidate:o,onAbort:null,onAfterPrint:null,onAuxClick:null,onBeforePrint:null,onBeforeUnload:null,onBlur:null,onCancel:null,onCanPlay:null,onCanPlayThrough:null,onChange:null,onClick:null,onClose:null,onContextMenu:null,onCopy:null,onCueChange:null,onCut:null,onDblClick:null,onDrag:null,onDragEnd:null,onDragEnter:null,onDragExit:null,onDragLeave:null,onDragOver:null,onDragStart:null,onDrop:null,onDurationChange:null,onEmptied:null,onEnded:null,onError:null,onFocus:null,onFormData:null,onHashChange:null,onInput:null,onInvalid:null,onKeyDown:null,onKeyPress:null,onKeyUp:null,onLanguageChange:null,onLoad:null,onLoadedData:null,onLoadedMetadata:null,onLoadEnd:null,onLoadStart:null,onMessage:null,onMessageError:null,onMouseDown:null,onMouseEnter:null,onMouseLeave:null,onMouseMove:null,onMouseOut:null,onMouseOver:null,onMouseUp:null,onOffline:null,onOnline:null,onPageHide:null,onPageShow:null,onPaste:null,onPause:null,onPlay:null,onPlaying:null,onPopState:null,onProgress:null,onRateChange:null,onRejectionHandled:null,onReset:null,onResize:null,onScroll:null,onSecurityPolicyViolation:null,onSeeked:null,onSeeking:null,onSelect:null,onSlotChange:null,onStalled:null,onStorage:null,onSubmit:null,onSuspend:null,onTimeUpdate:null,onToggle:null,onUnhandledRejection:null,onUnload:null,onVolumeChange:null,onWaiting:null,onWheel:null,open:o,optimum:c,pattern:null,ping:u,placeholder:null,playsInline:o,poster:null,preload:null,readOnly:o,referrerPolicy:null,rel:u,required:o,reversed:o,rows:c,rowSpan:c,sandbox:u,scope:null,scoped:o,seamless:o,selected:o,shape:null,size:c,sizes:null,slot:null,span:c,spellCheck:l,src:null,srcDoc:null,srcLang:null,srcSet:d,start:c,step:null,style:null,tabIndex:c,target:null,title:null,translate:null,type:null,typeMustMatch:o,useMap:null,value:l,width:c,wrap:null,align:null,aLink:null,archive:u,axis:null,background:null,bgColor:null,border:c,borderColor:null,bottomMargin:c,cellPadding:null,cellSpacing:null,char:null,charOff:null,classId:null,clear:null,code:null,codeBase:null,codeType:null,color:null,compact:o,declare:o,event:null,face:null,frame:null,frameBorder:null,hSpace:c,leftMargin:c,link:null,longDesc:null,lowSrc:null,marginHeight:c,marginWidth:c,noResize:o,noHref:o,noShade:o,noWrap:o,object:null,profile:null,prompt:null,rev:null,rightMargin:c,rules:null,scheme:null,scrolling:l,standby:null,summary:null,text:null,topMargin:c,valueType:null,version:null,vAlign:null,vLink:null,vSpace:c,allowTransparency:null,autoCorrect:null,autoSave:null,disablePictureInPicture:o,disableRemotePlayback:o,prefix:null,property:null,results:c,security:null,unselectable:null}})},53296:function(e,t,n){"use strict";var a=n(38781);e.exports=function(e,t){return a(e,t.toLowerCase())}},38781:function(e){"use strict";e.exports=function(e,t){return t in e?e[t]:t}},41757:function(e,t,n){"use strict";var a=n(96532),r=n(61723),i=n(51351);e.exports=function(e){var t,n,o=e.space,s=e.mustUseProperty||[],l=e.attributes||{},c=e.properties,u=e.transform,d={},p={};for(t in c)n=new i(t,u(l,t),c[t],o),-1!==s.indexOf(t)&&(n.mustUseProperty=!0),d[t]=n,p[a(t)]=t,p[a(n.attribute)]=t;return new r(d,p,o)}},51351:function(e,t,n){"use strict";var a=n(24192),r=n(20321);e.exports=s,s.prototype=new a,s.prototype.defined=!0;var i=["boolean","booleanish","overloadedBoolean","number","commaSeparated","spaceSeparated","commaOrSpaceSeparated"],o=i.length;function s(e,t,n,s){var l,c,u,d=-1;for(s&&(this.space=s),a.call(this,e,t);++d