From 2d5a50804b436818469e478488ce5db4bcfea945 Mon Sep 17 00:00:00 2001 From: Eric84626 <97266539+Eric84626@users.noreply.github.com> Date: Tue, 9 Dec 2025 03:27:52 +0800 Subject: [PATCH 01/29] fix: Return 403 exception when calling GET responses api (#17629) --- litellm/proxy/auth/auth_checks.py | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/litellm/proxy/auth/auth_checks.py b/litellm/proxy/auth/auth_checks.py index fc79a4d3591..309bd577606 100644 --- a/litellm/proxy/auth/auth_checks.py +++ b/litellm/proxy/auth/auth_checks.py @@ -402,13 +402,14 @@ def _allowed_routes_check(user_route: str, allowed_routes: list) -> bool: - user_route: str - the route the user is trying to call - allowed_routes: List[str|LiteLLMRoutes] - the list of allowed routes for the user. """ + from starlette.routing import compile_path for allowed_route in allowed_routes: - if ( - allowed_route in LiteLLMRoutes.__members__ - and user_route in LiteLLMRoutes[allowed_route].value - ): - return True + if allowed_route in LiteLLMRoutes.__members__: + for template in LiteLLMRoutes[allowed_route].value: + regex, _, _ = compile_path(template) + if regex.match(user_route): + return True elif allowed_route == user_route: return True return False From 958c1901341e0e124db42f01d3401e2c86f8c13e Mon Sep 17 00:00:00 2001 From: Alexsander Hamir Date: Mon, 8 Dec 2025 12:21:26 -0800 Subject: [PATCH 02/29] Fix flanky tests (#17665) * Fix test_delete_polling_removes_from_cache mock setup - Mock async_delete_cache to properly execute the real implementation path - Ensures init_async_client() is called and delete() is invoked on the returned client - Fixes AssertionError: Expected 'delete' to be called once. Called 0 times. * fix: resolve timeout in add_model_tab test by mocking useProviderFields hook - Mock useProviderFields hook to prevent network calls and React Query delays - Use waitFor to properly handle async operations - Test now passes reliably without 10s timeout * fix: add test timeout to prevent CI timeout failure - Add 15 second timeout to 'should display Test Connect and Add Model buttons' test - Test takes ~6 seconds locally, but CI was timing out at default 5 second limit - Ensures test has sufficient time to complete in CI environment * test: quarantine flaky test_oidc_circleci_with_azure Quarantine test that fails with 401 Unauthorized from Azure OAuth. The test is flaky and blocks CI builds. Marked with @pytest.mark.skip until Azure authentication can be fixed or migrated to our own account. --- .../test_secret_manager.py | 5 ++- .../test_response_polling_handler.py | 7 ++++ .../add_model/add_model_tab.test.tsx | 33 ++++++++++++++++--- 3 files changed, 37 insertions(+), 8 deletions(-) diff --git a/tests/litellm_utils_tests/test_secret_manager.py b/tests/litellm_utils_tests/test_secret_manager.py index 7099f6e13d0..da9c9d548a7 100644 --- a/tests/litellm_utils_tests/test_secret_manager.py +++ b/tests/litellm_utils_tests/test_secret_manager.py @@ -133,9 +133,8 @@ def test_oidc_circleci_v2(): print(f"secret_val: {redact_oidc_signature(secret_val)}") -@pytest.mark.skipif( - os.environ.get("CIRCLE_OIDC_TOKEN") is None, - reason="Cannot run without being in CircleCI Runner", +@pytest.mark.skip( + reason="Quarantined: Flaky test - fails with 401 Unauthorized from Azure OAuth. TODO: Switch to our own Azure account or fix authentication" ) def test_oidc_circleci_with_azure(): # TODO: Switch to our own Azure account, currently using ai.moda's account diff --git a/tests/proxy_unit_tests/test_response_polling_handler.py b/tests/proxy_unit_tests/test_response_polling_handler.py index 5d9b83969f7..cb4cd0efe57 100644 --- a/tests/proxy_unit_tests/test_response_polling_handler.py +++ b/tests/proxy_unit_tests/test_response_polling_handler.py @@ -519,6 +519,13 @@ class TestResponsePollingHandler: # init_async_client is a sync method that returns an async client mock_redis.init_async_client = Mock(return_value=mock_async_client) + # Mock async_delete_cache to actually call init_async_client and delete + async def mock_async_delete_cache(key): + client = mock_redis.init_async_client() + await client.delete(key) + + mock_redis.async_delete_cache = mock_async_delete_cache + handler = ResponsePollingHandler(redis_cache=mock_redis) result = await handler.delete_polling("litellm_poll_test") diff --git a/ui/litellm-dashboard/src/components/add_model/add_model_tab.test.tsx b/ui/litellm-dashboard/src/components/add_model/add_model_tab.test.tsx index 5f165bb7f85..0c353621654 100644 --- a/ui/litellm-dashboard/src/components/add_model/add_model_tab.test.tsx +++ b/ui/litellm-dashboard/src/components/add_model/add_model_tab.test.tsx @@ -1,5 +1,5 @@ import { QueryClient, QueryClientProvider } from "@tanstack/react-query"; -import { render, renderHook, screen } from "@testing-library/react"; +import { render, renderHook, screen, waitFor } from "@testing-library/react"; import { Form } from "antd"; import type { UploadProps } from "antd/es/upload"; import { describe, expect, it, vi } from "vitest"; @@ -37,6 +37,22 @@ vi.mock("../networking", async () => { }; }); +vi.mock("@/app/(dashboard)/hooks/providers/useProviderFields", () => ({ + useProviderFields: vi.fn().mockReturnValue({ + data: [ + { + provider: "OpenAI", + provider_display_name: "OpenAI", + litellm_provider: "openai", + default_model_placeholder: "gpt-3.5-turbo", + credential_fields: [], + }, + ], + isLoading: false, + error: null, + }), +})); + const createQueryClient = () => new QueryClient({ defaultOptions: { @@ -231,8 +247,15 @@ describe("Add Model Tab", () => { , ); - const testConnectButtons = await screen.findAllByRole("button", { name: "Test Connect" }); - expect(testConnectButtons.length).toBeGreaterThan(0); - expect(await screen.findByRole("button", { name: "Add Model" })).toBeInTheDocument(); - }, 10000); // 10 seconds timeout for complex logic + // Wait for async operations to complete and buttons to appear + await waitFor( + async () => { + const testConnectButtons = await screen.findAllByRole("button", { name: "Test Connect" }); + expect(testConnectButtons.length).toBeGreaterThan(0); + const addModelButton = await screen.findByRole("button", { name: "Add Model" }); + expect(addModelButton).toBeInTheDocument(); + }, + { timeout: 10000 }, + ); + }, 15000); // 15 second timeout to allow waitFor to complete }); From c87874c29e8ed467a4bc81efe0b2a766309b6bf6 Mon Sep 17 00:00:00 2001 From: vasilisazayka Date: Tue, 9 Dec 2025 00:31:06 +0400 Subject: [PATCH 03/29] [New provider] Sap gen ai hub (#16053) * add sap gen ai hub * add async tests * add async and streaming support * add embedding model support * add embedding support * remove unused import * fix structured output * clean-up * remove timeout and add tool support * remove unused code * fix(sap): improve streaming robustness; restore embed URL builder compatibility - sap/embed/transformation: add api_key and litellm_params to get_complete_url to align with core flow and prevent failures - sap/chat/handler: wrap async/sync streaming iterators to safely handle Stop(Async)Iteration and errors - sap/chat/transformation: remove unused imports and dead code * fix(sap): linter fix * fix(sap): made gen_ai_hub optional: import check + OptionalDependencyError with install hint if missing. * test(sap): add chat/stream/async tests and OptionalDependencyError check * Fix tool call handling in SAP GenAI Hub transformation Add sap models to model_prices_and_context_window.json and model_prices_and_context_window_backup.json * fix(sap): delete unnecessary code, linter fix * fix(sap): - refactor chat transformation - add support of list and dict content * fix(sap): - fix tests * fix(sap): - fix lint * Update transformation.py * fix(sap): fix model description and fix after rebase * change(sap): - http calls in chat handler, response transformation and auth handling without sap sdk. * change(sap): switching to v2 (chat handler, chat transformation), code clean up * add deployment discovery and improved crendentials handling * add deployment discovery and improved crendentials handling * change(sap): - fix sync stream * change(sap): - fix sync stream * fix(sap): - fix response format * fix(sap): - switch embedding to v2 and http request - reimplement stream creator - improve request transformation * fix async streaming * fix(sap): linters, transformation models, remove sap dependency test * fix(sap): code clean up * add unit test for sap chat completion * linters fix * move token, rg and base_url to properties * (sap): add embedding unit test Signed-off-by: Vasilisa Parshikova * fix(sap): bypass response format for some models Signed-off-by: Vasilisa Parshikova * fix(sap): fix chat transformation and list of supported params Signed-off-by: Vasilisa Parshikova * fix(sap): fix lint * add sap service key module parameter * fix(sap): remove unused code * fix(sap): remove prices * add service key support * fix(sap): - add message content validations - change get_supported_openai_params in chat transformation * typo in mock * fix(sap): - fix in supported params map * fix(sap): - fix in message content validation * fix(sap): - fix in message content validation * fix(sap): - use litellm client for credentials * fix(sap): - linter fix * fix(sap): - use build in custom_http_client - move credentials handling to transformation * fix(sap): - handle stream_options * fix(sap): - fix tests * fix(sap): - code clean up, linter fix * skip other authentication options when creds are provided * fix local variable --------- Signed-off-by: Vasilisa Parshikova Co-authored-by: Mathis Boerner Co-authored-by: karimmohraz <37623804+karimmohraz@users.noreply.github.com> Co-authored-by: Karim --- litellm/__init__.py | 15 +- .../get_llm_provider_logic.py | 2 + .../get_supported_openai_params.py | 5 + .../litellm_core_utils/streaming_handler.py | 1 - litellm/llms/sap/chat/__init__.py | 1 + litellm/llms/sap/chat/handler.py | 262 +++ litellm/llms/sap/chat/models.py | 112 ++ litellm/llms/sap/chat/transformation.py | 299 +++ litellm/llms/sap/credentials.py | 325 ++++ litellm/llms/sap/embed/transformation.py | 176 ++ litellm/main.py | 46 + litellm/proxy/utils.py | 2 +- litellm/types/utils.py | 1 + litellm/utils.py | 23 + .../llms/sap/chat/test_sap_chat_calls.py | 142 ++ .../llms/sap/embed/test_sap_embedding.py | 1607 +++++++++++++++++ 16 files changed, 3011 insertions(+), 8 deletions(-) create mode 100755 litellm/llms/sap/chat/__init__.py create mode 100755 litellm/llms/sap/chat/handler.py create mode 100644 litellm/llms/sap/chat/models.py create mode 100755 litellm/llms/sap/chat/transformation.py create mode 100644 litellm/llms/sap/credentials.py create mode 100644 litellm/llms/sap/embed/transformation.py create mode 100644 tests/test_litellm/llms/sap/chat/test_sap_chat_calls.py create mode 100644 tests/test_litellm/llms/sap/embed/test_sap_embedding.py diff --git a/litellm/__init__.py b/litellm/__init__.py index 5a441b99f6f..d2766be03c6 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -265,6 +265,7 @@ heroku_key: Optional[str] = None cometapi_key: Optional[str] = None ovhcloud_key: Optional[str] = None lemonade_key: Optional[str] = None +sap_service_key: Optional[str] = None amazon_nova_api_key: Optional[str] = None common_cloud_provider_auth_params: dict = { "params": ["project", "region_name", "token"], @@ -1069,7 +1070,7 @@ from litellm.litellm_core_utils.core_helpers import remove_index_from_tool_calls from litellm.litellm_core_utils.token_counter import get_modified_max_tokens # client must be imported immediately as it's used as a decorator at function definition time from .utils import client -# Note: Most other utils imports are lazy-loaded via __getattr__ to avoid loading utils.py +# Note: Most other utils imports are lazy-loaded via __getattr__ to avoid loading utils.py # (which imports tiktoken) at import time from .llms.bytez.chat.transformation import BytezChatConfig @@ -1241,6 +1242,7 @@ from .llms.topaz.common_utils import TopazModelInfo from .llms.topaz.image_variations.transformation import TopazImageVariationConfig from litellm.llms.openai.completion.transformation import OpenAITextCompletionConfig from .llms.groq.chat.transformation import GroqChatConfig +from .llms.sap.chat.transformation import GenAIHubOrchestrationConfig from .llms.voyage.embedding.transformation import VoyageEmbeddingConfig from .llms.voyage.embedding.transformation_contextual import ( VoyageContextualEmbeddingConfig, @@ -1339,6 +1341,7 @@ from .llms.azure.chat.o_series_transformation import AzureOpenAIO1Config from .llms.watsonx.completion.transformation import IBMWatsonXAIConfig from .llms.watsonx.chat.transformation import IBMWatsonXChatConfig from .llms.watsonx.embed.transformation import IBMWatsonXEmbeddingConfig +from .llms.sap.embed.transformation import GenAIHubEmbeddingConfig from .llms.watsonx.audio_transcription.transformation import ( IBMWatsonXAudioTranscriptionConfig, ) @@ -1511,13 +1514,13 @@ def set_global_gitlab_config(config: Dict[str, Any]) -> None: if TYPE_CHECKING: from litellm.types.utils import ModelInfo as _ModelInfoType - + # Cost calculator functions cost_per_token: Callable[..., Tuple[float, float]] completion_cost: Callable[..., float] response_cost_calculator: Any modify_integration: Any - + # Utils functions - type stubs for truly lazy loaded functions only # (functions NOT imported via "from .main import *") get_response_string: Callable[..., str] @@ -1547,7 +1550,7 @@ if TYPE_CHECKING: get_first_chars_messages: Callable[..., str] get_provider_fields: Callable[..., List] get_valid_models: Callable[..., list] - + # Response types - truly lazy loaded only (not in main.py or elsewhere) ModelResponseListIterator: Type[Any] @@ -1563,7 +1566,7 @@ def __getattr__(name: str) -> Any: if name in _cost_calculator_names: from ._lazy_imports import _lazy_import_cost_calculator return _lazy_import_cost_calculator(name) - + # Lazy load litellm_logging functions _litellm_logging_names = ( "Logging", @@ -1572,7 +1575,7 @@ def __getattr__(name: str) -> Any: if name in _litellm_logging_names: from ._lazy_imports import _lazy_import_litellm_logging return _lazy_import_litellm_logging(name) - + # Lazy load utils functions _utils_names = ( "exception_type", "get_optional_params", "get_response_string", "token_counter", diff --git a/litellm/litellm_core_utils/get_llm_provider_logic.py b/litellm/litellm_core_utils/get_llm_provider_logic.py index 168c837f1e5..677dac3d313 100644 --- a/litellm/litellm_core_utils/get_llm_provider_logic.py +++ b/litellm/litellm_core_utils/get_llm_provider_logic.py @@ -406,6 +406,8 @@ def get_llm_provider( # noqa: PLR0915 custom_llm_provider = "clarifai" elif model.startswith("amazon_nova"): custom_llm_provider = "amazon_nova" + elif model.startswith("sap/"): + custom_llm_provider = "sap" if not custom_llm_provider: if litellm.suppress_debug_info is False: print() # noqa diff --git a/litellm/litellm_core_utils/get_supported_openai_params.py b/litellm/litellm_core_utils/get_supported_openai_params.py index 19b52d2dace..4b40f44cbc4 100644 --- a/litellm/litellm_core_utils/get_supported_openai_params.py +++ b/litellm/litellm_core_utils/get_supported_openai_params.py @@ -116,6 +116,11 @@ def get_supported_openai_params( # noqa: PLR0915 f"Unsupported provider config: {transcription_provider_config} for model: {model}" ) return litellm.OpenAIConfig().get_supported_openai_params(model=model) + elif custom_llm_provider == "sap": + if request_type == "chat_completion": + return litellm.GenAIHubOrchestrationConfig().get_supported_openai_params(model=model) + elif request_type == "embeddings": + return litellm.GenAIHubEmbeddingConfig().get_supported_openai_params(model=model) elif custom_llm_provider == "azure": if litellm.AzureOpenAIO1Config().is_o_series_model(model=model): return litellm.AzureOpenAIO1Config().get_supported_openai_params( diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index 4ffb7ace5b0..d92af417175 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -441,7 +441,6 @@ class CustomStreamWrapper: finish_reason = None logprobs = None usage = None - if str_line and str_line.choices and len(str_line.choices) > 0: if ( str_line.choices[0].delta is not None diff --git a/litellm/llms/sap/chat/__init__.py b/litellm/llms/sap/chat/__init__.py new file mode 100755 index 00000000000..8b137891791 --- /dev/null +++ b/litellm/llms/sap/chat/__init__.py @@ -0,0 +1 @@ + diff --git a/litellm/llms/sap/chat/handler.py b/litellm/llms/sap/chat/handler.py new file mode 100755 index 00000000000..beabe255130 --- /dev/null +++ b/litellm/llms/sap/chat/handler.py @@ -0,0 +1,262 @@ +from __future__ import annotations + +import json +import time +import httpx + +from typing import Iterator, Optional, AsyncIterator + +from litellm.llms.base_llm.chat.transformation import BaseConfig +from litellm.types.llms.openai import OpenAIChatCompletionChunk +from ...custom_httpx.llm_http_handler import BaseLLMHTTPHandler + + +# ------------------------------- +# Errors +# ------------------------------- +class GenAIHubOrchestrationError(Exception): + def __init__(self, status_code: int, message: str): + super().__init__(message) + self.status_code = status_code + self.message = message + + +# ------------------------------- +# Stream parsing helpers +# ------------------------------- + + +def _now_ts() -> int: + return int(time.time()) + + +def _is_terminal_chunk(chunk: OpenAIChatCompletionChunk) -> bool: + """OpenAI-shaped chunk is terminal if any choice has a non-None finish_reason.""" + try: + for ch in chunk.choices or []: + if ch.finish_reason is not None: + return True + except Exception: + pass + return False + + +class _StreamParser: + """Normalize orchestration streaming events into OpenAI-like chunks.""" + + @staticmethod + def _from_orchestration_result(evt: dict) -> Optional[OpenAIChatCompletionChunk]: + """ + Accepts orchestration_result shape and maps it to an OpenAI-like *chunk*. + """ + orc = evt.get("orchestration_result") or {} + if not orc: + return None + + return OpenAIChatCompletionChunk.model_validate( + { + "id": orc.get("id") or evt.get("request_id") or "stream-chunk", + "object": orc.get("object") or "chat.completion.chunk", + "created": orc.get("created") or evt.get("created") or _now_ts(), + "model": orc.get("model") or "unknown", + "choices": [ + { + "index": c.get("index", 0), + "delta": c.get("delta") or {}, + "finish_reason": c.get("finish_reason"), + } + for c in (orc.get("choices") or []) + ], + } + ) + + @staticmethod + def to_openai_chunk(event_obj: dict) -> Optional[OpenAIChatCompletionChunk]: + """ + Accepts: + - {"final_result": } (IMPORTANT: this is just another chunk, NOT terminal) + - {"orchestration_result": {...}} (map to chunk) + - already-openai-shaped chunks + - other events (ignored) + Raises: + - ValueError for in-stream error objects + """ + # In-stream error per spec (surface as exception) + if "code" in event_obj or "error" in event_obj: + raise ValueError(json.dumps(event_obj)) + + # FINAL RESULT IS *NOT* TERMINAL: treat it as the next chunk + if "final_result" in event_obj: + fr = event_obj["final_result"] or {} + # ensure it looks like an OpenAI chunk + if "object" not in fr: + fr["object"] = "chat.completion.chunk" + return OpenAIChatCompletionChunk.model_validate(fr) + + # Orchestration incremental delta + if "orchestration_result" in event_obj: + return _StreamParser._from_orchestration_result(event_obj) + + # Already an OpenAI-like chunk + if "choices" in event_obj and "object" in event_obj: + return OpenAIChatCompletionChunk.model_validate(event_obj) + + # Unknown / heartbeat / metrics + return None + + +# ------------------------------- +# Iterators +# ------------------------------- +class SAPStreamIterator: + """ + Sync iterator over an httpx streaming response that yields OpenAIChatCompletionChunk. + Accepts both SSE `data: ...` and raw JSON lines. Closes on terminal chunk or [DONE]. + """ + + def __init__( + self, + response: Iterator, + event_prefix: str = "data: ", + final_msg: str = "[DONE]", + ): + self._resp = response + self._iter = response + self._prefix = event_prefix + self._final = final_msg + self._done = False + + def __iter__(self) -> Iterator[OpenAIChatCompletionChunk]: + return self + + def __next__(self) -> OpenAIChatCompletionChunk: + if self._done: + raise StopIteration + + for raw in self._iter: + line = (raw or "").strip() + if not line: + continue + + payload = ( + line[len(self._prefix) :] if line.startswith(self._prefix) else line + ) + if payload == self._final: + self._safe_close() + raise StopIteration + + try: + obj = json.loads(payload) + except Exception: + continue + + try: + chunk = _StreamParser.to_openai_chunk(obj) + except ValueError as e: + self._safe_close() + raise e + + if chunk is None: + continue + + # Close on terminal + if _is_terminal_chunk(chunk): + self._safe_close() + + return chunk + + self._safe_close() + raise StopIteration + + def _safe_close(self) -> None: + if self._done: + return + else: + self._done = True + + +class AsyncSAPStreamIterator: + sync_stream = False + + def __init__( + self, + response:AsyncIterator, + event_prefix: str = "data: ", + final_msg: str = "[DONE]", + ): + self._resp = response + self._prefix = event_prefix + self._final = final_msg + self._line_iter = None + self._done = False + + def __aiter__(self): + return self + + async def __anext__(self): + if self._done: + raise StopAsyncIteration + + if self._line_iter is None: + self._line_iter = self._resp + + while True: + try: + raw = await self._line_iter.__anext__() + except (StopAsyncIteration, httpx.ReadError, OSError): + await self._aclose() + raise StopAsyncIteration + + line = (raw or "").strip() + if not line: + continue + + # now = lambda: int(time.time() * 1000) + payload = ( + line[len(self._prefix) :] if line.startswith(self._prefix) else line + ) + if payload == self._final: + await self._aclose() + raise StopAsyncIteration + try: + obj = json.loads(payload) + except Exception: + continue + + try: + chunk = _StreamParser.to_openai_chunk(obj) + except ValueError as e: + await self._aclose() + raise GenAIHubOrchestrationError(502, str(e)) + + if chunk is None: + continue + + # If terminal, close BEFORE returning. Next __anext__() will stop immediately. + if any(c.finish_reason is not None for c in (chunk.choices or [])): + await self._aclose() + + return chunk + + async def _aclose(self): + if self._done: + return + else: + self._done = True + + +# ------------------------------- +# LLM handler +# ------------------------------- +class GenAIHubOrchestration(BaseLLMHTTPHandler): + def _add_stream_param_to_request_body( + self, + data: dict, + provider_config: BaseConfig, + fake_stream: bool + ): + if data.get("config", {}).get("stream", None) is not None: + data["config"]["stream"]["enabled"] = True + else: + data["config"]["stream"] = {"enabled": True} + return data diff --git a/litellm/llms/sap/chat/models.py b/litellm/llms/sap/chat/models.py new file mode 100644 index 00000000000..d8039ff5618 --- /dev/null +++ b/litellm/llms/sap/chat/models.py @@ -0,0 +1,112 @@ +from typing import Union, Literal + +from pydantic import BaseModel, Field, field_validator + + +def validate_different_content(v: Union[str, dict, list]) -> str: + if v in ((), {}, []): + return "" + elif isinstance(v, dict) and "text" in v: + return v['text'] + elif isinstance(v, list): + new_v = [] + for item in v: + if isinstance(item, dict) and "text" in item: + if item['text']: + new_v.append(item['text']) + elif isinstance(item, str): + new_v.append(item) + return '\n'.join(new_v) + elif isinstance(v, str): + return v + raise ValueError("Content must be a string") + return v + +class TextContent(BaseModel): + type_: Literal["text"] = Field(default="text", alias="type") + text: str + + +class ImageURLContent(BaseModel): + url: str + detail: str = "auto" + + +class ImageContent(BaseModel): + type_: Literal["image_url"] = Field(default="image_url", alias="type") + image_url: ImageURLContent + + +class FunctionObj(BaseModel): + name: str + arguments: str + + +class FunctionTool(BaseModel): + description: str = "" + name: str + parameters: dict = {} + strict: bool = False + + +class ChatCompletionTool(BaseModel): + type_: Literal["function"] = Field(default="function", alias="type") + function: FunctionTool + + +class MessageToolCall(BaseModel): + id: str + type_: Literal["function"] = Field(default="function", alias="type") + function: FunctionObj + + +class SAPMessage(BaseModel): + """ + Model for SystemChatMessage and DeveloperChatMessage + """ + + role: Literal["system", "developer"] = "system" + content: str + + _content_validator = field_validator("content", mode="before")(validate_different_content) + + +class SAPUserMessage(BaseModel): + role: Literal["user"] = "user" + content: Union[ + str, TextContent, ImageContent, list[Union[TextContent, ImageContent]] + ] + + +class SAPAssistantMessage(BaseModel): + role: Literal["assistant"] = "assistant" + content: str = "" + refusal: str = "" + tool_calls: list[MessageToolCall] = [] + + _content_validator = field_validator("content", mode="before")(validate_different_content) + + + +class SAPToolChatMessage(BaseModel): + role: Literal["tool"] = "tool" + tool_call_id: str + content: str + + _content_validator = field_validator("content", mode="before")(validate_different_content) + + +class ResponseFormat(BaseModel): + type_: Literal["text", "json_object"] = Field(default="text", alias="type") + + +class JSONResponseSchema(BaseModel): + description: str = "" + name: str + schema_: dict = Field(default_factory=dict, alias="schema") + strict: bool = False + + +class ResponseFormatJSONSchema(BaseModel): + type_: Literal["json_schema"] = Field(default="json_schema", alias="type") + json_schema: JSONResponseSchema diff --git a/litellm/llms/sap/chat/transformation.py b/litellm/llms/sap/chat/transformation.py new file mode 100755 index 00000000000..01ceb72c0de --- /dev/null +++ b/litellm/llms/sap/chat/transformation.py @@ -0,0 +1,299 @@ +""" +Translate from OpenAI's `/v1/chat/completions` to SAP Generative AI Hub's Orchestration Service`v2/completion` +""" +from typing import List, Optional, Union, Dict, Tuple, Any, TYPE_CHECKING, Iterator, AsyncIterator +from functools import cached_property +import litellm +import httpx + + +from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import ModelResponse + +from ...openai.chat.gpt_transformation import OpenAIGPTConfig + +if TYPE_CHECKING: + from litellm.litellm_core_utils.litellm_logging import Logging as _LiteLLMLoggingObj + + LiteLLMLoggingObj = _LiteLLMLoggingObj +else: + LiteLLMLoggingObj = Any + +from ..credentials import get_token_creator +from .models import ( + SAPMessage, + SAPAssistantMessage, + SAPToolChatMessage, + ChatCompletionTool, + ResponseFormatJSONSchema, + ResponseFormat, + SAPUserMessage, +) +from .handler import GenAIHubOrchestrationError, AsyncSAPStreamIterator, SAPStreamIterator + +def validate_dict(data: dict, model) -> dict: + return model(**data).model_dump(by_alias=True) + + +class GenAIHubOrchestrationConfig(OpenAIGPTConfig): + frequency_penalty: Optional[int] = None + function_call: Optional[Union[str, dict]] = None + functions: Optional[list] = None + logit_bias: Optional[dict] = None + max_tokens: Optional[int] = None + n: Optional[int] = None + presence_penalty: Optional[int] = None + stop: Optional[Union[str, list]] = None + temperature: Optional[int] = None + top_p: Optional[int] = None + response_format: Optional[dict] = None + tools: Optional[list] = None + tool_choice: Optional[Union[str, dict]] = None # + model_version: str = "latest" + + def __init__( + self, + frequency_penalty: Optional[int] = None, + function_call: Optional[Union[str, dict]] = None, + functions: Optional[list] = None, + logit_bias: Optional[dict] = None, + max_tokens: Optional[int] = None, + n: Optional[int] = None, + presence_penalty: Optional[int] = None, + stop: Optional[Union[str, list]] = None, + temperature: Optional[int] = None, + top_p: Optional[int] = None, + response_format: Optional[dict] = None, + tools: Optional[list] = None, + tool_choice: Optional[Union[str, dict]] = None, + ) -> None: + locals_ = locals().copy() + for key, value in locals_.items(): + if key != "self" and value is not None: + setattr(self.__class__, key, value) + self.token_creator = None + self._base_url = None + self._resource_group = None + + def run_env_setup(self, service_key: Optional[str] = None) -> None: + try: + self.token_creator, self._base_url, self._resource_group = get_token_creator(service_key) # type: ignore + except ValueError as err: + raise GenAIHubOrchestrationError(status_code=400, message=err.args[0]) + + + @property + def headers(self) -> Dict[str, str]: + if self.token_creator is None: + self.run_env_setup() + access_token = self.token_creator() # type: ignore + return { + "Authorization": access_token, + "AI-Resource-Group": self.resource_group, + "Content-Type": "application/json", + } + + @property + def base_url(self) -> str: + if self._base_url is None: + self.run_env_setup() + return self._base_url # type: ignore + + + @property + def resource_group(self) -> str: + if self._resource_group is None: + self.run_env_setup() + return self._resource_group # type: ignore + + @cached_property + def deployment_url(self) -> str: + # Keep a short, tight client lifecycle here to avoid fd leaks + client = litellm.module_level_client + # with httpx.Client(timeout=30) as client: + deployments = client.get( + f"{self.base_url}/lm/deployments", headers=self.headers + ).json() + valid: List[Tuple[str, str]] = [] + for dep in deployments.get("resources", []): + if dep.get("scenarioId") == "orchestration": + cfg = client.get( + f'{self.base_url}/lm/configurations/{dep["configurationId"]}', + headers=self.headers, + ).json() + if cfg.get("executableId") == "orchestration": + valid.append((dep["deploymentUrl"], dep["createdAt"])) + # newest first + return sorted(valid, key=lambda x: x[1], reverse=True)[0][0] + + @classmethod + def get_config(cls): + return super().get_config() + + def get_supported_openai_params(self, model): + params = [ + "frequency_penalty", + "logit_bias", + "logprobs", + "top_logprobs", + "max_tokens", + "max_completion_tokens", + "prediction", + "n", + "presence_penalty", + "seed", + "stop", + "stream", + "stream_options", + "temperature", + "top_p", + "tools", + "tool_choice", + "function_call", + "functions", + "extra_headers", + "parallel_tool_calls", + "response_format", + "timeout", + ] + if ( + model.startswith('anthropic') + or model.startswith("amazon") + or model.startswith("cohere") + or model.startswith("alephalpha") + or model == "gpt-4" + ): + params.remove("response_format") + if model.startswith("gemini") or model.startswith("amazon"): + params.remove("tool_choice") + return params + + def validate_environment( + self, + headers: dict, + model: str, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + api_key: Optional[str] = None, + api_base: Optional[str] = None, + ) -> dict: + if api_key: + self.run_env_setup(api_key) + return self.headers + + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ): + api_base_ = f"{self.deployment_url}/v2/completion" + return api_base_ + + def transform_request( + self, + model: str, + messages: List[Dict[str, str]], # type: ignore + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + supported_params = self.get_supported_openai_params(model) + model_params = { + k: v for k, v in optional_params.items() if k in supported_params + } + model_version = optional_params.pop("model_version", "latest") + template = [] + for message in messages: + if message["role"] == "user": + template.append(validate_dict(message, SAPUserMessage)) + elif message["role"] == "assistant": + template.append(validate_dict(message, SAPAssistantMessage)) + elif message["role"] == "tool": + template.append(validate_dict(message, SAPToolChatMessage)) + else: + template.append(validate_dict(message, SAPMessage)) + + tools_ = optional_params.pop("tools", []) + tools_ = [validate_dict(tool, ChatCompletionTool) for tool in tools_] + if tools_ != []: + tools = {"tools": tools_} + else: + tools = {} + + response_format = model_params.pop("response_format", {}) + resp_type = response_format.get("type", None) + if resp_type: + if resp_type== "json_schema": + response_format = validate_dict(response_format, ResponseFormatJSONSchema) + else: + response_format = validate_dict(response_format, ResponseFormat) + response_format = {"response_format": response_format} + model_params.pop("stream", False) + stream_config = {} + if "stream_options" in model_params: + # stream_config["enabled"] = True + stream_options = model_params.pop("stream_options", {}) + stream_config["chunk_size"] = stream_options.get("chunk_size", 100) + if "delimiters" in stream_options: + stream_config["delimiters"] = stream_options.get("delimiters") + # else: + # stream_config["enabled"] = False + config = { + "config": { + "modules": { + "prompt_templating": { + "prompt": { + "template": template, + **tools, + **response_format + }, + "model": { + "name": model, + "params": model_params, + "version": model_version, + }, + }, + }, + "stream": stream_config, + } + } + + return config + + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: LiteLLMLoggingObj, + request_data: dict, + messages: List[AllMessageValues], + optional_params: dict, + litellm_params: dict, + encoding: Any, + api_key: Optional[str] = None, + json_mode: Optional[bool] = None, + ) -> ModelResponse: + logging_obj.post_call( + input=messages, + api_key=api_key, + original_response=raw_response.text, + additional_args={"complete_input_dict": request_data}, + ) + return ModelResponse.model_validate(raw_response.json()["final_result"]) + + def get_model_response_iterator( + self, + streaming_response: Union[Iterator[str], AsyncIterator[str], "ModelResponse"], + sync_stream: bool, + json_mode: Optional[bool] = False, + ): + if sync_stream: + return SAPStreamIterator(response=streaming_response) # type: ignore + else: + return AsyncSAPStreamIterator(response=streaming_response) # type: ignore diff --git a/litellm/llms/sap/credentials.py b/litellm/llms/sap/credentials.py new file mode 100644 index 00000000000..e10bcbf7eae --- /dev/null +++ b/litellm/llms/sap/credentials.py @@ -0,0 +1,325 @@ +from __future__ import annotations +from typing import Any, Callable, Dict, Final, List, Optional, Sequence, Tuple +from datetime import datetime, timedelta, timezone +from threading import Lock +from pathlib import Path +from dataclasses import dataclass +import json +import os +import tempfile + +from litellm import sap_service_key +from litellm.llms.custom_httpx.http_handler import _get_httpx_client + +AUTH_ENDPOINT_SUFFIX = "/oauth/token" + +CONFIG_FILE_ENV_VAR = "AICORE_CONFIG" +HOME_PATH_ENV_VAR = "AICORE_HOME" +PROFILE_ENV_VAR = "AICORE_PROFILE" + +VCAP_SERVICES_ENV_VAR = "VCAP_SERVICES" +VCAP_AICORE_SERVICE_NAME = "aicore" +SERVICE_KEY_ENV_VAR = "AICORE_SERVICE_KEY" + +DEFAULT_HOME_PATH = os.path.join(os.path.expanduser("~"), ".aicore") + + +def _get_home() -> str: + return os.getenv(HOME_PATH_ENV_VAR, DEFAULT_HOME_PATH) + + +def _get_nested(d: Dict[str, Any], path: Sequence[str]) -> Any: + cur: Any = d + for k in path: + if not isinstance(cur, dict) or k not in cur: + raise KeyError(".".join(path)) + cur = cur[k] + return cur + + +def _load_json_env(var_name: str) -> Optional[Dict[str, Any]]: + raw = os.environ.get(var_name) + if not raw: + return None + try: + return json.loads(raw) + except json.JSONDecodeError: + return None + + +def _load_vcap() -> Dict[str, Any]: + return _load_json_env(VCAP_SERVICES_ENV_VAR) or {} + + +def _get_vcap_service(label: str) -> Optional[Dict[str, Any]]: + for services in _load_vcap().values(): + for svc in services: + if svc.get("label") == label: + return svc + return None + + +@dataclass(frozen=True) +class CredentialsValue: + name: str + vcap_key: Optional[Tuple[str, ...]] = None + default: Optional[str] = None + transform_fn: Optional[Callable[[str], str]] = None + + +CREDENTIAL_VALUES: Final[List[CredentialsValue]] = [ + CredentialsValue("client_id", ("clientid",)), + CredentialsValue("client_secret", ("clientsecret",)), + CredentialsValue( + "auth_url", + ("url",), + transform_fn=lambda url: url.rstrip("/") + + ("" if url.endswith(AUTH_ENDPOINT_SUFFIX) else AUTH_ENDPOINT_SUFFIX), + ), + CredentialsValue( + "base_url", + ("serviceurls", "AI_API_URL"), + transform_fn=lambda url: url.rstrip("/") + + ("" if url.endswith("/v2") else "/v2"), + ), + CredentialsValue("resource_group", default="default"), + CredentialsValue( + "cert_url", + ("certurl",), + transform_fn=lambda url: url.rstrip("/") + + ("" if url.endswith(AUTH_ENDPOINT_SUFFIX) else AUTH_ENDPOINT_SUFFIX), + ), + # file paths (kept for config compatibility) + CredentialsValue("cert_file_path"), + CredentialsValue("key_file_path"), + # inline PEMs from VCAP + CredentialsValue( + "cert_str", ("certificate",), transform_fn=lambda s: s.replace("\\n", "\n") + ), + CredentialsValue( + "key_str", ("key",), transform_fn=lambda s: s.replace("\\n", "\n") + ), +] + + +def init_conf(profile: Optional[str] = None) -> Dict[str, Any]: + """ + Loads config JSON from: + 1) $AICORE_CONFIG if set, otherwise + 2) $AICORE_HOME/config.json (or config_.json when profile is given/not default) + Returns {} when nothing is found. + """ + home = Path(_get_home()) + profile = profile or os.environ.get(PROFILE_ENV_VAR) + cfg_env = os.getenv(CONFIG_FILE_ENV_VAR) + cfg_path = ( + Path(cfg_env) + if cfg_env + else ( + home + / ( + "config.json" + if profile in (None, "", "default") + else f"config_{profile}.json" + ) + ) + ) + + if cfg_path and cfg_path.exists(): + try: + with cfg_path.open(encoding="utf-8") as f: + return json.load(f) + except json.JSONDecodeError: + raise KeyError(f"{cfg_path} is not valid JSON. Please fix or remove it!") + + # If an explicit non-default profile was requested but not found, raise. + if cfg_env or (profile not in (None, "", "default")): + raise FileNotFoundError( + f"Unable to locate profile config file at '{cfg_path}' in AICORE_HOME '{home}'" + ) + + return {} + + +def _env_name(name: str) -> str: + return f"AICORE_{name.upper()}" + + +def _resolve_value( + cred: CredentialsValue, + *, + kwargs: Dict[str, Any], + env: Dict[str, str], + config: Dict[str, Any], + service_like: Optional[Dict[str, Any]], +) -> Optional[str]: + # 1) explicit kwargs + if cred.name in kwargs and kwargs[cred.name] is not None: + return kwargs[cred.name] + + # 2) environment variables (primary name) + env_key = _env_name(cred.name) + if env_key in env and env[env_key] is not None: + return env[env_key] + + # 3) config file (accept both prefixed and plain keys) + for key in (env_key, cred.name): + if key in config and config[key] is not None: + return config[key] + + # 4) service-like source (AICORE_SERVICE_KEY first, else VCAP) + if service_like and cred.vcap_key: + try: + val = _get_nested(service_like, ("credentials",) + cred.vcap_key) + if val is not None: + return val + except KeyError: + pass + + # 5) default + return cred.default + + +def fetch_credentials(service_key: Optional[str] = None, profile: Optional[str] = None, **kwargs) -> Dict[str, str]: + """ + Resolution order per key: + kwargs + > env (AICORE_) + > config (AICORE_ or plain ) + > service-like source from JSON in $AICORE_SERVICE_KEY (same structure as a VCAP service object) + falling back to service entry in $VCAP_SERVICES with label 'aicore' + > default + """ + config = init_conf(profile) + env = os.environ # snapshot for testability + service_like = None + + if not config: + # Prefer AICORE_SERVICE_KEY if present; otherwise fall back to the VCAP service. + service_like = service_key or sap_service_key or _load_json_env(SERVICE_KEY_ENV_VAR) or _get_vcap_service( + VCAP_AICORE_SERVICE_NAME + ) + + out: Dict[str, str] = {} + for cred in CREDENTIAL_VALUES: + value = _resolve_value(cred, kwargs=kwargs, env=env, config=config, service_like=service_like) # type: ignore + if value is None: + continue + if cred.transform_fn: + value = cred.transform_fn(value) + out[cred.name] = value + if "cert_url" in out.keys(): + out["auth_url"] = out.pop("cert_url") + return out + + +def get_token_creator( + service_key: Optional[str] = None, + profile: Optional[str] = None, + *, + timeout: float = 30.0, + expiry_buffer_minutes: int = 60, + **overrides, +) -> Tuple[Callable[[], str], str, str]: + """ + Creates a callable that fetches and caches an OAuth2 bearer token + using credentials from `fetch_credentials()`. + + The callable: + - Automatically loads credentials via fetch_credentials(profile, **overrides) + - Fetches a new token only if expired or near expiry + - Caches token thread-safely with a configurable refresh buffer + + Args: + profile: Optional AICore profile name + timeout: HTTP request timeout in seconds (default 30s) + expiry_buffer_minutes: Refresh the token this many minutes before expiry + overrides: Any explicit credential overrides (client_id, client_secret, etc.) + + Returns: + Callable[[], str]: function returning a valid "Bearer " string. + """ + + # Resolve credentials using your helper + credentials: Dict[str, str] = fetch_credentials(service_key=service_key, profile=profile, **overrides) + + auth_url = credentials.get("auth_url") + client_id = credentials.get("client_id") + client_secret = credentials.get("client_secret") + cert_str = credentials.get("cert_str") + key_str = credentials.get("key_str") + cert_file_path = credentials.get("cert_file_path") + key_file_path = credentials.get("key_file_path") + + # Sanity check + if not auth_url or not client_id: + raise ValueError( + "fetch_credentials did not return valid 'auth_url' or 'client_id'" + ) + + modes = [ + client_secret is not None, + (cert_str is not None and key_str is not None), + (cert_file_path is not None and key_file_path is not None), + ] + if sum(bool(m) for m in modes) != 1: + raise ValueError( + "Invalid credentials: provide exactly one of client_secret, " + "(cert_str & key_str), or (cert_file_path & key_file_path)." + ) + + lock = Lock() + token: Optional[str] = None + token_expiry: Optional[datetime] = None + + def _request_token(cert_pair=None) -> tuple[str, datetime]: + data = {"grant_type": "client_credentials", "client_id": client_id} + if client_secret: + data["client_secret"] = client_secret + + client = _get_httpx_client() + # with httpx.Client(cert=cert_pair, timeout=timeout) as client: + resp = client.post(auth_url, data=data) + try: + resp.raise_for_status() + payload = resp.json() + access_token = payload["access_token"] + expires_in = int(payload.get("expires_in", 3600)) + expiry_date = datetime.now(timezone.utc) + timedelta(seconds=expires_in) + return f"Bearer {access_token}", expiry_date + except Exception as e: + msg = getattr(resp, "text", str(e)) + raise RuntimeError(f"Token request failed: {msg}") from e + + def _fetch_token() -> tuple[str, datetime]: + # Case 1: secret-based auth + if client_secret: + return _request_token() + # Case 2: cert/key strings + if cert_str and key_str: + cert_str_fixed = cert_str.replace("\\n", "\n") + key_str_fixed = key_str.replace("\\n", "\n") + with tempfile.TemporaryDirectory() as tmp: + cert_path = os.path.join(tmp, "cert.pem") + key_path = os.path.join(tmp, "key.pem") + with open(cert_path, "w") as f: + f.write(cert_str_fixed) + with open(key_path, "w") as f: + f.write(key_str_fixed) + return _request_token(cert_pair=(cert_path, key_path)) + # Case 3: file-based cert/key + return _request_token(cert_pair=(cert_file_path, key_file_path)) + + def get_token() -> str: + nonlocal token, token_expiry + with lock: + now = datetime.now(timezone.utc) + if ( + token is None + or token_expiry is None + or token_expiry - now < timedelta(minutes=expiry_buffer_minutes) + ): + token, token_expiry = _fetch_token() + return token + + return get_token, credentials["base_url"], credentials["resource_group"] diff --git a/litellm/llms/sap/embed/transformation.py b/litellm/llms/sap/embed/transformation.py new file mode 100644 index 00000000000..93f32c00abd --- /dev/null +++ b/litellm/llms/sap/embed/transformation.py @@ -0,0 +1,176 @@ +""" +Translates from OpenAI's `/v1/embeddings` to IBM's `/text/embeddings` route. +""" + +from typing import Optional, List, Dict, Literal +from pydantic import BaseModel, Field +from functools import cached_property + +import httpx + +from litellm.llms.base_llm.embedding.transformation import ( + BaseEmbeddingConfig, + LiteLLMLoggingObj, +) +from litellm.types.llms.openai import AllEmbeddingInputValues +from litellm.types.utils import EmbeddingResponse + +from ..chat.handler import GenAIHubOrchestrationError +from ..credentials import get_token_creator + + +class Usage(BaseModel): + prompt_tokens: int + total_tokens: int + + +class EmbeddingItem(BaseModel): + object: Literal["embedding"] + embedding: List[float] = Field( + ..., description="Vector of floats (length varies by model)." + ) + index: int + + +class FinalResult(BaseModel): + object: Literal["list"] + data: List[EmbeddingItem] + model: str + usage: Usage + + +class EmbeddingsResponse(BaseModel): + request_id: str + final_result: FinalResult + + +class EmbeddingModel(BaseModel): + name: str + version: str = "latest" + params: dict = Field(default_factory=dict, validation_alias="parameters") + + +class EmbeddingsModules(BaseModel): + embeddings: EmbeddingModel + + +class EmbeddingInput(BaseModel): + text: str | List[str] + type: Literal["text", "document", "query"] = "text" + + +class EmbeddingRequest(BaseModel): + config: EmbeddingsModules + input: EmbeddingInput + + +def validate_dict(data: dict, model) -> dict: + return model(**data).model_dump() + + +class GenAIHubEmbeddingConfig(BaseEmbeddingConfig): + def __init__(self): + super().__init__() + self._access_token_data = {} + self.token_creator, self.base_url, self.resource_group = get_token_creator() + + @property + def headers(self) -> Dict: + access_token = self.token_creator() + # headers for completions and embeddings requests + headers = { + "Authorization": access_token, + "AI-Resource-Group": self.resource_group, + "Content-Type": "application/json", + } + return headers + + @cached_property + def deployment_url(self) -> str: + with httpx.Client(timeout=30) as client: + valid_deployments = [] + deployments = client.get( + self.base_url + "/lm/deployments", headers=self.headers + ).json() + for deployment in deployments.get("resources", []): + if deployment["scenarioId"] == "orchestration": + config_details = client.get( + self.base_url + + f'/lm/configurations/{deployment["configurationId"]}', + headers=self.headers, + ).json() + if config_details["executableId"] == "orchestration": + valid_deployments.append( + (deployment["deploymentUrl"], deployment["createdAt"]) + ) + return sorted(valid_deployments, key=lambda x: x[1], reverse=True)[0][0] + + def get_error_class(self, error_message, status_code, headers): + return GenAIHubOrchestrationError(status_code, error_message) + + def get_supported_openai_params(self, model: str) -> list: + if "text-embedding-3" in model: + return ["encoding_format", "dimensions"] + else: + return [ + "encoding_format", + ] + + def map_openai_params( + self, + non_default_params: dict, + optional_params: dict, + model: str, + drop_params: bool, + ) -> dict: + return optional_params + + def validate_environment(self, headers: dict, *args, **kwargs) -> dict: + return self.headers + + def get_complete_url( + self, + api_base: Optional[str], + api_key: Optional[str], + model: str, + optional_params: dict, + litellm_params: dict, + stream: Optional[bool] = None, + ) -> str: + url = self.deployment_url.rstrip("/") + "/v2/embeddings" + return url + + def transform_embedding_request( + self, + model: str, + input: AllEmbeddingInputValues, + optional_params: dict, + headers: dict, + ) -> dict: + model_dict = {} + model_dict["name"] = model + model_dict["version"] = optional_params.get("version", "latest") + model_dict["params"] = optional_params.get("parameters", {}) + input_dict = {"text": input} + body = { + "config": { + "modules": { + "embeddings": {"model": validate_dict(model_dict, EmbeddingModel)} + } + }, + "input": validate_dict(input_dict, EmbeddingInput), + } + return body + + def transform_embedding_response( + self, + model: str, + raw_response: httpx.Response, + model_response: EmbeddingResponse, + logging_obj: LiteLLMLoggingObj, + api_key: Optional[str], + request_data: dict, + optional_params: dict, + litellm_params: dict, + ) -> EmbeddingResponse: + return EmbeddingResponse.model_validate(raw_response.json()["final_result"]) diff --git a/litellm/main.py b/litellm/main.py index 59838c2a035..20b2cbb7db8 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -176,6 +176,7 @@ from .llms.databricks.embed.handler import DatabricksEmbeddingHandler from .llms.deprecated_providers import aleph_alpha, palm from .llms.gemini.common_utils import get_api_key_from_env from .llms.groq.chat.handler import GroqChatCompletion +from .llms.sap.chat.handler import GenAIHubOrchestration from .llms.heroku.chat.transformation import HerokuChatConfig from .llms.huggingface.embedding.handler import HuggingFaceEmbedding from .llms.lemonade.chat.transformation import LemonadeChatConfig @@ -255,6 +256,8 @@ openai_text_completions = OpenAITextCompletion() openai_audio_transcriptions = OpenAIAudioTranscription() openai_image_variations = OpenAIImageVariationsHandler() groq_chat_completions = GroqChatCompletion() +sap_gen_ai_hub_chat_completions = GenAIHubOrchestration() +sap_gen_ai_hub_emb = GenAIHubOrchestration() azure_ai_embedding = AzureAIEmbedding() anthropic_chat_completions = AnthropicChatCompletion() azure_anthropic_chat_completions = AzureAnthropicChatCompletion() @@ -2093,6 +2096,34 @@ def completion( # type: ignore # noqa: PLR0915 logging_obj=logging, # model call logging done inside the class as we make need to modify I/O to fit aleph alpha's requirements client=client, ) + elif custom_llm_provider == "sap": + headers = headers or litellm.headers + ## LOAD CONFIG - if set + config = litellm.GenAIHubOrchestrationConfig.get_config() + for k, v in config.items(): + if ( + k not in optional_params + ): # completion(top_k=3) > openai_config(top_k=3) <- allows for dynamic variables to be passed in + optional_params[k] = v + + response = sap_gen_ai_hub_chat_completions.completion( + model=model, + messages=messages, + headers=headers, + model_response=model_response, + acompletion=acompletion, + logging_obj=logging, + optional_params=optional_params, + litellm_params=litellm_params, + timeout=timeout, # type: ignore + shared_session=shared_session, + client=client, + custom_llm_provider=custom_llm_provider, + encoding=encoding, + api_key=api_key, + api_base=api_base, + stream=stream, + ) elif custom_llm_provider == "aiohttp_openai": # NEW aiohttp provider for 10-100x higher RPS api_base = ( @@ -4858,6 +4889,21 @@ def embedding( # noqa: PLR0915 client=client, aembedding=aembedding, ) + elif custom_llm_provider == "sap": + response = base_llm_http_handler.embedding( + model=model, + input=input, + custom_llm_provider=custom_llm_provider, + api_base=api_base, + api_key=api_key, + logging_obj=logging, + timeout=timeout, + model_response=EmbeddingResponse(), + optional_params=optional_params, + litellm_params={}, + client=client, + aembedding=aembedding, + ) elif custom_llm_provider == "azure_ai": api_base = ( api_base # for deepinfra/perplexity/anyscale/groq/friendliai we check in get_llm_provider and pass in the api base from there diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index ec9daebbf70..09f3af20e7c 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -995,7 +995,7 @@ class ProxyLogging: ): result = await self._process_guardrail_callback( callback=_callback, - data=data, + data=data, # type: ignore user_api_key_dict=user_api_key_dict, call_type=call_type, ) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 5821ae3d233..3dc31a0771c 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2982,6 +2982,7 @@ class LlmProviders(str, Enum): LANGFUSE = "langfuse" HUMANLOOP = "humanloop" TOPAZ = "topaz" + SAP_GENERATIVE_AI_HUB = "sap" ASSEMBLYAI = "assemblyai" GITHUB_COPILOT = "github_copilot" SNOWFLAKE = "snowflake" diff --git a/litellm/utils.py b/litellm/utils.py index d58eb28a061..d77607fd3e9 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -2886,6 +2886,21 @@ def get_optional_params_embeddings( # noqa: PLR0915 model=model, drop_params=drop_params if drop_params is not None else False, ) + final_params = {**optional_params, **kwargs} + return final_params + elif custom_llm_provider == "sap": + supported_params = get_supported_openai_params( + model=model, + custom_llm_provider="sap", + request_type="embeddings", + ) + _check_valid_arg(supported_params=supported_params) + optional_params = litellm.GenAIHubEmbeddingConfig().map_openai_params( + non_default_params=non_default_params, + optional_params={}, + model=model, + drop_params=drop_params if drop_params is not None else False + ) elif custom_llm_provider == "infinity": supported_params = get_supported_openai_params( model=model, @@ -2899,6 +2914,10 @@ def get_optional_params_embeddings( # noqa: PLR0915 model=model, drop_params=drop_params if drop_params is not None else False, ) + + final_params = {**optional_params, **kwargs} + return final_params + elif custom_llm_provider == "fireworks_ai": supported_params = get_supported_openai_params( model=model, @@ -7216,6 +7235,8 @@ class ProviderConfigManager: return litellm.TritonConfig() elif litellm.LlmProviders.PETALS == provider: return litellm.PetalsConfig() + elif litellm.LlmProviders.SAP_GENERATIVE_AI_HUB == provider: + return litellm.GenAIHubOrchestrationConfig() elif litellm.LlmProviders.FEATHERLESS_AI == provider: return litellm.FeatherlessAIConfig() elif litellm.LlmProviders.NOVITA == provider: @@ -7276,6 +7297,8 @@ class ProviderConfigManager: return litellm.TritonEmbeddingConfig() elif litellm.LlmProviders.WATSONX == provider: return litellm.IBMWatsonXEmbeddingConfig() + elif litellm.LlmProviders.SAP_GENERATIVE_AI_HUB == provider: + return litellm.GenAIHubEmbeddingConfig() elif litellm.LlmProviders.INFINITY == provider: return litellm.InfinityEmbeddingConfig() elif litellm.LlmProviders.SAMBANOVA == provider: diff --git a/tests/test_litellm/llms/sap/chat/test_sap_chat_calls.py b/tests/test_litellm/llms/sap/chat/test_sap_chat_calls.py new file mode 100644 index 00000000000..3984bba27fa --- /dev/null +++ b/tests/test_litellm/llms/sap/chat/test_sap_chat_calls.py @@ -0,0 +1,142 @@ +import httpx +from unittest.mock import patch, PropertyMock + +import pytest + +mock_response = { + "request_id": "e86a0b4e-53e3-97dc-a5f7-82e451376b23", + "intermediate_results": { + "templating": [{"content": "Say hello", "role": "user"}], + "llm": { + "id": "chatcmpl-CUB63bLTYnfO2CQR0r0rArkrbe8CH", + "object": "chat.completion", + "created": 1761308531, + "model": "gpt-4o-2024-08-06", + "system_fingerprint": "fp_4a331a0222", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "Hello from SAP!"}, + "finish_reason": "stop", + } + ], + "usage": {"completion_tokens": 7, "prompt_tokens": 3, "total_tokens": 10}, + }, + }, + "final_result": { + "id": "chatcmpl-CUB63bLTYnfO2CQR0r0rArkrbe8CH", + "object": "chat.completion", + "created": 1761308531, + "model": "gpt-4o-2024-08-06", + "system_fingerprint": "fp_4a331a0222", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "Hello from SAP!"}, + "finish_reason": "stop", + } + ], + "usage": {"completion_tokens": 7, "prompt_tokens": 3, "total_tokens": 10}, + }, +} +mock_stream_response = [ + b'data: {"request_id": "a07127d3-cb74-9427-a4dc-ef9bf424fb43", "intermediate_results": {"templating": [{"content": "Hi", "role": "user"}]}, "final_result": {"id": \'\', "object": \'\', "created": 0, "model": \'\', "system_fingerprint": null, "choices": [{"index": 0, "delta": {"content": ""}}]}}\n\n', + b'data: {"request_id": "a07127d3-cb74-9427-a4dc-ef9bf424fb43", "intermediate_results": {"llm": {"id": "chatcmpl-HelloMsg", "object": "chat.completion.chunk", "created": 1761319270, "model": "gpt-4o-2024-08-06", "system_fingerprint": "fp_HelloMsg", "choices": [{"index": 0, "delta": {"role": "assistant", "content": "Hello "}}]}}, "final_result": {"id": "chatcmpl-HelloMsg", "object": "chat.completion.chunk", "created": 1761319270, "model": "gpt-4o-2024-08-06", "system_fingerprint": "fp_HelloMsg", "choices": [{"index": 0, "delta": {"role": "assistant", "content": "Hello "}}]}}\n\n', + b'data: {"request_id": "a07127d3-cb74-9427-a4dc-ef9bf424fb43", "intermediate_results": {"llm": {"id": "chatcmpl-CUDtFmLex96SxakzBIzhLq2h8Axmk", "object": "chat.completion.chunk", "created": 1761319269, "model": "gpt-4o-2024-08-06", "system_fingerprint": "fp_4a331a0222", "choices": [{"index": 0, "delta": {"role": "assistant", "content": "from SAP!"}, "finish_reason": "stop"}]}}, "final_result": {"id": "chatcmpl-CUDtFmLex96SxakzBIzhLq2h8Axmk", "object": "chat.completion.chunk", "created": 1761319269, "model": "gpt-4o-2024-08-06", "system_fingerprint": "fp_4a331a0222", "choices": [{"index": 0, "delta": {"role": "assistant", "content": "from SAP!"}, "finish_reason": "stop"}]}}\n\n', + b"data: [DONE]\n\n", +] + + +@pytest.fixture +def sap_api_response(): + return mock_response + + +@pytest.fixture +def sap_api_stream_response(): + return mock_response + + +@pytest.fixture +def fake_token_creator(): + return lambda: "Bearer FAKE_TOKEN", "https://api.ai.mock-sap.com", "fake-group" + + +@pytest.fixture +def fake_deployment_url(): + return "https://api.ai.mock-sap.com/v2/inference/deployments/mockid" + + +@pytest.mark.parametrize("sync_mode", [True, False]) +@pytest.mark.asyncio +async def test_sap_chat( + respx_mock, + sap_api_response, + fake_token_creator, + fake_deployment_url, + sync_mode, +): + import litellm + + litellm.disable_aiohttp_transport = True + with patch( + "litellm.llms.sap.chat.transformation.GenAIHubOrchestrationConfig.deployment_url", + new_callable=PropertyMock, + return_value=fake_deployment_url, + ), patch( + "litellm.llms.sap.chat.transformation.get_token_creator", + return_value=fake_token_creator, + ): + model = "sap/gpt-4o" + messages = [{"role": "user", "content": "Hello"}] + respx_mock.post(f"{fake_deployment_url}/v2/completion").respond( + json=sap_api_response + ) + + if sync_mode: + response = litellm.completion(model=model, messages=messages) + else: + response = await litellm.acompletion(model=model, messages=messages) + + assert response.choices[0].message.content == "Hello from SAP!" + assert response.model.startswith("gpt-4o") + assert response.usage.total_tokens == 10 + + +@pytest.mark.asyncio +async def test_sap_streaming( + respx_mock, + sap_api_stream_response, + fake_token_creator, + fake_deployment_url, +): + import litellm + + litellm.disable_aiohttp_transport = True + with patch( + "litellm.llms.sap.chat.transformation.GenAIHubOrchestrationConfig.deployment_url", + new_callable=PropertyMock, + return_value=fake_deployment_url, + ), patch( + "litellm.llms.sap.chat.transformation.get_token_creator", + return_value=fake_token_creator, + ): + model = "sap/gpt-4o" + messages = [{"role": "user", "content": "Hello"}] + + respx_mock.post(f"{fake_deployment_url}/v2/completion").mock( + return_value=httpx.Response( + 200, + content=mock_stream_response, + headers={"Content-Type": "text/event-stream"}, + ) + ) + + stream = litellm.completion(model=model, messages=messages, stream=True) + + full = "" + for chunk in stream: + delta = getattr(chunk.choices[0].delta, "content", None) or "" + full += delta + + assert full == "Hello from SAP!" diff --git a/tests/test_litellm/llms/sap/embed/test_sap_embedding.py b/tests/test_litellm/llms/sap/embed/test_sap_embedding.py new file mode 100644 index 00000000000..617740bb43f --- /dev/null +++ b/tests/test_litellm/llms/sap/embed/test_sap_embedding.py @@ -0,0 +1,1607 @@ +import httpx +from unittest.mock import patch, PropertyMock + +import pytest + +moke_response = { + "request_id": "9c18627f-ffce-9264-b441-e1f8967d5085", + "final_result": { + "object": "list", + "data": [ + { + "object": "embedding", + "embedding": [ + -0.0069594960659742355, + -0.035274259746074677, + 0.0015957315918058157, + 0.06534460932016373, + 0.03293841332197189, + -0.024201158434152603, + -0.02610827423632145, + 0.04937804862856865, + 0.01623266376554966, + -0.05168433114886284, + -0.013357206247746944, + -0.014599049463868141, + -0.026019571349024773, + -0.003257990349084139, + 0.024585537612438202, + 0.001171619864180684, + -0.05345839262008667, + 0.015057348646223545, + 0.011487049050629139, + 0.03394371271133423, + 0.04934848099946976, + 0.020372141152620316, + -0.01396334357559681, + 0.01887897402048111, + 0.017149262130260468, + 0.024156806990504265, + 0.01827283576130867, + -0.0011956436792388558, + 0.01955902948975563, + -0.03678221255540848, + 0.027675362303853035, + -0.028207581490278244, + 0.027645794674754143, + -0.01623266376554966, + -0.011716199107468128, + -0.01604047417640686, + -0.01407422311604023, + 0.03758053854107857, + 0.01887897402048111, + -0.037698812782764435, + 0.04343494400382042, + -0.012411040253937244, + 0.020948711782693863, + 0.013556787744164467, + 0.0019680997356772423, + 0.0002180617448175326, + -0.049171075224876404, + 0.00832330621778965, + 0.018331971019506454, + 0.029464207589626312, + -0.02678833156824112, + 0.007635856978595257, + 0.025014270097017288, + 0.10638456791639328, + 0.03624999523162842, + -0.010289558209478855, + 0.05954933911561966, + 0.02360980398952961, + -0.0040766457095742226, + 0.00014760748308617622, + -0.019056379795074463, + 0.009683419950306416, + 0.008338090032339096, + 0.004091429989784956, + -0.00712211849167943, + -0.013593747280538082, + -0.02390548214316368, + 0.012167106382548809, + -0.020933927968144417, + -0.013837681151926517, + -0.0013979976065456867, + 0.036220427602529526, + -0.028665879741311073, + -0.007184949703514576, + -0.00439819460734725, + -0.03018861636519432, + -0.08533236384391785, + -0.03775794804096222, + -0.0012566270306706429, + 0.0097425552085042, + -0.02742403745651245, + 0.02217577025294304, + -0.03710745647549629, + -0.011937957257032394, + -0.05629689246416092, + -0.006068769376724958, + -0.08083807677030563, + 0.016350936144590378, + -0.026684844866394997, + -0.00926947221159935, + -0.0157004464417696, + 0.04420370236039162, + -0.02841455489397049, + -0.02360980398952961, + 0.009350783191621304, + 0.01164967194199562, + -0.024733377620577812, + -0.0011577600380405784, + 0.059076253324747086, + -0.009380350820720196, + 0.03787621855735779, + -0.029375504702329636, + 0.03967984765768051, + -0.010548274964094162, + 0.011538793332874775, + 0.03926589712500572, + 0.008456360548734665, + -0.011228332296013832, + -0.053133148699998856, + 0.025590840727090836, + -0.06599509716033936, + -0.07817698270082474, + -0.0011134084779769182, + 0.010112151503562927, + 0.008959011174738407, + 0.04059644415974617, + 0.015138659626245499, + -0.05739089474081993, + -0.00017786816169973463, + -0.0665864497423172, + 0.003707049647346139, + -0.004886061418801546, + 0.02834063582122326, + -0.036604806780815125, + -0.06085031479597092, + 0.02111133374273777, + 0.0011392802698537707, + 0.016883153468370438, + -0.04736744612455368, + 0.0036054106894880533, + 0.05073816329240799, + 0.015463904477655888, + -0.02579781413078308, + -0.0072219097055494785, + -0.029863372445106506, + 0.03672307729721069, + -0.03749183565378189, + 0.028311068192124367, + -0.043789755553007126, + -0.022530583664774895, + 0.03128262236714363, + -0.008818564936518669, + 0.005717652849853039, + -0.015907419845461845, + 0.022027932107448578, + -0.015005605295300484, + 0.0012538550654426217, + 0.06705953180789948, + -0.028089310973882675, + -0.015389985404908657, + 0.027069224044680595, + 0.021820958703756332, + -0.052305251359939575, + 0.02448205091059208, + 0.012751067988574505, + -0.029523342847824097, + 0.015042564831674099, + -0.029464207589626312, + -0.023846345022320747, + 0.008545063436031342, + 0.0332932248711586, + 0.016099609434604645, + 0.01224102545529604, + -0.0526009276509285, + -0.0389702208340168, + 0.01159792859107256, + 0.028399771079421043, + 0.03678221255540848, + -0.032820142805576324, + -0.0012483111349865794, + -0.024866431951522827, + 0.029272018000483513, + -0.03234705701470375, + -0.007953709922730923, + -0.012654973194003105, + -0.005244569852948189, + 0.009299039840698242, + 0.00017197772103827447, + -0.0684787780046463, + -0.017962373793125153, + 0.016365719959139824, + 0.09745512157678604, + 0.005403496325016022, + 0.005754612386226654, + -0.032820142805576324, + -0.020593900233507156, + -0.011923172511160374, + 0.005255657713860273, + 0.021318307146430016, + 0.04334624111652374, + 0.000568716146517545, + 0.061914753168821335, + 0.004032294265925884, + 0.005163258872926235, + -0.006113120820373297, + -0.044321972876787186, + 0.0809563472867012, + -0.007366051897406578, + -0.005285225342959166, + -0.002382047474384308, + 0.021880093961954117, + -0.05535072460770607, + 0.01717883162200451, + -0.014488170854747295, + -0.024526402354240417, + -0.021273955702781677, + 0.022220123559236526, + 0.058011818677186966, + -0.00015661638462916017, + -0.04816577583551407, + 0.05248265713453293, + -0.03382544219493866, + 0.0070481994189321995, + 0.030129481106996536, + -0.013379381969571114, + -0.034712474793195724, + 0.045002032071352005, + 0.002792299259454012, + 0.049998972564935684, + 0.012329728342592716, + -0.009409918449819088, + 0.002725771861150861, + 0.06226956471800804, + 0.034180253744125366, + 0.021850526332855225, + 0.017844103276729584, + -0.013083704747259617, + -0.01316501572728157, + 0.015449120663106441, + -0.03420982137322426, + 0.02232361026108265, + 0.04923021048307419, + -0.047722261399030685, + -0.04656912013888359, + 0.019987761974334717, + 0.022841043770313263, + 0.030366022139787674, + -0.01254409458488226, + 0.016395287588238716, + -0.01499821338802576, + -0.03382544219493866, + 0.006460541393607855, + 0.0006504892953671515, + 0.02773449756205082, + 0.022160986438393593, + -0.00404707808047533, + 0.008655942976474762, + -0.06504892557859421, + 0.013194584287703037, + 0.015463904477655888, + 0.008582023903727531, + 0.010548274964094162, + 0.013401557691395283, + -0.022190555930137634, + -0.02746838890016079, + 0.02587173320353031, + 0.003599866759032011, + 0.03867454454302788, + -0.02485164813697338, + -0.04050774127244949, + -0.023698506876826286, + -0.03734399750828743, + -0.006582507863640785, + -0.04444024711847305, + -0.055942077189683914, + -0.042429640889167786, + 0.01164967194199562, + 0.03125305473804474, + -0.013926384039223194, + 0.005806356202811003, + -0.003505619941279292, + -0.026418736204504967, + 0.03536296263337135, + -0.010089975781738758, + -0.006885576993227005, + 0.015049956738948822, + 0.020534764975309372, + -0.016557909548282623, + -0.00034349344787187874, + 0.01642485521733761, + -0.046628255397081375, + -0.023713290691375732, + -0.006349662318825722, + 0.0355699360370636, + -0.06853791326284409, + 0.020194735378026962, + -0.009727771393954754, + -0.03456463664770126, + 0.03426895663142204, + 0.029567694291472435, + 0.03589517995715141, + 0.0104965316131711, + -0.01106570940464735, + -0.03344106301665306, + 0.002962313359603286, + 0.01793280616402626, + -0.006582507863640785, + -0.022146202623844147, + 0.023698506876826286, + -0.032406192272901535, + 0.0714946836233139, + 0.014842982403934002, + 0.009380350820720196, + -0.0008722469792701304, + -0.00832330621778965, + 0.028843285515904427, + 0.0036035627126693726, + -0.031164349988102913, + 0.008205035701394081, + 0.006678603123873472, + -0.035185556858778, + 0.014029871672391891, + 0.0348011776804924, + -0.02807452715933323, + -0.04103996232151985, + 0.003487139940261841, + -0.0004827850207220763, + -0.014658184722065926, + 0.002358023775741458, + -0.04420370236039162, + 0.0064790211617946625, + -0.04639171436429024, + 0.027571875602006912, + -0.01717883162200451, + 0.0006398633704520762, + -0.021022630855441093, + 0.06085031479597092, + 0.008648551069200039, + -0.01808064617216587, + -0.011516617611050606, + -0.0010561210801824927, + -0.023668939247727394, + 0.003780968952924013, + -0.007377139758318663, + 0.01914508268237114, + 0.007894574664533138, + -0.0431392677128315, + 0.02356545254588127, + -0.05224611610174179, + 0.05091556906700134, + -0.024807296693325043, + -0.05008767545223236, + -0.05836663022637367, + 0.010696114040911198, + 0.00684861745685339, + -0.010681330226361752, + 0.004294707905501127, + -0.010607410222291946, + -0.01737102121114731, + -0.008825956843793392, + 0.03663437440991402, + 0.03690048307180405, + -0.015996122732758522, + -0.025901300832629204, + -0.02072695456445217, + -0.03616129234433174, + 0.07108073681592941, + -0.0029087220318615437, + -0.01164967194199562, + 0.05008767545223236, + -0.05656300112605095, + 0.00023573306680191308, + 0.0010182374389842153, + -0.0366935096681118, + 0.05357666313648224, + 0.03208094835281372, + -0.060554638504981995, + 0.004283619578927755, + 0.020800873637199402, + 0.05103383958339691, + -0.013172407634556293, + 0.013652883470058441, + 0.001349950092844665, + 0.011775334365665913, + -0.04101039096713066, + 0.023121938109397888, + 0.03861540928483009, + -0.005329576786607504, + 0.011487049050629139, + -0.0016095914179459214, + 0.019455542787909508, + 0.05064946040511131, + 0.017563210800290108, + -0.028577176854014397, + 0.057302191853523254, + -0.007384531665593386, + 0.025014270097017288, + -0.038763247430324554, + -0.030957376584410667, + 0.015833500772714615, + 0.07805871218442917, + 0.0031988550908863544, + 0.042547911405563354, + -0.02356545254588127, + 0.018361538648605347, + -0.02035735733807087, + 0.03604302182793617, + 0.02273755706846714, + -0.009609500877559185, + -0.012204065918922424, + 0.021880093961954117, + -0.06034766510128975, + 0.010393044911324978, + 0.009180769324302673, + -0.01244799979031086, + -0.019381623715162277, + -0.04937804862856865, + 0.01159792859107256, + 0.020327789708971977, + -0.016025690361857414, + 0.015508255921304226, + -0.048638857901096344, + 0.05688824504613876, + 0.02278190851211548, + 0.027039656415581703, + -0.028311068192124367, + -0.013231543824076653, + -0.00849332008510828, + 0.015153443440794945, + 0.009587325155735016, + 0.012181890197098255, + 0.00012681768566835672, + -0.01982514001429081, + 0.025590840727090836, + -0.017193615436553955, + 0.046480417251586914, + 0.04316883534193039, + -0.06386622041463852, + 0.013918992131948471, + -0.07267739623785019, + -0.02300366573035717, + 0.01127268373966217, + -0.01212275493890047, + 0.040537308901548386, + -0.04044860601425171, + -0.012854555621743202, + 0.006870793178677559, + 0.038763247430324554, + 0.023846345022320747, + 0.023639371618628502, + -0.005876579321920872, + -0.007872398942708969, + -0.008382441475987434, + -0.039709415286779404, + -0.010016056708991528, + -0.04423326998949051, + -0.0039731590077281, + -0.007133206352591515, + -0.0228853952139616, + -0.018642431125044823, + -0.014865159057080746, + -0.0026592444628477097, + 0.007961101830005646, + 0.003590626874938607, + 0.006804266013205051, + 0.0219392292201519, + 0.010866127908229828, + -0.009956921450793743, + 0.00272392388433218, + -0.029419856145977974, + 0.024733377620577812, + -0.0003227036795578897, + 0.04760398715734482, + 0.035510800778865814, + 0.023624587804079056, + -0.03477161005139351, + 0.0021954013500362635, + -0.007717168424278498, + -0.022220123559236526, + -0.07575243711471558, + 0.02936072088778019, + 0.01184186153113842, + 0.04748571664094925, + -0.009299039840698242, + -0.014887334778904915, + -0.04003465920686722, + 0.0020715866703540087, + -0.019958194345235825, + 0.00602441793307662, + -0.015183011069893837, + 0.004202308598905802, + -0.006094641052186489, + -0.03423938900232315, + 0.06611336767673492, + -0.010245205834507942, + 0.07238171994686127, + 0.02440813183784485, + 0.021096549928188324, + -0.04399672895669937, + -0.034978583455085754, + 0.01963294856250286, + 0.019854707643389702, + 0.08704729378223419, + -0.01842067390680313, + -0.029183315113186836, + -0.015759581699967384, + 0.0019514678278937936, + -0.003065800294280052, + -0.0404781736433506, + -0.051506925374269485, + -0.015345633961260319, + 0.008293738588690758, + -0.008160683326423168, + 0.0518321692943573, + 0.04177915304899216, + -0.03879281505942345, + -0.03249489516019821, + 0.012551486492156982, + -0.012285376898944378, + -0.015049956738948822, + 0.006774697918444872, + 0.04523857310414314, + -0.012573662213981152, + -0.056503865867853165, + -0.024659456685185432, + -0.0035887788981199265, + -0.0016465509543195367, + -0.006275743246078491, + 0.02448205091059208, + -0.03420982137322426, + 0.02569432742893696, + 0.006767306011170149, + -0.025413433089852333, + -0.0068190498277544975, + 0.005026508122682571, + -0.016469206660985947, + -0.011494440957903862, + -0.031223485246300697, + -0.005362840835005045, + 0.0019662517588585615, + 0.038911085575819016, + -0.017415372654795647, + -0.032406192272901535, + 0.005717652849853039, + -0.028784150257706642, + 0.009609500877559185, + -0.005359144881367683, + -0.04438111186027527, + 0.0003402594884391874, + -0.02871023118495941, + 0.03435766324400902, + 0.006205520126968622, + 0.013054137118160725, + 0.011701415292918682, + -0.00207528262399137, + 0.024955134838819504, + 0.03314538672566414, + -0.0056733014062047005, + 0.041335638612508774, + 0.04183828830718994, + -0.01967730186879635, + -0.030957376584410667, + 0.04441067948937416, + -0.010200854390859604, + 0.021007847040891647, + -0.01846502535045147, + -0.025265594944357872, + -0.004475809633731842, + 0.009905178099870682, + 0.019470326602458954, + 0.00424666004255414, + -0.002463358687236905, + 0.013652883470058441, + 0.007236693520098925, + 0.0006560332258231938, + 0.03412111848592758, + -0.009417311288416386, + -0.028237149119377136, + 0.005802660249173641, + -0.023506317287683487, + -0.016217879951000214, + -0.008759429678320885, + -0.028917206451296806, + -0.01110266987234354, + 0.008655942976474762, + -0.015227362513542175, + -0.0034353965893387794, + 0.010385653004050255, + 0.0355699360370636, + 0.0097425552085042, + -0.024393348023295403, + -0.022427096962928772, + -0.008811173029243946, + -0.03317495435476303, + -0.023624587804079056, + 0.001332394196651876, + -0.010740465484559536, + 0.027971038594841957, + 0.02958247810602188, + 0.03057299740612507, + 0.016927504912018776, + -2.5496361558907665e-05, + -0.001735254074446857, + 0.027246631681919098, + 0.011627496220171452, + -0.004120997618883848, + -0.021421795710921288, + -0.045800358057022095, + -0.034328095614910126, + 0.004265139810740948, + 0.0019015723373740911, + -0.015404769219458103, + -0.0014174013631418347, + -0.06652731448411942, + 0.01269193273037672, + -0.0037698810920119286, + 0.0025964132510125637, + 0.02239752933382988, + -0.01686836965382099, + -0.023920265957713127, + -0.0007895498420111835, + 0.04089212045073509, + 0.011509224772453308, + -0.04130607098340988, + -0.03305668383836746, + 0.02300366573035717, + -0.001791617483831942, + 0.026832683011889458, + 0.016291799023747444, + -0.01284716371446848, + -0.015449120663106441, + -0.035540368407964706, + 0.007299524731934071, + 0.03772838041186333, + 0.03962071239948273, + 0.02122960425913334, + -0.023062802851200104, + -0.026049138978123665, + -0.034978583455085754, + 0.002077130600810051, + 0.019839923828840256, + 0.024378564208745956, + 0.023994185030460358, + -0.013660275377333164, + -0.027290983125567436, + -0.003294949885457754, + -0.006741434335708618, + -0.010223030112683773, + 0.015996122732758522, + -0.03923632949590683, + 0.010577842593193054, + -0.0032986460719257593, + 0.01633615233004093, + -0.000837135361507535, + 0.034150686115026474, + -0.0003019138821400702, + 0.005322184879332781, + 0.014421642757952213, + 0.011161805130541325, + -0.018967676907777786, + 0.0025760855060070753, + -0.04444024711847305, + -0.004697567317634821, + -0.01618831232190132, + -0.0034446364734321833, + -0.0031304797157645226, + -0.04030076786875725, + -0.006131600588560104, + -0.01237408071756363, + 0.005684389267116785, + 0.00957254134118557, + -0.009262080304324627, + -0.017297102138400078, + -0.018775485455989838, + -0.011021357960999012, + 0.009143809787929058, + 0.013128056190907955, + 0.004789966624230146, + -0.014983429573476315, + -0.021022630855441093, + 0.01349026057869196, + 0.002108546206727624, + 0.03690048307180405, + -0.028473690152168274, + 0.04778139665722847, + 0.005211306270211935, + 0.03988682106137276, + -0.0507085956633091, + -0.019396407529711723, + -0.03438723087310791, + -0.016705747693777084, + 0.005100427195429802, + -0.017696265131235123, + -0.016779666766524315, + 0.00019877344311680645, + 0.030336454510688782, + 0.03137132525444031, + -0.009284256026148796, + 0.003651610342785716, + 0.02826671674847603, + 0.00027858311659656465, + 0.009535581804811954, + 0.02965639717876911, + -0.01899724453687668, + 0.003202551044523716, + -0.03698918595910072, + 0.045800358057022095, + -0.025339514017105103, + 0.024940351024270058, + -0.07770390063524246, + 0.0022027932573109865, + -0.02807452715933323, + -0.016321366652846336, + 0.0020309309475123882, + -0.0023266079369932413, + 0.0060133300721645355, + -0.029715532436966896, + 0.018095429986715317, + -0.0025465176440775394, + 0.02579781413078308, + -0.02414202317595482, + 0.01660226099193096, + -0.017755400389432907, + -0.008404617197811604, + 0.04535684362053871, + 0.02455596998333931, + -0.013734194450080395, + -0.02943463996052742, + 0.022707989439368248, + -0.02183574251830578, + -0.010548274964094162, + -0.02397940121591091, + -0.02307758666574955, + -0.014369899407029152, + 0.01895289309322834, + -0.031075647100806236, + -0.03829016536474228, + 0.012100579217076302, + 0.0541088804602623, + 0.01244799979031086, + -0.012876731343567371, + 0.00900336354970932, + 0.013283287174999714, + 0.027897119522094727, + 0.03450550138950348, + 0.002345087705180049, + -0.031460028141736984, + -0.038763247430324554, + -0.020564332604408264, + -0.0597858801484108, + -0.0011956436792388558, + 0.012004484422504902, + 0.020401708781719208, + -0.004261443857103586, + 0.014347723685204983, + -0.02397940121591091, + 0.04166088253259659, + 0.04151304438710213, + -0.024053320288658142, + -0.006216607987880707, + 0.019115515053272247, + 0.012078403495252132, + -0.02220533974468708, + 0.012381472624838352, + 0.00907728262245655, + -0.027113575488328934, + 0.03766924515366554, + -0.0183467548340559, + 0.043494079262018204, + 0.01822848431766033, + 0.02281147614121437, + 0.03589517995715141, + -0.012884123250842094, + -0.016897937282919884, + -0.030055562034249306, + -0.012004484422504902, + -0.03645696863532066, + -0.018893757835030556, + -0.038231030106544495, + -0.012876731343567371, + 0.01822848431766033, + -0.019795572385191917, + -0.001042261254042387, + 0.003895543748512864, + 0.016853585839271545, + -0.01611439324915409, + -0.007768911775201559, + -0.012943258509039879, + -0.03571777418255806, + 0.019307704642415047, + 0.004956285003572702, + 0.026389168575406075, + 0.033766306936740875, + -0.00029914191691204906, + -0.02077130600810051, + -0.04707176983356476, + -0.008648551069200039, + 0.0019828835502266884, + -0.002997425151988864, + -0.007983277551829815, + -0.00564742973074317, + -0.03574734181165695, + 0.02671441249549389, + -0.028059743344783783, + 0.007480626925826073, + 0.02477772906422615, + 0.010356085374951363, + -0.019869491457939148, + 0.0202390868216753, + 0.00332451774738729, + -0.009372958913445473, + -0.021170469000935555, + 0.020712170749902725, + 0.018095429986715317, + -0.004017510451376438, + -0.002457814523950219, + -0.028798934072256088, + -0.008448968641459942, + -0.006527068559080362, + -0.008264170959591866, + -0.013113272376358509, + -0.00602441793307662, + -0.010577842593193054, + 0.007665425073355436, + 0.0021621377673000097, + -0.010940046980977058, + 0.011265291832387447, + -0.043257538229227066, + 0.013667667284607887, + 0.022027932107448578, + 0.04801793769001961, + 0.04834318161010742, + -0.015389985404908657, + -0.036870915442705154, + -0.0021418097894638777, + 0.026507439091801643, + 0.01975122094154358, + 0.000411175744375214, + 0.014000303111970425, + -0.04077384993433952, + 0.01401508692651987, + -0.03503771871328354, + -0.012980218045413494, + -0.02618219330906868, + 0.005100427195429802, + 0.05328098684549332, + 0.00936556700617075, + -0.01139095425605774, + -0.01556739117950201, + -0.033500198274850845, + 0.02258971892297268, + -0.009587325155735016, + 0.030025994405150414, + 0.003507467918097973, + 0.01604047417640686, + 0.029833804816007614, + -0.009321215562522411, + -0.01096961461007595, + -0.01728231832385063, + 2.100634628732223e-05, + -0.011775334365665913, + 0.007207125425338745, + 0.017829319462180138, + -0.02470380999147892, + 0.0017093823989853263, + -0.003806840628385544, + -0.02780841663479805, + 0.018331971019506454, + 0.02603435516357422, + -0.010962222702801228, + -0.04056687653064728, + 0.03775794804096222, + 0.03110521472990513, + 4.5015662180958316e-05, + 0.0038179284892976284, + -0.04719004034996033, + 0.021362660452723503, + -0.01660226099193096, + 0.025590840727090836, + 0.023447182029485703, + 0.012063619680702686, + -0.003446484450250864, + -0.02579781413078308, + -0.04423326998949051, + 0.02671441249549389, + 0.041631314903497696, + 0.015596958808600903, + 0.016143960878252983, + -0.008825956843793392, + -0.003448332427069545, + 0.040655579417943954, + 9.528651571599767e-05, + -0.0021344178821891546, + -0.002557605504989624, + 0.029523342847824097, + 0.0016659548273310065, + 0.011361386626958847, + 0.011886212974786758, + -0.02822236530482769, + 0.028931990265846252, + 0.008885092101991177, + -0.04101039096713066, + 0.013726802542805672, + -0.03500815108418465, + -0.008012845180928707, + 0.0035592112690210342, + -0.021480930969119072, + 0.009469054639339447, + -0.014828198589384556, + -0.0005927399033680558, + -0.02659614197909832, + -0.030927808955311775, + -0.015286498703062534, + -0.005625254008919001, + 0.008589415811002254, + 0.01139095425605774, + 0.030898241326212883, + -0.005780484527349472, + -0.001752809970639646, + -0.036220427602529526, + -0.0036867219023406506, + -0.02943463996052742, + 0.009321215562522411, + -0.012684540823101997, + -0.013911600224673748, + 0.014599049463868141, + -0.022678421810269356, + 0.008855524472892284, + 0.013623315840959549, + 0.0009013526723720133, + 0.000988669809885323, + -0.01970686949789524, + -0.004309491720050573, + 0.018450241535902023, + -0.038497138768434525, + -0.009469054639339447, + -0.011775334365665913, + 0.029257234185934067, + 0.0078502232208848, + 0.03098694421350956, + -0.02387591451406479, + 0.006859705317765474, + -0.008212427608668804, + 0.014850374311208725, + 0.02560562454164028, + 0.001327774254605174, + 0.024452483281493187, + 0.017001423984766006, + -0.017563210800290108, + 0.008071980439126492, + -0.011827077716588974, + -0.017001423984766006, + 0.027438821271061897, + 0.017237966880202293, + 0.019322488456964493, + 0.04795880243182182, + 0.004745615180581808, + 0.009838650934398174, + 0.0023986792657524347, + -0.0321696512401104, + 0.025635192170739174, + 0.008182859979569912, + 0.0317852720618248, + 0.03657523915171623, + -0.022294042631983757, + -0.03610215708613396, + 0.039709415286779404, + -0.01530128251761198, + 0.007173861842602491, + 0.03533339500427246, + -0.0052002184092998505, + -0.03011469729244709, + -0.02217577025294304, + -0.0015578479506075382, + 0.011923172511160374, + 0.018982460722327232, + 0.013955951668322086, + -0.02800060622394085, + 0.0012113514821976423, + 0.02814844623208046, + -0.03548123314976692, + 0.011812293902039528, + 0.03840843588113785, + 0.005292617250233889, + 0.027113575488328934, + -0.007325396407395601, + 0.01159792859107256, + 0.010903087444603443, + 0.026625709608197212, + -0.005758308805525303, + 0.004091429989784956, + 0.021954013034701347, + 0.032140083611011505, + -0.00607246533036232, + 0.0014931686455383897, + 0.0026001092046499252, + 0.0332932248711586, + 0.050412919372320175, + -0.0024337908253073692, + 0.023476749658584595, + 0.007658033166080713, + -0.0166466124355793, + -0.0017971614142879844, + 0.0057213488034904, + 0.007983277551829815, + -0.042843591421842575, + -0.019159866496920586, + 0.01369723491370678, + 0.033884577453136444, + 0.016350936144590378, + -0.004298403859138489, + -0.01926335319876671, + 0.05830749496817589, + 0.018553728237748146, + -0.0020106032025069, + 0.015508255921304226, + 0.06215129420161247, + 0.005558726843446493, + 0.01339416578412056, + -0.0033522373996675014, + 0.031992245465517044, + -0.030336454510688782, + 0.021747039631009102, + 0.009129025973379612, + 0.0157004464417696, + 0.026684844866394997, + 0.04293229430913925, + 0.030173832550644875, + -0.04949632287025452, + 0.006789481732994318, + 0.03456463664770126, + -0.02285582758486271, + -0.008293738588690758, + -0.02152528241276741, + 0.014259020797908306, + -0.018065862357616425, + 0.020327789708971977, + 0.00298818526789546, + 0.043523646891117096, + 0.046480417251586914, + -0.0035425794776529074, + 0.030898241326212883, + 0.015863068401813507, + 0.020446060225367546, + -0.01713447831571102, + 0.01967730186879635, + -0.03690048307180405, + -0.04151304438710213, + -0.010910479351878166, + -0.02096349559724331, + -0.023506317287683487, + -0.013978127390146255, + -0.004497985355556011, + -0.014103790745139122, + -0.07273653149604797, + 0.05910582095384598, + -0.009143809787929058, + -0.0008061816915869713, + 0.024659456685185432, + 0.006264655385166407, + 0.002077130600810051, + 0.004224484320729971, + -0.00976473093032837, + 0.006238783709704876, + 0.029612045735120773, + 0.009890394285321236, + -0.005887667182832956, + 0.00929164793342352, + 0.005710260942578316, + 0.024230726063251495, + -0.01883462257683277, + 0.002962313359603286, + -0.032524462789297104, + -0.027335334569215775, + 0.006349662318825722, + -0.04952589049935341, + 0.012226241640746593, + 0.008655942976474762, + 0.003224726766347885, + 0.021776607260107994, + 0.00597637053579092, + -0.011383562348783016, + 0.01454730611294508, + 0.011967524886131287, + -0.005676997359842062, + -0.01725275069475174, + -0.006534460466355085, + -0.04970329627394676, + 0.028798934072256088, + 0.017622346058487892, + 0.010282166302204132, + -0.01021563820540905, + 0.024378564208745956, + -0.012810204178094864, + -0.039177194237709045, + 0.0026222849264740944, + 0.031755704432725906, + -0.027364902198314667, + 0.041365206241607666, + 0.01744494028389454, + 0.0008010997553355992, + 0.013305462896823883, + -0.0202390868216753, + -0.018524160608649254, + -0.012861947529017925, + 0.004254051949828863, + 0.007569329813122749, + 0.05588294193148613, + 0.02167312055826187, + -0.004335363395512104, + -0.008922051638364792, + -0.00031831470550969243, + 0.02807452715933323, + 0.009661244228482246, + -0.006693386938422918, + -0.02511775679886341, + 0.01235190499573946, + 0.002814474981278181, + 0.001953315921127796, + 0.01473949570208788, + 0.024275077506899834, + 0.017016207799315453, + -0.00896640308201313, + -0.013556787744164467, + -0.006142688449472189, + -0.011117453686892986, + -0.0308391060680151, + -0.04441067948937416, + 0.0037791209761053324, + -0.03184440732002258, + 0.014089006930589676, + -0.018790269270539284, + 0.015670878812670708, + 0.00660098809748888, + -0.023639371618628502, + 0.013519828207790852, + 0.048520587384700775, + -0.015375201590359211, + -0.021702688187360764, + 0.027941470965743065, + 0.031755704432725906, + 0.01346808485686779, + 0.012721500359475613, + 0.0003790670889429748, + -0.011398346163332462, + 0.03666394203901291, + 0.0033818050287663937, + -0.034298524260520935, + -0.011560969054698944, + 0.026906602084636688, + 0.019307704642415047, + -0.03601345419883728, + 0.021894877776503563, + -0.003143415553495288, + -0.011472265236079693, + -0.001932988059706986, + -0.0192929208278656, + 0.13400079309940338, + -0.016483990475535393, + -0.04588906094431877, + 0.003272774163633585, + -0.00902553927153349, + -0.025206459686160088, + -0.007033415604382753, + 0.0149464700371027, + 0.0056104701943695545, + 0.027335334569215775, + -0.0014793087029829621, + -0.0021214820444583893, + -0.036604806780815125, + 0.022914962843060493, + 0.012736284174025059, + 0.03196267783641815, + 0.02122960425913334, + -0.014909510500729084, + 0.019987761974334717, + -0.015508255921304226, + -0.004316883627325296, + -0.003453876357525587, + -0.005237177945673466, + -0.00013894506264477968, + -0.007358659990131855, + -0.037698812782764435, + -0.023624587804079056, + 0.024718593806028366, + 0.038497138768434525, + -0.008633767254650593, + 0.015759581699967384, + 0.020076464861631393, + 0.042961861938238144, + 0.015730014070868492, + 0.007414099294692278, + -0.017563210800290108, + -0.014887334778904915, + -0.027217064052820206, + 0.023328911513090134, + -0.0080424128100276, + 0.008508103899657726, + 0.006442061625421047, + 0.03716659173369408, + -0.02443769946694374, + 0.02072695456445217, + -0.02198358066380024, + 0.01356417965143919, + 0.011856645345687866, + 0.0027534915134310722, + 0.008582023903727531, + -0.019322488456964493, + 0.03559950366616249, + -0.003697809763252735, + -0.004035990219563246, + 0.019425975158810616, + 0.01530128251761198, + 0.003289405955001712, + 0.024452483281493187, + 0.03026253543794155, + 0.0037329215556383133, + 0.022027932107448578, + -0.008271562866866589, + 0.01447338704019785, + 0.002790451282635331, + 0.04367148503661156, + 0.012063619680702686, + 0.005924626719206572, + 0.05546899512410164, + 0.014392075128853321, + -0.01167184766381979, + 0.007901966571807861, + 0.0023192160297185183, + 0.010304342024028301, + 0.00896640308201313, + -0.007550850044935942, + -0.00612051272764802, + -0.00669708289206028, + 0.008811173029243946, + -0.021998364478349686, + -0.020712170749902725, + 0.0015541519969701767, + -0.004826926160603762, + 0.020534764975309372, + 0.013519828207790852, + -0.003082432085648179, + 0.022027932107448578, + -0.01282498799264431, + -0.01698664017021656, + -0.0024984702467918396, + -0.020593900233507156, + 0.012285376898944378, + -0.002487382385879755, + 0.01638050377368927, + -0.016587477177381516, + 0.007099942769855261, + 0.012980218045413494, + -0.03627956286072731, + 0.030129481106996536, + 0.011376170441508293, + 0.01002344861626625, + -0.019278137013316154, + -0.00811633188277483, + 0.017075343057513237, + -0.02633003145456314, + -0.02837020345032215, + -0.01051870733499527, + 0.02708400785923004, + -0.008618983440101147, + -0.035806477069854736, + -0.05336968973278999, + 0.006068769376724958, + -0.005580902565270662, + -0.021480930969119072, + -0.0008976567187346518, + 0.019884275272488594, + -0.01676488295197487, + 0.007136902306228876, + -0.00917337741702795, + 0.016483990475535393, + 0.01237408071756363, + 0.017341453582048416, + 0.03426895663142204, + 0.009868218563497066, + -0.031607866287231445, + 0.023136721923947334, + 0.0020808265544474125, + 0.006275743246078491, + 0.030779970809817314, + 0.030750403180718422, + -0.0034797480329871178, + -0.0382014624774456, + -0.011834469623863697, + 0.009801690466701984, + 0.002282256493344903, + -0.0023672636598348618, + 0.01785888709127903, + -0.007232997566461563, + -0.021022630855441093, + 0.022264475002884865, + -0.010230422019958496, + -0.02239752933382988, + -0.030513860285282135, + 0.007643249351531267, + 0.018642431125044823, + 0.004132085479795933, + 0.018775485455989838, + 0.0157004464417696, + 0.008781605400145054, + -0.002657396486029029, + -0.021273955702781677, + -0.023314127698540688, + -0.019573813304305077, + 0.03314538672566414, + -0.022648854181170464, + 0.026344815269112587, + 0.02599000371992588, + -0.008862916380167007, + 0.00997170526534319, + 0.02829628437757492, + -0.0008098776452243328, + -0.02038692496716976, + -0.002106698229908943, + -0.01339416578412056, + -0.02569432742893696, + 0.023698506876826286, + 0.017090126872062683, + -0.000379529083147645, + 0.01907116360962391, + -0.003585082944482565, + -0.0015319761587306857, + 0.015360417775809765, + -0.031075647100806236, + -0.0008939607650972903, + 0.005713956896215677, + 0.021702688187360764, + 0.006527068559080362, + -0.0036793299950659275, + -0.002404223196208477, + 0.01997297815978527, + -0.002668484579771757, + -0.029523342847824097, + -0.005044987890869379, + 0.008064588531851768, + 0.015286498703062534, + -0.0366935096681118, + 0.02140701189637184, + -0.009986489079892635, + -0.021007847040891647, + -0.013549395836889744, + -0.01997297815978527, + -0.012935866601765156, + -0.0002760421484708786, + 0.04665782302618027, + -0.008929443545639515, + 0.008131115697324276, + 0.01108788512647152, + -0.0172083992511034, + -0.015330850146710873, + 0.0049710688181221485, + 0.009905178099870682, + -0.01960338093340397, + -0.002709140069782734, + -0.0013434821739792824, + 0.04041903838515282, + 0.044055864214897156, + -0.017430156469345093, + 0.011716199107468128, + 0.012403648346662521, + 0.008205035701394081, + -0.005928322672843933, + 0.012085795402526855, + -0.009446878917515278, + -0.02489599958062172, + -0.020904360339045525, + 0.047870099544525146, + 0.031341757625341415, + -0.00036359025398269296, + 0.046214308589696884, + 0.02822236530482769, + 0.01079960074275732, + 0.001308370498009026, + -0.020372141152620316, + -0.008914659731090069, + -0.02613784186542034, + -0.001027477439492941, + 0.007343876175582409, + -0.011405738070607185, + -0.01405204739421606, + 0.0010635129874572158, + 0.03332279250025749, + 0.030366022139787674, + -0.014295980334281921, + 0.010843952186405659, + 0.020401708781719208, + -0.01289890706539154, + 0.008271562866866589, + -0.049998972564935684, + 0.009010755456984043, + -0.019928626716136932, + -0.001308370498009026, + -0.004291011951863766, + -0.025590840727090836, + -0.018435457721352577, + -0.025487352162599564, + -0.015449120663106441, + 0.028384987264871597, + 0.06292005628347397, + -0.02190966159105301, + 0.014007695019245148, + 0.02474816143512726, + 0.031075647100806236, + 0.01982514001429081, + -0.0035721471067517996, + -0.014236845076084137, + -0.016070041805505753, + -0.030336454510688782, + -0.009528189897537231, + -0.006767306011170149, + 0.01502038910984993, + 0.02130352333188057, + -0.017888454720377922, + 0.016513558104634285, + 0.031016511842608452, + -0.009705595672130585, + 0.011989700607955456, + -0.01051870733499527, + 0.0005530082853510976, + 0.029301585629582405, + -0.05011724308133125, + 0.016587477177381516, + -0.0072847409173846245, + -0.0028495865408331156, + -0.02999642677605152, + -0.008611591532826424, + 0.015153443440794945, + 0.020874792709946632, + 0.00016527879051864147, + -0.005410888232290745, + 0.0022804085165262222, + -0.021968796849250793, + -0.015936987474560738, + 0.026093490421772003, + -0.0221314188092947, + 0.021200036630034447, + 0.0035573632922023535, + 0.002790451282635331, + 0.019869491457939148, + 0.02122960425913334, + -0.009328607469797134, + -0.03400284796953201, + 0.011376170441508293, + 0.020519981160759926, + -0.007724560331553221, + 0.015759581699967384, + 0.022087067365646362, + 0.031164349988102913, + -0.009786906652152538, + -0.020268654450774193, + -0.02227925881743431, + -7.975192420417443e-05, + -0.0004518313508015126, + 0.011738374829292297, + 0.044588085263967514, + -0.004930413328111172, + 0.007768911775201559, + 0.03559950366616249, + -0.008471144363284111, + 0.029035476967692375, + -0.007332788314670324, + -0.0065492442809045315, + -0.024112455546855927, + -0.0174005888402462, + -0.004656911827623844, + -0.002694356255233288, + -0.027438821271061897, + 0.022042715921998024, + -0.005968978628516197, + 0.05328098684549332, + 0.004571904893964529, + 0.03790578618645668, + 0.035806477069854736, + 0.006105728913098574, + 0.003136023646220565, + -0.018139781430363655, + 0.025590840727090836, + -0.001859992858953774, + 0.004634736105799675, + 0.005536550655961037, + 0.047308310866355896, + -0.010858736000955105, + -0.012322336435317993, + 0.007247781381011009, + 0.007613681256771088, + 0.0026370687410235405, + 0.015153443440794945, + -0.015168227255344391, + 0.00824938714504242, + -0.02035735733807087, + 0.03205138072371483, + -0.03157829865813255, + -0.015345633961260319, + -0.015256930142641068, + -0.002542821690440178, + 0.012773244641721249, + -0.015596958808600903, + 0.008752037771046162, + -0.0035277956631034613, + -0.017681481316685677, + 0.014429034665226936, + 0.050797298550605774, + -0.017622346058487892, + 0.001332394196651876, + 0.007439971435815096, + 0.001193795702420175, + -0.0047345273196697235, + 0.005455239675939083, + 0.02451161853969097, + -0.04970329627394676, + 0.008012845180928707, + -0.0057693966664373875, + 0.007366051897406578, + -0.04582992568612099, + -0.03734399750828743, + 0.02424550987780094, + 0.00841940101236105, + 0.06132340058684349, + -0.024038536474108696, + -0.003544427454471588, + -0.01728231832385063, + 0.0061796484515070915, + -0.008271562866866589, + -0.019322488456964493, + 2.53086764132604e-05, + -0.05845533311367035, + 0.00972037948668003, + -0.021540066227316856, + 0.032554030418395996, + 0.006412493996322155, + -0.0009674180182628334, + 0.00830113049596548, + 0.012603229843080044, + -0.034298524260520935, + -0.015803933143615723, + -7.484322850359604e-05, + 0.011812293902039528, + -0.002601957181468606, + -0.012980218045413494, + -0.01907116360962391, + -0.006017026025801897, + ], + "index": 0, + } + ], + "model": "text-embedding-3-small", + "usage": {"prompt_tokens": 1, "total_tokens": 1}, + }, +} + + +@pytest.fixture +def sap_api_response(): + return moke_response + + +@pytest.fixture +def fake_token_creator(): + return lambda: "Bearer FAKE_TOKEN", "https://api.ai.moke-sap.com", "fake-group" + + +@pytest.fixture +def fake_deployment_url(): + return "https://api.ai.moke-sap.com/v2/inference/deployments/mokeid" + + +@pytest.mark.parametrize("sync_mode", [True, False]) +@pytest.mark.asyncio +async def test_sap_chat( + respx_mock, + sap_api_response, + fake_token_creator, + fake_deployment_url, + sync_mode, +): + import litellm + + litellm.disable_aiohttp_transport = True + with patch( + "litellm.llms.sap.embed.transformation.GenAIHubEmbeddingConfig.deployment_url", + new_callable=PropertyMock, + return_value=fake_deployment_url, + ), patch( + "litellm.llms.sap.embed.transformation.get_token_creator", + return_value=fake_token_creator, + ): + model = "sap/text-embedding-3-small" + input = "Hi" + respx_mock.post(f"{fake_deployment_url}/v2/embeddings").respond( + json=sap_api_response + ) + + if sync_mode: + response = litellm.embedding(model=model, input=input) + else: + response = await litellm.aembedding(model=model, input=input) + + assert response + assert response.data[0]["embedding"] From ee0812a2975c98a46433d1d1d153e5d6b8d6c802 Mon Sep 17 00:00:00 2001 From: _juliettech Date: Mon, 8 Dec 2025 15:34:11 -0500 Subject: [PATCH 04/29] Add Helicone as a provider and update observability documentation (#17663) * Add Helicone as a provider to liteLLM * Add Helicone provider integration --- .../observability/helicone_integration.md | 436 +++++++++--------- docs/my-website/docs/providers/helicone.md | 268 +++++++++++ docs/my-website/sidebars.js | 3 +- litellm/constants.py | 3 + litellm/llms/openai_like/providers.json | 4 + litellm/types/utils.py | 1 + tests/llm_translation/test_helicone.py | 72 +++ 7 files changed, 564 insertions(+), 223 deletions(-) create mode 100644 docs/my-website/docs/providers/helicone.md create mode 100644 tests/llm_translation/test_helicone.py diff --git a/docs/my-website/docs/observability/helicone_integration.md b/docs/my-website/docs/observability/helicone_integration.md index 22ea051f7cd..92d0f5c3ebf 100644 --- a/docs/my-website/docs/observability/helicone_integration.md +++ b/docs/my-website/docs/observability/helicone_integration.md @@ -10,7 +10,7 @@ https://github.com/BerriAI/litellm ::: -[Helicone](https://helicone.ai/) is an open source observability platform that proxies your LLM requests and provides key insights into your usage, spend, latency and more. +[Helicone](https://helicone.ai/) is an open sourced observability platform providing key insights into your usage, spend, latency and more. ## Quick Start @@ -25,14 +25,10 @@ from litellm import completion ## Set env variables os.environ["HELICONE_API_KEY"] = "your-helicone-key" -os.environ["OPENAI_API_KEY"] = "your-openai-key" - -# Set callbacks -litellm.success_callback = ["helicone"] # OpenAI call response = completion( - model="gpt-4o", + model="helicone/gpt-4o-mini", messages=[{"role": "user", "content": "Hi ๐Ÿ‘‹ - I'm OpenAI"}], ) @@ -54,7 +50,7 @@ model_list: # Add Helicone callback litellm_settings: success_callback: ["helicone"] - + # Set Helicone API key environment_variables: HELICONE_API_KEY: "your-helicone-key" @@ -72,12 +68,12 @@ litellm --config config.yaml There are two main approaches to integrate Helicone with LiteLLM: -1. **Callbacks**: Log to Helicone while using any provider -2. **Proxy Mode**: Use Helicone as a proxy for advanced features +1. **As a Provider**: Use Helicone to log requests for [all models supported ](../providers/helicone) +2. **Callbacks**: Log to Helicone while using any provider ### Supported LLM Providers -Helicone can log requests across [various LLM providers](https://docs.helicone.ai/getting-started/quick-start), including: +Helicone can log requests across [all major LLM providers](https://helicone.ai/models), including: - OpenAI - Azure @@ -88,156 +84,149 @@ Helicone can log requests across [various LLM providers](https://docs.helicone.a - Replicate - And more -## Method 1: Using Callbacks +## Method 1: Using Helicone as a Provider + +Helicone's AI Gateway provides [advanced functionality](https://docs.helicone.ai) like caching, rate limiting, LLM security, and more. + + + + + Set Helicone as your base URL and pass authentication headers: + + ```python + import os + import litellm + from litellm import completion + + os.environ["HELICONE_API_KEY"] = "" # your Helicone API key + + messages = [{"content": "What is the capital of France?", "role": "user"}] + + # Helicone call - routes through Helicone gateway to any model + response = completion( + model="helicone/gpt-4o-mini", # or any 100+ models + messages=messages + ) + + print(response) + ``` + + ### Advanced Usage + + You can add custom metadata and properties to your requests using Helicone headers. Here are some examples: + + ```python + litellm.metadata = { + "Helicone-User-Id": "user-abc", # Specify the user making the request + "Helicone-Property-App": "web", # Custom property to add additional information + "Helicone-Property-Custom": "any-value", # Add any custom property + "Helicone-Prompt-Id": "prompt-supreme-court", # Assign an ID to associate this prompt with future versions + "Helicone-Cache-Enabled": "true", # Enable caching of responses + "Cache-Control": "max-age=3600", # Set cache limit to 1 hour + "Helicone-RateLimit-Policy": "10;w=60;s=user", # Set rate limit policy + "Helicone-Retry-Enabled": "true", # Enable retry mechanism + "helicone-retry-num": "3", # Set number of retries + "helicone-retry-factor": "2", # Set exponential backoff factor + "Helicone-Model-Override": "gpt-3.5-turbo-0613", # Override the model used for cost calculation + "Helicone-Session-Id": "session-abc-123", # Set session ID for tracking + "Helicone-Session-Path": "parent-trace/child-trace", # Set session path for hierarchical tracking + "Helicone-Omit-Response": "false", # Include response in logging (default behavior) + "Helicone-Omit-Request": "false", # Include request in logging (default behavior) + "Helicone-LLM-Security-Enabled": "true", # Enable LLM security features + "Helicone-Moderations-Enabled": "true", # Enable content moderation + } + ``` + + ### Caching and Rate Limiting + + Enable caching and set up rate limiting policies: + + ```python + litellm.metadata = { + "Helicone-Cache-Enabled": "true", # Enable caching of responses + "Cache-Control": "max-age=3600", # Set cache limit to 1 hour + "Helicone-RateLimit-Policy": "100;w=3600;s=user", # Set rate limit policy + } + ``` + + + + +## Method 2: Using Callbacks Log requests to Helicone while using any LLM provider directly. - + -```python -import os -import litellm -from litellm import completion + ```python + import os + import litellm + from litellm import completion -## Set env variables -os.environ["HELICONE_API_KEY"] = "your-helicone-key" -os.environ["OPENAI_API_KEY"] = "your-openai-key" -# os.environ["HELICONE_API_BASE"] = "" # [OPTIONAL] defaults to `https://api.helicone.ai` + ## Set env variables + os.environ["HELICONE_API_KEY"] = "your-helicone-key" + os.environ["OPENAI_API_KEY"] = "your-openai-key" + # os.environ["HELICONE_API_BASE"] = "" # [OPTIONAL] defaults to `https://api.helicone.ai` -# Set callbacks -litellm.success_callback = ["helicone"] + # Set callbacks + litellm.success_callback = ["helicone"] -# OpenAI call -response = completion( - model="gpt-4o", - messages=[{"role": "user", "content": "Hi ๐Ÿ‘‹ - I'm OpenAI"}], -) + # OpenAI call + response = completion( + model="gpt-4o", + messages=[{"role": "user", "content": "Hi ๐Ÿ‘‹ - I'm OpenAI"}], + ) -print(response) -``` + print(response) + ``` - - + + -```yaml title="config.yaml" -model_list: - - model_name: gpt-4 - litellm_params: - model: gpt-4 - api_key: os.environ/OPENAI_API_KEY - - model_name: claude-3 - litellm_params: - model: anthropic/claude-3-sonnet-20240229 - api_key: os.environ/ANTHROPIC_API_KEY + ```yaml title="config.yaml" + model_list: + - model_name: gpt-4 + litellm_params: + model: gpt-4 + api_key: os.environ/OPENAI_API_KEY + - model_name: claude-3 + litellm_params: + model: anthropic/claude-3-sonnet-20240229 + api_key: os.environ/ANTHROPIC_API_KEY -# Add Helicone logging -litellm_settings: - success_callback: ["helicone"] - -# Environment variables -environment_variables: - HELICONE_API_KEY: "your-helicone-key" - OPENAI_API_KEY: "your-openai-key" - ANTHROPIC_API_KEY: "your-anthropic-key" -``` + # Add Helicone logging + litellm_settings: + success_callback: ["helicone"] -Start the proxy: -```bash -litellm --config config.yaml -``` + # Environment variables + environment_variables: + HELICONE_API_KEY: "your-helicone-key" + OPENAI_API_KEY: "your-openai-key" + ANTHROPIC_API_KEY: "your-anthropic-key" + ``` -Make requests to your proxy: -```python -import openai + Start the proxy: + ```bash + litellm --config config.yaml + ``` -client = openai.OpenAI( - api_key="anything", # proxy doesn't require real API key - base_url="http://localhost:4000" -) + Make requests to your proxy: + ```python + import openai -response = client.chat.completions.create( - model="gpt-4", # This gets logged to Helicone - messages=[{"role": "user", "content": "Hello!"}] -) -``` + client = openai.OpenAI( + api_key="anything", # proxy doesn't require real API key + base_url="http://localhost:4000" + ) - - + response = client.chat.completions.create( + model="gpt-4", # This gets logged to Helicone + messages=[{"role": "user", "content": "Hello!"}] + ) + ``` -## Method 2: Using Helicone as a Proxy - -Helicone's proxy provides [advanced functionality](https://docs.helicone.ai/getting-started/proxy-vs-async) like caching, rate limiting, LLM security through [PromptArmor](https://promptarmor.com/) and more. - - - - -Set Helicone as your base URL and pass authentication headers: - -```python -import os -import litellm -from litellm import completion - -# Configure LiteLLM to use Helicone proxy -litellm.api_base = "https://oai.hconeai.com/v1" -litellm.headers = { - "Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}", -} - -# Set your OpenAI API key -os.environ["OPENAI_API_KEY"] = "your-openai-key" - -response = completion( - model="gpt-3.5-turbo", - messages=[{"role": "user", "content": "How does a court case get to the Supreme Court?"}] -) - -print(response) -``` - -### Advanced Usage - -You can add custom metadata and properties to your requests using Helicone headers. Here are some examples: - -```python -litellm.metadata = { - "Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}", # Authenticate to send requests to Helicone API - "Helicone-User-Id": "user-abc", # Specify the user making the request - "Helicone-Property-App": "web", # Custom property to add additional information - "Helicone-Property-Custom": "any-value", # Add any custom property - "Helicone-Prompt-Id": "prompt-supreme-court", # Assign an ID to associate this prompt with future versions - "Helicone-Cache-Enabled": "true", # Enable caching of responses - "Cache-Control": "max-age=3600", # Set cache limit to 1 hour - "Helicone-RateLimit-Policy": "10;w=60;s=user", # Set rate limit policy - "Helicone-Retry-Enabled": "true", # Enable retry mechanism - "helicone-retry-num": "3", # Set number of retries - "helicone-retry-factor": "2", # Set exponential backoff factor - "Helicone-Model-Override": "gpt-3.5-turbo-0613", # Override the model used for cost calculation - "Helicone-Session-Id": "session-abc-123", # Set session ID for tracking - "Helicone-Session-Path": "parent-trace/child-trace", # Set session path for hierarchical tracking - "Helicone-Omit-Response": "false", # Include response in logging (default behavior) - "Helicone-Omit-Request": "false", # Include request in logging (default behavior) - "Helicone-LLM-Security-Enabled": "true", # Enable LLM security features - "Helicone-Moderations-Enabled": "true", # Enable content moderation - "Helicone-Fallbacks": '["gpt-3.5-turbo", "gpt-4"]', # Set fallback models -} -``` - -### Caching and Rate Limiting - -Enable caching and set up rate limiting policies: - -```python -litellm.metadata = { - "Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}", # Authenticate to send requests to Helicone API - "Helicone-Cache-Enabled": "true", # Enable caching of responses - "Cache-Control": "max-age=3600", # Set cache limit to 1 hour - "Helicone-RateLimit-Policy": "100;w=3600;s=user", # Set rate limit policy -} -``` - - + ## Session Tracking and Tracing @@ -245,57 +234,62 @@ litellm.metadata = { Track multi-step and agentic LLM interactions using session IDs and paths: - + -```python -import litellm + ```python + import os + import litellm + from litellm import completion -litellm.api_base = "https://oai.hconeai.com/v1" -litellm.metadata = { - "Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}", - "Helicone-Session-Id": "session-abc-123", - "Helicone-Session-Path": "parent-trace/child-trace", -} + os.environ["HELICONE_API_KEY"] = "" # your Helicone API key -response = litellm.completion( - model="gpt-3.5-turbo", - messages=[{"role": "user", "content": "Start a conversation"}] -) -``` + messages = [{"content": "What is the capital of France?", "role": "user"}] - - + response = completion( + model="helicone/gpt-4", + messages=messages, + metadata={ + "Helicone-Session-Id": "session-abc-123", + "Helicone-Session-Path": "parent-trace/child-trace", + } + ) -```python -import openai + print(response) + ``` -client = openai.OpenAI( - api_key="anything", - base_url="http://localhost:4000" -) + + -# First request in session -response1 = client.chat.completions.create( - model="gpt-4", - messages=[{"role": "user", "content": "Hello"}], - extra_headers={ - "Helicone-Session-Id": "session-abc-123", - "Helicone-Session-Path": "conversation/greeting" - } -) + ```python + import openai -# Follow-up request in same session -response2 = client.chat.completions.create( - model="gpt-4", - messages=[{"role": "user", "content": "Tell me more"}], - extra_headers={ - "Helicone-Session-Id": "session-abc-123", - "Helicone-Session-Path": "conversation/follow-up" - } -) -``` + client = openai.OpenAI( + api_key="anything", + base_url="http://localhost:4000" + ) - + # First request in session + response1 = client.chat.completions.create( + model="gpt-4", + messages=[{"role": "user", "content": "Hello"}], + extra_headers={ + "Helicone-Session-Id": "session-abc-123", + "Helicone-Session-Path": "conversation/greeting" + } + ) + + # Follow-up request in same session + response2 = client.chat.completions.create( + model="gpt-4", + messages=[{"role": "user", "content": "Tell me more"}], + extra_headers={ + "Helicone-Session-Id": "session-abc-123", + "Helicone-Session-Path": "conversation/follow-up" + } + ) + ``` + + - `Helicone-Session-Id`: Unique identifier for the session to group related requests @@ -304,52 +298,50 @@ response2 = client.chat.completions.create( ## Retry and Fallback Mechanisms - + -```python -import litellm + ```python + import litellm -litellm.api_base = "https://oai.hconeai.com/v1" -litellm.metadata = { - "Helicone-Auth": f"Bearer {os.getenv('HELICONE_API_KEY')}", - "Helicone-Retry-Enabled": "true", - "helicone-retry-num": "3", - "helicone-retry-factor": "2", # Exponential backoff - "Helicone-Fallbacks": '["gpt-3.5-turbo", "gpt-4"]', -} + litellm.api_base = "https://ai-gateway.helicone.ai/" + litellm.metadata = { + "Helicone-Retry-Enabled": "true", + "helicone-retry-num": "3", + "helicone-retry-factor": "2", + } -response = litellm.completion( - model="gpt-4", - messages=[{"role": "user", "content": "Hello"}] -) -``` + response = litellm.completion( + model="helicone/gpt-4o-mini/openai,claude-3-5-sonnet-20241022/anthropic", # Try OpenAI first, then fallback to Anthropic, then continue with other models + messages=[{"role": "user", "content": "Hello"}] + ) + ``` - - + + -```yaml title="config.yaml" -model_list: - - model_name: gpt-4 - litellm_params: - model: gpt-4 - api_key: os.environ/OPENAI_API_KEY - api_base: "https://oai.hconeai.com/v1" + ```yaml title="config.yaml" + model_list: + - model_name: gpt-4 + litellm_params: + model: gpt-4 + api_key: os.environ/OPENAI_API_KEY + api_base: "https://oai.hconeai.com/v1" -default_litellm_params: - headers: - Helicone-Auth: "Bearer ${HELICONE_API_KEY}" - Helicone-Retry-Enabled: "true" - helicone-retry-num: "3" - helicone-retry-factor: "2" - Helicone-Fallbacks: '["gpt-3.5-turbo", "gpt-4"]' + default_litellm_params: + headers: + Helicone-Auth: "Bearer ${HELICONE_API_KEY}" + Helicone-Retry-Enabled: "true" + helicone-retry-num: "3" + helicone-retry-factor: "2" + Helicone-Fallbacks: '["gpt-3.5-turbo", "gpt-4"]' -environment_variables: - HELICONE_API_KEY: "your-helicone-key" - OPENAI_API_KEY: "your-openai-key" -``` + environment_variables: + HELICONE_API_KEY: "your-helicone-key" + OPENAI_API_KEY: "your-openai-key" + ``` - + -> **Supported Headers** - For a full list of supported Helicone headers and their descriptions, please refer to the [Helicone documentation](https://docs.helicone.ai/getting-started/quick-start). +> **Supported Headers** - For a full list of supported Helicone headers and their descriptions, please refer to the [Helicone documentation](https://docs.helicone.ai/features/advanced-usage/custom-properties). > By utilizing these headers and metadata options, you can gain deeper insights into your LLM usage, optimize performance, and better manage your AI workflows with Helicone and LiteLLM. diff --git a/docs/my-website/docs/providers/helicone.md b/docs/my-website/docs/providers/helicone.md new file mode 100644 index 00000000000..3f0cfcbcb28 --- /dev/null +++ b/docs/my-website/docs/providers/helicone.md @@ -0,0 +1,268 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Helicone + +## Overview + +| Property | Details | +|-------|-------| +| Description | Helicone is an AI gateway and observability platform that provides OpenAI-compatible endpoints with advanced monitoring, caching, and analytics capabilities. | +| Provider Route on LiteLLM | `helicone/` | +| Link to Provider Doc | [Helicone Documentation โ†—](https://docs.helicone.ai) | +| Base URL | `https://ai-gateway.helicone.ai/` | +| Supported Operations | [`/chat/completions`](#sample-usage), [`/completions`](#text-completion), [`/embeddings`](#embeddings) | + +
+ +**We support [ALL models available](https://helicone.ai/models) through Helicone's AI Gateway. Use `helicone/` as a prefix when sending requests.** + +## What is Helicone? + +Helicone is an open-source observability platform for LLM applications that provides: +- **Request Monitoring**: Track all LLM requests with detailed metrics +- **Caching**: Reduce costs and latency with intelligent caching +- **Rate Limiting**: Control request rates per user/key +- **Cost Tracking**: Monitor spend across models and users +- **Custom Properties**: Tag requests with metadata for filtering and analysis +- **Prompt Management**: Version control for prompts + +## Required Variables + +```python showLineNumbers title="Environment Variables" +os.environ["HELICONE_API_KEY"] = "" # your Helicone API key +``` + +Get your Helicone API key from your [Helicone dashboard](https://helicone.ai). + +## Usage - LiteLLM Python SDK + +### Non-streaming + +```python showLineNumbers title="Helicone Non-streaming Completion" +import os +import litellm +from litellm import completion + +os.environ["HELICONE_API_KEY"] = "" # your Helicone API key + +messages = [{"content": "What is the capital of France?", "role": "user"}] + +# Helicone call - routes through Helicone gateway to OpenAI +response = completion( + model="helicone/gpt-4", + messages=messages +) + +print(response) +``` + +### Streaming + +```python showLineNumbers title="Helicone Streaming Completion" +import os +import litellm +from litellm import completion + +os.environ["HELICONE_API_KEY"] = "" # your Helicone API key + +messages = [{"content": "Write a short poem about AI", "role": "user"}] + +# Helicone call with streaming +response = completion( + model="helicone/gpt-4", + messages=messages, + stream=True +) + +for chunk in response: + print(chunk) +``` + +### With Metadata (Helicone Custom Properties) + +```python showLineNumbers title="Helicone with Custom Properties" +import os +import litellm +from litellm import completion + +os.environ["HELICONE_API_KEY"] = "" # your Helicone API key + +response = completion( + model="helicone/gpt-4o-mini", + messages=[{"role": "user", "content": "What's the weather like?"}], + metadata={ + "Helicone-Property-Environment": "production", + "Helicone-Property-User-Id": "user_123", + "Helicone-Property-Session-Id": "session_abc" + } +) + +print(response) +``` + +### Text Completion + +```python showLineNumbers title="Helicone Text Completion" +import os +import litellm + +os.environ["HELICONE_API_KEY"] = "" # your Helicone API key + +response = litellm.completion( + model="helicone/gpt-4o-mini", # text completion model + prompt="Once upon a time" +) + +print(response) +``` + + +## Retry and Fallback Mechanisms + +```python +import litellm + +litellm.api_base = "https://ai-gateway.helicone.ai/" +litellm.metadata = { + "Helicone-Retry-Enabled": "true", + "helicone-retry-num": "3", + "helicone-retry-factor": "2", +} + +response = litellm.completion( + model="helicone/gpt-4o-mini/openai,claude-3-5-sonnet-20241022/anthropic", # Try OpenAI first, then fallback to Anthropic, then continue with other models, + messages=[{"role": "user", "content": "Hello"}] +) +``` + +## Supported OpenAI Parameters + +Helicone supports all standard OpenAI-compatible parameters: + +| Parameter | Type | Description | +|-----------|------|-------------| +| `messages` | array | **Required**. Array of message objects with 'role' and 'content' | +| `model` | string | **Required**. Model ID (e.g., gpt-4, claude-3-opus, etc.) | +| `stream` | boolean | Optional. Enable streaming responses | +| `temperature` | float | Optional. Sampling temperature | +| `top_p` | float | Optional. Nucleus sampling parameter | +| `max_tokens` | integer | Optional. Maximum tokens to generate | +| `frequency_penalty` | float | Optional. Penalize frequent tokens | +| `presence_penalty` | float | Optional. Penalize tokens based on presence | +| `stop` | string/array | Optional. Stop sequences | +| `n` | integer | Optional. Number of completions to generate | +| `tools` | array | Optional. List of available tools/functions | +| `tool_choice` | string/object | Optional. Control tool/function calling | +| `response_format` | object | Optional. Response format specification | +| `user` | string | Optional. User identifier | + +## Helicone-Specific Headers + +Pass these as metadata to leverage Helicone features: + +| Header | Description | +|--------|-------------| +| `Helicone-Property-*` | Custom properties for filtering (e.g., `Helicone-Property-User-Id`) | +| `Helicone-Cache-Enabled` | Enable caching for this request | +| `Helicone-User-Id` | User identifier for tracking | +| `Helicone-Session-Id` | Session identifier for grouping requests | +| `Helicone-Prompt-Id` | Prompt identifier for versioning | +| `Helicone-Rate-Limit-Policy` | Rate limiting policy name | + +Example with headers: + +```python showLineNumbers title="Helicone with Custom Headers" +import litellm + +response = litellm.completion( + model="helicone/gpt-4", + messages=[{"role": "user", "content": "Hello"}], + metadata={ + "Helicone-Cache-Enabled": "true", + "Helicone-Property-Environment": "production", + "Helicone-Property-User-Id": "user_123", + "Helicone-Session-Id": "session_abc", + "Helicone-Prompt-Id": "prompt_v1" + } +) +``` + +## Advanced Usage + +### Using with Different Providers + +Helicone acts as a gateway and supports multiple providers: + +```python showLineNumbers title="Helicone with Anthropic" +import litellm + +# Set both Helicone and Anthropic keys +os.environ["HELICONE_API_KEY"] = "your-helicone-key" + +response = litellm.completion( + model="helicone/claude-3.5-haiku/anthropic", + messages=[{"role": "user", "content": "Hello"}] +) +``` + +### Caching + +Enable caching to reduce costs and latency: + +```python showLineNumbers title="Helicone Caching" +import litellm + +response = litellm.completion( + model="helicone/gpt-4", + messages=[{"role": "user", "content": "What is 2+2?"}], + metadata={ + "Helicone-Cache-Enabled": "true" + } +) + +# Subsequent identical requests will be served from cache +response2 = litellm.completion( + model="helicone/gpt-4", + messages=[{"role": "user", "content": "What is 2+2?"}], + metadata={ + "Helicone-Cache-Enabled": "true" + } +) +``` + +## Features + +### Request Monitoring +- Track all requests with detailed metrics +- View request/response pairs +- Monitor latency and errors +- Filter by custom properties + +### Cost Tracking +- Per-model cost tracking +- Per-user cost tracking +- Cost alerts and budgets +- Historical cost analysis + +### Rate Limiting +- Per-user rate limits +- Per-API key rate limits +- Custom rate limit policies +- Automatic enforcement + +### Analytics +- Request volume trends +- Cost trends +- Latency percentiles +- Error rates + +Visit [Helicone Pricing](https://helicone.ai/pricing) for details. + +## Additional Resources + +- [Helicone Official Documentation](https://docs.helicone.ai) +- [Helicone Dashboard](https://helicone.ai) +- [Helicone GitHub](https://github.com/Helicone/helicone) +- [API Reference](https://docs.helicone.ai/rest/ai-gateway/post-v1-chat-completions) + diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 20b94963cf0..93d8a43578d 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -614,6 +614,7 @@ const sidebars = { "providers/github_copilot", "providers/gradient_ai", "providers/groq", + "providers/helicone", "providers/heroku", { type: "category", @@ -850,7 +851,7 @@ const sidebars = { "Learn how to deploy + call models from different providers on LiteLLM", slug: "/project", }, - items: [ + items: [ "projects/smolagents", "projects/mini-swe-agent", "projects/openai-agents", diff --git a/litellm/constants.py b/litellm/constants.py index 1c0f800eccd..ed5fee78462 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -345,6 +345,7 @@ LITELLM_CHAT_PROVIDERS = [ "huggingface", "together_ai", "datarobot", + "helicone", "openrouter", "cometapi", "vertex_ai", @@ -553,6 +554,7 @@ openai_compatible_endpoints: List = [ "https://api.morphllm.com/v1", "https://api.lambda.ai/v1", "https://api.hyperbolic.xyz/v1", + "https://ai-gateway.helicone.ai/", "https://ai-gateway.vercel.sh/v1", "https://api.inference.wandb.ai/v1", "https://api.clarifai.com/v2/ext/openai/v1", @@ -598,6 +600,7 @@ openai_compatible_providers: List = [ "moonshot", "publicai", "v0", + "helicone", "morph", "lambda_ai", "hyperbolic", diff --git a/litellm/llms/openai_like/providers.json b/litellm/llms/openai_like/providers.json index 3fb20b2dfc1..a6c19222619 100644 --- a/litellm/llms/openai_like/providers.json +++ b/litellm/llms/openai_like/providers.json @@ -10,5 +10,9 @@ "special_handling": { "convert_content_list_to_string": true } + }, + "helicone": { + "base_url": "https://ai-gateway.helicone.ai/", + "api_key_env": "HELICONE_API_KEY" } } diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 3dc31a0771c..9824b6db6d6 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2990,6 +2990,7 @@ class LlmProviders(str, Enum): LLAMA = "meta_llama" NSCALE = "nscale" PG_VECTOR = "pg_vector" + HELICONE = "helicone" HYPERBOLIC = "hyperbolic" RECRAFT = "recraft" FAL_AI = "fal_ai" diff --git a/tests/llm_translation/test_helicone.py b/tests/llm_translation/test_helicone.py new file mode 100644 index 00000000000..8ca2f62d2bc --- /dev/null +++ b/tests/llm_translation/test_helicone.py @@ -0,0 +1,72 @@ +import os +import sys +import pytest + +sys.path.insert( + 0, os.path.abspath("../..") +) # Adds the parent directory to the system path +import litellm + + +def test_completion_helicone(): + """Test basic completion through Helicone gateway""" + litellm._turn_on_debug() + resp = litellm.completion( + model="helicone/gpt-4o-mini", + messages=[{"role": "user", "content": "Say 'Hello from Helicone' and nothing else"}], + max_tokens=10, + ) + print(resp) + assert resp.choices[0].message.content is not None + assert len(resp.choices[0].message.content) > 0 + +def test_completion_helicone_specific_provider(): + """Test basic completion through Helicone gateway""" + litellm._turn_on_debug() + resp = litellm.completion( + model="helicone/claude-4.5-haiku/anthropic", + messages=[{"role": "user", "content": "Say 'Hello from Helicone' and nothing else"}], + max_tokens=10, + ) + print(resp) + assert resp.choices[0].message.content is not None + assert len(resp.choices[0].message.content) > 0 + + +def test_completion_helicone_streaming(): + """Test streaming completion through Helicone gateway""" + litellm._turn_on_debug() + resp = litellm.completion( + model="helicone/gpt-4o-mini", + messages=[{"role": "user", "content": "Count to 3"}], + max_tokens=20, + stream=True, + ) + + chunks = [] + for chunk in resp: + print(chunk) + if hasattr(chunk.choices[0], "delta") and hasattr(chunk.choices[0].delta, "content"): + if chunk.choices[0].delta.content: + chunks.append(chunk.choices[0].delta.content) + + full_response = "".join(chunks) + assert len(full_response) > 0 + print(f"Full response: {full_response}") + + +def test_completion_helicone_with_metadata(): + """Test Helicone with custom properties""" + litellm._turn_on_debug() + resp = litellm.completion( + model="helicone/gpt-4o-mini", + messages=[{"role": "user", "content": "Hello"}], + max_tokens=10, + metadata={ + "Helicone-Property-Environment": "test", + "Helicone-Property-Session": "test-session-123" + } + ) + print(resp) + assert resp.choices[0].message.content is not None + From 3a43042fad0164c8fcddcfca6c2e8d5a2fa2c626 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Mon, 8 Dec 2025 12:43:42 -0800 Subject: [PATCH 05/29] docs - add sap gen ai provider on LiteLLM (#17667) --- docs/my-website/docs/providers/sap.md | 121 ++++++++++++++++++ docs/my-website/sidebars.js | 1 + .../provider_create_fields.json | 18 +++ provider_endpoints_support.json | 17 +++ 4 files changed, 157 insertions(+) create mode 100644 docs/my-website/docs/providers/sap.md diff --git a/docs/my-website/docs/providers/sap.md b/docs/my-website/docs/providers/sap.md new file mode 100644 index 00000000000..a9183b9c0df --- /dev/null +++ b/docs/my-website/docs/providers/sap.md @@ -0,0 +1,121 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# SAP Generative AI Hub + +LiteLLM supports SAP Generative AI Hub's Orchestration Service. + +| Property | Details | +|-------|-------| +| Description | SAP's Generative AI Hub provides access to foundation models through the AI Core orchestration service. | +| Provider Route on LiteLLM | `sap/` | +| Supported Endpoints | `/chat/completions` | +| API Reference | [SAP AI Core Documentation](https://help.sap.com/docs/sap-ai-core) | + +## Authentication + +SAP Generative AI Hub uses service key authentication. You can provide credentials via: + +1. **Environment variable** - Set `AICORE_SERVICE_KEY` with your service key JSON +2. **Direct parameter** - Pass `api_key` with the service key JSON string + +```python showLineNumbers title="Environment Variable" +import os +os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}' +``` + +## Usage - LiteLLM Python SDK + +```python showLineNumbers title="SAP Chat Completion" +from litellm import completion +import os + +os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}' + +response = completion( + model="sap/gpt-4", + messages=[{"role": "user", "content": "Hello from LiteLLM"}] +) +print(response) +``` + +```python showLineNumbers title="SAP Chat Completion - Streaming" +from litellm import completion +import os + +os.environ["AICORE_SERVICE_KEY"] = '{"clientid": "...", "clientsecret": "...", ...}' + +response = completion( + model="sap/gpt-4", + messages=[{"role": "user", "content": "Hello from LiteLLM"}], + stream=True +) + +for chunk in response: + print(chunk.choices[0].delta.content or "", end="") +``` + +## Usage - LiteLLM Proxy + +Add to your LiteLLM Proxy config: + +```yaml showLineNumbers title="config.yaml" +model_list: + - model_name: sap-gpt4 + litellm_params: + model: sap/gpt-4 + api_key: os.environ/AICORE_SERVICE_KEY +``` + +Start the proxy: + +```bash showLineNumbers title="Start Proxy" +litellm --config config.yaml +``` + + + + +```bash showLineNumbers title="Test Request" +curl http://localhost:4000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer your-proxy-api-key" \ + -d '{ + "model": "sap-gpt4", + "messages": [{"role": "user", "content": "Hello"}] + }' +``` + + + + +```python showLineNumbers title="OpenAI SDK" +from openai import OpenAI + +client = OpenAI( + base_url="http://localhost:4000", + api_key="your-proxy-api-key" +) + +response = client.chat.completions.create( + model="sap-gpt4", + messages=[{"role": "user", "content": "Hello"}] +) +print(response.choices[0].message.content) +``` + + + + +## Supported Parameters + +| Parameter | Description | +|-----------|-------------| +| `temperature` | Controls randomness | +| `max_tokens` | Maximum tokens in response | +| `top_p` | Nucleus sampling | +| `tools` | Function calling tools | +| `tool_choice` | Tool selection behavior | +| `response_format` | Output format (json_object, json_schema) | +| `stream` | Enable streaming | + diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 93d8a43578d..583722a9393 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -668,6 +668,7 @@ const sidebars = { ] }, "providers/sambanova", + "providers/sap", "providers/snowflake", "providers/togetherai", "providers/topaz", diff --git a/litellm/proxy/public_endpoints/provider_create_fields.json b/litellm/proxy/public_endpoints/provider_create_fields.json index ddd41ca0b1d..629760a7dd2 100644 --- a/litellm/proxy/public_endpoints/provider_create_fields.json +++ b/litellm/proxy/public_endpoints/provider_create_fields.json @@ -2446,6 +2446,24 @@ ], "default_model_placeholder": "gpt-3.5-turbo" }, + { + "provider": "SAP", + "provider_display_name": "SAP Generative AI Hub", + "litellm_provider": "sap", + "credential_fields": [ + { + "key": "api_key", + "label": "SAP AI Core Service Key (JSON)", + "placeholder": null, + "tooltip": "Paste your SAP AI Core service key JSON. Contains clientid, clientsecret, and service URLs.", + "required": true, + "field_type": "textarea", + "options": null, + "default_value": null + } + ], + "default_model_placeholder": "sap/gpt-4" + }, { "provider": "Snowflake", "provider_display_name": "Snowflake", diff --git a/provider_endpoints_support.json b/provider_endpoints_support.json index a0e794ce59d..37c2ec17371 100644 --- a/provider_endpoints_support.json +++ b/provider_endpoints_support.json @@ -1554,6 +1554,23 @@ "a2a": true } }, + "sap": { + "display_name": "SAP Generative AI Hub (`sap`)", + "url": "https://docs.litellm.ai/docs/providers/sap", + "endpoints": { + "chat_completions": true, + "messages": true, + "responses": true, + "embeddings": false, + "image_generations": false, + "audio_transcriptions": false, + "audio_speech": false, + "moderations": false, + "batches": false, + "rerank": false, + "a2a": true + } + }, "snowflake": { "display_name": "Snowflake (`snowflake`)", "url": "https://docs.litellm.ai/docs/providers/snowflake", From 0c78cd7125959df5b1a2bb505f171a3d214c2ee5 Mon Sep 17 00:00:00 2001 From: Jason Nance <103449147+jason-nance@users.noreply.github.com> Date: Mon, 8 Dec 2025 15:57:49 -0500 Subject: [PATCH 06/29] Move query params to create_pass_through_route call (#17660) Fix error calling Langfuse passthrough endpoint. --- litellm/proxy/vertex_ai_endpoints/langfuse_endpoints.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/proxy/vertex_ai_endpoints/langfuse_endpoints.py b/litellm/proxy/vertex_ai_endpoints/langfuse_endpoints.py index 684e2ad0617..ce27c830f6f 100644 --- a/litellm/proxy/vertex_ai_endpoints/langfuse_endpoints.py +++ b/litellm/proxy/vertex_ai_endpoints/langfuse_endpoints.py @@ -128,12 +128,12 @@ async def langfuse_proxy_route( endpoint=endpoint, target=str(updated_url), custom_headers={"Authorization": langfuse_combined_key}, + query_params=dict(request.query_params), # type: ignore ) # dynamically construct pass-through endpoint based on incoming path received_value = await endpoint_func( request, fastapi_response, user_api_key_dict, - query_params=dict(request.query_params), # type: ignore ) return received_value From 7b47c0f583e9e3163562b260d4f7dc487af716d3 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Mon, 8 Dec 2025 12:58:21 -0800 Subject: [PATCH 07/29] docs: Explain default behavior of drop_params (#17658) Co-authored-by: Cursor Agent Co-authored-by: ishaan --- docs/my-website/docs/completion/drop_params.md | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/my-website/docs/completion/drop_params.md b/docs/my-website/docs/completion/drop_params.md index 590d9a45955..a81fd897b4e 100644 --- a/docs/my-website/docs/completion/drop_params.md +++ b/docs/my-website/docs/completion/drop_params.md @@ -5,6 +5,14 @@ import TabItem from '@theme/TabItem'; Drop unsupported OpenAI params by your LLM Provider. +## Default Behavior + +**By default, LiteLLM raises an exception** if you send a parameter to a model that doesn't support it. + +For example, if you send `temperature=0.2` to a model that doesn't support the `temperature` parameter, LiteLLM will raise an exception. + +**When `drop_params=True` is set**, LiteLLM will drop the unsupported parameter instead of raising an exception. This allows your code to work seamlessly across different providers without having to customize parameters for each one. + ## Quick Start ```python From dcf5217d1764dc423805022d8600d3d36e2c851a Mon Sep 17 00:00:00 2001 From: Cesar Garcia <128240629+Chesars@users.noreply.github.com> Date: Mon, 8 Dec 2025 18:05:50 -0300 Subject: [PATCH 08/29] docs: improve Getting Started page and SDK documentation structure (#17614) * docs: update Getting Started page with accurate endpoints and fix exception handling - Update endpoints list to include /responses, /audio, /batches - Change "Consistent output" to be endpoint-agnostic - Clarify Response Format title as "OpenAI Chat Completions Format" - Fix exception handling example: use litellm exceptions instead of deprecated openai.error - Add model prefix (anthropic/) to example * docs: reorganize sidebar and improve SDK documentation structure Sidebar changes: - Reorder: Python SDK first, then AI Gateway (Proxy) - Rename "LiteLLM - Getting Started" to "Getting Started" - Restructure SDK section with Core Functions, Configuration subsections - Move budget_manager to Guides - Move sdk_custom_pricing and migration to Extras - Remove duplicate embedding/async_embedding and embedding/moderation Content changes: - Add Response Format section to response_api.md - Add async aembedding() section to supported_embedding.md * docs: add deprecation notice for OpenAI Assistants API OpenAI has deprecated the Assistants API, shutting down on August 26, 2026. Added warning banner directing users to the Responses API. * docs: expand Core Functions in SDK sidebar Add more SDK functions to Core Functions category: - text_completion() - image_generation() - transcription() - speech() - Link to "All Supported Endpoints" for complete list * Rename Sidebar Item * docs: revert Getting Started label to original * Rename sidebar label from 'LiteLLM - Getting Started' to 'Getting Started' --- docs/my-website/docs/assistants.md | 8 ++ .../docs/embedding/supported_embedding.md | 20 ++++ docs/my-website/docs/index.md | 23 ++-- docs/my-website/docs/response_api.md | 32 ++++++ docs/my-website/sidebars.js | 100 ++++++++++++++---- 5 files changed, 152 insertions(+), 31 deletions(-) diff --git a/docs/my-website/docs/assistants.md b/docs/my-website/docs/assistants.md index d262b492a70..2960d0fded8 100644 --- a/docs/my-website/docs/assistants.md +++ b/docs/my-website/docs/assistants.md @@ -3,6 +3,14 @@ import TabItem from '@theme/TabItem'; # /assistants +:::warning Deprecation Notice + +OpenAI has deprecated the Assistants API. It will shut down on **August 26, 2026**. + +Consider migrating to the [Responses API](/docs/response_api) instead. See [OpenAI's migration guide](https://platform.openai.com/docs/guides/responses-vs-assistants) for details. + +::: + Covers Threads, Messages, Assistants. LiteLLM currently covers: diff --git a/docs/my-website/docs/embedding/supported_embedding.md b/docs/my-website/docs/embedding/supported_embedding.md index 0e8252b409b..11ca4da48a4 100644 --- a/docs/my-website/docs/embedding/supported_embedding.md +++ b/docs/my-website/docs/embedding/supported_embedding.md @@ -10,6 +10,26 @@ import os os.environ['OPENAI_API_KEY'] = "" response = embedding(model='text-embedding-ada-002', input=["good morning from litellm"]) ``` + +## Async Usage - `aembedding()` + +LiteLLM provides an asynchronous version of the `embedding` function called `aembedding`: + +```python +from litellm import aembedding +import asyncio + +async def get_embedding(): + response = await aembedding( + model='text-embedding-ada-002', + input=["good morning from litellm"] + ) + return response + +response = asyncio.run(get_embedding()) +print(response) +``` + ## Proxy Usage **NOTE** diff --git a/docs/my-website/docs/index.md b/docs/my-website/docs/index.md index 11d2963b7a3..c6e335e4cc3 100644 --- a/docs/my-website/docs/index.md +++ b/docs/my-website/docs/index.md @@ -7,8 +7,8 @@ https://github.com/BerriAI/litellm ## **Call 100+ LLMs using the OpenAI Input/Output Format** -- Translate inputs to provider's `completion`, `embedding`, and `image_generation` endpoints -- [Consistent output](https://docs.litellm.ai/docs/completion/output), text responses will always be available at `['choices'][0]['message']['content']` +- Translate inputs to provider's endpoints (`/chat/completions`, `/responses`, `/embeddings`, `/images`, `/audio`, `/batches`, and more) +- [Consistent output](https://docs.litellm.ai/docs/supported_endpoints) - same response format regardless of which provider you use - Retry/fallback logic across multiple deployments (e.g. Azure/OpenAI) - [Router](https://docs.litellm.ai/docs/routing) - Track spend & set budgets per project [LiteLLM Proxy Server](https://docs.litellm.ai/docs/simple_proxy) @@ -245,7 +245,7 @@ response = completion( -### Response Format (OpenAI Format) +### Response Format (OpenAI Chat Completions Format) ```json { @@ -514,15 +514,22 @@ response = completion( LiteLLM maps exceptions across all supported providers to the OpenAI exceptions. All our exceptions inherit from OpenAI's exception types, so any error-handling you have for that, should work out of the box with LiteLLM. ```python -from openai.error import OpenAIError +import litellm from litellm import completion +import os os.environ["ANTHROPIC_API_KEY"] = "bad-key" try: - # some code - completion(model="claude-instant-1", messages=[{"role": "user", "content": "Hey, how's it going?"}]) -except OpenAIError as e: - print(e) + completion(model="anthropic/claude-instant-1", messages=[{"role": "user", "content": "Hey, how's it going?"}]) +except litellm.AuthenticationError as e: + # Thrown when the API key is invalid + print(f"Authentication failed: {e}") +except litellm.RateLimitError as e: + # Thrown when you've exceeded your rate limit + print(f"Rate limited: {e}") +except litellm.APIError as e: + # Thrown for general API errors + print(f"API error: {e}") ``` ### See How LiteLLM Transforms Your Requests diff --git a/docs/my-website/docs/response_api.md b/docs/my-website/docs/response_api.md index 52e4c1e26f5..4e828c6c580 100644 --- a/docs/my-website/docs/response_api.md +++ b/docs/my-website/docs/response_api.md @@ -43,6 +43,38 @@ response = litellm.responses( print(response) ``` +#### Response Format (OpenAI Responses API Format) + +```json +{ + "id": "resp_abc123", + "object": "response", + "created_at": 1734366691, + "status": "completed", + "model": "o1-pro-2025-01-30", + "output": [ + { + "type": "message", + "id": "msg_abc123", + "status": "completed", + "role": "assistant", + "content": [ + { + "type": "output_text", + "text": "Once upon a time, a little unicorn named Stardust lived in a magical meadow where flowers sang lullabies. One night, she discovered that her horn could paint dreams across the sky, and she spent the evening creating the most beautiful aurora for all the forest creatures to enjoy. As the animals drifted off to sleep beneath her shimmering lights, Stardust curled up on a cloud of moonbeams, happy to have shared her magic with her friends.", + "annotations": [] + } + ] + } + ], + "usage": { + "input_tokens": 18, + "output_tokens": 98, + "total_tokens": 116 + } +} +``` + #### Streaming ```python showLineNumbers title="OpenAI Streaming Response" import litellm diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 583722a9393..1a0aa352390 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -118,11 +118,83 @@ const sidebars = { ], // But you can create a sidebar manually tutorialSidebar: [ - { type: "doc", id: "index" }, // NEW + { type: "doc", id: "index", label: "Getting Started" }, { type: "category", - label: "LiteLLM AI Gateway", + label: "LiteLLM Python SDK", + items: [ + { + type: "link", + label: "Quick Start", + href: "/docs/#litellm-python-sdk", + }, + { + type: "category", + label: "SDK Functions", + items: [ + { + type: "doc", + id: "completion/input", + label: "completion()", + }, + { + type: "doc", + id: "embedding/supported_embedding", + label: "embedding()", + }, + { + type: "doc", + id: "response_api", + label: "responses()", + }, + { + type: "doc", + id: "text_completion", + label: "text_completion()", + }, + { + type: "doc", + id: "image_generation", + label: "image_generation()", + }, + { + type: "doc", + id: "audio_transcription", + label: "transcription()", + }, + { + type: "doc", + id: "text_to_speech", + label: "speech()", + }, + { + type: "link", + label: "All Supported Endpoints โ†’", + href: "/docs/supported_endpoints", + }, + ], + }, + { + type: "category", + label: "Configuration", + items: [ + "set_keys", + "caching/all_caches", + ], + }, + "completion/token_usage", + "exception_mapping", + { + type: "category", + label: "LangChain, LlamaIndex, Instructor", + items: ["langchain/langchain", "tutorials/instructor"], + } + ], + }, + { + type: "category", + label: "LiteLLM AI Gateway (Proxy)", link: { type: "generated-index", title: "LiteLLM AI Gateway (LLM Proxy)", @@ -696,6 +768,7 @@ const sidebars = { type: "category", label: "Guides", items: [ + "budget_manager", "completion/computer_use", "completion/web_search", "completion/web_fetch", @@ -748,27 +821,6 @@ const sidebars = { "wildcard_routing" ], }, - { - type: "category", - label: "LiteLLM Python SDK", - items: [ - "set_keys", - "budget_manager", - "caching/all_caches", - "completion/token_usage", - "sdk_custom_pricing", - "embedding/async_embedding", - "embedding/moderation", - "migration", - "sdk_custom_pricing", - { - type: "category", - label: "LangChain, LlamaIndex, Instructor Integration", - items: ["langchain/langchain", "tutorials/instructor"], - } - ], - }, - { type: "category", label: "Load Testing", @@ -838,6 +890,8 @@ const sidebars = { type: "category", label: "Extras", items: [ + "sdk_custom_pricing", + "migration", "data_security", "data_retention", "proxy/security_encryption_faq", From 601da4a3d1324bfcc9cc0eed94d073493f0f26cf Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Mon, 8 Dec 2025 15:25:23 -0800 Subject: [PATCH 09/29] [Feat] New model - add nvidia nim `llama-3.2-nv-rerankqa-1b-v2` (#17670) * fix get_nvidia_nim_rerank_config * add NvidiaNimRankingConfig * add get_nvidia_nim_rerank_config * add test_nvidia_nim_rerank_ranking_endpoint * add /ranking model provider support * feat: add nvidia/llama-3.2-nv-rerankqa-1b-v2 --- .../docs/providers/nvidia_nim_rerank.md | 117 ++++++++++++++++-- litellm/__init__.py | 1 + .../llms/nvidia_nim/rerank/common_utils.py | 28 +++++ .../rerank/ranking_transformation.py | 75 +++++++++++ ...odel_prices_and_context_window_backup.json | 7 ++ litellm/utils.py | 6 +- model_prices_and_context_window.json | 7 ++ tests/llm_translation/test_nvidia_nim.py | 65 +++++++++- 8 files changed, 293 insertions(+), 13 deletions(-) create mode 100644 litellm/llms/nvidia_nim/rerank/common_utils.py create mode 100644 litellm/llms/nvidia_nim/rerank/ranking_transformation.py diff --git a/docs/my-website/docs/providers/nvidia_nim_rerank.md b/docs/my-website/docs/providers/nvidia_nim_rerank.md index 7373014a960..d28f056c24b 100644 --- a/docs/my-website/docs/providers/nvidia_nim_rerank.md +++ b/docs/my-website/docs/providers/nvidia_nim_rerank.md @@ -141,6 +141,111 @@ curl -X POST http://0.0.0.0:4000/rerank \ }' ``` +## `/v1/ranking` Models (llama-3.2-nv-rerankqa-1b-v2) + +Some Nvidia NIM rerank models use the `/v1/ranking` endpoint instead of the default `/v1/retrieval/{model}/reranking` endpoint. + +Use the `ranking/` prefix to force requests to the `/v1/ranking` endpoint: + +### LiteLLM Python SDK + +```python showLineNumbers title="Force /v1/ranking endpoint with ranking/ prefix" +import litellm +import os + +os.environ['NVIDIA_NIM_API_KEY'] = "nvapi-..." + +# Use "ranking/" prefix to force /v1/ranking endpoint +response = litellm.rerank( + model="nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2", + query="which way did the traveler go?", + documents=[ + "two roads diverged in a yellow wood...", + "then took the other, as just as fair...", + "i shall be telling this with a sigh somewhere ages and ages hence..." + ], + top_n=3, + truncate="END", # Optional: truncate long text from the end +) + +print(response) +``` + +### LiteLLM Proxy + +```yaml showLineNumbers title="config.yaml" +model_list: + - model_name: nvidia-ranking + litellm_params: + model: nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2 + api_key: os.environ/NVIDIA_NIM_API_KEY +``` + +```bash title="Request to LiteLLM Proxy" +curl -X POST http://0.0.0.0:4000/rerank \ + -H "Authorization: Bearer sk-1234" \ + -H "Content-Type: application/json" \ + -d '{ + "model": "nvidia-ranking", + "query": "which way did the traveler go?", + "documents": [ + "two roads diverged in a yellow wood...", + "then took the other, as just as fair..." + ], + "top_n": 2 + }' +``` + +### Understanding Model Resolution + +**Ranking Endpoint (`/v1/ranking`):** + +``` +model: nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2 + โ””โ”€โ”€โ”€โ”€โ”ฌโ”€โ”€โ”€โ”€โ”˜ โ””โ”€โ”€โ”ฌโ”€โ”€โ”˜ โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”ฌโ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”˜ + โ”‚ โ”‚ โ”‚ + โ”‚ โ”‚ โ””โ”€โ”€โ”€โ”€โ–ถ Model name sent to provider + โ”‚ โ”‚ + โ”‚ โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ–ถ Tells LiteLLM the request/response and url should be sent to Nvidia NIM /v1/ranking endpoint + โ”‚ + โ””โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ–ถ Provider prefix + +API URL: https://ai.api.nvidia.com/v1/ranking +``` + +**Visual Flow:** + +``` +Client Request LiteLLM Provider API +โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€ + +# Default reranking endpoint +model: "nvidia_nim/nvidia/model-name" + 1. Extracts model: nvidia/model-name + 2. Routes to default endpoint โ”€โ”€โ”€โ”€โ”€โ”€โ–ถ POST /v1/retrieval/nvidia/model-name/reranking + + +# Forced ranking endpoint +model: "nvidia_nim/ranking/nvidia/model-name" + 1. Detects "ranking/" prefix + 2. Extracts model: nvidia/model-name + 3. Routes to ranking endpoint โ”€โ”€โ”€โ”€โ”€โ”€โ–ถ POST /v1/ranking + Body: {"model": "nvidia/model-name", ...} +``` + +**When to use each endpoint:** + +| Endpoint | Model Prefix | Use Case | +|----------|--------------|----------| +| `/v1/retrieval/{model}/reranking` | `nvidia_nim/` | Default for most rerank models | +| `/v1/ranking` | `nvidia_nim/ranking/` | For models like `nvidia/llama-3.2-nv-rerankqa-1b-v2` that require this endpoint | + +:::tip + +Check the [Nvidia NIM model deployment page](https://build.nvidia.com/nvidia/llama-3_2-nv-rerankqa-1b-v2/deploy) to see which endpoint your model requires. + +::: + ## API Parameters ### Required Parameters @@ -203,16 +308,7 @@ response = litellm.rerank( -## API Endpoint - -The rerank endpoint uses a different base URL than chat/embeddings: - -- **Chat/Embeddings:** `https://integrate.api.nvidia.com/v1/` -- **Rerank:** `https://ai.api.nvidia.com/v1/` - -LiteLLM automatically uses the correct endpoint for rerank requests. - -### Custom API Base URL +## Custom API Base URL You can override the default base URL in several ways: @@ -258,4 +354,3 @@ Get your Nvidia NIM API key from [Nvidia's website](https://developer.nvidia.com - [Nvidia NIM Chat Completions](./nvidia_nim#sample-usage) - [LiteLLM Rerank Endpoint](../rerank) - [Nvidia NIM Official Docs โ†—](https://docs.api.nvidia.com/nim/reference/) - diff --git a/litellm/__init__.py b/litellm/__init__.py index d2766be03c6..34bfc778982 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1111,6 +1111,7 @@ from .llms.jina_ai.rerank.transformation import JinaAIRerankConfig from .llms.deepinfra.rerank.transformation import DeepinfraRerankConfig from .llms.hosted_vllm.rerank.transformation import HostedVLLMRerankConfig from .llms.nvidia_nim.rerank.transformation import NvidiaNimRerankConfig +from .llms.nvidia_nim.rerank.ranking_transformation import NvidiaNimRankingConfig from .llms.vertex_ai.rerank.transformation import VertexAIRerankConfig from .llms.fireworks_ai.rerank.transformation import FireworksAIRerankConfig from .llms.clarifai.chat.transformation import ClarifaiConfig diff --git a/litellm/llms/nvidia_nim/rerank/common_utils.py b/litellm/llms/nvidia_nim/rerank/common_utils.py new file mode 100644 index 00000000000..2bd8c123c90 --- /dev/null +++ b/litellm/llms/nvidia_nim/rerank/common_utils.py @@ -0,0 +1,28 @@ +""" +Common utilities for NVIDIA NIM rerank provider. +""" + + +def get_nvidia_nim_rerank_config(model: str): + """ + Get the appropriate NVIDIA NIM rerank config based on the model. + + Args: + model: The model string (e.g., "nvidia/llama-3.2-nv-rerankqa-1b-v2" or "ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2") + + Returns: + NvidiaNimRankingConfig if model starts with "ranking/", else NvidiaNimRerankConfig + + Example: + - "ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2" -> NvidiaNimRankingConfig + - "nvidia/llama-3.2-nv-rerankqa-1b-v2" -> NvidiaNimRerankConfig + """ + from litellm.llms.nvidia_nim.rerank.ranking_transformation import ( + NvidiaNimRankingConfig, + ) + from litellm.llms.nvidia_nim.rerank.transformation import NvidiaNimRerankConfig + + if model.startswith("ranking/"): + return NvidiaNimRankingConfig() + return NvidiaNimRerankConfig() + diff --git a/litellm/llms/nvidia_nim/rerank/ranking_transformation.py b/litellm/llms/nvidia_nim/rerank/ranking_transformation.py new file mode 100644 index 00000000000..72e3c039d4c --- /dev/null +++ b/litellm/llms/nvidia_nim/rerank/ranking_transformation.py @@ -0,0 +1,75 @@ +""" +Transformation for NVIDIA NIM Ranking models that use /v1/ranking endpoint. + +Use this by passing "nvidia_nim/ranking/" to force the /v1/ranking endpoint. + +Reference: https://build.nvidia.com/nvidia/llama-3_2-nv-rerankqa-1b-v2/deploy +""" + +from typing import Dict, Optional + +from litellm.llms.nvidia_nim.rerank.transformation import NvidiaNimRerankConfig + + +class NvidiaNimRankingConfig(NvidiaNimRerankConfig): + """ + Configuration for NVIDIA NIM models that use the /v1/ranking endpoint. + + Example: + curl -X "POST" 'https://ai.api.nvidia.com/v1/ranking' \ + -H 'Accept: application/json' \ + -H 'Content-Type: application/json' \ + -d '{ + "model": "nvidia/llama-3.2-nv-rerankqa-1b-v2", + "query": {"text": "which way did the traveler go?"}, + "passages": [{"text": "..."}, {"text": "..."}], + "truncate": "END" + }' + """ + + def _get_clean_model_name(self, model: str) -> str: + """Strip 'ranking/' prefix from model name.""" + if model.startswith("ranking/"): + return model[len("ranking/"):] + return model + + def get_complete_url( + self, + api_base: Optional[str], + model: str, + optional_params: Optional[dict] = None, + ) -> str: + """ + Construct the Nvidia NIM ranking URL. + + Format: {api_base}/v1/ranking + """ + if not api_base: + api_base = self.DEFAULT_NIM_RERANK_API_BASE + + api_base = api_base.rstrip("/") + + if api_base.endswith("/ranking"): + return api_base + + if api_base.endswith("/v1"): + api_base = api_base[:-3] + + return f"{api_base}/v1/ranking" + + def transform_rerank_request( + self, + model: str, + optional_rerank_params: Dict, + headers: dict, + ) -> dict: + """ + Transform request, using clean model name without 'ranking/' prefix. + """ + clean_model = self._get_clean_model_name(model) + return super().transform_rerank_request( + model=clean_model, + optional_rerank_params=optional_rerank_params, + headers=headers, + ) + diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index fde60a92370..f81bd214c5c 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -22865,6 +22865,13 @@ "mode": "rerank", "output_cost_per_token": 0.0 }, + "nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2": { + "input_cost_per_query": 0.0, + "input_cost_per_token": 0.0, + "litellm_provider": "nvidia_nim", + "mode": "rerank", + "output_cost_per_token": 0.0 + }, "sagemaker/meta-textgeneration-llama-2-13b": { "input_cost_per_token": 0.0, "litellm_provider": "sagemaker", diff --git a/litellm/utils.py b/litellm/utils.py index d77607fd3e9..00fc61b2288 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -7366,7 +7366,11 @@ class ProviderConfigManager: elif litellm.LlmProviders.DEEPINFRA == provider: return litellm.DeepinfraRerankConfig() elif litellm.LlmProviders.NVIDIA_NIM == provider: - return litellm.NvidiaNimRerankConfig() + from litellm.llms.nvidia_nim.rerank.common_utils import ( + get_nvidia_nim_rerank_config, + ) + + return get_nvidia_nim_rerank_config(model) elif litellm.LlmProviders.VERTEX_AI == provider: return litellm.VertexAIRerankConfig() elif litellm.LlmProviders.FIREWORKS_AI == provider: diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index fde60a92370..f81bd214c5c 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -22865,6 +22865,13 @@ "mode": "rerank", "output_cost_per_token": 0.0 }, + "nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2": { + "input_cost_per_query": 0.0, + "input_cost_per_token": 0.0, + "litellm_provider": "nvidia_nim", + "mode": "rerank", + "output_cost_per_token": 0.0 + }, "sagemaker/meta-textgeneration-llama-2-13b": { "input_cost_per_token": 0.0, "litellm_provider": "sagemaker", diff --git a/tests/llm_translation/test_nvidia_nim.py b/tests/llm_translation/test_nvidia_nim.py index 1705871258f..d0462efa6d5 100644 --- a/tests/llm_translation/test_nvidia_nim.py +++ b/tests/llm_translation/test_nvidia_nim.py @@ -184,13 +184,76 @@ def test_chat_completion_nvidia_nim_with_tools(): assert request_body["tool_choice"] == "auto" assert request_body["parallel_tool_calls"] == True +@pytest.mark.asyncio() +async def test_nvidia_nim_rerank_ranking_endpoint(): + """ + Test that using "nvidia_nim/ranking/" forces the /v1/ranking endpoint. + + This allows users to explicitly use the /v1/ranking endpoint for models like + nvidia/llama-3.2-nv-rerankqa-1b-v2. + + Reference: https://build.nvidia.com/nvidia/llama-3_2-nv-rerankqa-1b-v2/deploy + """ + mock_response = AsyncMock() + + def return_val(): + return { + "rankings": [ + {"index": 0, "logit": 0.95}, + {"index": 1, "logit": 0.75}, + ], + } + + mock_response.json = return_val + mock_response.headers = {"key": "value"} + mock_response.status_code = 200 + + with patch( + "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", + return_value=mock_response, + ) as mock_post: + # Use "ranking/" prefix to force /v1/ranking endpoint + response = await litellm.arerank( + model="nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2", + query="What is the GPU memory bandwidth?", + documents=["H100 delivers 3TB/s memory bandwidth", "A100 has 2TB/s memory bandwidth"], + top_n=2, + api_key="fake-api-key", + ) + + mock_post.assert_called_once() + + args_to_api = mock_post.call_args.kwargs["data"] + _url = mock_post.call_args.kwargs["url"] + print("url = ", _url) + + # Verify URL is /v1/ranking + assert _url == "https://ai.api.nvidia.com/v1/ranking" + + # Verify request body structure + request_data = json.loads(args_to_api) + print("request_data=", request_data) + + # Query should be an object with 'text' field + assert request_data["query"] == {"text": "What is the GPU memory bandwidth?"} + + # Documents should be 'passages' + assert request_data["passages"] == [ + {"text": "H100 delivers 3TB/s memory bandwidth"}, + {"text": "A100 has 2TB/s memory bandwidth"}, + ] + + # Model name in body should NOT have "ranking/" prefix + assert request_data["model"] == "nvidia/llama-3.2-nv-rerankqa-1b-v2" + + class TestNvidiaNim(BaseLLMRerankTest): def get_custom_llm_provider(self) -> litellm.LlmProviders: return litellm.LlmProviders.NVIDIA_NIM def get_base_rerank_call_args(self) -> dict: return { - "model": "nvidia_nim/nvidia/llama-3_2-nv-rerankqa-1b-v2", + "model": "nvidia_nim/nvidia/llama-3.2-nv-rerankqa-1b-v2", } def get_expected_cost(self) -> float: From fbe18a21c9472d77c7a8bc92ec6017cb070d2a4c Mon Sep 17 00:00:00 2001 From: Krish Dholakia Date: Mon, 8 Dec 2025 16:29:15 -0800 Subject: [PATCH 10/29] Docs: Add integration documentation instructions (#17644) Co-authored-by: Cursor Agent --- .../contribute_integration/custom_webhook_api.md | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/docs/my-website/docs/contribute_integration/custom_webhook_api.md b/docs/my-website/docs/contribute_integration/custom_webhook_api.md index 499c7fd51de..158937d2a43 100644 --- a/docs/my-website/docs/contribute_integration/custom_webhook_api.md +++ b/docs/my-website/docs/contribute_integration/custom_webhook_api.md @@ -95,11 +95,19 @@ curl -L -X POST 'http://0.0.0.0:4000/chat/completions' \ }' ``` -4. File a PR! +4. Add Documentation + +If you're adding a new integration, please add documentation for it under the `observability` folder: + +- Create a new file at `docs/my-website/docs/observability/_integration.md` +- Follow the format of existing integration docs, such as [Langsmith Integration](https://github.com/BerriAI/litellm/blob/main/docs/my-website/docs/observability/langsmith_integration.md) +- Include: Quick Start, SDK usage, Proxy usage, and any advanced configuration options + +5. File a PR! - Review our contribution guide [here](../../extras/contributing_code) -- push your fork to your GitHub repo -- submit a PR from there +- Push your fork to your GitHub repo +- Submit a PR from there ## What get's logged? From 8338bd9c539aa48d2a1bb20e593b277a46743074 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Mon, 8 Dec 2025 16:30:37 -0800 Subject: [PATCH 11/29] Change deprecation banner to only show on /sso/key/generate --- .../proxy/common_utils/html_forms/ui_login.py | 26 ++++++++++--- litellm/proxy/proxy_server.py | 10 +++-- tests/test_litellm/proxy/test_proxy_server.py | 38 +++++++++++++++++++ 3 files changed, 65 insertions(+), 9 deletions(-) diff --git a/litellm/proxy/common_utils/html_forms/ui_login.py b/litellm/proxy/common_utils/html_forms/ui_login.py index 8478d41e475..42cfb592a78 100644 --- a/litellm/proxy/common_utils/html_forms/ui_login.py +++ b/litellm/proxy/common_utils/html_forms/ui_login.py @@ -8,7 +8,22 @@ if server_root_path != "": url_to_redirect_to += server_root_path url_to_redirect_to += "/login" new_ui_login_url = get_custom_url("", "ui/login") -html_form = f""" + + +def build_ui_login_form(show_deprecation_banner: bool = False) -> str: + banner_html = ( + f""" +
+ Deprecated: Logging in with username and password on this page is deprecated. + Please use the new login page instead. + This page will be dedicated to signing in via SSO in the future. +
+ """ + if show_deprecation_banner + else "" + ) + + return f""" @@ -209,11 +224,7 @@ html_form = f"""
-
- Deprecated: Logging in with username and password on this page is deprecated. - Please use the new login page instead. - This page will be dedicated to signing in via SSO in the future. -
+ {banner_html}