From 615f76de888244753ed27e9d0bedc5125357b8da Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Tue, 4 Nov 2025 14:21:26 -0800 Subject: [PATCH] [Feature] UI - Litellm test key audio (#16251) * Add audio/transcriptions and audio/speech to Test key UI * Removed unused import in test * Fixed voice parameter not being passed in --- .../components/chat_ui/AudioRenderer.test.tsx | 13 ++ .../src/components/chat_ui/AudioRenderer.tsx | 23 +++ .../src/components/chat_ui/ChatUI.test.tsx | 59 +++++- .../src/components/chat_ui/ChatUI.tsx | 175 ++++++++++++++++-- .../components/chat_ui/CodeSnippets.test.tsx | 1 + .../src/components/chat_ui/CodeSnippets.tsx | 41 ++++ .../chat_ui/EndpointSelector.test.tsx | 5 +- .../components/chat_ui/EndpointSelector.tsx | 16 +- .../src/components/chat_ui/chatConstants.ts | 45 +++++ .../chat_ui/llm_calls/audio_speech.test.tsx | 87 +++++++++ .../chat_ui/llm_calls/audio_speech.tsx | 57 ++++++ .../llm_calls/audio_transcriptions.test.tsx | 107 +++++++++++ .../llm_calls/audio_transcriptions.tsx | 74 ++++++++ .../chat_ui/mode_endpoint_mapping.tsx | 6 + .../src/components/chat_ui/types.ts | 1 + 15 files changed, 676 insertions(+), 34 deletions(-) create mode 100644 ui/litellm-dashboard/src/components/chat_ui/AudioRenderer.test.tsx create mode 100644 ui/litellm-dashboard/src/components/chat_ui/AudioRenderer.tsx create mode 100644 ui/litellm-dashboard/src/components/chat_ui/chatConstants.ts create mode 100644 ui/litellm-dashboard/src/components/chat_ui/llm_calls/audio_speech.test.tsx create mode 100644 ui/litellm-dashboard/src/components/chat_ui/llm_calls/audio_speech.tsx create mode 100644 ui/litellm-dashboard/src/components/chat_ui/llm_calls/audio_transcriptions.test.tsx create mode 100644 ui/litellm-dashboard/src/components/chat_ui/llm_calls/audio_transcriptions.tsx diff --git a/ui/litellm-dashboard/src/components/chat_ui/AudioRenderer.test.tsx b/ui/litellm-dashboard/src/components/chat_ui/AudioRenderer.test.tsx new file mode 100644 index 00000000000..0b10e5533b4 --- /dev/null +++ b/ui/litellm-dashboard/src/components/chat_ui/AudioRenderer.test.tsx @@ -0,0 +1,13 @@ +import { render } from "@testing-library/react"; +import { describe, it, expect } from "vitest"; +import AudioRenderer from "./AudioRenderer"; +import { MessageType } from "./types"; + +describe("AudioRenderer", () => { + it("should render the audio renderer", () => { + const { container } = render( + , + ); + expect(container).toBeTruthy(); + }); +}); diff --git a/ui/litellm-dashboard/src/components/chat_ui/AudioRenderer.tsx b/ui/litellm-dashboard/src/components/chat_ui/AudioRenderer.tsx new file mode 100644 index 00000000000..48a766283e6 --- /dev/null +++ b/ui/litellm-dashboard/src/components/chat_ui/AudioRenderer.tsx @@ -0,0 +1,23 @@ +import React from "react"; +import { MessageType } from "./types"; + +interface AudioRendererProps { + message: MessageType; +} + +const AudioRenderer: React.FC = ({ message }) => { + // Check if this message contains audio + if (!message.isAudio || typeof message.content !== "string") { + return null; + } + + return ( +
+ +
+ ); +}; + +export default AudioRenderer; diff --git a/ui/litellm-dashboard/src/components/chat_ui/ChatUI.test.tsx b/ui/litellm-dashboard/src/components/chat_ui/ChatUI.test.tsx index 22d8d1fd2a0..f5d6d867786 100644 --- a/ui/litellm-dashboard/src/components/chat_ui/ChatUI.test.tsx +++ b/ui/litellm-dashboard/src/components/chat_ui/ChatUI.test.tsx @@ -1,7 +1,12 @@ -import { render } from "@testing-library/react"; -import { describe, it, expect } from "vitest"; +import { fireEvent, render, screen, waitFor } from "@testing-library/react"; +import { beforeEach, describe, expect, it } from "vitest"; import ChatUI from "./ChatUI"; +// Mock scrollIntoView which is not available in jsdom +beforeEach(() => { + Element.prototype.scrollIntoView = () => {}; +}); + describe("ChatUI", () => { it("should render the chat UI", () => { const { getByText } = render( @@ -15,4 +20,54 @@ describe("ChatUI", () => { ); expect(getByText("Test Key")).toBeInTheDocument(); }); + + it("should show the voice selector when the endpoint type is audio_speech", async () => { + const { getByText, container } = render( + , + ); + + // Wait for the component to render + await waitFor(() => { + expect(getByText("Test Key")).toBeInTheDocument(); + }); + + // Find the endpoint selector by looking for the "Endpoint Type:" text and its associated Select + const endpointTypeText = getByText("Endpoint Type:"); + const selectContainer = endpointTypeText.parentElement; + const selectElement = selectContainer?.querySelector(".ant-select-selector"); + + expect(selectElement).toBeInTheDocument(); + + // Click on the select to open the dropdown + if (selectElement) { + fireEvent.mouseDown(selectElement); + } + + // Wait for the dropdown to appear and find the audio_speech option + await waitFor(() => { + const audioSpeechOption = screen.getByText("/v1/audio/speech"); + expect(audioSpeechOption).toBeInTheDocument(); + }); + + // Click on the audio_speech option + const audioSpeechOption = screen.getByText("/v1/audio/speech"); + fireEvent.click(audioSpeechOption); + + // Verify the voice selector appears + await waitFor(() => { + expect(getByText("Voice")).toBeInTheDocument(); + }); + + // Verify the voice select component is present + const voiceText = getByText("Voice"); + const voiceSelectContainer = voiceText.parentElement; + const voiceSelectElement = voiceSelectContainer?.querySelector(".ant-select"); + expect(voiceSelectElement).toBeInTheDocument(); + }); }); diff --git a/ui/litellm-dashboard/src/components/chat_ui/ChatUI.tsx b/ui/litellm-dashboard/src/components/chat_ui/ChatUI.tsx index 164295261a6..5e85733749a 100644 --- a/ui/litellm-dashboard/src/components/chat_ui/ChatUI.tsx +++ b/ui/litellm-dashboard/src/components/chat_ui/ChatUI.tsx @@ -7,6 +7,8 @@ import { Select, Spin, Typography, Tooltip, Input, Upload, Modal, Button } from import { makeOpenAIChatCompletionRequest } from "./llm_calls/chat_completion"; import { makeOpenAIImageGenerationRequest } from "./llm_calls/image_generation"; import { makeOpenAIImageEditsRequest } from "./llm_calls/image_edits"; +import { makeOpenAIAudioSpeechRequest } from "./llm_calls/audio_speech"; +import { makeOpenAIAudioTranscriptionRequest } from "./llm_calls/audio_transcriptions"; import { makeOpenAIResponsesRequest } from "./llm_calls/responses_api"; import { makeAnthropicMessagesRequest } from "./llm_calls/anthropic_messages"; import { fetchAvailableModels, ModelGroup } from "./llm_calls/fetch_models"; @@ -29,6 +31,7 @@ import { createMultimodalMessage, createDisplayMessage } from "./ResponsesImageU import ChatImageUpload from "./ChatImageUpload"; import ChatImageRenderer from "./ChatImageRenderer"; import { createChatMultimodalMessage, createChatDisplayMessage } from "./ChatImageUtils"; +import AudioRenderer from "./AudioRenderer"; import SessionManagement from "./SessionManagement"; import MCPEventsDisplay, { MCPEvent } from "./MCPEventsDisplay"; import { SearchResultsDisplay } from "./SearchResultsDisplay"; @@ -46,6 +49,7 @@ import { SafetyOutlined, PictureOutlined, CodeOutlined, + SoundOutlined, ToolOutlined, FilePdfOutlined, ArrowUpOutlined, @@ -53,6 +57,7 @@ import { import NotificationsManager from "../molecules/notifications_manager"; import { makeOpenAIEmbeddingsRequest } from "./llm_calls/embeddings_api"; import { truncateString } from "./chatUtils"; +import { OPEN_AI_VOICE_SELECT_OPTIONS, OpenAIVoice } from "./chatConstants"; const { TextArea } = Input; const { Dragger } = Upload; @@ -122,6 +127,16 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d return []; } }); + const [selectedVoice, setSelectedVoice] = useState(() => { + const saved = sessionStorage.getItem("selectedVoice"); + if (!saved) return "alloy"; + try { + return JSON.parse(saved) as OpenAIVoice; + } catch { + // If stored value is not valid JSON, treat it as a plain string + return saved as OpenAIVoice; + } + }); const [selectedVectorStores, setSelectedVectorStores] = useState(() => { const saved = sessionStorage.getItem("selectedVectorStores"); try { @@ -156,6 +171,7 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d const [responsesImagePreviewUrl, setResponsesImagePreviewUrl] = useState(null); const [chatUploadedImage, setChatUploadedImage] = useState(null); const [chatImagePreviewUrl, setChatImagePreviewUrl] = useState(null); + const [uploadedAudio, setUploadedAudio] = useState(null); const [isGetCodeModalVisible, setIsGetCodeModalVisible] = useState(false); const [generatedCode, setGeneratedCode] = useState(""); const [selectedSdk, setSelectedSdk] = useState<"openai" | "azure">("openai"); @@ -200,6 +216,7 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d endpointType, selectedModel, selectedSdk, + selectedVoice, }); setGeneratedCode(code); } @@ -237,6 +254,7 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d sessionStorage.setItem("selectedVectorStores", JSON.stringify(selectedVectorStores)); sessionStorage.setItem("selectedGuardrails", JSON.stringify(selectedGuardrails)); sessionStorage.setItem("selectedMCPTools", JSON.stringify(selectedMCPTools)); + sessionStorage.setItem("selectedVoice", selectedVoice); if (selectedModel) { sessionStorage.setItem("selectedModel", selectedModel); @@ -322,7 +340,7 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d setChatHistory((prev) => { const last = prev[prev.length - 1]; // if the last message is already from this same role, append - if (last && last.role === role && !last.isImage) { + if (last && last.role === role && !last.isImage && !last.isAudio) { // build a new object, but only set `model` if it wasn't there already const updated: MessageType = { ...last, @@ -348,7 +366,7 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d setChatHistory((prevHistory) => { const lastMessage = prevHistory[prevHistory.length - 1]; - if (lastMessage && lastMessage.role === "assistant" && !lastMessage.isImage) { + if (lastMessage && lastMessage.role === "assistant" && !lastMessage.isImage && !lastMessage.isAudio) { return [ ...prevHistory.slice(0, prevHistory.length - 1), { @@ -500,11 +518,15 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d ]); }; + const updateAudioUI = (audioUrl: string, model: string) => { + setChatHistory((prevHistory) => [...prevHistory, { role: "assistant", content: audioUrl, model, isAudio: true }]); + }; + const updateChatImageUI = (imageUrl: string, model?: string) => { setChatHistory((prev) => { const last = prev[prev.length - 1]; // If the last message is from assistant and has content, add image to it - if (last && last.role === "assistant" && !last.isImage) { + if (last && last.role === "assistant" && !last.isImage && !last.isAudio) { const updated = { ...last, image: { @@ -602,8 +624,17 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d setChatImagePreviewUrl(null); }; + const handleAudioUpload = (file: File): false => { + setUploadedAudio(file); + return false; // Prevent default upload behavior + }; + + const handleRemoveAudio = () => { + setUploadedAudio(null); + }; + const handleSendMessage = async () => { - if (inputMessage.trim() === "") return; + if (inputMessage.trim() === "" && endpointType !== EndpointType.TRANSCRIPTION) return; // For image edits, require both image and prompt if (endpointType === EndpointType.IMAGE_EDITS && uploadedImages.length === 0) { @@ -611,6 +642,12 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d return; } + // For audio transcriptions, require audio file + if (endpointType === EndpointType.TRANSCRIPTION && !uploadedAudio) { + NotificationsManager.fromBackend("Please upload an audio file for transcription"); + return; + } + if (!token || !userRole || !userID) { return; } @@ -672,6 +709,12 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d chatImagePreviewUrl || undefined, chatUploadedImage.name, ); + } else if (endpointType === EndpointType.TRANSCRIPTION && uploadedAudio) { + // For audio transcription, show the audio file name and optional prompt + const audioMessage = inputMessage + ? `🎵 Audio file: ${uploadedAudio.name}\nPrompt: ${inputMessage}` + : `🎵 Audio file: ${uploadedAudio.name}`; + displayMessage = createDisplayMessage(audioMessage, false); } else { displayMessage = createDisplayMessage(inputMessage, false); } @@ -687,7 +730,7 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d // For chat completions, we preserve the multimodal content structure const apiChatHistory = [ ...chatHistory - .filter((msg) => !msg.isImage) + .filter((msg) => !msg.isImage && !msg.isAudio) .map(({ role, content }) => ({ role, content: typeof content === "string" ? content : "", @@ -722,6 +765,17 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d selectedTags, signal, ); + } else if (endpointType === EndpointType.SPEECH) { + // For audio speech + await makeOpenAIAudioSpeechRequest( + inputMessage, + selectedVoice, + (audioUrl, model) => updateAudioUI(audioUrl, model), + selectedModel || "", + effectiveApiKey, + selectedTags, + signal, + ); } else if (endpointType === EndpointType.IMAGE_EDITS) { // For image edits if (uploadedImages.length > 0) { @@ -745,7 +799,9 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d } else { // When using UI session management or starting new API session, send full history apiChatHistory = [ - ...chatHistory.filter((msg) => !msg.isImage).map(({ role, content }) => ({ role, content })), + ...chatHistory + .filter((msg) => !msg.isImage && !msg.isAudio) + .map(({ role, content }) => ({ role, content })), newUserMessage, ]; } @@ -770,7 +826,9 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d ); } else if (endpointType === EndpointType.ANTHROPIC_MESSAGES) { const apiChatHistory = [ - ...chatHistory.filter((msg) => !msg.isImage).map(({ role, content }) => ({ role, content })), + ...chatHistory + .filter((msg) => !msg.isImage && !msg.isAudio) + .map(({ role, content }) => ({ role, content })), newUserMessage, ]; @@ -797,6 +855,18 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d effectiveApiKey, selectedTags, ); + } else if (endpointType === EndpointType.TRANSCRIPTION) { + // For audio transcriptions + if (uploadedAudio) { + await makeOpenAIAudioTranscriptionRequest( + uploadedAudio, + (transcription, model) => updateTextUI("assistant", transcription, model), + selectedModel, + effectiveApiKey, + selectedTags, + signal, + ); + } } } } catch (error) { @@ -821,12 +891,23 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d if (endpointType === EndpointType.CHAT && chatUploadedImage) { handleRemoveChatImage(); } + // Clear audio after successful request for transcription + if (endpointType === EndpointType.TRANSCRIPTION && uploadedAudio) { + handleRemoveAudio(); + } } setInputMessage(""); }; const clearChatHistory = () => { + // Clean up audio object URLs before clearing history + chatHistory.forEach((message) => { + if (message.isAudio && typeof message.content === "string") { + URL.revokeObjectURL(message.content); + } + }); + setChatHistory([]); setMessageTraceId(null); setResponsesSessionId(null); // Clear responses session ID @@ -834,6 +915,7 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d handleRemoveAllImages(); // Clear any uploaded images for image edits handleRemoveResponsesImage(); // Clear any uploaded images for responses handleRemoveChatImage(); // Clear any uploaded images for chat completions + handleRemoveAudio(); // Clear any uploaded audio for transcription sessionStorage.removeItem("chatHistory"); sessionStorage.removeItem("messageTraceId"); sessionStorage.removeItem("responsesSessionId"); @@ -946,6 +1028,26 @@ const ChatUI: React.FC = ({ accessToken, token, userRole, userID, d className="mb-4" /> + {/* Voice Selector for Speech Endpoint */} + {endpointType === EndpointType.SPEECH && ( +
+ + + Voice + +