From fbaa4d109d6ca269b3e2b88039d88f7ed8804e45 Mon Sep 17 00:00:00 2001 From: devarakondasrikanth Date: Tue, 31 Mar 2026 11:21:20 -0700 Subject: [PATCH] feat(ui): support vllm reasoning field in playground streaming --- .../llm_calls/chat_completion.test.tsx | 56 +++++++++++++++++++ .../playground/llm_calls/chat_completion.tsx | 45 +++++++++++++-- 2 files changed, 97 insertions(+), 4 deletions(-) diff --git a/ui/litellm-dashboard/src/components/playground/llm_calls/chat_completion.test.tsx b/ui/litellm-dashboard/src/components/playground/llm_calls/chat_completion.test.tsx index 8649834b318..a2b07ed99e1 100644 --- a/ui/litellm-dashboard/src/components/playground/llm_calls/chat_completion.test.tsx +++ b/ui/litellm-dashboard/src/components/playground/llm_calls/chat_completion.test.tsx @@ -256,4 +256,60 @@ describe("chat_completion", () => { const callArgs = mockCreate.mock.calls[0][0]; expect(callArgs).not.toHaveProperty("mock_testing_fallbacks"); }); + + it("should forward vLLM reasoning delta via `reasoning` field", async () => { + const onReasoningContent = vi.fn(); + const mockChunks = [ + { + choices: [{ delta: { reasoning: "thinking..." }, index: 0 }], + model: "qwen3.5", + }, + ]; + async function* mockStream() { + for (const chunk of mockChunks) { + yield chunk; + } + } + mockCreate.mockResolvedValueOnce(mockStream()); + + await makeOpenAIChatCompletionRequest( + mockChatHistory, + mockUpdateUI, + "qwen3.5", + "test-token", + undefined, + undefined, + onReasoningContent, + ); + + expect(onReasoningContent).toHaveBeenCalledWith("thinking..."); + }); + + it("should support structured `reasoning` array deltas", async () => { + const onReasoningContent = vi.fn(); + const mockChunks = [ + { + choices: [{ delta: { reasoning: [{ text: "step 1 " }, { text: "step 2" }] }, index: 0 }], + model: "qwen3.5", + }, + ]; + async function* mockStream() { + for (const chunk of mockChunks) { + yield chunk; + } + } + mockCreate.mockResolvedValueOnce(mockStream()); + + await makeOpenAIChatCompletionRequest( + mockChatHistory, + mockUpdateUI, + "qwen3.5", + "test-token", + undefined, + undefined, + onReasoningContent, + ); + + expect(onReasoningContent).toHaveBeenCalledWith("step 1 step 2"); + }); }); diff --git a/ui/litellm-dashboard/src/components/playground/llm_calls/chat_completion.tsx b/ui/litellm-dashboard/src/components/playground/llm_calls/chat_completion.tsx index 3197c9409ce..9bd5be1b4d7 100644 --- a/ui/litellm-dashboard/src/components/playground/llm_calls/chat_completion.tsx +++ b/ui/litellm-dashboard/src/components/playground/llm_calls/chat_completion.tsx @@ -5,6 +5,41 @@ import { VectorStoreSearchResponse } from "../chat_ui/types"; import { getProxyBaseUrl } from "@/components/networking"; import { MCPServer, type MCPEvent } from "../../mcp_tools/types"; +function extractReasoningDeltaText(delta: any): string | undefined { + const legacyReasoning = delta?.reasoning_content; + if (typeof legacyReasoning === "string" && legacyReasoning.length > 0) { + return legacyReasoning; + } + + const reasoning = delta?.reasoning; + if (typeof reasoning === "string" && reasoning.length > 0) { + return reasoning; + } + + if (Array.isArray(reasoning)) { + const text = reasoning + .map((entry: any) => { + if (typeof entry === "string") return entry; + if (typeof entry?.text === "string") return entry.text; + if (typeof entry?.content === "string") return entry.content; + return ""; + }) + .join(""); + return text.length > 0 ? text : undefined; + } + + if (reasoning && typeof reasoning === "object") { + if (typeof reasoning.text === "string" && reasoning.text.length > 0) { + return reasoning.text; + } + if (typeof reasoning.content === "string" && reasoning.content.length > 0) { + return reasoning.content; + } + } + + return undefined; +} + export async function makeOpenAIChatCompletionRequest( chatHistory: { role: string; content: string | any[] }[], updateUI: (chunk: string, model?: string) => void, @@ -125,13 +160,15 @@ export async function makeOpenAIChatCompletionRequest( // Process content and measure time to first token const delta = chunk.choices[0]?.delta as any; + const reasoningDeltaText = extractReasoningDeltaText(delta); // Debug what's in the delta console.log("Delta content:", chunk.choices[0]?.delta?.content); console.log("Delta reasoning content:", delta?.reasoning_content); + console.log("Delta reasoning:", delta?.reasoning); // Measure time to first token for either content or reasoning_content - if (!firstTokenReceived && (chunk.choices[0]?.delta?.content || (delta && delta.reasoning_content))) { + if (!firstTokenReceived && (chunk.choices[0]?.delta?.content || reasoningDeltaText)) { firstTokenReceived = true; timeToFirstToken = Date.now() - startTime; console.log("First token received! Time:", timeToFirstToken, "ms"); @@ -156,9 +193,9 @@ export async function makeOpenAIChatCompletionRequest( onImageGenerated(delta.image.url, chunk.model); } - // Process reasoning content if present - using type assertion - if (delta && delta.reasoning_content) { - const reasoningContent = delta.reasoning_content; + // Process reasoning content if present + if (reasoningDeltaText) { + const reasoningContent = reasoningDeltaText; if (onReasoningContent) { onReasoningContent(reasoningContent); }