mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
feat(ui): support vllm reasoning field in playground streaming
This commit is contained in:
parent
7833eee344
commit
fbaa4d109d
2 changed files with 97 additions and 4 deletions
|
|
@ -256,4 +256,60 @@ describe("chat_completion", () => {
|
|||
const callArgs = mockCreate.mock.calls[0][0];
|
||||
expect(callArgs).not.toHaveProperty("mock_testing_fallbacks");
|
||||
});
|
||||
|
||||
it("should forward vLLM reasoning delta via `reasoning` field", async () => {
|
||||
const onReasoningContent = vi.fn();
|
||||
const mockChunks = [
|
||||
{
|
||||
choices: [{ delta: { reasoning: "thinking..." }, index: 0 }],
|
||||
model: "qwen3.5",
|
||||
},
|
||||
];
|
||||
async function* mockStream() {
|
||||
for (const chunk of mockChunks) {
|
||||
yield chunk;
|
||||
}
|
||||
}
|
||||
mockCreate.mockResolvedValueOnce(mockStream());
|
||||
|
||||
await makeOpenAIChatCompletionRequest(
|
||||
mockChatHistory,
|
||||
mockUpdateUI,
|
||||
"qwen3.5",
|
||||
"test-token",
|
||||
undefined,
|
||||
undefined,
|
||||
onReasoningContent,
|
||||
);
|
||||
|
||||
expect(onReasoningContent).toHaveBeenCalledWith("thinking...");
|
||||
});
|
||||
|
||||
it("should support structured `reasoning` array deltas", async () => {
|
||||
const onReasoningContent = vi.fn();
|
||||
const mockChunks = [
|
||||
{
|
||||
choices: [{ delta: { reasoning: [{ text: "step 1 " }, { text: "step 2" }] }, index: 0 }],
|
||||
model: "qwen3.5",
|
||||
},
|
||||
];
|
||||
async function* mockStream() {
|
||||
for (const chunk of mockChunks) {
|
||||
yield chunk;
|
||||
}
|
||||
}
|
||||
mockCreate.mockResolvedValueOnce(mockStream());
|
||||
|
||||
await makeOpenAIChatCompletionRequest(
|
||||
mockChatHistory,
|
||||
mockUpdateUI,
|
||||
"qwen3.5",
|
||||
"test-token",
|
||||
undefined,
|
||||
undefined,
|
||||
onReasoningContent,
|
||||
);
|
||||
|
||||
expect(onReasoningContent).toHaveBeenCalledWith("step 1 step 2");
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -5,6 +5,41 @@ import { VectorStoreSearchResponse } from "../chat_ui/types";
|
|||
import { getProxyBaseUrl } from "@/components/networking";
|
||||
import { MCPServer, type MCPEvent } from "../../mcp_tools/types";
|
||||
|
||||
function extractReasoningDeltaText(delta: any): string | undefined {
|
||||
const legacyReasoning = delta?.reasoning_content;
|
||||
if (typeof legacyReasoning === "string" && legacyReasoning.length > 0) {
|
||||
return legacyReasoning;
|
||||
}
|
||||
|
||||
const reasoning = delta?.reasoning;
|
||||
if (typeof reasoning === "string" && reasoning.length > 0) {
|
||||
return reasoning;
|
||||
}
|
||||
|
||||
if (Array.isArray(reasoning)) {
|
||||
const text = reasoning
|
||||
.map((entry: any) => {
|
||||
if (typeof entry === "string") return entry;
|
||||
if (typeof entry?.text === "string") return entry.text;
|
||||
if (typeof entry?.content === "string") return entry.content;
|
||||
return "";
|
||||
})
|
||||
.join("");
|
||||
return text.length > 0 ? text : undefined;
|
||||
}
|
||||
|
||||
if (reasoning && typeof reasoning === "object") {
|
||||
if (typeof reasoning.text === "string" && reasoning.text.length > 0) {
|
||||
return reasoning.text;
|
||||
}
|
||||
if (typeof reasoning.content === "string" && reasoning.content.length > 0) {
|
||||
return reasoning.content;
|
||||
}
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export async function makeOpenAIChatCompletionRequest(
|
||||
chatHistory: { role: string; content: string | any[] }[],
|
||||
updateUI: (chunk: string, model?: string) => void,
|
||||
|
|
@ -125,13 +160,15 @@ export async function makeOpenAIChatCompletionRequest(
|
|||
|
||||
// Process content and measure time to first token
|
||||
const delta = chunk.choices[0]?.delta as any;
|
||||
const reasoningDeltaText = extractReasoningDeltaText(delta);
|
||||
|
||||
// Debug what's in the delta
|
||||
console.log("Delta content:", chunk.choices[0]?.delta?.content);
|
||||
console.log("Delta reasoning content:", delta?.reasoning_content);
|
||||
console.log("Delta reasoning:", delta?.reasoning);
|
||||
|
||||
// Measure time to first token for either content or reasoning_content
|
||||
if (!firstTokenReceived && (chunk.choices[0]?.delta?.content || (delta && delta.reasoning_content))) {
|
||||
if (!firstTokenReceived && (chunk.choices[0]?.delta?.content || reasoningDeltaText)) {
|
||||
firstTokenReceived = true;
|
||||
timeToFirstToken = Date.now() - startTime;
|
||||
console.log("First token received! Time:", timeToFirstToken, "ms");
|
||||
|
|
@ -156,9 +193,9 @@ export async function makeOpenAIChatCompletionRequest(
|
|||
onImageGenerated(delta.image.url, chunk.model);
|
||||
}
|
||||
|
||||
// Process reasoning content if present - using type assertion
|
||||
if (delta && delta.reasoning_content) {
|
||||
const reasoningContent = delta.reasoning_content;
|
||||
// Process reasoning content if present
|
||||
if (reasoningDeltaText) {
|
||||
const reasoningContent = reasoningDeltaText;
|
||||
if (onReasoningContent) {
|
||||
onReasoningContent(reasoningContent);
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue