mirror of
https://github.com/RooVetGit/Roo-Code.git
synced 2026-09-29 01:42:19 +00:00
Merge 5762ad7e9b into 7adbfec2a4
This commit is contained in:
commit
dca49643d1
2 changed files with 278 additions and 2 deletions
|
|
@ -605,4 +605,233 @@ describe("NativeOllamaHandler", () => {
|
||||||
expect(firstEndIndex).toBeGreaterThan(lastPartialIndex)
|
expect(firstEndIndex).toBeGreaterThan(lastPartialIndex)
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
||||||
|
describe("native thinking support", () => {
|
||||||
|
it("should yield reasoning from native thinking field", async () => {
|
||||||
|
// Mock response with native thinking field (Ollama 0.5.0+)
|
||||||
|
mockChat.mockImplementation(async function* () {
|
||||||
|
yield {
|
||||||
|
message: {
|
||||||
|
content: "",
|
||||||
|
thinking: "Let me analyze this problem...",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
yield {
|
||||||
|
message: {
|
||||||
|
content: "",
|
||||||
|
thinking: " First, I need to consider X.",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
yield {
|
||||||
|
message: {
|
||||||
|
content: "The answer is 42",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
const stream = handler.createMessage("System", [{ role: "user" as const, content: "What is the answer?" }])
|
||||||
|
const results = []
|
||||||
|
|
||||||
|
for await (const chunk of stream) {
|
||||||
|
results.push(chunk)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Should have reasoning chunks from native thinking field
|
||||||
|
const reasoningChunks = results.filter((r) => r.type === "reasoning")
|
||||||
|
expect(reasoningChunks).toHaveLength(2)
|
||||||
|
expect(reasoningChunks[0]).toEqual({ type: "reasoning", text: "Let me analyze this problem..." })
|
||||||
|
expect(reasoningChunks[1]).toEqual({ type: "reasoning", text: " First, I need to consider X." })
|
||||||
|
|
||||||
|
// Should also have the text response
|
||||||
|
const textChunks = results.filter((r) => r.type === "text")
|
||||||
|
expect(textChunks.some((c) => c.text === "The answer is 42")).toBe(true)
|
||||||
|
})
|
||||||
|
|
||||||
|
it("should pass think option when reasoning is enabled with model support", async () => {
|
||||||
|
mockGetOllamaModels.mockResolvedValue({
|
||||||
|
"thinking-model": {
|
||||||
|
contextWindow: 4096,
|
||||||
|
maxTokens: 4096,
|
||||||
|
supportsImages: false,
|
||||||
|
supportsPromptCache: false,
|
||||||
|
supportsReasoningEffort: true,
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
const options: ApiHandlerOptions = {
|
||||||
|
apiModelId: "thinking-model",
|
||||||
|
ollamaModelId: "thinking-model",
|
||||||
|
ollamaBaseUrl: "http://localhost:11434",
|
||||||
|
enableReasoningEffort: true,
|
||||||
|
reasoningEffort: "high",
|
||||||
|
}
|
||||||
|
|
||||||
|
handler = new NativeOllamaHandler(options)
|
||||||
|
|
||||||
|
mockChat.mockImplementation(async function* () {
|
||||||
|
yield { message: { content: "Response" } }
|
||||||
|
})
|
||||||
|
|
||||||
|
const stream = handler.createMessage("System", [{ role: "user" as const, content: "Test" }])
|
||||||
|
|
||||||
|
for await (const _ of stream) {
|
||||||
|
// consume stream
|
||||||
|
}
|
||||||
|
|
||||||
|
// Verify think option was passed with high value
|
||||||
|
expect(mockChat).toHaveBeenCalledWith(
|
||||||
|
expect.objectContaining({
|
||||||
|
think: "high",
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it("should pass think: medium when reasoningEffort is medium", async () => {
|
||||||
|
mockGetOllamaModels.mockResolvedValue({
|
||||||
|
"thinking-model": {
|
||||||
|
contextWindow: 4096,
|
||||||
|
maxTokens: 4096,
|
||||||
|
supportsImages: false,
|
||||||
|
supportsPromptCache: false,
|
||||||
|
supportsReasoningEffort: true,
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
const options: ApiHandlerOptions = {
|
||||||
|
apiModelId: "thinking-model",
|
||||||
|
ollamaModelId: "thinking-model",
|
||||||
|
ollamaBaseUrl: "http://localhost:11434",
|
||||||
|
enableReasoningEffort: true,
|
||||||
|
reasoningEffort: "medium",
|
||||||
|
}
|
||||||
|
|
||||||
|
handler = new NativeOllamaHandler(options)
|
||||||
|
|
||||||
|
mockChat.mockImplementation(async function* () {
|
||||||
|
yield { message: { content: "Response" } }
|
||||||
|
})
|
||||||
|
|
||||||
|
const stream = handler.createMessage("System", [{ role: "user" as const, content: "Test" }])
|
||||||
|
|
||||||
|
for await (const _ of stream) {
|
||||||
|
// consume stream
|
||||||
|
}
|
||||||
|
|
||||||
|
expect(mockChat).toHaveBeenCalledWith(
|
||||||
|
expect.objectContaining({
|
||||||
|
think: "medium",
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it("should pass think: low when reasoningEffort is low or minimal", async () => {
|
||||||
|
mockGetOllamaModels.mockResolvedValue({
|
||||||
|
"thinking-model": {
|
||||||
|
contextWindow: 4096,
|
||||||
|
maxTokens: 4096,
|
||||||
|
supportsImages: false,
|
||||||
|
supportsPromptCache: false,
|
||||||
|
supportsReasoningEffort: true,
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
const options: ApiHandlerOptions = {
|
||||||
|
apiModelId: "thinking-model",
|
||||||
|
ollamaModelId: "thinking-model",
|
||||||
|
ollamaBaseUrl: "http://localhost:11434",
|
||||||
|
enableReasoningEffort: true,
|
||||||
|
reasoningEffort: "low",
|
||||||
|
}
|
||||||
|
|
||||||
|
handler = new NativeOllamaHandler(options)
|
||||||
|
|
||||||
|
mockChat.mockImplementation(async function* () {
|
||||||
|
yield { message: { content: "Response" } }
|
||||||
|
})
|
||||||
|
|
||||||
|
const stream = handler.createMessage("System", [{ role: "user" as const, content: "Test" }])
|
||||||
|
|
||||||
|
for await (const _ of stream) {
|
||||||
|
// consume stream
|
||||||
|
}
|
||||||
|
|
||||||
|
expect(mockChat).toHaveBeenCalledWith(
|
||||||
|
expect.objectContaining({
|
||||||
|
think: "low",
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it("should not pass think option when reasoning is not enabled", async () => {
|
||||||
|
mockGetOllamaModels.mockResolvedValue({
|
||||||
|
llama2: {
|
||||||
|
contextWindow: 4096,
|
||||||
|
maxTokens: 4096,
|
||||||
|
supportsImages: false,
|
||||||
|
supportsPromptCache: false,
|
||||||
|
// No supportsReasoningEffort
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
const options: ApiHandlerOptions = {
|
||||||
|
apiModelId: "llama2",
|
||||||
|
ollamaModelId: "llama2",
|
||||||
|
ollamaBaseUrl: "http://localhost:11434",
|
||||||
|
// No enableReasoningEffort
|
||||||
|
}
|
||||||
|
|
||||||
|
handler = new NativeOllamaHandler(options)
|
||||||
|
|
||||||
|
mockChat.mockImplementation(async function* () {
|
||||||
|
yield { message: { content: "Response" } }
|
||||||
|
})
|
||||||
|
|
||||||
|
const stream = handler.createMessage("System", [{ role: "user" as const, content: "Test" }])
|
||||||
|
|
||||||
|
for await (const _ of stream) {
|
||||||
|
// consume stream
|
||||||
|
}
|
||||||
|
|
||||||
|
// Verify think option was NOT passed (or is undefined)
|
||||||
|
expect(mockChat).toHaveBeenCalledWith(
|
||||||
|
expect.objectContaining({
|
||||||
|
think: undefined,
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
})
|
||||||
|
|
||||||
|
it("should handle both native thinking and tag-based reasoning", async () => {
|
||||||
|
// Some models might use both methods
|
||||||
|
mockChat.mockImplementation(async function* () {
|
||||||
|
yield {
|
||||||
|
message: {
|
||||||
|
content: "",
|
||||||
|
thinking: "Native thinking here",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
yield {
|
||||||
|
message: {
|
||||||
|
content: "<think>Tag-based thinking</think>The final answer",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
const stream = handler.createMessage("System", [{ role: "user" as const, content: "Test" }])
|
||||||
|
const results = []
|
||||||
|
|
||||||
|
for await (const chunk of stream) {
|
||||||
|
results.push(chunk)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Should have reasoning from both sources
|
||||||
|
const reasoningChunks = results.filter((r) => r.type === "reasoning")
|
||||||
|
expect(reasoningChunks.length).toBeGreaterThanOrEqual(2)
|
||||||
|
|
||||||
|
// Check native thinking was captured
|
||||||
|
expect(reasoningChunks.some((c) => c.text === "Native thinking here")).toBe(true)
|
||||||
|
|
||||||
|
// Check tag-based thinking was captured
|
||||||
|
expect(reasoningChunks.some((c) => c.text === "Tag-based thinking")).toBe(true)
|
||||||
|
})
|
||||||
|
})
|
||||||
})
|
})
|
||||||
|
|
|
||||||
|
|
@ -5,6 +5,7 @@ import { ModelInfo, openAiModelInfoSaneDefaults, DEEP_SEEK_DEFAULT_TEMPERATURE }
|
||||||
import { ApiStream } from "../transform/stream"
|
import { ApiStream } from "../transform/stream"
|
||||||
import { BaseProvider } from "./base-provider"
|
import { BaseProvider } from "./base-provider"
|
||||||
import type { ApiHandlerOptions } from "../../shared/api"
|
import type { ApiHandlerOptions } from "../../shared/api"
|
||||||
|
import { shouldUseReasoningEffort } from "../../shared/api"
|
||||||
import { getOllamaModels } from "./fetchers/ollama"
|
import { getOllamaModels } from "./fetchers/ollama"
|
||||||
import { TagMatcher } from "../../utils/tag-matcher"
|
import { TagMatcher } from "../../utils/tag-matcher"
|
||||||
import type { SingleCompletionHandler, ApiHandlerCreateMessageMetadata } from "../index"
|
import type { SingleCompletionHandler, ApiHandlerCreateMessageMetadata } from "../index"
|
||||||
|
|
@ -14,6 +15,9 @@ interface OllamaChatOptions {
|
||||||
num_ctx?: number
|
num_ctx?: number
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Ollama think option type: boolean or effort level
|
||||||
|
type OllamaThinkOption = boolean | "high" | "medium" | "low"
|
||||||
|
|
||||||
function convertToOllamaMessages(anthropicMessages: Anthropic.Messages.MessageParam[]): Message[] {
|
function convertToOllamaMessages(anthropicMessages: Anthropic.Messages.MessageParam[]): Message[] {
|
||||||
const ollamaMessages: Message[] = []
|
const ollamaMessages: Message[] = []
|
||||||
|
|
||||||
|
|
@ -155,6 +159,37 @@ export class NativeOllamaHandler extends BaseProvider implements SingleCompletio
|
||||||
this.options = options
|
this.options = options
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Determines the Ollama `think` option value based on model and settings.
|
||||||
|
* Returns undefined if thinking is not enabled, otherwise returns the
|
||||||
|
* appropriate effort level or true for basic thinking.
|
||||||
|
*/
|
||||||
|
private getThinkOption(modelInfo: ModelInfo): OllamaThinkOption | undefined {
|
||||||
|
// Check if reasoning should be enabled based on model and settings
|
||||||
|
const useReasoning = shouldUseReasoningEffort({
|
||||||
|
model: modelInfo,
|
||||||
|
settings: this.options,
|
||||||
|
})
|
||||||
|
|
||||||
|
if (!useReasoning) {
|
||||||
|
return undefined
|
||||||
|
}
|
||||||
|
|
||||||
|
// Map reasoning effort to Ollama think option
|
||||||
|
const effort = this.options.reasoningEffort
|
||||||
|
|
||||||
|
if (effort === "high" || effort === "xhigh") {
|
||||||
|
return "high"
|
||||||
|
} else if (effort === "medium") {
|
||||||
|
return "medium"
|
||||||
|
} else if (effort === "low" || effort === "minimal") {
|
||||||
|
return "low"
|
||||||
|
}
|
||||||
|
|
||||||
|
// Default to true (let Ollama decide) when reasoning is enabled but no specific effort
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
private ensureClient(): Ollama {
|
private ensureClient(): Ollama {
|
||||||
if (!this.client) {
|
if (!this.client) {
|
||||||
try {
|
try {
|
||||||
|
|
@ -206,7 +241,7 @@ export class NativeOllamaHandler extends BaseProvider implements SingleCompletio
|
||||||
metadata?: ApiHandlerCreateMessageMetadata,
|
metadata?: ApiHandlerCreateMessageMetadata,
|
||||||
): ApiStream {
|
): ApiStream {
|
||||||
const client = this.ensureClient()
|
const client = this.ensureClient()
|
||||||
const { id: modelId } = await this.fetchModel()
|
const { id: modelId, info: modelInfo } = await this.fetchModel()
|
||||||
const useR1Format = modelId.toLowerCase().includes("deepseek-r1")
|
const useR1Format = modelId.toLowerCase().includes("deepseek-r1")
|
||||||
|
|
||||||
const ollamaMessages: Message[] = [
|
const ollamaMessages: Message[] = [
|
||||||
|
|
@ -214,6 +249,9 @@ export class NativeOllamaHandler extends BaseProvider implements SingleCompletio
|
||||||
...convertToOllamaMessages(messages),
|
...convertToOllamaMessages(messages),
|
||||||
]
|
]
|
||||||
|
|
||||||
|
// Determine if native thinking should be enabled
|
||||||
|
const thinkOption = this.getThinkOption(modelInfo)
|
||||||
|
|
||||||
const matcher = new TagMatcher(
|
const matcher = new TagMatcher(
|
||||||
"think",
|
"think",
|
||||||
(chunk) =>
|
(chunk) =>
|
||||||
|
|
@ -235,12 +273,14 @@ export class NativeOllamaHandler extends BaseProvider implements SingleCompletio
|
||||||
}
|
}
|
||||||
|
|
||||||
// Create the actual API request promise
|
// Create the actual API request promise
|
||||||
|
// Include think option if reasoning is enabled (Ollama 0.5.0+)
|
||||||
const stream = await client.chat({
|
const stream = await client.chat({
|
||||||
model: modelId,
|
model: modelId,
|
||||||
messages: ollamaMessages,
|
messages: ollamaMessages,
|
||||||
stream: true,
|
stream: true,
|
||||||
options: chatOptions,
|
options: chatOptions,
|
||||||
tools: this.convertToolsToOllama(metadata?.tools),
|
tools: this.convertToolsToOllama(metadata?.tools),
|
||||||
|
think: thinkOption,
|
||||||
})
|
})
|
||||||
|
|
||||||
let totalInputTokens = 0
|
let totalInputTokens = 0
|
||||||
|
|
@ -252,8 +292,15 @@ export class NativeOllamaHandler extends BaseProvider implements SingleCompletio
|
||||||
|
|
||||||
try {
|
try {
|
||||||
for await (const chunk of stream) {
|
for await (const chunk of stream) {
|
||||||
|
// Handle native thinking field (Ollama 0.5.0+)
|
||||||
|
// This is the preferred method for models that support it
|
||||||
|
const thinking = (chunk.message as Message & { thinking?: string }).thinking
|
||||||
|
if (typeof thinking === "string" && thinking.length > 0) {
|
||||||
|
yield { type: "reasoning", text: thinking }
|
||||||
|
}
|
||||||
|
|
||||||
if (typeof chunk.message.content === "string" && chunk.message.content.length > 0) {
|
if (typeof chunk.message.content === "string" && chunk.message.content.length > 0) {
|
||||||
// Process content through matcher for reasoning detection
|
// Process content through matcher for reasoning detection (fallback for <think> tags)
|
||||||
for (const matcherChunk of matcher.update(chunk.message.content)) {
|
for (const matcherChunk of matcher.update(chunk.message.content)) {
|
||||||
yield matcherChunk
|
yield matcherChunk
|
||||||
}
|
}
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue