diff --git a/src/api/providers/openrouter.ts b/src/api/providers/openrouter.ts index 8d37cdf0a3..ccdde6c378 100644 --- a/src/api/providers/openrouter.ts +++ b/src/api/providers/openrouter.ts @@ -97,7 +97,11 @@ export class OpenRouterHandler implements ApiHandler { } // Removes messages in the middle when close to context window limit. Should not be applied to models that support prompt caching since it would continuously break the cache. - const shouldApplyMiddleOutTransform = !this.getModel().info.supportsPromptCache + let shouldApplyMiddleOutTransform = !this.getModel().info.supportsPromptCache + // except for deepseek (which we set supportsPromptCache to true for), where because the context window is so small our truncation algo might miss and we should use openrouter's middle-out transform as a fallback to ensure we don't exceed the context window + if (this.getModel().id === "deepseek/deepseek-chat") { + shouldApplyMiddleOutTransform = true + } // @ts-ignore-next-line const stream = await this.client.chat.completions.create({ diff --git a/src/core/Cline.ts b/src/core/Cline.ts index 1bffe8c9d1..9c233b245d 100644 --- a/src/core/Cline.ts +++ b/src/core/Cline.ts @@ -52,6 +52,7 @@ import { ClineProvider, GlobalFileNames } from "./webview/ClineProvider" import { showSystemNotification } from "../integrations/notifications" import { removeInvalidChars } from "../utils/string" import { fixModelHtmlEscaping } from "../utils/string" +import { OpenAiHandler } from "../api/providers/openai" const cwd = vscode.workspace.workspaceFolders?.map((folder) => folder.uri.fsPath).at(0) ?? path.join(os.homedir(), "Desktop") // may or may not exist but fs checking existence would immediately ask for permission which would be bad UX, need to come up with a better solution @@ -831,8 +832,26 @@ export class Cline { previousRequest.text, ) const totalTokens = (tokensIn || 0) + (tokensOut || 0) + (cacheWrites || 0) + (cacheReads || 0) - const contextWindow = this.api.getModel().info.contextWindow || 128_000 - const maxAllowedSize = Math.max(contextWindow - 40_000, contextWindow * 0.8) + let contextWindow = this.api.getModel().info.contextWindow || 128_000 + // FIXME: hack to get anyone using openai compatible with deepseek to have the proper context window instead of the default 128k. We need a way for the user to specify the context window for models they input through openai compatible + if (this.api instanceof OpenAiHandler && this.api.getModel().id.toLowerCase().includes("deepseek")) { + contextWindow = 64_000 + } + let maxAllowedSize: number + switch (contextWindow) { + case 64_000: // deepseek models + maxAllowedSize = contextWindow - 25_000 + break + case 128_000: // most models + maxAllowedSize = contextWindow - 40_000 + break + case 200_000: // claude models + maxAllowedSize = contextWindow - 40_000 + break + default: + maxAllowedSize = Math.max(contextWindow - 40_000, contextWindow * 0.8) // for deepseek, 80% of 64k meant only ~10k buffer which was too small and resulted in users getting context window errors. + } + if (totalTokens >= maxAllowedSize) { const truncatedMessages = truncateHalfConversation(this.apiConversationHistory) await this.overwriteApiConversationHistory(truncatedMessages) diff --git a/src/core/webview/ClineProvider.ts b/src/core/webview/ClineProvider.ts index 78cc71b460..4081e617dd 100644 --- a/src/core/webview/ClineProvider.ts +++ b/src/core/webview/ClineProvider.ts @@ -726,6 +726,11 @@ export class ClineProvider implements vscode.WebviewViewProvider { modelInfo.cacheWritesPrice = 0.3 modelInfo.cacheReadsPrice = 0.03 break + case "deepseek/deepseek-chat": + modelInfo.supportsPromptCache = true + modelInfo.cacheWritesPrice = 0.14 + modelInfo.cacheReadsPrice = 0.014 + break } models[rawModel.id] = modelInfo