From f87ca7fb3843f005bbc36922c036a828b0b66016 Mon Sep 17 00:00:00 2001 From: Roo Code Date: Thu, 20 Nov 2025 18:56:43 +0000 Subject: [PATCH] fix: sanitize URLs to prevent proxy corruption in TUN mode - Add URL sanitization logic to handle corrupted URLs from proxy tools like NekoBox - Support both production and development RooCode domains with subdomains - Preserve paths, query parameters, and fragments when fixing corrupted URLs - Add comprehensive test coverage for URL sanitization scenarios Fixes #9441 --- packages/cloud/src/__tests__/config.spec.ts | 178 ++++++++++++++++++++ packages/cloud/src/config.ts | 72 +++++++- 2 files changed, 248 insertions(+), 2 deletions(-) create mode 100644 packages/cloud/src/__tests__/config.spec.ts diff --git a/packages/cloud/src/__tests__/config.spec.ts b/packages/cloud/src/__tests__/config.spec.ts new file mode 100644 index 0000000000..b2567dfdb5 --- /dev/null +++ b/packages/cloud/src/__tests__/config.spec.ts @@ -0,0 +1,178 @@ +import { describe, it, expect, beforeEach, afterEach, vi } from "vitest" +import { getClerkBaseUrl, getRooCodeApiUrl, PRODUCTION_CLERK_BASE_URL, PRODUCTION_ROO_CODE_API_URL } from "../config.js" + +describe("config", () => { + let originalEnv: NodeJS.ProcessEnv + + beforeEach(() => { + // Save original environment + originalEnv = { ...process.env } + }) + + afterEach(() => { + // Restore original environment + process.env = originalEnv + vi.clearAllMocks() + }) + + describe("getClerkBaseUrl", () => { + it("should return production URL when environment variable is not set", () => { + delete process.env.CLERK_BASE_URL + expect(getClerkBaseUrl()).toBe(PRODUCTION_CLERK_BASE_URL) + }) + + it("should return valid custom URL from environment variable", () => { + process.env.CLERK_BASE_URL = "https://custom.clerk.com" + expect(getClerkBaseUrl()).toBe("https://custom.clerk.com") + }) + + it("should sanitize corrupted URL with proxy prefix (NekoBox issue)", () => { + // This is the exact issue reported: proxy adds its address before the actual URL + process.env.CLERK_BASE_URL = "http://127.0.0.1:2080clerk.roocode.com:443" + expect(getClerkBaseUrl()).toBe("https://clerk.roocode.com") + }) + + it("should sanitize corrupted URL with different proxy address", () => { + process.env.CLERK_BASE_URL = "http://192.168.1.1:8080clerk.roocode.com:443" + expect(getClerkBaseUrl()).toBe("https://clerk.roocode.com") + }) + + it("should preserve path in corrupted URL", () => { + process.env.CLERK_BASE_URL = "http://127.0.0.1:2080clerk.roocode.com:443/api/v1" + expect(getClerkBaseUrl()).toBe("https://clerk.roocode.com/api/v1") + }) + + it("should handle URL with clerk.roocode.com embedded anywhere", () => { + process.env.CLERK_BASE_URL = "garbage-text-clerk.roocode.com" + expect(getClerkBaseUrl()).toBe("https://clerk.roocode.com") + }) + + it("should return fallback for completely invalid URL", () => { + process.env.CLERK_BASE_URL = "not-a-url-at-all" + expect(getClerkBaseUrl()).toBe(PRODUCTION_CLERK_BASE_URL) + }) + + it("should handle empty string", () => { + process.env.CLERK_BASE_URL = "" + expect(getClerkBaseUrl()).toBe(PRODUCTION_CLERK_BASE_URL) + }) + + it("should preserve valid dev URL", () => { + process.env.CLERK_BASE_URL = "https://dev.clerk.roocode.com" + expect(getClerkBaseUrl()).toBe("https://dev.clerk.roocode.com") + }) + }) + + describe("getRooCodeApiUrl", () => { + it("should return production URL when environment variable is not set", () => { + delete process.env.ROO_CODE_API_URL + expect(getRooCodeApiUrl()).toBe(PRODUCTION_ROO_CODE_API_URL) + }) + + it("should return valid custom URL from environment variable", () => { + process.env.ROO_CODE_API_URL = "https://custom.api.com" + expect(getRooCodeApiUrl()).toBe("https://custom.api.com") + }) + + it("should sanitize corrupted URL with proxy prefix (NekoBox issue)", () => { + // This is the exact issue reported: proxy adds its address before the actual URL + process.env.ROO_CODE_API_URL = "http://127.0.0.1:2080app.roocode.com:443" + expect(getRooCodeApiUrl()).toBe("https://app.roocode.com") + }) + + it("should sanitize corrupted URL with different proxy address", () => { + process.env.ROO_CODE_API_URL = "http://192.168.1.1:8080app.roocode.com:443" + expect(getRooCodeApiUrl()).toBe("https://app.roocode.com") + }) + + it("should preserve path in corrupted URL", () => { + process.env.ROO_CODE_API_URL = "http://127.0.0.1:2080app.roocode.com:443/api/v1" + expect(getRooCodeApiUrl()).toBe("https://app.roocode.com/api/v1") + }) + + it("should handle URL with app.roocode.com embedded anywhere", () => { + process.env.ROO_CODE_API_URL = "garbage-text-app.roocode.com" + expect(getRooCodeApiUrl()).toBe("https://app.roocode.com") + }) + + it("should handle api.roocode.com domain", () => { + process.env.ROO_CODE_API_URL = "http://127.0.0.1:2080api.roocode.com:443" + expect(getRooCodeApiUrl()).toBe("https://api.roocode.com") + }) + + it("should return fallback for completely invalid URL", () => { + process.env.ROO_CODE_API_URL = "not-a-url-at-all" + expect(getRooCodeApiUrl()).toBe(PRODUCTION_ROO_CODE_API_URL) + }) + + it("should handle empty string", () => { + process.env.ROO_CODE_API_URL = "" + expect(getRooCodeApiUrl()).toBe(PRODUCTION_ROO_CODE_API_URL) + }) + + it("should preserve valid dev URL", () => { + process.env.ROO_CODE_API_URL = "https://dev.app.roocode.com" + expect(getRooCodeApiUrl()).toBe("https://dev.app.roocode.com") + }) + + it("should handle localhost URLs correctly", () => { + process.env.ROO_CODE_API_URL = "http://localhost:3000" + expect(getRooCodeApiUrl()).toBe("http://localhost:3000") + }) + + it("should handle local IP without proxy corruption", () => { + // User reported their local endpoint also gets corrupted + process.env.ROO_CODE_API_URL = "http://192.168.1.102:8000" + expect(getRooCodeApiUrl()).toBe("http://192.168.1.102:8000") + }) + + it("should fix corrupted local IP with proxy prefix", () => { + // Simulating what might happen if proxy corrupts local IP + process.env.ROO_CODE_API_URL = "http://127.0.0.1:2080192.168.1.102:8000" + // Since this doesn't contain a known domain, it should fall back + expect(getRooCodeApiUrl()).toBe(PRODUCTION_ROO_CODE_API_URL) + }) + }) + + describe("URL sanitization edge cases", () => { + it("should handle HTTP vs HTTPS detection based on port", () => { + // Port 443 should imply HTTPS + process.env.CLERK_BASE_URL = "proxygarbage-clerk.roocode.com:443" + expect(getClerkBaseUrl()).toBe("https://clerk.roocode.com") + + // No port 443 but fallback uses https + process.env.CLERK_BASE_URL = "proxygarbage-clerk.roocode.com" + expect(getClerkBaseUrl()).toBe("https://clerk.roocode.com") + }) + + it("should handle multiple corruptions in the same URL", () => { + // Multiple instances of corruption + process.env.CLERK_BASE_URL = "http://127.0.0.1:2080http://proxy:8888clerk.roocode.com:443" + expect(getClerkBaseUrl()).toBe("https://clerk.roocode.com") + }) + + it("should not be fooled by domain-like strings in paths", () => { + // Valid URL that happens to contain the domain string elsewhere + process.env.CLERK_BASE_URL = "https://valid.com/path/clerk.roocode.com/test" + expect(getClerkBaseUrl()).toBe("https://valid.com/path/clerk.roocode.com/test") + }) + + it("should handle undefined environment variables", () => { + process.env.CLERK_BASE_URL = undefined + expect(getClerkBaseUrl()).toBe(PRODUCTION_CLERK_BASE_URL) + + process.env.ROO_CODE_API_URL = undefined + expect(getRooCodeApiUrl()).toBe(PRODUCTION_ROO_CODE_API_URL) + }) + + it("should handle URLs with query parameters", () => { + process.env.CLERK_BASE_URL = "http://127.0.0.1:2080clerk.roocode.com:443?param=value" + expect(getClerkBaseUrl()).toBe("https://clerk.roocode.com?param=value") + }) + + it("should handle URLs with fragments", () => { + process.env.CLERK_BASE_URL = "http://127.0.0.1:2080clerk.roocode.com:443#fragment" + expect(getClerkBaseUrl()).toBe("https://clerk.roocode.com#fragment") + }) + }) +}) diff --git a/packages/cloud/src/config.ts b/packages/cloud/src/config.ts index cfff9d0f58..d00db88022 100644 --- a/packages/cloud/src/config.ts +++ b/packages/cloud/src/config.ts @@ -1,6 +1,74 @@ export const PRODUCTION_CLERK_BASE_URL = "https://clerk.roocode.com" export const PRODUCTION_ROO_CODE_API_URL = "https://app.roocode.com" -export const getClerkBaseUrl = () => process.env.CLERK_BASE_URL || PRODUCTION_CLERK_BASE_URL +/** + * Sanitizes a URL by removing any proxy prefixes that may have been incorrectly added + * by proxy tools like NekoBox in TUN mode. + * + * @param url - The URL to sanitize + * @param fallback - The fallback URL to use if sanitization fails + * @returns The sanitized URL + */ +function sanitizeUrl(url: string | undefined, fallback: string): string { + if (!url) { + return fallback + } -export const getRooCodeApiUrl = () => process.env.ROO_CODE_API_URL || PRODUCTION_ROO_CODE_API_URL + try { + // First, try to parse as a valid URL + const parsedUrl = new URL(url) + + // Check if it's already a valid RooCode URL + if ( + parsedUrl.hostname.endsWith(".roocode.com") || + parsedUrl.hostname === "clerk.roocode.com" || + parsedUrl.hostname === "app.roocode.com" || + parsedUrl.hostname === "api.roocode.com" + ) { + return url // URL is already valid + } + + // If it parses successfully and looks reasonable (not a RooCode domain), use it + if (parsedUrl.protocol && parsedUrl.hostname) { + return url + } + } catch (_error) { + // URL parsing failed, try to fix corrupted URL + } + + // Check if the URL contains known RooCode domains (for corrupted URLs) + const rooCodePatterns = [ + { pattern: /(dev\.|staging\.|test\.)?clerk\.roocode\.com/, domain: "clerk.roocode.com" }, + { pattern: /(dev\.|staging\.|test\.)?app\.roocode\.com/, domain: "app.roocode.com" }, + { pattern: /(dev\.|staging\.|test\.)?api\.roocode\.com/, domain: "api.roocode.com" }, + ] + + for (const { pattern } of rooCodePatterns) { + const match = url.match(pattern) + if (match) { + // Extract the full matched domain (including subdomain if present) + const fullDomain = match[0] + const domainIndex = url.indexOf(fullDomain) + + if (domainIndex !== -1) { + // The URL is corrupted, reconstruct it + const protocol = url.includes(":443") || fallback.startsWith("https://") ? "https://" : "http://" + + // Check for path, query, and fragment after the domain + const afterDomain = url.substring(domainIndex + fullDomain.length) + + // Match optional port, then path/query/fragment + const afterMatch = afterDomain.match(/^(?::\d+)?(.*)$/) + const pathQueryFragment = afterMatch?.[1] || "" + + return protocol + fullDomain + pathQueryFragment + } + } + } + + return fallback +} + +export const getClerkBaseUrl = () => sanitizeUrl(process.env.CLERK_BASE_URL, PRODUCTION_CLERK_BASE_URL) + +export const getRooCodeApiUrl = () => sanitizeUrl(process.env.ROO_CODE_API_URL, PRODUCTION_ROO_CODE_API_URL)