From 251e250935585962065aaedd8097c33ef56b023e Mon Sep 17 00:00:00 2001 From: vimzh Date: Sun, 24 May 2026 12:42:28 +0530 Subject: [PATCH] fix(onboarding): reject non-X hosts and anchor regex in extractHandle Copilot review flagged that 'lower.includes("x.com")' and the regex fallback match any substring of the input, so a URL like 'https://examplex.com/foo' or 'notwitter.com/foo' would parse, extract 'foo' as a handle, pass the 1-15 alphanumeric guard, and burn a Grok call on a wrong handle. Add an isXHost helper that checks the parsed hostname against x.com / twitter.com (and their subdomains) instead of relying on substring matches. Anchor the regex fallback to '^', '.', or '/' before the domain so the same substring trap does not apply there either. --- apps/web/app/api/onboarding/research/route.ts | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/apps/web/app/api/onboarding/research/route.ts b/apps/web/app/api/onboarding/research/route.ts index 47298303..3ebc8bc8 100644 --- a/apps/web/app/api/onboarding/research/route.ts +++ b/apps/web/app/api/onboarding/research/route.ts @@ -7,6 +7,16 @@ interface ResearchRequest { email?: string } +function isXHost(hostname: string): boolean { + const host = hostname.toLowerCase() + return ( + host === "x.com" || + host === "twitter.com" || + host.endsWith(".x.com") || + host.endsWith(".twitter.com") + ) +} + function extractHandle(input: string): string { const trimmed = input.trim() if (!trimmed) return "" @@ -21,9 +31,13 @@ function extractHandle(input: string): string { ? handle : `https://${handle}`, ) - handle = parsed.pathname.split("/").filter(Boolean)[0] ?? "" + handle = isXHost(parsed.hostname) + ? (parsed.pathname.split("/").filter(Boolean)[0] ?? "") + : "" } catch { - handle = handle.match(/(?:x\.com|twitter\.com)\/([^/\s?#]+)/i)?.[1] ?? "" + handle = + handle.match(/(?:^|[./])(?:x\.com|twitter\.com)\/([^/\s?#]+)/i)?.[1] ?? + "" } }