fix(onboarding): reject non-X hosts and anchor regex in extractHandle

Copilot review flagged that 'lower.includes("x.com")' and the regex
fallback match any substring of the input, so a URL like
'https://examplex.com/foo' or 'notwitter.com/foo' would parse, extract
'foo' as a handle, pass the 1-15 alphanumeric guard, and burn a Grok
call on a wrong handle.

Add an isXHost helper that checks the parsed hostname against x.com /
twitter.com (and their subdomains) instead of relying on substring
matches. Anchor the regex fallback to '^', '.', or '/' before the
domain so the same substring trap does not apply there either.
This commit is contained in:
vimzh 2026-05-24 12:42:28 +05:30
parent 923ed4e1f2
commit 251e250935

View file

@ -7,6 +7,16 @@ interface ResearchRequest {
email?: string
}
function isXHost(hostname: string): boolean {
const host = hostname.toLowerCase()
return (
host === "x.com" ||
host === "twitter.com" ||
host.endsWith(".x.com") ||
host.endsWith(".twitter.com")
)
}
function extractHandle(input: string): string {
const trimmed = input.trim()
if (!trimmed) return ""
@ -21,9 +31,13 @@ function extractHandle(input: string): string {
? handle
: `https://${handle}`,
)
handle = parsed.pathname.split("/").filter(Boolean)[0] ?? ""
handle = isXHost(parsed.hostname)
? (parsed.pathname.split("/").filter(Boolean)[0] ?? "")
: ""
} catch {
handle = handle.match(/(?:x\.com|twitter\.com)\/([^/\s?#]+)/i)?.[1] ?? ""
handle =
handle.match(/(?:^|[./])(?:x\.com|twitter\.com)\/([^/\s?#]+)/i)?.[1] ??
""
}
}