diff --git a/src/services/code-index/vector-store/qdrant-client.ts b/src/services/code-index/vector-store/qdrant-client.ts index ae1156112b..5d77cc39fc 100644 --- a/src/services/code-index/vector-store/qdrant-client.ts +++ b/src/services/code-index/vector-store/qdrant-client.ts @@ -1,7 +1,6 @@ import { QdrantClient, Schemas } from "@qdrant/js-client-rest" import { createHash } from "crypto" import * as path from "path" -import { getWorkspacePath } from "../../../utils/path" import { IVectorStore } from "../interfaces/vector-store" import { Payload, VectorStoreSearchResult } from "../interfaces" import { MAX_SEARCH_RESULTS, SEARCH_MIN_SCORE } from "../constants" @@ -12,6 +11,7 @@ import { MAX_SEARCH_RESULTS, SEARCH_MIN_SCORE } from "../constants" export class QdrantVectorStore implements IVectorStore { private readonly vectorSize!: number private readonly DISTANCE_METRIC = "Cosine" + private readonly workspacePath: string private client: QdrantClient private readonly collectionName: string @@ -74,6 +74,9 @@ export class QdrantVectorStore implements IVectorStore { }) } + // Store workspace path + this.workspacePath = workspacePath + // Generate collection name from workspace path const hash = createHash("sha256").update(workspacePath).digest("hex") this.vectorSize = vectorSize @@ -332,9 +335,8 @@ export class QdrantVectorStore implements IVectorStore { } try { - const workspaceRoot = getWorkspacePath() const normalizedPaths = filePaths.map((filePath) => { - const absolutePath = path.resolve(workspaceRoot, filePath) + const absolutePath = path.resolve(this.workspacePath, filePath) return path.normalize(absolutePath) }) diff --git a/src/services/code-index/worker-utils/get-relative-path.ts b/src/services/code-index/worker-utils/get-relative-path.ts new file mode 100644 index 0000000000..a4c7fcb345 --- /dev/null +++ b/src/services/code-index/worker-utils/get-relative-path.ts @@ -0,0 +1,31 @@ +import path from "path" + +/** + * Generates a normalized absolute path from a given file path and workspace root. + * Handles path resolution and normalization to ensure consistent absolute paths. + * + * @param filePath - The file path to normalize (can be relative or absolute) + * @param workspaceRoot - The root directory of the workspace + * @returns The normalized absolute path + */ +export function generateNormalizedAbsolutePath(filePath: string, workspaceRoot: string): string { + // Resolve the path to make it absolute if it's relative + const resolvedPath = path.resolve(workspaceRoot, filePath) + // Normalize to handle any . or .. segments and duplicate slashes + return path.normalize(resolvedPath) +} + +/** + * Generates a relative file path from a normalized absolute path and workspace root. + * Ensures consistent relative path generation across different platforms. + * + * @param normalizedAbsolutePath - The normalized absolute path to convert + * @param workspaceRoot - The root directory of the workspace + * @returns The relative path from workspaceRoot to the file + */ +export function generateRelativeFilePath(normalizedAbsolutePath: string, workspaceRoot: string): string { + // Generate the relative path + const relativePath = path.relative(workspaceRoot, normalizedAbsolutePath) + // Normalize to ensure consistent path separators + return path.normalize(relativePath) +} diff --git a/src/services/code-index/worker-utils/parser.ts b/src/services/code-index/worker-utils/parser.ts new file mode 100644 index 0000000000..832e388a4e --- /dev/null +++ b/src/services/code-index/worker-utils/parser.ts @@ -0,0 +1,376 @@ +import { readFile } from "fs/promises" +import { createHash } from "crypto" +import * as path from "path" +import { Node } from "web-tree-sitter" +import { LanguageParser, loadRequiredLanguageParsers } from "../../tree-sitter/languageParser" +import { ICodeParser, CodeBlock } from "../interfaces" +import { scannerExtensions } from "./supported-extensions" +import { MAX_BLOCK_CHARS, MIN_BLOCK_CHARS, MIN_CHUNK_REMAINDER_CHARS, MAX_CHARS_TOLERANCE_FACTOR } from "../constants" + +/** + * Worker-compatible implementation of the code parser interface + */ +export class CodeParser implements ICodeParser { + private loadedParsers: LanguageParser = {} + private pendingLoads: Map> = new Map() + // Markdown files are excluded because the current parser logic cannot effectively handle + // potentially large Markdown sections without a tree-sitter-like child node structure for chunking + + /** + * Parses a code file into code blocks + * @param filePath Path to the file to parse + * @param options Optional parsing options + * @returns Promise resolving to array of code blocks + */ + async parseFile( + filePath: string, + options?: { + content?: string + fileHash?: string + }, + ): Promise { + // Get file extension + const ext = path.extname(filePath).toLowerCase() + + // Skip if not a supported language + if (!this.isSupportedLanguage(ext)) { + return [] + } + + // Get file content + let content: string + let fileHash: string + + if (options?.content) { + content = options.content + fileHash = options.fileHash || this.createFileHash(content) + } else { + try { + content = await readFile(filePath, "utf8") + fileHash = this.createFileHash(content) + } catch (error) { + console.error(`Error reading file ${filePath}:`, error) + return [] + } + } + + // Parse the file + return this.parseContent(filePath, content, fileHash) + } + + /** + * Checks if a language is supported + * @param extension File extension + * @returns Boolean indicating if the language is supported + */ + private isSupportedLanguage(extension: string): boolean { + return scannerExtensions.includes(extension) + } + + /** + * Creates a hash for a file + * @param content File content + * @returns Hash string + */ + private createFileHash(content: string): string { + return createHash("sha256").update(content).digest("hex") + } + + /** + * Parses file content into code blocks + * @param filePath Path to the file + * @param content File content + * @param fileHash File hash + * @returns Array of code blocks + */ + private async parseContent(filePath: string, content: string, fileHash: string): Promise { + const ext = path.extname(filePath).slice(1).toLowerCase() + const seenSegmentHashes = new Set() + + // Check if we already have the parser loaded + if (!this.loadedParsers[ext]) { + const pendingLoad = this.pendingLoads.get(ext) + if (pendingLoad) { + try { + await pendingLoad + } catch (error) { + console.error(`Error in pending parser load for ${filePath}:`, error) + return [] + } + } else { + const loadPromise = loadRequiredLanguageParsers([filePath]) + this.pendingLoads.set(ext, loadPromise) + try { + const newParsers = await loadPromise + if (newParsers) { + this.loadedParsers = { ...this.loadedParsers, ...newParsers } + } + } catch (error) { + console.error(`Error loading language parser for ${filePath}:`, error) + return [] + } finally { + this.pendingLoads.delete(ext) + } + } + } + + const language = this.loadedParsers[ext] + if (!language) { + console.warn(`No parser available for file extension: ${ext}`) + return [] + } + + const tree = language.parser.parse(content) + + // We don't need to get the query string from languageQueries since it's already loaded + // in the language object + const captures = tree ? language.query.captures(tree.rootNode) : [] + + // Check if captures are empty + if (captures.length === 0) { + if (content.length >= MIN_BLOCK_CHARS) { + // Perform fallback chunking if content is large enough + const blocks = this._performFallbackChunking(filePath, content, fileHash, seenSegmentHashes) + return blocks + } else { + // Return empty if content is too small for fallback + return [] + } + } + + const results: CodeBlock[] = [] + + // Process captures if not empty + const queue: Node[] = Array.from(captures).map((capture) => capture.node) + + while (queue.length > 0) { + const currentNode = queue.shift()! + // const lineSpan = currentNode.endPosition.row - currentNode.startPosition.row + 1 // Removed as per lint error + + // Check if the node meets the minimum character requirement + if (currentNode.text.length >= MIN_BLOCK_CHARS) { + // If it also exceeds the maximum character limit, try to break it down + if (currentNode.text.length > MAX_BLOCK_CHARS * MAX_CHARS_TOLERANCE_FACTOR) { + if (currentNode.children.filter((child) => child !== null).length > 0) { + // If it has children, process them instead + queue.push(...currentNode.children.filter((child) => child !== null)) + } else { + // If it's a leaf node, chunk it (passing MIN_BLOCK_CHARS as per Task 1 Step 5) + // Note: _chunkLeafNodeByLines logic might need further adjustment later + const chunkedBlocks = this._chunkLeafNodeByLines( + currentNode, + filePath, + fileHash, + seenSegmentHashes, + ) + results.push(...chunkedBlocks) + } + } else { + // Node meets min chars and is within max chars, create a block + const identifier = + currentNode.childForFieldName("name")?.text || + currentNode.children.find((c) => c?.type === "identifier")?.text || + null + const type = currentNode.type + const start_line = currentNode.startPosition.row + 1 + const end_line = currentNode.endPosition.row + 1 + const content = currentNode.text + const segmentHash = createHash("sha256") + .update(`${filePath}-${start_line}-${end_line}-${content}`) + .digest("hex") + + if (!seenSegmentHashes.has(segmentHash)) { + seenSegmentHashes.add(segmentHash) + results.push({ + file_path: filePath, + identifier, + type, + start_line, + end_line, + content, + segmentHash, + fileHash, + }) + } + } + } + // Nodes smaller than MIN_BLOCK_CHARS are ignored + } + + return results + } + + /** + * Common helper function to chunk text by lines, avoiding tiny remainders. + */ + private _chunkTextByLines( + lines: string[], + filePath: string, + fileHash: string, + + chunkType: string, + seenSegmentHashes: Set, + baseStartLine: number = 1, // 1-based start line of the *first* line in the `lines` array + ): CodeBlock[] { + const chunks: CodeBlock[] = [] + let currentChunkLines: string[] = [] + let currentChunkLength = 0 + let chunkStartLineIndex = 0 // 0-based index within the `lines` array + const effectiveMaxChars = MAX_BLOCK_CHARS * MAX_CHARS_TOLERANCE_FACTOR + + const finalizeChunk = (endLineIndex: number) => { + if (currentChunkLength >= MIN_BLOCK_CHARS && currentChunkLines.length > 0) { + const chunkContent = currentChunkLines.join("\n") + const startLine = baseStartLine + chunkStartLineIndex + const endLine = baseStartLine + endLineIndex + const segmentHash = createHash("sha256") + .update(`${filePath}-${startLine}-${endLine}-${chunkContent}`) + .digest("hex") + + if (!seenSegmentHashes.has(segmentHash)) { + seenSegmentHashes.add(segmentHash) + chunks.push({ + file_path: filePath, + identifier: null, + type: chunkType, + start_line: startLine, + end_line: endLine, + content: chunkContent, + segmentHash, + fileHash, + }) + } + } + currentChunkLines = [] + currentChunkLength = 0 + chunkStartLineIndex = endLineIndex + 1 + } + + const createSegmentBlock = (segment: string, originalLineNumber: number, startCharIndex: number) => { + const segmentHash = createHash("sha256") + .update(`${filePath}-${originalLineNumber}-${originalLineNumber}-${startCharIndex}-${segment}`) + .digest("hex") + + if (!seenSegmentHashes.has(segmentHash)) { + seenSegmentHashes.add(segmentHash) + chunks.push({ + file_path: filePath, + identifier: null, + type: `${chunkType}_segment`, + start_line: originalLineNumber, + end_line: originalLineNumber, + content: segment, + segmentHash, + fileHash, + }) + } + } + + for (let i = 0; i < lines.length; i++) { + const line = lines[i] + const lineLength = line.length + (i < lines.length - 1 ? 1 : 0) // +1 for newline, except last line + const originalLineNumber = baseStartLine + i + + // Handle oversized lines (longer than effectiveMaxChars) + if (lineLength > effectiveMaxChars) { + // Finalize any existing normal chunk before processing the oversized line + if (currentChunkLines.length > 0) { + finalizeChunk(i - 1) + } + + // Split the oversized line into segments + let remainingLineContent = line + let currentSegmentStartChar = 0 + while (remainingLineContent.length > 0) { + const segment = remainingLineContent.substring(0, MAX_BLOCK_CHARS) + remainingLineContent = remainingLineContent.substring(MAX_BLOCK_CHARS) + createSegmentBlock(segment, originalLineNumber, currentSegmentStartChar) + currentSegmentStartChar += MAX_BLOCK_CHARS + } + continue + } + + // Handle normally sized lines + if (currentChunkLength > 0 && currentChunkLength + lineLength > effectiveMaxChars) { + // Re-balancing Logic + let splitIndex = i - 1 + let remainderLength = 0 + for (let j = i; j < lines.length; j++) { + remainderLength += lines[j].length + (j < lines.length - 1 ? 1 : 0) + } + + if ( + currentChunkLength >= MIN_BLOCK_CHARS && + remainderLength < MIN_CHUNK_REMAINDER_CHARS && + currentChunkLines.length > 1 + ) { + for (let k = i - 2; k >= chunkStartLineIndex; k--) { + const potentialChunkLines = lines.slice(chunkStartLineIndex, k + 1) + const potentialChunkLength = potentialChunkLines.join("\n").length + 1 + const potentialNextChunkLines = lines.slice(k + 1) + const potentialNextChunkLength = potentialNextChunkLines.join("\n").length + 1 + + if ( + potentialChunkLength >= MIN_BLOCK_CHARS && + potentialNextChunkLength >= MIN_CHUNK_REMAINDER_CHARS + ) { + splitIndex = k + break + } + } + } + + finalizeChunk(splitIndex) + + if (i >= chunkStartLineIndex) { + currentChunkLines.push(line) + currentChunkLength += lineLength + } else { + i = chunkStartLineIndex - 1 + continue + } + } else { + currentChunkLines.push(line) + currentChunkLength += lineLength + } + } + + // Process the last remaining chunk + if (currentChunkLines.length > 0) { + finalizeChunk(lines.length - 1) + } + + return chunks + } + + private _performFallbackChunking( + filePath: string, + content: string, + fileHash: string, + seenSegmentHashes: Set, + ): CodeBlock[] { + const lines = content.split("\n") + return this._chunkTextByLines(lines, filePath, fileHash, "fallback_chunk", seenSegmentHashes) + } + + private _chunkLeafNodeByLines( + node: Node, + filePath: string, + fileHash: string, + seenSegmentHashes: Set, + ): CodeBlock[] { + const lines = node.text.split("\n") + const baseStartLine = node.startPosition.row + 1 + return this._chunkTextByLines( + lines, + filePath, + fileHash, + node.type, // Use the node's type + seenSegmentHashes, + baseStartLine, + ) + } +} + +// Export a singleton instance for convenience +export const codeParser = new CodeParser() diff --git a/src/services/code-index/worker-utils/supported-extensions.ts b/src/services/code-index/worker-utils/supported-extensions.ts new file mode 100644 index 0000000000..1385b4df7e --- /dev/null +++ b/src/services/code-index/worker-utils/supported-extensions.ts @@ -0,0 +1,69 @@ +// Worker-compatible version of supported extensions +// Defines extensions directly without importing from tree-sitter + +const allExtensions = [ + ".tla", + ".js", + ".jsx", + ".ts", + ".vue", + ".tsx", + ".py", + // Rust + ".rs", + ".go", + // C + ".c", + ".h", + // C++ + ".cpp", + ".hpp", + // C# + ".cs", + // Ruby + ".rb", + ".java", + ".php", + ".swift", + // Solidity + ".sol", + // Kotlin + ".kt", + ".kts", + // Elixir + ".ex", + ".exs", + // Elisp + ".el", + // HTML + ".html", + ".htm", + // Markdown + ".md", + ".markdown", + // JSON + ".json", + // CSS + ".css", + // SystemRDL + ".rdl", + // OCaml + ".ml", + ".mli", + // Lua + ".lua", + // Scala + ".scala", + // TOML + ".toml", + // Zig + ".zig", + // Elm + ".elm", + // Embedded Template + ".ejs", + ".erb", +] + +// Filter out markdown extensions for the scanner +export const scannerExtensions = allExtensions.filter((ext) => ext !== ".md" && ext !== ".markdown") diff --git a/src/workers/indexing-worker.ts b/src/workers/indexing-worker.ts index 7fdcd65da6..d771870d2f 100644 --- a/src/workers/indexing-worker.ts +++ b/src/workers/indexing-worker.ts @@ -15,7 +15,7 @@ import { OpenAiEmbedder } from "../services/code-index/embedders/openai" import { CodeIndexOllamaEmbedder } from "../services/code-index/embedders/ollama" import { OpenAICompatibleEmbedder } from "../services/code-index/embedders/openai-compatible" import { QdrantVectorStore } from "../services/code-index/vector-store/qdrant-client" -import { codeParser } from "../services/code-index/processors" +import { codeParser } from "../services/code-index/worker-utils/parser" import { EmbedderProvider, getDefaultModelId, getModelDimension } from "../shared/embeddingModels" class IndexingWorker {