mirror of
https://github.com/RooVetGit/Roo-Code.git
synced 2026-09-05 08:10:14 +00:00
fix: create worker-compatible parser and supported-extensions modules
- Created worker-utils/parser.ts that imports from worker-utils/supported-extensions - Created worker-utils/supported-extensions.ts with extensions defined directly - Updated indexing-worker to import codeParser from worker-utils version - This eliminates all vscode dependencies from the worker bundle
This commit is contained in:
parent
cb4bb8476f
commit
55d3e4f168
5 changed files with 482 additions and 4 deletions
|
|
@ -1,7 +1,6 @@
|
|||
import { QdrantClient, Schemas } from "@qdrant/js-client-rest"
|
||||
import { createHash } from "crypto"
|
||||
import * as path from "path"
|
||||
import { getWorkspacePath } from "../../../utils/path"
|
||||
import { IVectorStore } from "../interfaces/vector-store"
|
||||
import { Payload, VectorStoreSearchResult } from "../interfaces"
|
||||
import { MAX_SEARCH_RESULTS, SEARCH_MIN_SCORE } from "../constants"
|
||||
|
|
@ -12,6 +11,7 @@ import { MAX_SEARCH_RESULTS, SEARCH_MIN_SCORE } from "../constants"
|
|||
export class QdrantVectorStore implements IVectorStore {
|
||||
private readonly vectorSize!: number
|
||||
private readonly DISTANCE_METRIC = "Cosine"
|
||||
private readonly workspacePath: string
|
||||
|
||||
private client: QdrantClient
|
||||
private readonly collectionName: string
|
||||
|
|
@ -74,6 +74,9 @@ export class QdrantVectorStore implements IVectorStore {
|
|||
})
|
||||
}
|
||||
|
||||
// Store workspace path
|
||||
this.workspacePath = workspacePath
|
||||
|
||||
// Generate collection name from workspace path
|
||||
const hash = createHash("sha256").update(workspacePath).digest("hex")
|
||||
this.vectorSize = vectorSize
|
||||
|
|
@ -332,9 +335,8 @@ export class QdrantVectorStore implements IVectorStore {
|
|||
}
|
||||
|
||||
try {
|
||||
const workspaceRoot = getWorkspacePath()
|
||||
const normalizedPaths = filePaths.map((filePath) => {
|
||||
const absolutePath = path.resolve(workspaceRoot, filePath)
|
||||
const absolutePath = path.resolve(this.workspacePath, filePath)
|
||||
return path.normalize(absolutePath)
|
||||
})
|
||||
|
||||
|
|
|
|||
31
src/services/code-index/worker-utils/get-relative-path.ts
Normal file
31
src/services/code-index/worker-utils/get-relative-path.ts
Normal file
|
|
@ -0,0 +1,31 @@
|
|||
import path from "path"
|
||||
|
||||
/**
|
||||
* Generates a normalized absolute path from a given file path and workspace root.
|
||||
* Handles path resolution and normalization to ensure consistent absolute paths.
|
||||
*
|
||||
* @param filePath - The file path to normalize (can be relative or absolute)
|
||||
* @param workspaceRoot - The root directory of the workspace
|
||||
* @returns The normalized absolute path
|
||||
*/
|
||||
export function generateNormalizedAbsolutePath(filePath: string, workspaceRoot: string): string {
|
||||
// Resolve the path to make it absolute if it's relative
|
||||
const resolvedPath = path.resolve(workspaceRoot, filePath)
|
||||
// Normalize to handle any . or .. segments and duplicate slashes
|
||||
return path.normalize(resolvedPath)
|
||||
}
|
||||
|
||||
/**
|
||||
* Generates a relative file path from a normalized absolute path and workspace root.
|
||||
* Ensures consistent relative path generation across different platforms.
|
||||
*
|
||||
* @param normalizedAbsolutePath - The normalized absolute path to convert
|
||||
* @param workspaceRoot - The root directory of the workspace
|
||||
* @returns The relative path from workspaceRoot to the file
|
||||
*/
|
||||
export function generateRelativeFilePath(normalizedAbsolutePath: string, workspaceRoot: string): string {
|
||||
// Generate the relative path
|
||||
const relativePath = path.relative(workspaceRoot, normalizedAbsolutePath)
|
||||
// Normalize to ensure consistent path separators
|
||||
return path.normalize(relativePath)
|
||||
}
|
||||
376
src/services/code-index/worker-utils/parser.ts
Normal file
376
src/services/code-index/worker-utils/parser.ts
Normal file
|
|
@ -0,0 +1,376 @@
|
|||
import { readFile } from "fs/promises"
|
||||
import { createHash } from "crypto"
|
||||
import * as path from "path"
|
||||
import { Node } from "web-tree-sitter"
|
||||
import { LanguageParser, loadRequiredLanguageParsers } from "../../tree-sitter/languageParser"
|
||||
import { ICodeParser, CodeBlock } from "../interfaces"
|
||||
import { scannerExtensions } from "./supported-extensions"
|
||||
import { MAX_BLOCK_CHARS, MIN_BLOCK_CHARS, MIN_CHUNK_REMAINDER_CHARS, MAX_CHARS_TOLERANCE_FACTOR } from "../constants"
|
||||
|
||||
/**
|
||||
* Worker-compatible implementation of the code parser interface
|
||||
*/
|
||||
export class CodeParser implements ICodeParser {
|
||||
private loadedParsers: LanguageParser = {}
|
||||
private pendingLoads: Map<string, Promise<LanguageParser>> = new Map()
|
||||
// Markdown files are excluded because the current parser logic cannot effectively handle
|
||||
// potentially large Markdown sections without a tree-sitter-like child node structure for chunking
|
||||
|
||||
/**
|
||||
* Parses a code file into code blocks
|
||||
* @param filePath Path to the file to parse
|
||||
* @param options Optional parsing options
|
||||
* @returns Promise resolving to array of code blocks
|
||||
*/
|
||||
async parseFile(
|
||||
filePath: string,
|
||||
options?: {
|
||||
content?: string
|
||||
fileHash?: string
|
||||
},
|
||||
): Promise<CodeBlock[]> {
|
||||
// Get file extension
|
||||
const ext = path.extname(filePath).toLowerCase()
|
||||
|
||||
// Skip if not a supported language
|
||||
if (!this.isSupportedLanguage(ext)) {
|
||||
return []
|
||||
}
|
||||
|
||||
// Get file content
|
||||
let content: string
|
||||
let fileHash: string
|
||||
|
||||
if (options?.content) {
|
||||
content = options.content
|
||||
fileHash = options.fileHash || this.createFileHash(content)
|
||||
} else {
|
||||
try {
|
||||
content = await readFile(filePath, "utf8")
|
||||
fileHash = this.createFileHash(content)
|
||||
} catch (error) {
|
||||
console.error(`Error reading file ${filePath}:`, error)
|
||||
return []
|
||||
}
|
||||
}
|
||||
|
||||
// Parse the file
|
||||
return this.parseContent(filePath, content, fileHash)
|
||||
}
|
||||
|
||||
/**
|
||||
* Checks if a language is supported
|
||||
* @param extension File extension
|
||||
* @returns Boolean indicating if the language is supported
|
||||
*/
|
||||
private isSupportedLanguage(extension: string): boolean {
|
||||
return scannerExtensions.includes(extension)
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates a hash for a file
|
||||
* @param content File content
|
||||
* @returns Hash string
|
||||
*/
|
||||
private createFileHash(content: string): string {
|
||||
return createHash("sha256").update(content).digest("hex")
|
||||
}
|
||||
|
||||
/**
|
||||
* Parses file content into code blocks
|
||||
* @param filePath Path to the file
|
||||
* @param content File content
|
||||
* @param fileHash File hash
|
||||
* @returns Array of code blocks
|
||||
*/
|
||||
private async parseContent(filePath: string, content: string, fileHash: string): Promise<CodeBlock[]> {
|
||||
const ext = path.extname(filePath).slice(1).toLowerCase()
|
||||
const seenSegmentHashes = new Set<string>()
|
||||
|
||||
// Check if we already have the parser loaded
|
||||
if (!this.loadedParsers[ext]) {
|
||||
const pendingLoad = this.pendingLoads.get(ext)
|
||||
if (pendingLoad) {
|
||||
try {
|
||||
await pendingLoad
|
||||
} catch (error) {
|
||||
console.error(`Error in pending parser load for ${filePath}:`, error)
|
||||
return []
|
||||
}
|
||||
} else {
|
||||
const loadPromise = loadRequiredLanguageParsers([filePath])
|
||||
this.pendingLoads.set(ext, loadPromise)
|
||||
try {
|
||||
const newParsers = await loadPromise
|
||||
if (newParsers) {
|
||||
this.loadedParsers = { ...this.loadedParsers, ...newParsers }
|
||||
}
|
||||
} catch (error) {
|
||||
console.error(`Error loading language parser for ${filePath}:`, error)
|
||||
return []
|
||||
} finally {
|
||||
this.pendingLoads.delete(ext)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const language = this.loadedParsers[ext]
|
||||
if (!language) {
|
||||
console.warn(`No parser available for file extension: ${ext}`)
|
||||
return []
|
||||
}
|
||||
|
||||
const tree = language.parser.parse(content)
|
||||
|
||||
// We don't need to get the query string from languageQueries since it's already loaded
|
||||
// in the language object
|
||||
const captures = tree ? language.query.captures(tree.rootNode) : []
|
||||
|
||||
// Check if captures are empty
|
||||
if (captures.length === 0) {
|
||||
if (content.length >= MIN_BLOCK_CHARS) {
|
||||
// Perform fallback chunking if content is large enough
|
||||
const blocks = this._performFallbackChunking(filePath, content, fileHash, seenSegmentHashes)
|
||||
return blocks
|
||||
} else {
|
||||
// Return empty if content is too small for fallback
|
||||
return []
|
||||
}
|
||||
}
|
||||
|
||||
const results: CodeBlock[] = []
|
||||
|
||||
// Process captures if not empty
|
||||
const queue: Node[] = Array.from(captures).map((capture) => capture.node)
|
||||
|
||||
while (queue.length > 0) {
|
||||
const currentNode = queue.shift()!
|
||||
// const lineSpan = currentNode.endPosition.row - currentNode.startPosition.row + 1 // Removed as per lint error
|
||||
|
||||
// Check if the node meets the minimum character requirement
|
||||
if (currentNode.text.length >= MIN_BLOCK_CHARS) {
|
||||
// If it also exceeds the maximum character limit, try to break it down
|
||||
if (currentNode.text.length > MAX_BLOCK_CHARS * MAX_CHARS_TOLERANCE_FACTOR) {
|
||||
if (currentNode.children.filter((child) => child !== null).length > 0) {
|
||||
// If it has children, process them instead
|
||||
queue.push(...currentNode.children.filter((child) => child !== null))
|
||||
} else {
|
||||
// If it's a leaf node, chunk it (passing MIN_BLOCK_CHARS as per Task 1 Step 5)
|
||||
// Note: _chunkLeafNodeByLines logic might need further adjustment later
|
||||
const chunkedBlocks = this._chunkLeafNodeByLines(
|
||||
currentNode,
|
||||
filePath,
|
||||
fileHash,
|
||||
seenSegmentHashes,
|
||||
)
|
||||
results.push(...chunkedBlocks)
|
||||
}
|
||||
} else {
|
||||
// Node meets min chars and is within max chars, create a block
|
||||
const identifier =
|
||||
currentNode.childForFieldName("name")?.text ||
|
||||
currentNode.children.find((c) => c?.type === "identifier")?.text ||
|
||||
null
|
||||
const type = currentNode.type
|
||||
const start_line = currentNode.startPosition.row + 1
|
||||
const end_line = currentNode.endPosition.row + 1
|
||||
const content = currentNode.text
|
||||
const segmentHash = createHash("sha256")
|
||||
.update(`${filePath}-${start_line}-${end_line}-${content}`)
|
||||
.digest("hex")
|
||||
|
||||
if (!seenSegmentHashes.has(segmentHash)) {
|
||||
seenSegmentHashes.add(segmentHash)
|
||||
results.push({
|
||||
file_path: filePath,
|
||||
identifier,
|
||||
type,
|
||||
start_line,
|
||||
end_line,
|
||||
content,
|
||||
segmentHash,
|
||||
fileHash,
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
// Nodes smaller than MIN_BLOCK_CHARS are ignored
|
||||
}
|
||||
|
||||
return results
|
||||
}
|
||||
|
||||
/**
|
||||
* Common helper function to chunk text by lines, avoiding tiny remainders.
|
||||
*/
|
||||
private _chunkTextByLines(
|
||||
lines: string[],
|
||||
filePath: string,
|
||||
fileHash: string,
|
||||
|
||||
chunkType: string,
|
||||
seenSegmentHashes: Set<string>,
|
||||
baseStartLine: number = 1, // 1-based start line of the *first* line in the `lines` array
|
||||
): CodeBlock[] {
|
||||
const chunks: CodeBlock[] = []
|
||||
let currentChunkLines: string[] = []
|
||||
let currentChunkLength = 0
|
||||
let chunkStartLineIndex = 0 // 0-based index within the `lines` array
|
||||
const effectiveMaxChars = MAX_BLOCK_CHARS * MAX_CHARS_TOLERANCE_FACTOR
|
||||
|
||||
const finalizeChunk = (endLineIndex: number) => {
|
||||
if (currentChunkLength >= MIN_BLOCK_CHARS && currentChunkLines.length > 0) {
|
||||
const chunkContent = currentChunkLines.join("\n")
|
||||
const startLine = baseStartLine + chunkStartLineIndex
|
||||
const endLine = baseStartLine + endLineIndex
|
||||
const segmentHash = createHash("sha256")
|
||||
.update(`${filePath}-${startLine}-${endLine}-${chunkContent}`)
|
||||
.digest("hex")
|
||||
|
||||
if (!seenSegmentHashes.has(segmentHash)) {
|
||||
seenSegmentHashes.add(segmentHash)
|
||||
chunks.push({
|
||||
file_path: filePath,
|
||||
identifier: null,
|
||||
type: chunkType,
|
||||
start_line: startLine,
|
||||
end_line: endLine,
|
||||
content: chunkContent,
|
||||
segmentHash,
|
||||
fileHash,
|
||||
})
|
||||
}
|
||||
}
|
||||
currentChunkLines = []
|
||||
currentChunkLength = 0
|
||||
chunkStartLineIndex = endLineIndex + 1
|
||||
}
|
||||
|
||||
const createSegmentBlock = (segment: string, originalLineNumber: number, startCharIndex: number) => {
|
||||
const segmentHash = createHash("sha256")
|
||||
.update(`${filePath}-${originalLineNumber}-${originalLineNumber}-${startCharIndex}-${segment}`)
|
||||
.digest("hex")
|
||||
|
||||
if (!seenSegmentHashes.has(segmentHash)) {
|
||||
seenSegmentHashes.add(segmentHash)
|
||||
chunks.push({
|
||||
file_path: filePath,
|
||||
identifier: null,
|
||||
type: `${chunkType}_segment`,
|
||||
start_line: originalLineNumber,
|
||||
end_line: originalLineNumber,
|
||||
content: segment,
|
||||
segmentHash,
|
||||
fileHash,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const line = lines[i]
|
||||
const lineLength = line.length + (i < lines.length - 1 ? 1 : 0) // +1 for newline, except last line
|
||||
const originalLineNumber = baseStartLine + i
|
||||
|
||||
// Handle oversized lines (longer than effectiveMaxChars)
|
||||
if (lineLength > effectiveMaxChars) {
|
||||
// Finalize any existing normal chunk before processing the oversized line
|
||||
if (currentChunkLines.length > 0) {
|
||||
finalizeChunk(i - 1)
|
||||
}
|
||||
|
||||
// Split the oversized line into segments
|
||||
let remainingLineContent = line
|
||||
let currentSegmentStartChar = 0
|
||||
while (remainingLineContent.length > 0) {
|
||||
const segment = remainingLineContent.substring(0, MAX_BLOCK_CHARS)
|
||||
remainingLineContent = remainingLineContent.substring(MAX_BLOCK_CHARS)
|
||||
createSegmentBlock(segment, originalLineNumber, currentSegmentStartChar)
|
||||
currentSegmentStartChar += MAX_BLOCK_CHARS
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Handle normally sized lines
|
||||
if (currentChunkLength > 0 && currentChunkLength + lineLength > effectiveMaxChars) {
|
||||
// Re-balancing Logic
|
||||
let splitIndex = i - 1
|
||||
let remainderLength = 0
|
||||
for (let j = i; j < lines.length; j++) {
|
||||
remainderLength += lines[j].length + (j < lines.length - 1 ? 1 : 0)
|
||||
}
|
||||
|
||||
if (
|
||||
currentChunkLength >= MIN_BLOCK_CHARS &&
|
||||
remainderLength < MIN_CHUNK_REMAINDER_CHARS &&
|
||||
currentChunkLines.length > 1
|
||||
) {
|
||||
for (let k = i - 2; k >= chunkStartLineIndex; k--) {
|
||||
const potentialChunkLines = lines.slice(chunkStartLineIndex, k + 1)
|
||||
const potentialChunkLength = potentialChunkLines.join("\n").length + 1
|
||||
const potentialNextChunkLines = lines.slice(k + 1)
|
||||
const potentialNextChunkLength = potentialNextChunkLines.join("\n").length + 1
|
||||
|
||||
if (
|
||||
potentialChunkLength >= MIN_BLOCK_CHARS &&
|
||||
potentialNextChunkLength >= MIN_CHUNK_REMAINDER_CHARS
|
||||
) {
|
||||
splitIndex = k
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
finalizeChunk(splitIndex)
|
||||
|
||||
if (i >= chunkStartLineIndex) {
|
||||
currentChunkLines.push(line)
|
||||
currentChunkLength += lineLength
|
||||
} else {
|
||||
i = chunkStartLineIndex - 1
|
||||
continue
|
||||
}
|
||||
} else {
|
||||
currentChunkLines.push(line)
|
||||
currentChunkLength += lineLength
|
||||
}
|
||||
}
|
||||
|
||||
// Process the last remaining chunk
|
||||
if (currentChunkLines.length > 0) {
|
||||
finalizeChunk(lines.length - 1)
|
||||
}
|
||||
|
||||
return chunks
|
||||
}
|
||||
|
||||
private _performFallbackChunking(
|
||||
filePath: string,
|
||||
content: string,
|
||||
fileHash: string,
|
||||
seenSegmentHashes: Set<string>,
|
||||
): CodeBlock[] {
|
||||
const lines = content.split("\n")
|
||||
return this._chunkTextByLines(lines, filePath, fileHash, "fallback_chunk", seenSegmentHashes)
|
||||
}
|
||||
|
||||
private _chunkLeafNodeByLines(
|
||||
node: Node,
|
||||
filePath: string,
|
||||
fileHash: string,
|
||||
seenSegmentHashes: Set<string>,
|
||||
): CodeBlock[] {
|
||||
const lines = node.text.split("\n")
|
||||
const baseStartLine = node.startPosition.row + 1
|
||||
return this._chunkTextByLines(
|
||||
lines,
|
||||
filePath,
|
||||
fileHash,
|
||||
node.type, // Use the node's type
|
||||
seenSegmentHashes,
|
||||
baseStartLine,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// Export a singleton instance for convenience
|
||||
export const codeParser = new CodeParser()
|
||||
69
src/services/code-index/worker-utils/supported-extensions.ts
Normal file
69
src/services/code-index/worker-utils/supported-extensions.ts
Normal file
|
|
@ -0,0 +1,69 @@
|
|||
// Worker-compatible version of supported extensions
|
||||
// Defines extensions directly without importing from tree-sitter
|
||||
|
||||
const allExtensions = [
|
||||
".tla",
|
||||
".js",
|
||||
".jsx",
|
||||
".ts",
|
||||
".vue",
|
||||
".tsx",
|
||||
".py",
|
||||
// Rust
|
||||
".rs",
|
||||
".go",
|
||||
// C
|
||||
".c",
|
||||
".h",
|
||||
// C++
|
||||
".cpp",
|
||||
".hpp",
|
||||
// C#
|
||||
".cs",
|
||||
// Ruby
|
||||
".rb",
|
||||
".java",
|
||||
".php",
|
||||
".swift",
|
||||
// Solidity
|
||||
".sol",
|
||||
// Kotlin
|
||||
".kt",
|
||||
".kts",
|
||||
// Elixir
|
||||
".ex",
|
||||
".exs",
|
||||
// Elisp
|
||||
".el",
|
||||
// HTML
|
||||
".html",
|
||||
".htm",
|
||||
// Markdown
|
||||
".md",
|
||||
".markdown",
|
||||
// JSON
|
||||
".json",
|
||||
// CSS
|
||||
".css",
|
||||
// SystemRDL
|
||||
".rdl",
|
||||
// OCaml
|
||||
".ml",
|
||||
".mli",
|
||||
// Lua
|
||||
".lua",
|
||||
// Scala
|
||||
".scala",
|
||||
// TOML
|
||||
".toml",
|
||||
// Zig
|
||||
".zig",
|
||||
// Elm
|
||||
".elm",
|
||||
// Embedded Template
|
||||
".ejs",
|
||||
".erb",
|
||||
]
|
||||
|
||||
// Filter out markdown extensions for the scanner
|
||||
export const scannerExtensions = allExtensions.filter((ext) => ext !== ".md" && ext !== ".markdown")
|
||||
|
|
@ -15,7 +15,7 @@ import { OpenAiEmbedder } from "../services/code-index/embedders/openai"
|
|||
import { CodeIndexOllamaEmbedder } from "../services/code-index/embedders/ollama"
|
||||
import { OpenAICompatibleEmbedder } from "../services/code-index/embedders/openai-compatible"
|
||||
import { QdrantVectorStore } from "../services/code-index/vector-store/qdrant-client"
|
||||
import { codeParser } from "../services/code-index/processors"
|
||||
import { codeParser } from "../services/code-index/worker-utils/parser"
|
||||
import { EmbedderProvider, getDefaultModelId, getModelDimension } from "../shared/embeddingModels"
|
||||
|
||||
class IndexingWorker {
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue