From 57ac6face56b1cf043022aca52956a331fb12833 Mon Sep 17 00:00:00 2001 From: abhigyantrumio Date: Tue, 5 Aug 2025 09:12:42 +0530 Subject: [PATCH] Advance edge case handling for python codebases ( eg: decorators, similar function name, etc --- repomix-output.txt | 14244 ++++++++++++++++++++++ src/core/ingestion/call-processor.ts | 784 +- src/core/ingestion/parsing-processor.ts | 293 +- src/ui/pages/HomePage.tsx | 127 +- 4 files changed, 15432 insertions(+), 16 deletions(-) create mode 100644 repomix-output.txt diff --git a/repomix-output.txt b/repomix-output.txt new file mode 100644 index 000000000..6f62e8a91 --- /dev/null +++ b/repomix-output.txt @@ -0,0 +1,14244 @@ +This file is a merged representation of the entire codebase, combining all repository files into a single document. +Generated by Repomix on: 2025-08-05T03:05:30.093Z + +================================================================ +File Summary +================================================================ + +Purpose: +-------- +This file contains a packed representation of the entire repository's contents. +It is designed to be easily consumable by AI systems for analysis, code review, +or other automated processes. + +File Format: +------------ +The content is organized as follows: +1. This summary section +2. Repository information +3. Directory structure +4. Multiple file entries, each consisting of: + a. A separator line (================) + b. The file path (File: path/to/file) + c. Another separator line + d. The full contents of the file + e. A blank line + +Usage Guidelines: +----------------- +- This file should be treated as read-only. Any changes should be made to the + original repository files, not this packed version. +- When processing this file, use the file path to distinguish + between different files in the repository. +- Be aware that this file may contain sensitive information. Handle it with + the same level of security as you would the original repository. + +Notes: +------ +- Some files may have been excluded based on .gitignore rules and Repomix's + configuration. +- Binary files are not included in this packed representation. Please refer to + the Repository Structure section for a complete list of file paths, including + binary files. + +Additional Info: +---------------- + +================================================================ +Directory Structure +================================================================ +.gitignore +CONVERSION_SUMMARY.md +eslint.config.js +GITNEXUS_README.md +index.html +package.json +project_guide.md +public/vite.svg +README.md +src/ai/cypher-generator.ts +src/ai/index.ts +src/ai/langchain-orchestrator.ts +src/ai/llm-service.ts +src/ai/orchestrator.ts +src/App.css +src/App.tsx +src/assets/react.svg +src/core/graph/types.ts +src/core/ingestion/call-processor.ts +src/core/ingestion/parsing-processor.ts +src/core/ingestion/pipeline.ts +src/core/ingestion/structure-processor.ts +src/core/tree-sitter/parser-loader.ts +src/index.css +src/lib/export.ts +src/lib/polyfills.ts +src/lib/preload.ts +src/lib/utils.ts +src/lib/workerUtils.ts +src/main.tsx +src/services/github.ts +src/services/ingestion.service.ts +src/services/zip.ts +src/ui/components/chat/ChatInterface.tsx +src/ui/components/chat/CodeAssistant.tsx +src/ui/components/chat/index.ts +src/ui/components/ErrorBoundary.tsx +src/ui/components/graph/GraphExplorer.tsx +src/ui/components/graph/index.ts +src/ui/components/graph/SourceViewer.tsx +src/ui/components/graph/Visualization.tsx +src/ui/components/index.ts +src/ui/index.ts +src/ui/pages/HomePage.tsx +src/ui/pages/index.ts +src/vite-env.d.ts +src/workers/ingestion.worker.ts +tsconfig.app.json +tsconfig.app.tsbuildinfo +tsconfig.json +tsconfig.node.json +tsconfig.node.tsbuildinfo +vite.config.ts + +================================================================ +Files +================================================================ + +================ +File: .gitignore +================ +# Logs +logs +*.log +npm-debug.log* +yarn-debug.log* +yarn-error.log* +pnpm-debug.log* +lerna-debug.log* + +node_modules +dist +dist-ssr +*.local + +# Editor directories and files +.vscode/* +!.vscode/extensions.json +.idea +.DS_Store +*.suo +*.ntvs* +*.njsproj +*.sln +*.sw? + +================ +File: CONVERSION_SUMMARY.md +================ +# Deno to Node.js Conversion Summary + +## Overview +Successfully converted the GitNexus repository from Deno to Node.js while maintaining all functionality. + +## Changes Made + +### 1. Configuration Files +- **Removed**: `deno.json`, `deno.lock` +- **Updated**: `package.json` with all dependencies from `deno.json` + - Added all npm dependencies: jszip, axios, cytoscape, web-tree-sitter, langchain packages, etc. + - Updated version to 1.0.0 + - Kept existing build scripts (Vite-based) + +### 2. Import Statements +- **Removed**: All `npm:` prefixes from import statements +- **Removed**: All `@ts-expect-error` comments related to npm: imports +- **Files affected**: 13+ TypeScript files across the codebase + +### 3. Dependencies Successfully Converted +- `react` & `react-dom` (already present) +- `jszip` for ZIP file processing +- `axios` for HTTP requests +- `cytoscape` & `cytoscape-dagre` for graph visualization +- `web-tree-sitter` for code parsing +- `comlink` for web workers +- `@langchain/*` packages for AI functionality +- `zod` for schema validation + +### 4. Build System +- **Unchanged**: Vite configuration remains the same +- **Unchanged**: TypeScript configuration +- **Working**: Development server starts successfully on port 5173 +- **Note**: Some TypeScript errors remain but don't prevent the dev server from running + +## Current Status +✅ **Development server running** - The application starts and runs on Node.js +✅ **All dependencies installed** - npm install completed successfully +✅ **Import statements fixed** - All Deno-style imports converted to Node.js style +⚠️ **TypeScript errors** - Some type errors remain but don't block functionality + +## Next Steps (Optional) +The conversion is complete and functional, but to achieve a clean build: +1. Fix TypeScript errors in langchain imports +2. Update type definitions for cytoscape +3. Fix unused variable warnings +4. Address JSZip type compatibility issues + +## Files Modified +- `package.json` - Added all dependencies +- 13+ TypeScript files - Removed npm: prefixes and Deno comments +- Removed `deno.json` and `deno.lock` + +The repository is now fully converted to Node.js and ready for development! + +================ +File: eslint.config.js +================ +import js from '@eslint/js' +import globals from 'globals' +import reactHooks from 'eslint-plugin-react-hooks' +import reactRefresh from 'eslint-plugin-react-refresh' +import tseslint from 'typescript-eslint' + +export default tseslint.config( + { ignores: ['dist'] }, + { + extends: [js.configs.recommended, ...tseslint.configs.recommended], + files: ['**/*.{ts,tsx}'], + languageOptions: { + ecmaVersion: 2020, + globals: globals.browser, + }, + plugins: { + 'react-hooks': reactHooks, + 'react-refresh': reactRefresh, + }, + rules: { + ...reactHooks.configs.recommended.rules, + 'react-refresh/only-export-components': [ + 'warn', + { allowConstantExport: true }, + ], + }, + }, +) + +================ +File: GITNEXUS_README.md +================ +# 🔍 CodeNexus - Edge Knowledge Graph Creator with Graph RAG + +**Transform any codebase into an interactive knowledge graph in your browser. No servers, no setup - just instant Graph RAG-powered code intelligence.** + +CodeNexus is a client-side knowledge graph creator that runs entirely in your browser. Drop in a GitHub repo or ZIP file, and get an interactive knowledge graph with AI-powered chat interface. Perfect for code exploration, documentation, and understanding complex codebases through Graph RAG (Retrieval-Augmented Generation). + +## ✨ Features + +### 📊 **Code Analysis & Visualization** +- **GitHub Integration**: Analyze any public GitHub repository directly from URL +- **ZIP File Support**: Upload and analyze local code archives +- **Interactive Knowledge Graph**: Visualize code structure with Cytoscape.js +- **Multi-language Support**: Currently optimized for Python with extensible architecture +- **Smart Filtering**: Directory and file pattern filters to focus analysis scope +- **Performance Optimization**: Configurable file limits with confirmation dialogs for large repositories + +### 🤖 **AI-Powered Chat Interface** +- **Multiple LLM Providers**: OpenAI, Anthropic (Claude), Google Gemini +- **ReAct Agent Pattern**: Uses proper LangChain ReAct implementation for reasoning +- **Tool-Augmented Responses**: Graph queries, code retrieval, file search +- **Context-Aware**: Maintains conversation history with configurable memory + +### 🔧 **Advanced Processing Pipeline** +- **3-Pass Ingestion Strategy**: + 1. **Structure Analysis**: Project hierarchy and file organization + 2. **Code Parsing**: AST-based extraction using Tree-sitter + 3. **Call Resolution**: Function/method call relationship mapping +- **Web Worker Processing**: Non-blocking UI with progress tracking +- **Intelligent Caching**: AST and processing result optimization +- **Error Resilience**: Comprehensive error boundaries and recovery mechanisms + +### 🎨 **Modern UI/UX** +- **Responsive Design**: Adaptive layout for different screen sizes +- **Real-time Progress**: Live updates during repository processing +- **Interactive Graph**: Node selection, zooming, panning +- **Split-Panel Layout**: Graph visualization + AI chat interface +- **Settings Management**: Persistent configuration for API keys and preferences +- **Export Functionality**: Download knowledge graphs as JSON with metadata +- **Performance Controls**: File limits, filtering, and optimization settings + +### 🛡️ **Reliability & Performance** +- **Error Boundaries**: Graceful error handling with user-friendly recovery options +- **Performance Monitoring**: Real-time processing statistics and export size calculation +- **Memory Management**: Efficient handling of large repositories with configurable limits +- **Progress Tracking**: Detailed progress indicators with phase-specific messaging +- **Confirmation Dialogs**: Smart warnings for potentially expensive operations + +## 🏗️ Architecture + +### **Frontend Stack** +- **React 18** with TypeScript +- **Vite** for fast development and building +- **Cytoscape.js** for graph visualization +- **Custom CSS** with modern design patterns +- **Error Boundaries** for robust error handling + +### **Processing Engine** +- **Deno Runtime** for TypeScript execution +- **Tree-sitter WASM** for syntax parsing +- **Web Workers** for background processing +- **Comlink** for worker communication + +### **AI Integration** +- **LangChain.js** with proper ReAct agent implementation +- **Multiple LLM Support**: OpenAI, Anthropic, Gemini +- **Tool-based Architecture**: Graph queries, code retrieval, file search +- **Cypher Query Generation**: Natural language to graph queries + +### **Services Layer** +``` +src/ +├── services/ # External API integrations +│ ├── github.ts # GitHub REST API client +│ └── zip.ts # ZIP file processing +├── core/ # Core processing logic +│ ├── graph/ # Knowledge graph types +│ ├── ingestion/ # 3-pass processing pipeline +│ └── tree-sitter/ # Syntax parsing infrastructure +├── ai/ # AI and RAG components +│ ├── llm-service.ts # Multi-provider LLM client +│ ├── cypher-generator.ts # NL to Cypher translation +│ ├── orchestrator.ts # Custom ReAct implementation +│ └── langchain-orchestrator.ts # Standard LangChain ReAct +├── workers/ # Web Worker implementations +├── ui/ # React components and pages +│ ├── components/ # Reusable UI components +│ │ ├── ErrorBoundary.tsx # Error handling component +│ │ ├── graph/ # Graph visualization components +│ │ └── chat/ # Chat interface components +│ └── pages/ # Application pages +├── lib/ # Shared utilities +│ └── export.ts # Graph export functionality +└── App.tsx # Main application entry point +``` + +## 🚀 Getting Started + +### Prerequisites +- **Node.js 18+** and **npm/yarn** +- **Deno 1.40+** for development +- **API Keys** for AI features (OpenAI, Anthropic, or Gemini) + +### Installation + +1. **Clone the repository** + ```bash + git clone + cd gitnexus + ``` + +2. **Install dependencies** + ```bash + npm install + ``` + +3. **Start development server** + ```bash + npm run dev + ``` + +4. **Open in browser** + ``` + http://localhost:5173 + ``` + +### Configuration + +1. **GitHub Token (Optional)** + - Increases rate limit from 60 to 5,000 requests/hour + - Generate at: https://github.com/settings/tokens + - Requires no special permissions for public repos + +2. **AI API Keys** + - **OpenAI**: Get from https://platform.openai.com/api-keys + - **Anthropic**: Get from https://console.anthropic.com/ + - **Gemini**: Get from https://makersuite.google.com/app/apikey + +3. **Performance Settings** + - **File Limit**: Configure maximum files to process (default: 500) + - **Directory Filters**: Focus on specific directories (e.g., "src", "lib") + - **File Patterns**: Filter by file types (e.g., "*.py", "*.js", "*.ts") + +## 💡 Usage + +### Analyzing a Repository + +1. **GitHub Repository** + ``` + 1. Enter GitHub URL: https://github.com/owner/repo + 2. Optional: Set directory/file filters to focus analysis + 3. Click "Analyze" + 4. For large repos: Confirm processing or adjust filters + 5. Wait for processing (structure → parsing → call resolution) + 6. Explore the interactive graph + ``` + +2. **ZIP File Upload** + ``` + 1. Click "Choose File" and select a .zip file + 2. Optional: Configure filters before processing + 3. Click "Analyze" + 4. Processing will extract and analyze text files + 5. Explore results in the graph visualization + ``` + +### Performance Optimization + +1. **Directory Filtering** + ``` + - Enter directory names: "src", "lib", "components" + - Focuses analysis on specific parts of the codebase + - Reduces processing time and memory usage + ``` + +2. **File Pattern Filtering** + ``` + - Use patterns: "*.py", "*.js", "*.ts" + - Supports wildcards: "test*.py", "*util*" + - Comma-separated: "*.py,*.js,*.ts" + ``` + +3. **File Limits** + ``` + - Default limit: 500 files + - Configurable in settings (50-2000 files) + - Large repositories show confirmation dialog + - Automatic truncation to limit if confirmed + ``` + +### Using the AI Chat + +1. **Configure API Key** + ``` + 1. Click the ⚙️ settings button + 2. Choose your preferred LLM provider + 3. Enter your API key + 4. Select model (e.g., gpt-4o-mini, claude-3-haiku) + ``` + +2. **Ask Questions** + ``` + - "What functions are in the main.py file?" + - "Show me all classes that inherit from BaseClass" + - "How does the authentication system work?" + - "Find all functions that call the database" + ``` + +### Exporting Data + +1. **Export Knowledge Graph** + ``` + 1. Click the 📥 Export button after processing + 2. Downloads JSON file with graph data and metadata + 3. Includes processing statistics and timestamps + 4. File size shown in UI before export + ``` + +2. **Export Format** + ```json + { + "metadata": { + "exportedAt": "2024-01-01T12:00:00.000Z", + "version": "1.0.0", + "nodeCount": 150, + "relationshipCount": 200, + "fileCount": 25, + "processingDuration": 5000 + }, + "graph": { + "nodes": [...], + "relationships": [...] + }, + "fileContents": {...} + } + ``` + +### Graph Interaction + +- **Node Selection**: Click any node to highlight and view details +- **Zoom & Pan**: Mouse wheel to zoom, drag to pan +- **Node Types**: Different colors/shapes for files, functions, classes, etc. +- **Relationships**: Arrows show CONTAINS, CALLS, INHERITS relationships + +## 🔧 Development + +### Project Structure +``` +GitNexus/ +├── src/ +│ ├── services/ # External integrations +│ ├── core/ # Processing pipeline +│ ├── ai/ # AI and RAG systems +│ ├── workers/ # Web Workers +│ ├── ui/ # React components +│ │ ├── components/ # Reusable components +│ │ │ ├── ErrorBoundary.tsx +│ │ │ ├── graph/ # Graph components +│ │ │ └── chat/ # Chat components +│ │ └── pages/ # Application pages +│ ├── lib/ # Utilities +│ │ └── export.ts # Export functionality +│ └── App.tsx # Main application +├── public/ +│ └── wasm/ # Tree-sitter WASM files +├── package.json +├── vite.config.ts +└── tsconfig.json +``` + +### Key Components + +#### **Error Handling** +```typescript +// ErrorBoundary component with recovery options + { + console.error('Application error:', error); + }} +> + + +``` + +#### **Performance Optimization** +```typescript +// File filtering and limits +const filterFiles = (files: any[]) => { + return files + .filter(file => matchesDirectoryFilter(file)) + .filter(file => matchesPatternFilter(file)) + .slice(0, maxFiles); +}; +``` + +#### **Export Functionality** +```typescript +// Export with metadata +exportAndDownloadGraph(graph, { + projectName: 'my-project', + includeMetadata: true, + prettyPrint: true +}, fileContents, { duration: 5000 }); +``` + +#### **Processing Pipeline** +```typescript +// 3-pass ingestion strategy with progress tracking +const pipeline = new GraphPipeline(); +const result = await pipeline.run({ + projectRoot: '/', + projectName: 'MyProject', + filePaths: ['src/main.py', 'src/utils.py'], + fileContents: new Map([ + ['src/main.py', 'def main(): pass'], + ['src/utils.py', 'def helper(): pass'] + ]) +}); +``` + +#### **AI Integration** +```typescript +// LangChain ReAct agent with error handling +const orchestrator = new LangChainRAGOrchestrator(llmService, cypherGenerator); +await orchestrator.setContext({ graph, fileContents }, llmConfig); +const response = await orchestrator.answerQuestion("How does auth work?"); +``` + +#### **Graph Visualization** +```typescript +// Interactive graph component with error boundaries + + setSelectedNode(nodeId)} + /> + +``` + +### Adding New Features + +1. **New Language Support** + ```typescript + // Add parser in core/tree-sitter/ + export const loadJavaScriptParser = async () => { + // Load JS Tree-sitter grammar + }; + ``` + +2. **Custom AI Tools** + ```typescript + // Add tools in ai/langchain-orchestrator.ts + const customTool = tool( + async (input: { query: string }) => { + // Tool implementation + }, + { + name: "custom_tool", + description: "Custom functionality", + schema: z.object({ query: z.string() }) + } + ); + ``` + +3. **Export Formats** + ```typescript + // Add new export formats in lib/export.ts + export function exportToCSV(graph: KnowledgeGraph): string { + // CSV export implementation + } + ``` + +## 🧪 Testing & Quality Assurance + +### Error Handling +- **Error Boundaries**: Catch and display JavaScript errors gracefully +- **User Recovery**: Allow users to reset component state after errors +- **Detailed Logging**: Console logging for debugging and error reporting +- **Fallback UI**: User-friendly error messages with recovery options + +### Performance Testing +1. **Large Repository Handling** + - Test with repositories containing 1000+ files + - Verify confirmation dialogs for file limits + - Monitor memory usage during processing + - Test filtering effectiveness + +2. **UI Responsiveness** + - Ensure non-blocking processing with Web Workers + - Verify progress indicators update correctly + - Test error recovery mechanisms + - Validate export functionality with large graphs + +3. **Error Scenarios** + - Network failures during GitHub API calls + - Corrupted ZIP files + - Invalid API keys + - Memory exhaustion scenarios + +### Manual Testing Checklist +- [ ] GitHub repository analysis with various sizes +- [ ] ZIP file upload and extraction +- [ ] Directory and file pattern filtering +- [ ] Large repository confirmation dialog +- [ ] Export functionality with different options +- [ ] Error boundary activation and recovery +- [ ] API key validation for all providers +- [ ] Settings persistence across sessions +- [ ] Graph visualization interactions +- [ ] Chat interface with different LLM providers + +## 🚀 Deployment + +### Production Build +```bash +npm run build +npm run preview +``` + +### Environment Variables +```env +# Optional: Pre-configure API keys +VITE_OPENAI_API_KEY=sk-... +VITE_ANTHROPIC_API_KEY=sk-ant-... +VITE_GEMINI_API_KEY=... + +# Performance settings +VITE_DEFAULT_MAX_FILES=500 +VITE_ENABLE_DEBUG_LOGGING=false +``` + +### Docker Deployment +```dockerfile +FROM node:18-alpine +WORKDIR /app +COPY package*.json ./ +RUN npm ci --only=production +COPY . . +RUN npm run build +EXPOSE 3000 +CMD ["npm", "run", "preview", "--", "--host", "0.0.0.0"] +``` + +### Performance Monitoring +```javascript +// Add performance monitoring +const observer = new PerformanceObserver((list) => { + for (const entry of list.getEntries()) { + if (entry.entryType === 'measure') { + console.log(`${entry.name}: ${entry.duration}ms`); + } + } +}); +observer.observe({ entryTypes: ['measure'] }); +``` + +## 🔒 Security & Privacy + +- **Client-Side Processing**: All analysis happens in your browser +- **API Keys**: Stored locally, never transmitted to our servers +- **GitHub Access**: Uses public API, respects repository permissions +- **Data Privacy**: No code or analysis results are stored remotely +- **Error Logging**: Sensitive data excluded from error reports +- **Export Security**: User-controlled data export with no server interaction + +## 🤝 Contributing + +### Development Setup +1. Fork the repository +2. Create feature branch: `git checkout -b feature/amazing-feature` +3. Make changes and test thoroughly +4. Run the testing checklist above +5. Commit: `git commit -m 'Add amazing feature'` +6. Push: `git push origin feature/amazing-feature` +7. Open a Pull Request + +### Code Style +- **TypeScript**: Strict mode enabled +- **ESLint**: Follow configured rules +- **Prettier**: Auto-formatting +- **Comments**: Minimal, only when necessary +- **Error Handling**: Comprehensive error boundaries and recovery +- **Performance**: Consider memory usage and processing time + +### Testing Guidelines +- Test error scenarios and edge cases +- Verify performance with large datasets +- Ensure graceful degradation +- Test all export functionality +- Validate error boundary behavior + +## 📚 Technical Details + +### Knowledge Graph Schema +```typescript +interface KnowledgeGraph { + nodes: GraphNode[]; // Code entities + relationships: GraphRelationship[]; // Connections +} + +// Node types: Project, Folder, File, Module, Class, Function, Method, Variable +// Relationship types: CONTAINS, CALLS, INHERITS, OVERRIDES, IMPORTS +``` + +### Export Format +```typescript +interface ExportedGraph { + metadata: { + exportedAt: string; + version: string; + nodeCount: number; + relationshipCount: number; + fileCount?: number; + processingDuration?: number; + }; + graph: KnowledgeGraph; + fileContents?: Record; +} +``` + +### Error Boundary Implementation +- **Component-Level**: Individual components wrapped for isolation +- **Application-Level**: Top-level boundary for catastrophic failures +- **Recovery Options**: Reset state, reload page, or continue with fallback +- **Error Reporting**: Detailed technical information for developers + +### Performance Optimizations +- **Web Workers**: Non-blocking processing +- **AST Caching**: Reuse parsed syntax trees +- **Progressive Loading**: Stream results as available +- **Memory Management**: Efficient data structures +- **File Filtering**: Reduce processing scope +- **Confirmation Dialogs**: Prevent accidental expensive operations + +### ReAct Agent Implementation +- **Standard LangChain**: Uses `createReactAgent` from `@langchain/langgraph/prebuilt` +- **Custom Implementation**: Manual ReAct loop for educational purposes +- **Tools**: Graph queries, code retrieval, file search +- **Memory**: Conversation persistence with thread management +- **Error Recovery**: Graceful handling of API failures + +## 📄 License + +This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details. + +## 🙏 Acknowledgments + +- **Tree-sitter**: Syntax parsing infrastructure +- **LangChain.js**: AI agent framework +- **Cytoscape.js**: Graph visualization +- **React**: UI framework with error boundaries +- **Vite**: Build tool and dev server + +--- + +**CodeNexus** - Edge Knowledge Graph Creator with instant Graph RAG. Zero setup, maximum insight. 🚀 + +*Browser-native code intelligence that runs anywhere, anytime - no servers required.* + +================ +File: index.html +================ + + + + + + + GitNexus - Code Knowledge Graph Explorer + + + +
+ + + + +================ +File: package.json +================ +{ + "name": "gitnexus", + "private": true, + "version": "1.0.0", + "type": "module", + "scripts": { + "dev": "vite", + "build": "tsc -b && vite build", + "lint": "eslint .", + "preview": "vite preview" + }, + "dependencies": { + "@langchain/anthropic": "^0.1.21", + "@langchain/core": "^0.3.66", + "@langchain/google-genai": "^0.2.16", + "@langchain/langgraph": "^0.0.26", + "@langchain/openai": "^0.0.28", + "@types/d3": "^7.4.3", + "axios": "^1.6.0", + "comlink": "^4.4.1", + "d3": "^7.9.0", + "jszip": "^3.10.1", + "react": "^18.3.1", + "react-dom": "^18.3.1", + "uuid": "^11.1.0", + "web-tree-sitter": "^0.20.8", + "zod": "^3.25.76" + }, + "devDependencies": { + "@eslint/js": "^9.11.1", + "@types/react": "^18.3.10", + "@types/react-dom": "^18.3.0", + "@vitejs/plugin-react": "^4.3.2", + "eslint": "^9.11.1", + "eslint-plugin-react-hooks": "^5.1.0-rc.0", + "eslint-plugin-react-refresh": "^0.4.12", + "globals": "^15.9.0", + "typescript": "^5.5.3", + "typescript-eslint": "^8.7.0", + "vite": "^5.4.8" + } +} + +================ +File: project_guide.md +================ +# GitNexus: Edge-Based Code Knowledge Graph Generator for Deno - Step-by-Step Implementation Guide + +This guide will walk you through building a fully edge-based code knowledge graph generator from scratch using Deno. I'll explain each concept before showing the implementation, so you understand **why** we're doing something, not just **how** to do it. + +## Phase 1: Project Setup & Core Infrastructure + +### Step 1: Project Structure and Tooling Setup + +**Why this matters:** Before writing any code, we need to set up our development environment properly. A well-structured project makes it easier to add features later and keeps everything organized. + +**Key concepts:** + +- We're using Vite (a modern build tool) with React and TypeScript +- We need special configuration for WebAssembly (WASM) files +- A clear directory structure helps us scale to multiple languages later + +**Implementation Steps:** + +1. **Create the base project:** + +```bash +# Create project root +mkdir GitNexus +cd GitNexus + +# Initialize Vite project with React and TypeScript +npm create vite@latest . -- --template react-ts + +# Initialize Deno project +deno init +``` + +2. **Create the application directory structure:** + +```bash +# Create directories for our core components +mkdir -p src/{core,core/tree-sitter,core/graph,core/ingestion,services,ai,ai/agents,ai/prompts,ui,ui/components,ui/components/graph,ui/components/chat,ui/hooks,workers,lib,config,store} +``` + +**Why this structure?** + +- `core/`: Contains the engine that builds the knowledge graph +- `services/`: Handles external interactions (GitHub API, ZIP processing) +- `ai/`: Contains the RAG and chat functionality +- `ui/`: All user interface components +- `workers/`: Web Workers for heavy processing (keeps UI responsive) +- `lib/`: Utility functions used throughout the app + +### Step 2: Configure Build Tools for WASM + +**Why this matters:** WebAssembly (WASM) is how we'll run the Tree-sitter parsers in the browser. We need special configuration to handle these binary files correctly. + +**Key concepts:** + +- WASM files are binary files that run at near-native speed in browsers +- Vite needs special configuration to handle them properly +- We want to avoid inlining large WASM files in our JavaScript bundles + +**Implementation:** + +1. **Update `vite.config.ts`:** + +```typescript +import { defineConfig } from 'vite' +import react from '@vitejs/plugin-react' +export default defineConfig({ + plugins: [react()], + worker: { + format: 'es' + }, + assetsInclude: ['**/*.wasm'], + build: { + target: 'esnext', + assetsInlineLimit: 0 // Don't inline WASM files + } +}) +``` + +**What this does:** + +- `assetsInclude: ['**/*.wasm']` tells Vite to treat WASM files as assets +- `assetsInlineLimit: 0` ensures WASM files aren't inlined into JavaScript (they're too large) +- `worker: { format: 'es' }` configures Web Workers to use ES modules + +2. **Configure TypeScript** with a `tsconfig.json` that has strict settings for better code quality. + +**Why strict settings?** They help catch errors early and make the code more maintainable as the project grows. + +### Step 3: Set Up WASM Parser Infrastructure + +**Why this matters:** Tree-sitter is the engine that parses code into ASTs (Abstract Syntax Trees). We need to get these parsers working in the browser via WASM. + +**Key concepts:** + +- Tree-sitter parsers for different languages are written in C +- We compile them to WASM so they can run in browsers +- We need to load these parsers on demand + +**Implementation:** + +1. **Create a public directory for WASM files:** + +```bash +mkdir -p public/wasm/python +``` + +2. **Download the Tree-sitter Python parser:** + - Get `tree-sitter-python.wasm` from [tree-sitter-python releases](https://github.com/tree-sitter/tree-sitter-python/releases) + - Place it in `public/wasm/python/` + +**Why host WASM files separately?** Browsers can't access the user's file system directly for security reasons. We need to serve the WASM files from a URL. + +3. **Create a loader for Tree-sitter parsers:** + +```typescript +import WebTreeSitter from 'web-tree-sitter'; +let parserInstance: WebTreeSitter | null = null; +const parserCache = new Map(); + +export async function initTreeSitter() { + if (parserInstance) return parserInstance; + parserInstance = await WebTreeSitter.init(); + return parserInstance; +} + +export async function loadPythonParser(): Promise { + if (parserCache.has('python')) { + return parserCache.get('python')!; + } + const Parser = await initTreeSitter(); + const pythonLang = await Parser.Language.load( + '/wasm/python/tree-sitter-python.wasm' + ); + parserCache.set('python', pythonLang); + return pythonLang; +} +``` + +**How this works:** + +1. `initTreeSitter()` initializes the WebAssembly module once +2. `loadPythonParser()` loads the Python parser from the WASM file +3. We cache parsers to avoid reloading them multiple times + +**Why cache parsers?** Loading WASM files is relatively slow, so we want to do it once and reuse the parsers. + +## Phase 2: Code Acquisition Module + +### Step 4: Implement GitHub API Integration + +**Why this matters:** Users will want to analyze public GitHub repositories, so we need a way to fetch code from GitHub. + +**Key concepts:** + +- GitHub has a REST API for accessing repository contents +- We need to handle rate limits (GitHub limits how many requests you can make) +- We'll let users provide their own API tokens for higher limits + +**Implementation:** + +```typescript +export class GitHubService { + private token: string | null = null; + + setToken(token: string) { + this.token = token; + } + + async getRepoContents(owner: string, repo: string, path = '') { + const headers: HeadersInit = { + 'Accept': 'application/vnd.github.v3+json' + }; + if (this.token) { + headers['Authorization'] = `token ${this.token}`; + } + + const response = await fetch( + `https://api.github.com/repos/${owner}/${repo}/contents/${path}`, + { headers } + ); + + if (!response.ok) { + throw new Error(`GitHub API error: ${response.status}`); + } + + return response.json(); + } +} +``` + +**How this works:** + +- `getRepoContents()` fetches the directory structure of a repository +- It uses the GitHub API with proper headers +- It handles authentication via a token + +**Important note:** GitHub API has rate limits. For unauthenticated requests, it's about 60 requests/hour. With a token, it's 5,000/hour. + +### Step 5: Implement ZIP Processing + +**Why this matters:** Not all code is on GitHub. Users might want to analyze local code or private repositories by uploading a ZIP file. + +**Key concepts:** + +- JSZip is a library for handling ZIP files in JavaScript +- We need to extract files and their contents from the ZIP +- We'll use a Map to store file paths and contents + +**Implementation:** + +```typescript +import JSZip from 'jszip'; + +export class ZipService { + async processZip(file: File): Promise> { + const zip = await JSZip.loadAsync(file); + const files = new Map(); + + for (const [filePath, zipEntry] of Object.entries(zip.files)) { + if (!zipEntry.dir) { + const content = await zipEntry.async('text'); + files.set(filePath, content); + } + } + + return files; + } +} +``` + +**How this works:** + +1. `JSZip.loadAsync(file)` loads the ZIP file +2. We iterate through all entries in the ZIP +3. For each file (not directory), we extract its content as text +4. We store the file path and content in a Map + +**Why use a Map?** It provides O(1) lookups by file path, which is important when we need to find files during graph construction. + +## Phase 3: Graph Construction Pipeline + +### Step 6: Define Graph Data Structures + +**Why this matters:** Before we can build a graph, we need to define what nodes and relationships look like. + +**Key concepts:** + +- A knowledge graph consists of nodes and relationships +- Nodes represent code elements (functions, classes, etc.) +- Relationships represent connections between elements (calls, contains, etc.) + +**Implementation:** + +```typescript +export type NodeLabel = + | 'Project' + | 'Package' + | 'Module' + | 'Folder' + | 'File' + | 'Class' + | 'Function' + | 'Method' + | 'Variable'; + +export interface GraphNode { + id: string; + label: NodeLabel; + properties: Record; +} + +export type RelationshipType = + | 'CONTAINS' + | 'CALLS' + | 'INHERITS' + | 'OVERRIDES' + | 'IMPORTS'; + +export interface GraphRelationship { + id: string; + type: RelationshipType; + source: string; + target: string; + properties?: Record; +} + +export interface KnowledgeGraph { + nodes: GraphNode[]; + relationships: GraphRelationship[]; +} +``` + +**Why these specific types?** + +- `NodeLabel` defines all possible types of code elements we'll track +- `RelationshipType` defines how code elements connect to each other +- `KnowledgeGraph` is the complete structure we'll build + +**Important relationships:** + +- `CONTAINS`: A folder contains files, a file contains functions +- `CALLS`: A function calls another function +- `IMPORTS`: One module imports from another + +### Step 7: Implement the 3-Pass Ingestion Pipeline + +**Why this matters:** Building a complete knowledge graph requires multiple passes to handle cross-file references properly. + +**Key concepts:** + +- **Pass 1**: Identify the overall structure (folders, modules) +- **Pass 2**: Parse individual files and cache ASTs +- **Pass 3**: Process function calls across files (the hardest part) + +This three-pass approach solves the "island problem" - where functions in different files appear disconnected. + +#### Pass 1: Structure Identification + +```typescript +export class StructureProcessor { + private graph: KnowledgeGraph; + private projectRoot: string; + private projectName: string; + + constructor(graph: KnowledgeGraph, projectRoot: string, projectName: string) { + this.graph = graph; + this.projectRoot = projectRoot; + this.projectName = projectName; + } + + identifyStructure(filePaths: string[]): void { + // Add Project node + this.graph.nodes.push({ + id: `project:${this.projectName}`, + label: 'Project', + properties: { name: this.projectName } + }); + + // Track directory structure + const directories = new Set(); + for (const filePath of filePaths) { + const dirPath = filePath.substring(0, filePath.lastIndexOf('/')); + if (dirPath && !directories.has(dirPath)) { + directories.add(dirPath); + // Create Folder node + this.graph.nodes.push({ + id: `folder:${dirPath}`, + label: 'Folder', + properties: { path: dirPath } + }); + + // Create CONTAINS relationship with parent + if (dirPath.includes('/')) { + const parentPath = dirPath.substring(0, dirPath.lastIndexOf('/')); + this.graph.relationships.push({ + id: `rel:folder:${dirPath}:parent`, + type: 'CONTAINS', + source: `folder:${parentPath}`, + target: `folder:${dirPath}` + }); + } else { + // Root folder connects to project + this.graph.relationships.push({ + id: `rel:folder:${dirPath}:project`, + type: 'CONTAINS', + source: `project:${this.projectName}`, + target: `folder:${dirPath}` + }); + } + } + } + } +} +``` + +**How this works:** + +1. Creates a root Project node +2. Walks through all file paths to identify directories +3. Creates Folder nodes and CONTAINS relationships + +**Why identify structure first?** We need to know the overall organization before parsing individual files. + +#### Pass 2: File Parsing + +```typescript +export class ParsingProcessor { + private graph: KnowledgeGraph; + private astCache = new Map(); + + constructor(graph: KnowledgeGraph) { + this.graph = graph; + } + + async parseFiles(filePaths: string[], fileContents: Map): Promise> { + for (const [filePath, content] of fileContents) { + if (filePath.endsWith('.py')) { + await this.parsePythonFile(filePath, content); + } + } + return this.astCache; + } + + private async parsePythonFile(filePath: string, content: string): Promise { + const parser = await loadPythonParser(); + const tree = parser.parse(content); + // Cache the AST + this.astCache.set(filePath, tree); + // Extract definitions from the AST + this.extractDefinitions(filePath, tree, content); + } + + private extractDefinitions(filePath: string, tree: any, content: string): void { + // Extract modules + this.graph.nodes.push({ + id: `module:${filePath}`, + label: 'Module', + properties: { + path: filePath, + name: filePath.split('/').pop()!.replace('.py', ''), + extension: '.py' + } + }); + + // Extract functions from the AST + const rootNode = tree.rootNode; + const functionDefs = rootNode.descendantsOfType('function_definition'); + for (const funcNode of functionDefs) { + const nameNode = funcNode.childForFieldName('name'); + const name = nameNode ? nameNode.text : 'unknown'; + + // Calculate position + const startLine = funcNode.startPosition.row + 1; + + // Create function node + this.graph.nodes.push({ + id: `function:${filePath}:${name}`, + label: 'Function', + properties: { + name, + qualified_name: `${this.getModuleName(filePath)}.${name}`, + path: filePath, + start_line: startLine + } + }); + + // Create CONTAINS relationship with module + this.graph.relationships.push({ + id: `rel:function:${filePath}:${name}:module`, + type: 'CONTAINS', + source: `module:${filePath}`, + target: `function:${filePath}:${name}` + }); + } + } +} +``` + +**How this works:** + +1. Parses each file with the appropriate Tree-sitter parser +2. Caches the AST for later use +3. Extracts definitions (functions, classes) from the AST +4. Creates nodes and relationships in the graph + +**Why cache ASTs?** We need them in Pass 3 to resolve cross-file function calls. + +#### Pass 3: Call Resolution + +```typescript +export class CallProcessor { + private graph: KnowledgeGraph; + private astCache: Map; + private projectRoot: string; + private projectName: string; + + constructor( + graph: KnowledgeGraph, + astCache: Map, + projectRoot: string, + projectName: string + ) { + this.graph = graph; + this.astCache = astCache; + this.projectRoot = projectRoot; + this.projectName = projectName; + } + + processCalls(): void { + for (const [filePath, tree] of this.astCache) { + if (filePath.endsWith('.py')) { + this.processPythonCalls(filePath, tree); + } + } + } + + private processPythonCalls(filePath: string, tree: any): void { + const rootNode = tree.rootNode; + // Find all call expressions + const callExpressions = rootNode.descendantsOfType('call'); + for (const callNode of callExpressions) { + const functionNameNode = callNode.childForFieldName('function'); + if (!functionNameNode) continue; + + // Handle different types of function references + let targetFunctionName = ''; + if (functionNameNode.type === 'identifier') { + targetFunctionName = functionNameNode.text; + } else if (functionNameNode.type === 'attribute') { + // Handle method calls like obj.method() + const attrNode = functionNameNode; + const objectNode = attrNode.childForFieldName('object'); + const attrNameNode = attrNode.childForFieldName('attribute'); + if (objectNode && attrNameNode) { + const objectName = objectNode.text; + const methodName = attrNameNode.text; + targetFunctionName = `${objectName}.${methodName}`; + } + } + + if (!targetFunctionName) continue; + + // Try to resolve the target function + const targetNode = this.resolveTargetFunction(targetFunctionName, filePath); + if (targetNode) { + // Create CALLS relationship + const callerId = this.getCallerId(callNode, filePath); + this.graph.relationships.push({ + id: `rel:call:${callerId}:${targetNode.id}`, + type: 'CALLS', + source: callerId, + target: targetNode.id + }); + } + } + } + + private resolveTargetFunction(targetName: string, currentFilePath: string): { id: string; type: string } | null { + // 1. Check if it's a built-in function + if (this.isBuiltInFunction(targetName)) { + return { + id: `builtin:${targetName}`, + type: 'builtin' + }; + } + + // 2. Check if it's an imported function + const importInfo = this.findImportForFunction(targetName, currentFilePath); + if (importInfo) { + const targetId = `function:${importInfo.sourceFile}:${importInfo.targetName}`; + return { + id: targetId, + type: 'imported' + }; + } + + // 3. Check if it's defined in the current file + for (const node of this.graph.nodes) { + if (node.label === 'Function' && + node.properties.name === targetName && + node.properties.path === currentFilePath) { + return { + id: node.id, + type: 'local' + }; + } + } + + return null; + } +} +``` + +**How this works:** + +1. Finds all function calls in the AST +2. Determines what function is being called +3. Resolves the target function across files using imports +4. Creates CALLS relationships in the graph + +**Why is this the hardest part?** Resolving cross-file references requires understanding: + +- How imports work in the language +- How to map a simple name to a fully qualified name +- Handling edge cases like aliases (`import helper as h`) + +### Step 8: Implement Web Workers for Performance + +**Why this matters:** Parsing code and building graphs can be CPU-intensive. Web Workers keep the UI responsive. + +**Key concepts:** + +- Web Workers run JavaScript in background threads +- They can't access the DOM directly +- We use Comlink to simplify communication + +**Implementation:** + +```typescript +// src/workers/ingestion.worker.ts +import { expose } from 'comlink'; +import { GraphPipeline } from '../core/ingestion/pipeline'; + +class IngestionWorker { + async processRepository( + projectRoot: string, + projectName: string, + filePaths: string[], + fileContents: Record + ) { + const pipeline = new GraphPipeline(projectRoot, projectName); + return pipeline.run(filePaths, new Map(Object.entries(fileContents))); + } +} + +expose(new IngestionWorker()); +``` + +**How this works:** + +1. The worker runs the heavy processing in a background thread +2. We expose methods via Comlink to call them from the main thread +3. The main thread can call these methods without blocking the UI + +**Why use Web Workers?** Without them, large repositories would freeze the browser tab while processing. + +## Phase 4: Graph Visualization + +### Step 9: Implement Graph Visualization Components + +**Why this matters:** A knowledge graph is useless if users can't see and interact with it. + +**Key concepts:** + +- Cytoscape.js is a powerful graph visualization library +- We need to convert our graph data to Cytoscape's format +- Users need controls to filter and navigate the graph + +**Implementation:** + +```tsx +import React, { useEffect, useRef } from 'react'; +import cytoscape from 'cytoscape'; +import dagre from 'cytoscape-dagre'; +import { KnowledgeGraph } from '@/core/graph/types'; + +cytoscape.use(dagre); + +interface GraphVisualizationProps { + graph: KnowledgeGraph; + onNodeClick?: (nodeId: string) => void; + filter?: (node: any) => boolean; +} + +export const GraphVisualization: React.FC = ({ + graph, + onNodeClick, + filter +}) => { + const containerRef = useRef(null); + const cyRef = useRef(null); + + useEffect(() => { + if (!containerRef.current) return; + + // Clean up previous instance + if (cyRef.current) { + cyRef.current.destroy(); + } + + // Convert our graph to Cytoscape format + const cyElements = convertToCytoscapeElements(graph, filter); + + const cy = cytoscape({ + container: containerRef.current, + elements: cyElements, + style: [ + { + selector: 'node', + style: { + 'label': 'data(label)', + 'width': 'mapData(size, 0, 100, 20, 80)', + 'height': 'mapData(size, 0, 100, 20, 80)', + 'background-color': 'data(color)', + 'text-valign': 'center', + 'text-halign': 'center', + 'font-size': '8px' + } + }, + { + selector: 'edge', + style: { + 'width': 2, + 'line-color': '#ccc', + 'target-arrow-color': '#ccc', + 'target-arrow-shape': 'triangle' + } + } + ], + layout: { + name: 'dagre', + rankDir: 'TB', + padding: 20 + } + }); + + // Add interactions + cy.on('tap', 'node', (event) => { + const node = event.target; + const nodeId = node.data('id'); + if (onNodeClick) { + onNodeClick(nodeId); + } + }); + + cyRef.current = cy; + + return () => { + if (cyRef.current) { + cyRef.current.destroy(); + cyRef.current = null; + } + }; + }, [graph, filter]); + + return ( +
+ ); +}; + +function convertToCytoscapeElements( + graph: KnowledgeGraph, + filter?: (node: any) => boolean +) { + const elements: any[] = []; + + // Add nodes + for (const node of graph.nodes) { + if (filter && !filter(node)) continue; + elements.push({ + data: { + id: node.id, + label: getNodeLabel(node), + type: node.label, + color: getNodeColor(node.label), + size: getNodeSize(node) + } + }); + } + + // Add edges + for (const rel of graph.relationships) { + elements.push({ + data: { + id: rel.id, + source: rel.source, + target: rel.target, + label: rel.type + } + }); + } + + return elements; +} +``` + +**How this works:** + +1. Converts our graph data to Cytoscape's format +2. Sets up visual styles based on node type +3. Applies a hierarchical layout (dagre) +4. Adds interaction handlers for node clicks + +**Why use Cytoscape.js?** It's specifically designed for graph visualization with: + +- Multiple layout algorithms +- Good performance for medium-sized graphs +- Extensive customization options + +### Step 10: Create Source Code Viewer + +**Why this matters:** Seeing the graph isn't enough - users need to see the actual code behind the nodes. + +**Implementation:** + +```tsx +import React, { useState, useEffect } from 'react'; +import { KnowledgeGraph } from '@/core/graph/types'; + +interface SourceViewerProps { + graph: KnowledgeGraph; + selectedNodeId: string | null; +} + +export const SourceViewer: React.FC = ({ graph, selectedNodeId }) => { + const [sourceCode, setSourceCode] = useState(''); + const [fileName, setFileName] = useState(''); + const [lineNumber, setLineNumber] = useState(null); + + useEffect(() => { + if (!selectedNodeId) { + setSourceCode(''); + setFileName(''); + setLineNumber(null); + return; + } + + // Find the node in the graph + const node = graph.nodes.find(n => n.id === selectedNodeId); + if (!node) return; + + // For functions, get the source code + if (node.label === 'Function' || node.label === 'Method') { + const filePath = node.properties.path; + const startLine = node.properties.start_line; + + // In a real implementation, you'd have the source code available + setFileName(filePath); + setLineNumber(startLine); + setSourceCode(`# Source code for ${node.properties.qualified_name} +# Line ${startLine} and following...`); + } + }, [graph, selectedNodeId]); + + if (!selectedNodeId || !sourceCode) { + return ( +
+

Select a node to view source code

+
+ ); + } + + return ( +
+
+ {fileName} + {lineNumber && ( + Line {lineNumber} + )} +
+
+
{sourceCode}
+
+
+ ); +}; +``` + +**How this works:** + +1. When a node is selected, it finds the corresponding code element +2. It displays the source code with line numbers +3. It highlights the relevant part of the code + +**Why is this important?** It bridges the gap between the abstract graph and the concrete code, helping users understand what they're seeing. + +## Phase 5: RAG Chat Interface + +### Step 11: Implement LLM Service + +**Why this matters:** The chat interface needs to connect to LLMs (Large Language Models) to translate natural language to graph queries. + +**Key concepts:** + +- We'll support multiple LLM providers (OpenAI, Anthropic, Gemini) +- Users provide their own API keys (privacy-focused) +- We need a consistent interface for different providers + +**Implementation:** + +```typescript +import { ChatOpenAI } from 'langchain/chat_models/openai'; +import { ChatAnthropic } from 'langchain/chat_models/anthropic'; +import { ChatGoogleGenerativeAI } from '@langchain/google-genai'; + +export type LLMProvider = 'openai' | 'anthropic' | 'gemini'; + +export interface LLMConfig { + provider: LLMProvider; + apiKey: string; + model?: string; +} + +export class LLMService { + private config: LLMConfig; + + constructor(config: LLMConfig) { + this.config = config; + } + + getChatModel() { + switch (this.config.provider) { + case 'openai': + return new ChatOpenAI({ + apiKey: this.config.apiKey, + modelName: this.config.model || 'gpt-4-turbo', + temperature: 0 + }); + case 'anthropic': + return new ChatAnthropic({ + apiKey: this.config.apiKey, + modelName: this.config.model || 'claude-3-sonnet-20240229', + temperature: 0 + }); + case 'gemini': + return new ChatGoogleGenerativeAI({ + apiKey: this.config.apiKey, + modelName: this.config.model || 'gemini-1.5-pro-latest', + temperature: 0 + }); + default: + throw new Error(`Unsupported LLM provider: ${this.config.provider}`); + } + } +} +``` + +**How this works:** + +1. The service takes an LLM configuration (provider, API key, model) +2. It returns a consistent chat model interface regardless of provider +3. It handles provider-specific initialization + +**Why support multiple providers?** Different users have different preferences and API key availability. + +### Step 12: Implement Cypher Generator + +**Why this matters:** The core of the RAG system - translating natural language questions to graph queries. + +**Key concepts:** + +- We use a system prompt to instruct the LLM +- The prompt includes our graph schema +- We clean the response to get a valid Cypher query + +**Implementation:** + +```typescript +import { BaseChatModel } from 'langchain/chat_models/base'; +import { CYPHER_SYSTEM_PROMPT } from '../prompts/cypher'; + +export class CypherGenerator { + private llm: BaseChatModel; + + constructor(llm: BaseChatModel) { + this.llm = llm; + } + + async generate(naturalLanguageQuery: string): Promise { + const response = await this.llm.call([ + { role: 'system', content: CYPHER_SYSTEM_PROMPT }, + { role: 'user', content: naturalLanguageQuery } + ]); + + return this.cleanResponse(response.content); + } + + private cleanResponse(response: string): string { + // Remove markdown code blocks + let cleaned = response.replace(/```cypher/g, '').replace(/```/g, ''); + // Ensure it ends with a semicolon + if (!cleaned.trim().endsWith(';')) { + cleaned = cleaned.trim() + ';'; + } + return cleaned; + } +} +``` + +**How this works:** + +1. It sends the natural language query with a system prompt to the LLM +2. The system prompt teaches the LLM about our graph structure +3. It cleans the response to extract a valid Cypher query + +**Why is the system prompt important?** It provides the LLM with the context it needs to generate correct queries. Without it, the LLM wouldn't know about our graph schema. + +### Step 13: Implement RAG Orchestrator + +**Why this matters:** This is the "brain" of the system that coordinates the query process. + +**Key concepts:** + +- It follows a ReAct (Reason + Act) pattern +- It plans steps, uses tools, observes results, and responds +- It prevents hallucination by sticking to tool results + +**Implementation:** + +```typescript +import { BaseChatModel } from 'langchain/chat_models/base'; +import { RAG_ORCHESTRATOR_SYSTEM_PROMPT } from '../prompts/rag-orchestrator'; + +export class RAGOrchestrator { + private llm: BaseChatModel; + + constructor(llm: BaseChatModel) { + this.llm = llm; + } + + async query( + userQuery: string, + queryGraph: (cypher: string) => Promise, + retrieveCode: (nodeId: string) => Promise + ) { + // Start with the system prompt + let conversation = [ + { role: 'system', content: RAG_ORCHESTRATOR_SYSTEM_PROMPT } + ]; + + // Add the user's question + conversation.push({ role: 'user', content: userQuery }); + + // Simple ReAct loop + for (let i = 0; i < 5; i++) { // Max 5 steps + const response = await this.llm.call(conversation); + const responseContent = response.content; + + // Check if the response contains a tool call + if (responseContent.includes('Action: query_graph')) { + const match = responseContent.match(/Action Input: (.*)/); + if (match) { + const cypherQuery = match[1].trim(); + + // Execute the query + const queryResults = await queryGraph(cypherQuery); + + // Add the observation to the conversation + conversation.push({ + role: 'assistant', + content: responseContent + }); + + conversation.push({ + role: 'system', + content: `Observation: ${JSON.stringify(queryResults)}` + }); + + // If we have results, we might be done + if (queryResults.length > 0) { + break; + } + } + } + else if (responseContent.includes('Action: retrieve_code')) { + // Similar handling for code retrieval + } + else { + // This appears to be the final answer + return responseContent; + } + } + + // If we got here without a final answer, generate one + conversation.push({ + role: 'user', + content: 'Please provide your final answer based on the information gathered.' + }); + + const finalResponse = await this.llm.call(conversation); + return finalResponse.content; + } +} +``` + +**How this works:** + +1. It starts with a system prompt that defines the rules +2. It sends the user's query to the LLM +3. The LLM responds with either: + - A tool call (query_graph or retrieve_code) + - A final answer +4. If it's a tool call, it executes the tool and adds the result to the conversation +5. It repeats until it gets a final answer or hits the step limit + +**Why the step limit?** To prevent infinite loops if the LLM gets stuck. + +## Phase 6: Main Application Integration + +### Step 14: Create Main Application Component + +**Why this matters:** This brings all the pieces together into a cohesive UI. + +**Implementation:** + +```tsx +import React, { useState, useRef } from 'react'; +import { GraphVisualization } from '@/ui/components/graph/Visualization'; +import { GraphControls } from '@/ui/components/graph/Controls'; +import { SourceViewer } from '@/ui/components/graph/SourceViewer'; +import { ChatInterface } from '@/ui/components/chat/ChatInterface'; +import { KnowledgeGraph } from '@/core/graph/types'; +import { GitHubService } from '@/services/github'; +import { ZipService } from '@/services/zip'; +import { ingestionWorkerApi } from '@/lib/workerUtils'; + +export const HomePage = () => { + const [graph, setGraph] = useState(null); + const [selectedNodeId, setSelectedNodeId] = useState(null); + const [isLoading, setIsLoading] = useState(false); + const [error, setError] = useState(null); + const [repoUrl, setRepoUrl] = useState(''); + const fileInputRef = useRef(null); + + const handleRepoSubmit = async (e: React.FormEvent) => { + e.preventDefault(); + if (!repoUrl.trim() || isLoading) return; + + setIsLoading(true); + setError(null); + + try { + // Parse the GitHub URL + const urlMatch = repoUrl.match(/github\.com\/([^/]+)\/([^/]+)/); + if (!urlMatch) { + throw new Error('Invalid GitHub repository URL'); + } + + const owner = urlMatch[1]; + const repo = urlMatch[2].replace(/\.git$/, ''); + + // Fetch repository contents + const githubService = new GitHubService(); + const contents = await githubService.getRepoContents(owner, repo); + + // Filter for Python files + const pythonFiles = contents + .filter((item: any) => item.type === 'file' && item.name.endsWith('.py')) + .map((item: any) => item.path); + + // Fetch file contents + const fileContents: Record = {}; + for (const filePath of pythonFiles) { + fileContents[filePath] = await githubService.getFileContent(owner, repo, filePath); + } + + // Process the repository + const projectName = `${owner}/${repo}`; + const processedGraph = await ingestionWorkerApi.processRepository( + repoUrl, + projectName, + pythonFiles, + fileContents + ); + + setGraph(processedGraph); + } catch (err) { + setError(err instanceof Error ? err.message : 'Failed to process repository'); + console.error('Processing error:', err); + } finally { + setIsLoading(false); + } + }; + + const handleQuery = async (query: string): Promise => { + if (!graph) { + throw new Error('No graph available'); + } + + // In a real implementation, this would use the RAG orchestrator + return `I found information related to "${query}" in the codebase.`; + }; + + return ( +
+ {/* Header with repository input */} +
+
+

GitNexus

+ +
+
+
+ setRepoUrl(e.target.value)} + placeholder="https://github.com/owner/repo.git" + className="flex-1 px-3 py-2 border border-gray-300 rounded-md focus:outline-none focus:ring-1 focus:ring-blue-500 focus:border-blue-500" + /> + +
+
+ +
+ or + + +
+
+
+
+ +
+ {/* Graph Visualization Pane */} +
+
+ {graph ? ( + <> + {}} + onLayoutChange={() => {}} + /> +
+ +
+ + ) : ( +
+ {isLoading ? ( +
+
+

Processing repository...

+
+ ) : ( +

Enter a repository URL or upload a ZIP to get started

+ )} +
+ )} +
+
+ + {/* Right Panel */} +
+ {/* Source Viewer */} +
+
+

Source Code

+
+ {graph ? ( + + ) : ( +
+

Select a node to view source

+
+ )} +
+ + {/* Chat Interface */} +
+
+

Ask About Code

+
+ {graph ? ( + + ) : ( +
+

Process a repository to ask questions about the code

+
+ )} +
+
+
+
+ ); +}; +``` + +**How this works:** + +1. The header has inputs for GitHub URLs and ZIP uploads +2. The main area has two panes: + - Left: Graph visualization + - Right: Source viewer and chat interface +3. When a repository is processed, the graph is displayed +4. Users can click nodes to see source code and ask questions + +**Why this layout?** It provides a cohesive experience where users can: + +- See the big picture (graph) +- Drill down to specific code (source viewer) +- Ask questions about what they're seeing (chat) + +## Final Steps: Testing and Optimization + +### Step 15: Add Error Boundaries + +**Why this matters:** Inevitably, something will go wrong. We want to handle errors gracefully. + +**Implementation:** + +```tsx +import React, { Component, ErrorInfo, ReactNode } from 'react'; + +interface Props { + children: ReactNode; +} + +interface State { + hasError: boolean; + error: Error | null; +} + +export class ErrorBoundary extends Component { + public state: State = { + hasError: false, + error: null + }; + + public static getDerivedStateFromError(error: Error): State { + return { hasError: true, error }; + } + + public componentDidCatch(error: Error, errorInfo: ErrorInfo) { + console.error("Uncaught error:", error, errorInfo); + } + + public render() { + if (this.state.hasError) { + return ( +
+

Something went wrong

+

{this.state.error?.message}

+ +
+ ); + } + + return this.props.children; + } +} +``` + +**How this works:** + +- It catches JavaScript errors in child components +- It displays a friendly error message instead of a blank screen +- It allows users to try again without losing their work + +**Why use error boundaries?** They prevent a single error from breaking the entire application. + +### Step 16: Implement Performance Optimizations + +**Why this matters:** Large repositories can be slow to process. We need to keep the UI responsive. + +**Key optimizations:** + +1. **Web Workers**: Already implemented for graph processing +2. **Progress Indicators**: Show users what's happening +3. **File Filtering**: Only process relevant files +4. **Lazy Loading**: Load components as needed + +**Implementation:** + +```tsx +// Add to your GitHub processing function +const MAX_FILES = 500; // Limit for free tier +if (pythonFiles.length > MAX_FILES) { + // Offer to filter by directory or file pattern + const shouldFilter = window.confirm( + `Repository has ${pythonFiles.length} Python files (max ${MAX_FILES}). ` + + `Would you like to filter by directory or file pattern?` + ); + if (shouldFilter) { + const filterPattern = prompt( + "Enter a directory path or file pattern to filter (e.g., 'src/', '*.py')", + "src/" + ); + if (filterPattern) { + const filteredFiles = pythonFiles.filter(file => + file.includes(filterPattern) || file.endsWith(filterPattern) + ); + pythonFiles = filteredFiles; + } + } +} +``` + +**Why limit file processing?** Processing too many files can: + +- Freeze the browser tab +- Exceed GitHub API rate limits +- Use excessive memory + +### Step 17: Add Export Functionality + +**Why this matters:** Users might want to save or share their generated graphs. + +**Implementation:** + +```tsx +export function exportGraphToJson(graph: KnowledgeGraph): string { + return JSON.stringify(graph, null, 2); +} + +export function downloadGraph(graph: KnowledgeGraph, filename: string = 'gitnexus-graph.json') { + const json = exportGraphToJson(graph); + const blob = new Blob([json], { type: 'application/json' }); + const url = URL.createObjectURL(blob); + const a = document.createElement('a'); + a.href = url; + a.download = filename; + document.body.appendChild(a); + a.click(); + document.body.removeChild(a); + URL.revokeObjectURL(url); +} +``` + +**How this works:** + +1. Converts the graph to JSON +2. Creates a downloadable file +3. Triggers a download + +**Why include export?** It allows users to: + +- Save their work for later +- Share graphs with teammates +- Use the data in other tools + +## Conclusion + +This implementation guide has walked you through building a complete edge-based code knowledge graph generator using Deno. By following these steps, you'll create a privacy-focused tool that runs entirely in the user's browser. + +**Key advantages of this approach:** + +- **Zero server costs**: All processing happens in the user's browser +- **Strong privacy**: Code never leaves the user's machine +- **Modular architecture**: Easy to add more languages later +- **Clear separation of concerns**: Makes the codebase maintainable +- **Deno compatibility**: Modern runtime with built-in TypeScript support + +Remember to start small (Python support only) and iterate, adding more features and language support as you validate the core functionality. The most important part is getting the graph construction pipeline working correctly - everything else builds on that foundation. + +Good luck with your implementation of GitNexus! + +================ +File: public/vite.svg +================ + + +================ +File: README.md +================ +# GitNexus: Edge-Based Code Knowledge Graph Generator for Deno - Step-by-Step Implementation Guide + +This guide will walk you through building a fully edge-based code knowledge graph generator from scratch using Deno. I'll explain each concept before showing the implementation, so you understand **why** we're doing something, not just **how** to do it. + +## Phase 1: Project Setup & Core Infrastructure + +### Step 1: Project Structure and Tooling Setup + +**Why this matters:** Before writing any code, we need to set up our development environment properly. A well-structured project makes it easier to add features later and keeps everything organized. + +**Key concepts:** + +- We're using Vite (a modern build tool) with React and TypeScript +- We need special configuration for WebAssembly (WASM) files +- A clear directory structure helps us scale to multiple languages later + +**Implementation Steps:** + +1. **Create the base project:** + +```bash +# Create project root +mkdir GitNexus +cd GitNexus + +# Initialize Vite project with React and TypeScript +npm create vite@latest . -- --template react-ts + +# Initialize Deno project +deno init +``` + +2. **Create the application directory structure:** + +```bash +# Create directories for our core components +mkdir -p src/{core,core/tree-sitter,core/graph,core/ingestion,services,ai,ai/agents,ai/prompts,ui,ui/components,ui/components/graph,ui/components/chat,ui/hooks,workers,lib,config,store} +``` + +**Why this structure?** + +- `core/`: Contains the engine that builds the knowledge graph +- `services/`: Handles external interactions (GitHub API, ZIP processing) +- `ai/`: Contains the RAG and chat functionality +- `ui/`: All user interface components +- `workers/`: Web Workers for heavy processing (keeps UI responsive) +- `lib/`: Utility functions used throughout the app + +### Step 2: Configure Build Tools for WASM + +**Why this matters:** WebAssembly (WASM) is how we'll run the Tree-sitter parsers in the browser. We need special configuration to handle these binary files correctly. + +**Key concepts:** + +- WASM files are binary files that run at near-native speed in browsers +- Vite needs special configuration to handle them properly +- We want to avoid inlining large WASM files in our JavaScript bundles + +**Implementation:** + +1. **Update `vite.config.ts`:** + +```typescript +import { defineConfig } from 'vite' +import react from '@vitejs/plugin-react' +export default defineConfig({ + plugins: [react()], + worker: { + format: 'es' + }, + assetsInclude: ['**/*.wasm'], + build: { + target: 'esnext', + assetsInlineLimit: 0 // Don't inline WASM files + } +}) +``` + +**What this does:** + +- `assetsInclude: ['**/*.wasm']` tells Vite to treat WASM files as assets +- `assetsInlineLimit: 0` ensures WASM files aren't inlined into JavaScript (they're too large) +- `worker: { format: 'es' }` configures Web Workers to use ES modules + +2. **Configure TypeScript** with a `tsconfig.json` that has strict settings for better code quality. + +**Why strict settings?** They help catch errors early and make the code more maintainable as the project grows. + +### Step 3: Set Up WASM Parser Infrastructure + +**Why this matters:** Tree-sitter is the engine that parses code into ASTs (Abstract Syntax Trees). We need to get these parsers working in the browser via WASM. + +**Key concepts:** + +- Tree-sitter parsers for different languages are written in C +- We compile them to WASM so they can run in browsers +- We need to load these parsers on demand + +**Implementation:** + +1. **Create a public directory for WASM files:** + +```bash +mkdir -p public/wasm/python +``` + +2. **Download the Tree-sitter Python parser:** + - Get `tree-sitter-python.wasm` from [tree-sitter-python releases](https://github.com/tree-sitter/tree-sitter-python/releases) + - Place it in `public/wasm/python/` + +**Why host WASM files separately?** Browsers can't access the user's file system directly for security reasons. We need to serve the WASM files from a URL. + +I'm building a Deno 2.4.1-based edge code knowledge graph generator called GitNexus. I've completed Phase 1 (project setup and core infrastructure) and now need to implement Phase 2 (Code Acquisition Module) with Deno 2.4.1 compatibility. + +Please generate the following two services with these specific requirements: + +## 1. GitHub Service (Deno 2.4.1 Implementation) + +Create a GitHubService class in `src/services/github.ts` that: + +- Uses Deno 2.4.1's native fetch API (no Node.js dependencies) +- Handles GitHub API authentication via personal access tokens +- Implements rate limit handling (GitHub allows 5,000 requests/hour with token) +- Includes methods to: + * getRepoContents(owner: string, repo: string, path = '') - fetches directory structure + * getFileContent(owner: string, repo: string, filePath: string) - fetches individual file content +- Properly handles GitHub API rate limits and errors +- Uses Deno 2.4.1-specific error handling patterns +- Includes TypeScript interfaces for return types +- Has comprehensive comments explaining key implementation choices + +Important Deno 2.4.1 considerations: + +- Use ES modules (no CommonJS) +- No Node.js-specific modules (use Deno's built-in APIs where possible) +- Handle fetch responses with proper Deno error patterns +- Include proper types for all functions +- Follow Deno 2.4.1's security model (permissions) + +## 2. ZIP Processing Service (Deno 2.4.1 Implementation) + +Create a ZipService class in `src/services/zip.ts` that: + +- Uses Deno-compatible ZIP processing (use `https://deno.land/x/zip@v1.2.3/mod.ts` instead of JSZip) +- Processes uploaded ZIP files containing code repositories +- Extracts file paths and contents into a Map +- Handles binary data properly in Deno 2.4.1 environment +- Includes error handling for corrupted ZIP files +- Has TypeScript interfaces for all types +- Includes comprehensive comments + +Important Deno 2.4.1 considerations: + +- Use Deno's file system APIs where appropriate +- Handle file reading with Deno.readFile() +- Process ZIP entries without blocking the event loop +- Use Deno's native text decoding for file contents +- Implement streaming where possible for large ZIP files +- Note that this is for a browser-based application, so the ZIP service should work with File objects from HTML inputs + +## Additional Requirements + +- All code must be Deno 2.4.1 compatible +- Use strict TypeScript with deno-lint directives where needed +- Include proper error messages that help with debugging +- Add unit test stubs for both services (using Deno's built-in test runner) +- Follow the same directory structure as Phase 1 (services directory already exists) +- Maintain the same coding style and patterns established in Phase 1 +- Include necessary imports from Deno's standard library +- Document any Deno-specific permissions required + +I'm building a Deno 2.4.1-based edge code knowledge graph generator called GitNexus. I've completed Phase 1 (project setup and core infrastructure) and now need to implement Phase 2 (Code Acquisition Module) with Deno 2.4.1 compatibility. + +Please generate the following two services with these specific requirements: + +## 1. GitHub Service (Deno 2.4.1 Implementation) + +Create a GitHubService class in `src/services/github.ts` that: + +- Uses Deno 2.4.1's native fetch API (no Node.js dependencies) +- Handles GitHub API authentication via personal access tokens +- Implements rate limit handling (GitHub allows 5,000 requests/hour with token) +- Includes methods to: + * getRepoContents(owner: string, repo: string, path = '') - fetches directory structure + * getFileContent(owner: string, repo: string, filePath: string) - fetches individual file content +- Properly handles GitHub API rate limits and errors +- Uses Deno 2.4.1-specific error handling patterns +- Includes TypeScript interfaces for return types +- Has comprehensive comments explaining key implementation choices + +Important Deno 2.4.1 considerations: + +- Use ES modules (no CommonJS) +- No Node.js-specific modules (use Deno's built-in APIs where possible) +- Handle fetch responses with proper Deno error patterns +- Include proper types for all functions +- Follow Deno 2.4.1's security model (permissions) + +## 2. ZIP Processing Service (Deno 2.4.1 Implementation) + +Create a ZipService class in `src/services/zip.ts` that: + +- Uses Deno-compatible ZIP processing (use `https://deno.land/x/zip@v1.2.3/mod.ts` instead of JSZip) +- Processes uploaded ZIP files containing code repositories +- Extracts file paths and contents into a Map +- Handles binary data properly in Deno 2.4.1 environment +- Includes error handling for corrupted ZIP files +- Has TypeScript interfaces for all types +- Includes comprehensive comments + +Important Deno 2.4.1 considerations: + +- Use Deno's file system APIs where appropriate +- Handle file reading with Deno.readFile() +- Process ZIP entries without blocking the event loop +- Use Deno's native text decoding for file contents +- Implement streaming where possible for large ZIP files +- Note that this is for a browser-based application, so the ZIP service should work with File objects from HTML inputs + +## Additional Requirements + +- All code must be Deno 2.4.1 compatible +- Use strict TypeScript with deno-lint directives where needed +- Include proper error messages that help with debugging +- Add unit test stubs for both services (using Deno's built-in test runner) +- Follow the same directory structure as Phase 1 (services directory already exists) +- Maintain the same coding style and patterns established in Phase 1 +- Include necessary imports from Deno's standard library +- Document any Deno-specific permissions required + +I'm building a Deno 2.4.1-based edge code knowledge graph generator called GitNexus. I've completed Phase 1 (project setup and core infrastructure) and now need to implement Phase 2 (Code Acquisition Module) with Deno 2.4.1 compatibility. + +Please generate the following two services with these specific requirements: + +## 1. GitHub Service (Deno 2.4.1 Implementation) + +Create a GitHubService class in `src/services/github.ts` that: + +- Uses Deno 2.4.1's native fetch API (no Node.js dependencies) +- Handles GitHub API authentication via personal access tokens +- Implements rate limit handling (GitHub allows 5,000 requests/hour with token) +- Includes methods to: + * getRepoContents(owner: string, repo: string, path = '') - fetches directory structure + * getFileContent(owner: string, repo: string, filePath: string) - fetches individual file content +- Properly handles GitHub API rate limits and errors +- Uses Deno 2.4.1-specific error handling patterns +- Includes TypeScript interfaces for return types +- Has comprehensive comments explaining key implementation choices + +Important Deno 2.4.1 considerations: + +- Use ES modules (no CommonJS) +- No Node.js-specific modules (use Deno's built-in APIs where possible) +- Handle fetch responses with proper Deno error patterns +- Include proper types for all functions +- Follow Deno 2.4.1's security model (permissions) + +## 2. ZIP Processing Service (Deno 2.4.1 Implementation) + +Create a ZipService class in `src/services/zip.ts` that: + +- Uses Deno-compatible ZIP processing (use `https://deno.land/x/zip@v1.2.3/mod.ts` instead of JSZip) +- Processes uploaded ZIP files containing code repositories +- Extracts file paths and contents into a Map +- Handles binary data properly in Deno 2.4.1 environment +- Includes error handling for corrupted ZIP files +- Has TypeScript interfaces for all types +- Includes comprehensive comments + +Important Deno 2.4.1 considerations: + +- Use Deno's file system APIs where appropriate +- Handle file reading with Deno.readFile() +- Process ZIP entries without blocking the event loop +- Use Deno's native text decoding for file contents +- Implement streaming where possible for large ZIP files +- Note that this is for a browser-based application, so the ZIP service should work with File objects from HTML inputs + +## Additional Requirements + +- All code must be Deno 2.4.1 compatible +- Use strict TypeScript with deno-lint directives where needed +- Include proper error messages that help with debugging +- Add unit test stubs for both services (using Deno's built-in test runner) +- Follow the same directory structure as Phase 1 (services directory already exists) +- Maintain the same coding style and patterns established in Phase 1 +- Include necessary imports from Deno's standard library +- Document any Deno-specific permissions required + +I'm building a Deno 2.4.1-based edge code knowledge graph generator called GitNexus. I've completed Phase 1 (project setup and core infrastructure) and now need to implement Phase 2 (Code Acquisition Module) with Deno 2.4.1 compatibility. + +Please generate the following two services with these specific requirements: + +## 1. GitHub Service (Deno 2.4.1 Implementation) + +Create a GitHubService class in `src/services/github.ts` that: + +- Uses Deno 2.4.1's native fetch API (no Node.js dependencies) +- Handles GitHub API authentication via personal access tokens +- Implements rate limit handling (GitHub allows 5,000 requests/hour with token) +- Includes methods to: + * getRepoContents(owner: string, repo: string, path = '') - fetches directory structure + * getFileContent(owner: string, repo: string, filePath: string) - fetches individual file content +- Properly handles GitHub API rate limits and errors +- Uses Deno 2.4.1-specific error handling patterns +- Includes TypeScript interfaces for return types +- Has comprehensive comments explaining key implementation choices + +Important Deno 2.4.1 considerations: + +- Use ES modules (no CommonJS) +- No Node.js-specific modules (use Deno's built-in APIs where possible) +- Handle fetch responses with proper Deno error patterns +- Include proper types for all functions +- Follow Deno 2.4.1's security model (permissions) + +## 2. ZIP Processing Service (Deno 2.4.1 Implementation) + +Create a ZipService class in `src/services/zip.ts` that: + +- Uses Deno-compatible ZIP processing (use `https://deno.land/x/zip@v1.2.3/mod.ts` instead of JSZip) +- Processes uploaded ZIP files containing code repositories +- Extracts file paths and contents into a Map +- Handles binary data properly in Deno 2.4.1 environment +- Includes error handling for corrupted ZIP files +- Has TypeScript interfaces for all types +- Includes comprehensive comments + +Important Deno 2.4.1 considerations: + +- Use Deno's file system APIs where appropriate +- Handle file reading with Deno.readFile() +- Process ZIP entries without blocking the event loop +- Use Deno's native text decoding for file contents +- Implement streaming where possible for large ZIP files +- Note that this is for a browser-based application, so the ZIP service should work with File objects from HTML inputs + +## Additional Requirements + +- All code must be Deno 2.4.1 compatible +- Use strict TypeScript with deno-lint directives where needed +- Include proper error messages that help with debugging +- Add unit test stubs for both services (using Deno's built-in test runner) +- Follow the same directory structure as Phase 1 (services directory already exists) +- Maintain the same coding style and patterns established in Phase 1 +- Include necessary imports from Deno's standard library +- Document any Deno-specific permissions required + +I'm building a Deno 2.4.1-based edge code knowledge graph generator called GitNexus. I've completed Phase 1 (project setup and core infrastructure) and now need to implement Phase 2 (Code Acquisition Module) with Deno 2.4.1 compatibility. + +Please generate the following two services with these specific requirements: + +## 1. GitHub Service (Deno 2.4.1 Implementation) + +Create a GitHubService class in `src/services/github.ts` that: + +- Uses Deno 2.4.1's native fetch API (no Node.js dependencies) +- Handles GitHub API authentication via personal access tokens +- Implements rate limit handling (GitHub allows 5,000 requests/hour with token) +- Includes methods to: + * getRepoContents(owner: string, repo: string, path = '') - fetches directory structure + * getFileContent(owner: string, repo: string, filePath: string) - fetches individual file content +- Properly handles GitHub API rate limits and errors +- Uses Deno 2.4.1-specific error handling patterns +- Includes TypeScript interfaces for return types +- Has comprehensive comments explaining key implementation choices + +Important Deno 2.4.1 considerations: + +- Use ES modules (no CommonJS) +- No Node.js-specific modules (use Deno's built-in APIs where possible) +- Handle fetch responses with proper Deno error patterns +- Include proper types for all functions +- Follow Deno 2.4.1's security model (permissions) + +## 2. ZIP Processing Service (Deno 2.4.1 Implementation) + +Create a ZipService class in `src/services/zip.ts` that: + +- Uses Deno-compatible ZIP processing (use `https://deno.land/x/zip@v1.2.3/mod.ts` instead of JSZip) +- Processes uploaded ZIP files containing code repositories +- Extracts file paths and contents into a Map +- Handles binary data properly in Deno 2.4.1 environment +- Includes error handling for corrupted ZIP files +- Has TypeScript interfaces for all types +- Includes comprehensive comments + +Important Deno 2.4.1 considerations: + +- Use Deno's file system APIs where appropriate +- Handle file reading with Deno.readFile() +- Process ZIP entries without blocking the event loop +- Use Deno's native text decoding for file contents +- Implement streaming where possible for large ZIP files +- Note that this is for a browser-based application, so the ZIP service should work with File objects from HTML inputs + +## Additional Requirements + +- All code must be Deno 2.4.1 compatible +- Use strict TypeScript with deno-lint directives where needed +- Include proper error messages that help with debugging +- Add unit test stubs for both services (using Deno's built-in test runner) +- Follow the same directory structure as Phase 1 (services directory already exists) +- Maintain the same coding style and patterns established in Phase - Include necessary imports from Deno's standard library +- Document any Deno-specific permissions required + +I'm building a Deno 2.4.1-based edge code knowledge graph generator called GitNexus. I've completed Phase 1 (project setup and core infrastructure) and now need to implement Phase 2 (Code Acquisition Module) with Deno 2.4.1 compatibility. + +Please generate the following two services with these specific requirements: + +## 1. GitHub Service (Deno 2.4.1 Implementation) + +Create a GitHubService class in `src/services/github.ts` that: + +- Uses Deno 2.4.1's native fetch API (no Node.js dependencies) +- Handles GitHub API authentication via personal access tokens +- Implements rate limit handling (GitHub allows 5,000 requests/hour with token) +- Includes methods to: + * getRepoContents(owner: string, repo: string, path = '') - fetches directory structure + * getFileContent(owner: string, repo: string, filePath: string) - fetches individual file content +- Properly handles GitHub API rate limits and errors +- Uses Deno 2.4.1-specific error handling patterns +- Includes TypeScript interfaces for return types +- Has comprehensive comments explaining key implementation choices + +Important Deno 2.4.1 considerations: + +- Use ES modules (no CommonJS) +- No Node.js-specific modules (use Deno's built-in APIs where possible) +- Handle fetch responses with proper Deno error patterns +- Include proper types for all functions +- Follow Deno 2.4.1's security model (permissions) + +## 2. ZIP Processing Service (Deno 2.4.1 Implementation) + +Create a ZipService class in `src/services/zip.ts` that: + +- Uses Deno-compatible ZIP processing (use `https://deno.land/x/zip@v1.2.3/mod.ts` instead of JSZip) +- Processes uploaded ZIP files containing code repositories +- Extracts file paths and contents into a Map +- Handles binary data properly in Deno 2.4.1 environment +- Includes error handling for corrupted ZIP files +- Has TypeScript interfaces for all types +- Includes comprehensive comments + +Important Deno 2.4.1 considerations: + +- Use Deno's file system APIs where appropriate +- Handle file reading with Deno.readFile() +- Process ZIP entries without blocking the event loop +- Use Deno's native text decoding for file contents +- Implement streaming where possible for large ZIP files +- Note that this is for a browser-based application, so the ZIP service should work with File objects from HTML inputs + +## Additional Requirements + +- All code must be Deno 2.4.1 compatible +- Use strict TypeScript with deno-lint directives where needed +- Include proper error messages that help with debugging +- Add unit test stubs for both services (using Deno's built-in test runner) +- Follow the same directory structure as Phase 1 (services directory already exists) +- Maintain the same coding style and patterns established in Phase - Include necessary imports from Deno's standard library +- Document any Deno-specific permissions required + +I'm building a Deno 2.4.1-based edge code knowledge graph generator called GitNexus. I've completed Phase 1 (project setup and core infrastructure) and now need to implement Phase 2 (Code Acquisition Module) with Deno 2.4.1 compatibility. + +Please generate the following two services with these specific requirements: + +## 1. GitHub Service (Deno 2.4.1 Implementation) + +Create a GitHubService class in `src/services/github.ts` that: + +- Uses Deno 2.4.1's native fetch API (no Node.js dependencies) +- Handles GitHub API authentication via personal access tokens +- Implements rate limit handling (GitHub allows 5,000 requests/hour with token) +- Includes methods to: + * getRepoContents(owner: string, repo: string, path = '') - fetches directory structure + * getFileContent(owner: string, repo: string, filePath: string) - fetches individual file content +- Properly handles GitHub API rate limits and errors +- Uses Deno 2.4.1-specific error handling patterns +- Includes TypeScript interfaces for return types +- Has comprehensive comments explaining key implementation choices + +Important Deno 2.4.1 considerations: + +- Use ES modules (no CommonJS) +- No Node.js-specific modules (use Deno's built-in APIs where possible) +- Handle fetch responses with proper Deno error patterns +- Include proper types for all functions +- Follow Deno 2.4.1's security model (permissions) + +## 2. ZIP Processing Service (Deno 2.4.1 Implementation) + +Create a ZipService class in `src/services/zip.ts` that: + +- Uses Deno-compatible ZIP processing (use `https://deno.land/x/zip@v1.2.3/mod.ts` instead of JSZip) +- Processes uploaded ZIP files containing code repositories +- Extracts file paths and contents into a Map +- Handles binary data properly in Deno 2.4.1 environment +- Includes error handling for corrupted ZIP files +- Has TypeScript interfaces for all types +- Includes comprehensive comments + +Important Deno 2.4.1 considerations: + +- Use Deno's file system APIs where appropriate +- Handle file reading with Deno.readFile() +- Process ZIP entries without blocking the event loop +- Use Deno's native text decoding for file contents +- Implement streaming where possible for large ZIP files +- Note that this is for a browser-based application, so the ZIP service should work with File objects from HTML inputs + +## Additional Requirements + +- All code must be Deno 2.4.1 compatible +- Use strict TypeScript with deno-lint directives where needed +- Include proper error messages that help with debugging +- Add unit test stubs for both services (using Deno's built-in test runner) +- Follow the same directory structure as Phase 1 (services directory already exists) +- Maintain the same coding style and patterns established in Phase - Include necessary imports from Deno's standard library +- Document any Deno-specific permissions required + +3. **Create a loader for Tree-sitter parsers:** + +```typescript +import WebTreeSitter from 'web-tree-sitter'; +let parserInstance: WebTreeSitter | null = null; +const parserCache = new Map(); + +export async function initTreeSitter() { + if (parserInstance) return parserInstance; + parserInstance = await WebTreeSitter.init(); + return parserInstance; +} + +export async function loadPythonParser(): Promise { + if (parserCache.has('python')) { + return parserCache.get('python')!; + } + const Parser = await initTreeSitter(); + const pythonLang = await Parser.Language.load( + '/wasm/python/tree-sitter-python.wasm' + ); + parserCache.set('python', pythonLang); + return pythonLang; +} +``` + +**How this works:** + +1. `initTreeSitter()` initializes the WebAssembly module once +2. `loadPythonParser()` loads the Python parser from the WASM file +3. We cache parsers to avoid reloading them multiple times + +**Why cache parsers?** Loading WASM files is relatively slow, so we want to do it once and reuse the parsers. + +## Phase 2: Code Acquisition Module + +### Step 4: Implement GitHub API Integration + +**Why this matters:** Users will want to analyze public GitHub repositories, so we need a way to fetch code from GitHub. + +**Key concepts:** + +- GitHub has a REST API for accessing repository contents +- We need to handle rate limits (GitHub limits how many requests you can make) +- We'll let users provide their own API tokens for higher limits + +**Implementation:** + +```typescript +export class GitHubService { + private token: string | null = null; + + setToken(token: string) { + this.token = token; + } + + async getRepoContents(owner: string, repo: string, path = '') { + const headers: HeadersInit = { + 'Accept': 'application/vnd.github.v3+json' + }; + if (this.token) { + headers['Authorization'] = `token ${this.token}`; + } + + const response = await fetch( + `https://api.github.com/repos/${owner}/${repo}/contents/${path}`, + { headers } + ); + + if (!response.ok) { + throw new Error(`GitHub API error: ${response.status}`); + } + + return response.json(); + } +} +``` + +**How this works:** + +- `getRepoContents()` fetches the directory structure of a repository +- It uses the GitHub API with proper headers +- It handles authentication via a token + +**Important note:** GitHub API has rate limits. For unauthenticated requests, it's about 60 requests/hour. With a token, it's 5,000/hour. + +### Step 5: Implement ZIP Processing + +**Why this matters:** Not all code is on GitHub. Users might want to analyze local code or private repositories by uploading a ZIP file. + +**Key concepts:** + +- JSZip is a library for handling ZIP files in JavaScript +- We need to extract files and their contents from the ZIP +- We'll use a Map to store file paths and contents + +**Implementation:** + +```typescript +import JSZip from 'jszip'; + +export class ZipService { + async processZip(file: File): Promise> { + const zip = await JSZip.loadAsync(file); + const files = new Map(); + + for (const [filePath, zipEntry] of Object.entries(zip.files)) { + if (!zipEntry.dir) { + const content = await zipEntry.async('text'); + files.set(filePath, content); + } + } + + return files; + } +} +``` + +**How this works:** + +1. `JSZip.loadAsync(file)` loads the ZIP file +2. We iterate through all entries in the ZIP +3. For each file (not directory), we extract its content as text +4. We store the file path and content in a Map + +**Why use a Map?** It provides O(1) lookups by file path, which is important when we need to find files during graph construction. + +## Phase 3: Graph Construction Pipeline + +### Step 6: Define Graph Data Structures + +**Why this matters:** Before we can build a graph, we need to define what nodes and relationships look like. + +**Key concepts:** + +- A knowledge graph consists of nodes and relationships +- Nodes represent code elements (functions, classes, etc.) +- Relationships represent connections between elements (calls, contains, etc.) + +**Implementation:** + +```typescript +export type NodeLabel = + | 'Project' + | 'Package' + | 'Module' + | 'Folder' + | 'File' + | 'Class' + | 'Function' + | 'Method' + | 'Variable'; + +export interface GraphNode { + id: string; + label: NodeLabel; + properties: Record; +} + +export type RelationshipType = + | 'CONTAINS' + | 'CALLS' + | 'INHERITS' + | 'OVERRIDES' + | 'IMPORTS'; + +export interface GraphRelationship { + id: string; + type: RelationshipType; + source: string; + target: string; + properties?: Record; +} + +export interface KnowledgeGraph { + nodes: GraphNode[]; + relationships: GraphRelationship[]; +} +``` + +**Why these specific types?** + +- `NodeLabel` defines all possible types of code elements we'll track +- `RelationshipType` defines how code elements connect to each other +- `KnowledgeGraph` is the complete structure we'll build + +**Important relationships:** + +- `CONTAINS`: A folder contains files, a file contains functions +- `CALLS`: A function calls another function +- `IMPORTS`: One module imports from another + +### Step 7: Implement the 3-Pass Ingestion Pipeline + +**Why this matters:** Building a complete knowledge graph requires multiple passes to handle cross-file references properly. + +**Key concepts:** + +- **Pass 1**: Identify the overall structure (folders, modules) +- **Pass 2**: Parse individual files and cache ASTs +- **Pass 3**: Process function calls across files (the hardest part) + +This three-pass approach solves the "island problem" - where functions in different files appear disconnected. + +#### Pass 1: Structure Identification + +```typescript +export class StructureProcessor { + private graph: KnowledgeGraph; + private projectRoot: string; + private projectName: string; + + constructor(graph: KnowledgeGraph, projectRoot: string, projectName: string) { + this.graph = graph; + this.projectRoot = projectRoot; + this.projectName = projectName; + } + + identifyStructure(filePaths: string[]): void { + // Add Project node + this.graph.nodes.push({ + id: `project:${this.projectName}`, + label: 'Project', + properties: { name: this.projectName } + }); + + // Track directory structure + const directories = new Set(); + for (const filePath of filePaths) { + const dirPath = filePath.substring(0, filePath.lastIndexOf('/')); + if (dirPath && !directories.has(dirPath)) { + directories.add(dirPath); + // Create Folder node + this.graph.nodes.push({ + id: `folder:${dirPath}`, + label: 'Folder', + properties: { path: dirPath } + }); + + // Create CONTAINS relationship with parent + if (dirPath.includes('/')) { + const parentPath = dirPath.substring(0, dirPath.lastIndexOf('/')); + this.graph.relationships.push({ + id: `rel:folder:${dirPath}:parent`, + type: 'CONTAINS', + source: `folder:${parentPath}`, + target: `folder:${dirPath}` + }); + } else { + // Root folder connects to project + this.graph.relationships.push({ + id: `rel:folder:${dirPath}:project`, + type: 'CONTAINS', + source: `project:${this.projectName}`, + target: `folder:${dirPath}` + }); + } + } + } + } +} +``` + +**How this works:** + +1. Creates a root Project node +2. Walks through all file paths to identify directories +3. Creates Folder nodes and CONTAINS relationships + +**Why identify structure first?** We need to know the overall organization before parsing individual files. + +#### Pass 2: File Parsing + +```typescript +export class ParsingProcessor { + private graph: KnowledgeGraph; + private astCache = new Map(); + + constructor(graph: KnowledgeGraph) { + this.graph = graph; + } + + async parseFiles(filePaths: string[], fileContents: Map): Promise> { + for (const [filePath, content] of fileContents) { + if (filePath.endsWith('.py')) { + await this.parsePythonFile(filePath, content); + } + } + return this.astCache; + } + + private async parsePythonFile(filePath: string, content: string): Promise { + const parser = await loadPythonParser(); + const tree = parser.parse(content); + // Cache the AST + this.astCache.set(filePath, tree); + // Extract definitions from the AST + this.extractDefinitions(filePath, tree, content); + } + + private extractDefinitions(filePath: string, tree: any, content: string): void { + // Extract modules + this.graph.nodes.push({ + id: `module:${filePath}`, + label: 'Module', + properties: { + path: filePath, + name: filePath.split('/').pop()!.replace('.py', ''), + extension: '.py' + } + }); + + // Extract functions from the AST + const rootNode = tree.rootNode; + const functionDefs = rootNode.descendantsOfType('function_definition'); + for (const funcNode of functionDefs) { + const nameNode = funcNode.childForFieldName('name'); + const name = nameNode ? nameNode.text : 'unknown'; + + // Calculate position + const startLine = funcNode.startPosition.row + 1; + + // Create function node + this.graph.nodes.push({ + id: `function:${filePath}:${name}`, + label: 'Function', + properties: { + name, + qualified_name: `${this.getModuleName(filePath)}.${name}`, + path: filePath, + start_line: startLine + } + }); + + // Create CONTAINS relationship with module + this.graph.relationships.push({ + id: `rel:function:${filePath}:${name}:module`, + type: 'CONTAINS', + source: `module:${filePath}`, + target: `function:${filePath}:${name}` + }); + } + } +} +``` + +**How this works:** + +1. Parses each file with the appropriate Tree-sitter parser +2. Caches the AST for later use +3. Extracts definitions (functions, classes) from the AST +4. Creates nodes and relationships in the graph + +**Why cache ASTs?** We need them in Pass 3 to resolve cross-file function calls. + +#### Pass 3: Call Resolution + +```typescript +export class CallProcessor { + private graph: KnowledgeGraph; + private astCache: Map; + private projectRoot: string; + private projectName: string; + + constructor( + graph: KnowledgeGraph, + astCache: Map, + projectRoot: string, + projectName: string + ) { + this.graph = graph; + this.astCache = astCache; + this.projectRoot = projectRoot; + this.projectName = projectName; + } + + processCalls(): void { + for (const [filePath, tree] of this.astCache) { + if (filePath.endsWith('.py')) { + this.processPythonCalls(filePath, tree); + } + } + } + + private processPythonCalls(filePath: string, tree: any): void { + const rootNode = tree.rootNode; + // Find all call expressions + const callExpressions = rootNode.descendantsOfType('call'); + for (const callNode of callExpressions) { + const functionNameNode = callNode.childForFieldName('function'); + if (!functionNameNode) continue; + + // Handle different types of function references + let targetFunctionName = ''; + if (functionNameNode.type === 'identifier') { + targetFunctionName = functionNameNode.text; + } else if (functionNameNode.type === 'attribute') { + // Handle method calls like obj.method() + const attrNode = functionNameNode; + const objectNode = attrNode.childForFieldName('object'); + const attrNameNode = attrNode.childForFieldName('attribute'); + if (objectNode && attrNameNode) { + const objectName = objectNode.text; + const methodName = attrNameNode.text; + targetFunctionName = `${objectName}.${methodName}`; + } + } + + if (!targetFunctionName) continue; + + // Try to resolve the target function + const targetNode = this.resolveTargetFunction(targetFunctionName, filePath); + if (targetNode) { + // Create CALLS relationship + const callerId = this.getCallerId(callNode, filePath); + this.graph.relationships.push({ + id: `rel:call:${callerId}:${targetNode.id}`, + type: 'CALLS', + source: callerId, + target: targetNode.id + }); + } + } + } + + private resolveTargetFunction(targetName: string, currentFilePath: string): { id: string; type: string } | null { + // 1. Check if it's a built-in function + if (this.isBuiltInFunction(targetName)) { + return { + id: `builtin:${targetName}`, + type: 'builtin' + }; + } + + // 2. Check if it's an imported function + const importInfo = this.findImportForFunction(targetName, currentFilePath); + if (importInfo) { + const targetId = `function:${importInfo.sourceFile}:${importInfo.targetName}`; + return { + id: targetId, + type: 'imported' + }; + } + + // 3. Check if it's defined in the current file + for (const node of this.graph.nodes) { + if (node.label === 'Function' && + node.properties.name === targetName && + node.properties.path === currentFilePath) { + return { + id: node.id, + type: 'local' + }; + } + } + + return null; + } +} +``` + +**How this works:** + +1. Finds all function calls in the AST +2. Determines what function is being called +3. Resolves the target function across files using imports +4. Creates CALLS relationships in the graph + +**Why is this the hardest part?** Resolving cross-file references requires understanding: + +- How imports work in the language +- How to map a simple name to a fully qualified name +- Handling edge cases like aliases (`import helper as h`) + +### Step 8: Implement Web Workers for Performance + +**Why this matters:** Parsing code and building graphs can be CPU-intensive. Web Workers keep the UI responsive. + +**Key concepts:** + +- Web Workers run JavaScript in background threads +- They can't access the DOM directly +- We use Comlink to simplify communication + +**Implementation:** + +```typescript +// src/workers/ingestion.worker.ts +import { expose } from 'comlink'; +import { GraphPipeline } from '../core/ingestion/pipeline'; + +class IngestionWorker { + async processRepository( + projectRoot: string, + projectName: string, + filePaths: string[], + fileContents: Record + ) { + const pipeline = new GraphPipeline(projectRoot, projectName); + return pipeline.run(filePaths, new Map(Object.entries(fileContents))); + } +} + +expose(new IngestionWorker()); +``` + +**How this works:** + +1. The worker runs the heavy processing in a background thread +2. We expose methods via Comlink to call them from the main thread +3. The main thread can call these methods without blocking the UI + +**Why use Web Workers?** Without them, large repositories would freeze the browser tab while processing. + +## Phase 4: Graph Visualization + +### Step 9: Implement Graph Visualization Components + +**Why this matters:** A knowledge graph is useless if users can't see and interact with it. + +**Key concepts:** + +- Cytoscape.js is a powerful graph visualization library +- We need to convert our graph data to Cytoscape's format +- Users need controls to filter and navigate the graph + +**Implementation:** + +```tsx +import React, { useEffect, useRef } from 'react'; +import cytoscape from 'cytoscape'; +import dagre from 'cytoscape-dagre'; +import { KnowledgeGraph } from '@/core/graph/types'; + +cytoscape.use(dagre); + +interface GraphVisualizationProps { + graph: KnowledgeGraph; + onNodeClick?: (nodeId: string) => void; + filter?: (node: any) => boolean; +} + +export const GraphVisualization: React.FC = ({ + graph, + onNodeClick, + filter +}) => { + const containerRef = useRef(null); + const cyRef = useRef(null); + + useEffect(() => { + if (!containerRef.current) return; + + // Clean up previous instance + if (cyRef.current) { + cyRef.current.destroy(); + } + + // Convert our graph to Cytoscape format + const cyElements = convertToCytoscapeElements(graph, filter); + + const cy = cytoscape({ + container: containerRef.current, + elements: cyElements, + style: [ + { + selector: 'node', + style: { + 'label': 'data(label)', + 'width': 'mapData(size, 0, 100, 20, 80)', + 'height': 'mapData(size, 0, 100, 20, 80)', + 'background-color': 'data(color)', + 'text-valign': 'center', + 'text-halign': 'center', + 'font-size': '8px' + } + }, + { + selector: 'edge', + style: { + 'width': 2, + 'line-color': '#ccc', + 'target-arrow-color': '#ccc', + 'target-arrow-shape': 'triangle' + } + } + ], + layout: { + name: 'dagre', + rankDir: 'TB', + padding: 20 + } + }); + + // Add interactions + cy.on('tap', 'node', (event) => { + const node = event.target; + const nodeId = node.data('id'); + if (onNodeClick) { + onNodeClick(nodeId); + } + }); + + cyRef.current = cy; + + return () => { + if (cyRef.current) { + cyRef.current.destroy(); + cyRef.current = null; + } + }; + }, [graph, filter]); + + return ( +
+ ); +}; + +function convertToCytoscapeElements( + graph: KnowledgeGraph, + filter?: (node: any) => boolean +) { + const elements: any[] = []; + + // Add nodes + for (const node of graph.nodes) { + if (filter && !filter(node)) continue; + elements.push({ + data: { + id: node.id, + label: getNodeLabel(node), + type: node.label, + color: getNodeColor(node.label), + size: getNodeSize(node) + } + }); + } + + // Add edges + for (const rel of graph.relationships) { + elements.push({ + data: { + id: rel.id, + source: rel.source, + target: rel.target, + label: rel.type + } + }); + } + + return elements; +} +``` + +**How this works:** + +1. Converts our graph data to Cytoscape's format +2. Sets up visual styles based on node type +3. Applies a hierarchical layout (dagre) +4. Adds interaction handlers for node clicks + +**Why use Cytoscape.js?** It's specifically designed for graph visualization with: + +- Multiple layout algorithms +- Good performance for medium-sized graphs +- Extensive customization options + +### Step 10: Create Source Code Viewer + +**Why this matters:** Seeing the graph isn't enough - users need to see the actual code behind the nodes. + +**Implementation:** + +```tsx +import React, { useState, useEffect } from 'react'; +import { KnowledgeGraph } from '@/core/graph/types'; + +interface SourceViewerProps { + graph: KnowledgeGraph; + selectedNodeId: string | null; +} + +export const SourceViewer: React.FC = ({ graph, selectedNodeId }) => { + const [sourceCode, setSourceCode] = useState(''); + const [fileName, setFileName] = useState(''); + const [lineNumber, setLineNumber] = useState(null); + + useEffect(() => { + if (!selectedNodeId) { + setSourceCode(''); + setFileName(''); + setLineNumber(null); + return; + } + + // Find the node in the graph + const node = graph.nodes.find(n => n.id === selectedNodeId); + if (!node) return; + + // For functions, get the source code + if (node.label === 'Function' || node.label === 'Method') { + const filePath = node.properties.path; + const startLine = node.properties.start_line; + + // In a real implementation, you'd have the source code available + setFileName(filePath); + setLineNumber(startLine); + setSourceCode(`# Source code for ${node.properties.qualified_name} +# Line ${startLine} and following...`); + } + }, [graph, selectedNodeId]); + + if (!selectedNodeId || !sourceCode) { + return ( +
+

Select a node to view source code

+
+ ); + } + + return ( +
+
+ {fileName} + {lineNumber && ( + Line {lineNumber} + )} +
+
+
{sourceCode}
+
+
+ ); +}; +``` + +**How this works:** + +1. When a node is selected, it finds the corresponding code element +2. It displays the source code with line numbers +3. It highlights the relevant part of the code + +**Why is this important?** It bridges the gap between the abstract graph and the concrete code, helping users understand what they're seeing. + +## Phase 5: RAG Chat Interface + +### Step 11: Implement LLM Service + +**Why this matters:** The chat interface needs to connect to LLMs (Large Language Models) to translate natural language to graph queries. + +**Key concepts:** + +- We'll support multiple LLM providers (OpenAI, Anthropic, Gemini) +- Users provide their own API keys (privacy-focused) +- We need a consistent interface for different providers + +**Implementation:** + +```typescript +import { ChatOpenAI } from 'langchain/chat_models/openai'; +import { ChatAnthropic } from 'langchain/chat_models/anthropic'; +import { ChatGoogleGenerativeAI } from '@langchain/google-genai'; + +export type LLMProvider = 'openai' | 'anthropic' | 'gemini'; + +export interface LLMConfig { + provider: LLMProvider; + apiKey: string; + model?: string; +} + +export class LLMService { + private config: LLMConfig; + + constructor(config: LLMConfig) { + this.config = config; + } + + getChatModel() { + switch (this.config.provider) { + case 'openai': + return new ChatOpenAI({ + apiKey: this.config.apiKey, + modelName: this.config.model || 'gpt-4-turbo', + temperature: 0 + }); + case 'anthropic': + return new ChatAnthropic({ + apiKey: this.config.apiKey, + modelName: this.config.model || 'claude-3-sonnet-20240229', + temperature: 0 + }); + case 'gemini': + return new ChatGoogleGenerativeAI({ + apiKey: this.config.apiKey, + modelName: this.config.model || 'gemini-1.5-pro-latest', + temperature: 0 + }); + default: + throw new Error(`Unsupported LLM provider: ${this.config.provider}`); + } + } +} +``` + +**How this works:** + +1. The service takes an LLM configuration (provider, API key, model) +2. It returns a consistent chat model interface regardless of provider +3. It handles provider-specific initialization + +**Why support multiple providers?** Different users have different preferences and API key availability. + +### Step 12: Implement Cypher Generator + +**Why this matters:** The core of the RAG system - translating natural language questions to graph queries. + +**Key concepts:** + +- We use a system prompt to instruct the LLM +- The prompt includes our graph schema +- We clean the response to get a valid Cypher query + +**Implementation:** + +```typescript +import { BaseChatModel } from 'langchain/chat_models/base'; +import { CYPHER_SYSTEM_PROMPT } from '../prompts/cypher'; + +export class CypherGenerator { + private llm: BaseChatModel; + + constructor(llm: BaseChatModel) { + this.llm = llm; + } + + async generate(naturalLanguageQuery: string): Promise { + const response = await this.llm.call([ + { role: 'system', content: CYPHER_SYSTEM_PROMPT }, + { role: 'user', content: naturalLanguageQuery } + ]); + + return this.cleanResponse(response.content); + } + + private cleanResponse(response: string): string { + // Remove markdown code blocks + let cleaned = response.replace(/```cypher/g, '').replace(/```/g, ''); + // Ensure it ends with a semicolon + if (!cleaned.trim().endsWith(';')) { + cleaned = cleaned.trim() + ';'; + } + return cleaned; + } +} +``` + +**How this works:** + +1. It sends the natural language query with a system prompt to the LLM +2. The system prompt teaches the LLM about our graph structure +3. It cleans the response to extract a valid Cypher query + +**Why is the system prompt important?** It provides the LLM with the context it needs to generate correct queries. Without it, the LLM wouldn't know about our graph schema. + +### Step 13: Implement RAG Orchestrator + +**Why this matters:** This is the "brain" of the system that coordinates the query process. + +**Key concepts:** + +- It follows a ReAct (Reason + Act) pattern +- It plans steps, uses tools, observes results, and responds +- It prevents hallucination by sticking to tool results + +**Implementation:** + +```typescript +import { BaseChatModel } from 'langchain/chat_models/base'; +import { RAG_ORCHESTRATOR_SYSTEM_PROMPT } from '../prompts/rag-orchestrator'; + +export class RAGOrchestrator { + private llm: BaseChatModel; + + constructor(llm: BaseChatModel) { + this.llm = llm; + } + + async query( + userQuery: string, + queryGraph: (cypher: string) => Promise, + retrieveCode: (nodeId: string) => Promise + ) { + // Start with the system prompt + let conversation = [ + { role: 'system', content: RAG_ORCHESTRATOR_SYSTEM_PROMPT } + ]; + + // Add the user's question + conversation.push({ role: 'user', content: userQuery }); + + // Simple ReAct loop + for (let i = 0; i < 5; i++) { // Max 5 steps + const response = await this.llm.call(conversation); + const responseContent = response.content; + + // Check if the response contains a tool call + if (responseContent.includes('Action: query_graph')) { + const match = responseContent.match(/Action Input: (.*)/); + if (match) { + const cypherQuery = match[1].trim(); + + // Execute the query + const queryResults = await queryGraph(cypherQuery); + + // Add the observation to the conversation + conversation.push({ + role: 'assistant', + content: responseContent + }); + + conversation.push({ + role: 'system', + content: `Observation: ${JSON.stringify(queryResults)}` + }); + + // If we have results, we might be done + if (queryResults.length > 0) { + break; + } + } + } + else if (responseContent.includes('Action: retrieve_code')) { + // Similar handling for code retrieval + } + else { + // This appears to be the final answer + return responseContent; + } + } + + // If we got here without a final answer, generate one + conversation.push({ + role: 'user', + content: 'Please provide your final answer based on the information gathered.' + }); + + const finalResponse = await this.llm.call(conversation); + return finalResponse.content; + } +} +``` + +**How this works:** + +1. It starts with a system prompt that defines the rules +2. It sends the user's query to the LLM +3. The LLM responds with either: + - A tool call (query_graph or retrieve_code) + - A final answer +4. If it's a tool call, it executes the tool and adds the result to the conversation +5. It repeats until it gets a final answer or hits the step limit + +**Why the step limit?** To prevent infinite loops if the LLM gets stuck. + +## Phase 6: Main Application Integration + +### Step 14: Create Main Application Component + +**Why this matters:** This brings all the pieces together into a cohesive UI. + +**Implementation:** + +```tsx +import React, { useState, useRef } from 'react'; +import { GraphVisualization } from '@/ui/components/graph/Visualization'; +import { GraphControls } from '@/ui/components/graph/Controls'; +import { SourceViewer } from '@/ui/components/graph/SourceViewer'; +import { ChatInterface } from '@/ui/components/chat/ChatInterface'; +import { KnowledgeGraph } from '@/core/graph/types'; +import { GitHubService } from '@/services/github'; +import { ZipService } from '@/services/zip'; +import { ingestionWorkerApi } from '@/lib/workerUtils'; + +export const HomePage = () => { + const [graph, setGraph] = useState(null); + const [selectedNodeId, setSelectedNodeId] = useState(null); + const [isLoading, setIsLoading] = useState(false); + const [error, setError] = useState(null); + const [repoUrl, setRepoUrl] = useState(''); + const fileInputRef = useRef(null); + + const handleRepoSubmit = async (e: React.FormEvent) => { + e.preventDefault(); + if (!repoUrl.trim() || isLoading) return; + + setIsLoading(true); + setError(null); + + try { + // Parse the GitHub URL + const urlMatch = repoUrl.match(/github\.com\/([^/]+)\/([^/]+)/); + if (!urlMatch) { + throw new Error('Invalid GitHub repository URL'); + } + + const owner = urlMatch[1]; + const repo = urlMatch[2].replace(/\.git$/, ''); + + // Fetch repository contents + const githubService = new GitHubService(); + const contents = await githubService.getRepoContents(owner, repo); + + // Filter for Python files + const pythonFiles = contents + .filter((item: any) => item.type === 'file' && item.name.endsWith('.py')) + .map((item: any) => item.path); + + // Fetch file contents + const fileContents: Record = {}; + for (const filePath of pythonFiles) { + fileContents[filePath] = await githubService.getFileContent(owner, repo, filePath); + } + + // Process the repository + const projectName = `${owner}/${repo}`; + const processedGraph = await ingestionWorkerApi.processRepository( + repoUrl, + projectName, + pythonFiles, + fileContents + ); + + setGraph(processedGraph); + } catch (err) { + setError(err instanceof Error ? err.message : 'Failed to process repository'); + console.error('Processing error:', err); + } finally { + setIsLoading(false); + } + }; + + const handleQuery = async (query: string): Promise => { + if (!graph) { + throw new Error('No graph available'); + } + + // In a real implementation, this would use the RAG orchestrator + return `I found information related to "${query}" in the codebase.`; + }; + + return ( +
+ {/* Header with repository input */} +
+
+

GitNexus

+ +
+
+
+ setRepoUrl(e.target.value)} + placeholder="https://github.com/owner/repo.git" + className="flex-1 px-3 py-2 border border-gray-300 rounded-md focus:outline-none focus:ring-1 focus:ring-blue-500 focus:border-blue-500" + /> + +
+
+ +
+ or + + +
+
+
+
+ +
+ {/* Graph Visualization Pane */} +
+
+ {graph ? ( + <> + {}} + onLayoutChange={() => {}} + /> +
+ +
+ + ) : ( +
+ {isLoading ? ( +
+
+

Processing repository...

+
+ ) : ( +

Enter a repository URL or upload a ZIP to get started

+ )} +
+ )} +
+
+ + {/* Right Panel */} +
+ {/* Source Viewer */} +
+
+

Source Code

+
+ {graph ? ( + + ) : ( +
+

Select a node to view source

+
+ )} +
+ + {/* Chat Interface */} +
+
+

Ask About Code

+
+ {graph ? ( + + ) : ( +
+

Process a repository to ask questions about the code

+
+ )} +
+
+
+
+ ); +}; +``` + +**How this works:** + +1. The header has inputs for GitHub URLs and ZIP uploads +2. The main area has two panes: + - Left: Graph visualization + - Right: Source viewer and chat interface +3. When a repository is processed, the graph is displayed +4. Users can click nodes to see source code and ask questions + +**Why this layout?** It provides a cohesive experience where users can: + +- See the big picture (graph) +- Drill down to specific code (source viewer) +- Ask questions about what they're seeing (chat) + +## Final Steps: Testing and Optimization + +### Step 15: Add Error Boundaries + +**Why this matters:** Inevitably, something will go wrong. We want to handle errors gracefully. + +**Implementation:** + +```tsx +import React, { Component, ErrorInfo, ReactNode } from 'react'; + +interface Props { + children: ReactNode; +} + +interface State { + hasError: boolean; + error: Error | null; +} + +export class ErrorBoundary extends Component { + public state: State = { + hasError: false, + error: null + }; + + public static getDerivedStateFromError(error: Error): State { + return { hasError: true, error }; + } + + public componentDidCatch(error: Error, errorInfo: ErrorInfo) { + console.error("Uncaught error:", error, errorInfo); + } + + public render() { + if (this.state.hasError) { + return ( +
+

Something went wrong

+

{this.state.error?.message}

+ +
+ ); + } + + return this.props.children; + } +} +``` + +**How this works:** + +- It catches JavaScript errors in child components +- It displays a friendly error message instead of a blank screen +- It allows users to try again without losing their work + +**Why use error boundaries?** They prevent a single error from breaking the entire application. + +### Step 16: Implement Performance Optimizations + +**Why this matters:** Large repositories can be slow to process. We need to keep the UI responsive. + +**Key optimizations:** + +1. **Web Workers**: Already implemented for graph processing +2. **Progress Indicators**: Show users what's happening +3. **File Filtering**: Only process relevant files +4. **Lazy Loading**: Load components as needed + +**Implementation:** + +```tsx +// Add to your GitHub processing function +const MAX_FILES = 500; // Limit for free tier +if (pythonFiles.length > MAX_FILES) { + // Offer to filter by directory or file pattern + const shouldFilter = window.confirm( + `Repository has ${pythonFiles.length} Python files (max ${MAX_FILES}). ` + + `Would you like to filter by directory or file pattern?` + ); + if (shouldFilter) { + const filterPattern = prompt( + "Enter a directory path or file pattern to filter (e.g., 'src/', '*.py')", + "src/" + ); + if (filterPattern) { + const filteredFiles = pythonFiles.filter(file => + file.includes(filterPattern) || file.endsWith(filterPattern) + ); + pythonFiles = filteredFiles; + } + } +} +``` + +**Why limit file processing?** Processing too many files can: + +- Freeze the browser tab +- Exceed GitHub API rate limits +- Use excessive memory + +### Step 17: Add Export Functionality + +**Why this matters:** Users might want to save or share their generated graphs. + +**Implementation:** + +```tsx +export function exportGraphToJson(graph: KnowledgeGraph): string { + return JSON.stringify(graph, null, 2); +} + +export function downloadGraph(graph: KnowledgeGraph, filename: string = 'gitnexus-graph.json') { + const json = exportGraphToJson(graph); + const blob = new Blob([json], { type: 'application/json' }); + const url = URL.createObjectURL(blob); + const a = document.createElement('a'); + a.href = url; + a.download = filename; + document.body.appendChild(a); + a.click(); + document.body.removeChild(a); + URL.revokeObjectURL(url); +} +``` + +**How this works:** + +1. Converts the graph to JSON +2. Creates a downloadable file +3. Triggers a download + +**Why include export?** It allows users to: + +- Save their work for later +- Share graphs with teammates +- Use the data in other tools + +## Conclusion + +This implementation guide has walked you through building a complete edge-based code knowledge graph generator using Deno. By following these steps, you'll create a privacy-focused tool that runs entirely in the user's browser. + +**Key advantages of this approach:** + +- **Zero server costs**: All processing happens in the user's browser +- **Strong privacy**: Code never leaves the user's machine +- **Modular architecture**: Easy to add more languages later +- **Clear separation of concerns**: Makes the codebase maintainable +- **Deno compatibility**: Modern runtime with built-in TypeScript support + +Remember to start small (Python support only) and iterate, adding more features and language support as you validate the core functionality. The most important part is getting the graph construction pipeline working correctly - everything else builds on that foundation. + +================ +File: src/ai/cypher-generator.ts +================ +import { HumanMessage, SystemMessage } from '@langchain/core/messages'; +import type { LLMService, LLMConfig } from './llm-service.ts'; +import type { KnowledgeGraph } from '../core/graph/types.ts'; + +export interface CypherQuery { + cypher: string; + explanation: string; + confidence: number; + warnings?: string[]; +} + +export interface CypherGenerationOptions { + maxRetries?: number; + includeExamples?: boolean; + strictMode?: boolean; +} + +export class CypherGenerator { + private llmService: LLMService; + private graphSchema: string = ''; + + // Common Cypher patterns and examples + private static readonly CYPHER_EXAMPLES = [ + { + question: "What functions are in the main.py file?", + cypher: "MATCH (f:File {name: 'main.py'})-[:CONTAINS]->(func:Function) RETURN func.name" + }, + { + question: "Which functions call the authenticate function?", + cypher: "MATCH (caller)-[:CALLS]->(target:Function {name: 'authenticate'}) RETURN caller.name" + }, + { + question: "Show me all classes in the project", + cypher: "MATCH (c:Class) RETURN c.name, c.filePath" + }, + { + question: "What modules import numpy?", + cypher: "MATCH (m:Module)-[:IMPORTS]->(lib) WHERE lib.name CONTAINS 'numpy' RETURN m.name" + } + ]; + + constructor(llmService: LLMService) { + this.llmService = llmService; + } + + /** + * Update the graph schema for better query generation + */ + public updateSchema(graph: KnowledgeGraph): void { + this.graphSchema = this.generateSchemaDescription(graph); + } + + /** + * Generate a Cypher query from natural language + */ + public async generateQuery( + question: string, + llmConfig: LLMConfig, + options: CypherGenerationOptions = {} + ): Promise { + const { maxRetries = 2, includeExamples = true, strictMode = false } = options; + + let lastError: string | null = null; + + for (let attempt = 0; attempt <= maxRetries; attempt++) { + try { + const systemPrompt = this.buildSystemPrompt(includeExamples, strictMode, lastError); + const userPrompt = this.buildUserPrompt(question); + + const messages = [ + new SystemMessage(systemPrompt), + new HumanMessage(userPrompt) + ]; + + const response = await this.llmService.chat(llmConfig, messages); + const result = this.parseResponse(response.content); + + // Validate the generated query + const validation = this.validateQuery(result.cypher); + if (!validation.isValid) { + lastError = validation.error!; + if (attempt < maxRetries) { + console.warn(`Query validation failed (attempt ${attempt + 1}): ${validation.error}`); + continue; + } + } + + return { + ...result, + warnings: validation.warnings + }; + + } catch (error) { + lastError = error instanceof Error ? error.message : 'Unknown error'; + if (attempt < maxRetries) { + console.warn(`Query generation failed (attempt ${attempt + 1}): ${lastError}`); + continue; + } + + throw new Error(`Failed to generate Cypher query after ${maxRetries + 1} attempts: ${lastError}`); + } + } + + throw new Error('Unexpected error in query generation'); + } + + /** + * Build the system prompt with schema and examples + */ + private buildSystemPrompt(includeExamples: boolean, strictMode: boolean, lastError?: string | null): string { + let prompt = `You are a Cypher query expert for a code knowledge graph. Your task is to convert natural language questions into valid Cypher queries. + +GRAPH SCHEMA: +${this.graphSchema} + +NODE TYPES: +- Project: Root project node +- Folder: Directory containers +- File: Source code files +- Module: Python modules (.py files) +- Class: Class definitions +- Function: Function definitions +- Method: Class method definitions +- Variable: Variable declarations + +RELATIONSHIP TYPES: +- CONTAINS: Hierarchical containment (Project->Folder, Folder->File, File->Function, etc.) +- CALLS: Function/method calls between code entities +- INHERITS: Class inheritance relationships +- IMPORTS: Module import relationships +- OVERRIDES: Method override relationships + +IMPORTANT RULES: +1. Always use MATCH patterns to find nodes +2. Use WHERE clauses for filtering by properties +3. Node properties include: name, filePath, startLine, endLine, type +4. Return meaningful information, not just node IDs +5. Use case-insensitive matching when appropriate (e.g., WHERE toLower(n.name) CONTAINS toLower('search')) +6. Prefer specific node types over generic matches +7. Always return results in a readable format`; + + if (includeExamples) { + prompt += `\n\nEXAMPLES:`; + for (const example of CypherGenerator.CYPHER_EXAMPLES) { + prompt += `\nQ: "${example.question}"\nA: ${example.cypher}\n`; + } + } + + if (strictMode) { + prompt += `\n\nSTRICT MODE: Only generate queries that exactly match the schema. Do not make assumptions about node properties that aren't explicitly defined.`; + } + + if (lastError) { + prompt += `\n\nPREVIOUS ERROR: The last query attempt failed with: "${lastError}". Please fix this issue in your new query.`; + } + + prompt += `\n\nRESPONSE FORMAT: +Provide your response in this exact JSON format: +{ + "cypher": "your cypher query here", + "explanation": "brief explanation of what the query does", + "confidence": 0.85 +} + +The confidence should be a number between 0 and 1 indicating how confident you are in the query.`; + + return prompt; + } + + /** + * Build the user prompt with the question + */ + private buildUserPrompt(question: string): string { + return `Please convert this question to a Cypher query: "${question}"`; + } + + /** + * Parse the LLM response to extract Cypher query + */ + private parseResponse(response: string): { cypher: string; explanation: string; confidence: number } { + try { + // Try to extract JSON from the response + const jsonMatch = response.match(/\{[\s\S]*\}/); + if (jsonMatch) { + const parsed = JSON.parse(jsonMatch[0]); + return { + cypher: parsed.cypher || '', + explanation: parsed.explanation || '', + confidence: parsed.confidence || 0.5 + }; + } + + // Fallback: try to extract Cypher from code blocks + const cypherMatch = response.match(/```(?:cypher)?\s*(.*?)\s*```/s); + if (cypherMatch) { + return { + cypher: cypherMatch[1].trim(), + explanation: 'Generated Cypher query', + confidence: 0.7 + }; + } + + // Last resort: use the entire response as cypher + return { + cypher: response.trim(), + explanation: 'Raw LLM response', + confidence: 0.3 + }; + + } catch (error) { + throw new Error(`Failed to parse LLM response: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + /** + * Validate the generated Cypher query + */ + private validateQuery(cypher: string): { isValid: boolean; error?: string; warnings?: string[] } { + const warnings: string[] = []; + + if (!cypher || cypher.trim().length === 0) { + return { isValid: false, error: 'Empty query generated' }; + } + + // Basic syntax checks + const upperCypher = cypher.toUpperCase(); + + // Must have MATCH or CREATE or other valid starting keywords + if (!upperCypher.match(/^\s*(MATCH|CREATE|MERGE|WITH|RETURN|CALL|SHOW)/)) { + return { isValid: false, error: 'Query must start with a valid Cypher keyword (MATCH, CREATE, etc.)' }; + } + + // Check for balanced parentheses + const openParens = (cypher.match(/\(/g) || []).length; + const closeParens = (cypher.match(/\)/g) || []).length; + if (openParens !== closeParens) { + return { isValid: false, error: 'Unbalanced parentheses in query' }; + } + + // Check for balanced brackets + const openBrackets = (cypher.match(/\[/g) || []).length; + const closeBrackets = (cypher.match(/\]/g) || []).length; + if (openBrackets !== closeBrackets) { + return { isValid: false, error: 'Unbalanced brackets in query' }; + } + + // Check for balanced braces + const openBraces = (cypher.match(/\{/g) || []).length; + const closeBraces = (cypher.match(/\}/g) || []).length; + if (openBraces !== closeBraces) { + return { isValid: false, error: 'Unbalanced braces in query' }; + } + + // Warn about potentially expensive operations + if (upperCypher.includes('MATCH ()') || upperCypher.includes('MATCH (*)')) { + warnings.push('Query matches all nodes - this could be expensive'); + } + + if (!upperCypher.includes('RETURN') && !upperCypher.includes('DELETE') && !upperCypher.includes('SET')) { + warnings.push('Query does not return results'); + } + + return { isValid: true, warnings }; + } + + /** + * Generate a schema description from the knowledge graph + */ + private generateSchemaDescription(graph: KnowledgeGraph): string { + const nodeTypes = new Set(); + const relationshipTypes = new Set(); + const nodeProperties = new Map>(); + + // Analyze nodes + graph.nodes.forEach(node => { + nodeTypes.add(node.label); + + if (!nodeProperties.has(node.label)) { + nodeProperties.set(node.label, new Set()); + } + + Object.keys(node.properties).forEach(prop => { + nodeProperties.get(node.label)!.add(prop); + }); + }); + + // Analyze relationships + graph.relationships.forEach(rel => { + relationshipTypes.add(rel.type); + }); + + let schema = `NODES (${graph.nodes.length} total):\n`; + for (const nodeType of Array.from(nodeTypes).sort()) { + const props = nodeProperties.get(nodeType); + const propList = props ? Array.from(props).sort().join(', ') : 'none'; + const count = graph.nodes.filter(n => n.label === nodeType).length; + schema += `- ${nodeType} (${count}): ${propList}\n`; + } + + schema += `\nRELATIONSHIPS (${graph.relationships.length} total):\n`; + for (const relType of Array.from(relationshipTypes).sort()) { + const count = graph.relationships.filter(r => r.type === relType).length; + schema += `- ${relType} (${count})\n`; + } + + return schema; + } + + /** + * Get the current schema description + */ + public getSchema(): string { + return this.graphSchema; + } + + /** + * Clean and format a Cypher query + */ + public cleanQuery(cypher: string): string { + return cypher + .trim() + .replace(/\s+/g, ' ') + .replace(/\s*([(),\[\]{}])\s*/g, '$1') + .replace(/\s*([=<>!]+)\s*/g, ' $1 ') + .replace(/\s+/g, ' ') + .trim(); + } +} + +================ +File: src/ai/index.ts +================ +export { LLMService, type LLMProvider, type LLMConfig, type ChatResponse } from './llm-service.ts'; +export { CypherGenerator, type CypherQuery, type CypherGenerationOptions } from './cypher-generator.ts'; +export { + RAGOrchestrator, + type RAGContext, + type RAGResponse, + type RAGOptions, + type ToolResult, + type ReasoningStep +} from './orchestrator.ts'; +export { + LangChainRAGOrchestrator, + type LangChainRAGContext, + type LangChainRAGResponse, + type LangChainRAGOptions +} from './langchain-orchestrator.ts'; + +================ +File: src/ai/langchain-orchestrator.ts +================ +import { createReactAgent } from '@langchain/langgraph/prebuilt'; +import { MemorySaver } from '@langchain/langgraph'; +import { DynamicStructuredTool } from '@langchain/core/tools'; +import { z } from 'zod'; +import { SystemMessage } from '@langchain/core/messages'; +import type { LLMService, LLMConfig } from './llm-service.ts'; +import type { CypherGenerator } from './cypher-generator.ts'; +import type { KnowledgeGraph } from '../core/graph/types.ts'; + +export interface LangChainRAGContext { + graph: KnowledgeGraph; + fileContents: Map; +} + +export interface ToolCall { + tool: string; + input: Record; + output: string; +} + +export interface LangChainRAGResponse { + answer: string; + sources: string[]; + confidence: number; + toolCalls: ToolCall[]; +} + +export interface LangChainRAGOptions { + maxIterations?: number; + temperature?: number; + enableMemory?: boolean; + threadId?: string; +} + +type AgentType = ReturnType; +type MemoryType = InstanceType; + +export class LangChainRAGOrchestrator { + private llmService: LLMService; + private cypherGenerator: CypherGenerator; + private context: LangChainRAGContext | null = null; + private agent: AgentType | null = null; + private memory: MemoryType | null = null; + + constructor(llmService: LLMService, cypherGenerator: CypherGenerator) { + this.llmService = llmService; + this.cypherGenerator = cypherGenerator; + this.memory = new MemorySaver(); + } + + /** + * Set the current context and initialize the agent + */ + public async setContext(context: LangChainRAGContext, llmConfig: LLMConfig): Promise { + this.context = context; + this.cypherGenerator.updateSchema(context.graph); + + // Create LangChain-compliant tools + const tools = this.createTools(llmConfig); + + // Get the LLM from our service + const llm = this.llmService.getChatModel(llmConfig); + + // Create system message for ReAct behavior + const systemMessage = this.buildSystemMessage(); + + // Create the ReAct agent using LangGraph + this.agent = createReactAgent({ + llm: llm, + tools: tools, + checkpointSaver: this.memory || undefined, + messageModifier: systemMessage + }); + } + + /** + * Answer a question using the LangChain ReAct agent + */ + public async answerQuestion( + question: string, + options: LangChainRAGOptions = {} + ): Promise { + if (!this.agent || !this.context) { + throw new Error('Agent not initialized. Call setContext() first.'); + } + + const { + maxIterations = 10, + enableMemory = false, + threadId = 'default' + } = options; + + try { + const config = enableMemory + ? { + configurable: { thread_id: threadId }, + recursionLimit: maxIterations + } + : { recursionLimit: maxIterations }; + + // Stream the agent execution + const stream = await this.agent.stream( + { messages: [{ role: "user", content: question }] }, + config + ); + + let finalAnswer = ''; + const toolCalls: ToolCall[] = []; + const sources: string[] = []; + + // Process the stream + for await (const chunk of stream) { + if (chunk.agent) { + finalAnswer = chunk.agent.messages[chunk.agent.messages.length - 1].content; + } + + if (chunk.tools) { + const toolMessage = chunk.tools.messages[chunk.tools.messages.length - 1]; + if (toolMessage.tool_calls) { + toolMessage.tool_calls.forEach((toolCall: { name: string; args: Record }) => { + toolCalls.push({ + tool: toolCall.name, + input: toolCall.args, + output: toolMessage.content || '' + }); + }); + } + } + } + + // Calculate confidence based on successful tool usage + const confidence = this.calculateConfidence(toolCalls); + + return { + answer: finalAnswer || 'I was unable to find a complete answer to your question.', + sources: Array.from(new Set(sources)), + confidence, + toolCalls + }; + + } catch (error) { + throw new Error(`LangChain RAG orchestration failed: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + /** + * Create LangChain-compliant tools + */ + private createTools(llmConfig: LLMConfig) { + const queryGraphTool = new DynamicStructuredTool({ + name: "query_graph", + description: "Query the code knowledge graph using natural language. Use this to find information about code structure, relationships, functions, classes, etc.", + schema: z.object({ + question: z.string().describe("Natural language question about the codebase") + }), + func: async (input: { question: string }) => { + try { + const cypherQuery = await this.cypherGenerator.generateQuery(input.question, llmConfig); + const mockResults = this.simulateGraphQuery(cypherQuery.cypher); + + return `Query: ${cypherQuery.cypher}\n\nResults:\n${mockResults}\n\nExplanation: ${cypherQuery.explanation}`; + } catch (error) { + return `Error generating graph query: ${error instanceof Error ? error.message : 'Unknown error'}`; + } + } + }); + + const getCodeContentTool = new DynamicStructuredTool({ + name: "get_code_content", + description: "Retrieve the source code content of a specific file. Use this when you need to examine the actual code implementation.", + schema: z.object({ + filePath: z.string().describe("The path to the file whose content you want to retrieve") + }), + func: async (input: { filePath: string }) => { + if (!this.context) { + return 'No context available'; + } + + const content = this.context.fileContents.get(input.filePath); + if (!content) { + // Try to find similar file paths + const similarFiles = Array.from(this.context.fileContents.keys()) + .filter(path => path.includes(input.filePath) || input.filePath.includes(path)) + .slice(0, 3); + + if (similarFiles.length > 0) { + return `File not found. Similar files available: ${similarFiles.join(', ')}`; + } + return 'File not found'; + } + + return `File: ${input.filePath}\n\n${content}`; + } + }); + + const searchFilesTool = new DynamicStructuredTool({ + name: "search_files", + description: "Search for files matching a pattern or containing specific text. Use this to find relevant files in the codebase.", + schema: z.object({ + pattern: z.string().describe("Search pattern or text to look for in file names or content") + }), + func: async (input: { pattern: string }) => { + if (!this.context) { + return 'No context available'; + } + + const matchingFiles: string[] = []; + const lowerPattern = input.pattern.toLowerCase(); + + // Search in file paths + for (const filePath of this.context.fileContents.keys()) { + if (filePath.toLowerCase().includes(lowerPattern)) { + matchingFiles.push(filePath); + } + } + + // Search in file contents + for (const [filePath, content] of this.context.fileContents.entries()) { + if (!matchingFiles.includes(filePath) && + content.toLowerCase().includes(lowerPattern)) { + matchingFiles.push(filePath); + } + } + + return matchingFiles.length > 0 + ? `Found ${matchingFiles.length} files:\n${matchingFiles.slice(0, 10).join('\n')}${matchingFiles.length > 10 ? '\n... and more' : ''}` + : 'No files found matching the pattern'; + } + }); + + return [queryGraphTool, getCodeContentTool, searchFilesTool]; + } + + /** + * Build system message for ReAct behavior + */ + private buildSystemMessage(): SystemMessage { + const systemPrompt = `You are an expert code analyst that helps users understand codebases by using available tools. + +You have access to the following tools: +1. query_graph: Query the code knowledge graph using natural language +2. get_code_content: Retrieve the source code content of a specific file +3. search_files: Search for files matching a pattern or containing specific text + +IMPORTANT GUIDELINES: +- Always think step by step about what information you need +- Use tools to gather factual information before providing answers +- Base your responses ONLY on information retrieved from tools +- If you cannot find information, say so explicitly +- Provide helpful, accurate answers about the codebase structure and functionality +- When referencing code, always cite the specific files you examined + +Your goal is to provide accurate, evidence-based answers about the codebase using the available tools.`; + + return new SystemMessage(systemPrompt); + } + + /** + * Simulate graph query execution (same as before) + */ + private simulateGraphQuery(cypher: string): string { + if (!this.context) return 'No context available'; + + const upperCypher = cypher.toUpperCase(); + + if (upperCypher.includes('FUNCTION') && upperCypher.includes('RETURN')) { + const functions = this.context.graph.nodes + .filter(n => n.label === 'Function') + .slice(0, 5) + .map(n => `${n.properties.name} (${n.properties.filePath})`) + .join('\n'); + return functions || 'No functions found'; + } + + if (upperCypher.includes('CLASS') && upperCypher.includes('RETURN')) { + const classes = this.context.graph.nodes + .filter(n => n.label === 'Class') + .slice(0, 5) + .map(n => `${n.properties.name} (${n.properties.filePath})`) + .join('\n'); + return classes || 'No classes found'; + } + + if (upperCypher.includes('FILE') && upperCypher.includes('RETURN')) { + const files = this.context.graph.nodes + .filter(n => n.label === 'File') + .slice(0, 5) + .map(n => n.properties.name) + .join('\n'); + return files || 'No files found'; + } + + return 'Query executed successfully (simulated)'; + } + + /** + * Calculate confidence based on tool usage + */ + private calculateConfidence(toolCalls: ToolCall[]): number { + if (toolCalls.length === 0) return 0.3; + + const successfulCalls = toolCalls.filter(call => + !call.output.includes('Error') && + !call.output.includes('not found') && + call.output.length > 10 + ).length; + + const baseConfidence = 0.5; + const toolBonus = (successfulCalls / toolCalls.length) * 0.4; + + return Math.min(0.95, Math.max(0.1, baseConfidence + toolBonus)); + } + + /** + * Get current context information + */ + public getContextInfo(): { nodeCount: number; fileCount: number; hasContext: boolean; hasAgent: boolean } { + return { + nodeCount: this.context?.graph.nodes.length || 0, + fileCount: this.context?.fileContents.size || 0, + hasContext: !!this.context, + hasAgent: !!this.agent + }; + } + + /** + * Clear memory for a specific thread + */ + public async clearMemory(threadId: string): Promise { + if (this.memory) { + // Note: MemorySaver doesn't have a direct clear method in the current API + // This would need to be implemented based on the specific memory backend + console.log(`Memory clearing not implemented for thread: ${threadId}`); + } + } +} + +================ +File: src/ai/llm-service.ts +================ +import { ChatOpenAI } from '@langchain/openai'; +import { AzureChatOpenAI } from '@langchain/openai'; +import { ChatAnthropic } from '@langchain/anthropic'; +import { ChatGoogleGenerativeAI } from '@langchain/google-genai'; +import type { BaseMessage } from '@langchain/core/messages'; +import type { BaseChatModel } from '@langchain/core/language_models/chat_models'; + +export type LLMProvider = 'openai' | 'azure-openai' | 'anthropic' | 'gemini'; + +export interface LLMConfig { + provider: LLMProvider; + apiKey: string; + model?: string; + temperature?: number; + maxTokens?: number; + maxRetries?: number; + // Azure OpenAI specific fields + azureOpenAIEndpoint?: string; + azureOpenAIApiVersion?: string; + azureOpenAIDeploymentName?: string; +} + +export interface ChatResponse { + content: string; + usage?: { + promptTokens: number; + completionTokens: number; + totalTokens: number; + }; + model?: string; + finishReason?: string; +} + +export class LLMService { + private models: Map = new Map(); + private defaultConfig: Partial = { + temperature: 0.1, + maxTokens: 4000, + maxRetries: 3, + azureOpenAIApiVersion: '2024-02-01' // Default Azure OpenAI API version + }; + + // Default models for each provider + private static readonly DEFAULT_MODELS: Record = { + openai: 'gpt-4o-mini', + 'azure-openai': 'gpt-4o-mini', + anthropic: 'claude-3-haiku-20240307', + gemini: 'gemini-2.5-flash' // Use the latest 2.5 Flash model as default (2025) + }; + + constructor() {} + + /** + * Initialize or get a chat model for the specified provider + */ + public getChatModel(config: LLMConfig): BaseChatModel { + const cacheKey = this.getCacheKey(config); + + if (this.models.has(cacheKey)) { + return this.models.get(cacheKey)!; + } + + const model = this.createChatModel(config); + this.models.set(cacheKey, model); + return model; + } + + /** + * Send a chat message and get a response + */ + public async chat( + config: LLMConfig, + messages: BaseMessage[], + options?: { stream?: boolean } + ): Promise { + try { + const model = this.getChatModel(config); + + if (options?.stream) { + // For streaming, we'd need to handle this differently + // For now, we'll just use regular invoke + console.warn('Streaming not implemented yet, falling back to regular invoke'); + } + + const response = await model.invoke(messages); + + return { + content: response.content as string, + usage: response.response_metadata?.usage ? { + promptTokens: response.response_metadata.usage.prompt_tokens || 0, + completionTokens: response.response_metadata.usage.completion_tokens || 0, + totalTokens: response.response_metadata.usage.total_tokens || 0 + } : undefined, + model: config.model || LLMService.DEFAULT_MODELS[config.provider], + finishReason: response.response_metadata?.finish_reason + }; + } catch (error) { + throw new Error(`LLM chat failed: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + /** + * Validate API key format for different providers + */ + public validateApiKey(provider: LLMProvider, apiKey: string): boolean { + if (!apiKey || apiKey.trim().length === 0) { + return false; + } + + switch (provider) { + case 'openai': + return apiKey.startsWith('sk-') && apiKey.length > 20; + case 'azure-openai': + // Azure OpenAI keys are typically 32 characters long and don't have a specific prefix + return apiKey.length >= 20; // More flexible validation for Azure keys + case 'anthropic': + return apiKey.startsWith('sk-ant-') && apiKey.length > 20; + case 'gemini': + return apiKey.length > 20; // Google API keys don't have a consistent prefix + default: + return false; + } + } + + /** + * Validate Azure OpenAI configuration + */ + public validateAzureOpenAIConfig(config: LLMConfig): { valid: boolean; error?: string } { + if (config.provider !== 'azure-openai') { + return { valid: false, error: 'Provider must be azure-openai' }; + } + + if (!config.azureOpenAIEndpoint) { + return { valid: false, error: 'Azure OpenAI endpoint is required' }; + } + + if (!config.azureOpenAIEndpoint.includes('openai.azure.com')) { + return { valid: false, error: 'Invalid Azure OpenAI endpoint format. Should contain "openai.azure.com"' }; + } + + if (!config.azureOpenAIDeploymentName) { + return { valid: false, error: 'Azure OpenAI deployment name is required' }; + } + + return { valid: true }; + } + + /** + * Get available models for a provider + */ + public getAvailableModels(provider: LLMProvider): string[] { + switch (provider) { + case 'openai': + return [ + 'gpt-4o', + 'gpt-4o-mini', + 'gpt-4-turbo', + 'gpt-4', + 'gpt-3.5-turbo' + ]; + case 'azure-openai': + return [ + 'gpt-4o', + 'gpt-4o-mini', + 'gpt-4.1-mini-v2', // Common deployment name + 'gpt-4-turbo', + 'gpt-4', + 'gpt-35-turbo', // Note: Azure uses gpt-35-turbo instead of gpt-3.5-turbo + 'gpt-4-32k' + ]; + case 'anthropic': + return [ + 'claude-3-5-sonnet-20241022', + 'claude-3-5-haiku-20241022', + 'claude-3-opus-20240229', + 'claude-3-sonnet-20240229', + 'claude-3-haiku-20240307' + ]; + case 'gemini': + return [ + 'gemini-2.5-flash', // Latest and fastest (2025) - NEW DEFAULT + 'gemini-2.5-pro', // Latest pro model (2025) - PREMIUM + 'gemini-1.5-flash', // Most stable and widely available + 'gemini-1.5-pro', // Stable pro model + 'gemini-1.0-pro', // Legacy but very stable + 'gemini-1.5-flash-8b', // Smaller, efficient version + 'gemini-2.0-flash', // Newer model (may not be available to all users) + 'gemini-2.0-flash-lite' // Lightweight version + ]; + default: + return []; + } + } + + /** + * Get provider display name + */ + public getProviderDisplayName(provider: LLMProvider): string { + switch (provider) { + case 'openai': + return 'OpenAI'; + case 'azure-openai': + return 'Azure OpenAI'; + case 'anthropic': + return 'Anthropic'; + case 'gemini': + return 'Google Gemini'; + default: + return provider; + } + } + + /** + * Test connection with the provider + */ + public async testConnection(config: LLMConfig): Promise<{ success: boolean; error?: string }> { + try { + // Validate Azure OpenAI config if needed + if (config.provider === 'azure-openai') { + const validation = this.validateAzureOpenAIConfig(config); + if (!validation.valid) { + return { + success: false, + error: validation.error + }; + } + } + + const model = this.createChatModel(config); + + // Send a simple test message + const testMessages = [{ + role: 'user' as const, + content: 'Hello, this is a connection test. Please respond with "OK".' + }]; + + const response = await model.invoke(testMessages); + + return { + success: true + }; + } catch (error) { + return { + success: false, + error: error instanceof Error ? error.message : 'Connection test failed' + }; + } + } + + /** + * Clear cached models (useful for updating API keys) + */ + public clearCache(): void { + this.models.clear(); + } + + /** + * Create a chat model instance based on the provider + */ + private createChatModel(config: LLMConfig): BaseChatModel { + const mergedConfig = { ...this.defaultConfig, ...config }; + const model = mergedConfig.model || LLMService.DEFAULT_MODELS[config.provider]; + + switch (config.provider) { + case 'openai': + return new ChatOpenAI({ + apiKey: config.apiKey, + model, + temperature: mergedConfig.temperature, + maxTokens: mergedConfig.maxTokens, + maxRetries: mergedConfig.maxRetries, + timeout: 30000 + }); + + case 'azure-openai': + return new AzureChatOpenAI({ + azureOpenAIApiKey: config.apiKey, + model: config.azureOpenAIDeploymentName, // Use deployment name as model + temperature: mergedConfig.temperature, + maxTokens: mergedConfig.maxTokens, + maxRetries: mergedConfig.maxRetries, + timeout: 30000, + azureOpenAIApiInstanceName: config.azureOpenAIEndpoint?.replace('https://', '').split('.')[0], + azureOpenAIApiVersion: config.azureOpenAIApiVersion, + azureOpenAIApiDeploymentName: config.azureOpenAIDeploymentName + }); + + case 'anthropic': + return new ChatAnthropic({ + apiKey: config.apiKey, + model, + temperature: mergedConfig.temperature, + maxTokens: mergedConfig.maxTokens, + maxRetries: mergedConfig.maxRetries, + timeout: 30000 + }); + + case 'gemini': + return new ChatGoogleGenerativeAI({ + apiKey: config.apiKey, + model, + temperature: mergedConfig.temperature, + maxOutputTokens: mergedConfig.maxTokens, + maxRetries: mergedConfig.maxRetries + }); + + default: + throw new Error(`Unsupported provider: ${config.provider}`); + } + } + + /** + * Generate cache key for model instances + */ + private getCacheKey(config: LLMConfig): string { + const model = config.model || LLMService.DEFAULT_MODELS[config.provider]; + let baseKey = `${config.provider}:${model}:${config.apiKey.slice(-8)}:${config.temperature}:${config.maxTokens}`; + + // Add Azure OpenAI specific fields to cache key + if (config.provider === 'azure-openai') { + baseKey += `:${config.azureOpenAIEndpoint}:${config.azureOpenAIDeploymentName}:${config.azureOpenAIApiVersion}`; + } + + return baseKey; + } +} + +================ +File: src/ai/orchestrator.ts +================ +import { HumanMessage, SystemMessage, AIMessage } from '@langchain/core/messages'; +import type { LLMService, LLMConfig } from './llm-service.ts'; +import type { CypherGenerator, CypherQuery } from './cypher-generator.ts'; +import type { KnowledgeGraph, GraphNode } from '../core/graph/types.ts'; + +export interface RAGContext { + graph: KnowledgeGraph; + fileContents: Map; +} + +export interface ToolResult { + toolName: string; + input: string; + output: string; + success: boolean; + error?: string; +} + +export interface ReasoningStep { + step: number; + thought: string; + action: string; + actionInput: string; + observation: string; + toolResult?: ToolResult; +} + +export interface RAGResponse { + answer: string; + reasoning: ReasoningStep[]; + cypherQueries: CypherQuery[]; + confidence: number; + sources: string[]; +} + +export interface RAGOptions { + maxReasoningSteps?: number; + includeReasoning?: boolean; + strictMode?: boolean; + temperature?: number; +} + +export class RAGOrchestrator { + private llmService: LLMService; + private cypherGenerator: CypherGenerator; + private context: RAGContext | null = null; + + constructor(llmService: LLMService, cypherGenerator: CypherGenerator) { + this.llmService = llmService; + this.cypherGenerator = cypherGenerator; + } + + /** + * Set the current context (graph and file contents) + */ + public setContext(context: RAGContext): void { + this.context = context; + this.cypherGenerator.updateSchema(context.graph); + } + + /** + * Answer a question using ReAct pattern + */ + public async answerQuestion( + question: string, + llmConfig: LLMConfig, + options: RAGOptions = {} + ): Promise { + if (!this.context) { + throw new Error('Context not set. Call setContext() first.'); + } + + const { + maxReasoningSteps = 5, + includeReasoning = true, + strictMode = false, + temperature = 0.1 + } = options; + + const reasoning: ReasoningStep[] = []; + const cypherQueries: CypherQuery[] = []; + const sources: string[] = []; + + // Enhanced LLM config for reasoning + const reasoningConfig: LLMConfig = { + ...llmConfig, + temperature: temperature + }; + + let currentStep = 1; + let finalAnswer = ''; + let confidence = 0.5; + + try { + // Initial system prompt for ReAct + const systemPrompt = this.buildReActSystemPrompt(strictMode); + const conversation = [new SystemMessage(systemPrompt)]; + + // Add the user question + conversation.push(new HumanMessage(`Question: ${question}`)); + + while (currentStep <= maxReasoningSteps) { + // Get reasoning from LLM + const response = await this.llmService.chat(reasoningConfig, conversation); + const reasoning_step = this.parseReasoningStep(response.content, currentStep); + + reasoning.push(reasoning_step); + + // Check if we have a final answer + if (reasoning_step.action.toLowerCase().includes('final_answer')) { + finalAnswer = reasoning_step.actionInput; + confidence = this.calculateConfidence(reasoning, cypherQueries); + break; + } + + // Execute the action + let toolResult: ToolResult | null = null; + + try { + if (reasoning_step.action.toLowerCase().includes('query_graph')) { + toolResult = await this.executeGraphQuery(reasoning_step.actionInput, reasoningConfig); + if (toolResult.success) { + cypherQueries.push(...this.extractCypherQueries(toolResult)); + } + } else if (reasoning_step.action.toLowerCase().includes('get_code')) { + toolResult = await this.getCodeContent(reasoning_step.actionInput); + if (toolResult.success) { + sources.push(...this.extractSources(toolResult)); + } + } else if (reasoning_step.action.toLowerCase().includes('search_files')) { + toolResult = await this.searchFiles(reasoning_step.actionInput); + } else { + toolResult = { + toolName: 'unknown', + input: reasoning_step.actionInput, + output: 'Unknown action type', + success: false, + error: `Unknown action: ${reasoning_step.action}` + }; + } + } catch (error) { + toolResult = { + toolName: reasoning_step.action, + input: reasoning_step.actionInput, + output: '', + success: false, + error: error instanceof Error ? error.message : 'Unknown error' + }; + } + + // Update the reasoning step with tool result + reasoning_step.observation = toolResult.output; + reasoning_step.toolResult = toolResult; + + // Add the tool result to conversation + conversation.push(new AIMessage(response.content)); + conversation.push(new HumanMessage(`Observation: ${toolResult.output}`)); + + currentStep++; + } + + // If we didn't get a final answer, generate one based on the reasoning + if (!finalAnswer && reasoning.length > 0) { + const summaryPrompt = this.buildSummaryPrompt(question, reasoning); + conversation.push(new HumanMessage(summaryPrompt)); + + const summaryResponse = await this.llmService.chat(reasoningConfig, conversation); + finalAnswer = summaryResponse.content; + confidence = Math.max(0.3, confidence - 0.2); // Lower confidence for incomplete reasoning + } + + return { + answer: finalAnswer || 'I was unable to find a complete answer to your question.', + reasoning: includeReasoning ? reasoning : [], + cypherQueries, + confidence, + sources: Array.from(new Set(sources)) // Remove duplicates + }; + + } catch (error) { + throw new Error(`RAG orchestration failed: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + /** + * Build the ReAct system prompt + */ + private buildReActSystemPrompt(strictMode: boolean): string { + const prompt = `You are an expert code analyst using a ReAct (Reasoning + Acting) approach to answer questions about a codebase. + +You have access to the following tools: +1. query_graph(question): Query the code knowledge graph using natural language +2. get_code(file_path): Retrieve the source code content of a specific file +3. search_files(pattern): Search for files matching a pattern or containing specific text + +IMPORTANT INSTRUCTIONS: +- Always think step by step using the format: Thought: [your reasoning] +- Then specify an action using: Action: [tool_name] +- Provide the input using: Action Input: [input for the tool] +- After receiving an observation, continue reasoning or provide a final answer +- Use Final Answer: [your answer] when you have sufficient information +- Base your answers ONLY on the information retrieved from tools +- Do not make assumptions or hallucinate information +- If you cannot find information, say so explicitly + +RESPONSE FORMAT: +Thought: [Your reasoning about what to do next] +Action: [query_graph, get_code, search_files, or Final Answer] +Action Input: [The input for the action] + +After receiving an Observation, continue with: +Thought: [Your analysis of the observation] +Action: [Next action or Final Answer] +Action Input: [Input for next action or your final answer] + +${strictMode ? '\nSTRICT MODE: Only use information explicitly found in the tools. Do not infer or assume anything.' : ''} + +Remember: Your goal is to provide accurate, evidence-based answers about the codebase.`; + + return prompt; + } + + /** + * Parse a reasoning step from LLM response + */ + private parseReasoningStep(response: string, stepNumber: number): ReasoningStep { + const thoughtMatch = response.match(/Thought:\s*(.*?)(?=\n|Action:|$)/s); + const actionMatch = response.match(/Action:\s*(.*?)(?=\n|Action Input:|$)/s); + const inputMatch = response.match(/Action Input:\s*(.*?)(?=\n|$)/s); + + return { + step: stepNumber, + thought: thoughtMatch?.[1]?.trim() || 'No thought provided', + action: actionMatch?.[1]?.trim() || 'unknown', + actionInput: inputMatch?.[1]?.trim() || '', + observation: '' // Will be filled after tool execution + }; + } + + /** + * Execute a graph query using the Cypher generator + */ + private async executeGraphQuery(question: string, llmConfig: LLMConfig): Promise { + try { + const cypherQuery = await this.cypherGenerator.generateQuery(question, llmConfig); + + // For now, we'll simulate query execution since we don't have a real graph database + // In a real implementation, this would execute the Cypher query against Neo4j or similar + const mockResults = this.simulateGraphQuery(cypherQuery.cypher); + + return { + toolName: 'query_graph', + input: question, + output: `Query: ${cypherQuery.cypher}\n\nResults:\n${mockResults}\n\nExplanation: ${cypherQuery.explanation}`, + success: true + }; + } catch (error) { + return { + toolName: 'query_graph', + input: question, + output: '', + success: false, + error: error instanceof Error ? error.message : 'Query execution failed' + }; + } + } + + /** + * Get code content from a file + */ + private async getCodeContent(filePath: string): Promise { + if (!this.context) { + return { + toolName: 'get_code', + input: filePath, + output: '', + success: false, + error: 'No context available' + }; + } + + const content = this.context.fileContents.get(filePath); + if (!content) { + // Try to find similar file paths + const similarFiles = Array.from(this.context.fileContents.keys()) + .filter(path => path.includes(filePath) || filePath.includes(path)) + .slice(0, 3); + + if (similarFiles.length > 0) { + return { + toolName: 'get_code', + input: filePath, + output: `File not found. Similar files available: ${similarFiles.join(', ')}`, + success: false, + error: 'File not found' + }; + } + + return { + toolName: 'get_code', + input: filePath, + output: 'File not found', + success: false, + error: 'File not found' + }; + } + + return { + toolName: 'get_code', + input: filePath, + output: `File: ${filePath}\n\n${content}`, + success: true + }; + } + + /** + * Search for files matching a pattern + */ + private async searchFiles(pattern: string): Promise { + if (!this.context) { + return { + toolName: 'search_files', + input: pattern, + output: '', + success: false, + error: 'No context available' + }; + } + + const matchingFiles: string[] = []; + const lowerPattern = pattern.toLowerCase(); + + // Search in file paths + for (const filePath of this.context.fileContents.keys()) { + if (filePath.toLowerCase().includes(lowerPattern)) { + matchingFiles.push(filePath); + } + } + + // Search in file contents + for (const [filePath, content] of this.context.fileContents.entries()) { + if (!matchingFiles.includes(filePath) && + content.toLowerCase().includes(lowerPattern)) { + matchingFiles.push(filePath); + } + } + + return { + toolName: 'search_files', + input: pattern, + output: matchingFiles.length > 0 + ? `Found ${matchingFiles.length} files:\n${matchingFiles.slice(0, 10).join('\n')}${matchingFiles.length > 10 ? '\n... and more' : ''}` + : 'No files found matching the pattern', + success: true + }; + } + + /** + * Simulate graph query execution (placeholder for real implementation) + */ + private simulateGraphQuery(cypher: string): string { + if (!this.context) return 'No context available'; + + // Simple simulation based on common patterns + const upperCypher = cypher.toUpperCase(); + + if (upperCypher.includes('FUNCTION') && upperCypher.includes('RETURN')) { + const functions = this.context.graph.nodes + .filter(n => n.label === 'Function') + .slice(0, 5) + .map(n => `${n.properties.name} (${n.properties.filePath})`) + .join('\n'); + return functions || 'No functions found'; + } + + if (upperCypher.includes('CLASS') && upperCypher.includes('RETURN')) { + const classes = this.context.graph.nodes + .filter(n => n.label === 'Class') + .slice(0, 5) + .map(n => `${n.properties.name} (${n.properties.filePath})`) + .join('\n'); + return classes || 'No classes found'; + } + + if (upperCypher.includes('FILE') && upperCypher.includes('RETURN')) { + const files = this.context.graph.nodes + .filter(n => n.label === 'File') + .slice(0, 5) + .map(n => n.properties.name) + .join('\n'); + return files || 'No files found'; + } + + return 'Query executed successfully (simulated)'; + } + + /** + * Extract Cypher queries from tool results + */ + private extractCypherQueries(toolResult: ToolResult): CypherQuery[] { + const queries: CypherQuery[] = []; + const queryMatch = toolResult.output.match(/Query: (.*?)(?=\n|$)/); + + if (queryMatch) { + queries.push({ + cypher: queryMatch[1], + explanation: 'Generated during reasoning', + confidence: 0.8 + }); + } + + return queries; + } + + /** + * Extract sources from tool results + */ + private extractSources(toolResult: ToolResult): string[] { + const sources: string[] = []; + const fileMatch = toolResult.output.match(/File: (.*?)(?=\n|$)/); + + if (fileMatch) { + sources.push(fileMatch[1]); + } + + return sources; + } + + /** + * Calculate confidence based on reasoning quality + */ + private calculateConfidence(reasoning: ReasoningStep[], queries: CypherQuery[]): number { + let confidence = 0.5; + + // Boost confidence for successful tool usage + const successfulSteps = reasoning.filter(step => step.toolResult?.success).length; + confidence += (successfulSteps / reasoning.length) * 0.3; + + // Boost confidence for high-quality queries + const avgQueryConfidence = queries.length > 0 + ? queries.reduce((sum, q) => sum + q.confidence, 0) / queries.length + : 0.5; + confidence += avgQueryConfidence * 0.2; + + // Cap at reasonable bounds + return Math.min(0.95, Math.max(0.1, confidence)); + } + + /** + * Build summary prompt for incomplete reasoning + */ + private buildSummaryPrompt(question: string, reasoning: ReasoningStep[]): string { + const observations = reasoning + .map(step => `Step ${step.step}: ${step.observation}`) + .join('\n'); + + return `Based on the following observations from your reasoning process, please provide a final answer to the question: "${question}" + +Observations: +${observations} + +Final Answer:`; + } + + /** + * Get current context information + */ + public getContextInfo(): { nodeCount: number; fileCount: number; hasContext: boolean } { + if (!this.context) { + return { nodeCount: 0, fileCount: 0, hasContext: false }; + } + + return { + nodeCount: this.context.graph.nodes.length, + fileCount: this.context.fileContents.size, + hasContext: true + }; + } +} + +================ +File: src/App.css +================ +#root { + max-width: 1280px; + margin: 0 auto; + padding: 2rem; + text-align: center; +} + +.logo { + height: 6em; + padding: 1.5em; + will-change: filter; + transition: filter 300ms; +} +.logo:hover { + filter: drop-shadow(0 0 2em #646cffaa); +} +.logo.react:hover { + filter: drop-shadow(0 0 2em #61dafbaa); +} + +@keyframes logo-spin { + from { + transform: rotate(0deg); + } + to { + transform: rotate(360deg); + } +} + +@media (prefers-reduced-motion: no-preference) { + a:nth-of-type(2) .logo { + animation: logo-spin infinite 20s linear; + } +} + +.card { + padding: 2em; +} + +.read-the-docs { + color: #888; +} + +================ +File: src/App.tsx +================ +import React from 'react'; +import { HomePage, ErrorBoundary } from './ui/index.ts'; + +const App: React.FC = () => { + return ( + + + + ); +}; + +export default App; + +================ +File: src/assets/react.svg +================ + + +================ +File: src/core/graph/types.ts +================ +export type NodeLabel = + | 'Project' + | 'Package' + | 'Module' + | 'Folder' + | 'File' + | 'Class' + | 'Function' + | 'Method' + | 'Variable'; + +export interface GraphNode { + id: string; + label: NodeLabel; + properties: Record; +} + +export type RelationshipType = + | 'CONTAINS' + | 'CALLS' + | 'INHERITS' + | 'OVERRIDES' + | 'IMPORTS'; + +export interface GraphRelationship { + id: string; + type: RelationshipType; + source: string; + target: string; + properties?: Record; +} + +export interface KnowledgeGraph { + nodes: GraphNode[]; + relationships: GraphRelationship[]; +} + +================ +File: src/core/ingestion/call-processor.ts +================ +import type { KnowledgeGraph, GraphNode, GraphRelationship } from '../graph/types.ts'; +import { generateId } from '../../lib/utils.ts'; + +import type Parser from 'web-tree-sitter'; + +export interface CallResolutionInput { + graph: KnowledgeGraph; + astCache: Map; + fileContents: Map; +} + +interface FunctionCall { + callerFilePath: string; + callerFunction: string; + calledName: string; + callType: 'function' | 'method' | 'attribute' | 'super'; + line: number; + column: number; + isChained?: boolean; + objectName?: string; + superMethodName?: string; + assignedToVariable?: string; + chainedFromVariable?: string; +} + +interface ImportInfo { + filePath: string; + importedName: string; + alias?: string; + fromModule: string; + importType: 'function' | 'class' | 'module' | 'attribute'; +} + +interface VariableTypeInfo { + variableName: string; + inferredType: string; + filePath: string; + functionContext: string; + line: number; + confidence: 'high' | 'medium' | 'low'; + source: 'constructor' | 'method_return' | 'factory' | 'assignment' | 'parameter'; +} + +interface MethodReturnTypeInfo { + methodId: string; + methodName: string; + className: string; + returnType?: string; + filePath: string; +} + +const BUILTIN_FUNCTIONS = new Set([ + 'print', 'len', 'str', 'int', 'float', 'bool', 'list', 'dict', 'tuple', 'set', + 'range', 'enumerate', 'zip', 'map', 'filter', 'sum', 'max', 'min', 'abs', + 'round', 'sorted', 'reversed', 'any', 'all', 'type', 'isinstance', 'hasattr', + 'getattr', 'setattr', 'delattr', 'dir', 'vars', 'id', 'hash', 'repr', + 'open', 'input', 'format', 'exec', 'eval', 'compile', 'globals', 'locals' +]); + +export class CallProcessor { + private importCache: Map = new Map(); + private functionNodes: Map = new Map(); + private variableTypes: Map = new Map(); + private methodReturnTypes: Map = new Map(); + private classConstructors: Map = new Map(); + + public async process(input: CallResolutionInput): Promise { + const { graph, astCache, fileContents } = input; + + // Build function node lookup for fast resolution + this.buildFunctionNodeLookup(graph); + + console.log(`CallProcessor: Processing ${astCache.size} files with ASTs`); + + // Extract imports from all files first + for (const [filePath, ast] of astCache) { + if (this.isPythonFile(filePath)) { + const content = fileContents.get(filePath); + if (content && ast) { + await this.extractImports(filePath, ast, content); + } + } + } + + // For files without ASTs, try regex-based import extraction + const pythonFiles = Array.from(fileContents.keys()).filter(path => this.isPythonFile(path)); + const filesWithoutAST = pythonFiles.filter(path => !astCache.has(path)); + + if (filesWithoutAST.length > 0) { + console.log(`CallProcessor: ${filesWithoutAST.length} files don't have ASTs, using regex fallback for imports`); + for (const filePath of filesWithoutAST) { + const content = fileContents.get(filePath); + if (content) { + await this.extractImportsRegex(filePath, content); + } + } + } + + // Process function calls in all files + for (const [filePath, ast] of astCache) { + if (this.isPythonFile(filePath)) { + const content = fileContents.get(filePath); + if (content && ast) { + await this.processFunctionCalls(graph, filePath, ast, content); + } + } + } + + console.log(`CallProcessor: Found imports in ${this.importCache.size} files`); + + // Create import relationships between files + this.createImportRelationships(graph); + } + + private createImportRelationships(graph: KnowledgeGraph): void { + let importRelationshipsCreated = 0; + + for (const [filePath, imports] of this.importCache) { + const sourceFileNode = graph.nodes.find(node => + node.label === 'File' && node.properties.filePath === filePath + ); + + if (!sourceFileNode) continue; + + for (const importInfo of imports) { + // Try to find the target file based on the module name + const targetFileNode = this.findTargetFileForImport(graph, importInfo); + + if (targetFileNode) { + // Create IMPORTS relationship + const relationship: GraphRelationship = { + id: generateId('relationship', `${sourceFileNode.id}-imports-${targetFileNode.id}`), + type: 'IMPORTS', + source: sourceFileNode.id, + target: targetFileNode.id, + properties: { + importedName: importInfo.importedName, + fromModule: importInfo.fromModule, + importType: importInfo.importType + } + }; + + // Check if relationship already exists + const existingRel = graph.relationships.find(rel => + rel.source === sourceFileNode.id && + rel.target === targetFileNode.id && + rel.type === 'IMPORTS' + ); + + if (!existingRel) { + graph.relationships.push(relationship); + importRelationshipsCreated++; + } + } + } + } + + console.log(`CallProcessor: Created ${importRelationshipsCreated} import relationships`); + } + + private findTargetFileForImport(graph: KnowledgeGraph, importInfo: ImportInfo): GraphNode | null { + const moduleName = importInfo.fromModule; + + // Try different strategies to find the target file + const fileNodes = graph.nodes.filter(node => node.label === 'File'); + + // Strategy 1: Direct module name match (e.g., "utils" -> "utils.py") + let targetFile = fileNodes.find(node => { + const fileName = node.properties.name as string; + return fileName === `${moduleName}.py`; + }); + + if (targetFile) return targetFile; + + // Strategy 2: Last part of module path (e.g., "myproject.utils" -> "utils.py") + const lastPart = moduleName.split('.').pop(); + if (lastPart) { + targetFile = fileNodes.find(node => { + const fileName = node.properties.name as string; + return fileName === `${lastPart}.py`; + }); + } + + if (targetFile) return targetFile; + + // Strategy 3: Check if module path matches file path + targetFile = fileNodes.find(node => { + const filePath = node.properties.filePath as string; + return filePath && filePath.includes(moduleName.replace('.', '/')); + }); + + return targetFile || null; + } + + private buildFunctionNodeLookup(graph: KnowledgeGraph): void { + for (const node of graph.nodes) { + if (node.label === 'Function' || node.label === 'Method') { + const filePath = node.properties.filePath as string; + const functionName = node.properties.name as string; + + // Create different keys for Functions vs Methods to avoid conflicts + if (node.label === 'Method') { + const parentClass = node.properties.parentClass as string; + const methodKey = `${filePath}:method:${parentClass}.${functionName}`; + this.functionNodes.set(methodKey, node); + + // Also add a general method key for resolution when class context is unknown + const generalMethodKey = `${filePath}:method:${functionName}`; + if (!this.functionNodes.has(generalMethodKey)) { + this.functionNodes.set(generalMethodKey, node); + } + + // Track method return type information + this.buildMethodReturnTypeInfo(node, parentClass, functionName, filePath); + } else { + const functionKey = `${filePath}:function:${functionName}`; + this.functionNodes.set(functionKey, node); + } + } else if (node.label === 'Class') { + // Track class constructors for instantiation inference + const className = node.properties.name as string; + const filePath = node.properties.filePath as string; + const classKey = `${filePath}:${className}`; + this.classConstructors.set(classKey, node); + } + } + } + + private buildMethodReturnTypeInfo(methodNode: GraphNode, className: string, methodName: string, filePath: string): void { + const methodId = methodNode.id; + + // Infer return type based on method name and class context + let returnType: string | undefined; + + // Constructor methods return the class instance + if (methodName === '__init__' || methodName === '__new__') { + returnType = className; + } + // Factory methods often return class instances + else if (methodName.startsWith('create_') || methodName.startsWith('build_') || + methodName.startsWith('make_') || methodName.includes('factory')) { + returnType = this.inferFactoryReturnType(methodName, className); + } + // Getter methods often return specific types + else if (methodName.startsWith('get_')) { + returnType = this.inferGetterReturnType(methodName, className); + } + // Property methods (decorated with @property) return the property type + else if (methodNode.properties.decorators) { + const decorators = methodNode.properties.decorators as string[]; + if (decorators.includes('property')) { + returnType = this.inferPropertyReturnType(methodName, className); + } + } + + const returnTypeInfo: MethodReturnTypeInfo = { + methodId, + methodName, + className, + returnType, + filePath + }; + + this.methodReturnTypes.set(methodId, returnTypeInfo); + } + + private inferFactoryReturnType(methodName: string, className: string): string | undefined { + // Factory methods like create_user, build_report, make_connection + if (methodName.startsWith('create_')) { + const typeName = methodName.substring(7); // Remove 'create_' + return this.capitalizeFirstLetter(typeName); + } + if (methodName.startsWith('build_')) { + const typeName = methodName.substring(6); // Remove 'build_' + return this.capitalizeFirstLetter(typeName); + } + if (methodName.startsWith('make_')) { + const typeName = methodName.substring(5); // Remove 'make_' + return this.capitalizeFirstLetter(typeName); + } + + // If it's a factory class, it might return instances of the main entity + if (className.endsWith('Factory')) { + const entityName = className.substring(0, className.length - 7); // Remove 'Factory' + return entityName; + } + + return undefined; + } + + private inferGetterReturnType(methodName: string, className: string): string | undefined { + // Common getter patterns + if (methodName === 'get_name' || methodName === 'get_title') return 'str'; + if (methodName === 'get_id' || methodName === 'get_count') return 'int'; + if (methodName === 'get_price' || methodName === 'get_amount') return 'float'; + if (methodName === 'get_active' || methodName === 'get_enabled') return 'bool'; + if (methodName.includes('_list') || methodName.includes('_all')) return 'list'; + if (methodName.includes('_dict') || methodName.includes('_data')) return 'dict'; + + return undefined; + } + + private inferPropertyReturnType(methodName: string, className: string): string | undefined { + // Property return type inference based on naming patterns + if (methodName === 'name' || methodName === 'title' || methodName === 'description') return 'str'; + if (methodName === 'id' || methodName === 'count' || methodName === 'size') return 'int'; + if (methodName === 'price' || methodName === 'amount' || methodName === 'rate') return 'float'; + if (methodName === 'active' || methodName === 'enabled' || methodName === 'valid') return 'bool'; + + return undefined; + } + + private capitalizeFirstLetter(str: string): string { + return str.charAt(0).toUpperCase() + str.slice(1); + } + + private async extractImports(filePath: string, ast: any, content: string): Promise { + const imports: ImportInfo[] = []; + + this.traverseNode(ast.rootNode, (node: any) => { + if (node.type === 'import_statement') { + this.processImportStatement(node, imports, filePath); + } else if (node.type === 'import_from_statement') { + this.processImportFromStatement(node, imports, filePath); + } + }); + + this.importCache.set(filePath, imports); + } + + private processImportStatement(node: any, imports: ImportInfo[], filePath: string): void { + const nameNode = node.childForFieldName('name'); + if (nameNode) { + const moduleName = nameNode.text; + + // Handle aliased imports (import module as alias) + const asNode = node.children.find((child: any) => child.type === 'as_pattern'); + const alias = asNode ? asNode.childForFieldName('alias')?.text : undefined; + + imports.push({ + filePath, + importedName: moduleName, + alias, + fromModule: moduleName, + importType: 'module' + }); + } + } + + private processImportFromStatement(node: any, imports: ImportInfo[], filePath: string): void { + const moduleNameNode = node.childForFieldName('module_name'); + const importListNode = node.children.find((child: any) => child.type === 'import_list'); + + if (moduleNameNode && importListNode) { + const moduleName = moduleNameNode.text; + + this.traverseNode(importListNode, (child: any) => { + if (child.type === 'dotted_name' || child.type === 'identifier') { + const importedName = child.text; + imports.push({ + filePath, + importedName, + fromModule: moduleName, + importType: 'function' // Default assumption, could be refined + }); + } else if (child.type === 'as_pattern') { + const nameNode = child.childForFieldName('name'); + const aliasNode = child.childForFieldName('alias'); + + if (nameNode && aliasNode) { + imports.push({ + filePath, + importedName: nameNode.text, + alias: aliasNode.text, + fromModule: moduleName, + importType: 'function' + }); + } + } + }); + } + } + + private async processFunctionCalls( + graph: KnowledgeGraph, + filePath: string, + ast: any, + content: string + ): Promise { + const functionCalls: FunctionCall[] = []; + let currentFunction: string | null = null; + + this.traverseNode(ast.rootNode, (node: any) => { + // Track current function context + if (node.type === 'function_definition') { + const nameNode = node.childForFieldName('name'); + if (nameNode) { + currentFunction = nameNode.text; + } + } + + // Track variable assignments for type inference + if (node.type === 'assignment') { + this.processVariableAssignment(node, filePath, currentFunction); + } + + // Find function calls + if (node.type === 'call') { + const callInfo = this.extractCallInfo(node, filePath, currentFunction, content); + if (callInfo) { + // Check if this call is part of an assignment + const assignmentInfo = this.extractAssignmentInfo(node); + if (assignmentInfo) { + callInfo.assignedToVariable = assignmentInfo.variableName; + } + + functionCalls.push(callInfo); + } + } + }); + + // Resolve calls and create relationships + for (const call of functionCalls) { + await this.resolveAndCreateCallRelationship(graph, call); + + // Track variable type if this call is assigned to a variable + if (call.assignedToVariable) { + this.inferAndTrackVariableType(graph, call); + } + } + } + + private processVariableAssignment(assignmentNode: any, filePath: string, currentFunction: string | null): void { + const leftNode = assignmentNode.childForFieldName('left'); + const rightNode = assignmentNode.childForFieldName('right'); + + if (!leftNode || !rightNode) return; + + // Extract variable name from left side + let variableName: string | null = null; + if (leftNode.type === 'identifier') { + variableName = leftNode.text; + } + + if (!variableName) return; + + // Analyze right side for type inference + if (rightNode.type === 'call') { + // This will be handled in the call processing + return; + } else if (rightNode.type === 'identifier') { + // Variable assignment from another variable + const sourceVariable = rightNode.text; + this.copyVariableType(filePath, currentFunction || '', sourceVariable, variableName); + } + } + + private extractAssignmentInfo(callNode: any): { variableName: string } | null { + // Walk up the AST to find if this call is part of an assignment + let parent = callNode.parent; + + while (parent) { + if (parent.type === 'assignment') { + const leftNode = parent.childForFieldName('left'); + if (leftNode && leftNode.type === 'identifier') { + return { variableName: leftNode.text }; + } + } + parent = parent.parent; + } + + return null; + } + + private inferAndTrackVariableType(graph: KnowledgeGraph, call: FunctionCall): void { + if (!call.assignedToVariable) return; + + let inferredType: string | undefined; + let confidence: 'high' | 'medium' | 'low' = 'low'; + let source: VariableTypeInfo['source'] = 'assignment'; + + // Try to infer type based on the call + if (call.callType === 'function' || call.callType === 'method') { + // Check if it's a class constructor call + const constructorType = this.inferConstructorType(graph, call); + if (constructorType) { + inferredType = constructorType; + confidence = 'high'; + source = 'constructor'; + } else { + // Check if it's a method call with known return type + const methodReturnType = this.inferMethodReturnType(graph, call); + if (methodReturnType) { + inferredType = methodReturnType.returnType; + confidence = methodReturnType.confidence; + source = 'method_return'; + } + } + } + + if (inferredType) { + const variableKey = `${call.callerFilePath}:${call.callerFunction}:${call.assignedToVariable}`; + const typeInfo: VariableTypeInfo = { + variableName: call.assignedToVariable, + inferredType, + filePath: call.callerFilePath, + functionContext: call.callerFunction, + line: call.line, + confidence, + source + }; + + this.variableTypes.set(variableKey, typeInfo); + this.logTypeInference(call, inferredType, confidence); + } + } + + private inferConstructorType(graph: KnowledgeGraph, call: FunctionCall): string | undefined { + // Check if the called function is a class constructor + const classKey = `${call.callerFilePath}:${call.calledName}`; + if (this.classConstructors.has(classKey)) { + return call.calledName; + } + + // Check for imported class constructors + const imports = this.importCache.get(call.callerFilePath) || []; + for (const importInfo of imports) { + if (importInfo.importedName === call.calledName && importInfo.importType === 'class') { + return call.calledName; + } + } + + return undefined; + } + + private inferMethodReturnType(graph: KnowledgeGraph, call: FunctionCall): { returnType?: string, confidence: 'high' | 'medium' | 'low' } | null { + // Find the method node that was called + const targetNode = this.resolveMethodCall(graph, call) || this.resolveLocalFunction(call); + + if (targetNode && this.methodReturnTypes.has(targetNode.id)) { + const returnTypeInfo = this.methodReturnTypes.get(targetNode.id)!; + return { + returnType: returnTypeInfo.returnType, + confidence: returnTypeInfo.returnType ? 'medium' : 'low' + }; + } + + return null; + } + + private copyVariableType(filePath: string, functionContext: string, sourceVar: string, targetVar: string): void { + const sourceKey = `${filePath}:${functionContext}:${sourceVar}`; + const sourceType = this.variableTypes.get(sourceKey); + + if (sourceType) { + const targetKey = `${filePath}:${functionContext}:${targetVar}`; + const copiedType: VariableTypeInfo = { + ...sourceType, + variableName: targetVar + }; + this.variableTypes.set(targetKey, copiedType); + } + } + + private extractCallInfo( + callNode: any, + filePath: string, + currentFunction: string | null, + content: string + ): FunctionCall | null { + const functionNode = callNode.childForFieldName('function'); + if (!functionNode) return null; + + const position = callNode.startPosition; + const line = position.row + 1; + const column = position.column + 1; + + if (functionNode.type === 'identifier') { + // Simple function call: func() + return { + callerFilePath: filePath, + callerFunction: currentFunction || '', + calledName: functionNode.text, + callType: 'function', + line, + column + }; + } else if (functionNode.type === 'attribute') { + // Method call: obj.method() or super().method() or chained call + const objectNode = functionNode.childForFieldName('object'); + const attributeNode = functionNode.childForFieldName('attribute'); + + if (objectNode && attributeNode) { + // Check if this is a super() call + if (objectNode.type === 'call') { + const superFunctionNode = objectNode.childForFieldName('function'); + if (superFunctionNode && superFunctionNode.type === 'identifier' && superFunctionNode.text === 'super') { + // This is a super().method() call + return { + callerFilePath: filePath, + callerFunction: currentFunction || '', + calledName: 'super', + callType: 'super', + line, + column, + superMethodName: attributeNode.text + }; + } + } + + // Check if the object is a variable with known type (for chained calls) + let objectName = objectNode.text; + let chainedFromVariable: string | undefined; + + if (objectNode.type === 'identifier') { + // This might be a method call on a typed variable + const variableType = this.getVariableType(filePath, currentFunction || '', objectName); + if (variableType) { + chainedFromVariable = objectName; + } + } + + // Regular method call + return { + callerFilePath: filePath, + callerFunction: currentFunction || '', + calledName: attributeNode.text, + callType: 'method', + line, + column, + objectName: objectName, + chainedFromVariable + }; + } + } + + return null; + } + + private async resolveAndCreateCallRelationship( + graph: KnowledgeGraph, + call: FunctionCall + ): Promise { + const callerNode = this.findCallerNode(graph, call); + if (!callerNode) return; + + // Improved resolution order: prioritize imports to avoid self-referential calls + const targetNode = + this.resolveBuiltinFunction(call) || + this.resolveImportedFunction(call) || // Check imports first + this.resolveSuperCall(graph, call) || // Check super() calls + this.resolveMethodCall(graph, call) || + this.resolveLocalFunction(call) || // Check local functions + this.resolveWithAdvancedFallback(graph, call); // Advanced fallback for ambiguous calls + + if (targetNode) { + // Prevent self-referential calls unless it's actually recursive + if (callerNode.id === targetNode.id) { + // Only allow self-calls if the function name matches exactly (true recursion) + const callerName = callerNode.properties.name as string; + if (callerName !== call.calledName) { + console.warn(`Prevented incorrect self-referential call from ${callerName} to ${call.calledName}`); + return; + } + } + + const existingNode = graph.nodes.find(node => node.id === targetNode.id); + if (!existingNode) { + graph.nodes.push(targetNode); + } + + const relationship = this.createCallRelationship(callerNode.id, targetNode.id, call); + graph.relationships.push(relationship); + } + } + + private findCallerNode(graph: KnowledgeGraph, call: FunctionCall): GraphNode | null { + if (call.callerFunction === '') { + // Call is at module level + return graph.nodes.find(node => + node.label === 'Module' && + node.properties.path === call.callerFilePath + ) || null; + } else { + // Call is within a function + const key = `${call.callerFilePath}:${call.callerFunction}`; + return this.functionNodes.get(key) || null; + } + } + + private resolveBuiltinFunction(call: FunctionCall): GraphNode | null { + if (BUILTIN_FUNCTIONS.has(call.calledName)) { + return this.getOrCreateBuiltinNode(call.calledName); + } + return null; + } + + private resolveImportedFunction(call: FunctionCall): GraphNode | null { + const imports = this.importCache.get(call.callerFilePath) || []; + + for (const importInfo of imports) { + // Handle direct import match (from module import function) + if (importInfo.importedName === call.calledName || importInfo.alias === call.calledName) { + return this.getOrCreateImportedNode(call.calledName, importInfo.fromModule); + } + + // Handle module.function pattern (import module; module.function()) + if (call.callType === 'method' && call.objectName) { + // Check if objectName matches imported module or alias + if (importInfo.importedName === call.objectName || importInfo.alias === call.objectName) { + // This is a call to an imported module's function + return this.getOrCreateImportedNode(call.calledName, importInfo.fromModule); + } + } + + // Handle import module as alias patterns + if (call.callType === 'method' && call.objectName === importInfo.alias && importInfo.importType === 'module') { + return this.getOrCreateImportedNode(call.calledName, importInfo.fromModule); + } + } + + return null; + } + + private resolveWithAdvancedFallback(graph: KnowledgeGraph, call: FunctionCall): GraphNode | null { + // Find all potential candidates across the entire graph + const candidates = this.findAllCandidates(graph, call); + + if (candidates.length === 0) { + return null; + } + + if (candidates.length === 1) { + return candidates[0]; + } + + // Multiple candidates found - use heuristics to pick the best one + const rankedCandidates = this.rankCandidatesByProximity(call.callerFilePath, candidates); + + if (rankedCandidates.length > 0) { + const bestCandidate = rankedCandidates[0]; + + // Enable detailed logging for debugging (can be controlled via environment variable) + const enableDetailedLogging = process.env.GITNEXUS_DEBUG_FALLBACK === 'true'; + if (enableDetailedLogging) { + this.logCandidateAnalysis(call, rankedCandidates); + } else { + console.log(`Advanced fallback: Selected ${bestCandidate.node.properties.filePath}:${bestCandidate.node.properties.name} for call to ${call.calledName} from ${call.callerFilePath} (score: ${bestCandidate.score.toFixed(2)})`); + } + + return bestCandidate.node; + } + + return null; + } + + private applySpecialCaseHeuristics(call: FunctionCall, candidates: GraphNode[]): GraphNode[] { + // Apply special case filtering and prioritization + const filtered = candidates.filter(candidate => { + const candidatePath = candidate.properties.filePath as string; + const candidateName = candidate.properties.name as string; + + // Skip obvious non-matches + if (this.isObviousNonMatch(call, candidate)) { + return false; + } + + return true; + }); + + // Apply framework-specific heuristics + return this.applyFrameworkHeuristics(call, filtered); + } + + private isObviousNonMatch(call: FunctionCall, candidate: GraphNode): boolean { + const candidatePath = candidate.properties.filePath as string; + const candidateName = candidate.properties.name as string; + const callerPath = call.callerFilePath; + + // Skip if candidate is in a completely different domain + const callerDomain = this.extractDomain(callerPath); + const candidateDomain = this.extractDomain(candidatePath); + + if (callerDomain && candidateDomain && callerDomain !== candidateDomain) { + const commonDomains = ['utils', 'helpers', 'common', 'shared', 'lib', 'core']; + if (!commonDomains.includes(candidateDomain.toLowerCase())) { + return true; + } + } + + // Skip private/internal functions when caller is not in same module + if (candidateName.startsWith('_') && !this.areInSameModule(callerPath, candidatePath)) { + return true; + } + + return false; + } + + private applyFrameworkHeuristics(call: FunctionCall, candidates: GraphNode[]): GraphNode[] { + // Django-specific heuristics + if (this.isDjangoProject(call.callerFilePath)) { + return this.applyDjangoHeuristics(call, candidates); + } + + // Flask-specific heuristics + if (this.isFlaskProject(call.callerFilePath)) { + return this.applyFlaskHeuristics(call, candidates); + } + + // FastAPI-specific heuristics + if (this.isFastAPIProject(call.callerFilePath)) { + return this.applyFastAPIHeuristics(call, candidates); + } + + return candidates; + } + + private extractDomain(filePath: string): string | null { + const parts = filePath.split('/').filter(p => p.length > 0); + if (parts.length >= 2) { + return parts[parts.length - 2]; // Directory containing the file + } + return null; + } + + private areInSameModule(path1: string, path2: string): boolean { + const parts1 = path1.split('/').slice(0, -1); // Remove filename + const parts2 = path2.split('/').slice(0, -1); // Remove filename + + // Consider same module if they share at least 2 path segments + let commonSegments = 0; + const minLength = Math.min(parts1.length, parts2.length); + + for (let i = 0; i < minLength; i++) { + if (parts1[i] === parts2[i]) { + commonSegments++; + } else { + break; + } + } + + return commonSegments >= 2; + } + + private isDjangoProject(filePath: string): boolean { + return filePath.includes('django') || + filePath.includes('models.py') || + filePath.includes('views.py') || + filePath.includes('urls.py'); + } + + private isFlaskProject(filePath: string): boolean { + return filePath.includes('flask') || + filePath.includes('app.py') || + filePath.includes('routes.py'); + } + + private isFastAPIProject(filePath: string): boolean { + return filePath.includes('fastapi') || + filePath.includes('main.py') || + filePath.includes('routers/'); + } + + private applyDjangoHeuristics(call: FunctionCall, candidates: GraphNode[]): GraphNode[] { + // Prefer models.py for model-related functions, views.py for view functions, etc. + return candidates.sort((a, b) => { + const pathA = a.properties.filePath as string; + const pathB = b.properties.filePath as string; + + if (call.callerFilePath.includes('views.py') && pathA.includes('models.py')) { + return -1; // Prefer models.py when called from views.py + } + + return 0; + }); + } + + private applyFlaskHeuristics(call: FunctionCall, candidates: GraphNode[]): GraphNode[] { + // Flask-specific prioritization logic + return candidates; + } + + private applyFastAPIHeuristics(call: FunctionCall, candidates: GraphNode[]): GraphNode[] { + // FastAPI-specific prioritization logic + return candidates; + } + + private findAllCandidates(graph: KnowledgeGraph, call: FunctionCall): GraphNode[] { + const candidates: GraphNode[] = []; + + // Look for functions and methods with matching names across all files + for (const node of graph.nodes) { + if ((node.label === 'Function' || node.label === 'Method') && + node.properties.name === call.calledName && + node.properties.filePath !== call.callerFilePath) { // Exclude same file (already checked) + candidates.push(node); + } + } + + // Apply special case heuristics to filter and prioritize candidates + return this.applySpecialCaseHeuristics(call, candidates); + } + + private rankCandidatesByProximity(callerFilePath: string, candidates: GraphNode[]): Array<{node: GraphNode, score: number}> { + const scored = candidates.map(candidate => ({ + node: candidate, + score: this.calculateProximityScore(callerFilePath, candidate.properties.filePath as string) + })); + + // Sort by score (higher is better) + return scored.sort((a, b) => b.score - a.score); + } + + private calculateProximityScore(callerPath: string, candidatePath: string): number { + // Normalize paths (convert backslashes to forward slashes) + const normalizedCaller = this.normalizePath(callerPath); + const normalizedCandidate = this.normalizePath(candidatePath); + + const callerParts = normalizedCaller.split('/').filter(part => part.length > 0); + const candidateParts = normalizedCandidate.split('/').filter(part => part.length > 0); + + let score = 0; + + // Base score: prefer shorter paths (closer to root) + score += Math.max(0, 10 - candidateParts.length); + + // Proximity score: count common path segments from the beginning + let commonPrefixLength = 0; + const minLength = Math.min(callerParts.length, candidateParts.length); + + for (let i = 0; i < minLength - 1; i++) { // Exclude filename + if (callerParts[i] === candidateParts[i]) { + commonPrefixLength++; + } else { + break; + } + } + + // Higher score for more common path segments + score += commonPrefixLength * 20; + + // Same directory bonus + if (commonPrefixLength === Math.min(callerParts.length - 1, candidateParts.length - 1)) { + score += 30; + } + + // Sibling directory bonus (same parent, different immediate directory) + if (commonPrefixLength === Math.min(callerParts.length - 2, candidateParts.length - 2) && + commonPrefixLength > 0) { + score += 15; + } + + // Naming convention bonuses + score += this.calculateNamingConventionScore(normalizedCaller, normalizedCandidate); + + // Penalize deep nested paths + const nestingPenalty = Math.max(0, candidateParts.length - 5) * 2; + score -= nestingPenalty; + + return score; + } + + private calculateNamingConventionScore(callerPath: string, candidatePath: string): number { + let score = 0; + + // Extract directory and file names + const callerParts = callerPath.split('/'); + const candidateParts = candidatePath.split('/'); + + const callerDir = callerParts[callerParts.length - 2] || ''; + const candidateDir = candidateParts[candidateParts.length - 2] || ''; + + const callerFile = callerParts[callerParts.length - 1].replace(/\.[^.]*$/, ''); + const candidateFile = candidateParts[candidateParts.length - 1].replace(/\.[^.]*$/, ''); + + // Prefer utils, helpers, common files + if (candidateFile.match(/^(utils?|helpers?|common|shared|lib|core)$/i)) { + score += 10; + } + + // Prefer files with similar names + if (this.calculateStringSimilarity(callerFile, candidateFile) > 0.6) { + score += 8; + } + + // Prefer similar directory names + if (this.calculateStringSimilarity(callerDir, candidateDir) > 0.7) { + score += 5; + } + + // Avoid test files unless caller is also a test + if (candidateFile.match(/test|spec/i) && !callerFile.match(/test|spec/i)) { + score -= 20; + } + + // Avoid legacy/deprecated paths + if (candidatePath.match(/(legacy|deprecated|old|archive)/i)) { + score -= 15; + } + + return score; + } + + private calculateStringSimilarity(str1: string, str2: string): number { + if (str1 === str2) return 1.0; + if (str1.length === 0 || str2.length === 0) return 0.0; + + const longer = str1.length > str2.length ? str1 : str2; + const shorter = str1.length > str2.length ? str2 : str1; + + if (longer.length === 0) return 1.0; + + const editDistance = this.calculateLevenshteinDistance(longer, shorter); + return (longer.length - editDistance) / longer.length; + } + + private calculateLevenshteinDistance(str1: string, str2: string): number { + const matrix = Array(str2.length + 1).fill(null).map(() => Array(str1.length + 1).fill(null)); + + for (let i = 0; i <= str1.length; i++) matrix[0][i] = i; + for (let j = 0; j <= str2.length; j++) matrix[j][0] = j; + + for (let j = 1; j <= str2.length; j++) { + for (let i = 1; i <= str1.length; i++) { + const indicator = str1[i - 1] === str2[j - 1] ? 0 : 1; + matrix[j][i] = Math.min( + matrix[j][i - 1] + 1, // deletion + matrix[j - 1][i] + 1, // insertion + matrix[j - 1][i - 1] + indicator // substitution + ); + } + } + + return matrix[str2.length][str1.length]; + } + + private normalizePath(path: string): string { + return path.replace(/\\/g, '/').toLowerCase(); + } + + private resolveSuperCall(graph: KnowledgeGraph, call: FunctionCall): GraphNode | null { + if (call.callType !== 'super' || !call.superMethodName) return null; + + // Find the current method node to get its class context + const currentMethodNode = graph.nodes.find(node => + node.label === 'Method' && + node.properties.filePath === call.callerFilePath && + node.properties.name === call.callerFunction + ); + + if (!currentMethodNode || !currentMethodNode.properties.parentClass) return null; + + const currentClassName = currentMethodNode.properties.parentClass as string; + + // Find the current class node to get its base classes + const currentClassNode = graph.nodes.find(node => + node.label === 'Class' && + node.properties.filePath === call.callerFilePath && + node.properties.name === currentClassName + ); + + if (!currentClassNode || !currentClassNode.properties.baseClasses) return null; + + const baseClasses = currentClassNode.properties.baseClasses as string[]; + + // Look for the method in each base class (in order) + for (const baseClassName of baseClasses) { + const baseMethodNode = graph.nodes.find(node => + node.label === 'Method' && + node.properties.name === call.superMethodName && + node.properties.parentClass === baseClassName + ); + + if (baseMethodNode) { + return baseMethodNode; + } + } + + return null; + } + + private resolveLocalFunction(call: FunctionCall): GraphNode | null { + // Try function key first + const functionKey = `${call.callerFilePath}:function:${call.calledName}`; + const functionNode = this.functionNodes.get(functionKey); + if (functionNode) { + return functionNode; + } + + // Try method key if function not found + const methodKey = `${call.callerFilePath}:method:${call.calledName}`; + return this.functionNodes.get(methodKey) || null; + } + + private getVariableType(filePath: string, functionContext: string, variableName: string): VariableTypeInfo | null { + const variableKey = `${filePath}:${functionContext}:${variableName}`; + return this.variableTypes.get(variableKey) || null; + } + + private resolveMethodCall(graph: KnowledgeGraph, call: FunctionCall): GraphNode | null { + if (call.callType !== 'method') return null; + + // If this is a chained method call, use type information to resolve + if (call.chainedFromVariable) { + const variableType = this.getVariableType(call.callerFilePath, call.callerFunction, call.chainedFromVariable); + if (variableType) { + return this.resolveMethodOnType(graph, call, variableType.inferredType); + } + } + + // First try to find methods in the same file using the new key format + const methodKey = `${call.callerFilePath}:method:${call.calledName}`; + const methodNode = this.functionNodes.get(methodKey); + if (methodNode) { + return methodNode; + } + + // Fallback: look for method nodes in the graph (for cases where key lookup fails) + const methods = graph.nodes.filter(node => + node.label === 'Method' && + node.properties.filePath === call.callerFilePath && + node.properties.name === call.calledName + ); + + return methods[0] || null; + } + + private resolveMethodOnType(graph: KnowledgeGraph, call: FunctionCall, typeName: string): GraphNode | null { + // Look for methods of the specified type across all files + const methods = graph.nodes.filter(node => + node.label === 'Method' && + node.properties.name === call.calledName && + node.properties.parentClass === typeName + ); + + if (methods.length > 0) { + // Prefer methods in the same file, then use proximity-based ranking + const sameFileMethod = methods.find(method => + method.properties.filePath === call.callerFilePath + ); + + if (sameFileMethod) { + return sameFileMethod; + } + + // Use advanced fallback ranking for cross-file method resolution + const rankedMethods = this.rankCandidatesByProximity(call.callerFilePath, methods); + return rankedMethods.length > 0 ? rankedMethods[0].node : methods[0]; + } + + return null; + } + + private getOrCreateBuiltinNode(functionName: string): GraphNode { + const id = generateId('builtin', functionName); + return { + id, + label: 'Function', + properties: { + name: functionName, + type: 'builtin', + language: 'python' + } + }; + } + + private getOrCreateImportedNode(functionName: string, moduleName: string): GraphNode { + const id = generateId('imported', `${moduleName}.${functionName}`); + return { + id, + label: 'Function', + properties: { + name: functionName, + module: moduleName, + type: 'imported', + language: 'python' + } + }; + } + + private createCallRelationship( + callerId: string, + targetId: string, + call: FunctionCall + ): GraphRelationship { + return { + id: generateId('relationship', `${callerId}-calls-${targetId}`), + type: 'CALLS', + source: callerId, + target: targetId, + properties: { + callType: call.callType, + line: call.line, + column: call.column, + ...(call.objectName && { objectName: call.objectName }) + } + }; + } + + private traverseNode(node: any, callback: (node: any) => void): void { + callback(node); + + for (let i = 0; i < node.childCount; i++) { + const child = node.child(i); + if (child) { + this.traverseNode(child, callback); + } + } + } + + private isPythonFile(filePath: string): boolean { + return filePath.endsWith('.py'); + } + + public getImportInfo(filePath: string): ImportInfo[] { + return this.importCache.get(filePath) || []; + } + + public getCallStats(): { totalCalls: number; callTypes: Record } { + const stats = { + totalCalls: 0, + callTypes: { + function: 0, + method: 0, + builtin: 0, + imported: 0, + local: 0 + } + }; + + // This would be populated during processing + return stats; + } + + public getTypeInferenceStats(): { + totalVariablesTyped: number; + typesByConfidence: Record; + typesBySource: Record; + mostCommonTypes: Array<{type: string, count: number}>; + } { + const stats = { + totalVariablesTyped: this.variableTypes.size, + typesByConfidence: { high: 0, medium: 0, low: 0 }, + typesBySource: { constructor: 0, method_return: 0, factory: 0, assignment: 0, parameter: 0 }, + mostCommonTypes: [] as Array<{type: string, count: number}> + }; + + const typeCounts = new Map(); + + for (const typeInfo of this.variableTypes.values()) { + stats.typesByConfidence[typeInfo.confidence]++; + stats.typesBySource[typeInfo.source]++; + + const currentCount = typeCounts.get(typeInfo.inferredType) || 0; + typeCounts.set(typeInfo.inferredType, currentCount + 1); + } + + // Sort types by frequency + stats.mostCommonTypes = Array.from(typeCounts.entries()) + .map(([type, count]) => ({ type, count })) + .sort((a, b) => b.count - a.count) + .slice(0, 10); + + return stats; + } + + public getVariableTypeInfo(filePath: string, functionContext: string, variableName: string): VariableTypeInfo | null { + return this.getVariableType(filePath, functionContext, variableName); + } + + public getAllVariableTypes(): VariableTypeInfo[] { + return Array.from(this.variableTypes.values()); + } + + private logTypeInference(call: FunctionCall, inferredType: string, confidence: string): void { + if (process.env.GITNEXUS_DEBUG_TYPES === 'true') { + console.log(`Type inference: ${call.assignedToVariable} = ${call.calledName}() -> ${inferredType} (${confidence} confidence)`); + } + } + + public getAdvancedFallbackStats(): { + totalFallbackResolutions: number; + successfulResolutions: number; + ambiguousCallsResolved: number; + } { + // This would be populated during processing in a real implementation + return { + totalFallbackResolutions: 0, + successfulResolutions: 0, + ambiguousCallsResolved: 0 + }; + } + + private logCandidateAnalysis(call: FunctionCall, candidates: Array<{node: GraphNode, score: number}>): void { + console.log(`\n=== Advanced Fallback Analysis for ${call.calledName} ===`); + console.log(`Caller: ${call.callerFilePath}:${call.callerFunction}`); + console.log(`Found ${candidates.length} candidates:`); + + candidates.forEach((candidate, index) => { + const filePath = candidate.node.properties.filePath as string; + const name = candidate.node.properties.name as string; + const label = candidate.node.label; + + console.log(` ${index + 1}. ${filePath}:${name} (${label}) - Score: ${candidate.score.toFixed(2)}`); + + if (index === 0) { + console.log(` ✅ SELECTED`); + } + }); + console.log(`==========================================\n`); + } + + private async extractImportsRegex(filePath: string, content: string): Promise { + const imports: ImportInfo[] = []; + const lines = content.split('\n'); + + for (const line of lines) { + const trimmedLine = line.trim(); + + // Match: import module + const importMatch = trimmedLine.match(/^import\s+([a-zA-Z_][a-zA-Z0-9_]*(?:\.[a-zA-Z_][a-zA-Z0-9_]*)*)/); + if (importMatch) { + const moduleName = importMatch[1]; + imports.push({ + filePath, + importedName: moduleName, + fromModule: moduleName, + importType: 'module' + }); + } + + // Match: from module import item + const fromImportMatch = trimmedLine.match(/^from\s+([a-zA-Z_][a-zA-Z0-9_]*(?:\.[a-zA-Z_][a-zA-Z0-9_]*)*)\s+import\s+([a-zA-Z_][a-zA-Z0-9_]*)/); + if (fromImportMatch) { + const moduleName = fromImportMatch[1]; + const importedName = fromImportMatch[2]; + imports.push({ + filePath, + importedName, + fromModule: moduleName, + importType: 'function' // Default assumption + }); + } + } + + if (imports.length > 0) { + this.importCache.set(filePath, imports); + console.log(`Regex-extracted ${imports.length} imports from ${filePath}`); + } + } +} + +================ +File: src/core/ingestion/parsing-processor.ts +================ +import type { KnowledgeGraph, GraphNode, GraphRelationship } from '../graph/types.ts'; +import { initTreeSitter, loadPythonParser } from '../tree-sitter/parser-loader.ts'; +import { generateId } from '../../lib/utils.ts'; + +import type Parser from 'web-tree-sitter'; + +export interface ParsingInput { + filePaths: string[]; + fileContents: Map; +} + +interface ParsedDefinition { + name: string; + type: 'function' | 'class' | 'method'; + startLine: number; + endLine: number; + parentClass?: string; + decorators?: string[]; + baseClasses?: string[]; +} + +export class ParsingProcessor { + private parser: Parser | null = null; + private astCache: Map = new Map(); + + public async process(graph: KnowledgeGraph, input: ParsingInput): Promise { + const { filePaths, fileContents } = input; + + console.log('ParsingProcessor: Processing', filePaths.length, 'files'); + + // Memory optimization: Process files in batches to prevent OOM + const BATCH_SIZE = 10; // Process 10 files at a time + const sourceFiles = filePaths.filter(path => this.isSourceFile(path)); + + console.log(`ParsingProcessor: Found ${sourceFiles.length} source files, processing in batches of ${BATCH_SIZE}`); + + // Enable tree-sitter parsing once WASM files are available + try { + await this.initializeParser(); + + let successfullyParsed = 0; + let failedToParse = 0; + + // Process files in batches + for (let i = 0; i < sourceFiles.length; i += BATCH_SIZE) { + const batch = sourceFiles.slice(i, i + BATCH_SIZE); + console.log(`Processing batch ${Math.floor(i / BATCH_SIZE) + 1}/${Math.ceil(sourceFiles.length / BATCH_SIZE)} (${batch.length} files)`); + + for (const filePath of batch) { + const fileContent = fileContents.get(filePath); + if (!fileContent) { + console.warn(`No content found for source file: ${filePath}`); + continue; + } + + // Memory optimization: Skip very large files + if (fileContent.length > 500000) { // Skip files larger than 500KB + console.warn(`Skipping large file (${fileContent.length} chars): ${filePath}`); + continue; + } + + try { + const definitions = this.parseFile(filePath, fileContent); + + // Find the existing file node created by StructureProcessor + const existingFileNode = graph.nodes.find(node => + node.label === 'File' && + (node.properties.path === filePath || node.properties.filePath === filePath) + ); + + if (existingFileNode) { + // Update the existing file node with parsing results + const extension = this.getFileExtension(filePath); + const language = this.getLanguageFromExtension(extension); + + existingFileNode.properties.filePath = filePath; // Ensure filePath is set + existingFileNode.properties.extension = extension; + existingFileNode.properties.language = language; + existingFileNode.properties.definitionCount = definitions.length; + + // Create definition nodes and relationships + for (const definition of definitions) { + const defNode = this.createDefinitionNode(filePath, definition); + graph.nodes.push(defNode); + + // Create CONTAINS relationship from file to definition + graph.relationships.push({ + id: generateId('relationship', `${existingFileNode.id}-contains-${defNode.id}`), + type: 'CONTAINS', + source: existingFileNode.id, + target: defNode.id, + properties: {} + }); + } + + // Create inheritance and override relationships + this.createInheritanceRelationships(graph, filePath, definitions); + + if (definitions.length > 0) { + successfullyParsed++; + console.log(`✅ Successfully parsed ${filePath} - found ${definitions.length} definitions`); + } else { + console.log(`⚠️ No definitions found in ${filePath} (file may be empty or contain only imports)`); + } + } else { + console.warn(`⚠️ File node not found for ${filePath}, creating new one`); + + // Fallback: create file node if StructureProcessor missed it + const extension = this.getFileExtension(filePath); + const language = this.getLanguageFromExtension(extension); + + const fileNode: GraphNode = { + id: generateId('file', filePath), + label: 'File', + properties: { + name: filePath.split('/').pop() || filePath, + filePath, + extension, + language, + definitionCount: definitions.length + } + }; + graph.nodes.push(fileNode); + + // Create definition nodes and relationships + for (const definition of definitions) { + const defNode = this.createDefinitionNode(filePath, definition); + graph.nodes.push(defNode); + + // Create CONTAINS relationship from file to definition + graph.relationships.push({ + id: generateId('relationship', `${fileNode.id}-contains-${defNode.id}`), + type: 'CONTAINS', + source: fileNode.id, + target: defNode.id, + properties: {} + }); + } + + // Create inheritance and override relationships + this.createInheritanceRelationships(graph, filePath, definitions); + } + + } catch (parseError) { + failedToParse++; + console.error(`❌ Failed to parse ${filePath}:`, parseError); + + // Find existing file node and mark it as failed to parse + const existingFileNode = graph.nodes.find(node => + node.label === 'File' && + (node.properties.path === filePath || node.properties.filePath === filePath) + ); + + if (existingFileNode) { + existingFileNode.properties.parseError = true; + existingFileNode.properties.definitionCount = 0; + } + } + + // Clear AST cache periodically to save memory + if (this.astCache.size > 20) { + const oldestKeys = Array.from(this.astCache.keys()).slice(0, 10); + for (const key of oldestKeys) { + this.astCache.delete(key); + } + } + } + + // Small delay between batches to prevent overwhelming the system + await new Promise(resolve => setTimeout(resolve, 10)); + } + + console.log(`ParsingProcessor: Completed - ${successfullyParsed} successful, ${failedToParse} failed`); + } catch (error) { + console.warn('Tree-sitter parsing failed, falling back to basic file nodes:', error); + + // Fallback: create basic file nodes without parsing (also in batches) + for (let i = 0; i < sourceFiles.length; i += BATCH_SIZE) { + const batch = sourceFiles.slice(i, i + BATCH_SIZE); + + for (const filePath of batch) { + const extension = this.getFileExtension(filePath); + const language = this.getLanguageFromExtension(extension); + + const fileNode: GraphNode = { + id: generateId('file', filePath), + label: 'File', + properties: { + name: filePath.split('/').pop() || filePath, + filePath, + extension, + language + } + }; + graph.nodes.push(fileNode); + } + } + } + + console.log('ParsingProcessor: Created', graph.nodes.filter(n => n.label === 'File').length, 'file nodes'); + } + + private async initializeParser(): Promise { + if (this.parser) return; + + try { + await initTreeSitter(); + await loadPythonParser(); + + this.parser = await initTreeSitter(); + const pythonLang = await loadPythonParser(); + this.parser.setLanguage(pythonLang); + + console.log('Tree-sitter parser initialized successfully'); + } catch (error) { + console.error('Failed to initialize parser:', error); + throw new Error(`Parser initialization failed: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + private parseFile(filePath: string, fileContent: string): ParsedDefinition[] { + const extension = this.getFileExtension(filePath); + + if (extension === '.py') { + // Try Tree-sitter first, fallback to regex if it fails + if (this.parser) { + return this.parsePythonFile(filePath, fileContent); + } else { + console.warn(`Tree-sitter not available for ${filePath}, using regex fallback`); + return this.parsePythonFileRegex(filePath, fileContent); + } + } + + // Only Python files are processed now + return []; + } + + private parsePythonFileRegex(filePath: string, fileContent: string): ParsedDefinition[] { + const definitions: ParsedDefinition[] = []; + const lines = fileContent.split('\n'); + + for (let i = 0; i < lines.length; i++) { + const line = lines[i].trim(); + const lineNumber = i + 1; + + // Extract decorators for upcoming function/class definitions + const decorators = this.extractDecoratorsRegex(lines, i); + + // Match function definitions: def function_name( + const functionMatch = line.match(/^def\s+(\w+)\s*\(/); + if (functionMatch) { + definitions.push({ + name: functionMatch[1], + type: 'function', + startLine: lineNumber, + endLine: lineNumber, + decorators: decorators.length > 0 ? decorators : undefined + }); + } + + // Match class definitions: class ClassName + const classMatch = line.match(/^class\s+(\w+)(?:\s*\(.*\))?\s*:/); + if (classMatch) { + const baseClasses = this.extractBaseClassesRegex(line); + definitions.push({ + name: classMatch[1], + type: 'class', + startLine: lineNumber, + endLine: lineNumber, + decorators: decorators.length > 0 ? decorators : undefined, + baseClasses: baseClasses.length > 0 ? baseClasses : undefined + }); + } + + // Match method definitions inside classes (indented def) + const methodMatch = line.match(/^\s+def\s+(\w+)\s*\(/); + if (methodMatch) { + // Find the parent class by looking backwards + let parentClass = 'UnknownClass'; + for (let j = i - 1; j >= 0; j--) { + const prevLine = lines[j].trim(); + const classMatch = prevLine.match(/^class\s+(\w+)(?:\s*\(.*\))?\s*:/); + if (classMatch) { + parentClass = classMatch[1]; + break; + } + // Stop if we hit another function or class at root level + if (prevLine.match(/^(def|class)\s+/)) { + break; + } + } + + // Extract decorators for methods (look for indented decorators) + const methodDecorators = this.extractMethodDecoratorsRegex(lines, i); + + definitions.push({ + name: methodMatch[1], + type: 'method', + startLine: lineNumber, + endLine: lineNumber, + parentClass, + decorators: methodDecorators.length > 0 ? methodDecorators : undefined + }); + } + } + + console.log(`Regex-parsed ${definitions.length} Python definitions from ${filePath}`); + return definitions; + } + + private extractDecoratorsRegex(lines: string[], currentIndex: number): string[] { + const decorators: string[] = []; + + // Look backwards for decorator lines + for (let i = currentIndex - 1; i >= 0; i--) { + const line = lines[i].trim(); + + // Stop if we hit a non-decorator, non-empty line + if (line && !line.startsWith('@')) { + break; + } + + if (line.startsWith('@')) { + const decoratorName = this.parseDecoratorNameRegex(line); + if (decoratorName) { + decorators.unshift(decoratorName); + } + } + } + + return decorators; + } + + private extractMethodDecoratorsRegex(lines: string[], currentIndex: number): string[] { + const decorators: string[] = []; + + // Look backwards for indented decorator lines + for (let i = currentIndex - 1; i >= 0; i--) { + const line = lines[i]; + const trimmedLine = line.trim(); + + // Stop if we hit a non-decorator line that's not just whitespace + if (trimmedLine && !trimmedLine.startsWith('@')) { + break; + } + + if (trimmedLine.startsWith('@') && line.match(/^\s+@/)) { + const decoratorName = this.parseDecoratorNameRegex(trimmedLine); + if (decoratorName) { + decorators.unshift(decoratorName); + } + } + } + + return decorators; + } + + private extractBaseClassesRegex(classLine: string): string[] { + const baseClasses: string[] = []; + + // Match class definition with parentheses: class Child(Parent1, Parent2): + const match = classLine.match(/^class\s+\w+\s*\(([^)]+)\)\s*:/); + if (match) { + const baseClassesStr = match[1].trim(); + if (baseClassesStr) { + // Split by comma and clean up each base class name + const classes = baseClassesStr.split(',').map(cls => cls.trim()); + for (const cls of classes) { + // Handle simple names and qualified names + const cleanClass = cls.replace(/\s+/g, ''); + if (cleanClass && cleanClass.match(/^[a-zA-Z_][a-zA-Z0-9_.]*$/)) { + baseClasses.push(cleanClass); + } + } + } + } + + return baseClasses; + } + + private parseDecoratorNameRegex(decoratorLine: string): string | null { + // Remove @ symbol and extract decorator name + const withoutAt = decoratorLine.substring(1); + + // Handle simple decorators: @decorator_name + const simpleMatch = withoutAt.match(/^([a-zA-Z_][a-zA-Z0-9_]*(?:\.[a-zA-Z_][a-zA-Z0-9_]*)*)/); + if (simpleMatch) { + return simpleMatch[1]; + } + + // Handle decorators with arguments: @decorator_name(args) + const withArgsMatch = withoutAt.match(/^([a-zA-Z_][a-zA-Z0-9_]*(?:\.[a-zA-Z_][a-zA-Z0-9_]*)*)\s*\(/); + if (withArgsMatch) { + return withArgsMatch[1]; + } + + return null; + } + + private parsePythonFile(filePath: string, fileContent: string): ParsedDefinition[] { + if (!this.parser) { + console.warn('Parser not initialized. Cannot parse Python file:', filePath); + return this.parsePythonFileRegex(filePath, fileContent); + } + + try { + const tree = this.parser.parse(fileContent); + + this.astCache.set(filePath, tree); + + const definitions: ParsedDefinition[] = []; + const rootNode = tree.rootNode; + const processedMethodNodes = new Set(); + + // First pass: identify all methods inside classes + this.traverseNode(rootNode, (currentNode: Parser.SyntaxNode) => { + if (currentNode.type === 'class_definition') { + this.traverseNode(currentNode, (methodNode: Parser.SyntaxNode) => { + if (methodNode.type === 'function_definition' && methodNode !== currentNode) { + processedMethodNodes.add(methodNode); + } + }); + } + }); + + // Second pass: process all definitions, skipping methods that will be handled as class methods + this.traverseNode(rootNode, (currentNode: Parser.SyntaxNode) => { + if (currentNode.type === 'function_definition' && !processedMethodNodes.has(currentNode)) { + const nameNode = currentNode.childForFieldName('name'); + if (nameNode) { + const decorators = this.extractDecorators(currentNode); + definitions.push({ + name: nameNode.text, + type: 'function', + startLine: currentNode.startPosition.row + 1, + endLine: currentNode.endPosition.row + 1, + decorators: decorators.length > 0 ? decorators : undefined + }); + } + } else if (currentNode.type === 'class_definition') { + const nameNode = currentNode.childForFieldName('name'); + if (nameNode) { + const className = nameNode.text; + const decorators = this.extractDecorators(currentNode); + const baseClasses = this.extractBaseClasses(currentNode); + definitions.push({ + name: className, + type: 'class', + startLine: currentNode.startPosition.row + 1, + endLine: currentNode.endPosition.row + 1, + decorators: decorators.length > 0 ? decorators : undefined, + baseClasses: baseClasses.length > 0 ? baseClasses : undefined + }); + + const methods = this.extractMethodsFromClass(currentNode, className); + definitions.push(...methods); + } + } + }); + + console.log(`Tree-sitter parsed ${definitions.length} Python definitions from ${filePath}`); + return definitions; + } catch (error) { + console.error(`Error parsing Python file ${filePath}:`, error); + console.log(`Falling back to regex parsing for ${filePath}`); + return this.parsePythonFileRegex(filePath, fileContent); + } + } + + private createDefinitionNode(filePath: string, definition: ParsedDefinition): GraphNode { + const nodeLabel = definition.type === 'function' ? 'Function' : + (definition.type === 'method' ? 'Method' : 'Class'); + + return { + id: generateId(definition.type, `${filePath}:${definition.name}`), + label: nodeLabel as 'Function' | 'Method' | 'Class', + properties: { + name: definition.name, + filePath, + startLine: definition.startLine, + endLine: definition.endLine, + ...(definition.parentClass && { parentClass: definition.parentClass }), + ...(definition.decorators && { decorators: definition.decorators }), + ...(definition.baseClasses && { baseClasses: definition.baseClasses }) + } + }; + } + + private createInheritanceRelationships( + graph: KnowledgeGraph, + filePath: string, + definitions: ParsedDefinition[] + ): void { + // Create INHERITS relationships for classes with base classes + const classDefinitions = definitions.filter(def => def.type === 'class'); + + for (const classDef of classDefinitions) { + if (classDef.baseClasses && classDef.baseClasses.length > 0) { + const childClassId = generateId('class', `${filePath}:${classDef.name}`); + + for (const baseClassName of classDef.baseClasses) { + // Try to find the base class in the same file first + let baseClassId = generateId('class', `${filePath}:${baseClassName}`); + let baseClassExists = graph.nodes.some(node => node.id === baseClassId); + + if (!baseClassExists) { + // If not found in same file, look for it in other files + const baseClassNode = graph.nodes.find(node => + node.label === 'Class' && + node.properties.name === baseClassName + ); + + if (baseClassNode) { + baseClassId = baseClassNode.id; + baseClassExists = true; + } + } + + if (baseClassExists) { + // Create INHERITS relationship + const inheritanceRelationship: GraphRelationship = { + id: generateId('relationship', `${childClassId}-inherits-${baseClassId}`), + type: 'INHERITS', + source: childClassId, + target: baseClassId, + properties: {} + }; + + graph.relationships.push(inheritanceRelationship); + + // Create OVERRIDES relationships for methods + this.createOverrideRelationships(graph, filePath, classDef, baseClassName, definitions); + } + } + } + } + } + + private createOverrideRelationships( + graph: KnowledgeGraph, + filePath: string, + childClass: ParsedDefinition, + baseClassName: string, + allDefinitions: ParsedDefinition[] + ): void { + // Get all methods from the child class + const childMethods = allDefinitions.filter(def => + def.type === 'method' && def.parentClass === childClass.name + ); + + for (const childMethod of childMethods) { + // Look for a method with the same name in the base class + const baseMethodId = this.findBaseClassMethod(graph, baseClassName, childMethod.name); + + if (baseMethodId) { + const childMethodId = generateId('method', `${filePath}:${childMethod.name}`); + + // Create OVERRIDES relationship + const overrideRelationship: GraphRelationship = { + id: generateId('relationship', `${childMethodId}-overrides-${baseMethodId}`), + type: 'OVERRIDES', + source: childMethodId, + target: baseMethodId, + properties: { + methodName: childMethod.name, + childClass: childClass.name, + baseClass: baseClassName + } + }; + + graph.relationships.push(overrideRelationship); + } + } + } + + private findBaseClassMethod(graph: KnowledgeGraph, baseClassName: string, methodName: string): string | null { + // Look for a method with the given name in the base class + const baseMethod = graph.nodes.find(node => + node.label === 'Method' && + node.properties.name === methodName && + node.properties.parentClass === baseClassName + ); + + return baseMethod ? baseMethod.id : null; + } + + private createDefinitionRelationships( + graph: KnowledgeGraph, + filePath: string, + definitions: ParsedDefinition[] + ): void { + const fileNodeId = generateId('file', filePath); + + for (const definition of definitions) { + const defNodeId = generateId(definition.type, `${filePath}:${definition.name}`); + + const relationship: GraphRelationship = { + id: generateId('relationship', `${fileNodeId}-contains-${defNodeId}`), + source: fileNodeId, + target: defNodeId, + type: 'CONTAINS', + properties: {} + }; + + graph.relationships.push(relationship); + + if (definition.type === 'method' && definition.parentClass) { + const classNodeId = generateId('class', `${filePath}:${definition.parentClass}`); + const methodRelationship: GraphRelationship = { + id: generateId('relationship', `${classNodeId}-has-method-${defNodeId}`), + source: classNodeId, + target: defNodeId, + type: 'CONTAINS', + properties: {} + }; + + graph.relationships.push(methodRelationship); + } + } + } + + private extractDefinitions(node: Parser.SyntaxNode): ParsedDefinition[] { + const definitions: ParsedDefinition[] = []; + + this.traverseNode(node, (currentNode: Parser.SyntaxNode) => { + if (currentNode.type === 'function_definition') { + const nameNode = currentNode.childForFieldName('name'); + if (nameNode) { + definitions.push({ + name: nameNode.text, + type: 'function', + startLine: currentNode.startPosition.row + 1, + endLine: currentNode.endPosition.row + 1 + }); + } + } else if (currentNode.type === 'class_definition') { + const nameNode = currentNode.childForFieldName('name'); + if (nameNode) { + const className = nameNode.text; + definitions.push({ + name: className, + type: 'class', + startLine: currentNode.startPosition.row + 1, + endLine: currentNode.endPosition.row + 1 + }); + + const methods = this.extractMethodsFromClass(currentNode, className); + definitions.push(...methods); + } + } + }); + + return definitions; + } + + private extractMethodsFromClass(classNode: Parser.SyntaxNode, className: string): ParsedDefinition[] { + const methods: ParsedDefinition[] = []; + + this.traverseNode(classNode, (node: Parser.SyntaxNode) => { + if (node.type === 'function_definition') { + const nameNode = node.childForFieldName('name'); + if (nameNode) { + const decorators = this.extractDecorators(node); + methods.push({ + name: nameNode.text, + type: 'method', + startLine: node.startPosition.row + 1, + endLine: node.endPosition.row + 1, + parentClass: className, + decorators: decorators.length > 0 ? decorators : undefined + }); + } + } + }); + + return methods; + } + + private extractBaseClasses(classNode: Parser.SyntaxNode): string[] { + const baseClasses: string[] = []; + + // Look for argument_list node which contains base classes + const argumentList = classNode.childForFieldName('superclasses'); + if (argumentList) { + this.traverseNode(argumentList, (node: Parser.SyntaxNode) => { + if (node.type === 'identifier') { + baseClasses.push(node.text); + } else if (node.type === 'attribute') { + // Handle qualified names like module.ClassName + baseClasses.push(node.text); + } + }); + } + + return baseClasses; + } + + private extractDecorators(node: Parser.SyntaxNode): string[] { + const decorators: string[] = []; + + // Look for decorator nodes that are siblings before the function/class definition + let currentNode = node.previousSibling; + + while (currentNode && currentNode.type === 'decorator') { + const decoratorName = this.getDecoratorName(currentNode); + if (decoratorName) { + decorators.unshift(decoratorName); // Add to beginning to maintain order + } + currentNode = currentNode.previousSibling; + } + + return decorators; + } + + private getDecoratorName(decoratorNode: Parser.SyntaxNode): string | null { + // Find the identifier or attribute after the '@' symbol + for (let i = 0; i < decoratorNode.childCount; i++) { + const child = decoratorNode.child(i); + if (!child) continue; + + if (child.type === 'identifier') { + return child.text; + } else if (child.type === 'attribute') { + // Handle dotted decorators like @app.route + return child.text; + } else if (child.type === 'call') { + // Handle decorators with arguments like @retry(attempts=3) + const functionNode = child.childForFieldName('function'); + if (functionNode) { + if (functionNode.type === 'identifier') { + return functionNode.text; + } else if (functionNode.type === 'attribute') { + return functionNode.text; + } + } + } + } + + return null; + } + + private traverseNode(node: Parser.SyntaxNode, callback: (node: Parser.SyntaxNode) => void): void { + callback(node); + + for (let i = 0; i < node.childCount; i++) { + const child = node.child(i); + if (child) { + this.traverseNode(child, callback); + } + } + } + + private isSourceFile(filePath: string): boolean { + // Only process Python source files + if (!filePath) return false; + const extension = this.getFileExtension(filePath).toLowerCase(); + return extension === '.py'; + } + + private getFileExtension(filePath: string): string { + if (!filePath) return ''; + const lastDotIndex = filePath.lastIndexOf('.'); + if (lastDotIndex === -1) { + return ''; + } + return filePath.substring(lastDotIndex); + } + + private getLanguageFromExtension(extension: string): string { + switch (extension.toLowerCase()) { + case '.py': + case '.pyx': + case '.pyi': + return 'python'; + default: + return 'unknown'; + } + } + + public getAst(filePath: string): Parser.Tree | undefined { + return this.astCache.get(filePath); + } + + public getCachedAsts(): Map { + return new Map(this.astCache); + } +} + +================ +File: src/core/ingestion/pipeline.ts +================ +import type { KnowledgeGraph, GraphRelationship } from '../graph/types.ts'; +import { StructureProcessor } from './structure-processor.ts'; +import { ParsingProcessor } from './parsing-processor.ts'; +import { CallProcessor } from './call-processor.ts'; + +export interface PipelineInput { + projectRoot: string; + projectName: string; + filePaths: string[]; + fileContents: Map; +} + +export class GraphPipeline { + private structureProcessor: StructureProcessor; + private parsingProcessor: ParsingProcessor; + private callProcessor: CallProcessor; + + constructor() { + this.structureProcessor = new StructureProcessor(); + this.parsingProcessor = new ParsingProcessor(); + this.callProcessor = new CallProcessor(); + } + + public async run(input: PipelineInput): Promise { + const { projectRoot, projectName, filePaths, fileContents } = input; + + const graph: KnowledgeGraph = { + nodes: [], + relationships: [] + }; + + console.log(`Starting 3-pass ingestion for project: ${projectName}`); + + // Pass 1: Structure Analysis + console.log('Pass 1: Analyzing project structure...'); + await this.structureProcessor.process(graph, { + projectRoot, + projectName, + filePaths + }); + + // Pass 2: Code Parsing and Definition Extraction + console.log('Pass 2: Parsing code and extracting definitions...'); + await this.parsingProcessor.process(graph, { + filePaths, + fileContents + }); + + // Pass 3: Call Resolution + console.log('Pass 3: Resolving function calls and references...'); + await this.callProcessor.process({ + graph, + astCache: this.parsingProcessor.getCachedAsts(), + fileContents + }); + + console.log(`Ingestion complete. Graph contains ${graph.nodes.length} nodes and ${graph.relationships.length} relationships.`); + + // Debug: Show graph structure + const nodesByType = graph.nodes.reduce((acc, node) => { + acc[node.label] = (acc[node.label] || 0) + 1; + return acc; + }, {} as Record); + + const relationshipsByType = graph.relationships.reduce((acc, rel) => { + acc[rel.type] = (acc[rel.type] || 0) + 1; + return acc; + }, {} as Record); + + console.log('Node distribution:', nodesByType); + console.log('Relationship distribution:', relationshipsByType); + + // Validate graph integrity + this.validateGraph(graph); + + return graph; + } + + private validateGraph(graph: KnowledgeGraph): void { + const nodeIds = new Set(graph.nodes.map(node => node.id)); + const orphanedRelationships: GraphRelationship[] = []; + + for (const relationship of graph.relationships) { + if (!nodeIds.has(relationship.source)) { + console.warn(`Orphaned relationship: source node '${relationship.source}' not found for relationship '${relationship.id}'`); + orphanedRelationships.push(relationship); + } + + if (!nodeIds.has(relationship.target)) { + console.warn(`Orphaned relationship: target node '${relationship.target}' not found for relationship '${relationship.id}'`); + orphanedRelationships.push(relationship); + } + } + + // Remove orphaned relationships to prevent graph errors + if (orphanedRelationships.length > 0) { + console.warn(`Removing ${orphanedRelationships.length} orphaned relationships`); + const orphanedIds = new Set(orphanedRelationships.map(rel => rel.id)); + graph.relationships = graph.relationships.filter(rel => !orphanedIds.has(rel.id)); + } + + console.log(`Graph validation complete. Final graph: ${graph.nodes.length} nodes, ${graph.relationships.length} relationships`); + } + + public getStats(graph: KnowledgeGraph): { nodeStats: Record; relationshipStats: Record } { + const nodeStats: Record = {}; + const relationshipStats: Record = {}; + + for (const node of graph.nodes) { + nodeStats[node.label] = (nodeStats[node.label] || 0) + 1; + } + + for (const relationship of graph.relationships) { + relationshipStats[relationship.type] = (relationshipStats[relationship.type] || 0) + 1; + } + + return { nodeStats, relationshipStats }; + } + + public getCallStats(): { totalCalls: number; callTypes: Record } { + return this.callProcessor.getCallStats(); + } +} + +================ +File: src/core/ingestion/structure-processor.ts +================ +import type { KnowledgeGraph, GraphNode, GraphRelationship } from '../graph/types.ts'; +import { generateId } from '../../lib/utils.ts'; + +export interface StructureInput { + projectRoot: string; + projectName: string; + filePaths: string[]; +} + +export class StructureProcessor { + private nodeIdMap: Map = new Map(); + + public async process(graph: KnowledgeGraph, input: StructureInput): Promise { + const { projectRoot, projectName, filePaths } = input; + + // Create project root node + const projectNode = this.createProjectNode(projectName, projectRoot); + graph.nodes.push(projectNode); + + // Extract unique folder paths from file paths + const folderPaths = this.extractFolderPaths(filePaths); + + // Create folder nodes and establish hierarchy + const folderNodes = this.createFolderNodes(folderPaths); + graph.nodes.push(...folderNodes); + + // Create file nodes + const fileNodes = this.createFileNodes(filePaths); + graph.nodes.push(...fileNodes); + + // Establish CONTAINS relationships + this.createContainsRelationships(graph, projectNode.id, folderPaths, filePaths); + } + + private createProjectNode(projectName: string, projectRoot: string): GraphNode { + const id = generateId('project', projectName); + this.nodeIdMap.set('', id); // Empty path represents project root + + return { + id, + label: 'Project', + properties: { + name: projectName, + path: projectRoot, + createdAt: new Date().toISOString() + } + }; + } + + private extractFolderPaths(filePaths: string[]): string[] { + const folderSet = new Set(); + + for (const filePath of filePaths) { + const pathParts = filePath.split('/'); + + // Generate all parent folder paths + for (let i = 1; i < pathParts.length; i++) { + const folderPath = pathParts.slice(0, i).join('/'); + if (folderPath) { + folderSet.add(folderPath); + } + } + } + + return Array.from(folderSet).sort(); + } + + private createFolderNodes(folderPaths: string[]): GraphNode[] { + const nodes: GraphNode[] = []; + + for (const folderPath of folderPaths) { + const pathParts = folderPath.split('/'); + const folderName = pathParts[pathParts.length - 1]; + const id = generateId('folder', folderPath); + + this.nodeIdMap.set(folderPath, id); + + nodes.push({ + id, + label: 'Folder', + properties: { + name: folderName, + path: folderPath, + depth: pathParts.length + } + }); + } + + return nodes; + } + + private createFileNodes(filePaths: string[]): GraphNode[] { + const nodes: GraphNode[] = []; + + for (const filePath of filePaths) { + const pathParts = filePath.split('/'); + const fileName = pathParts[pathParts.length - 1]; + const fileExtension = this.getFileExtension(fileName); + const id = generateId('file', filePath); + + this.nodeIdMap.set(filePath, id); + + nodes.push({ + id, + label: 'File', + properties: { + name: fileName, + filePath: filePath, // Use filePath consistently + path: filePath, // Keep path for backward compatibility + extension: fileExtension, + isSourceFile: this.isSourceFile(fileExtension) + } + }); + } + + return nodes; + } + + private createContainsRelationships( + graph: KnowledgeGraph, + projectId: string, + folderPaths: string[], + filePaths: string[] + ): void { + const relationships: GraphRelationship[] = []; + + // Project contains root folders + const rootFolders = folderPaths.filter(path => path && !path.includes('/')); + for (const rootFolder of rootFolders) { + const folderId = this.nodeIdMap.get(rootFolder); + if (folderId) { + relationships.push({ + id: generateId('relationship', `${projectId}-contains-${folderId}`), + type: 'CONTAINS', + source: projectId, + target: folderId + }); + } + } + + // Folders contain subfolders + for (const folderPath of folderPaths) { + const parentPath = this.getParentPath(folderPath); + const parentId = this.nodeIdMap.get(parentPath); + const folderId = this.nodeIdMap.get(folderPath); + + if (parentId && folderId && parentPath !== folderPath) { + relationships.push({ + id: generateId('relationship', `${parentId}-contains-${folderId}`), + type: 'CONTAINS', + source: parentId, + target: folderId + }); + } + } + + // Folders and project contain files + for (const filePath of filePaths) { + const parentPath = this.getParentPath(filePath); + const parentId = this.nodeIdMap.get(parentPath); + const fileId = this.nodeIdMap.get(filePath); + + if (parentId && fileId) { + relationships.push({ + id: generateId('relationship', `${parentId}-contains-${fileId}`), + type: 'CONTAINS', + source: parentId, + target: fileId + }); + } + } + + graph.relationships.push(...relationships); + } + + private getParentPath(path: string): string { + const pathParts = path.split('/'); + if (pathParts.length <= 1) return ''; + return pathParts.slice(0, -1).join('/'); + } + + private getFileExtension(fileName: string): string { + const lastDotIndex = fileName.lastIndexOf('.'); + return lastDotIndex !== -1 ? fileName.substring(lastDotIndex) : ''; + } + + private isSourceFile(extension: string): boolean { + const sourceExtensions = new Set([ + '.py', '.js', '.ts', '.tsx', '.jsx', '.java', '.cpp', '.c', '.h', '.hpp', + '.cs', '.php', '.rb', '.go', '.rs', '.swift', '.kt', '.scala' + ]); + return sourceExtensions.has(extension); + } + + public getNodeId(path: string): string | undefined { + return this.nodeIdMap.get(path); + } +} + +================ +File: src/core/tree-sitter/parser-loader.ts +================ +// Import tree-sitter explicitly to ensure Vite pre-optimizes it +import Parser from "web-tree-sitter"; + +let parserInstance: Parser | null = null; +const parserCache = new Map(); + +export async function initTreeSitter(): Promise { + if (parserInstance) return parserInstance; + + try { + // Initialize WebAssembly with proper configuration + await Parser.init({ + locateFile(scriptName: string, scriptDirectory: string) { + // Return the correct path for WASM files + if (scriptName.endsWith('.wasm')) { + return `/wasm/${scriptName}`; + } + return scriptDirectory + scriptName; + } + }); + parserInstance = new Parser(); + return parserInstance; + } catch (error) { + console.error('Failed to initialize Tree-sitter:', error); + throw new Error(`Tree-sitter initialization failed: ${error instanceof Error ? error.message : 'Unknown error'}`); + } +} + +export async function loadPythonParser(): Promise { + if (parserCache.has('python')) { + return parserCache.get('python')!; + } + + try { + // Load Python language from WASM file + const wasmPath = '/wasm/python/tree-sitter-python.wasm'; + console.log('Loading Python parser from:', wasmPath); + + const pythonLang = await Parser.Language.load(wasmPath); + + parserCache.set('python', pythonLang); + console.log('Python parser loaded successfully'); + return pythonLang; + } catch (error) { + console.error('Failed to load Python parser:', error); + throw new Error(`Python parser loading failed: ${error instanceof Error ? error.message : 'Unknown error'}`); + } +} + +================ +File: src/index.css +================ +:root { + font-family: Inter, system-ui, Avenir, Helvetica, Arial, sans-serif; + line-height: 1.5; + font-weight: 400; + + color-scheme: light dark; + color: rgba(255, 255, 255, 0.87); + background-color: #242424; + + font-synthesis: none; + text-rendering: optimizeLegibility; + -webkit-font-smoothing: antialiased; + -moz-osx-font-smoothing: grayscale; +} + +a { + font-weight: 500; + color: #646cff; + text-decoration: inherit; +} +a:hover { + color: #535bf2; +} + +body { + margin: 0; + display: flex; + place-items: center; + min-width: 320px; + min-height: 100vh; +} + +h1 { + font-size: 3.2em; + line-height: 1.1; +} + +button { + border-radius: 8px; + border: 1px solid transparent; + padding: 0.6em 1.2em; + font-size: 1em; + font-weight: 500; + font-family: inherit; + background-color: #1a1a1a; + cursor: pointer; + transition: border-color 0.25s; +} +button:hover { + border-color: #646cff; +} +button:focus, +button:focus-visible { + outline: 4px auto -webkit-focus-ring-color; +} + +@media (prefers-color-scheme: light) { + :root { + color: #213547; + background-color: #ffffff; + } + a:hover { + color: #747bff; + } + button { + background-color: #f9f9f9; + } +} + +================ +File: src/lib/export.ts +================ +import type { KnowledgeGraph } from '../core/graph/types.ts'; + +export interface ExportOptions { + filename?: string; + includeMetadata?: boolean; + prettyPrint?: boolean; + includeTimestamp?: boolean; +} + +export interface ExportMetadata { + exportedAt: string; + version: string; + nodeCount: number; + relationshipCount: number; + fileCount?: number; + processingDuration?: number; +} + +export interface ExportedGraph { + metadata: ExportMetadata; + graph: KnowledgeGraph; + fileContents?: Record; +} + +/** + * Export a KnowledgeGraph to JSON format + */ +export function exportGraphToJSON( + graph: KnowledgeGraph, + options: ExportOptions = {}, + fileContents?: Map, + processingStats?: { duration: number } +): string { + const { + includeMetadata = true, + prettyPrint = true, + includeTimestamp = true + } = options; + + let exportData: ExportedGraph | KnowledgeGraph; + + if (includeMetadata) { + const metadata: ExportMetadata = { + exportedAt: includeTimestamp ? new Date().toISOString() : '', + version: '1.0.0', + nodeCount: graph.nodes.length, + relationshipCount: graph.relationships.length, + fileCount: fileContents?.size, + processingDuration: processingStats?.duration + }; + + exportData = { + metadata, + graph, + ...(fileContents && { fileContents: Object.fromEntries(fileContents) }) + }; + } else { + exportData = graph; + } + + return JSON.stringify(exportData, null, prettyPrint ? 2 : 0); +} + +/** + * Trigger download of a JSON file + */ +export function downloadJSON(content: string, filename: string): void { + try { + // Create blob with JSON content + const blob = new Blob([content], { type: 'application/json' }); + + // Create download URL + const url = URL.createObjectURL(blob); + + // Create temporary download link + const link = document.createElement('a'); + link.href = url; + link.download = filename; + link.style.display = 'none'; + + // Add to DOM, click, and remove + document.body.appendChild(link); + link.click(); + document.body.removeChild(link); + + // Clean up URL + URL.revokeObjectURL(url); + } catch (error) { + console.error('Failed to download JSON file:', error); + throw new Error('Failed to download file. Please check your browser permissions.'); + } +} + +/** + * Generate a default filename for the export + */ +export function generateExportFilename( + projectName?: string, + includeTimestamp: boolean = true +): string { + const baseName = projectName + ? `gitnexus-${projectName.replace(/[^a-zA-Z0-9-_]/g, '-')}` + : 'gitnexus-graph'; + + if (includeTimestamp) { + const timestamp = new Date().toISOString() + .replace(/[:.]/g, '-') + .replace('T', '_') + .split('.')[0]; // Remove milliseconds + return `${baseName}_${timestamp}.json`; + } + + return `${baseName}.json`; +} + +/** + * Export and download a KnowledgeGraph + */ +export function exportAndDownloadGraph( + graph: KnowledgeGraph, + options: ExportOptions & { projectName?: string } = {}, + fileContents?: Map, + processingStats?: { duration: number } +): void { + const { + filename, + projectName, + includeTimestamp = true, + ...exportOptions + } = options; + + try { + // Generate filename if not provided + const finalFilename = filename || generateExportFilename(projectName, includeTimestamp); + + // Export to JSON + const jsonContent = exportGraphToJSON(graph, exportOptions, fileContents, processingStats); + + // Trigger download + downloadJSON(jsonContent, finalFilename); + + console.log(`Successfully exported graph to ${finalFilename}`); + } catch (error) { + console.error('Export failed:', error); + throw error; + } +} + +/** + * Calculate export file size (approximate) + */ +export function calculateExportSize( + graph: KnowledgeGraph, + includeFileContents: boolean = false, + fileContents?: Map +): { sizeBytes: number; sizeFormatted: string } { + // Create a sample export to measure size + const sampleExport = exportGraphToJSON( + graph, + { includeMetadata: true, prettyPrint: false }, + includeFileContents ? fileContents : undefined + ); + + const sizeBytes = new Blob([sampleExport]).size; + const sizeFormatted = formatFileSize(sizeBytes); + + return { sizeBytes, sizeFormatted }; +} + +/** + * Format file size in human-readable format + */ +export function formatFileSize(bytes: number): string { + if (bytes === 0) return '0 Bytes'; + + const k = 1024; + const sizes = ['Bytes', 'KB', 'MB', 'GB']; + const i = Math.floor(Math.log(bytes) / Math.log(k)); + + return parseFloat((bytes / Math.pow(k, i)).toFixed(2)) + ' ' + sizes[i]; +} + +/** + * Validate if a graph can be exported + */ +export function validateGraphForExport(graph: KnowledgeGraph): { + isValid: boolean; + errors: string[]; + warnings: string[]; +} { + const errors: string[] = []; + const warnings: string[] = []; + + // Check if graph exists + if (!graph) { + errors.push('Graph is null or undefined'); + return { isValid: false, errors, warnings }; + } + + // Check if graph has nodes + if (!graph.nodes || graph.nodes.length === 0) { + warnings.push('Graph has no nodes'); + } + + // Check if graph has relationships + if (!graph.relationships || graph.relationships.length === 0) { + warnings.push('Graph has no relationships'); + } + + // Check for invalid node IDs + const nodeIds = new Set(graph.nodes.map(n => n.id)); + if (nodeIds.size !== graph.nodes.length) { + errors.push('Graph contains duplicate node IDs'); + } + + // Check for invalid relationships + graph.relationships.forEach((rel, index) => { + if (!nodeIds.has(rel.source)) { + errors.push(`Relationship ${index} has invalid source node ID: ${rel.source}`); + } + if (!nodeIds.has(rel.target)) { + errors.push(`Relationship ${index} has invalid target node ID: ${rel.target}`); + } + }); + + // Check for very large exports + const approximateSize = JSON.stringify(graph).length; + if (approximateSize > 50 * 1024 * 1024) { // 50MB + warnings.push('Export file will be very large (>50MB). Consider filtering the data.'); + } + + return { + isValid: errors.length === 0, + errors, + warnings + }; +} + +/** + * Import a KnowledgeGraph from JSON string + */ +export function importGraphFromJSON(jsonString: string): { + graph: KnowledgeGraph; + metadata?: ExportMetadata; + fileContents?: Map; +} { + try { + const parsed = JSON.parse(jsonString); + + // Check if it's an exported graph with metadata + if (parsed.metadata && parsed.graph) { + const result: { + graph: KnowledgeGraph; + metadata: ExportMetadata; + fileContents?: Map; + } = { + graph: parsed.graph, + metadata: parsed.metadata + }; + + // Convert file contents back to Map if present + if (parsed.fileContents) { + result.fileContents = new Map(Object.entries(parsed.fileContents)); + } + + return result; + } + + // Assume it's a raw graph + return { graph: parsed }; + } catch (error) { + throw new Error(`Failed to import graph: ${error instanceof Error ? error.message : 'Invalid JSON'}`); + } +} + +/** + * Create a filtered export of the graph + */ +export function createFilteredExport( + graph: KnowledgeGraph, + filters: { + nodeTypes?: string[]; + relationshipTypes?: string[]; + filePatterns?: string[]; + maxNodes?: number; + } +): KnowledgeGraph { + const { nodeTypes, relationshipTypes, filePatterns, maxNodes } = filters; + + let filteredNodes = graph.nodes; + let filteredRelationships = graph.relationships; + + // Filter by node types + if (nodeTypes && nodeTypes.length > 0) { + filteredNodes = filteredNodes.filter(node => nodeTypes.includes(node.label)); + } + + // Filter by file patterns + if (filePatterns && filePatterns.length > 0) { + filteredNodes = filteredNodes.filter(node => { + const filePath = node.properties.filePath as string; + if (!filePath) return true; // Keep nodes without file paths + + return filePatterns.some(pattern => { + const regex = new RegExp(pattern.replace(/\*/g, '.*'), 'i'); + return regex.test(filePath); + }); + }); + } + + // Limit number of nodes + if (maxNodes && filteredNodes.length > maxNodes) { + filteredNodes = filteredNodes.slice(0, maxNodes); + } + + // Get filtered node IDs + const filteredNodeIds = new Set(filteredNodes.map(n => n.id)); + + // Filter relationships to only include those between filtered nodes + filteredRelationships = filteredRelationships.filter(rel => + filteredNodeIds.has(rel.source) && filteredNodeIds.has(rel.target) + ); + + // Filter by relationship types + if (relationshipTypes && relationshipTypes.length > 0) { + filteredRelationships = filteredRelationships.filter(rel => + relationshipTypes.includes(rel.type) + ); + } + + return { + nodes: filteredNodes, + relationships: filteredRelationships + }; +} + +================ +File: src/lib/polyfills.ts +================ +// Browser polyfill for Node.js AsyncLocalStorage +export class AsyncLocalStorage { + private storage = new Map(); + private currentId = 0; + + constructor() {} + + run(store: T, callback: () => R): R { + const id = (++this.currentId).toString(); + this.storage.set(id, store); + try { + return callback(); + } finally { + this.storage.delete(id); + } + } + + getStore(): T | undefined { + // In browser context, we can't truly replicate AsyncLocalStorage + // Return undefined as fallback + return undefined; + } +} + +// Export as both named and default to match different import styles +export { AsyncLocalStorage as default }; + +// Polyfill for global async_hooks if not available +if (typeof globalThis !== 'undefined' && !(globalThis as any).AsyncLocalStorage) { + (globalThis as any).AsyncLocalStorage = AsyncLocalStorage; +} + +================ +File: src/lib/preload.ts +================ +// Preload all dependencies that might be loaded during processing +// This ensures Vite optimizes them during initial build rather than during runtime + +import 'web-tree-sitter'; +import 'comlink'; + +// Import all the processing modules to ensure their dependencies are discovered +import '../core/ingestion/pipeline'; +import '../core/ingestion/parsing-processor'; +import '../core/ingestion/call-processor'; +import '../core/ingestion/structure-processor'; +import '../core/tree-sitter/parser-loader'; + +console.log('Dependencies preloaded to prevent runtime optimization'); + +================ +File: src/lib/utils.ts +================ +export function generateId(type: string, identifier: string): string { + const timestamp = Date.now(); + const hash = simpleHash(identifier); + return `${type}_${hash}_${timestamp}`; +} + +function simpleHash(str: string): string { + let hash = 0; + for (let i = 0; i < str.length; i++) { + const char = str.charCodeAt(i); + hash = ((hash << 5) - hash) + char; + hash = hash & hash; + } + return Math.abs(hash).toString(36); +} + +================ +File: src/lib/workerUtils.ts +================ +// @ts-expect-error - imports are resolved at runtime in Deno +import * as Comlink from 'comlink'; +import type { IngestionWorker, IngestionProgress, IngestionResult } from '../workers/ingestion.worker.ts'; +import type { PipelineInput } from '../core/ingestion/pipeline.ts'; + +// Export types for external use +export type { IngestionProgress, IngestionResult }; + +export interface WorkerProxy { + processRepository(input: PipelineInput): Promise; + processFiles(projectName: string, files: { path: string; content: string }[]): Promise; + validateRepository(input: PipelineInput): Promise<{ valid: boolean; errors: string[] }>; + getWorkerInfo(): Promise<{ version: string; capabilities: string[] }>; + setProgressCallback(callback: (progress: IngestionProgress) => void): Promise; + terminate(): Promise; +} + +export class IngestionWorkerManager { + private worker: Worker | null = null; + private workerProxy: WorkerProxy | null = null; + private isInitialized = false; + + public async initialize(): Promise { + if (this.isInitialized) { + return; + } + + try { + // Create the worker + this.worker = new Worker( + new URL('../workers/ingestion.worker.ts', import.meta.url).href, + { + type: 'module', + name: 'ingestion-worker' + } + ); + + // Wrap with Comlink + this.workerProxy = Comlink.wrap(this.worker) as WorkerProxy; + + this.isInitialized = true; + console.log('Ingestion worker initialized successfully'); + + } catch (error) { + throw new Error(`Failed to initialize ingestion worker: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + public async processRepository(input: PipelineInput): Promise { + await this.ensureInitialized(); + + try { + return await this.workerProxy!.processRepository(input); + } catch (error) { + throw new Error(`Worker processing failed: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + public async processFiles( + projectName: string, + files: { path: string; content: string }[] + ): Promise { + await this.ensureInitialized(); + + try { + return await this.workerProxy!.processFiles(projectName, files); + } catch (error) { + throw new Error(`Worker file processing failed: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + public async validateRepository(input: PipelineInput): Promise<{ valid: boolean; errors: string[] }> { + await this.ensureInitialized(); + + try { + return await this.workerProxy!.validateRepository(input); + } catch (error) { + return { + valid: false, + errors: [`Validation failed: ${error instanceof Error ? error.message : 'Unknown error'}`] + }; + } + } + + public async setProgressCallback(callback: (progress: IngestionProgress) => void): Promise { + await this.ensureInitialized(); + + try { + // Wrap callback with Comlink.proxy to allow it to be called from worker + const proxiedCallback = Comlink.proxy(callback); + await this.workerProxy!.setProgressCallback(proxiedCallback); + } catch (error) { + console.warn('Failed to set progress callback:', error); + } + } + + public async getWorkerInfo(): Promise<{ version: string; capabilities: string[] }> { + await this.ensureInitialized(); + + try { + return await this.workerProxy!.getWorkerInfo(); + } catch (error) { + throw new Error(`Failed to get worker info: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + public async terminate(): Promise { + if (!this.isInitialized || !this.worker) { + return; + } + + try { + // Notify worker to cleanup + if (this.workerProxy) { + await this.workerProxy.terminate(); + } + + // Terminate the worker + this.worker.terminate(); + + // Cleanup references + this.worker = null; + this.workerProxy = null; + this.isInitialized = false; + + console.log('Ingestion worker terminated'); + + } catch (error) { + console.warn('Error during worker termination:', error); + + // Force terminate if cleanup fails + if (this.worker) { + this.worker.terminate(); + this.worker = null; + this.workerProxy = null; + this.isInitialized = false; + } + } + } + + public isWorkerReady(): boolean { + return this.isInitialized && this.worker !== null && this.workerProxy !== null; + } + + private async ensureInitialized(): Promise { + if (!this.isInitialized) { + await this.initialize(); + } + } +} + +// Singleton instance for easy access +let workerManager: IngestionWorkerManager | null = null; + +export function getIngestionWorker(): IngestionWorkerManager { + if (!workerManager) { + workerManager = new IngestionWorkerManager(); + } + return workerManager; +} + +export async function createIngestionWorker(): Promise { + const manager = new IngestionWorkerManager(); + await manager.initialize(); + return manager; +} + +// Utility function for processing with automatic cleanup +export async function processWithWorker( + processor: (worker: IngestionWorkerManager) => Promise +): Promise { + const worker = await createIngestionWorker(); + + try { + return await processor(worker); + } finally { + await worker.terminate(); + } +} + +// Error handling utilities +export class WorkerError extends Error { + constructor(message: string, public readonly cause?: Error) { + super(message); + this.name = 'WorkerError'; + } +} + +export function isWorkerSupported(): boolean { + try { + return typeof Worker !== 'undefined'; + } catch { + return false; + } +} + +================ +File: src/main.tsx +================ +import './lib/polyfills.ts'; +import './lib/preload.ts'; +import { StrictMode } from 'react' +import { createRoot } from 'react-dom/client' +import App from './App.tsx' +import './index.css' + +createRoot(document.getElementById('root')!).render( + +) + +================ +File: src/services/github.ts +================ +import axios, { type AxiosInstance, type AxiosResponse } from 'axios'; + +interface GitHubFile { + name: string; + path: string; + sha: string; + size: number; + url: string; + html_url: string; + git_url: string; + download_url: string | null; + type: 'file' | 'dir'; + content?: string; + encoding?: string; +} + +interface GitHubDirectory { + name: string; + path: string; + sha: string; + size: number; + url: string; + html_url: string; + git_url: string; + download_url: string | null; + type: 'file' | 'dir'; +} + +interface RateLimitInfo { + limit: number; + remaining: number; + reset: number; + used: number; +} + +interface GitHubError { + message: string; + documentation_url?: string; +} + +export class GitHubService { + private client: AxiosInstance; + private baseURL = 'https://api.github.com'; + private rateLimitInfo: RateLimitInfo | null = null; + + constructor(token?: string) { + this.client = axios.create({ + baseURL: this.baseURL, + headers: { + 'Accept': 'application/vnd.github.v3+json', + ...(token && { 'Authorization': `Bearer ${token}` }) + }, + timeout: 30000 + }); + + this.setupInterceptors(); + } + + private setupInterceptors(): void { + this.client.interceptors.response.use( + (response: AxiosResponse) => { + this.updateRateLimitInfo(response); + return response; + }, + (error: { response?: AxiosResponse; message: string }) => { + if (error.response) { + this.updateRateLimitInfo(error.response); + + if (error.response.status === 403 && this.isRateLimited()) { + const resetTime = new Date(this.rateLimitInfo!.reset * 1000); + throw new Error(`GitHub API rate limit exceeded. Resets at ${resetTime.toISOString()}`); + } + + if (error.response.status === 401) { + throw new Error('GitHub API authentication failed. Please check your token.'); + } + + if (error.response.status === 404) { + throw new Error('Repository or resource not found.'); + } + + const githubError: GitHubError = error.response.data; + throw new Error(`GitHub API error: ${githubError.message}`); + } + + throw new Error(`Network error: ${error.message}`); + } + ); + } + + private updateRateLimitInfo(response: AxiosResponse): void { + const headers = response.headers; + if (headers['x-ratelimit-limit']) { + this.rateLimitInfo = { + limit: parseInt(headers['x-ratelimit-limit'], 10), + remaining: parseInt(headers['x-ratelimit-remaining'], 10), + reset: parseInt(headers['x-ratelimit-reset'], 10), + used: parseInt(headers['x-ratelimit-used'], 10) + }; + } + } + + private isRateLimited(): boolean { + return this.rateLimitInfo !== null && this.rateLimitInfo.remaining === 0; + } + + public getRateLimitInfo(): RateLimitInfo | null { + return this.rateLimitInfo; + } + + public async checkRateLimit(): Promise { + if (this.isRateLimited()) { + const resetTime = new Date(this.rateLimitInfo!.reset * 1000); + const now = new Date(); + + if (now < resetTime) { + const waitTime = Math.ceil((resetTime.getTime() - now.getTime()) / 1000); + throw new Error(`Rate limit exceeded. Wait ${waitTime} seconds before making another request.`); + } + } + } + + public async getRepositoryContents( + owner: string, + repo: string, + path: string = '' + ): Promise<(GitHubFile | GitHubDirectory)[]> { + await this.checkRateLimit(); + + try { + const response = await this.client.get(`/repos/${owner}/${repo}/contents/${path}`); + + if (!Array.isArray(response.data)) { + throw new Error('Expected directory contents, but received a single file.'); + } + + return response.data as (GitHubFile | GitHubDirectory)[]; + } catch (error) { + if (error instanceof Error) { + throw error; + } + throw new Error('Failed to fetch repository contents'); + } + } + + public async getFileContent( + owner: string, + repo: string, + path: string + ): Promise { + await this.checkRateLimit(); + + try { + const response = await this.client.get(`/repos/${owner}/${repo}/contents/${path}`); + const file = response.data as GitHubFile; + + if (file.type !== 'file') { + throw new Error(`Path ${path} is not a file`); + } + + // If content or encoding is missing, try to download directly + if (!file.content || !file.encoding) { + if (file.download_url) { + console.warn(`File ${path} missing content/encoding, downloading directly`); + return await this.downloadFileRaw(owner, repo, path); + } else { + throw new Error('File content, encoding, and download URL are all missing'); + } + } + + if (file.encoding === 'base64') { + try { + return atob(file.content.replace(/\s/g, '')); + } catch { + throw new Error('Failed to decode base64 content'); + } + } + + return file.content; + } catch (error) { + if (error instanceof Error) { + throw error; + } + throw new Error('Failed to fetch file content'); + } + } + + public async downloadFileRaw( + owner: string, + repo: string, + path: string + ): Promise { + await this.checkRateLimit(); + + try { + const response = await this.client.get(`/repos/${owner}/${repo}/contents/${path}`); + const file = response.data as GitHubFile; + + if (file.type !== 'file' || !file.download_url) { + throw new Error(`Cannot download file: ${path}`); + } + + const downloadResponse = await axios.get(file.download_url, { + timeout: 30000 + }); + + return downloadResponse.data; + } catch (error) { + if (error instanceof Error) { + throw error; + } + throw new Error('Failed to download file'); + } + } + + public async getAllFilesRecursively(owner: string, repo: string, path: string = ''): Promise { + const files: GitHubFile[] = []; + + try { + const contents = await this.getRepositoryContents(owner, repo, path); + + for (const item of contents) { + if (item.type === 'dir') { + // Skip common directories that shouldn't be processed + if (this.shouldSkipDirectory(item.path)) { + console.log(`Skipping directory: ${item.path}`); + continue; + } + + // Recursively get files from subdirectories + const subFiles = await this.getAllFilesRecursively(owner, repo, item.path); + files.push(...subFiles); + } else if (item.type === 'file') { + // Only include files that should be processed + if (this.shouldIncludeFile(item.path)) { + files.push(item); + } else { + console.log(`Skipping file: ${item.path}`); + } + } + } + } catch (error) { + console.error(`Error fetching contents for ${path}:`, error); + } + + return files; + } + + private shouldSkipDirectory(path: string): boolean { + if (!path) return true; // Skip if path is undefined/null + + const skipDirs = [ + // Git version control + '.git', + // JavaScript dependencies (common in full-stack projects) + 'node_modules', + // Python bytecode cache + '__pycache__', + // Python virtual environments + 'venv', + 'env', + '.venv', + 'envs', + 'virtualenv', + // Build, distribution, and temporary directories + 'build', + 'dist', + 'docs', + 'logs', + 'tmp', + '.tmp', + // Additional common directories to skip + 'coverage', + '.coverage', + 'htmlcov', + 'vendor', + 'deps', + '_build', + '.gradle', + 'bin', + 'obj', + '.vs', + '.vscode', + '.idea', + 'temp' + ]; + + // Check each directory component in the path + const pathParts = path.split('/'); + for (const part of pathParts) { + const dirName = part.toLowerCase(); + + // Check for exact matches + if (skipDirs.includes(dirName) || dirName.startsWith('.')) { + return true; + } + + // Check for .egg-info directories + if (dirName.endsWith('.egg-info')) { + return true; + } + } + + // Check for virtual environment patterns anywhere in the path + const fullPathLower = path.toLowerCase(); + const venvPatterns = [ + '/.venv/', + '/venv/', + '/env/', + '/.env/', + '/envs/', + '/virtualenv/', + '/site-packages/', + '/lib/python', + '/lib64/python', + '/scripts/', + '/bin/python' + ]; + + if (venvPatterns.some(pattern => fullPathLower.includes(pattern))) { + return true; + } + + return false; + } + + private shouldIncludeFile(path: string): boolean { + if (!path) return false; // Skip if path is undefined/null + + const fileName = path.split('/').pop() || ''; + + // Skip hidden files except specific config files + if (fileName.startsWith('.') && !fileName.endsWith('.env.example')) { + return false; + } + + // Skip Python-specific file patterns + const skipPatterns = [ + // Python compiled bytecode + /\.pyc$/, + /\.pyo$/, + // Python extension modules (binary) + /\.pyd$/, + /\.so$/, + // Python packages + /\.egg$/, + /\.whl$/, + // Lock files + /\.lock$/, + /poetry\.lock$/, + /Pipfile\.lock$/, + // Editor swap files + /\..*\.swp$/, + /\..*\.swo$/, + // OS metadata files + /^Thumbs\.db$/, + /^\.DS_Store$/, + // General binary and archive files + /\.zip$/, + /\.tar$/, + /\.rar$/, + /\.7z$/, + /\.gz$/, + // Media files + /\.(jpg|jpeg|png|gif|bmp|svg|ico)$/i, + /\.(mp4|avi|mov|wmv|flv|webm)$/i, + /\.(mp3|wav|flac|aac|ogg)$/i, + // Document files + /\.(pdf|doc|docx|xls|xlsx|ppt|pptx)$/i, + // Other binary files + /\.(exe|dll|dylib)$/i, + // Minified files and source maps + /\.min\.(js|css)$/, + /\.map$/, + // Log and temporary files + /\.log$/, + /\.tmp$/, + /\.cache$/, + /\.pid$/, + /\.seed$/ + ]; + + if (skipPatterns.some(pattern => pattern.test(fileName))) { + return false; + } + + // Only include Python files and essential config files + const extension = '.' + fileName.split('.').pop()?.toLowerCase(); + + // Python source files + if (extension === '.py') { + return true; + } + + // Essential Python config files + const importantPythonFiles = [ + 'pyproject.toml', + 'setup.py', + 'requirements.txt', + 'setup.cfg', + 'tox.ini', + 'pytest.ini', + 'Pipfile', + 'poetry.toml', + 'README.md', + 'LICENSE', + 'CHANGELOG.md', + 'MANIFEST.in' + ]; + + if (importantPythonFiles.includes(fileName)) { + return true; + } + + return false; + } + + public getAuthenticationStatus(): { authenticated: boolean; rateLimitInfo: RateLimitInfo | null } { + const authHeader = this.client.defaults.headers['Authorization']; + return { + authenticated: !!authHeader, + rateLimitInfo: this.rateLimitInfo + }; + } +} + +================ +File: src/services/ingestion.service.ts +================ +import { GitHubService } from './github.ts'; +import { ZipService } from './zip.ts'; +import { getIngestionWorker, type IngestionProgress } from '../lib/workerUtils.ts'; +import type { KnowledgeGraph } from '../core/graph/types.ts'; + +export interface IngestionOptions { + directoryFilter?: string; + fileExtensions?: string; + onProgress?: (message: string) => void; +} + +export interface IngestionResult { + graph: KnowledgeGraph; + fileContents: Map; +} + +export class IngestionService { + private githubService: GitHubService; + private zipService: ZipService; + + constructor(githubToken?: string) { + this.githubService = new GitHubService(githubToken); + this.zipService = new ZipService(); + } + + async processGitHubRepo( + githubUrl: string, + options: IngestionOptions = {} + ): Promise { + const { directoryFilter, fileExtensions, onProgress } = options; + + // Parse GitHub URL + const match = githubUrl.match(/^https:\/\/github\.com\/([^/]+)\/([^/]+)(?:\/.*)?$/); + if (!match) { + throw new Error('Invalid GitHub repository URL'); + } + + const [, owner, repo] = match; + + onProgress?.('Fetching repository structure...'); + + // Get all files from the repository + const allFiles = await this.githubService.getAllFilesRecursively(owner, repo); + + // Filter files based on options + const filteredFiles = this.filterFiles(allFiles, directoryFilter, fileExtensions); + + onProgress?.(`Found ${filteredFiles.length} files. Downloading content...`); + + // Download file contents + const fileContents = new Map(); + let processedFiles = 0; + + for (const file of filteredFiles) { + try { + const content = await this.githubService.getFileContent(owner, repo, file.path); + if (content && content.length <= 1000000) { // Skip files larger than 1MB + fileContents.set(file.path, content); + } + processedFiles++; + + if (processedFiles % 10 === 0) { + onProgress?.(`Downloaded ${processedFiles}/${filteredFiles.length} files...`); + } + } catch (error) { + console.warn(`Failed to download ${file.path}:`, error); + } + } + + onProgress?.('Processing files with knowledge graph engine...'); + + // Process with ingestion worker + const graph = await this.processWithWorker( + fileContents, + `${owner}/${repo}`, + Array.from(fileContents.keys()), + onProgress + ); + + return { graph, fileContents }; + } + + async processZipFile( + file: File, + options: IngestionOptions = {} + ): Promise { + const { directoryFilter, fileExtensions, onProgress } = options; + + onProgress?.('Extracting ZIP file...'); + + // Extract ZIP contents + const allFileContents = await this.zipService.extractTextFiles(file); + + // Filter files based on options + const filteredFileContents = new Map(); + const allPaths = Array.from(allFileContents.keys()); + const filteredPaths = this.filterPaths(allPaths, directoryFilter, fileExtensions); + + filteredPaths.forEach(path => { + const content = allFileContents.get(path); + if (content) { + filteredFileContents.set(path, content); + } + }); + + onProgress?.(`Extracted ${filteredFileContents.size} files. Processing...`); + + // Process with ingestion worker + const projectName = file.name.replace(/\.zip$/i, ''); + const graph = await this.processWithWorker( + filteredFileContents, + projectName, + Array.from(filteredFileContents.keys()), + onProgress + ); + + return { graph, fileContents: filteredFileContents }; + } + + private async processWithWorker( + fileContents: Map, + projectName: string, + filePaths: string[], + onProgress?: (message: string) => void + ): Promise { + const worker = getIngestionWorker(); + + try { + await worker.initialize(); + + // Set up progress callback + await worker.setProgressCallback((progress: IngestionProgress) => { + onProgress?.(progress.message); + }); + + onProgress?.('Building knowledge graph...'); + + // Process repository + const result = await worker.processRepository({ + projectRoot: '/', + projectName, + filePaths, + fileContents + }); + + if (!result.success || !result.graph) { + throw new Error(result.error || 'Failed to process repository'); + } + + return result.graph; + } catch (error) { + throw new Error(`Processing failed: ${error instanceof Error ? error.message : 'Unknown error'}`); + } + } + + private filterFiles(files: any[], directoryFilter?: string, fileExtensions?: string): any[] { + let filtered = files; + + // Filter by directory + if (directoryFilter?.trim()) { + const dirPatterns = directoryFilter.toLowerCase().split(',').map(p => p.trim()); + filtered = filtered.filter(file => + dirPatterns.some(pattern => file.path.toLowerCase().includes(pattern)) + ); + } + + // Filter by file extensions + if (fileExtensions?.trim()) { + const extensions = fileExtensions.toLowerCase().split(',').map(ext => ext.trim()); + filtered = filtered.filter(file => + extensions.some(ext => file.path.toLowerCase().endsWith(ext)) + ); + } + + return filtered; + } + + private filterPaths(paths: string[], directoryFilter?: string, fileExtensions?: string): string[] { + let filtered = paths; + + // Filter by directory + if (directoryFilter?.trim()) { + const dirPatterns = directoryFilter.toLowerCase().split(',').map(p => p.trim()); + filtered = filtered.filter(path => + dirPatterns.some(pattern => path.toLowerCase().includes(pattern)) + ); + } + + // Filter by file extensions + if (fileExtensions?.trim()) { + const extensions = fileExtensions.toLowerCase().split(',').map(ext => ext.trim()); + filtered = filtered.filter(path => + extensions.some(ext => path.toLowerCase().endsWith(ext)) + ); + } + + return filtered; + } +} + +================ +File: src/services/zip.ts +================ +import JSZip from 'jszip'; + +interface JSZipObjectWithData extends JSZip.JSZipObject { + _data?: { + uncompressedSize: number; + }; +} + +interface ZipFileEntry { + path: string; + content: string; + isDirectory: boolean; + size: number; + lastModified: Date; +} + +interface ExtractionOptions { + maxFileSize?: number; + maxTotalSize?: number; + allowedExtensions?: string[]; + excludeDirectories?: boolean; +} + +export class ZipService { + private static readonly DEFAULT_MAX_FILE_SIZE = 10 * 1024 * 1024; // 10MB + private static readonly DEFAULT_MAX_TOTAL_SIZE = 100 * 1024 * 1024; // 100MB + private static readonly TEXT_EXTENSIONS = new Set([ + '.js', '.ts', '.tsx', '.jsx', '.py', '.java', '.cpp', '.c', '.h', '.hpp', + '.cs', '.php', '.rb', '.go', '.rs', '.swift', '.kt', '.scala', '.clj', + '.html', '.htm', '.xml', '.css', '.scss', '.sass', '.less', '.json', + '.yaml', '.yml', '.toml', '.ini', '.cfg', '.conf', '.md', '.txt', + '.sql', '.sh', '.bash', '.zsh', '.fish', '.ps1', '.bat', '.cmd', + '.dockerfile', '.gitignore', '.gitattributes', '.env', '.properties' + ]); + + constructor() {} + + public async extractTextFiles( + file: File, + options: ExtractionOptions = {} + ): Promise> { + const { + maxFileSize = ZipService.DEFAULT_MAX_FILE_SIZE, + maxTotalSize = ZipService.DEFAULT_MAX_TOTAL_SIZE, + allowedExtensions = Array.from(ZipService.TEXT_EXTENSIONS), + excludeDirectories = true + } = options; + + if (!file) { + throw new Error('No file provided'); + } + + if (file.type !== 'application/zip' && !file.name.toLowerCase().endsWith('.zip')) { + throw new Error('File must be a ZIP archive'); + } + + try { + const arrayBuffer = await this.fileToArrayBuffer(file); + const zip = new JSZip(); + const zipContent = await zip.loadAsync(arrayBuffer); + + const extractedFiles = new Map(); + let totalExtractedSize = 0; + + const files = Object.keys(zipContent.files); + + for (const filePath of files) { + const zipFile = zipContent.files[filePath]; + + if (zipFile.dir && excludeDirectories) { + continue; + } + + if (zipFile.dir) { + extractedFiles.set(filePath, ''); + continue; + } + + // Skip directories and files that shouldn't be processed + if (this.shouldSkipPath(filePath)) { + console.log(`Skipping filtered path: ${filePath}`); + continue; + } + + if (!this.isTextFile(filePath, allowedExtensions)) { + continue; + } + + const zipFileWithData = zipFile as JSZipObjectWithData; + if (zipFileWithData._data && zipFileWithData._data.uncompressedSize > maxFileSize) { + console.warn(`Skipping file ${filePath}: exceeds maximum file size (${maxFileSize} bytes)`); + continue; + } + + if (totalExtractedSize + (zipFileWithData._data?.uncompressedSize || 0) > maxTotalSize) { + console.warn(`Stopping extraction: total size would exceed maximum (${maxTotalSize} bytes)`); + break; + } + + try { + const content = await zipFile.async('text'); + + if (content.length > maxFileSize) { + console.warn(`Skipping file ${filePath}: content exceeds maximum file size`); + continue; + } + + extractedFiles.set(filePath, content); + totalExtractedSize += content.length; + + } catch (fileError) { + console.warn(`Failed to extract file ${filePath}:`, fileError); + continue; + } + } + + return extractedFiles; + + } catch (error) { + if (error instanceof Error) { + if (error.message.includes('corrupt') || error.message.includes('invalid')) { + throw new Error('ZIP file appears to be corrupted or invalid'); + } + throw new Error(`Failed to extract ZIP file: ${error.message}`); + } + throw new Error('Unknown error occurred while extracting ZIP file'); + } + } + + public async getZipFileInfo(file: File): Promise { + if (!file) { + throw new Error('No file provided'); + } + + try { + const arrayBuffer = await this.fileToArrayBuffer(file); + const zip = new JSZip(); + const zipContent = await zip.loadAsync(arrayBuffer); + + const fileInfos: ZipFileEntry[] = []; + + for (const [path, zipFile] of Object.entries(zipContent.files)) { + const file = zipFile as { dir: boolean; _data?: { uncompressedSize: number }; date?: Date }; + fileInfos.push({ + path, + content: '', // Don't load content for info request + isDirectory: file.dir, + size: file._data?.uncompressedSize || 0, + lastModified: file.date || new Date() + }); + } + + return fileInfos.sort((a, b) => a.path.localeCompare(b.path)); + + } catch (error) { + if (error instanceof Error) { + throw new Error(`Failed to read ZIP file info: ${error.message}`); + } + throw new Error('Unknown error occurred while reading ZIP file info'); + } + } + + public async extractSpecificFiles( + file: File, + filePaths: string[] + ): Promise> { + if (!file) { + throw new Error('No file provided'); + } + + if (!filePaths || filePaths.length === 0) { + throw new Error('No file paths specified'); + } + + try { + const arrayBuffer = await this.fileToArrayBuffer(file); + const zip = new JSZip(); + const zipContent = await zip.loadAsync(arrayBuffer); + + const extractedFiles = new Map(); + + for (const filePath of filePaths) { + const zipFile = zipContent.files[filePath]; + + if (!zipFile) { + console.warn(`File not found in ZIP: ${filePath}`); + continue; + } + + if (zipFile.dir) { + extractedFiles.set(filePath, ''); + continue; + } + + try { + const content = await zipFile.async('text'); + extractedFiles.set(filePath, content); + } catch (fileError) { + console.warn(`Failed to extract file ${filePath}:`, fileError); + continue; + } + } + + return extractedFiles; + + } catch (error) { + if (error instanceof Error) { + throw new Error(`Failed to extract specific files from ZIP: ${error.message}`); + } + throw new Error('Unknown error occurred while extracting specific files'); + } + } + + public isValidZipFile(file: File): boolean { + if (!file) return false; + + return ( + file.type === 'application/zip' || + file.type === 'application/x-zip-compressed' || + file.name.toLowerCase().endsWith('.zip') + ); + } + + private async fileToArrayBuffer(file: File): Promise { + return new Promise((resolve, reject) => { + const reader = new FileReader(); + + reader.onload = () => { + if (reader.result instanceof ArrayBuffer) { + resolve(reader.result); + } else { + reject(new Error('Failed to read file as ArrayBuffer')); + } + }; + + reader.onerror = () => { + reject(new Error('Error reading file')); + }; + + reader.readAsArrayBuffer(file); + }); + } + + private isTextFile(filePath: string, allowedExtensions: string[]): boolean { + if (!filePath || filePath.endsWith('/')) { + return false; + } + + const extension = filePath.toLowerCase().split('.').pop(); + return extension ? allowedExtensions.includes(`.${extension}`) : false; + } + + private shouldSkipPath(filePath: string): boolean { + // Skip directories that shouldn't be processed + if (this.shouldSkipDirectory(filePath)) { + return true; + } + + // Skip files that shouldn't be processed + if (!this.shouldIncludeFile(filePath)) { + return true; + } + + return false; + } + + private shouldSkipDirectory(path: string): boolean { + if (!path) return true; // Skip if path is undefined/null + + const skipDirs = [ + // Git version control + '.git', + // JavaScript dependencies (common in full-stack projects) + 'node_modules', + // Python bytecode cache + '__pycache__', + // Python virtual environments + 'venv', + 'env', + '.venv', + 'envs', + 'virtualenv', + // Build, distribution, and temporary directories + 'build', + 'dist', + 'docs', + 'logs', + 'tmp', + '.tmp', + // Additional common directories to skip + 'coverage', + '.coverage', + 'htmlcov', + 'vendor', + 'deps', + '_build', + '.gradle', + 'bin', + 'obj', + '.vs', + '.vscode', + '.idea', + 'temp' + ]; + + // Check each directory component in the path + const pathParts = path.split('/'); + for (const part of pathParts) { + const dirName = part.toLowerCase(); + + // Check for exact matches + if (skipDirs.includes(dirName) || dirName.startsWith('.')) { + return true; + } + + // Check for .egg-info directories + if (dirName.endsWith('.egg-info')) { + return true; + } + } + + // Check for virtual environment patterns anywhere in the path + const fullPathLower = path.toLowerCase(); + const venvPatterns = [ + '/.venv/', + '/venv/', + '/env/', + '/.env/', + '/envs/', + '/virtualenv/', + '/site-packages/', + '/lib/python', + '/lib64/python', + '/scripts/', + '/bin/python' + ]; + + if (venvPatterns.some(pattern => fullPathLower.includes(pattern))) { + return true; + } + + return false; + } + + private shouldIncludeFile(path: string): boolean { + if (!path) return false; // Skip if path is undefined/null + + const fileName = path.split('/').pop() || ''; + + // Skip hidden files except specific config files + if (fileName.startsWith('.') && !fileName.endsWith('.env.example')) { + return false; + } + + // Skip Python-specific file patterns + const skipPatterns = [ + // Python compiled bytecode + /\.pyc$/, + /\.pyo$/, + // Python extension modules (binary) + /\.pyd$/, + /\.so$/, + // Python packages + /\.egg$/, + /\.whl$/, + // Lock files + /\.lock$/, + /poetry\.lock$/, + /Pipfile\.lock$/, + // Editor swap files + /\..*\.swp$/, + /\..*\.swo$/, + // OS metadata files + /^Thumbs\.db$/, + /^\.DS_Store$/, + // General binary and archive files + /\.zip$/, + /\.tar$/, + /\.rar$/, + /\.7z$/, + /\.gz$/, + // Media files + /\.(jpg|jpeg|png|gif|bmp|svg|ico)$/i, + /\.(mp4|avi|mov|wmv|flv|webm)$/i, + /\.(mp3|wav|flac|aac|ogg)$/i, + // Document files + /\.(pdf|doc|docx|xls|xlsx|ppt|pptx)$/i, + // Other binary files + /\.(exe|dll|dylib)$/i, + // Minified files and source maps + /\.min\.(js|css)$/, + /\.map$/, + // Log and temporary files + /\.log$/, + /\.tmp$/, + /\.cache$/, + /\.pid$/, + /\.seed$/ + ]; + + if (skipPatterns.some(pattern => pattern.test(fileName))) { + return false; + } + + // Only include Python files and essential config files + const extension = '.' + fileName.split('.').pop()?.toLowerCase(); + + // Python source files + if (extension === '.py') { + return true; + } + + // Essential Python config files + const importantPythonFiles = [ + 'pyproject.toml', + 'setup.py', + 'requirements.txt', + 'setup.cfg', + 'tox.ini', + 'pytest.ini', + 'Pipfile', + 'poetry.toml', + 'README.md', + 'LICENSE', + 'CHANGELOG.md', + 'MANIFEST.in' + ]; + + if (importantPythonFiles.includes(fileName)) { + return true; + } + + return false; + } + + public getDefaultTextExtensions(): string[] { + return Array.from(ZipService.TEXT_EXTENSIONS); + } + + public async validateZipFile(file: File): Promise<{ valid: boolean; error?: string }> { + try { + if (!this.isValidZipFile(file)) { + return { valid: false, error: 'File is not a valid ZIP archive' }; + } + + const arrayBuffer = await this.fileToArrayBuffer(file); + const zip = new JSZip(); + await zip.loadAsync(arrayBuffer); + + return { valid: true }; + + } catch (error) { + return { + valid: false, + error: error instanceof Error ? error.message : 'Unknown validation error' + }; + } + } +} + +================ +File: src/ui/components/chat/ChatInterface.tsx +================ +import React, { useState, useRef, useEffect } from 'react'; +import type { KnowledgeGraph } from '../../../core/graph/types.ts'; +import { LLMService, type LLMProvider, type LLMConfig } from '../../../ai/llm-service.ts'; +import { CypherGenerator } from '../../../ai/cypher-generator.ts'; +import { RAGOrchestrator, type RAGResponse, type RAGOptions } from '../../../ai/orchestrator.ts'; + +interface ChatMessage { + id: string; + role: 'user' | 'assistant'; + content: string; + timestamp: Date; + metadata?: { + cypherQueries?: Array<{ cypher: string; explanation: string }>; + sources?: string[]; + confidence?: number; + reasoning?: Array<{ step: number; thought: string; action: string }>; + }; +} + +interface ChatInterfaceProps { + graph: KnowledgeGraph; + fileContents: Map; + className?: string; + style?: React.CSSProperties; +} + +interface LLMSettings { + provider: LLMProvider; + apiKey: string; + model: string; + temperature: number; + maxTokens: number; + // Azure OpenAI specific fields + azureOpenAIEndpoint?: string; + azureOpenAIDeploymentName?: string; + azureOpenAIApiVersion?: string; +} + +const ChatInterface: React.FC = ({ + graph, + fileContents, + className = '', + style = {} +}) => { + const [messages, setMessages] = useState([]); + const [inputValue, setInputValue] = useState(''); + const [isLoading, setIsLoading] = useState(false); + const [showSettings, setShowSettings] = useState(false); + const [showReasoning, setShowReasoning] = useState(false); + + // LLM Configuration + const [llmSettings, setLLMSettings] = useState({ + provider: 'openai', + apiKey: '', + model: 'gpt-4o-mini', + temperature: 0.1, + maxTokens: 4000, + azureOpenAIEndpoint: '', + azureOpenAIDeploymentName: '', + azureOpenAIApiVersion: '2024-02-01' + }); + + // Services + const [llmService] = useState(new LLMService()); + const [cypherGenerator] = useState(new CypherGenerator(llmService)); + const [ragOrchestrator] = useState(new RAGOrchestrator(llmService, cypherGenerator)); + + const messagesEndRef = useRef(null); + const inputRef = useRef(null); + + // Initialize RAG context when graph or fileContents change + useEffect(() => { + ragOrchestrator.setContext({ graph, fileContents }); + }, [graph, fileContents, ragOrchestrator]); + + // Load settings from localStorage on mount + useEffect(() => { + const savedProvider = localStorage.getItem('llm_provider') as LLMProvider; + const savedApiKey = localStorage.getItem('llm_api_key'); + const savedAzureEndpoint = localStorage.getItem('azure_openai_endpoint'); + const savedAzureDeployment = localStorage.getItem('azure_openai_deployment'); + const savedAzureApiVersion = localStorage.getItem('azure_openai_api_version'); + + if (savedProvider || savedApiKey || savedAzureEndpoint) { + setLLMSettings(prev => ({ + ...prev, + provider: savedProvider || prev.provider, + apiKey: savedApiKey || prev.apiKey, + azureOpenAIEndpoint: savedAzureEndpoint || prev.azureOpenAIEndpoint, + azureOpenAIDeploymentName: savedAzureDeployment || prev.azureOpenAIDeploymentName, + azureOpenAIApiVersion: savedAzureApiVersion || prev.azureOpenAIApiVersion, + // For Azure OpenAI, use deployment name as model, otherwise use default model + model: savedProvider === 'azure-openai' + ? (savedAzureDeployment || 'gpt-4.1-mini-v2') + : (savedProvider ? llmService.getAvailableModels(savedProvider)[0] : prev.model) + })); + } + }, [llmService]); + + // Auto-scroll to bottom when new messages arrive + useEffect(() => { + messagesEndRef.current?.scrollIntoView({ behavior: 'smooth' }); + }, [messages]); + + // Handle form submission + const handleSubmit = async (e: React.FormEvent) => { + e.preventDefault(); + if (!inputValue.trim() || isLoading) return; + + // Validate API key + if (!llmSettings.apiKey.trim()) { + alert('Please configure your API key in settings'); + setShowSettings(true); + return; + } + + if (!llmService.validateApiKey(llmSettings.provider, llmSettings.apiKey)) { + alert('Invalid API key format. Please check your settings.'); + setShowSettings(true); + return; + } + + // Additional validation for Azure OpenAI + if (llmSettings.provider === 'azure-openai') { + if (!llmSettings.azureOpenAIEndpoint?.trim()) { + alert('Please configure your Azure OpenAI endpoint in settings'); + setShowSettings(true); + return; + } + if (!llmSettings.azureOpenAIDeploymentName?.trim()) { + alert('Please configure your Azure OpenAI deployment name in settings'); + setShowSettings(true); + return; + } + } + + const userMessage: ChatMessage = { + id: generateId(), + role: 'user', + content: inputValue.trim(), + timestamp: new Date() + }; + + setMessages(prev => [...prev, userMessage]); + setInputValue(''); + setIsLoading(true); + + try { + const llmConfig: LLMConfig = { + provider: llmSettings.provider, + apiKey: llmSettings.apiKey, + model: llmSettings.model, + temperature: llmSettings.temperature, + maxTokens: llmSettings.maxTokens, + // Azure OpenAI specific fields + azureOpenAIEndpoint: llmSettings.azureOpenAIEndpoint, + azureOpenAIDeploymentName: llmSettings.azureOpenAIDeploymentName, + azureOpenAIApiVersion: llmSettings.azureOpenAIApiVersion + }; + + const ragOptions: RAGOptions = { + maxReasoningSteps: 5, + includeReasoning: showReasoning, + strictMode: false, + temperature: llmSettings.temperature + }; + + const response: RAGResponse = await ragOrchestrator.answerQuestion( + userMessage.content, + llmConfig, + ragOptions + ); + + const assistantMessage: ChatMessage = { + id: generateId(), + role: 'assistant', + content: response.answer, + timestamp: new Date(), + metadata: { + cypherQueries: response.cypherQueries.map(q => ({ + cypher: q.cypher, + explanation: q.explanation + })), + sources: response.sources, + confidence: response.confidence, + reasoning: showReasoning ? response.reasoning.map(r => ({ + step: r.step, + thought: r.thought, + action: r.action + })) : undefined + } + }; + + setMessages(prev => [...prev, assistantMessage]); + + } catch (error) { + const errorMessage: ChatMessage = { + id: generateId(), + role: 'assistant', + content: `I apologize, but I encountered an error while processing your question: ${error instanceof Error ? error.message : 'Unknown error'}`, + timestamp: new Date() + }; + + setMessages(prev => [...prev, errorMessage]); + } finally { + setIsLoading(false); + } + }; + + // Handle key press in textarea + const handleKeyPress = (e: React.KeyboardEvent) => { + if (e.key === 'Enter' && !e.shiftKey) { + e.preventDefault(); + handleSubmit(e as any); + } + }; + + // Clear conversation + const clearConversation = () => { + setMessages([]); + }; + + // Generate unique ID + const generateId = () => { + return `msg_${Date.now()}_${Math.random().toString(36).substr(2, 9)}`; + }; + + // Get available models for current provider + const getAvailableModels = () => { + return llmService.getAvailableModels(llmSettings.provider); + }; + + const containerStyle: React.CSSProperties = { + display: 'flex', + flexDirection: 'column', + height: '600px', + border: '1px solid #ddd', + borderRadius: '8px', + backgroundColor: '#fff', + ...style + }; + + const headerStyle: React.CSSProperties = { + padding: '16px', + borderBottom: '1px solid #eee', + backgroundColor: '#f8f9fa', + borderRadius: '8px 8px 0 0', + display: 'flex', + justifyContent: 'space-between', + alignItems: 'center' + }; + + const messagesStyle: React.CSSProperties = { + flex: 1, + overflowY: 'auto', + padding: '16px', + display: 'flex', + flexDirection: 'column', + gap: '16px' + }; + + const inputAreaStyle: React.CSSProperties = { + padding: '16px', + borderTop: '1px solid #eee' + }; + + const messageStyle = (role: 'user' | 'assistant'): React.CSSProperties => ({ + padding: '12px 16px', + borderRadius: '12px', + maxWidth: '80%', + alignSelf: role === 'user' ? 'flex-end' : 'flex-start', + backgroundColor: role === 'user' ? '#007bff' : '#f1f3f4', + color: role === 'user' ? '#fff' : '#333', + wordWrap: 'break-word' + }); + + const buttonStyle: React.CSSProperties = { + padding: '8px 16px', + border: 'none', + borderRadius: '4px', + backgroundColor: '#007bff', + color: '#fff', + cursor: 'pointer', + fontSize: '14px' + }; + + const textareaStyle: React.CSSProperties = { + width: '100%', + minHeight: '60px', + padding: '12px', + border: '1px solid #ddd', + borderRadius: '4px', + resize: 'vertical', + fontSize: '14px', + fontFamily: 'inherit' + }; + + return ( +
+ {/* Header */} +
+
+ 💬 + Code Assistant + + {llmService.getProviderDisplayName(llmSettings.provider)} + +
+ +
+ + + +
+
+ + {/* Settings Panel */} + {showSettings && ( +
+

LLM Configuration

+ +
+
+ + +
+ +
+ + {llmSettings.provider === 'azure-openai' ? ( + setLLMSettings(prev => ({ ...prev, model: e.target.value }))} + placeholder="gpt-4.1-mini-v2" + style={{ width: '100%', padding: '6px', fontSize: '14px' }} + /> + ) : ( + + )} +
+
+ +
+ + setLLMSettings(prev => ({ ...prev, apiKey: e.target.value }))} + placeholder={ + llmSettings.provider === 'azure-openai' ? 'Your Azure OpenAI key...' : + llmSettings.provider === 'anthropic' ? 'sk-ant-...' : + llmSettings.provider === 'gemini' ? 'Your Google API key...' : 'sk-...' + } + style={{ width: '100%', padding: '6px', fontSize: '14px' }} + /> +
+ + {/* Azure OpenAI Specific Fields */} + {llmSettings.provider === 'azure-openai' && ( + <> +
+ + setLLMSettings(prev => ({ ...prev, azureOpenAIEndpoint: e.target.value }))} + placeholder="https://your-resource.openai.azure.com" + style={{ width: '100%', padding: '6px', fontSize: '14px' }} + /> +
+ +
+ + setLLMSettings(prev => ({ ...prev, azureOpenAIApiVersion: e.target.value }))} + placeholder="2024-02-01" + style={{ width: '100%', padding: '6px', fontSize: '14px' }} + /> +
+ + )} + +
+
+ + setLLMSettings(prev => ({ ...prev, temperature: parseFloat(e.target.value) }))} + style={{ width: '100%' }} + /> +
+ +
+ + setLLMSettings(prev => ({ ...prev, maxTokens: parseInt(e.target.value) }))} + style={{ width: '100%', padding: '6px', fontSize: '14px' }} + /> +
+
+ + {/* Save Button */} +
+ + +
+
+ )} + + {/* Messages */} +
+ {messages.length === 0 && ( +
+
💬
+
Ask me anything about the codebase!
+
+ I can help you understand functions, classes, dependencies, and more. +
+
+ )} + + {messages.map((message) => ( +
+
+
+ {message.content} +
+ + {/* Metadata */} + {message.metadata && ( +
+ {message.metadata.confidence && ( +
+ Confidence: {Math.round(message.metadata.confidence * 100)}% +
+ )} + + {message.metadata.sources && message.metadata.sources.length > 0 && ( +
+ Sources: {message.metadata.sources.join(', ')} +
+ )} + + {message.metadata.cypherQueries && message.metadata.cypherQueries.length > 0 && ( +
+ View Queries ({message.metadata.cypherQueries.length}) + {message.metadata.cypherQueries.map((query, index) => ( +
+
Query: {query.cypher}
+
Explanation: {query.explanation}
+
+ ))} +
+ )} + + {message.metadata.reasoning && message.metadata.reasoning.length > 0 && ( +
+ View Reasoning ({message.metadata.reasoning.length} steps) + {message.metadata.reasoning.map((step, index) => ( +
+
Step {step.step}: {step.thought}
+
Action: {step.action}
+
+ ))} +
+ )} +
+ )} +
+ +
+ {message.timestamp.toLocaleTimeString()} +
+
+ ))} + + {isLoading && ( +
+
+
+ Thinking... +
+
+ )} + +
+
+ + {/* Input Area */} +
+
+