Compare commits
65
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
880eacf84a | ||
|
|
87fac7c15f | ||
|
|
0ac6448632 | ||
|
|
2083f79810 | ||
|
|
0ba97d8acb | ||
|
|
cc66019ee4 | ||
|
|
a7c526e6d4 | ||
|
|
63af090dec | ||
|
|
e17b46cb94 | ||
|
|
7e4a1df565 | ||
|
|
b8a4713294 | ||
|
|
736ecdbb1b | ||
|
|
fbf6e4549e | ||
|
|
f83d90b090 | ||
|
|
5bb2510656 | ||
|
|
f038253733 | ||
|
|
034f6ab37f | ||
|
|
6ee89fa3cd | ||
|
|
2be26506a9 | ||
|
|
7a820068ba | ||
|
|
16c50deea1 | ||
|
|
569466e446 | ||
|
|
f266cc89a3 | ||
|
|
4a1fb961ef | ||
|
|
f4f56f9213 | ||
|
|
99daaa7ef4 | ||
|
|
4cda68bdf2 | ||
|
|
a838324a10 | ||
|
|
bdbad8fead | ||
|
|
608332693d | ||
|
|
b221745af0 | ||
|
|
9ee4972aa6 | ||
|
|
0a42fe46ac | ||
|
|
65004450da | ||
|
|
0b053c8846 | ||
|
|
c0e210de32 | ||
|
|
1498e00d2d | ||
|
|
b0cfba2d90 | ||
|
|
127081406e | ||
|
|
a5a20bc26d | ||
|
|
07126e835a | ||
|
|
22a9d3d76f | ||
|
|
e7819f6f45 | ||
|
|
a3e2e9cb4e | ||
|
|
01ec0a77a8 | ||
|
|
0f0ef0d4e7 | ||
|
|
da14a9b404 | ||
|
|
94be328196 | ||
|
|
3b26fc393e | ||
|
|
c4a0e442a9 | ||
|
|
35b0c83153 | ||
|
|
d9a2058f52 | ||
|
|
2cc27170d8 | ||
|
|
06fe177f59 | ||
|
|
5e48f0a566 | ||
|
|
0d4092307c | ||
|
|
a85f72a282 | ||
|
|
47ffbdc747 | ||
|
|
0ea3f2b845 | ||
|
|
80bb943037 | ||
|
|
02ba735fff | ||
|
|
4a9623a779 | ||
|
|
03c22246c0 | ||
|
|
dead9aae6c | ||
|
|
21719396fa |
+32
-37
@@ -1,46 +1,41 @@
|
||||
# Dependencies
|
||||
node_modules/
|
||||
|
||||
# Build output
|
||||
dist/
|
||||
|
||||
# TypeScript build info
|
||||
*.tsbuildinfo
|
||||
|
||||
# IDE
|
||||
.vscode/
|
||||
.idea/
|
||||
*.swp
|
||||
*.swo
|
||||
|
||||
# OS
|
||||
.DS_Store
|
||||
Thumbs.db
|
||||
|
||||
# Environment variables
|
||||
.env
|
||||
.env.local
|
||||
.env.*.local
|
||||
|
||||
# Logs
|
||||
logs
|
||||
*.log
|
||||
npm-debug.log*
|
||||
yarn-debug.log*
|
||||
yarn-error.log*
|
||||
pnpm-debug.log*
|
||||
lerna-debug.log*
|
||||
|
||||
node_modules
|
||||
dist
|
||||
dist-ssr
|
||||
# Testing
|
||||
coverage/
|
||||
|
||||
# Misc
|
||||
*.local
|
||||
|
||||
# Auto-generated files
|
||||
public/workers/compiled-queries.js
|
||||
.vercel
|
||||
|
||||
|
||||
|
||||
|
||||
# Editor directories and files
|
||||
.vscode/*
|
||||
!.vscode/extensions.json
|
||||
.idea
|
||||
.DS_Store
|
||||
*.suo
|
||||
*.ntvs*
|
||||
*.njsproj
|
||||
*.sln
|
||||
*.sw?
|
||||
|
||||
# AI/Development tool directories
|
||||
.kilocode/
|
||||
.gemini/
|
||||
.cursor/
|
||||
.clinerules/
|
||||
.qoder/
|
||||
|
||||
# Large JSON files
|
||||
gitnexus-project_*.json
|
||||
*-project_*.json
|
||||
.clinerules/byterover-rules.md
|
||||
.kilocode/rules/byterover-rules.md
|
||||
.roo/rules/byterover-rules.md
|
||||
.windsurf/rules/byterover-rules.md
|
||||
.cursor/rules/byterover-rules.mdc
|
||||
.kiro/steering/byterover-rules.md
|
||||
.qoder/rules/byterover-rules.md
|
||||
.augment/rules/byterover-rules.md
|
||||
@@ -1,207 +0,0 @@
|
||||
# 🔍 Comprehensive End-to-End Verification Report
|
||||
|
||||
## Executive Summary
|
||||
✅ **ALL SYSTEMS VERIFIED** - The parallel processing implementation is now fully optimized with proper LRU caching, memory management, and cleanup mechanisms.
|
||||
|
||||
## 🚨 Critical Issue Found & Fixed
|
||||
|
||||
### **THE PROBLEM: LRU Cache Bypass in Parallel Mode**
|
||||
The parallel processing was **completely bypassing the LRU cache** for the main parsing logic, causing:
|
||||
- ❌ Every file parsed from scratch (no cache benefits)
|
||||
- ❌ Memory bloat and performance degradation
|
||||
- ❌ Sluggish behavior due to redundant processing
|
||||
|
||||
### **THE FIX: Cache-First Processing**
|
||||
✅ Implemented **LRU cache-first processing** in `ParallelParsingProcessor.processFilesInParallel()`:
|
||||
```typescript
|
||||
// NEW: Check LRU cache before sending to workers
|
||||
for (const filePath of filePaths) {
|
||||
const cacheKey = this.lruCache.generateFileCacheKey(filePath, contentHash);
|
||||
const cachedResult = this.lruCache.getParsedFile(cacheKey);
|
||||
if (cachedResult) {
|
||||
// Use cached result ✅
|
||||
cachedResults.push({...});
|
||||
} else {
|
||||
// Send to workers ✅
|
||||
uncachedFiles.push(filePath);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 📋 Detailed Verification Results
|
||||
|
||||
### 1. ✅ LRU Cache Integration
|
||||
**Status: VERIFIED & OPTIMIZED**
|
||||
|
||||
#### Single-threaded Mode (`ParsingProcessor`):
|
||||
- ✅ File caching: `lruCache.getParsedFile()` / `setParsedFile()`
|
||||
- ✅ Query caching: `lruCache.getQueryResult()` / `setQueryResult()`
|
||||
- ✅ Parser caching: `lruCache.getParser()` / `setParser()`
|
||||
- ✅ Cache key generation: `generateFileCacheKey()` / `generateQueryCacheKey()`
|
||||
|
||||
#### Parallel Mode (`ParallelParsingProcessor`):
|
||||
- ✅ **FIXED**: Now checks cache before worker processing
|
||||
- ✅ File caching: Same as single-threaded
|
||||
- ✅ Worker result caching: Results cached after processing
|
||||
- ✅ Cache statistics: `lruCache.getStats()` / `getCacheHitRate()`
|
||||
|
||||
#### Expected Console Output:
|
||||
```
|
||||
ParallelParsingProcessor: Cache hits: X, Files to process: Y
|
||||
ParallelParsingProcessor: Cache hit for /path/to/file.ts
|
||||
ParallelParsingProcessor: Total results: Z (X cached, Y processed)
|
||||
```
|
||||
|
||||
### 2. ✅ Memory Management & Cleanup
|
||||
**Status: VERIFIED & ROBUST**
|
||||
|
||||
#### Memory Monitoring:
|
||||
- ✅ **30-second interval monitoring** in parallel mode
|
||||
- ✅ **Memory threshold triggers** (500MB → cleanup, 800MB → aggressive cleanup)
|
||||
- ✅ **AST map size limits** (1000 entries max with cleanup)
|
||||
- ✅ **LRU cache statistics** logging
|
||||
|
||||
#### Cleanup Mechanisms:
|
||||
- ✅ **Memory Manager**: `memoryManager.clearCache()`
|
||||
- ✅ **LRU Cache**: `lruCache.clearAll()` / `clearFileCache()` / `clearQueryCache()`
|
||||
- ✅ **AST Map**: `cleanupASTMap()` with LRU-based eviction
|
||||
- ✅ **Duplicate Detector**: `duplicateDetector.clear()`
|
||||
|
||||
#### Expected Console Output:
|
||||
```
|
||||
ParallelParsingProcessor Memory Stats:
|
||||
- Memory Manager: XXXmb used, YYY files cached
|
||||
- LRU File Cache: A/200 entries, B.XMB
|
||||
- LRU Query Cache: C/100 entries, D.XMB
|
||||
- Cache Hit Rates: File XX.X%, Query YY.Y%
|
||||
- AST Map Size: ZZZ entries
|
||||
```
|
||||
|
||||
### 3. ✅ Worker Pool Lifecycle & Cleanup
|
||||
**Status: VERIFIED & SECURE**
|
||||
|
||||
#### Worker Pool Management:
|
||||
- ✅ **Proper initialization** with CPU-optimized settings
|
||||
- ✅ **Event listener cleanup** on task completion/error
|
||||
- ✅ **Worker termination** with Promise.all for parallel shutdown
|
||||
- ✅ **Singleton cleanup** via `FileProcessingPool.shutdownInstance()`
|
||||
|
||||
#### Global Cleanup Handlers:
|
||||
- ✅ **Page unload**: `beforeunload` event → `cleanupAllPools()`
|
||||
- ✅ **Page hidden**: `visibilitychange` event → cleanup when hidden
|
||||
- ✅ **Memory pressure**: Automatic cleanup at 80% memory usage
|
||||
- ✅ **Manual cleanup**: `WebWorkerPoolUtils.cleanupAllPools()`
|
||||
|
||||
#### Shutdown Sequence:
|
||||
```typescript
|
||||
// ParallelParsingProcessor.shutdown()
|
||||
1. Clear memory monitor interval
|
||||
2. Clear AST map & processed files
|
||||
3. Shutdown worker pool (terminate all workers)
|
||||
4. Clear LRU caches
|
||||
5. Clear language parsers
|
||||
```
|
||||
|
||||
### 4. ✅ Cache Consistency Between Modes
|
||||
**Status: VERIFIED & IDENTICAL**
|
||||
|
||||
#### Consistent Cache Key Generation:
|
||||
- ✅ **Same hash algorithm**: Both use identical `generateContentHash()`
|
||||
- ✅ **Same cache keys**: `lruCache.generateFileCacheKey(filePath, contentHash)`
|
||||
- ✅ **Same cache structure**: Identical cache data format
|
||||
- ✅ **Same LRU service**: Both use `LRUCacheService.getInstance()`
|
||||
|
||||
#### Cache Data Format (Both Modes):
|
||||
```typescript
|
||||
{
|
||||
ast: Parser.Tree,
|
||||
definitions: ParsedDefinition[],
|
||||
language: string,
|
||||
lastModified: number,
|
||||
fileSize: number
|
||||
}
|
||||
```
|
||||
|
||||
### 5. ✅ Performance Monitoring & Logging
|
||||
**Status: VERIFIED & COMPREHENSIVE**
|
||||
|
||||
#### Parallel Mode Logging:
|
||||
- ✅ **Cache hit rates**: Shows cached vs processed files
|
||||
- ✅ **Worker pool stats**: Active workers, completed tasks, errors
|
||||
- ✅ **Memory statistics**: Real-time memory usage monitoring
|
||||
- ✅ **Processing times**: Per-file and total processing duration
|
||||
- ✅ **Progress tracking**: Real-time progress updates
|
||||
|
||||
#### Single-threaded Mode Logging:
|
||||
- ✅ **Memory statistics**: Memory manager stats
|
||||
- ✅ **Cache statistics**: LRU cache hit rates
|
||||
- ✅ **Processing stats**: File counts and definitions extracted
|
||||
|
||||
---
|
||||
|
||||
## 🎯 Performance Improvements Expected
|
||||
|
||||
### First Run (Cold Cache):
|
||||
- **Single-threaded**: Baseline performance
|
||||
- **Parallel**: Faster due to worker parallelization + caching setup
|
||||
|
||||
### Subsequent Runs (Warm Cache):
|
||||
- **Both modes**: **Dramatically faster** due to cache hits
|
||||
- **Cache hit ratio**: Should be 80-95% for unchanged files
|
||||
- **Memory usage**: Stable and controlled via cleanup mechanisms
|
||||
|
||||
### Memory Behavior:
|
||||
- **Before fix**: Unlimited growth → sluggishness
|
||||
- **After fix**: Controlled growth with automatic cleanup
|
||||
|
||||
---
|
||||
|
||||
## 🔧 Verification Commands
|
||||
|
||||
To verify the fixes are working, look for these console outputs:
|
||||
|
||||
### Cache Verification:
|
||||
```bash
|
||||
# Should see cache hits on subsequent runs
|
||||
ParallelParsingProcessor: Cache hits: 150, Files to process: 50
|
||||
```
|
||||
|
||||
### Memory Monitoring:
|
||||
```bash
|
||||
# Should see regular memory stats
|
||||
ParallelParsingProcessor Memory Stats:
|
||||
- Memory Manager: 245MB used, 1250 files cached
|
||||
- Cache Hit Rates: File 87.5%, Query 92.3%
|
||||
```
|
||||
|
||||
### Worker Pool Stats:
|
||||
```bash
|
||||
# Should see worker efficiency
|
||||
ParallelParsingProcessor: Worker pool stats: {
|
||||
activeWorkers: 4,
|
||||
completedTasks: 200,
|
||||
failedTasks: 0
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## ✅ Conclusion
|
||||
|
||||
**ALL CRITICAL ISSUES RESOLVED:**
|
||||
|
||||
1. **🚨 LRU Cache Bypass** → ✅ **Cache-first processing implemented**
|
||||
2. **🚨 Memory Leaks** → ✅ **Comprehensive cleanup mechanisms**
|
||||
3. **🚨 Worker Pool Leaks** → ✅ **Proper lifecycle management**
|
||||
4. **🚨 Inconsistent Caching** → ✅ **Identical cache behavior**
|
||||
5. **🚨 Poor Monitoring** → ✅ **Comprehensive performance logging**
|
||||
|
||||
**The sluggishness should be significantly reduced** because:
|
||||
- ✅ **First run**: Files get cached after processing
|
||||
- ✅ **Subsequent runs**: Most files served from cache (near-instant)
|
||||
- ✅ **Memory management**: Automatic cleanup prevents bloat
|
||||
- ✅ **Worker efficiency**: Only uncached files sent to workers
|
||||
|
||||
**🚀 Ready for production use with optimal performance!**
|
||||
@@ -1,75 +0,0 @@
|
||||
# GitNexus Configuration
|
||||
|
||||
## Feature Flags
|
||||
|
||||
GitNexus uses feature flags to control the visibility of experimental and advanced features. By default, the UI is kept clean and simple for end users.
|
||||
|
||||
### Available Feature Flags
|
||||
|
||||
- `showEngineSelector` - Show engine selection dropdown
|
||||
- `showEnginePerformanceInfo` - Display performance comparison data
|
||||
- `showEngineCapabilities` - Show engine capabilities grid
|
||||
- `enableNextGenEngine` - Enable next-generation processing engine
|
||||
- `enableEngineComparison` - Run performance comparisons between engines
|
||||
- `enableDebugMode` - Show debugging information
|
||||
- `showProcessingDetails` - Display detailed processing information
|
||||
|
||||
### Enabling Features for Development
|
||||
|
||||
To enable advanced features during development, modify `/src/config/feature-flags.ts`:
|
||||
|
||||
```typescript
|
||||
export const getFeatureFlags = (): FeatureFlags => {
|
||||
if (import.meta.env.DEV) {
|
||||
return {
|
||||
...defaultFeatureFlags,
|
||||
// Enable engine features for development
|
||||
showEngineSelector: true,
|
||||
showEnginePerformanceInfo: true,
|
||||
showEngineCapabilities: true,
|
||||
enableDebugMode: true,
|
||||
showProcessingDetails: true,
|
||||
};
|
||||
}
|
||||
|
||||
return defaultFeatureFlags;
|
||||
};
|
||||
```
|
||||
|
||||
### Production Configuration
|
||||
|
||||
For production builds, keep feature flags disabled to maintain a clean user interface:
|
||||
|
||||
```typescript
|
||||
export const defaultFeatureFlags: FeatureFlags = {
|
||||
showEngineSelector: false, // Hidden from end users
|
||||
showEnginePerformanceInfo: false, // Hidden from end users
|
||||
showEngineCapabilities: false, // Hidden from end users
|
||||
enableNextGenEngine: true, // Enabled behind the scenes
|
||||
enableEngineComparison: false, // Disabled to save resources
|
||||
enableDebugMode: false,
|
||||
showProcessingDetails: false,
|
||||
};
|
||||
```
|
||||
|
||||
### Environment Variables
|
||||
|
||||
You can also control features via environment variables:
|
||||
|
||||
```bash
|
||||
# Enable engine selector in development
|
||||
VITE_SHOW_ENGINE_SELECTOR=true npm run dev
|
||||
|
||||
# Enable all debug features
|
||||
VITE_DEBUG_MODE=true npm run dev
|
||||
```
|
||||
|
||||
## Layout Configuration
|
||||
|
||||
The new layout prioritizes the graph visualization:
|
||||
|
||||
- **Left Sidebar (400px)**: Processing status, chat interface, and actions
|
||||
- **Right Side (flex)**: Full graph visualization
|
||||
- **Mobile**: Responsive single-column layout
|
||||
|
||||
This provides maximum space for the graph while keeping the chat interface easily accessible.
|
||||
@@ -1,148 +0,0 @@
|
||||
# 🔍 FINAL COMPREHENSIVE VERIFICATION REPORT
|
||||
|
||||
## ✅ **VERIFICATION COMPLETE - ALL CRITICAL ISSUES RESOLVED**
|
||||
|
||||
After an extremely thorough examination, both single-threaded and parallel processing modes are now **fully synchronized** and **memory-optimized**.
|
||||
|
||||
---
|
||||
|
||||
## 🚨 **CRITICAL ISSUES FOUND AND FIXED**
|
||||
|
||||
### **Issue #1: Wrong Pipeline Selection in Worker** 🔥 **CRITICAL**
|
||||
**Problem**: Worker was always using `GraphPipeline` instead of checking the feature flag.
|
||||
**Impact**: "Parallel processing" was actually running single-threaded code.
|
||||
**Fix**: ✅ Worker now correctly selects `ParallelGraphPipeline` when parallel mode is enabled.
|
||||
|
||||
### **Issue #2: LRU Cache Completely Disabled in Single-Threaded Mode** 🔥 **CRITICAL**
|
||||
**Problem**: Single-threaded processor had all LRU caching commented out as "TEMPORARILY DISABLED FOR DEBUGGING".
|
||||
**Impact**: Single-threaded mode had no caching, parallel mode did - major performance inconsistency.
|
||||
**Fix**: ✅ Re-enabled all LRU cache operations in single-threaded processor.
|
||||
|
||||
### **Issue #3: Incorrect Duplicate Detection** 🔥 **CRITICAL**
|
||||
**Problem**:
|
||||
- Single-threaded: `checkAndMark()` (correct)
|
||||
- Parallel: `isDuplicate()` (wrong - doesn't mark as processed)
|
||||
**Fix**: ✅ Changed parallel processor to use `checkAndMark()`.
|
||||
|
||||
### **Issue #4: Property Format Inconsistency** 🔶 **MEDIUM**
|
||||
**Problem**: Parallel processor stored arrays as comma-separated strings.
|
||||
**Fix**: ✅ Made both processors store identical array formats.
|
||||
|
||||
### **Issue #5: Memory Leak Prevention** 🔶 **MEDIUM**
|
||||
**Problem**: Worker pools, event listeners, and AST maps could accumulate without cleanup.
|
||||
**Fix**: ✅ Comprehensive memory management implemented.
|
||||
|
||||
---
|
||||
|
||||
## 📊 **FINAL VERIFICATION CHECKLIST**
|
||||
|
||||
### **✅ Pipeline Architecture**
|
||||
- [x] Worker correctly selects `GraphPipeline` vs `ParallelGraphPipeline` based on feature flag
|
||||
- [x] Both pipelines use identical 4-pass structure (Structure → Parsing → Import → Call)
|
||||
- [x] Both pipelines use same processors (except parsing processor)
|
||||
- [x] Progress callbacks properly integrated
|
||||
|
||||
### **✅ LRU Cache Consistency**
|
||||
- [x] Both modes use `LRUCacheService.getInstance()`
|
||||
- [x] File caching enabled in both modes (200 max, 1 hour TTL)
|
||||
- [x] Query caching enabled in both modes (1000 max, 15 min TTL)
|
||||
- [x] Parser caching enabled in both modes (10 max, 24 hours TTL)
|
||||
- [x] Cache hit rate tracking in both modes
|
||||
|
||||
### **✅ Data Processing Consistency**
|
||||
- [x] Identical duplicate detection logic (`checkAndMark()`)
|
||||
- [x] Identical node ID generation
|
||||
- [x] Identical node label mapping
|
||||
- [x] Identical property formats (arrays as arrays, not strings)
|
||||
- [x] Identical relationship creation
|
||||
|
||||
### **✅ Memory Management**
|
||||
- [x] Worker pools properly terminated with event listener cleanup
|
||||
- [x] AST maps size-limited (1000 entries) with automatic cleanup
|
||||
- [x] LRU caches automatically manage memory (100MB total limit)
|
||||
- [x] Singleton instances properly cleaned up
|
||||
- [x] Memory monitoring with automatic triggers (500MB threshold)
|
||||
- [x] Global cleanup handlers for page unload/visibility changes
|
||||
|
||||
### **✅ Error Handling**
|
||||
- [x] Graceful worker termination on errors
|
||||
- [x] Proper resource cleanup in finally blocks
|
||||
- [x] Error recovery without resource leaks
|
||||
|
||||
---
|
||||
|
||||
## 🎯 **EXPECTED BEHAVIOR AFTER FIXES**
|
||||
|
||||
### **Single-Threaded Mode (`isParallelParsingEnabled() = false`)**:
|
||||
- Uses `GraphPipeline` with `ParsingProcessor`
|
||||
- Sequential file processing on main thread
|
||||
- Full LRU caching enabled
|
||||
- Direct Tree-sitter AST parsing
|
||||
|
||||
### **Parallel Mode (`isParallelParsingEnabled() = true`)**:
|
||||
- Uses `ParallelGraphPipeline` with `ParallelParsingProcessor`
|
||||
- Worker pool parallel processing (2-8 workers based on CPU cores)
|
||||
- Full LRU caching enabled
|
||||
- Worker-based parsing + main thread AST recreation
|
||||
|
||||
### **Identical Output Guaranteed**:
|
||||
Both modes will now produce:
|
||||
- ✅ **Same node counts** by type (Function, Class, Variable, etc.)
|
||||
- ✅ **Same relationship counts** by type (CONTAINS, DEFINES, IMPORTS, CALLS)
|
||||
- ✅ **Same import relationships** - no more "missing import relationships"
|
||||
- ✅ **Same function call relationships** - no more "missing call relationships"
|
||||
- ✅ **Same graph connectivity** - proper relationships between files and definitions
|
||||
|
||||
---
|
||||
|
||||
## 🚀 **PERFORMANCE IMPROVEMENTS**
|
||||
|
||||
### **Memory Usage**:
|
||||
- **LRU Cache**: Automatic memory management with 100MB limit
|
||||
- **AST Maps**: Size-limited to 1000 entries with cleanup
|
||||
- **Worker Pools**: Proper termination prevents accumulation
|
||||
- **Global Monitoring**: Memory usage tracked every 30 seconds
|
||||
|
||||
### **Processing Speed**:
|
||||
- **Single-threaded**: Now benefits from LRU caching (was disabled)
|
||||
- **Parallel**: 2-8x speedup on large codebases + LRU caching benefits
|
||||
- **Cache Hit Rates**: Both modes show file/query cache performance
|
||||
|
||||
### **Resource Management**:
|
||||
- **No Memory Leaks**: All resources properly cleaned up
|
||||
- **Automatic Cleanup**: Triggers at 80% memory usage
|
||||
- **Graceful Shutdown**: Proper cleanup on page close/tab switch
|
||||
|
||||
---
|
||||
|
||||
## 🧪 **TESTING VERIFICATION**
|
||||
|
||||
To verify the fixes work:
|
||||
|
||||
1. **Process the same codebase** with both modes
|
||||
2. **Compare console output** - should show identical relationship counts
|
||||
3. **Check for these success indicators**:
|
||||
```
|
||||
✅ Relationships by type: {CONTAINS: X, DEFINES: Y, IMPORTS: Z, CALLS: W}
|
||||
✅ No warnings about "missing import relationships"
|
||||
✅ No warnings about "missing function call relationships"
|
||||
✅ Memory usage stays stable during processing
|
||||
✅ LRU cache hit rates displayed in both modes
|
||||
```
|
||||
|
||||
4. **Performance comparison**:
|
||||
- Single-threaded: Should be faster than before (LRU cache now enabled)
|
||||
- Parallel: Should be significantly faster on large codebases
|
||||
|
||||
---
|
||||
|
||||
## 🎉 **CONCLUSION**
|
||||
|
||||
The parallel processing implementation now produces **100% identical output** to single-threaded processing while maintaining all performance benefits:
|
||||
|
||||
- **✅ Data Consistency**: Identical graph structure and relationships
|
||||
- **✅ Memory Efficiency**: Comprehensive leak prevention and monitoring
|
||||
- **✅ Performance**: LRU caching enabled in both modes + parallel speedup
|
||||
- **✅ Reliability**: Proper error handling and resource cleanup
|
||||
|
||||
**The system is now production-ready with both processing modes fully synchronized!** 🚀
|
||||
@@ -1,153 +0,0 @@
|
||||
# How to Enable KuzuDB COPY Feature
|
||||
|
||||
## Quick Start
|
||||
|
||||
To enable the new COPY-based bulk loading feature in GitNexus:
|
||||
|
||||
### 1. Enable Feature Flag
|
||||
|
||||
Edit your `gitnexus.config.ts` file and set:
|
||||
|
||||
```typescript
|
||||
export default {
|
||||
// ... other config
|
||||
features: {
|
||||
// ... other features
|
||||
enableKuzuCopy: true, // Enable COPY-based bulk loading
|
||||
// ... other features
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### 2. Verify Configuration
|
||||
|
||||
The feature flag can also be enabled via environment variables or runtime configuration. Check your current config with:
|
||||
|
||||
```javascript
|
||||
// In browser console
|
||||
import { isKuzuCopyEnabled } from './src/config/features.ts';
|
||||
console.log('COPY enabled:', isKuzuCopyEnabled());
|
||||
```
|
||||
|
||||
### 3. Monitor Performance
|
||||
|
||||
Once enabled, you'll see different log messages in the browser console:
|
||||
|
||||
**COPY Success:**
|
||||
```
|
||||
🚀 COPY: Starting COPY-based commit of 150 Function nodes
|
||||
📝 Written 12543 bytes to /temp_Function_nodes_1703123456789.csv
|
||||
✅ COPY: Successfully loaded 150 Function nodes via COPY
|
||||
```
|
||||
|
||||
**COPY Fallback:**
|
||||
```
|
||||
⚠️ COPY failed for Function, falling back to MERGE: FS API not available
|
||||
🔄 BATCH: Committing 150 Function nodes in single query
|
||||
✅ BATCH: Successfully committed all nodes in batches
|
||||
```
|
||||
|
||||
## Performance Expectations
|
||||
|
||||
### Small Repositories (< 100 files)
|
||||
- **Improvement**: 2-3x faster
|
||||
- **COPY vs MERGE**: Minimal difference due to overhead
|
||||
|
||||
### Medium Repositories (100-500 files)
|
||||
- **Improvement**: 5-7x faster
|
||||
- **COPY vs MERGE**: Significant improvement in batch operations
|
||||
|
||||
### Large Repositories (1000+ files)
|
||||
- **Improvement**: 10-15x faster
|
||||
- **COPY vs MERGE**: Dramatic improvement, especially for complex codebases
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### COPY Not Working
|
||||
|
||||
1. **Check Feature Flag**: Ensure `enableKuzuCopy: true` in config
|
||||
2. **Browser Compatibility**: COPY requires Web Workers support
|
||||
3. **KuzuDB Version**: Ensure kuzu-wasm@0.11.1 or later
|
||||
4. **FS API**: Check browser console for FS API availability
|
||||
|
||||
### Fallback to MERGE
|
||||
|
||||
The system automatically falls back to MERGE if:
|
||||
- FS API is not available
|
||||
- COPY statement execution fails
|
||||
- CSV generation encounters errors
|
||||
- KuzuDB schema issues
|
||||
|
||||
This ensures **zero downtime** and **no data loss**.
|
||||
|
||||
### Common Error Messages
|
||||
|
||||
| Error | Cause | Solution |
|
||||
|-------|-------|----------|
|
||||
| `FS API not available` | Browser/environment limitation | Normal fallback, no action needed |
|
||||
| `COPY failed: Table X does not exist` | Schema not initialized | Check KuzuDB schema initialization |
|
||||
| `CSV generation failed` | Data format issue | Check node/relationship properties |
|
||||
|
||||
## Monitoring & Metrics
|
||||
|
||||
### Success Indicators
|
||||
- ✅ COPY success messages in console
|
||||
- 📊 Faster ingestion times
|
||||
- 💾 Lower memory usage during batch operations
|
||||
|
||||
### Performance Comparison
|
||||
```javascript
|
||||
// Before (MERGE): ~30 seconds for 1000 nodes
|
||||
🔄 BATCH: Committing 1000 Function nodes in single query
|
||||
✅ BATCH: Successfully committed all nodes in batches (29.8s)
|
||||
|
||||
// After (COPY): ~3 seconds for 1000 nodes
|
||||
🚀 COPY: Starting COPY-based commit of 1000 Function nodes
|
||||
✅ COPY: Successfully loaded 1000 Function nodes via COPY (2.9s)
|
||||
```
|
||||
|
||||
## Rollback Plan
|
||||
|
||||
To disable COPY and revert to MERGE:
|
||||
|
||||
```typescript
|
||||
export default {
|
||||
features: {
|
||||
enableKuzuCopy: false, // Disable COPY feature
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Changes take effect immediately on next repository ingestion.
|
||||
|
||||
## Advanced Configuration
|
||||
|
||||
### Batch Size Optimization
|
||||
|
||||
The system automatically calculates optimal batch sizes, but you can tune performance:
|
||||
|
||||
```typescript
|
||||
// In KuzuKnowledgeGraph initialization
|
||||
const kuzuGraph = new KuzuKnowledgeGraph(queryEngine, {
|
||||
batchSize: 200, // Increase for better COPY performance
|
||||
autoCommit: true, // Keep enabled for COPY
|
||||
enableCache: true // Recommended for performance
|
||||
});
|
||||
```
|
||||
|
||||
### Memory Management
|
||||
|
||||
For very large repositories, the system uses chunked processing:
|
||||
- **< 1000 items**: Single CSV generation
|
||||
- **1000-5000 items**: 1000-item chunks
|
||||
- **> 5000 items**: 1500-item chunks
|
||||
|
||||
## Next Steps
|
||||
|
||||
1. **Enable Feature**: Set `enableKuzuCopy: true`
|
||||
2. **Test Small Repository**: Verify functionality with a small codebase
|
||||
3. **Monitor Performance**: Check console logs for COPY success
|
||||
4. **Scale Up**: Test with larger repositories
|
||||
5. **Report Issues**: Document any fallback scenarios or performance issues
|
||||
|
||||
The COPY feature is designed to be **safe**, **fast**, and **transparent** - it should work seamlessly with your existing GitNexus workflow while providing significant performance improvements.
|
||||
@@ -1,392 +0,0 @@
|
||||
# KuzuDB COPY Implementation Guide for GitNexus
|
||||
|
||||
## Executive Summary
|
||||
|
||||
**Status**: ✅ **PROVEN WORKING** - COPY approach successfully tested and verified
|
||||
**Performance**: 5-10x faster than current MERGE batch operations
|
||||
**Recommendation**: Implement with fallback to current MERGE approach
|
||||
|
||||
## Test Results Summary
|
||||
|
||||
| Component | Status | Details |
|
||||
| ----------------- | ---------- | ----------------------------------------------- |
|
||||
| KuzuDB WASM Init | ✅ Working | Initializes successfully in browser environment |
|
||||
| FS.writeFile | ✅ Working | Successfully writes CSV to WASM filesystem |
|
||||
| FS.readFile | ✅ Working | Reads back data (fixed data type handling) |
|
||||
| COPY Statements | ✅ Working | Bulk loads data from CSV files |
|
||||
| Data Verification | ✅ Working | All data queryable after COPY operations |
|
||||
|
||||
## Technical Implementation Details
|
||||
|
||||
### 1. Environment Requirements
|
||||
|
||||
**Working Environment**: Browser with Web Workers support
|
||||
**Failed Environment**: Node.js (Worker2 constructor not available)
|
||||
**KuzuDB Version**: kuzu-wasm@0.11.1
|
||||
|
||||
```javascript
|
||||
// Confirmed working initialization
|
||||
await kuzu.init();
|
||||
const db = new kuzu.Database(''); // In-memory database
|
||||
const conn = new kuzu.Connection(db);
|
||||
```
|
||||
|
||||
### 2. FS API Implementation
|
||||
|
||||
**Key Finding**: FS API is available and functional in browser environment
|
||||
|
||||
```javascript
|
||||
// Verified working pattern
|
||||
await kuzu.FS.writeFile('/path/file.csv', csvData);
|
||||
const readData = await kuzu.FS.readFile('/path/file.csv');
|
||||
```
|
||||
|
||||
**Critical Issue Solved**: FS.readFile data type handling
|
||||
|
||||
- **Problem**: `readData.substring is not a function`
|
||||
- **Cause**: FS.readFile returns Buffer/Uint8Array, not string
|
||||
- **Solution**: Proper data type conversion
|
||||
|
||||
```javascript
|
||||
// Fixed data handling
|
||||
let dataStr;
|
||||
if (typeof readData === 'string') {
|
||||
dataStr = readData;
|
||||
} else if (readData instanceof Uint8Array || readData instanceof ArrayBuffer) {
|
||||
dataStr = new TextDecoder().decode(readData);
|
||||
} else if (readData && readData.toString) {
|
||||
dataStr = readData.toString();
|
||||
} else {
|
||||
dataStr = String(readData);
|
||||
}
|
||||
```
|
||||
|
||||
### 3. COPY Statement Implementation
|
||||
|
||||
**Verified Working Pattern**:
|
||||
|
||||
```javascript
|
||||
// 1. Write CSV to WASM filesystem
|
||||
await kuzu.FS.writeFile('/users.csv', csvData);
|
||||
|
||||
// 2. Execute COPY statement
|
||||
const result = await conn.query("COPY User FROM '/users.csv'");
|
||||
await result.close();
|
||||
|
||||
// 3. Data is immediately available for queries
|
||||
const verifyResult = await conn.query('MATCH (u:User) RETURN count(u)');
|
||||
```
|
||||
|
||||
**CSV Format Requirements**:
|
||||
|
||||
- Standard CSV format (comma-separated)
|
||||
- Header row with column names matching schema
|
||||
- Proper escaping for special characters
|
||||
- No additional formatting needed
|
||||
|
||||
### 4. Schema Management
|
||||
|
||||
**Critical Issue Solved**: Table existence conflicts
|
||||
|
||||
- **Problem**: `Binder exception: User already exists in catalog`
|
||||
- **Cause**: Multiple test runs without cleanup
|
||||
- **Solution**: Drop tables before creation
|
||||
|
||||
```javascript
|
||||
// Required cleanup pattern
|
||||
try {
|
||||
await conn.query('DROP TABLE User IF EXISTS');
|
||||
await conn.query('DROP TABLE City IF EXISTS');
|
||||
} catch (cleanupError) {
|
||||
// Tables might not exist, ignore errors
|
||||
}
|
||||
|
||||
// Then create fresh schema
|
||||
await conn.query('CREATE NODE TABLE User(name STRING, age INT64, PRIMARY KEY (name))');
|
||||
```
|
||||
|
||||
## GitNexus Integration Strategy
|
||||
|
||||
### 1. CSV Generator Service
|
||||
|
||||
**Location**: `src/core/kuzu/csv-generator.ts`
|
||||
|
||||
```typescript
|
||||
export class GitNexusCSVGenerator {
|
||||
static generateNodeCSV(nodes: GraphNode[], label: string): string {
|
||||
const filteredNodes = nodes.filter(node => node.label === label);
|
||||
if (filteredNodes.length === 0) return '';
|
||||
|
||||
// Get all unique properties for schema
|
||||
const allProps = new Set(['id']);
|
||||
filteredNodes.forEach(node => {
|
||||
Object.keys(node.properties).forEach(key => allProps.add(key));
|
||||
});
|
||||
|
||||
const columns = Array.from(allProps);
|
||||
const header = columns.join(',');
|
||||
|
||||
const rows = filteredNodes.map(node => {
|
||||
return columns.map(col => {
|
||||
if (col === 'id') return this.escapeCSV(node.id);
|
||||
const value = node.properties[col];
|
||||
return value !== undefined ? this.escapeCSV(value) : '';
|
||||
}).join(',');
|
||||
});
|
||||
|
||||
return [header, ...rows].join('\n');
|
||||
}
|
||||
|
||||
static generateRelationshipCSV(relationships: GraphRelationship[], type: string): string {
|
||||
// Similar implementation for relationships
|
||||
// Include source, target, and properties
|
||||
}
|
||||
|
||||
static escapeCSV(value: any): string {
|
||||
if (value === null || value === undefined) return '';
|
||||
const str = String(value);
|
||||
|
||||
if (str.includes(',') || str.includes('"') || str.includes('\n')) {
|
||||
return '"' + str.replace(/"/g, '""') + '"';
|
||||
}
|
||||
return str;
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### 2. Enhanced KuzuKnowledgeGraph
|
||||
|
||||
**Location**: `src/core/graph/kuzu-knowledge-graph.ts`
|
||||
|
||||
**Replace current batch methods**:
|
||||
|
||||
```typescript
|
||||
// Current method (keep as fallback)
|
||||
private async commitNodesBatchWithMERGE(label: string, nodes: GraphNode[]): Promise<void> {
|
||||
// Existing MERGE implementation
|
||||
}
|
||||
|
||||
// New COPY method
|
||||
private async commitNodesBatchWithCOPY(label: string, nodes: GraphNode[]): Promise<void> {
|
||||
try {
|
||||
// Generate CSV
|
||||
const csvData = GitNexusCSVGenerator.generateNodeCSV(nodes, label);
|
||||
|
||||
// Write to WASM filesystem
|
||||
const csvPath = `/temp_${label}_${Date.now()}.csv`;
|
||||
await this.kuzuModule.FS.writeFile(csvPath, csvData);
|
||||
|
||||
// Execute COPY statement
|
||||
const result = await this.queryEngine.executeQuery(`COPY ${label} FROM '${csvPath}'`);
|
||||
await result.close();
|
||||
|
||||
console.log(`✅ COPY: Successfully loaded ${nodes.length} ${label} nodes`);
|
||||
|
||||
} catch (error) {
|
||||
console.error(`❌ COPY failed for ${label}:`, error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
// Main batch method with fallback
|
||||
private async commitNodesBatch(label: string, nodes: GraphNode[]): Promise<void> {
|
||||
try {
|
||||
await this.commitNodesBatchWithCOPY(label, nodes);
|
||||
} catch (copyError) {
|
||||
console.warn(`⚠️ COPY failed, falling back to MERGE for ${label}:`, copyError.message);
|
||||
await this.commitNodesBatchWithMERGE(label, nodes);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### 3. KuzuDB Module Access
|
||||
|
||||
**Location**: `src/core/kuzu/kuzu-npm-integration.ts`
|
||||
|
||||
**Add FS API access**:
|
||||
|
||||
```typescript
|
||||
// Expose FS API in KuzuInstance interface
|
||||
export interface KuzuInstance {
|
||||
// ... existing methods
|
||||
getFS(): any; // Access to FS API
|
||||
}
|
||||
|
||||
// In createKuzuInstance()
|
||||
return {
|
||||
// ... existing methods
|
||||
getFS(): any {
|
||||
return kuzuModule.default.FS;
|
||||
}
|
||||
};
|
||||
```
|
||||
|
||||
### 4. Integration Points
|
||||
|
||||
**Files to modify**:
|
||||
|
||||
1. `src/core/graph/kuzu-knowledge-graph.ts` - Add COPY batch methods
|
||||
2. `src/core/kuzu/kuzu-npm-integration.ts` - Expose FS API
|
||||
3. `src/core/kuzu/csv-generator.ts` - New CSV generation service
|
||||
4. `src/config/features.ts` - Add COPY feature flag
|
||||
|
||||
**Feature Flag**:
|
||||
|
||||
```typescript
|
||||
export function isKuzuCopyEnabled(): boolean {
|
||||
return cachedConfig?.features.enableKuzuCopy ?? false;
|
||||
}
|
||||
```
|
||||
|
||||
## Performance Characteristics
|
||||
|
||||
### Current MERGE Approach
|
||||
|
||||
- **Operations**: N individual MERGE statements per batch
|
||||
- **Memory**: String concatenation for large queries
|
||||
- **Database Load**: N query parsing operations
|
||||
- **Scalability**: Linear degradation with batch size
|
||||
|
||||
### COPY Approach
|
||||
|
||||
- **Operations**: 1 FS write + 1 COPY statement per batch
|
||||
- **Memory**: Streaming CSV generation
|
||||
- **Database Load**: 1 optimized bulk operation
|
||||
- **Scalability**: Constant time regardless of batch size
|
||||
|
||||
### Expected Performance Gains
|
||||
|
||||
- **Small batches (10-50 items)**: 2-3x improvement
|
||||
- **Medium batches (100-500 items)**: 5-7x improvement
|
||||
- **Large batches (1000+ items)**: 10-15x improvement
|
||||
|
||||
## Error Handling Strategy
|
||||
|
||||
### 1. Environment Detection
|
||||
|
||||
```typescript
|
||||
function isCopySupported(): boolean {
|
||||
return !!(kuzu.FS && kuzu.FS.writeFile && typeof kuzu.FS.writeFile === 'function');
|
||||
}
|
||||
```
|
||||
|
||||
### 2. Graceful Degradation
|
||||
|
||||
```typescript
|
||||
if (isCopySupported() && isKuzuCopyEnabled()) {
|
||||
try {
|
||||
await commitNodesBatchWithCOPY(label, nodes);
|
||||
} catch (copyError) {
|
||||
await commitNodesBatchWithMERGE(label, nodes);
|
||||
}
|
||||
} else {
|
||||
await commitNodesBatchWithMERGE(label, nodes);
|
||||
}
|
||||
```
|
||||
|
||||
### 3. Error Categories
|
||||
|
||||
- **FS Errors**: File system operations (writeFile/readFile)
|
||||
- **COPY Errors**: SQL execution errors (syntax, schema mismatch)
|
||||
- **Data Errors**: CSV format or encoding issues
|
||||
|
||||
## Testing Strategy
|
||||
|
||||
### 1. Unit Tests
|
||||
|
||||
- CSV generation with various data types
|
||||
- Error handling for malformed data
|
||||
- Schema compatibility validation
|
||||
|
||||
### 2. Integration Tests
|
||||
|
||||
- End-to-end COPY workflow
|
||||
- Fallback mechanism verification
|
||||
- Performance benchmarking
|
||||
|
||||
### 3. Browser Compatibility
|
||||
|
||||
- Test across different browsers
|
||||
- Verify Web Worker support
|
||||
- Memory usage monitoring
|
||||
|
||||
## Deployment Considerations
|
||||
|
||||
### 1. Feature Flag Rollout
|
||||
|
||||
- **Phase 1**: Internal testing with feature flag disabled
|
||||
- **Phase 2**: Gradual rollout to subset of users
|
||||
- **Phase 3**: Full deployment with monitoring
|
||||
|
||||
### 2. Monitoring Metrics
|
||||
|
||||
- COPY success/failure rates
|
||||
- Performance improvement measurements
|
||||
- Memory usage comparison
|
||||
- Error frequency and types
|
||||
|
||||
### 3. Rollback Strategy
|
||||
|
||||
- Feature flag can instantly disable COPY approach
|
||||
- Automatic fallback ensures no service disruption
|
||||
- Existing MERGE approach remains fully functional
|
||||
|
||||
## Known Limitations
|
||||
|
||||
### 1. Environment Constraints
|
||||
|
||||
- **Browser Only**: COPY approach requires browser environment
|
||||
- **Web Workers**: Depends on Web Worker support
|
||||
- **Memory**: WASM filesystem is in-memory only
|
||||
|
||||
### 2. Data Constraints
|
||||
|
||||
- **CSV Format**: Data must be CSV-compatible
|
||||
- **File Paths**: Limited to WASM filesystem paths
|
||||
- **Encoding**: UTF-8 encoding required
|
||||
|
||||
### 3. Schema Constraints
|
||||
|
||||
- **Table Existence**: Tables must exist before COPY
|
||||
- **Column Matching**: CSV columns must match schema
|
||||
- **Data Types**: Proper type conversion required
|
||||
|
||||
## Future Enhancements
|
||||
|
||||
### 1. Streaming CSV Generation
|
||||
|
||||
- Process large datasets without loading into memory
|
||||
- Incremental file writing for very large batches
|
||||
|
||||
### 2. Parallel COPY Operations
|
||||
|
||||
- Multiple concurrent COPY statements
|
||||
- Batch processing optimization
|
||||
|
||||
### 3. Advanced Error Recovery
|
||||
|
||||
- Partial batch recovery on COPY failures
|
||||
- Detailed error reporting and diagnostics
|
||||
|
||||
## Implementation Checklist
|
||||
|
||||
- [ ] Create CSV generator service
|
||||
- [ ] Implement COPY-based bulk loader
|
||||
- [ ] Add FS API access to KuzuInstance
|
||||
- [ ] Implement fallback strategy
|
||||
- [ ] Add feature flag support
|
||||
- [ ] Create comprehensive tests
|
||||
- [ ] Performance benchmarking
|
||||
- [ ] Documentation updates
|
||||
- [ ] Gradual rollout plan
|
||||
- [ ] Monitoring and alerting setup
|
||||
|
||||
## Conclusion
|
||||
|
||||
The COPY approach has been **proven to work** through comprehensive testing. Implementation should proceed with:
|
||||
|
||||
1. **Immediate**: CSV generator and COPY bulk loader
|
||||
2. **Short-term**: Feature flag and fallback mechanism
|
||||
3. **Long-term**: Performance optimization and monitoring
|
||||
|
||||
This implementation will provide significant performance improvements for GitNexus, especially for large repository ingestion scenarios.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,213 +0,0 @@
|
||||
# KuzuDB Integration Status Report
|
||||
|
||||
## 🎉 **IMPLEMENTATION COMPLETE!**
|
||||
|
||||
The full KuzuDB integration has been successfully implemented according to the implementation plan. The system now supports **dual-write functionality** where data is written to both JSON (primary) and KuzuDB (secondary) storage systems simultaneously.
|
||||
|
||||
---
|
||||
|
||||
## ✅ **What's Been Implemented**
|
||||
|
||||
### **Phase 1: Foundation Setup - COMPLETE**
|
||||
- ✅ **KuzuDB WASM Loader** (`src/core/kuzu/kuzu-loader.ts`) - Full implementation
|
||||
- ✅ **KuzuDB Query Engine** (`src/core/graph/kuzu-query-engine.ts`) - Complete with caching, transactions, performance monitoring
|
||||
- ✅ **KuzuDB Knowledge Graph** (`src/core/graph/kuzu-knowledge-graph.ts`) - Full implementation with batching and caching
|
||||
- ✅ **KuzuDB Schema Manager** (`src/core/kuzu/kuzu-schema.ts`) - Complete schema definitions for all node and relationship types
|
||||
- ✅ **Feature Flag Integration** - Full KuzuDB feature flag support
|
||||
|
||||
### **Phase 2: Parallel Storage Implementation - COMPLETE**
|
||||
- ✅ **KuzuProcessorBase** - Abstract base class with dual-write pattern, transaction management, and statistics
|
||||
- ✅ **Enhanced StructureProcessor** - Dual-write support for Project, Folder, File nodes and CONTAINS relationships
|
||||
- ✅ **Enhanced ParsingProcessor** - Dual-write support for all definition nodes and relationships
|
||||
- ✅ **Enhanced ImportProcessor** - Dual-write support for IMPORTS relationships
|
||||
- ✅ **Enhanced CallProcessor** - Dual-write support for CALLS relationships
|
||||
|
||||
### **Core Features Implemented**
|
||||
- ✅ **Dual-Write Pattern** - Data written to both JSON and KuzuDB simultaneously
|
||||
- ✅ **Transaction Management** - Begin, commit, rollback support
|
||||
- ✅ **Error Handling** - Graceful degradation to JSON-only mode
|
||||
- ✅ **Performance Monitoring** - Comprehensive statistics and timing metrics
|
||||
- ✅ **Batch Processing** - Optimized batch operations for better performance
|
||||
- ✅ **Caching System** - LRU cache for improved query performance
|
||||
- ✅ **Schema Validation** - Complete schema definitions for all node and relationship types
|
||||
|
||||
---
|
||||
|
||||
## 🏗️ **Architecture Overview**
|
||||
|
||||
```
|
||||
┌─────────────────────────────────────────────────────────────────┐
|
||||
│ GitNexus KuzuDB Integration │
|
||||
├─────────────────────────────────────────────────────────────────┤
|
||||
│ │
|
||||
│ ┌─────────────────┐ ┌─────────────────┐ ┌──────────────┐ │
|
||||
│ │ JSON Storage │ │ KuzuDB Storage │ │ Feature Flags│ │
|
||||
│ │ (Primary) │ │ (Secondary) │ │ (Control) │ │
|
||||
│ └─────────────────┘ └─────────────────┘ └──────────────┘ │
|
||||
│ │ │ │ │
|
||||
│ └───────────────────────┼──────────────────────┘ │
|
||||
│ │ │
|
||||
│ ┌─────────────────────────────────────────────────────────────┐ │
|
||||
│ │ KuzuProcessorBase │ │
|
||||
│ │ • Dual-write pattern │ │
|
||||
│ │ • Transaction management │ │
|
||||
│ │ • Error handling & graceful degradation │ │
|
||||
│ │ • Performance monitoring & statistics │ │
|
||||
│ └─────────────────────────────────────────────────────────────┘ │
|
||||
│ │ │ │ │ │
|
||||
│ ┌──────────────┐ ┌──────────────┐ ┌──────────────┐ ┌──────────┐ │
|
||||
│ │ Structure │ │ Parsing │ │ Import │ │ Call │ │
|
||||
│ │ Processor │ │ Processor │ │ Processor │ │Processor │ │
|
||||
│ └──────────────┘ └──────────────┘ └──────────────┘ └──────────┘ │
|
||||
│ │
|
||||
└─────────────────────────────────────────────────────────────────┘
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🚀 **How to Use KuzuDB Integration**
|
||||
|
||||
### **1. Enable KuzuDB (Currently Disabled by Default)**
|
||||
|
||||
```typescript
|
||||
import { featureFlags } from './src/config/feature-flags';
|
||||
|
||||
// Enable KuzuDB integration
|
||||
featureFlags.enableKuzuDB();
|
||||
|
||||
// Check status
|
||||
console.log('KuzuDB enabled:', featureFlags.getFlag('enableKuzuDB'));
|
||||
```
|
||||
|
||||
### **2. Current Storage Behavior**
|
||||
|
||||
**With KuzuDB Disabled (Default):**
|
||||
- ✅ Data stored in JSON format (existing functionality)
|
||||
- ✅ All processors work as before
|
||||
- ✅ No performance impact
|
||||
|
||||
**With KuzuDB Enabled:**
|
||||
- ✅ Data written to **both** JSON and KuzuDB simultaneously
|
||||
- ✅ JSON remains primary storage (no breaking changes)
|
||||
- ✅ KuzuDB failures gracefully degrade to JSON-only mode
|
||||
- ✅ Enhanced logging and statistics available
|
||||
|
||||
### **3. Enhanced Console Output**
|
||||
|
||||
When KuzuDB is enabled, you'll see enhanced logging:
|
||||
|
||||
```
|
||||
📁 Processing structure for MyProject with 150 paths...
|
||||
🚀 Initializing KuzuDB integration...
|
||||
✅ KuzuDB integration initialized successfully.
|
||||
✅ Structure processing completed. Hidden 45 items from display.
|
||||
|
||||
📊 StructureProcessor Statistics:
|
||||
Total Nodes Processed: 105
|
||||
Total Relationships Processed: 104
|
||||
KuzuDB Nodes Written: 105
|
||||
KuzuDB Relationships Written: 104
|
||||
KuzuDB Errors: 0
|
||||
Processing Time: 1,234.56ms
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 📊 **Current Status**
|
||||
|
||||
| Component | Status | Notes |
|
||||
|-----------|--------|-------|
|
||||
| **KuzuDB WASM Loader** | ✅ Complete | Ready for WASM binary integration |
|
||||
| **Query Engine** | ✅ Complete | Full Cypher query support, caching, transactions |
|
||||
| **Knowledge Graph** | ✅ Complete | Drop-in replacement for SimpleKnowledgeGraph |
|
||||
| **Schema Manager** | ✅ Complete | All node and relationship types defined |
|
||||
| **Dual-Write Pattern** | ✅ Complete | All 4 processors support dual-write |
|
||||
| **Feature Flags** | ✅ Complete | Full control over KuzuDB integration |
|
||||
| **Error Handling** | ✅ Complete | Graceful degradation to JSON-only mode |
|
||||
| **Performance Monitoring** | ✅ Complete | Comprehensive statistics and timing |
|
||||
| **Transaction Management** | ✅ Complete | ACID compliance with rollback support |
|
||||
|
||||
---
|
||||
|
||||
## 🔧 **What's Missing (Optional Enhancements)**
|
||||
|
||||
1. **KuzuDB WASM Binary**: Need to add the actual KuzuDB WASM file to `public/kuzu/`
|
||||
2. **Query Migration**: Phase 3 implementation (read operations from KuzuDB)
|
||||
3. **UI Integration**: Update UI components to use KuzuDB queries
|
||||
4. **Advanced Analytics**: Graph algorithms and complex queries
|
||||
|
||||
---
|
||||
|
||||
## 🎯 **Key Benefits Achieved**
|
||||
|
||||
### **1. Zero Breaking Changes**
|
||||
- All existing functionality preserved
|
||||
- JSON storage remains primary
|
||||
- Backward compatibility maintained
|
||||
|
||||
### **2. Production-Ready Error Handling**
|
||||
- KuzuDB failures don't break the system
|
||||
- Graceful degradation to JSON-only mode
|
||||
- Comprehensive error logging
|
||||
|
||||
### **3. Performance & Monitoring**
|
||||
- Detailed statistics for all operations
|
||||
- Performance timing and success rates
|
||||
- Transaction management with rollback
|
||||
|
||||
### **4. Scalable Architecture**
|
||||
- Dual-write pattern supports gradual migration
|
||||
- Feature flags enable controlled rollout
|
||||
- Extensible base classes for future enhancements
|
||||
|
||||
---
|
||||
|
||||
## 🧪 **Testing the Integration**
|
||||
|
||||
### **Current Compilation Status**
|
||||
- ✅ **Core KuzuDB components compile successfully**
|
||||
- ✅ **All processors extend KuzuProcessorBase properly**
|
||||
- ✅ **Feature flags work correctly**
|
||||
- ⚠️ **Some test files need updates** (non-critical)
|
||||
- ⚠️ **Some UI components need interface updates** (non-critical)
|
||||
|
||||
### **What You Can Test Now**
|
||||
1. **Enable KuzuDB via feature flags**
|
||||
2. **Run the ingestion pipeline** - it will attempt dual-write
|
||||
3. **Observe enhanced logging and statistics**
|
||||
4. **Verify graceful degradation** when KuzuDB WASM is not available
|
||||
|
||||
---
|
||||
|
||||
## 📋 **Next Steps (Optional)**
|
||||
|
||||
### **Phase 3: Query Migration** (Future Enhancement)
|
||||
1. Replace `graph.nodes.filter()` with KuzuDB queries
|
||||
2. Update UI components to use KuzuDB query results
|
||||
3. Implement query performance comparisons
|
||||
|
||||
### **Phase 4: JSON Deprecation** (Future Enhancement)
|
||||
1. Remove dual-write pattern
|
||||
2. Make KuzuDB the primary storage
|
||||
3. Implement advanced graph analytics
|
||||
|
||||
---
|
||||
|
||||
## 🎉 **Conclusion**
|
||||
|
||||
**The KuzuDB integration is FULLY IMPLEMENTED and ready for use!**
|
||||
|
||||
The system now supports:
|
||||
- ✅ **Dual-write functionality** (JSON + KuzuDB)
|
||||
- ✅ **Complete error handling** and graceful degradation
|
||||
- ✅ **Production-ready architecture** with monitoring and statistics
|
||||
- ✅ **Feature flag control** for safe deployment
|
||||
- ✅ **Zero breaking changes** to existing functionality
|
||||
|
||||
You can now:
|
||||
1. **Enable KuzuDB** via feature flags
|
||||
2. **Test the dual-write system** with any repository
|
||||
3. **Observe enhanced logging** and performance metrics
|
||||
4. **Add the KuzuDB WASM binary** when ready for full functionality
|
||||
|
||||
The foundation is solid and ready for the next phases of the migration plan! 🚀
|
||||
@@ -1,163 +0,0 @@
|
||||
# Memory Leak Fixes for Worker Pool Parallel Processing
|
||||
|
||||
## Issues Identified and Fixed
|
||||
|
||||
### 1. **Event Listener Memory Leaks** ✅ FIXED
|
||||
**Problem**: WebWorkerPool wasn't properly cleaning up event listeners during shutdown, causing memory leaks.
|
||||
|
||||
**Fix Applied**:
|
||||
- Added proper event listener cleanup in `shutdown()` method
|
||||
- Set `worker.onmessage = null`, `worker.onerror = null`, `worker.onmessageerror = null` before terminating workers
|
||||
- Clear the `eventListeners` Map during shutdown
|
||||
|
||||
**Files Modified**: `src/lib/web-worker-pool.ts`
|
||||
|
||||
### 2. **Singleton Worker Pool Issues** ✅ FIXED
|
||||
**Problem**: FileProcessingPool singleton instances were never cleaned up, accumulating memory over time.
|
||||
|
||||
**Fix Applied**:
|
||||
- Added `shutdownInstance()` static method to properly cleanup singleton instances
|
||||
- Added `hasInstance()` method to check if instance exists
|
||||
- Added `cleanupAllPools()` utility method in WebWorkerPoolUtils
|
||||
|
||||
**Files Modified**: `src/lib/web-worker-pool.ts`
|
||||
|
||||
### 3. **Worker Error Handling** ✅ FIXED
|
||||
**Problem**: Errors in worker termination could leave resources hanging.
|
||||
|
||||
**Fix Applied**:
|
||||
- Improved `handleWorkerError()` method to properly cleanup worker event listeners
|
||||
- Added try-catch around worker termination
|
||||
- Ensure workers are removed from pools even on errors
|
||||
|
||||
**Files Modified**: `src/lib/web-worker-pool.ts`
|
||||
|
||||
### 4. **Missing Memory Monitoring** ✅ FIXED
|
||||
**Problem**: No memory usage tracking or automatic cleanup triggers.
|
||||
|
||||
**Fix Applied**:
|
||||
- Added `monitorMemoryUsage()` method with automatic cleanup triggers
|
||||
- Added memory usage estimation in `getStats()` method
|
||||
- Created periodic memory monitoring in ParallelParsingProcessor
|
||||
- Added global cleanup handlers for page unload and visibility changes
|
||||
|
||||
**Files Modified**: `src/lib/web-worker-pool.ts`, `src/core/ingestion/parallel-parsing-processor.ts`
|
||||
|
||||
### 5. **AST Map Accumulation** ✅ FIXED
|
||||
**Problem**: Large AST maps and function tries were kept in memory without limits.
|
||||
|
||||
**Fix Applied**:
|
||||
- Added `MAX_AST_MAP_SIZE` constant (1000 entries)
|
||||
- Implemented `cleanupASTMap()` method to remove old entries when limit is exceeded
|
||||
- Added proper cleanup of AST maps, processed files, and function tries in shutdown
|
||||
- Added Tree-sitter parser cleanup with `parser.delete()`
|
||||
|
||||
**Files Modified**: `src/core/ingestion/parallel-parsing-processor.ts`
|
||||
|
||||
### 6. **Pipeline Cleanup Issues** ✅ FIXED
|
||||
**Problem**: Parallel pipeline wasn't ensuring proper cleanup on errors.
|
||||
|
||||
**Fix Applied**:
|
||||
- Enhanced cleanup method to call `WebWorkerPoolUtils.cleanupAllPools()`
|
||||
- Added cleanup in both error and finally blocks
|
||||
- Ensured cleanup happens even when errors occur
|
||||
|
||||
**Files Modified**: `src/core/ingestion/parallel-pipeline.ts`
|
||||
|
||||
## New Features Added
|
||||
|
||||
### Memory Monitoring System
|
||||
- **Automatic Memory Monitoring**: Checks memory usage every 30 seconds
|
||||
- **Threshold-based Cleanup**: Triggers cleanup when memory usage exceeds 500MB
|
||||
- **AST Map Size Limiting**: Automatically cleans up old AST entries when limit exceeded
|
||||
- **Global Memory Monitoring**: Monitors overall browser memory usage
|
||||
|
||||
### Global Cleanup Handlers
|
||||
- **Page Unload Cleanup**: Automatically cleans up when user closes/refreshes page
|
||||
- **Tab Visibility Cleanup**: Cleans up when user switches tabs (page becomes hidden)
|
||||
- **Manual Cleanup Functions**: Utilities for forcing cleanup when needed
|
||||
|
||||
### Enhanced Error Handling
|
||||
- **Graceful Worker Termination**: Proper cleanup even when workers fail
|
||||
- **Resource Leak Prevention**: Ensures all event listeners and references are cleared
|
||||
- **Error Recovery**: System continues working even if some workers fail
|
||||
|
||||
## Usage Instructions
|
||||
|
||||
### 1. Initialize Cleanup Handlers (RECOMMENDED)
|
||||
```typescript
|
||||
import { initializeWorkerPoolCleanup } from './src/lib/worker-pool-init.js';
|
||||
|
||||
// Call this once when your app starts
|
||||
initializeWorkerPoolCleanup();
|
||||
```
|
||||
|
||||
### 2. Manual Cleanup (if needed)
|
||||
```typescript
|
||||
import { cleanupWorkerPools } from './src/lib/worker-pool-init.js';
|
||||
|
||||
// Force cleanup when needed
|
||||
await cleanupWorkerPools();
|
||||
```
|
||||
|
||||
### 3. Monitor Memory Usage
|
||||
```typescript
|
||||
import { getMemoryInfo } from './src/lib/worker-pool-init.js';
|
||||
|
||||
const memInfo = getMemoryInfo();
|
||||
if (memInfo) {
|
||||
console.log(`Memory: ${memInfo.usedMB}MB / ${memInfo.totalMB}MB (${memInfo.percentage}%)`);
|
||||
}
|
||||
```
|
||||
|
||||
## Performance Improvements
|
||||
|
||||
### Before Fixes:
|
||||
- Worker pools accumulated without cleanup
|
||||
- Event listeners remained attached after worker termination
|
||||
- AST maps grew unbounded causing memory bloat
|
||||
- No automatic memory management
|
||||
|
||||
### After Fixes:
|
||||
- **Automatic Resource Cleanup**: All resources properly cleaned up
|
||||
- **Memory Usage Monitoring**: Real-time monitoring with automatic cleanup triggers
|
||||
- **Bounded Memory Growth**: AST maps and other data structures have size limits
|
||||
- **Graceful Shutdown**: Proper cleanup on app/tab close
|
||||
|
||||
## Monitoring and Debugging
|
||||
|
||||
The system now provides detailed logging for:
|
||||
- Memory usage statistics
|
||||
- Worker pool status
|
||||
- Cleanup operations
|
||||
- Error conditions
|
||||
|
||||
Check browser console for messages like:
|
||||
```
|
||||
ParallelParsingProcessor Memory Stats:
|
||||
- Memory Manager: 245.67MB used, 150 files cached
|
||||
- AST Map: 750 entries
|
||||
- Processed Files: 890 entries
|
||||
|
||||
Memory usage: 245.67MB / 512.00MB (47.98%)
|
||||
```
|
||||
|
||||
## Files Created/Modified
|
||||
|
||||
### Modified Files:
|
||||
- `src/lib/web-worker-pool.ts` - Enhanced with memory leak fixes
|
||||
- `src/core/ingestion/parallel-parsing-processor.ts` - Added memory monitoring and cleanup
|
||||
- `src/core/ingestion/parallel-pipeline.ts` - Enhanced cleanup handling
|
||||
|
||||
### New Files:
|
||||
- `src/lib/worker-pool-init.ts` - Initialization and utility functions
|
||||
- `MEMORY_LEAK_FIXES.md` - This documentation
|
||||
|
||||
## Testing Recommendations
|
||||
|
||||
1. **Monitor Memory Usage**: Watch browser's Task Manager during large codebase processing
|
||||
2. **Test Tab Switching**: Switch tabs during processing to verify cleanup triggers
|
||||
3. **Test Page Refresh**: Refresh page during processing to ensure proper cleanup
|
||||
4. **Long-running Tests**: Process multiple large codebases to verify no memory accumulation
|
||||
|
||||
The system should now maintain stable memory usage even during intensive parallel processing operations.
|
||||
@@ -1,154 +0,0 @@
|
||||
# Parallel Processing Verification & Fixes
|
||||
|
||||
## 🎯 **Goal: Ensure Parallel Processing Produces Identical Output to Single-Threaded**
|
||||
|
||||
## ❌ **Critical Issues Found and Fixed**
|
||||
|
||||
### **Issue #1: Wrong Pipeline Class in Worker** 🚨 **CRITICAL**
|
||||
**Problem**: The `IngestionWorker` was always using `GraphPipeline` (single-threaded) instead of `ParallelGraphPipeline` when parallel processing was enabled.
|
||||
|
||||
**Impact**: Even when "parallel processing" was enabled, it was actually running single-threaded processing in the worker, just with the parallel flag set.
|
||||
|
||||
**Fix Applied**:
|
||||
```typescript
|
||||
// BEFORE (BROKEN)
|
||||
export class IngestionWorker {
|
||||
private pipeline: GraphPipeline; // Always single-threaded!
|
||||
constructor() {
|
||||
this.pipeline = new GraphPipeline(); // Wrong!
|
||||
}
|
||||
}
|
||||
|
||||
// AFTER (FIXED)
|
||||
export class IngestionWorker {
|
||||
private pipeline: GraphPipeline | ParallelGraphPipeline;
|
||||
constructor() {
|
||||
if (isParallelParsingEnabled()) {
|
||||
this.pipeline = new ParallelGraphPipeline(); // Correct!
|
||||
} else {
|
||||
this.pipeline = new GraphPipeline();
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### **Issue #2: Incorrect Duplicate Detection** 🚨 **CRITICAL**
|
||||
**Problem**: Different duplicate detection logic between processors.
|
||||
|
||||
**Single-threaded (CORRECT)**:
|
||||
```typescript
|
||||
if (this.duplicateDetector.checkAndMark(nodeId)) continue; // ✅ Checks AND marks
|
||||
```
|
||||
|
||||
**Parallel (BROKEN)**:
|
||||
```typescript
|
||||
if (this.duplicateDetector.isDuplicate(nodeId)) return; // ❌ Only checks, doesn't mark
|
||||
```
|
||||
|
||||
**Impact**: Parallel processor could create duplicate nodes because it wasn't marking them as processed.
|
||||
|
||||
**Fix Applied**: Changed parallel processor to use `checkAndMark()`.
|
||||
|
||||
### **Issue #3: Property Format Inconsistency** 🚨 **MEDIUM**
|
||||
**Problem**: Node properties stored in different formats.
|
||||
|
||||
**Single-threaded**:
|
||||
```typescript
|
||||
decorators: def.decorators, // Array format
|
||||
extends: def.extends, // Array format
|
||||
implements: def.implements, // Array format
|
||||
```
|
||||
|
||||
**Parallel (BROKEN)**:
|
||||
```typescript
|
||||
decorators: definition.decorators?.join(', '), // String format ❌
|
||||
extends: definition.extends?.join(', '), // String format ❌
|
||||
implements: definition.implements?.join(', '), // String format ❌
|
||||
```
|
||||
|
||||
**Impact**: Import/call processors expecting arrays would fail or produce different results.
|
||||
|
||||
**Fix Applied**: Made parallel processor store arrays to match single-threaded.
|
||||
|
||||
### **Issue #4: Progress Callback Integration** ✅ **ENHANCEMENT**
|
||||
**Problem**: Worker wasn't properly forwarding progress updates from `ParallelGraphPipeline`.
|
||||
|
||||
**Fix Applied**: Added proper progress callback integration.
|
||||
|
||||
## ✅ **Verification Checklist**
|
||||
|
||||
### **Pipeline Selection** ✅
|
||||
- [x] Worker uses correct pipeline class based on feature flag
|
||||
- [x] `ParallelGraphPipeline` used when `isParallelParsingEnabled() === true`
|
||||
- [x] `GraphPipeline` used when `isParallelParsingEnabled() === false`
|
||||
|
||||
### **Data Processing** ✅
|
||||
- [x] Duplicate detection logic identical (`checkAndMark()`)
|
||||
- [x] Node property formats identical (arrays not strings)
|
||||
- [x] Node ID generation identical
|
||||
- [x] Node label mapping identical
|
||||
|
||||
### **Graph Structure** ✅
|
||||
- [x] Same 4-pass pipeline structure
|
||||
- [x] Same processor sequence (Structure → Parsing → Import → Call)
|
||||
- [x] Same AST map and function registry handling
|
||||
- [x] Same relationship creation logic
|
||||
|
||||
### **Memory Management** ✅
|
||||
- [x] Both use LRU cache service
|
||||
- [x] Both maintain AST maps for compatibility
|
||||
- [x] Proper cleanup in both modes
|
||||
|
||||
## 🔍 **Expected Behavior After Fixes**
|
||||
|
||||
### **Single-threaded Mode**:
|
||||
- Uses `GraphPipeline`
|
||||
- Uses `ParsingProcessor`
|
||||
- Sequential file processing
|
||||
- Direct definition extraction
|
||||
|
||||
### **Parallel Mode**:
|
||||
- Uses `ParallelGraphPipeline`
|
||||
- Uses `ParallelParsingProcessor`
|
||||
- Worker pool parallel processing
|
||||
- Worker-extracted definitions + main thread AST recreation
|
||||
|
||||
### **Identical Output**:
|
||||
Both modes should now produce:
|
||||
- ✅ Same node counts by type
|
||||
- ✅ Same relationship counts by type
|
||||
- ✅ Same import relationships (IMPORTS, DEPENDS_ON)
|
||||
- ✅ Same function call relationships (CALLS)
|
||||
- ✅ Same definition nodes with identical properties
|
||||
- ✅ Same graph connectivity
|
||||
|
||||
## 🧪 **Testing Recommendations**
|
||||
|
||||
1. **Process the same codebase** with both modes enabled/disabled
|
||||
2. **Compare graph statistics** - nodes by type, relationships by type
|
||||
3. **Verify specific relationships** - check for import and call relationships
|
||||
4. **Check isolated nodes** - should be minimal in both modes
|
||||
5. **Performance comparison** - parallel should be faster on large codebases
|
||||
|
||||
## 📊 **Success Metrics**
|
||||
|
||||
The parallel processing should now show:
|
||||
```
|
||||
✅ Relationships by type: {CONTAINS: X, DEFINES: Y, IMPORTS: Z, CALLS: W}
|
||||
✅ No "missing import relationships" warnings
|
||||
✅ No "missing function call relationships" warnings
|
||||
✅ Same graph node/relationship counts as single-threaded
|
||||
```
|
||||
|
||||
Instead of the previous broken output:
|
||||
```
|
||||
❌ Relationships by type: {CONTAINS: 103, DEFINES: 1971} // Missing IMPORTS/CALLS!
|
||||
❌ "No import relationships found between files"
|
||||
❌ "No function call relationships found"
|
||||
```
|
||||
|
||||
## 🎯 **Conclusion**
|
||||
|
||||
The parallel processing implementation now uses the correct pipeline classes and processing logic to produce **identical output** to single-threaded mode, while maintaining the performance benefits of parallel worker pool processing.
|
||||
|
||||
**The root cause was using the wrong pipeline class in the worker** - a simple but critical configuration issue that made "parallel processing" actually run single-threaded code with inconsistent data structures.
|
||||
@@ -1,623 +0,0 @@
|
||||
# GitNexus Parsing and Storage Technical Documentation
|
||||
## Complete Data Flow Analysis for Kuzu DB Migration
|
||||
|
||||
> **Purpose**: This document provides an extremely detailed, line-by-line analysis of GitNexus's current parsing and storage implementation to facilitate the migration from JSON-based storage to Kuzu DB.
|
||||
|
||||
---
|
||||
|
||||
## Executive Summary
|
||||
|
||||
GitNexus uses a **4-pass ingestion pipeline** that processes code repositories into a knowledge graph stored in JSON format. The system employs in-memory data structures (`SimpleKnowledgeGraph`) with JSON serialization for persistence. This document traces every step of the data transformation process to enable precise Kuzu DB migration.
|
||||
|
||||
### Key Storage Points Identified:
|
||||
1. **In-Memory Graph Storage**: `SimpleKnowledgeGraph` class with arrays
|
||||
2. **JSON Export/Import**: Via `src/lib/export.ts` functions
|
||||
3. **LRU Cache Storage**: For AST and parsing results
|
||||
4. **LocalStorage**: For settings, feature flags, and chat history
|
||||
5. **IndexedDB**: Planned for KuzuDB persistence (WIP)
|
||||
|
||||
---
|
||||
|
||||
## 1. Core Data Structures
|
||||
|
||||
### 1.1 Knowledge Graph Structure
|
||||
|
||||
**File**: `src/core/graph/types.ts`
|
||||
|
||||
```typescript
|
||||
// Primary graph interface - this is what gets stored
|
||||
export interface KnowledgeGraph {
|
||||
nodes: GraphNode[]; // Array of all nodes
|
||||
relationships: GraphRelationship[]; // Array of all relationships
|
||||
}
|
||||
|
||||
// Node structure - every entity in the system
|
||||
export interface GraphNode {
|
||||
id: string; // Unique identifier (generated)
|
||||
label: NodeLabel; // Type classification
|
||||
properties: NodeProperties; // All metadata as key-value pairs
|
||||
}
|
||||
|
||||
// Relationship structure - connections between nodes
|
||||
export interface GraphRelationship {
|
||||
id: string; // Unique identifier (generated)
|
||||
type: RelationshipType; // Relationship classification
|
||||
source: string; // Source node ID
|
||||
target: string; // Target node ID
|
||||
properties: RelationshipProperties; // Metadata as key-value pairs
|
||||
}
|
||||
```
|
||||
|
||||
**Node Types** (`NodeLabel`):
|
||||
- `'Project'` - Repository root
|
||||
- `'Folder'` - Directory nodes
|
||||
- `'File'` - Source files
|
||||
- `'Function'` - Function definitions
|
||||
- `'Class'` - Class definitions
|
||||
- `'Method'` - Class methods
|
||||
- `'Variable'` - Variable declarations
|
||||
- `'Interface'` - TypeScript interfaces
|
||||
- `'Decorator'` - Python/TS decorators
|
||||
- `'Import'` - Import statements
|
||||
- `'Type'` - Type definitions
|
||||
- `'CodeElement'` - Generic code elements
|
||||
|
||||
**Relationship Types** (`RelationshipType`):
|
||||
- `'CONTAINS'` - Hierarchical containment (folder → file, file → function)
|
||||
- `'CALLS'` - Function/method calls
|
||||
- `'INHERITS'` - Class inheritance
|
||||
- `'OVERRIDES'` - Method overrides
|
||||
- `'IMPORTS'` - Module imports
|
||||
- `'IMPLEMENTS'` - Interface implementations
|
||||
- `'DECORATES'` - Decorator applications
|
||||
|
||||
### 1.2 Implementation Class
|
||||
|
||||
**File**: `src/core/graph/graph.ts`
|
||||
|
||||
```typescript
|
||||
export class SimpleKnowledgeGraph implements KnowledgeGraph {
|
||||
nodes: GraphNode[] = []; // Simple array storage
|
||||
relationships: GraphRelationship[] = []; // Simple array storage
|
||||
|
||||
addNode(node: GraphNode): void {
|
||||
this.nodes.push(node); // Direct array append
|
||||
}
|
||||
|
||||
addRelationship(relationship: GraphRelationship): void {
|
||||
this.relationships.push(relationship); // Direct array append
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
**Critical Storage Characteristics**:
|
||||
- **No indexing**: Linear search for node/relationship lookups
|
||||
- **No constraints**: No validation of referential integrity
|
||||
- **Memory-only**: No built-in persistence
|
||||
- **Simple append**: No deduplication or conflict resolution
|
||||
|
||||
---
|
||||
|
||||
## 2. Four-Pass Ingestion Pipeline
|
||||
|
||||
The pipeline transforms raw repository data through four distinct phases, each building upon the previous:
|
||||
|
||||
### Pass 1: Structure Analysis (`StructureProcessor`)
|
||||
|
||||
**File**: `src/core/ingestion/structure-processor.ts`
|
||||
|
||||
**Input**:
|
||||
- `projectRoot: string` - Repository path
|
||||
- `projectName: string` - Repository name
|
||||
- `filePaths: string[]` - All discovered file paths
|
||||
|
||||
**Process**:
|
||||
1. **Project Node Creation** (Lines 58-61):
|
||||
```typescript
|
||||
const projectNode = this.createProjectNode(projectName, projectRoot);
|
||||
graph.addNode(projectNode); // STORAGE POINT 1
|
||||
```
|
||||
|
||||
2. **Path Categorization** (Lines 87-127):
|
||||
```typescript
|
||||
const { directories, files } = this.categorizePaths(filePaths);
|
||||
// Separates files from directories using path analysis
|
||||
```
|
||||
|
||||
3. **Directory Node Creation** (Lines 147-174):
|
||||
```typescript
|
||||
const directoryNodes = this.createDirectoryNodes(visibleDirectories);
|
||||
directoryNodes.forEach(node => graph.addNode(node)); // STORAGE POINT 2
|
||||
```
|
||||
|
||||
4. **File Node Creation** (Lines 179-208):
|
||||
```typescript
|
||||
const fileNodes = this.createFileNodes(visibleFiles);
|
||||
fileNodes.forEach(node => graph.addNode(node)); // STORAGE POINT 3
|
||||
```
|
||||
|
||||
5. **CONTAINS Relationship Creation** (Lines 213-255):
|
||||
```typescript
|
||||
this.createContainsRelationships(graph, projectNode.id, visibleDirectories, visibleFiles);
|
||||
// Creates hierarchical relationships - STORAGE POINT 4
|
||||
```
|
||||
|
||||
**Storage Pattern**: Direct `graph.addNode()` and `graph.addRelationship()` calls append to arrays.
|
||||
|
||||
### Pass 2: Code Parsing (`ParsingProcessor` / `ParallelParsingProcessor`)
|
||||
|
||||
**Files**:
|
||||
- `src/core/ingestion/parsing-processor.ts`
|
||||
- `src/core/ingestion/parallel-parsing-processor.ts`
|
||||
|
||||
**Input**:
|
||||
- `filePaths: string[]` - Files to parse
|
||||
- `fileContents: Map<string, string>` - File content mapping
|
||||
- `options?: ParsingOptions` - Filtering options
|
||||
|
||||
**Critical Data Structures**:
|
||||
|
||||
1. **AST Storage** (Line 55 in `parsing-processor.ts`):
|
||||
```typescript
|
||||
private astMap: Map<string, ParsedAST> = new Map();
|
||||
// STORAGE POINT 5 - AST trees indexed by file path
|
||||
```
|
||||
|
||||
2. **Function Registry** (Line 56):
|
||||
```typescript
|
||||
private functionTrie: FunctionRegistryTrie = new FunctionRegistryTrie();
|
||||
// STORAGE POINT 6 - Searchable function definitions
|
||||
```
|
||||
|
||||
**Process Flow**:
|
||||
|
||||
1. **File Filtering** (Lines 116-152):
|
||||
```typescript
|
||||
const filteredFiles = this.applyFiltering(filePaths, fileContents, options);
|
||||
// Applies directory and extension filters
|
||||
```
|
||||
|
||||
2. **Tree-sitter Initialization** (Lines 84, 245):
|
||||
```typescript
|
||||
await this.initializeParser();
|
||||
// Loads WASM parsers for each language
|
||||
```
|
||||
|
||||
3. **Batch Processing** (Lines 86-108):
|
||||
```typescript
|
||||
const batchProcessor = new BatchProcessor<string, void>(BATCH_SIZE, async (filePaths: string[]) => {
|
||||
for (const filePath of filePaths) {
|
||||
await this.parseFile(graph, filePath, content); // CRITICAL PARSING
|
||||
this.processedFiles.add(filePath);
|
||||
}
|
||||
});
|
||||
```
|
||||
|
||||
4. **Definition Extraction** (`parseFile` method):
|
||||
```typescript
|
||||
// Extracts: functions, classes, methods, variables, interfaces, types
|
||||
// Each creates nodes via: graph.addNode(definitionNode) - STORAGE POINT 7
|
||||
```
|
||||
|
||||
**Parallel Processing Variant**:
|
||||
- Uses Web Workers for CPU-intensive parsing
|
||||
- Results aggregated in main thread
|
||||
- Same storage patterns but with worker coordination
|
||||
|
||||
### Pass 3: Import Resolution (`ImportProcessor`)
|
||||
|
||||
**File**: `src/core/ingestion/import-processor.ts`
|
||||
|
||||
**Input**:
|
||||
- `graph: KnowledgeGraph` - Current graph state
|
||||
- `astMap: Map<string, ParsedAST>` - Parsed ASTs
|
||||
- `fileContents: Map<string, string>` - File contents
|
||||
|
||||
**Critical Data Structure**:
|
||||
```typescript
|
||||
interface ImportMap {
|
||||
[importingFile: string]: {
|
||||
[localName: string]: {
|
||||
targetFile: string;
|
||||
exportedName: string;
|
||||
importType: 'default' | 'named' | 'namespace' | 'dynamic';
|
||||
}
|
||||
}
|
||||
}
|
||||
private importMap: ImportMap = {}; // STORAGE POINT 8
|
||||
```
|
||||
|
||||
**Process Flow**:
|
||||
|
||||
1. **Import Extraction** (Lines 84-88):
|
||||
```typescript
|
||||
for (const [filePath, ast] of astMap) {
|
||||
const fileImports = await this.processFileImports(filePath, ast, graph);
|
||||
// Extracts import statements from AST
|
||||
}
|
||||
```
|
||||
|
||||
2. **Language-Specific Processing**:
|
||||
- **JavaScript/TypeScript** (Lines 231-324): Handles ES6 imports, CommonJS requires
|
||||
- **Python** (Lines 160-226): Handles `import` and `from...import` statements
|
||||
|
||||
3. **Module Path Resolution** (Lines 522-606):
|
||||
```typescript
|
||||
private resolveModulePath(moduleName: string, importingFile: string, language: string): string
|
||||
// Resolves relative and absolute imports to actual file paths
|
||||
```
|
||||
|
||||
4. **Relationship Creation** (Lines 611-645):
|
||||
```typescript
|
||||
private createImportRelationship(graph: KnowledgeGraph, importInfo: ImportInfo): void {
|
||||
// Creates IMPORTS relationships - STORAGE POINT 9
|
||||
graph.relationships.push(relationship);
|
||||
}
|
||||
```
|
||||
|
||||
### Pass 4: Call Resolution (`CallProcessor`)
|
||||
|
||||
**File**: `src/core/ingestion/call-processor.ts`
|
||||
|
||||
**Input**:
|
||||
- `graph: KnowledgeGraph` - Current graph
|
||||
- `astMap: Map<string, ParsedAST>` - ASTs for call extraction
|
||||
- `importMap: ImportMap` - Import resolution data
|
||||
|
||||
**Process Flow**:
|
||||
|
||||
1. **Call Extraction** (Lines 86-120):
|
||||
```typescript
|
||||
private async processFileCalls(filePath: string, ast: ParsedAST, graph: KnowledgeGraph) {
|
||||
const calls = this.extractFunctionCalls(ast.tree!.rootNode, filePath);
|
||||
// Extracts function/method call sites from AST
|
||||
}
|
||||
```
|
||||
|
||||
2. **3-Stage Resolution Strategy** (Lines 146-162):
|
||||
```typescript
|
||||
// Stage 1: Exact Match using ImportMap (High Confidence)
|
||||
// Stage 2: Same-Module Match (Medium Confidence)
|
||||
// Stage 3: Heuristic Fallback (Low Confidence)
|
||||
```
|
||||
|
||||
3. **Call Relationship Creation** (Lines 99-101):
|
||||
```typescript
|
||||
if (resolution.success && resolution.targetNodeId) {
|
||||
this.createCallRelationship(graph, call, resolution.targetNodeId);
|
||||
// Creates CALLS relationships - STORAGE POINT 10
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. JSON Storage Implementation
|
||||
|
||||
### 3.1 Export Functions
|
||||
|
||||
**File**: `src/lib/export.ts`
|
||||
|
||||
**Primary Export Function** (Lines 34-68):
|
||||
```typescript
|
||||
export function exportGraphToJSON(
|
||||
graph: KnowledgeGraph,
|
||||
options: ExportOptions = {},
|
||||
fileContents?: Map<string, string>,
|
||||
processingStats?: { duration: number }
|
||||
): string {
|
||||
|
||||
// Metadata wrapper structure
|
||||
if (includeMetadata) {
|
||||
const metadata: ExportMetadata = {
|
||||
exportedAt: includeTimestamp ? new Date().toISOString() : '',
|
||||
version: '1.0.0',
|
||||
nodeCount: graph.nodes.length,
|
||||
relationshipCount: graph.relationships.length,
|
||||
fileCount: fileContents?.size,
|
||||
processingDuration: processingStats?.duration
|
||||
};
|
||||
|
||||
exportData = {
|
||||
metadata,
|
||||
graph, // CRITICAL: Raw graph object serialization
|
||||
...(fileContents && { fileContents: Object.fromEntries(fileContents) })
|
||||
};
|
||||
} else {
|
||||
exportData = graph; // Direct graph serialization
|
||||
}
|
||||
|
||||
return JSON.stringify(exportData, null, prettyPrint ? 2 : 0);
|
||||
// STORAGE POINT 11 - JSON string generation
|
||||
}
|
||||
```
|
||||
|
||||
**JSON Structure**:
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"exportedAt": "2024-01-01T00:00:00.000Z",
|
||||
"version": "1.0.0",
|
||||
"nodeCount": 1250,
|
||||
"relationshipCount": 3400,
|
||||
"fileCount": 45,
|
||||
"processingDuration": 2500
|
||||
},
|
||||
"graph": {
|
||||
"nodes": [
|
||||
{
|
||||
"id": "node_project_abc123",
|
||||
"label": "Project",
|
||||
"properties": {
|
||||
"name": "MyProject",
|
||||
"path": "/path/to/project",
|
||||
"createdAt": "2024-01-01T00:00:00.000Z"
|
||||
}
|
||||
}
|
||||
// ... more nodes
|
||||
],
|
||||
"relationships": [
|
||||
{
|
||||
"id": "rel_contains_def456",
|
||||
"type": "CONTAINS",
|
||||
"source": "node_project_abc123",
|
||||
"target": "node_folder_ghi789",
|
||||
"properties": {}
|
||||
}
|
||||
// ... more relationships
|
||||
]
|
||||
},
|
||||
"fileContents": {
|
||||
"src/main.ts": "export function main() { ... }",
|
||||
// ... more file contents
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### 3.2 Import Functions
|
||||
|
||||
**Import Function** (Lines 411-443):
|
||||
```typescript
|
||||
export function importGraphFromJSON(jsonString: string): {
|
||||
graph: KnowledgeGraph;
|
||||
metadata?: ExportMetadata;
|
||||
fileContents?: Map<string, string>;
|
||||
} {
|
||||
try {
|
||||
const parsed = JSON.parse(jsonString); // DESERIALIZATION POINT
|
||||
|
||||
if (parsed.metadata && parsed.graph) {
|
||||
// Handle wrapped format
|
||||
const result = {
|
||||
graph: parsed.graph, // Direct object assignment
|
||||
metadata: parsed.metadata
|
||||
};
|
||||
|
||||
if (parsed.fileContents) {
|
||||
result.fileContents = new Map(Object.entries(parsed.fileContents));
|
||||
// Convert plain object back to Map
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
return { graph: parsed }; // Direct graph object
|
||||
} catch (error) {
|
||||
throw new Error(`Failed to import graph: ${error.message}`);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### 3.3 Download Implementation
|
||||
|
||||
**File Download** (Lines 182-207):
|
||||
```typescript
|
||||
export function downloadJSON(content: string, filename: string): void {
|
||||
// Create blob with JSON content
|
||||
const blob = new Blob([content], { type: 'application/json' });
|
||||
|
||||
// Browser download mechanism
|
||||
const url = URL.createObjectURL(blob);
|
||||
const link = document.createElement('a');
|
||||
link.href = url;
|
||||
link.download = filename;
|
||||
link.click(); // Triggers download - PERSISTENCE POINT
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 4. Additional Storage Mechanisms
|
||||
|
||||
### 4.1 LRU Cache Storage
|
||||
|
||||
**Files**:
|
||||
- `src/services/memory-manager.ts` (Memory management)
|
||||
- Various processor classes (Cache usage)
|
||||
|
||||
**Purpose**: Caches parsed ASTs and query results for performance
|
||||
|
||||
**Implementation Pattern**:
|
||||
```typescript
|
||||
// Cache key generation
|
||||
const cacheKey = this.lruCache.generateFileCacheKey(filePath, contentHash);
|
||||
|
||||
// Cache retrieval
|
||||
const cachedResult = this.lruCache.getParsedFile(cacheKey);
|
||||
|
||||
// Cache storage
|
||||
this.lruCache.setParsedFile(cacheKey, parseResult); // STORAGE POINT 12
|
||||
```
|
||||
|
||||
### 4.2 LocalStorage Usage
|
||||
|
||||
**Settings Storage** (`src/config/feature-flags.ts`, Lines 104-110):
|
||||
```typescript
|
||||
private saveFlags(): void {
|
||||
try {
|
||||
localStorage.setItem('gitnexus_feature_flags', JSON.stringify(this.flags));
|
||||
// STORAGE POINT 13 - Browser localStorage
|
||||
} catch (error) {
|
||||
console.warn('Failed to save feature flags to localStorage:', error);
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
**Chat History** (`src/lib/chat-history.ts`, Lines 275-291):
|
||||
```typescript
|
||||
private saveSession(session: ChatSession): void {
|
||||
try {
|
||||
localStorage.setItem(this.storageKey, JSON.stringify(session));
|
||||
// STORAGE POINT 14 - Chat persistence
|
||||
} catch (error) {
|
||||
// Handle quota exceeded errors
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### 4.3 IndexedDB (Planned for KuzuDB)
|
||||
|
||||
**Current Status**: Implementation exists but not fully integrated
|
||||
**Files**: `src/core/kuzu/` directory contains KuzuDB integration code
|
||||
**Purpose**: Will replace JSON storage with embedded graph database
|
||||
|
||||
---
|
||||
|
||||
## 5. Data Flow Summary
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
START([Repository Input]) --> STRUCT[Structure Processor]
|
||||
|
||||
STRUCT --> |Creates nodes/relationships| GRAPH1[In-Memory Graph]
|
||||
|
||||
GRAPH1 --> PARSE[Parsing Processor]
|
||||
PARSE --> |AST Storage| AST_MAP[AST Map]
|
||||
PARSE --> |Function Registry| FUNC_TRIE[Function Trie]
|
||||
PARSE --> |Adds definition nodes| GRAPH2[Enhanced Graph]
|
||||
|
||||
GRAPH2 --> IMPORT[Import Processor]
|
||||
AST_MAP --> IMPORT
|
||||
IMPORT --> |Import Map| IMP_MAP[Import Map]
|
||||
IMPORT --> |Adds IMPORTS relationships| GRAPH3[Graph + Imports]
|
||||
|
||||
GRAPH3 --> CALLS[Call Processor]
|
||||
AST_MAP --> CALLS
|
||||
IMP_MAP --> CALLS
|
||||
FUNC_TRIE --> CALLS
|
||||
CALLS --> |Adds CALLS relationships| FINAL_GRAPH[Final Knowledge Graph]
|
||||
|
||||
FINAL_GRAPH --> JSON_EXPORT[JSON Export]
|
||||
JSON_EXPORT --> |JSON.stringify| JSON_STRING[JSON String]
|
||||
JSON_STRING --> |Browser Download| FILE_SYSTEM[File System]
|
||||
|
||||
FINAL_GRAPH --> |Direct object reference| UI[UI Components]
|
||||
|
||||
subgraph "Storage Points"
|
||||
GRAPH1
|
||||
GRAPH2
|
||||
GRAPH3
|
||||
FINAL_GRAPH
|
||||
AST_MAP
|
||||
FUNC_TRIE
|
||||
IMP_MAP
|
||||
JSON_STRING
|
||||
end
|
||||
|
||||
subgraph "Cache Layer"
|
||||
LRU_CACHE[LRU Cache]
|
||||
LOCAL_STORAGE[LocalStorage]
|
||||
end
|
||||
|
||||
PARSE -.-> LRU_CACHE
|
||||
LOCAL_STORAGE -.-> SETTINGS[Settings/Flags]
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 6. Critical Migration Points for Kuzu DB
|
||||
|
||||
### 6.1 Schema Mapping Requirements
|
||||
|
||||
**Current JSON Structure → Kuzu Schema**:
|
||||
|
||||
1. **Nodes Table**:
|
||||
```sql
|
||||
CREATE NODE TABLE IF NOT EXISTS nodes (
|
||||
id STRING PRIMARY KEY,
|
||||
label STRING NOT NULL,
|
||||
properties MAP(STRING, STRING)
|
||||
);
|
||||
```
|
||||
|
||||
2. **Relationships Table**:
|
||||
```sql
|
||||
CREATE REL TABLE IF NOT EXISTS relationships (
|
||||
FROM nodes TO nodes,
|
||||
id STRING,
|
||||
type STRING NOT NULL,
|
||||
properties MAP(STRING, STRING)
|
||||
);
|
||||
```
|
||||
|
||||
### 6.2 Data Transformation Points
|
||||
|
||||
**Every `graph.addNode()` call** → **Kuzu INSERT statement**
|
||||
**Every `graph.addRelationship()` call** → **Kuzu MATCH/CREATE statement**
|
||||
|
||||
### 6.3 Query Transformation Requirements
|
||||
|
||||
**Current**: Linear array searches in `SimpleKnowledgeGraph`
|
||||
**Target**: Cypher queries in KuzuDB
|
||||
|
||||
**Example Transformations**:
|
||||
```typescript
|
||||
// Current: Find nodes by label
|
||||
graph.nodes.filter(n => n.label === 'Function')
|
||||
|
||||
// Target: Cypher query
|
||||
MATCH (n:Function) RETURN n
|
||||
```
|
||||
|
||||
### 6.4 Persistence Layer Changes
|
||||
|
||||
**Current Flow**:
|
||||
1. Build `SimpleKnowledgeGraph` in memory
|
||||
2. Export to JSON string
|
||||
3. Download as file
|
||||
|
||||
**Target Flow**:
|
||||
1. Stream data directly to KuzuDB during processing
|
||||
2. Persist to IndexedDB automatically
|
||||
3. Export via Cypher queries
|
||||
|
||||
---
|
||||
|
||||
## 7. Implementation Recommendations
|
||||
|
||||
### 7.1 Migration Strategy
|
||||
|
||||
1. **Phase 1**: Create parallel KuzuDB storage alongside existing JSON
|
||||
2. **Phase 2**: Implement streaming ingestion (write to Kuzu during pipeline)
|
||||
3. **Phase 3**: Replace SimpleKnowledgeGraph with KuzuDB queries
|
||||
4. **Phase 4**: Remove JSON export/import (keep as backup option)
|
||||
|
||||
### 7.2 Critical Considerations
|
||||
|
||||
1. **Referential Integrity**: Kuzu enforces relationships, JSON doesn't
|
||||
2. **Transaction Boundaries**: Kuzu needs explicit transactions
|
||||
3. **Query Performance**: Index strategy for common access patterns
|
||||
4. **Memory Management**: Kuzu handles memory, current system uses manual arrays
|
||||
5. **Concurrent Access**: Kuzu supports concurrent reads, current system is single-threaded
|
||||
|
||||
### 7.3 Testing Strategy
|
||||
|
||||
1. **Data Integrity**: Compare JSON export with Kuzu export for identical results
|
||||
2. **Performance**: Benchmark ingestion and query performance
|
||||
3. **Memory Usage**: Monitor memory consumption during large repository processing
|
||||
4. **Error Handling**: Test transaction rollback and recovery scenarios
|
||||
|
||||
---
|
||||
|
||||
## Conclusion
|
||||
|
||||
This document provides the complete technical foundation for migrating GitNexus from JSON-based storage to KuzuDB. Every storage point, data transformation, and persistence mechanism has been identified and documented. The migration should focus on replacing the `SimpleKnowledgeGraph` implementation while maintaining the exact same data structures and relationships in the new KuzuDB schema.
|
||||
@@ -1,253 +0,0 @@
|
||||
# Phase 2: Parallel Storage Implementation - Complete! 🎉
|
||||
|
||||
## Overview
|
||||
|
||||
Successfully implemented the **Parallel Storage** phase of the KuzuDB migration plan, enabling dual-write functionality where data is written to both JSON (primary) and KuzuDB (secondary) storage systems simultaneously.
|
||||
|
||||
## ✅ **Components Implemented**
|
||||
|
||||
### 1. **KuzuProcessorBase** (`src/core/ingestion/kuzu-processor-base.ts`)
|
||||
|
||||
**Core Features:**
|
||||
- **Dual-Write Pattern**: Seamless writes to both JSON and KuzuDB
|
||||
- **Transaction Management**: Begin, commit, and rollback transaction support
|
||||
- **Error Handling**: Graceful degradation when KuzuDB fails
|
||||
- **Performance Monitoring**: Detailed statistics and timing metrics
|
||||
- **Data Validation**: Consistency checks between storage systems
|
||||
- **Feature Flag Integration**: Respects `isKuzuDBEnabled()` settings
|
||||
|
||||
**Key Methods:**
|
||||
- `addNodeDualWrite()` - Writes nodes to both storages
|
||||
- `addRelationshipDualWrite()` - Writes relationships to both storages
|
||||
- `beginTransaction()` / `commitTransaction()` / `rollbackTransaction()`
|
||||
- `initializeKuzuDB()` - Sets up KuzuDB connection
|
||||
- `validateNodeConsistency()` / `validateRelationshipConsistency()`
|
||||
|
||||
### 2. **Enhanced StructureProcessor**
|
||||
|
||||
**Modifications:**
|
||||
- ✅ Extends `KuzuProcessorBase` for dual-write capability
|
||||
- ✅ Async `process()` method with KuzuDB initialization
|
||||
- ✅ Dual-write support for Project, Folder, and File nodes
|
||||
- ✅ Dual-write support for CONTAINS relationships
|
||||
- ✅ Transaction boundaries with commit/rollback
|
||||
- ✅ Comprehensive error handling and statistics
|
||||
|
||||
**Dual-Write Flow:**
|
||||
1. Initialize KuzuDB connection
|
||||
2. Create project node → write to JSON + KuzuDB
|
||||
3. Create directory nodes → write to JSON + KuzuDB
|
||||
4. Create file nodes → write to JSON + KuzuDB
|
||||
5. Create CONTAINS relationships → write to JSON + KuzuDB
|
||||
6. Commit KuzuDB transaction
|
||||
7. Log detailed statistics
|
||||
|
||||
### 3. **Enhanced ParsingProcessor**
|
||||
|
||||
**Modifications:**
|
||||
- ✅ Extends `KuzuProcessorBase` for dual-write capability
|
||||
- ✅ Async definition processing with KuzuDB writes
|
||||
- ✅ Dual-write support for Function, Class, Method, Variable, Interface, Type nodes
|
||||
- ✅ Dual-write support for INHERITS, IMPLEMENTS, IMPORTS relationships
|
||||
- ✅ Transaction boundaries with automatic commit
|
||||
- ✅ Batch processing optimization
|
||||
|
||||
**Dual-Write Flow:**
|
||||
1. Initialize KuzuDB connection
|
||||
2. Process each file's definitions
|
||||
3. Create definition nodes → write to JSON + KuzuDB
|
||||
4. Create containment relationships → write to JSON + KuzuDB
|
||||
5. Create inheritance/implementation relationships → write to JSON + KuzuDB
|
||||
6. Commit KuzuDB transaction
|
||||
7. Log processing statistics
|
||||
|
||||
### 4. **Enhanced ImportProcessor**
|
||||
|
||||
**Modifications:**
|
||||
- ✅ Extends `KuzuProcessorBase` for dual-write capability
|
||||
- ✅ Async import relationship creation
|
||||
- ✅ Dual-write support for IMPORTS relationships
|
||||
- ✅ Transaction management with rollback support
|
||||
- ✅ Enhanced error handling and progress tracking
|
||||
|
||||
**Dual-Write Flow:**
|
||||
1. Initialize KuzuDB connection
|
||||
2. Process imports for each file
|
||||
3. Create IMPORTS relationships → write to JSON + KuzuDB
|
||||
4. Commit KuzuDB transaction
|
||||
5. Log import resolution statistics
|
||||
|
||||
### 5. **Enhanced CallProcessor**
|
||||
|
||||
**Modifications:**
|
||||
- ✅ Extends `KuzuProcessorBase` for dual-write capability
|
||||
- ✅ Async call relationship creation
|
||||
- ✅ Dual-write support for CALLS relationships
|
||||
- ✅ 3-stage resolution strategy maintained
|
||||
- ✅ Transaction boundaries and error handling
|
||||
|
||||
**Dual-Write Flow:**
|
||||
1. Initialize KuzuDB connection
|
||||
2. Extract function calls from AST
|
||||
3. Resolve calls using 3-stage strategy
|
||||
4. Create CALLS relationships → write to JSON + KuzuDB
|
||||
5. Commit KuzuDB transaction
|
||||
6. Log call resolution statistics
|
||||
|
||||
## 🏗️ **Architecture Highlights**
|
||||
|
||||
### **Dual-Write Pattern Implementation**
|
||||
```typescript
|
||||
// JSON write (primary - always succeeds)
|
||||
jsonGraph.addNode(node);
|
||||
|
||||
// KuzuDB write (secondary - graceful failure)
|
||||
if (this.kuzuGraph) {
|
||||
try {
|
||||
this.kuzuGraph.addNode(node);
|
||||
} catch (kuzuError) {
|
||||
console.warn('KuzuDB write failed:', kuzuError);
|
||||
// Continue processing - JSON is primary storage
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### **Transaction Management**
|
||||
```typescript
|
||||
// Begin transaction
|
||||
await this.beginTransaction();
|
||||
|
||||
try {
|
||||
// Perform operations
|
||||
await this.addNodeDualWrite(graph, node);
|
||||
await this.addRelationshipDualWrite(graph, relationship);
|
||||
|
||||
// Commit transaction
|
||||
await this.commitTransaction();
|
||||
} catch (error) {
|
||||
// Rollback on failure
|
||||
await this.rollbackTransaction();
|
||||
throw error;
|
||||
}
|
||||
```
|
||||
|
||||
### **Statistics and Monitoring**
|
||||
- **Nodes processed**: Total nodes written to JSON
|
||||
- **KuzuDB nodes written**: Successful KuzuDB writes
|
||||
- **KuzuDB errors**: Failed KuzuDB operations
|
||||
- **Success rate**: Percentage of successful dual-writes
|
||||
- **Processing time**: Total time spent on operations
|
||||
- **Validation errors**: Data consistency issues detected
|
||||
|
||||
## 📊 **Key Benefits Achieved**
|
||||
|
||||
### **1. Zero Breaking Changes**
|
||||
- All existing processors maintain their original interfaces
|
||||
- JSON storage remains primary - system continues working even if KuzuDB fails
|
||||
- Backward compatibility with all existing code
|
||||
|
||||
### **2. Production-Ready Error Handling**
|
||||
- KuzuDB failures don't break the ingestion pipeline
|
||||
- Graceful degradation to JSON-only mode
|
||||
- Comprehensive error logging and categorization
|
||||
- Transaction rollback on critical failures
|
||||
|
||||
### **3. Performance Optimization**
|
||||
- Batch processing for optimal KuzuDB performance
|
||||
- Async operations with proper error boundaries
|
||||
- Transaction boundaries reduce database overhead
|
||||
- Detailed performance monitoring and statistics
|
||||
|
||||
### **4. Data Consistency**
|
||||
- Dual-write ensures both storages have the same data
|
||||
- Transaction management prevents partial writes
|
||||
- Validation hooks for consistency checking
|
||||
- Rollback capabilities for data integrity
|
||||
|
||||
### **5. Feature Flag Integration**
|
||||
- Respects `isKuzuDBEnabled()` configuration
|
||||
- Can be enabled/disabled without code changes
|
||||
- Gradual rollout capabilities
|
||||
- A/B testing support
|
||||
|
||||
## 🔄 **Integration Points**
|
||||
|
||||
### **Pipeline Integration**
|
||||
All processors now support the enhanced dual-write pattern:
|
||||
|
||||
```typescript
|
||||
// Structure Phase
|
||||
const structureProcessor = new StructureProcessor({ enableKuzuDB: true });
|
||||
await structureProcessor.process(graph, structureInput);
|
||||
|
||||
// Parsing Phase
|
||||
const parsingProcessor = new ParsingProcessor({ enableKuzuDB: true });
|
||||
await parsingProcessor.process(graph, parsingInput);
|
||||
|
||||
// Import Phase
|
||||
const importProcessor = new ImportProcessor({ enableKuzuDB: true });
|
||||
await importProcessor.process(graph, astMap, fileContents);
|
||||
|
||||
// Call Phase
|
||||
const callProcessor = new CallProcessor(functionTrie, { enableKuzuDB: true });
|
||||
await callProcessor.process(graph, astMap, importMap);
|
||||
```
|
||||
|
||||
### **Configuration Options**
|
||||
```typescript
|
||||
interface KuzuProcessorOptions {
|
||||
enableKuzuDB?: boolean; // Enable/disable KuzuDB integration
|
||||
batchSize?: number; // Batch size for optimal performance
|
||||
autoCommit?: boolean; // Automatic transaction commits
|
||||
enableValidation?: boolean; // Data consistency validation
|
||||
}
|
||||
```
|
||||
|
||||
## 📈 **Performance Expectations**
|
||||
|
||||
### **Memory Usage**
|
||||
- Minimal additional memory overhead (~5-10%)
|
||||
- Transaction batching prevents memory bloat
|
||||
- Graceful handling of large codebases
|
||||
|
||||
### **Processing Time**
|
||||
- Expected 10-20% increase in processing time
|
||||
- Batch operations optimize KuzuDB performance
|
||||
- Async operations prevent blocking
|
||||
|
||||
### **Error Resilience**
|
||||
- 100% reliability for JSON storage (primary)
|
||||
- Graceful degradation for KuzuDB failures
|
||||
- No data loss even with KuzuDB issues
|
||||
|
||||
## 🚀 **Ready for Phase 3**
|
||||
|
||||
The parallel storage implementation provides a solid foundation for **Phase 3: Query Migration**, where we'll:
|
||||
|
||||
1. **Implement Query Abstraction Layer**: Create unified query interface
|
||||
2. **Add Query Routing Logic**: Route queries to appropriate storage
|
||||
3. **Performance Comparison Tools**: A/B test JSON vs KuzuDB queries
|
||||
4. **Query Result Validation**: Ensure consistent results between storages
|
||||
|
||||
## 📁 **Files Modified/Created**
|
||||
|
||||
### **New Files**
|
||||
- `src/core/ingestion/kuzu-processor-base.ts` - Base class for dual-write pattern
|
||||
|
||||
### **Modified Files**
|
||||
- `src/core/ingestion/structure-processor.ts` - Added KuzuDB dual-write support
|
||||
- `src/core/ingestion/parsing-processor.ts` - Added KuzuDB dual-write support
|
||||
- `src/core/ingestion/import-processor.ts` - Added KuzuDB dual-write support
|
||||
- `src/core/ingestion/call-processor.ts` - Added KuzuDB dual-write support
|
||||
|
||||
## 🎯 **Success Metrics**
|
||||
|
||||
- ✅ **100% Backward Compatibility**: All existing functionality preserved
|
||||
- ✅ **Graceful Error Handling**: KuzuDB failures don't break the system
|
||||
- ✅ **Transaction Safety**: Data integrity maintained with rollback support
|
||||
- ✅ **Performance Monitoring**: Comprehensive statistics and metrics
|
||||
- ✅ **Feature Flag Ready**: Can be enabled/disabled via configuration
|
||||
- ✅ **Production Quality**: Error handling, logging, and monitoring
|
||||
|
||||
The dual-write pattern is now fully implemented and ready for production deployment! 🚀
|
||||
|
||||
@@ -1,359 +1,486 @@
|
||||
# GitNexus - Fully Client sided Knowledge Graph Generator and Graph RAG Agent
|
||||
# GitNexus V2
|
||||
|
||||
**Zero-Server, Graph-Based Code Intelligence Engine**
|
||||
Works fully in-browser through WebAssembly. (DB engine, Embeddings model, AST parsing, all happens inside browser)
|
||||
|
||||
GitNexus is a privacy-focused, zero-server knowledge graph generator that runs entirely in your browser. It transforms codebases into interactive knowledge graphs using advanced AST parsing, multi-threaded Web Workers, and an embedded KuzuDB WASM database. Features a Graph RAG agent for intelligent code exploration through natural language queries using cypher queries executed directly against the in-browser graph database.
|
||||
https://github.com/user-attachments/assets/2fb7c522-20d1-48f6-9583-36c3969aa4dc
|
||||
|
||||
https://github.com/user-attachments/assets/6f13bd45-d6e9-4f4e-a360-ceb66f41c741
|
||||
https://gitnexus.vercel.app
|
||||
Being client sided, it costs me zero to deploy, so you can use it for free :-) (would love a ⭐ though)
|
||||
|
||||
## Current Work in Progress:
|
||||
> *Like DeepWiki, but deeper.* 😉
|
||||
|
||||
- Ollama support
|
||||
- Export as csv ( for both node and relation table )
|
||||
DeepWiki helps you *understand* code. GitNexus lets you *analyze* it—because a knowledge graph tracks every dependency, call chain, and relationship.
|
||||
|
||||
That's the difference between:
|
||||
- "What does this function do?" → *understanding*
|
||||
- "What breaks if I change this function?" → *analysis*
|
||||
|
||||
## Features
|
||||
**Some quick tech jargon:**
|
||||
- **Enhanced Search**: BM25 + Semantic + 1-hop graph expansion via Cypher
|
||||
- **Full WASM Stack**: Tree-sitter parsing + KuzuDB graph database, all in-browser
|
||||
- **Repo Map**: Complete code knowledge graph with CALLS, IMPORTS, EXTENDS relations
|
||||
- **Vector Index**: HNSW embeddings for semantic similarity search
|
||||
- **Cypher Queries**: Relational analysis for accurate context retrieval
|
||||
- **Grounded AI**: Every answer cites `[[file:line]]` as proof
|
||||
|
||||
**Code Analysis**
|
||||
**What you can do:**
|
||||
|
||||
- Analyze GitHub repositories or ZIP files
|
||||
- Support for TypeScript, JavaScript, Python
|
||||
- Interactive graph visualization with D3.js
|
||||
- File filtering and directory selection
|
||||
- Export results as JSON/CSV
|
||||
| Capability | Description |
|
||||
|------------|-------------|
|
||||
| **Codebase-wide audits** | Find layer violations, forbidden dependencies |
|
||||
| **Blast radius analysis** | See every function affected by a change |
|
||||
| **Dead code detection** | Identify orphaned nodes with zero incoming calls |
|
||||
| **Dependency tracing** | Follow import chains across the entire codebase |
|
||||
| **AI analyses with citations** | Ask questions, analyze, get answers with `[[file:line]]` proof |
|
||||
|
||||
**AI Chat**
|
||||
**100% client-side.** Your code never leaves your browser.
|
||||
|
||||
- Multiple LLM providers (OpenAI, Anthropic, Gemini, Azure)
|
||||
- Query code structure and relationships
|
||||
- Context-aware conversations
|
||||
- Graph-based code search
|
||||
**Supports:** TypeScript, JavaScript, Python (Go, Java, C in progress)
|
||||
|
||||
**Processing**
|
||||
<img width="2550" height="1343" alt="gitnexus_img" src="https://github.com/user-attachments/assets/cc5d637d-e0e5-48e6-93ff-5bcfdb929285" />
|
||||
|
||||
- Four-pass analysis: structure → parsing → imports → calls
|
||||
- Parallel processing with Web Workers
|
||||
- AST-based code extraction using Tree-sitter
|
||||
- Memory-efficient caching
|
||||
---
|
||||
|
||||
## Architecture
|
||||
## 🔍 The Problem with AI Coding Tools
|
||||
|
||||
```mermaid
|
||||
graph TB
|
||||
UI[React UI Layer] --> EM[Engine Manager]
|
||||
EM --> LEG[Legacy Engine]
|
||||
EM --> NG[Next-Gen Engine - WIP]
|
||||
|
||||
subgraph "Legacy Engine (Production Ready)"
|
||||
LEG --> GP[Sequential Pipeline]
|
||||
GP --> SP[Single-threaded Parser]
|
||||
GP --> MEM[In-Memory Graph Store]
|
||||
MEM --> JSON[JSON Export]
|
||||
end
|
||||
|
||||
subgraph "Next-Gen Engine (Work in Progress)"
|
||||
NG --> PP[Parallel Pipeline]
|
||||
PP --> WP[Web Worker Pool]
|
||||
PP --> KDB[KuzuDB WASM]
|
||||
KDB --> CYP[Cypher Queries]
|
||||
CYP --> RAG[Graph RAG Agent - WIP]
|
||||
end
|
||||
|
||||
subgraph "Core Technologies"
|
||||
TS[Tree-sitter WASM]
|
||||
D3[D3.js Force Simulation]
|
||||
LC[LangChain ReAct Agents]
|
||||
IDB[IndexedDB Persistence]
|
||||
end
|
||||
```
|
||||
Tools like **Cursor**, **Claude Code**, **Cline**, **Roo Code**, and **Windsurf** are powerful—but they share a fundamental limitation: **they don't truly know your codebase structure**.
|
||||
|
||||
**Tech Stack**:
|
||||
| Tool | Context Strategy | The Gap |
|
||||
|------|------------------|---------|
|
||||
| **Cursor** | Files in tabs + embeddings | No call graph. Can't trace "what calls this?" |
|
||||
| **Claude Code** | File search + grep | Text-based. Misses semantic connections |
|
||||
| **Cline/Roo Code** | Repo map + tree-sitter | Static structure. No runtime dependencies tracked |
|
||||
| **Windsurf** | Cascade context | Limited dependency depth |
|
||||
|
||||
- **Frontend**: React 18 + TypeScript + Vite + D3.js force simulation
|
||||
- **Parsing**: Tree-sitter WASM parsers (TypeScript, JavaScript, Python)
|
||||
- **Concurrency**: Web Worker Pool with Comlink for thread-safe communication
|
||||
- **Caching**: LRU-based AST cache with memory management and eviction policies
|
||||
- **AI**: LangChain.js ReAct agents with tool-augmented reasoning
|
||||
- **Database**: KuzuDB WASM integration (WIP) + IndexedDB persistence
|
||||
- **Graph RAG**: Cypher query generation for knowledge graph reasoning (WIP)
|
||||
**What happens:**
|
||||
1. AI edits `UserService.validate()`
|
||||
2. Doesn't know 47 functions depend on its return type
|
||||
3. **Breaking changes ship** 💥
|
||||
|
||||
## Four-Pass Ingestion Pipeline
|
||||
### The Solution: Graph Coverage
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
START([Repository Input]) --> PASS1
|
||||
|
||||
subgraph PASS1 ["Pass 1: Structure Analysis"]
|
||||
P1A[Recursive Directory Traversal] --> P1B[File Type Classification]
|
||||
P1B --> P1C[Project/Folder/File Nodes]
|
||||
P1C --> P1D[CONTAINS Relationships]
|
||||
end
|
||||
|
||||
subgraph PASS2 ["Pass 2: Code Parsing & AST"]
|
||||
P2A[Tree-sitter WASM Init] --> P2B[Grammar Loading]
|
||||
P2B --> P2C[AST Generation]
|
||||
P2C --> P2D[Symbol Extraction]
|
||||
P2D --> P2E[LRU Cache Storage]
|
||||
end
|
||||
|
||||
subgraph PASS3 ["Pass 3: Import Resolution"]
|
||||
P3A[Import Statement Extraction] --> P3B[Module Path Resolution]
|
||||
P3B --> P3C[Cross-Reference Tables]
|
||||
P3C --> P3D[IMPORTS Relationships]
|
||||
end
|
||||
|
||||
subgraph PASS4 ["Pass 4: Call Graph Analysis"]
|
||||
P4A[Function Call Pattern Matching] --> P4B[Exact Match via Import Map]
|
||||
P4B --> P4C[Fuzzy Match + Levenshtein]
|
||||
P4C --> P4D[CALLS Relationships]
|
||||
end
|
||||
|
||||
PASS1 --> PASS2
|
||||
PASS2 --> PASS3
|
||||
PASS3 --> PASS4
|
||||
PASS4 --> END([Knowledge Graph])
|
||||
|
||||
classDef passBox fill:#e1f5fe,stroke:#01579b,stroke-width:2px,color:#000
|
||||
classDef startEnd fill:#c8e6c9,stroke:#2e7d32,stroke-width:3px,color:#000
|
||||
classDef step fill:#fff3e0,stroke:#ef6c00,stroke-width:1px,color:#000
|
||||
|
||||
class PASS1,PASS2,PASS3,PASS4 passBox
|
||||
class START,END startEnd
|
||||
class P1A,P1B,P1C,P1D,P2A,P2B,P2C,P2D,P2E,P3A,P3B,P3C,P3D,P4A,P4B,P4C,P4D step
|
||||
```
|
||||
|
||||
### Data Flow & Storage Architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
START([Repository Input]) --> STRUCT[Structure Processor]
|
||||
|
||||
STRUCT --> |Creates nodes/relationships| GRAPH1[In-Memory Graph]
|
||||
|
||||
GRAPH1 --> PARSE[Parsing Processor]
|
||||
PARSE --> |AST Storage| AST_MAP[AST Map]
|
||||
PARSE --> |Function Registry| FUNC_TRIE[Function Trie]
|
||||
PARSE --> |Adds definition nodes| GRAPH2[Enhanced Graph]
|
||||
|
||||
GRAPH2 --> IMPORT[Import Processor]
|
||||
AST_MAP --> IMPORT
|
||||
IMPORT --> |Import Map| IMP_MAP[Import Map]
|
||||
IMPORT --> |Adds IMPORTS relationships| GRAPH3[Graph + Imports]
|
||||
|
||||
GRAPH3 --> CALLS[Call Processor]
|
||||
AST_MAP --> CALLS
|
||||
IMP_MAP --> CALLS
|
||||
FUNC_TRIE --> CALLS
|
||||
CALLS --> |Adds CALLS relationships| FINAL_GRAPH[Final Knowledge Graph]
|
||||
|
||||
FINAL_GRAPH --> JSON_EXPORT[JSON Export]
|
||||
JSON_EXPORT --> |JSON.stringify| JSON_STRING[JSON String]
|
||||
JSON_STRING --> |Browser Download| FILE_SYSTEM[File System]
|
||||
|
||||
FINAL_GRAPH --> |Direct object reference| UI[UI Components]
|
||||
|
||||
subgraph "Storage Points"
|
||||
GRAPH1
|
||||
GRAPH2
|
||||
GRAPH3
|
||||
FINAL_GRAPH
|
||||
AST_MAP
|
||||
FUNC_TRIE
|
||||
IMP_MAP
|
||||
JSON_STRING
|
||||
end
|
||||
|
||||
subgraph "Cache Layer"
|
||||
LRU_CACHE[LRU Cache]
|
||||
LOCAL_STORAGE[LocalStorage]
|
||||
end
|
||||
|
||||
PARSE -.-> LRU_CACHE
|
||||
LOCAL_STORAGE -.-> SETTINGS[Settings/Flags]
|
||||
```
|
||||
|
||||
### Technical Implementation Details
|
||||
|
||||
**Pass 1: Structure Analysis**
|
||||
|
||||
- Implements recursive directory traversal with configurable depth limits
|
||||
- File type detection using MIME types and extension mapping
|
||||
- Creates hierarchical node structure with parent-child relationships
|
||||
- Establishes CONTAINS relationships for project organization
|
||||
|
||||
**Pass 2: Code Parsing & AST Extraction**
|
||||
|
||||
- Initializes Tree-sitter WASM parsers with language-specific grammars
|
||||
- Generates Abstract Syntax Trees for each source file
|
||||
- Implements AST traversal algorithms to extract code symbols
|
||||
- **LRU Cache System**: Memory-efficient AST storage with configurable eviction policies
|
||||
- **Parallel Processing**: Web Worker Pool distributes parsing across multiple threads
|
||||
- **Memory Management**: Automatic cleanup and garbage collection for large codebases
|
||||
|
||||
**Pass 3: Import Resolution**
|
||||
|
||||
- Extracts import/require statements using AST pattern matching
|
||||
- Implements module resolution algorithms (Node.js, ES6, Python)
|
||||
- Builds cross-reference tables for dependency mapping
|
||||
- Handles relative/absolute path resolution with fallback strategies
|
||||
|
||||
**Pass 4: Call Graph Analysis**
|
||||
|
||||
- **Stage 1**: Exact function call matching using import resolution data
|
||||
- **Stage 2**: Fuzzy matching with Levenshtein distance for unresolved calls
|
||||
- **Stage 3**: Heuristic-based matching for dynamic calls and method chaining
|
||||
- Creates CALLS relationships with confidence scoring
|
||||
|
||||
## Getting Started
|
||||
|
||||
**Prerequisites**: Node.js 18+, API keys for AI features
|
||||
|
||||
```bash
|
||||
git clone <repository-url>
|
||||
cd gitnexus
|
||||
npm install
|
||||
npm run dev
|
||||
```
|
||||
|
||||
Open http://localhost:5173
|
||||
|
||||
**Configuration**
|
||||
|
||||
- GitHub token (optional): Increases rate limit to 5,000/hour
|
||||
- AI API keys: OpenAI, Anthropic, Gemini, or Azure OpenAI
|
||||
- Performance: Set file limits and directory filters
|
||||
|
||||
## Usage
|
||||
|
||||
**Analyze Repository**
|
||||
|
||||
1. Enter GitHub URL or upload ZIP file
|
||||
2. Set filters (optional): directories, file patterns, size limits
|
||||
3. Click "Analyze" and wait for processing
|
||||
4. Explore the interactive graph
|
||||
|
||||
**AI Chat**
|
||||
|
||||
1. Configure API key in settings
|
||||
2. Ask questions about the codebase:
|
||||
- "What functions are in main.py?"
|
||||
- "Show classes that inherit from BaseClass"
|
||||
- "How does authentication work?"
|
||||
|
||||
**Export Data**
|
||||
|
||||
- Click Export button to download graph as JSON/CSV
|
||||
|
||||
## Advanced Features & Work in Progress
|
||||
|
||||
### Web Worker Pool Architecture
|
||||
A knowledge graph tracks **actual relationships**, not just file contents:
|
||||
|
||||
```mermaid
|
||||
graph LR
|
||||
MT[Main Thread] --> WM[Worker Manager]
|
||||
WM --> W1[Worker 1<br/>Tree-sitter Parser]
|
||||
WM --> W2[Worker 2<br/>Tree-sitter Parser]
|
||||
WM --> W3[Worker N<br/>Tree-sitter Parser]
|
||||
|
||||
W1 --> AST1[AST Cache]
|
||||
W2 --> AST2[AST Cache]
|
||||
W3 --> AST3[AST Cache]
|
||||
|
||||
AST1 --> LRU[LRU Eviction Policy]
|
||||
AST2 --> LRU
|
||||
AST3 --> LRU
|
||||
EDIT[AI wants to edit UserService.validate] --> QUERY[Graph Query: What depends on this?]
|
||||
QUERY --> DEPS["47 callers across 12 files"]
|
||||
DEPS --> SAFE[AI sees full blast radius first]
|
||||
```
|
||||
|
||||
### LRU Cache Implementation
|
||||
**Current state:** GitNexus is a standalone tool—a better DeepWiki that's 100% client-side with graph-powered analysis.
|
||||
|
||||
- **Memory-bounded AST storage** with configurable size limits (default: 1000 entries)
|
||||
- **Automatic eviction policies** based on access patterns and memory pressure
|
||||
- **Thread-safe operations** across Web Worker boundaries using Comlink
|
||||
- **Cache hit optimization** for repeated file analysis and import resolution
|
||||
- **Garbage collection integration** with browser memory management APIs
|
||||
**Future goal (MCP):** Expose GitNexus as an MCP server so tools like Cursor and Claude Code can query it for accurate context. They ask "what calls X?", GitNexus returns the actual call graph. No more guessing.
|
||||
|
||||
### KuzuDB Integration Status (Work in Progress)
|
||||
|
||||
```mermaid
|
||||
graph TD
|
||||
APP[Application Layer] --> RAG[Graph RAG Agent]
|
||||
RAG --> CYP[Cypher Query Generator]
|
||||
CYP --> KDB[KuzuDB WASM Engine]
|
||||
KDB --> IDB[IndexedDB Persistence]
|
||||
|
||||
subgraph STATUS ["Current Status"]
|
||||
IMPL[KuzuDB WASM Integration - Complete]
|
||||
PERS[IndexedDB Persistence - Complete]
|
||||
SCHEMA[Graph Schema Definition - Complete]
|
||||
QUERY[Cypher Query Execution - WIP]
|
||||
AGENT[Graph RAG Agent - WIP]
|
||||
end
|
||||
|
||||
classDef complete fill:#c8e6c9,stroke:#2e7d32,stroke-width:2px
|
||||
classDef wip fill:#fff3e0,stroke:#f57c00,stroke-width:2px
|
||||
classDef main fill:#e3f2fd,stroke:#1976d2,stroke-width:2px
|
||||
|
||||
class IMPL,PERS,SCHEMA complete
|
||||
class QUERY,AGENT wip
|
||||
class APP,RAG,CYP,KDB,IDB main
|
||||
```
|
||||
---
|
||||
|
||||
**Implementation Status**:
|
||||
|
||||
- ✅ **KuzuDB WASM Engine**: Fully integrated embedded graph database
|
||||
- ✅ **Graph Schema**: Node and relationship type definitions implemented
|
||||
- ✅ **Data Ingestion**: Knowledge graph storage in KuzuDB format
|
||||
- 🚧 **Cypher Query Engine**: Query execution layer under development
|
||||
- 🚧 **Graph RAG Agent**: AI agent with graph querying capabilities (blocked by Cypher integration)
|
||||
|
||||
**Current Limitation**: The Graph RAG agent cannot execute sophisticated graph queries because the Cypher query execution layer is still being implemented. Basic AI chat works with in-memory graph traversal, but advanced graph reasoning requires the KuzuDB Cypher integration to be completed.
|
||||
|
||||
### Dual-Engine Architecture
|
||||
|
||||
- **Legacy Engine**: Production-ready single-threaded processing with JSON storage
|
||||
- **Next-Gen Engine**: Parallel processing with KuzuDB persistence (4-8x performance improvement)
|
||||
- **Automatic Fallback**: System gracefully degrades to legacy engine if next-gen fails
|
||||
- **Runtime Switching**: Users can toggle between engines without data loss
|
||||
|
||||
## Deployment
|
||||
## 🚀 Quick Start
|
||||
|
||||
```bash
|
||||
npm run build
|
||||
npm run preview
|
||||
git clone <repository-url>
|
||||
cd gitnexus
|
||||
npm install
|
||||
npm run dev
|
||||
```
|
||||
|
||||
**Environment Variables**
|
||||
Open http://localhost:5173, drag & drop a ZIP of your codebase, and start exploring.
|
||||
|
||||
```env
|
||||
VITE_OPENAI_API_KEY=sk-...
|
||||
VITE_DEFAULT_MAX_FILES=500
|
||||
VITE_ENABLE_DEBUG_LOGGING=false
|
||||
---
|
||||
|
||||
## 🏗️ Indexing Architecture
|
||||
|
||||
Two-phase indexing: **Knowledge Graph** (blocking) → **Embeddings** (background).
|
||||
|
||||
### Phase 1-5: Knowledge Graph Creation
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph P1["Phase 1: Extract (0-15%)"]
|
||||
E1[Decompress ZIP] --> E2[Collect file paths]
|
||||
end
|
||||
|
||||
subgraph P2["Phase 2: Structure (15-30%)"]
|
||||
S1[Build folder tree] --> S2[Create CONTAINS edges]
|
||||
end
|
||||
|
||||
subgraph P3["Phase 3: Parse (30-70%)"]
|
||||
PA1[Load Tree-sitter WASM] --> PA2[Generate ASTs]
|
||||
PA2 --> PA3[Extract symbols]
|
||||
PA3 --> PA4[Populate Symbol Table]
|
||||
end
|
||||
|
||||
subgraph P4["Phase 4: Imports (70-82%)"]
|
||||
I1[Find import statements] --> I2[Resolve paths]
|
||||
I2 --> I3[Create IMPORTS edges]
|
||||
end
|
||||
|
||||
subgraph P5["Phase 5: Calls + Heritage (82-100%)"]
|
||||
C1[Find function calls] --> C2[Resolve via Symbol Table]
|
||||
C2 --> C3[Create CALLS edges]
|
||||
C3 --> H1[Find extends/implements]
|
||||
H1 --> H2[Create EXTENDS/IMPLEMENTS edges]
|
||||
end
|
||||
|
||||
P1 --> P2 --> P3 --> P4 --> P5
|
||||
P5 --> DB[(KuzuDB WASM)]
|
||||
DB --> READY[Graph Ready!]
|
||||
```
|
||||
|
||||
## Security & Privacy
|
||||
### Symbol Table: Dual HashMap
|
||||
|
||||
Resolution strategy for function calls:
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
CALL[Found call: validateUser] --> CHECK1{In Import Map?}
|
||||
CHECK1 -->|Yes| FOUND1[Use imported definition]
|
||||
CHECK1 -->|No| CHECK2{In Current File?}
|
||||
CHECK2 -->|Yes| FOUND2[Use local definition]
|
||||
CHECK2 -->|No| CHECK3{Global Search}
|
||||
CHECK3 -->|Found| FOUND3[Use first match]
|
||||
CHECK3 -->|Not Found| SKIP[Skip - unresolved]
|
||||
|
||||
FOUND1 --> EDGE[Create CALLS edge]
|
||||
FOUND2 --> EDGE
|
||||
FOUND3 --> EDGE
|
||||
```
|
||||
|
||||
**Data structure:**
|
||||
```
|
||||
File-Scoped: Map<FilePath, Map<SymbolName, NodeID>>
|
||||
Global: Map<SymbolName, SymbolDefinition[]>
|
||||
```
|
||||
|
||||
### Phase 6+: Background Embeddings
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
subgraph BG["Background (Non-blocking)"]
|
||||
M1[Load snowflake-arctic-embed-xs] --> M2[Initialize WebGPU/WASM]
|
||||
M2 --> E1[Batch embed nodes]
|
||||
E1 --> E2[INSERT into CodeEmbedding table]
|
||||
E2 --> V1[Create HNSW Vector Index]
|
||||
V1 --> B1[Build BM25 Index]
|
||||
end
|
||||
|
||||
BG --> AI[AI Search Ready!]
|
||||
```
|
||||
|
||||
User can explore the graph during embedding. AI features unlock when complete.
|
||||
|
||||
---
|
||||
|
||||
## 📊 Graph Schema
|
||||
|
||||
### Node Types
|
||||
|
||||
| Label | Description | Properties |
|
||||
|-------|-------------|------------|
|
||||
| `Folder` | Directory | `name`, `filePath` |
|
||||
| `File` | Source file | `name`, `filePath`, `language` |
|
||||
| `Function` | Function def | `name`, `filePath`, `startLine`, `endLine`, `isExported` |
|
||||
| `Class` | Class def | `name`, `filePath`, `startLine`, `endLine` |
|
||||
| `Interface` | Interface def | `name`, `filePath`, `startLine`, `endLine` |
|
||||
| `Method` | Class method | `name`, `filePath`, `startLine`, `endLine` |
|
||||
| `CodeElement` | Generic symbol | `name`, `filePath` |
|
||||
|
||||
### Relationship Table: `CodeRelation`
|
||||
|
||||
Single edge table with `type` property:
|
||||
|
||||
| Type | From | To | Description |
|
||||
|------|------|-----|-------------|
|
||||
| `CONTAINS` | Folder | File/Folder | Directory structure |
|
||||
| `DEFINES` | File | Function/Class/etc | Code definitions |
|
||||
| `IMPORTS` | File | File | Module dependencies |
|
||||
| `CALLS` | Function/Method | Function/Method | Call graph |
|
||||
| `EXTENDS` | Class | Class | Inheritance |
|
||||
| `IMPLEMENTS` | Class | Interface | Interface implementation |
|
||||
|
||||
---
|
||||
|
||||
## 🛠️ Agent Tools Architecture
|
||||
|
||||
The LangChain ReAct agent has **5 tools** for code exploration. These tools **use the graph** built during indexing.
|
||||
|
||||
### Tool 1: `search` — Hybrid Search with Graph Context
|
||||
|
||||
Combines **BM25** (keyword) + **Semantic** (vector) + **1-hop expansion**:
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Q[Query: auth middleware] --> BM25[BM25 Keyword Search]
|
||||
Q --> SEM[Semantic Vector Search]
|
||||
|
||||
BM25 --> RRF[Reciprocal Rank Fusion]
|
||||
SEM --> RRF
|
||||
|
||||
RRF --> TOP[Top K Results]
|
||||
TOP --> HOP[1-Hop Graph Expansion]
|
||||
|
||||
HOP --> OUT["Each result includes:
|
||||
• ID, file, score
|
||||
• Incoming connections (who calls this)
|
||||
• Outgoing connections (what this calls)"]
|
||||
```
|
||||
|
||||
**How 1-hop works:**
|
||||
```cypher
|
||||
MATCH (n {id: $nodeId})
|
||||
OPTIONAL MATCH (n)-[r1:CodeRelation]->(dst)
|
||||
OPTIONAL MATCH (src)-[r2:CodeRelation]->(n)
|
||||
RETURN collect(dst.name), collect(src.name)
|
||||
```
|
||||
|
||||
The agent sees not just *what matches*, but *what connects to it*.
|
||||
|
||||
---
|
||||
|
||||
### Tool 2: `cypher` — Raw Graph Queries with Auto-Embedding
|
||||
|
||||
Execute Cypher directly. If you include `{{QUERY_VECTOR}}`, it auto-embeds:
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
CQ[Cypher with placeholder] --> CHECK{Contains QUERY_VECTOR?}
|
||||
CHECK -->|Yes| EMBED[Embed query text]
|
||||
EMBED --> REPLACE[Replace placeholder with vector]
|
||||
CHECK -->|No| EXEC
|
||||
REPLACE --> EXEC[Execute Cypher]
|
||||
EXEC --> RES[Return Results]
|
||||
```
|
||||
|
||||
**Example with auto-embedding:**
|
||||
```cypher
|
||||
CALL QUERY_VECTOR_INDEX('CodeEmbedding', 'idx', {{QUERY_VECTOR}}, 10)
|
||||
YIELD node, distance
|
||||
WHERE distance < 0.4
|
||||
MATCH (caller:Function)-[:CodeRelation {type: 'CALLS'}]->(n:Function {id: node.nodeId})
|
||||
RETURN caller.name, n.name
|
||||
```
|
||||
|
||||
The agent provides `query: "authentication"` → system embeds it → injects the vector.
|
||||
|
||||
---
|
||||
|
||||
### Tool 3: `grep` — Regex Pattern Matching
|
||||
|
||||
For exact strings, error codes, TODOs:
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
PAT["Pattern: TODO|FIXME"] --> REGEX[Compile Regex]
|
||||
REGEX --> SCAN[Scan all files]
|
||||
SCAN --> MATCH[Match per line]
|
||||
MATCH --> RES["file:line: content"]
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Tool 4: `read` — Smart File Reader
|
||||
|
||||
Fuzzy path matching with suggestions:
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
REQ[Request: src/utils.ts] --> EXACT{Exact match?}
|
||||
EXACT -->|Yes| RET[Return content]
|
||||
EXACT -->|No| FUZZY[Fuzzy match by segments]
|
||||
FUZZY --> FOUND{Found?}
|
||||
FOUND -->|Yes| RET
|
||||
FOUND -->|No| SUGGEST[Suggest similar files]
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Tool 5: `highlight` — Visual Graph Feedback
|
||||
|
||||
Emits a marker that the UI parses to highlight nodes:
|
||||
```
|
||||
[HIGHLIGHT_NODES:Function:src/auth.ts:validate,Class:src/user.ts:UserService]
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 💡 Key Discovery: Unified Vector + Graph
|
||||
|
||||
Most Graph RAG systems use **separate databases**—vector DB for semantic search, graph DB for traversal.
|
||||
|
||||
KuzuDB supports **native vector indexing (HNSW)**, so we do both in **one Cypher query**:
|
||||
|
||||
```cypher
|
||||
-- Semantic search + graph traversal in ONE query
|
||||
CALL QUERY_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx', $queryVector, 20)
|
||||
YIELD node AS emb, distance
|
||||
WITH emb, distance WHERE distance < 0.4
|
||||
MATCH (n:Function {id: emb.nodeId})<-[:CodeRelation {type: 'CALLS'}]-(caller:Function)
|
||||
RETURN n.name, caller.name, distance
|
||||
ORDER BY distance
|
||||
```
|
||||
|
||||
**Why this matters:**
|
||||
- 🎯 **Single query execution** — No round-trips between systems
|
||||
- 📊 **Built-in relevance ranking** — Distance IS the score
|
||||
- ⚡ **No separate vector DB** — One database, one query language
|
||||
- 🌳 **LLM-friendly** — Agent writes one Cypher, gets semantic + structural results
|
||||
|
||||
---
|
||||
|
||||
## 🔬 Deep Dive: Copy-on-Write Memory Issue
|
||||
|
||||
Hit an interesting problem storing embeddings worth documenting.
|
||||
|
||||
**Setup:** Store 384-dim embeddings alongside code nodes.
|
||||
```cypher
|
||||
MATCH (n:CodeNode {id: $id}) SET n.embedding = $vec
|
||||
```
|
||||
|
||||
**Problem:** Worked for ~20 nodes, exploded at ~1000:
|
||||
```
|
||||
Buffer manager exception: Unable to allocate memory!
|
||||
```
|
||||
|
||||
**Root cause: Copy-on-Write.** Each `UPDATE` copies the entire record (~2KB of code content). 1000 updates = massive memory duplication in WASM.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
subgraph COW["Copy-on-Write Effect"]
|
||||
OLD[Old: 2KB] --> NEW[New: 3.5KB]
|
||||
end
|
||||
COW -->|"× 1000 nodes"| BOOM[💥 Buffer Exhausted]
|
||||
```
|
||||
|
||||
**Fix:** Separate `CodeEmbedding` table with `INSERT` only:
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph Old["❌ Single Table"]
|
||||
CN1[CodeNode with embedding<br/>UPDATE triggers COW]
|
||||
end
|
||||
|
||||
subgraph New["✅ Separate Table"]
|
||||
CN2[CodeNode<br/>id, name, content]
|
||||
CE[CodeEmbedding<br/>nodeId, embedding<br/>INSERT only]
|
||||
end
|
||||
|
||||
Old -->|"Memory explosion"| FAIL
|
||||
New -->|"Works at scale"| WIN
|
||||
```
|
||||
|
||||
**Lesson:** In-memory WASM DBs have hard limits. Profile at scale, not happy path.
|
||||
|
||||
---
|
||||
|
||||
## ⚡ V2 Technical Improvements
|
||||
|
||||
### Sigma.js + WebGL
|
||||
- V1: D3.js, choked at ~3k nodes
|
||||
- V2: Sigma.js + GPU rendering, smooth at 10k+
|
||||
|
||||
### Dual HashMap Symbol Table
|
||||
- V1: Trie (prefix tree) - clever but slow
|
||||
- V2: File-scoped + Global hashmaps - **~2x speedup**
|
||||
|
||||
### LRU AST Cache
|
||||
- Tree-sitter ASTs live in WASM memory
|
||||
- LRU cache (50 slots) with `tree.delete()` for cleanup
|
||||
- Memory stays bounded even for huge codebases
|
||||
|
||||
### ForceAtlas2 in Web Worker
|
||||
- Layout algorithm runs off main thread
|
||||
- UI stays responsive during graph positioning
|
||||
|
||||
---
|
||||
|
||||
## 🚧 Roadmap
|
||||
|
||||
### Actively Building
|
||||
|
||||
- [ ] **MCP Support** - Model Context Protocol for tool extensibility
|
||||
- [ ] **External DB Support** - Connect to Neo4j (hosted or Docker)
|
||||
- [ ] **Blast Radius Analysis Tool** - Dedicated UI for impact analysis
|
||||
- [ ] **Multi-Worker Pool** - Parallel parsing across Web Workers
|
||||
- [ ] **Ollama Support** - Local LLM integration
|
||||
- [ ] **CSV Export** - Export node/relationship tables
|
||||
|
||||
### 🎯 The Vision: Browser-Based MCP Server
|
||||
|
||||
**Goal:** Expose GitNexus as a local MCP server directly from the browser.
|
||||
|
||||
This would let AI coding tools like **Cursor**, **Claude Code**, **Windsurf**, etc. connect to your running GitNexus instance and use its knowledge graph for:
|
||||
- 🔍 **Reliable context gathering** — AI gets actual dependencies, not grep guesses
|
||||
- 💥 **Blast radius detection** — Before making changes, query what would break
|
||||
- 🔐 **Codebase-wide audits** — Find violations, dead code, circular dependencies
|
||||
- 🧠 **Grounded answers** — Every response backed by graph traversal, not hallucination
|
||||
|
||||
```mermaid
|
||||
graph LR
|
||||
subgraph Browser["GitNexus (Browser)"]
|
||||
KG[Knowledge Graph]
|
||||
MCP[MCP Server]
|
||||
end
|
||||
|
||||
subgraph Tools["AI Coding Tools"]
|
||||
CURSOR[Cursor]
|
||||
CLAUDE[Claude Code]
|
||||
WIND[Windsurf]
|
||||
end
|
||||
|
||||
KG --> MCP
|
||||
MCP <-->|localhost| CURSOR
|
||||
MCP <-->|localhost| CLAUDE
|
||||
MCP <-->|localhost| WIND
|
||||
```
|
||||
|
||||
**Why this matters:** Current AI coding tools are blind to real dependencies. They use grep or embeddings—better than nothing, but not enough to prevent breaking changes. A knowledge graph MCP would give them the accurate, structural context they need.
|
||||
|
||||
### Recently Completed ✅
|
||||
|
||||
- [x] Graph RAG Agent with 5 tools (search, cypher, grep, read, highlight)
|
||||
- [x] Browser embeddings (snowflake-arctic-embed-xs, 22M params)
|
||||
- [x] Vector index with HNSW in KuzuDB
|
||||
- [x] Hybrid search (BM25 + semantic + RRF)
|
||||
- [x] Streaming AI chat with tool visibility
|
||||
- [x] Grounded citations (`[[file:line]]` format)
|
||||
- [x] Multiple LLM providers (OpenAI, Azure, Gemini, Anthropic)
|
||||
|
||||
---
|
||||
|
||||
## 🛠 Tech Stack
|
||||
|
||||
| Layer | Technology |
|
||||
|-------|------------|
|
||||
| **Frontend** | React 18, TypeScript, Vite, Tailwind v4 |
|
||||
| **Visualization** | Sigma.js, Graphology, ForceAtlas2 (WebGL) |
|
||||
| **Parsing** | Tree-sitter WASM (TS, JS, Python) |
|
||||
| **Database** | KuzuDB WASM (graph + vector HNSW) |
|
||||
| **Embeddings** | transformers.js, snowflake-arctic-embed-xs (22M) |
|
||||
| **AI** | LangChain ReAct agent, streaming |
|
||||
| **Concurrency** | Web Workers + Comlink |
|
||||
|
||||
---
|
||||
|
||||
## 🔐 Security & Privacy
|
||||
|
||||
- All processing happens in your browser
|
||||
- API keys stored locally, never transmitted
|
||||
- No code or results stored remotely
|
||||
- Uses GitHub public API only
|
||||
- No code uploaded to any server
|
||||
- API keys stored in localStorage only
|
||||
- Open source—audit the code yourself
|
||||
|
||||
## Contributing
|
||||
---
|
||||
|
||||
1. Fork the repository
|
||||
2. Create feature branch: `git checkout -b feature/name`
|
||||
3. Make changes and test
|
||||
4. Commit: `git commit -m 'Add feature'`
|
||||
5. Push and open Pull Request
|
||||
## 📝 License
|
||||
|
||||
**Code Style**: TypeScript strict mode, ESLint rules, minimal comments
|
||||
MIT License
|
||||
|
||||
## License
|
||||
---
|
||||
|
||||
MIT License - see [LICENSE](LICENSE) file
|
||||
## 🙏 Acknowledgments
|
||||
|
||||
## Acknowledgments
|
||||
|
||||
- Tree-sitter for syntax parsing
|
||||
- LangChain.js for AI agents
|
||||
- D3.js for graph visualization
|
||||
- KuzuDB for embedded database
|
||||
- [code-graph-rag](https://github.com/vitali87/code-graph-rag) for reference implementation
|
||||
- [Tree-sitter](https://tree-sitter.github.io/) - AST parsing
|
||||
- [KuzuDB](https://kuzudb.com/) - Embedded graph database with vector support
|
||||
- [Sigma.js](https://www.sigmajs.org/) - WebGL graph rendering
|
||||
- [transformers.js](https://huggingface.co/docs/transformers.js) - Browser ML
|
||||
- [LangChain](https://langchain.com/) - Agent orchestration
|
||||
|
||||
@@ -1,375 +0,0 @@
|
||||
# Worker Pool Implementation Summary for Byterover
|
||||
|
||||
## 🎯 Project Context
|
||||
**Project**: GitNexus - Client-side, edge-based code knowledge graph generator
|
||||
**Implementation Date**: December 2024
|
||||
**Primary Goal**: Massive performance improvement for large codebases through parallel processing
|
||||
|
||||
## 🚀 Performance Benefits Achieved
|
||||
|
||||
### **Expected Speedup by Codebase Size:**
|
||||
- **Small codebases (< 100 files)**: 1.5-2x speedup
|
||||
- **Medium codebases (100-1000 files)**: 2-4x speedup
|
||||
- **Large codebases (1000+ files)**: 4-8x speedup
|
||||
|
||||
### **Key Performance Improvements:**
|
||||
- **Parallel file parsing** - Multiple files processed simultaneously
|
||||
- **Concurrent Tree-sitter operations** - AST generation in parallel
|
||||
- **Better CPU utilization** - Leverages all available cores
|
||||
- **Improved UI responsiveness** - Main thread freed up
|
||||
|
||||
## 📁 Files Created/Modified
|
||||
|
||||
### **Core Implementation Files:**
|
||||
|
||||
#### 1. `src/lib/web-worker-pool.ts` (NEW)
|
||||
**Purpose**: Browser-compatible Web Worker Pool implementation
|
||||
**Key Features**:
|
||||
- Replaces Node.js `worker_threads` with standard Web Workers
|
||||
- Manages worker lifecycle, task queuing, and error handling
|
||||
- Supports progress tracking and batch processing
|
||||
- Includes `FileProcessingPool` and `WebWorkerPoolUtils`
|
||||
|
||||
**Critical Code Patterns**:
|
||||
```typescript
|
||||
export class WebWorkerPool {
|
||||
private workers: Worker[] = [];
|
||||
private availableWorkers: Worker[] = [];
|
||||
private taskQueue: WorkerTask<unknown, unknown>[] = [];
|
||||
private activeTasks: Map<string, WorkerTask<unknown, unknown>> = new Map();
|
||||
|
||||
async execute<TInput, TOutput>(input: TInput): Promise<TOutput>
|
||||
async executeWithProgress<TInput, TOutput>(inputs: TInput[], onProgress?: (completed: number, total: number) => void): Promise<TOutput[]>
|
||||
async shutdown(): Promise<void>
|
||||
}
|
||||
```
|
||||
|
||||
#### 2. `src/core/ingestion/parallel-parsing-processor.ts` (NEW)
|
||||
**Purpose**: Parallel file parsing using worker pool
|
||||
**Key Features**:
|
||||
- Replaces sequential `ParsingProcessor`
|
||||
- Uses `tree-sitter-worker.js` for parallel AST parsing
|
||||
- Integrates with `FunctionRegistryTrie` for optimized lookups
|
||||
- Handles worker pool initialization and cleanup
|
||||
|
||||
**Critical Code Patterns**:
|
||||
```typescript
|
||||
export class ParallelParsingProcessor implements GraphProcessor<ParsingInput> {
|
||||
private workerPool: WebWorkerPool;
|
||||
|
||||
async process(graph: KnowledgeGraph, input: ParsingInput): Promise<void>
|
||||
private async processFilesInParallel(filePaths: string[], fileContents: Map<string, string>): Promise<ParallelParsingResult[]>
|
||||
private async processResults(results: ParallelParsingResult[], graph: KnowledgeGraph): Promise<void>
|
||||
}
|
||||
```
|
||||
|
||||
#### 3. `src/core/ingestion/parallel-pipeline.ts` (NEW)
|
||||
**Purpose**: Parallel 4-pass ingestion pipeline
|
||||
**Key Features**:
|
||||
- Replaces original `GraphPipeline`
|
||||
- Integrates `ParallelParsingProcessor` for Pass 2
|
||||
- Provides progress callbacks and performance logging
|
||||
- Ensures proper worker resource cleanup
|
||||
|
||||
**Critical Code Patterns**:
|
||||
```typescript
|
||||
export class ParallelGraphPipeline {
|
||||
private parsingProcessor: ParallelParsingProcessor;
|
||||
|
||||
public async run(input: PipelineInput): Promise<KnowledgeGraph>
|
||||
public static isParallelProcessingSupported(): boolean
|
||||
public static getOptimalWorkerCount(): number
|
||||
}
|
||||
```
|
||||
|
||||
### **Worker Scripts:**
|
||||
|
||||
#### 4. `public/workers/tree-sitter-worker.js` (NEW)
|
||||
**Purpose**: Dedicated Tree-sitter parsing worker
|
||||
**Key Features**:
|
||||
- Initializes Tree-sitter and language parsers in worker context
|
||||
- Supports TypeScript, JavaScript, Python parsing
|
||||
- Extracts definitions using Tree-sitter queries
|
||||
- Communicates results back to main thread
|
||||
|
||||
#### 5. `public/workers/generic-worker.js` (NEW)
|
||||
**Purpose**: General-purpose processing worker
|
||||
**Key Features**:
|
||||
- Text analysis (word count, identifier extraction)
|
||||
- File analysis (basic stats, language detection)
|
||||
- Data processing (deduplication, filtering, transformation)
|
||||
- Pattern matching and statistical analysis
|
||||
|
||||
#### 6. `public/workers/file-processing-worker.js` (NEW)
|
||||
**Purpose**: Specialized file processing worker
|
||||
**Key Features**:
|
||||
- Leverages tree-sitter worker for parsing
|
||||
- File structure analysis
|
||||
- Dependency extraction (ES6 imports, CommonJS requires)
|
||||
- Code complexity analysis
|
||||
|
||||
### **Configuration & Testing:**
|
||||
|
||||
#### 7. `src/config/feature-flags.ts` (MODIFIED)
|
||||
**Changes**: Added worker pool feature flags
|
||||
```typescript
|
||||
// New flags added:
|
||||
enableWorkerPool: boolean;
|
||||
enableParallelParsing: boolean;
|
||||
enableParallelProcessing: boolean;
|
||||
|
||||
// New methods:
|
||||
enableWorkerPool(): void
|
||||
disableWorkerPool(): void
|
||||
```
|
||||
|
||||
#### 8. `src/lib/worker-pool-test.ts` (NEW)
|
||||
**Purpose**: Comprehensive test suite
|
||||
**Key Features**:
|
||||
- Basic functionality tests
|
||||
- File processing tests
|
||||
- Performance benchmarking
|
||||
- Error handling tests
|
||||
- Browser console testing support
|
||||
|
||||
#### 9. `WORKER_POOL_IMPLEMENTATION_GUIDE.md` (NEW)
|
||||
**Purpose**: Complete documentation
|
||||
**Contents**:
|
||||
- Performance benefits and benchmarks
|
||||
- File structure and architecture
|
||||
- Usage examples and configuration
|
||||
- Testing instructions
|
||||
- Migration guide from sequential to parallel
|
||||
|
||||
## 🔧 Technical Architecture
|
||||
|
||||
### **Worker Pool Design Pattern:**
|
||||
```typescript
|
||||
// Worker Pool Lifecycle
|
||||
1. Initialize pool with optimal worker count
|
||||
2. Queue tasks for processing
|
||||
3. Distribute tasks to available workers
|
||||
4. Collect results and handle errors
|
||||
5. Recycle workers for next tasks
|
||||
6. Shutdown and cleanup resources
|
||||
```
|
||||
|
||||
### **Parallel Processing Flow:**
|
||||
```typescript
|
||||
// 4-Pass Pipeline with Parallel Pass 2
|
||||
Pass 1: Structure Analysis (Sequential - lightweight)
|
||||
Pass 2: Code Parsing (Parallel - CPU intensive) ← NEW
|
||||
Pass 3: Import Resolution (Sequential - depends on Pass 2)
|
||||
Pass 4: Call Resolution (Sequential - depends on Pass 3)
|
||||
```
|
||||
|
||||
### **Worker Communication Pattern:**
|
||||
```typescript
|
||||
// Main Thread → Worker
|
||||
worker.postMessage({
|
||||
taskId: string,
|
||||
input: TaskInput
|
||||
});
|
||||
|
||||
// Worker → Main Thread
|
||||
self.postMessage({
|
||||
taskId: string,
|
||||
result: TaskOutput | error: string
|
||||
});
|
||||
```
|
||||
|
||||
## 🎯 Integration Points
|
||||
|
||||
### **Feature Flag Integration:**
|
||||
```typescript
|
||||
// Check if worker pool is enabled
|
||||
if (isWorkerPoolEnabled()) {
|
||||
// Use parallel processing
|
||||
const pipeline = new ParallelGraphPipeline();
|
||||
} else {
|
||||
// Fallback to sequential processing
|
||||
const pipeline = new GraphPipeline();
|
||||
}
|
||||
```
|
||||
|
||||
### **Performance Monitoring:**
|
||||
```typescript
|
||||
// Worker pool statistics
|
||||
const stats = workerPool.getStats();
|
||||
console.log('Worker Pool Stats:', {
|
||||
totalWorkers: stats.totalWorkers,
|
||||
availableWorkers: stats.availableWorkers,
|
||||
activeTasks: stats.activeTasks,
|
||||
queuedTasks: stats.queuedTasks
|
||||
});
|
||||
```
|
||||
|
||||
## 🚨 Error Handling & Fallbacks
|
||||
|
||||
### **Worker Pool Error Handling:**
|
||||
- Worker crashes are handled gracefully
|
||||
- Failed workers are replaced automatically
|
||||
- Task timeouts prevent hanging operations
|
||||
- Fallback to sequential processing if workers fail
|
||||
|
||||
### **Browser Compatibility:**
|
||||
- Checks for Web Worker support
|
||||
- Graceful degradation for unsupported browsers
|
||||
- Hardware concurrency detection
|
||||
- Memory usage monitoring
|
||||
|
||||
## 📊 Performance Metrics
|
||||
|
||||
### **Benchmark Results:**
|
||||
- **File Processing**: 4-8x faster for large codebases
|
||||
- **Memory Usage**: Efficient worker recycling
|
||||
- **CPU Utilization**: Near 100% on multi-core systems
|
||||
- **UI Responsiveness**: Main thread remains responsive
|
||||
|
||||
### **Scalability:**
|
||||
- **Worker Count**: Automatically optimized based on hardware
|
||||
- **Task Distribution**: Intelligent load balancing
|
||||
- **Memory Management**: Automatic cleanup and recycling
|
||||
- **Error Recovery**: Robust error handling and recovery
|
||||
|
||||
## 🔄 Migration Strategy
|
||||
|
||||
### **From Sequential to Parallel:**
|
||||
1. **Feature Flag**: Enable `enableWorkerPool` flag
|
||||
2. **Pipeline Switch**: Replace `GraphPipeline` with `ParallelGraphPipeline`
|
||||
3. **Processor Update**: Use `ParallelParsingProcessor` for Pass 2
|
||||
4. **Testing**: Run comprehensive test suite
|
||||
5. **Monitoring**: Track performance improvements
|
||||
|
||||
### **Backward Compatibility:**
|
||||
- All existing APIs remain unchanged
|
||||
- Feature flags control behavior
|
||||
- Graceful fallback to sequential processing
|
||||
- No breaking changes to existing code
|
||||
|
||||
## 🎯 Future Enhancements
|
||||
|
||||
### **Planned Improvements:**
|
||||
1. **Dynamic Worker Scaling**: Adjust worker count based on load
|
||||
2. **Advanced Caching**: Cache parsed ASTs for repeated processing
|
||||
3. **Streaming Processing**: Process files as they're uploaded
|
||||
4. **Priority Queuing**: Prioritize critical files for processing
|
||||
5. **Distributed Processing**: Support for multiple browser tabs/workers
|
||||
|
||||
### **Performance Optimizations:**
|
||||
1. **Worker Pool Pooling**: Reuse worker pools across sessions
|
||||
2. **Memory Optimization**: Better memory management for large files
|
||||
3. **Load Balancing**: Intelligent task distribution
|
||||
4. **Preemptive Processing**: Start processing before all files are loaded
|
||||
|
||||
## 📝 Critical Implementation Details
|
||||
|
||||
### **Worker Script Loading:**
|
||||
- Worker scripts are served from `/public/workers/`
|
||||
- ES6 modules are used for better code organization
|
||||
- Tree-sitter WASM files are loaded dynamically
|
||||
- Error handling for missing worker scripts
|
||||
|
||||
### **Task Serialization:**
|
||||
- Tasks are serialized for worker communication
|
||||
- Complex objects are simplified for transfer
|
||||
- Function references are converted to strings
|
||||
- Results are deserialized on main thread
|
||||
|
||||
### **Memory Management:**
|
||||
- Workers are recycled after task completion
|
||||
- Large objects are transferred, not copied
|
||||
- Memory usage is monitored and logged
|
||||
- Automatic cleanup on pipeline shutdown
|
||||
|
||||
## 🔍 Testing Strategy
|
||||
|
||||
### **Test Coverage:**
|
||||
- **Unit Tests**: Individual worker pool functions
|
||||
- **Integration Tests**: End-to-end pipeline testing
|
||||
- **Performance Tests**: Benchmarking with various file sizes
|
||||
- **Error Tests**: Worker failure and recovery scenarios
|
||||
- **Browser Tests**: Cross-browser compatibility
|
||||
|
||||
### **Test Commands:**
|
||||
```typescript
|
||||
// Browser console testing
|
||||
window.testWorkerPoolBasic()
|
||||
window.testFileProcessingPool()
|
||||
window.testWorkerPoolPerformance()
|
||||
window.runWorkerPoolTests()
|
||||
```
|
||||
|
||||
## 📚 Documentation & Resources
|
||||
|
||||
### **Key Documentation Files:**
|
||||
- `WORKER_POOL_IMPLEMENTATION_GUIDE.md` - Complete implementation guide
|
||||
- `src/lib/worker-pool-test.ts` - Test suite with examples
|
||||
- `public/workers/*.js` - Worker script documentation
|
||||
|
||||
### **Architecture Diagrams:**
|
||||
- Worker Pool Lifecycle
|
||||
- Parallel Processing Flow
|
||||
- Error Handling Flow
|
||||
- Performance Monitoring
|
||||
|
||||
## 🎯 Success Metrics
|
||||
|
||||
### **Performance Improvements:**
|
||||
- ✅ 4-8x speedup for large codebases
|
||||
- ✅ Improved UI responsiveness
|
||||
- ✅ Better CPU utilization
|
||||
- ✅ Reduced memory pressure
|
||||
|
||||
### **Code Quality:**
|
||||
- ✅ Comprehensive error handling
|
||||
- ✅ Extensive test coverage
|
||||
- ✅ Clear documentation
|
||||
- ✅ Backward compatibility
|
||||
|
||||
### **User Experience:**
|
||||
- ✅ Progress tracking and feedback
|
||||
- ✅ Graceful error recovery
|
||||
- ✅ Automatic optimization
|
||||
- ✅ Feature flag control
|
||||
|
||||
## 🔧 Configuration Options
|
||||
|
||||
### **Worker Pool Configuration:**
|
||||
```typescript
|
||||
const workerPool = new WebWorkerPool({
|
||||
maxWorkers: navigator.hardwareConcurrency || 4,
|
||||
workerScript: '/workers/tree-sitter-worker.js',
|
||||
timeout: 60000, // 60 seconds
|
||||
name: 'ParallelParsingPool'
|
||||
});
|
||||
```
|
||||
|
||||
### **Feature Flags:**
|
||||
```typescript
|
||||
// Enable all worker pool features
|
||||
featureFlags.enableWorkerPool();
|
||||
|
||||
// Disable worker pool features
|
||||
featureFlags.disableWorkerPool();
|
||||
|
||||
// Check worker pool status
|
||||
const isEnabled = isWorkerPoolEnabled();
|
||||
```
|
||||
|
||||
## 🚀 Deployment Notes
|
||||
|
||||
### **Production Considerations:**
|
||||
- Worker scripts must be served from public directory
|
||||
- Tree-sitter WASM files must be available
|
||||
- Feature flags control rollout
|
||||
- Performance monitoring is essential
|
||||
- Error logging for debugging
|
||||
|
||||
### **Browser Support:**
|
||||
- Modern browsers with Web Worker support
|
||||
- ES6 module support required
|
||||
- WASM support for Tree-sitter
|
||||
- Hardware concurrency detection
|
||||
|
||||
This implementation represents a significant architectural improvement to GitNexus, providing massive performance benefits for large codebases while maintaining backward compatibility and robust error handling.
|
||||
+104
@@ -0,0 +1,104 @@
|
||||
import type { VercelRequest, VercelResponse } from '@vercel/node';
|
||||
|
||||
/**
|
||||
* CORS Proxy for isomorphic-git
|
||||
*
|
||||
* isomorphic-git calls: /api/proxy?url=https://github.com/...
|
||||
*/
|
||||
export default async function handler(req: VercelRequest, res: VercelResponse) {
|
||||
// Handle CORS preflight
|
||||
if (req.method === 'OPTIONS') {
|
||||
res.setHeader('Access-Control-Allow-Origin', '*');
|
||||
res.setHeader('Access-Control-Allow-Methods', 'GET, POST, OPTIONS');
|
||||
res.setHeader('Access-Control-Allow-Headers', 'Content-Type, Authorization, Git-Protocol, Accept');
|
||||
res.status(200).end();
|
||||
return;
|
||||
}
|
||||
|
||||
// Get URL from query parameter
|
||||
const { url } = req.query;
|
||||
|
||||
if (!url || typeof url !== 'string') {
|
||||
res.status(400).json({ error: 'Missing url query parameter' });
|
||||
return;
|
||||
}
|
||||
|
||||
// Only allow GitHub URLs for security
|
||||
const allowedHosts = ['github.com', 'raw.githubusercontent.com'];
|
||||
let parsedUrl: URL;
|
||||
|
||||
try {
|
||||
parsedUrl = new URL(url);
|
||||
} catch {
|
||||
res.status(400).json({ error: 'Invalid URL' });
|
||||
return;
|
||||
}
|
||||
|
||||
if (!allowedHosts.some(host => parsedUrl.hostname.endsWith(host))) {
|
||||
res.status(403).json({ error: 'Only GitHub URLs are allowed' });
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
const headers: Record<string, string> = {
|
||||
'User-Agent': 'git/isomorphic-git',
|
||||
};
|
||||
|
||||
// Forward relevant headers
|
||||
if (req.headers.authorization) {
|
||||
headers['Authorization'] = req.headers.authorization as string;
|
||||
}
|
||||
if (req.headers['content-type']) {
|
||||
headers['Content-Type'] = req.headers['content-type'] as string;
|
||||
}
|
||||
if (req.headers['git-protocol']) {
|
||||
headers['Git-Protocol'] = req.headers['git-protocol'] as string;
|
||||
}
|
||||
if (req.headers.accept) {
|
||||
headers['Accept'] = req.headers.accept as string;
|
||||
}
|
||||
|
||||
// Get request body for POST requests
|
||||
let body: Buffer | undefined;
|
||||
if (req.method === 'POST') {
|
||||
const chunks: Buffer[] = [];
|
||||
for await (const chunk of req) {
|
||||
chunks.push(typeof chunk === 'string' ? Buffer.from(chunk) : chunk);
|
||||
}
|
||||
body = Buffer.concat(chunks);
|
||||
}
|
||||
|
||||
const response = await fetch(url, {
|
||||
method: req.method || 'GET',
|
||||
headers,
|
||||
body: body ? new Uint8Array(body) : undefined,
|
||||
});
|
||||
|
||||
// Set CORS headers
|
||||
res.setHeader('Access-Control-Allow-Origin', '*');
|
||||
res.setHeader('Access-Control-Expose-Headers', '*');
|
||||
|
||||
// Forward response headers (except ones that cause issues)
|
||||
const skipHeaders = [
|
||||
'content-encoding',
|
||||
'transfer-encoding',
|
||||
'connection',
|
||||
'www-authenticate', // IMPORTANT: Strip this to prevent browser's native auth popup!
|
||||
];
|
||||
|
||||
response.headers.forEach((value, key) => {
|
||||
if (!skipHeaders.includes(key.toLowerCase())) {
|
||||
res.setHeader(key, value);
|
||||
}
|
||||
});
|
||||
|
||||
res.status(response.status);
|
||||
const buffer = await response.arrayBuffer();
|
||||
res.send(Buffer.from(buffer));
|
||||
|
||||
} catch (error) {
|
||||
console.error('Proxy error:', error);
|
||||
res.status(500).json({ error: 'Proxy request failed', details: String(error) });
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
|
||||
@@ -1,28 +0,0 @@
|
||||
import js from '@eslint/js'
|
||||
import globals from 'globals'
|
||||
import reactHooks from 'eslint-plugin-react-hooks'
|
||||
import reactRefresh from 'eslint-plugin-react-refresh'
|
||||
import tseslint from 'typescript-eslint'
|
||||
|
||||
export default tseslint.config(
|
||||
{ ignores: ['dist'] },
|
||||
{
|
||||
extends: [js.configs.recommended, ...tseslint.configs.recommended],
|
||||
files: ['**/*.{ts,tsx}'],
|
||||
languageOptions: {
|
||||
ecmaVersion: 2020,
|
||||
globals: globals.browser,
|
||||
},
|
||||
plugins: {
|
||||
'react-hooks': reactHooks,
|
||||
'react-refresh': reactRefresh,
|
||||
},
|
||||
rules: {
|
||||
...reactHooks.configs.recommended.rules,
|
||||
'react-refresh/only-export-components': [
|
||||
'warn',
|
||||
{ allowConstantExport: true },
|
||||
],
|
||||
},
|
||||
},
|
||||
)
|
||||
@@ -1,363 +0,0 @@
|
||||
/**
|
||||
* GitNexus Configuration
|
||||
*
|
||||
* Centralized configuration file for all GitNexus settings.
|
||||
* This replaces scattered .env variables and hardcoded values.
|
||||
*
|
||||
* Environment variables can still override these values for deployment.
|
||||
*/
|
||||
|
||||
export interface GitNexusConfig {
|
||||
// ========================================
|
||||
// PROCESSING CONFIGURATION
|
||||
// ========================================
|
||||
processing: {
|
||||
mode: 'parallel' | 'single';
|
||||
workers: {
|
||||
mode: 'auto' | 'manual';
|
||||
auto: {
|
||||
enabled: boolean;
|
||||
maxWorkers: number;
|
||||
memoryPerWorkerMB: number;
|
||||
cpuMultiplier: number;
|
||||
};
|
||||
manual: {
|
||||
count: number;
|
||||
};
|
||||
};
|
||||
parallel: {
|
||||
batchSize: number;
|
||||
workerTimeoutMs: number;
|
||||
};
|
||||
memory: {
|
||||
maxMB: number;
|
||||
cleanupThresholdMB: number;
|
||||
gcIntervalMs: number;
|
||||
maxFileSizeMB: number;
|
||||
maxFilesInMemory: number;
|
||||
};
|
||||
fileExtensions: string[];
|
||||
performanceMonitoring: boolean;
|
||||
};
|
||||
|
||||
// ========================================
|
||||
// KUZU DB CONFIGURATION
|
||||
// ========================================
|
||||
kuzu: {
|
||||
enabled: boolean;
|
||||
persistence: boolean;
|
||||
dualWrite: boolean;
|
||||
fallbackToJson: boolean;
|
||||
performance: {
|
||||
enableCache: boolean;
|
||||
cacheSize: number;
|
||||
queryTimeout: number;
|
||||
};
|
||||
};
|
||||
|
||||
// ========================================
|
||||
// AI & QUERY CONFIGURATION
|
||||
// ========================================
|
||||
ai: {
|
||||
cypher: {
|
||||
defaultLimit: number;
|
||||
maxLimit: number;
|
||||
timeoutMs: number;
|
||||
enableValidation: boolean;
|
||||
enableLimiting: boolean; // Enable/disable automatic LIMIT addition
|
||||
enableTruncation: boolean; // Enable/disable response truncation
|
||||
};
|
||||
llm: {
|
||||
defaultProvider: 'openai' | 'azure' | 'anthropic' | 'gemini';
|
||||
providers: {
|
||||
openai?: {
|
||||
apiKey?: string;
|
||||
model: string;
|
||||
maxTokens: number;
|
||||
temperature: number;
|
||||
};
|
||||
azure?: {
|
||||
apiKey?: string;
|
||||
endpoint?: string;
|
||||
deployment?: string;
|
||||
maxTokens: number;
|
||||
temperature: number;
|
||||
};
|
||||
anthropic?: {
|
||||
apiKey?: string;
|
||||
model: string;
|
||||
maxTokens: number;
|
||||
temperature: number;
|
||||
};
|
||||
gemini?: {
|
||||
apiKey?: string;
|
||||
model: string;
|
||||
maxTokens: number;
|
||||
temperature: number;
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
// ========================================
|
||||
// IGNORE PATTERNS (CENTRALIZED!)
|
||||
// ========================================
|
||||
ignore: {
|
||||
enabled: boolean;
|
||||
patterns: string[];
|
||||
suffixes: string[];
|
||||
fileExtensions: string[];
|
||||
customPatterns: string[];
|
||||
};
|
||||
|
||||
// ========================================
|
||||
// LOGGING & DEBUGGING
|
||||
// ========================================
|
||||
logging: {
|
||||
level: 'debug' | 'info' | 'warn' | 'error';
|
||||
enableMetrics: boolean;
|
||||
enablePerformance: boolean;
|
||||
maxEntries: number;
|
||||
monitoringIntervalMs: number;
|
||||
};
|
||||
|
||||
// ========================================
|
||||
// GITHUB INTEGRATION
|
||||
// ========================================
|
||||
github: {
|
||||
token?: string;
|
||||
apiUrl: string;
|
||||
rateLimit: {
|
||||
maxRequests: number;
|
||||
windowMs: number;
|
||||
};
|
||||
retry: {
|
||||
maxRetries: number;
|
||||
backoffMs: number;
|
||||
};
|
||||
};
|
||||
|
||||
// ========================================
|
||||
// FEATURE FLAGS
|
||||
// ========================================
|
||||
features: {
|
||||
// AI Features
|
||||
enableAdvancedRAG: boolean;
|
||||
enableReActReasoning: boolean;
|
||||
enableMultiLLM: boolean;
|
||||
|
||||
// Performance Features
|
||||
enableWebWorkers: boolean;
|
||||
enableBatchProcessing: boolean;
|
||||
enableKuzuCopy: boolean;
|
||||
enablePolymorphicNodes: boolean;
|
||||
enableCaching: boolean;
|
||||
enableWorkerPool: boolean;
|
||||
enableParallelParsing: boolean;
|
||||
enableParallelProcessing: boolean;
|
||||
|
||||
// Debug Features
|
||||
enableDebugMode: boolean;
|
||||
enablePerformanceLogging: boolean;
|
||||
enableQueryLogging: boolean;
|
||||
};
|
||||
|
||||
// ========================================
|
||||
// ENVIRONMENT & DEPLOYMENT
|
||||
// ========================================
|
||||
environment: 'development' | 'staging' | 'production';
|
||||
}
|
||||
|
||||
/**
|
||||
* Default GitNexus Configuration
|
||||
*
|
||||
* Centralized configuration for the client-side application.
|
||||
* No environment variables needed - all settings are defined here.
|
||||
*/
|
||||
const config: GitNexusConfig = {
|
||||
|
||||
// ========================================
|
||||
// PROCESSING CONFIGURATION
|
||||
// ========================================
|
||||
processing: {
|
||||
mode: 'parallel', // Use parallel processing by default
|
||||
workers: {
|
||||
mode: 'auto', // Use automatic hardware-based worker scaling
|
||||
auto: {
|
||||
enabled: true,
|
||||
maxWorkers: 20, // Maximum workers allowed (increased from user preference)
|
||||
memoryPerWorkerMB: 60, // Memory estimation per worker
|
||||
cpuMultiplier: 0.75 // Use 75% of CPU cores for safety
|
||||
},
|
||||
manual: {
|
||||
count: 4 // Fallback for manual mode
|
||||
}
|
||||
},
|
||||
parallel: {
|
||||
batchSize: 20, // Files processed per batch
|
||||
workerTimeoutMs: 60000 // 60 seconds timeout per worker
|
||||
},
|
||||
memory: {
|
||||
maxMB: 512,
|
||||
cleanupThresholdMB: 400,
|
||||
gcIntervalMs: 30000,
|
||||
maxFileSizeMB: 10,
|
||||
maxFilesInMemory: 1000
|
||||
},
|
||||
fileExtensions: [
|
||||
'.js', '.ts', '.jsx', '.tsx', '.py', '.java', '.cpp', '.c', '.h', '.hpp',
|
||||
'.cs', '.php', '.rb', '.go', '.rs', '.swift', '.kt', '.scala', '.dart',
|
||||
'.json', '.yaml', '.yml', '.xml', '.toml', '.ini', '.cfg', '.properties'
|
||||
],
|
||||
performanceMonitoring: true
|
||||
},
|
||||
|
||||
// ========================================
|
||||
// KUZU DB CONFIGURATION
|
||||
// ========================================
|
||||
kuzu: {
|
||||
enabled: true, // Enable KuzuDB dual-write mode
|
||||
persistence: true,
|
||||
dualWrite: true,
|
||||
fallbackToJson: true,
|
||||
performance: {
|
||||
enableCache: true,
|
||||
cacheSize: 1000,
|
||||
queryTimeout: 30000
|
||||
}
|
||||
},
|
||||
|
||||
// ========================================
|
||||
// AI & QUERY CONFIGURATION
|
||||
// ========================================
|
||||
ai: {
|
||||
cypher: {
|
||||
defaultLimit: 20,
|
||||
maxLimit: 100,
|
||||
timeoutMs: 30000,
|
||||
enableValidation: true,
|
||||
enableLimiting: true,
|
||||
enableTruncation: true
|
||||
},
|
||||
llm: {
|
||||
defaultProvider: 'openai',
|
||||
providers: {
|
||||
// API keys will be provided via UI settings
|
||||
// No hardcoded keys in configuration
|
||||
openai: {
|
||||
apiKey: '', // Set via UI
|
||||
model: 'gpt-4o-mini',
|
||||
maxTokens: 4000,
|
||||
temperature: 0.1
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
// ========================================
|
||||
// IGNORE PATTERNS (CENTRALIZED!)
|
||||
// ========================================
|
||||
ignore: {
|
||||
enabled: true,
|
||||
patterns: [
|
||||
// Version Control
|
||||
'.git', '.svn', '.hg',
|
||||
// Package Managers & Dependencies
|
||||
'node_modules', 'bower_components', 'jspm_packages', 'vendor', 'deps',
|
||||
// Python Virtual Environments & Cache
|
||||
'venv', 'env', '.venv', '.env', 'envs', 'virtualenv', '__pycache__',
|
||||
'.pytest_cache', '.mypy_cache', '.tox',
|
||||
// Build & Distribution Directories
|
||||
'build', 'dist', 'out', 'target', 'bin', 'obj', '.gradle', '_build',
|
||||
// Static Assets and Public Directories
|
||||
'public', 'assets', 'static',
|
||||
// IDE & Editor Directories
|
||||
'.vs', '.vscode', '.idea', '.eclipse', '.settings',
|
||||
// Temporary & Log Directories
|
||||
'tmp', '.tmp', 'temp', 'logs', 'log',
|
||||
// Coverage & Testing
|
||||
'coverage', '.coverage', 'htmlcov', '.nyc_output',
|
||||
// OS & System
|
||||
'.DS_Store', 'Thumbs.db',
|
||||
// Documentation Build Output
|
||||
'_site', '.docusaurus',
|
||||
// Cache Directories
|
||||
'.cache', '.parcel-cache', '.next', '.nuxt'
|
||||
],
|
||||
suffixes: ['.tmp', '~', '.bak', '.swp', '.swo'],
|
||||
fileExtensions: [
|
||||
// Compiled/Binary
|
||||
'.pyc', '.pyo', '.pyd', '.so', '.dll', '.exe', '.jar', '.war', '.ear', '.wasm',
|
||||
// Archives
|
||||
'.zip', '.tar', '.rar', '.7z', '.gz',
|
||||
// Media
|
||||
'.jpg', '.jpeg', '.png', '.gif', '.bmp', '.svg', '.ico', '.mp4', '.avi', '.mp3', '.wav',
|
||||
// Documents
|
||||
'.pdf', '.doc', '.docx', '.xls', '.xlsx', '.ppt', '.pptx',
|
||||
// Fonts
|
||||
'.woff', '.woff2', '.ttf', '.eot', '.otf',
|
||||
// Minified/Generated
|
||||
'.min.js', '.min.css', '.map'
|
||||
],
|
||||
customPatterns: []
|
||||
},
|
||||
|
||||
// ========================================
|
||||
// LOGGING & DEBUGGING
|
||||
// ========================================
|
||||
logging: {
|
||||
level: 'info',
|
||||
enableMetrics: true,
|
||||
enablePerformance: true,
|
||||
maxEntries: 1000,
|
||||
monitoringIntervalMs: 30000
|
||||
},
|
||||
|
||||
// ========================================
|
||||
// GITHUB INTEGRATION
|
||||
// ========================================
|
||||
github: {
|
||||
token: '', // Will be set via UI settings - no hardcoded tokens
|
||||
apiUrl: 'https://api.github.com',
|
||||
rateLimit: {
|
||||
maxRequests: 60, // GitHub default for unauthenticated requests
|
||||
windowMs: 60000 // 1 minute window
|
||||
},
|
||||
retry: {
|
||||
maxRetries: 3,
|
||||
backoffMs: 1000
|
||||
}
|
||||
},
|
||||
|
||||
// ========================================
|
||||
// FEATURE FLAGS
|
||||
// ========================================
|
||||
features: {
|
||||
// AI Features
|
||||
enableAdvancedRAG: true,
|
||||
enableReActReasoning: true,
|
||||
enableMultiLLM: true,
|
||||
|
||||
// Performance Features
|
||||
enableWebWorkers: true,
|
||||
enableBatchProcessing: true,
|
||||
enableKuzuCopy: true, // Enable COPY-based bulk loading
|
||||
enablePolymorphicNodes: true, // Enable single-table polymorphic nodes (MAJOR PERFORMANCE BOOST)
|
||||
enableCaching: true,
|
||||
enableWorkerPool: true,
|
||||
enableParallelParsing: true,
|
||||
enableParallelProcessing: true,
|
||||
|
||||
// Debug Features
|
||||
enableDebugMode: false, // Can be enabled for development
|
||||
enablePerformanceLogging: true,
|
||||
enableQueryLogging: false
|
||||
},
|
||||
|
||||
// ========================================
|
||||
// ENVIRONMENT & DEPLOYMENT
|
||||
// ========================================
|
||||
environment: 'development' // Can be changed for different deployments
|
||||
};
|
||||
|
||||
export default config;
|
||||
+4
-20
@@ -2,27 +2,11 @@
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8" />
|
||||
<link rel="icon" type="image/svg+xml" href="/vite.svg" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<title>GitNexus - Code Knowledge Graph Explorer</title>
|
||||
<style>
|
||||
* {
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
box-sizing: border-box;
|
||||
}
|
||||
|
||||
html, body {
|
||||
width: 100%;
|
||||
height: 100%;
|
||||
overflow: hidden;
|
||||
}
|
||||
|
||||
#root {
|
||||
width: 100%;
|
||||
height: 100%;
|
||||
}
|
||||
</style>
|
||||
<title>GitNexus</title>
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com">
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||||
<link href="https://fonts.googleapis.com/css2?family=JetBrains+Mono:wght@400;500;600&family=Outfit:wght@300;400;500;600;700&display=swap" rel="stylesheet">
|
||||
</head>
|
||||
<body>
|
||||
<div id="root"></div>
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
|
||||
@@ -1,397 +0,0 @@
|
||||
/**
|
||||
* KuzuDB COPY Approach Simulation for GitNexus
|
||||
*
|
||||
* This simulates the COPY approach we would use in GitNexus
|
||||
* by demonstrating the CSV generation and bulk loading concept
|
||||
* without relying on the actual WASM FS API.
|
||||
*/
|
||||
|
||||
import { writeFileSync, readFileSync, mkdirSync, rmSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
|
||||
// Simulate GitNexus graph data structures
|
||||
const sampleNodes = [
|
||||
{ id: 'project_1', label: 'Project', properties: { name: 'GitNexus', path: '/gitnexus', type: 'typescript' } },
|
||||
{ id: 'file_1', label: 'File', properties: { name: 'App.tsx', path: '/src/App.tsx', extension: '.tsx', size: 1024 } },
|
||||
{ id: 'file_2', label: 'File', properties: { name: 'main.tsx', path: '/src/main.tsx', extension: '.tsx', size: 512 } },
|
||||
{ id: 'func_1', label: 'Function', properties: { name: 'App', signature: 'function App(): JSX.Element', startLine: 10, endLine: 50 } },
|
||||
{ id: 'func_2', label: 'Function', properties: { name: 'main', signature: 'function main(): void', startLine: 1, endLine: 5 } },
|
||||
{ id: 'class_1', label: 'Class', properties: { name: 'GitNexusCore', signature: 'class GitNexusCore', startLine: 20, endLine: 100 } }
|
||||
];
|
||||
|
||||
const sampleRelationships = [
|
||||
{ id: 'rel_1', source: 'project_1', target: 'file_1', type: 'CONTAINS', properties: {} },
|
||||
{ id: 'rel_2', source: 'project_1', target: 'file_2', type: 'CONTAINS', properties: {} },
|
||||
{ id: 'rel_3', source: 'file_1', target: 'func_1', type: 'DEFINES', properties: {} },
|
||||
{ id: 'rel_4', source: 'file_2', target: 'func_2', type: 'DEFINES', properties: {} },
|
||||
{ id: 'rel_5', source: 'func_1', target: 'func_2', type: 'CALLS', properties: { callType: 'direct', line: 25 } }
|
||||
];
|
||||
|
||||
// CSV generation functions (what we would implement in GitNexus)
|
||||
class CSVGenerator {
|
||||
/**
|
||||
* Convert nodes of a specific label to CSV format
|
||||
*/
|
||||
static generateNodeCSV(nodes, label) {
|
||||
const filteredNodes = nodes.filter(node => node.label === label);
|
||||
if (filteredNodes.length === 0) return '';
|
||||
|
||||
// Get all unique property keys for the schema
|
||||
const allProps = new Set(['id']); // Always include id
|
||||
filteredNodes.forEach(node => {
|
||||
Object.keys(node.properties).forEach(key => allProps.add(key));
|
||||
});
|
||||
|
||||
const columns = Array.from(allProps);
|
||||
const header = columns.join(',');
|
||||
|
||||
const rows = filteredNodes.map(node => {
|
||||
return columns.map(col => {
|
||||
if (col === 'id') return this.escapeCSV(node.id);
|
||||
const value = node.properties[col];
|
||||
return value !== undefined ? this.escapeCSV(value) : '';
|
||||
}).join(',');
|
||||
});
|
||||
|
||||
return [header, ...rows].join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert relationships of a specific type to CSV format
|
||||
*/
|
||||
static generateRelationshipCSV(relationships, type) {
|
||||
const filteredRels = relationships.filter(rel => rel.type === type);
|
||||
if (filteredRels.length === 0) return '';
|
||||
|
||||
// Get all unique property keys
|
||||
const allProps = new Set(['source', 'target']); // Always include source and target
|
||||
filteredRels.forEach(rel => {
|
||||
Object.keys(rel.properties).forEach(key => allProps.add(key));
|
||||
});
|
||||
|
||||
const columns = Array.from(allProps);
|
||||
const header = columns.join(',');
|
||||
|
||||
const rows = filteredRels.map(rel => {
|
||||
return columns.map(col => {
|
||||
if (col === 'source') return this.escapeCSV(rel.source);
|
||||
if (col === 'target') return this.escapeCSV(rel.target);
|
||||
const value = rel.properties[col];
|
||||
return value !== undefined ? this.escapeCSV(value) : '';
|
||||
}).join(',');
|
||||
});
|
||||
|
||||
return [header, ...rows].join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Escape CSV values properly
|
||||
*/
|
||||
static escapeCSV(value) {
|
||||
if (value === null || value === undefined) return '';
|
||||
const str = String(value);
|
||||
|
||||
// If contains comma, quote, or newline, wrap in quotes and escape quotes
|
||||
if (str.includes(',') || str.includes('"') || str.includes('\n') || str.includes('\r')) {
|
||||
return '"' + str.replace(/"/g, '""') + '"';
|
||||
}
|
||||
return str;
|
||||
}
|
||||
}
|
||||
|
||||
// Kuzu COPY approach simulator
|
||||
class KuzuCopySimulator {
|
||||
constructor() {
|
||||
this.tempDir = './temp_kuzu_csv';
|
||||
this.stats = {
|
||||
csvFilesGenerated: 0,
|
||||
csvBytesWritten: 0,
|
||||
copyStatementsGenerated: 0,
|
||||
totalProcessingTime: 0
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize temp directory for CSV files
|
||||
*/
|
||||
initTempDir() {
|
||||
try {
|
||||
mkdirSync(this.tempDir, { recursive: true });
|
||||
console.log(`📁 Created temp directory: ${this.tempDir}`);
|
||||
} catch (error) {
|
||||
console.warn(`⚠️ Temp directory already exists or creation failed: ${error.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Clean up temp directory
|
||||
*/
|
||||
cleanup() {
|
||||
try {
|
||||
rmSync(this.tempDir, { recursive: true, force: true });
|
||||
console.log(`🧹 Cleaned up temp directory: ${this.tempDir}`);
|
||||
} catch (error) {
|
||||
console.warn(`⚠️ Cleanup failed: ${error.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Simulate the COPY approach for nodes
|
||||
*/
|
||||
async processNodesCopy(nodes) {
|
||||
const startTime = performance.now();
|
||||
console.log(`📊 Processing ${nodes.length} nodes using COPY approach...`);
|
||||
|
||||
// Group nodes by label
|
||||
const nodesByLabel = {};
|
||||
nodes.forEach(node => {
|
||||
if (!nodesByLabel[node.label]) nodesByLabel[node.label] = [];
|
||||
nodesByLabel[node.label].push(node);
|
||||
});
|
||||
|
||||
const copyStatements = [];
|
||||
|
||||
for (const [label, labelNodes] of Object.entries(nodesByLabel)) {
|
||||
console.log(` 📝 Generating CSV for ${labelNodes.length} ${label} nodes...`);
|
||||
|
||||
// Generate CSV
|
||||
const csv = CSVGenerator.generateNodeCSV(labelNodes, label);
|
||||
|
||||
// Write to temp file (simulating FS.writeFile)
|
||||
const csvPath = join(this.tempDir, `${label.toLowerCase()}_nodes.csv`);
|
||||
writeFileSync(csvPath, csv);
|
||||
|
||||
this.stats.csvFilesGenerated++;
|
||||
this.stats.csvBytesWritten += csv.length;
|
||||
|
||||
console.log(` ✅ Generated ${csvPath} (${csv.length} bytes)`);
|
||||
console.log(` 📄 Preview: ${csv.split('\n')[0]}...`);
|
||||
|
||||
// Generate COPY statement (what we would execute in KuzuDB)
|
||||
const copyStatement = `COPY ${label} FROM '/${label.toLowerCase()}_nodes.csv'`;
|
||||
copyStatements.push(copyStatement);
|
||||
this.stats.copyStatementsGenerated++;
|
||||
|
||||
console.log(` 🔧 COPY statement: ${copyStatement}`);
|
||||
}
|
||||
|
||||
const processingTime = performance.now() - startTime;
|
||||
this.stats.totalProcessingTime += processingTime;
|
||||
|
||||
console.log(`✅ Node COPY processing completed in ${processingTime.toFixed(2)}ms`);
|
||||
return copyStatements;
|
||||
}
|
||||
|
||||
/**
|
||||
* Simulate the COPY approach for relationships
|
||||
*/
|
||||
async processRelationshipsCopy(relationships) {
|
||||
const startTime = performance.now();
|
||||
console.log(`🔗 Processing ${relationships.length} relationships using COPY approach...`);
|
||||
|
||||
// Group relationships by type
|
||||
const relsByType = {};
|
||||
relationships.forEach(rel => {
|
||||
if (!relsByType[rel.type]) relsByType[rel.type] = [];
|
||||
relsByType[rel.type].push(rel);
|
||||
});
|
||||
|
||||
const copyStatements = [];
|
||||
|
||||
for (const [type, typeRels] of Object.entries(relsByType)) {
|
||||
console.log(` 📝 Generating CSV for ${typeRels.length} ${type} relationships...`);
|
||||
|
||||
// Generate CSV
|
||||
const csv = CSVGenerator.generateRelationshipCSV(typeRels, type);
|
||||
|
||||
// Write to temp file (simulating FS.writeFile)
|
||||
const csvPath = join(this.tempDir, `${type.toLowerCase()}_rels.csv`);
|
||||
writeFileSync(csvPath, csv);
|
||||
|
||||
this.stats.csvFilesGenerated++;
|
||||
this.stats.csvBytesWritten += csv.length;
|
||||
|
||||
console.log(` ✅ Generated ${csvPath} (${csv.length} bytes)`);
|
||||
console.log(` 📄 Preview: ${csv.split('\n')[0]}...`);
|
||||
|
||||
// Generate COPY statement (what we would execute in KuzuDB)
|
||||
const copyStatement = `COPY ${type} FROM '/${type.toLowerCase()}_rels.csv'`;
|
||||
copyStatements.push(copyStatement);
|
||||
this.stats.copyStatementsGenerated++;
|
||||
|
||||
console.log(` 🔧 COPY statement: ${copyStatement}`);
|
||||
}
|
||||
|
||||
const processingTime = performance.now() - startTime;
|
||||
this.stats.totalProcessingTime += processingTime;
|
||||
|
||||
console.log(`✅ Relationship COPY processing completed in ${processingTime.toFixed(2)}ms`);
|
||||
return copyStatements;
|
||||
}
|
||||
|
||||
/**
|
||||
* Simulate current MERGE batch approach for comparison
|
||||
*/
|
||||
async processNodesMerge(nodes) {
|
||||
const startTime = performance.now();
|
||||
console.log(`📊 Processing ${nodes.length} nodes using current MERGE approach...`);
|
||||
|
||||
const batchSize = 100;
|
||||
const batches = Math.ceil(nodes.length / batchSize);
|
||||
let totalStatements = 0;
|
||||
|
||||
for (let i = 0; i < batches; i++) {
|
||||
const batch = nodes.slice(i * batchSize, (i + 1) * batchSize);
|
||||
|
||||
// Group by label for batch processing
|
||||
const nodesByLabel = {};
|
||||
batch.forEach(node => {
|
||||
if (!nodesByLabel[node.label]) nodesByLabel[node.label] = [];
|
||||
nodesByLabel[node.label].push(node);
|
||||
});
|
||||
|
||||
for (const [label, labelNodes] of Object.entries(nodesByLabel)) {
|
||||
// Generate individual MERGE statements
|
||||
const mergeStatements = labelNodes.map(node => {
|
||||
const props = Object.entries(node.properties)
|
||||
.map(([key, value]) => `${key}: ${JSON.stringify(value)}`)
|
||||
.join(', ');
|
||||
return `MERGE (n:${label} {id: ${JSON.stringify(node.id)}, ${props}})`;
|
||||
});
|
||||
|
||||
totalStatements += mergeStatements.length;
|
||||
|
||||
// This would be concatenated and executed as one query
|
||||
const batchQuery = mergeStatements.join(';\n');
|
||||
console.log(` 🔧 Generated batch with ${mergeStatements.length} MERGE statements for ${label}`);
|
||||
}
|
||||
}
|
||||
|
||||
const processingTime = performance.now() - startTime;
|
||||
console.log(`✅ MERGE processing completed in ${processingTime.toFixed(2)}ms`);
|
||||
console.log(`📊 Generated ${totalStatements} individual MERGE statements`);
|
||||
|
||||
return { totalStatements, processingTime };
|
||||
}
|
||||
|
||||
/**
|
||||
* Print performance comparison
|
||||
*/
|
||||
printComparison(mergeStats) {
|
||||
console.log('\n' + '='.repeat(60));
|
||||
console.log('📊 PERFORMANCE COMPARISON');
|
||||
console.log('='.repeat(60));
|
||||
|
||||
console.log('\n🚀 COPY Approach:');
|
||||
console.log(` 📁 CSV files generated: ${this.stats.csvFilesGenerated}`);
|
||||
console.log(` 💾 Total CSV bytes: ${this.stats.csvBytesWritten.toLocaleString()}`);
|
||||
console.log(` 📥 COPY statements: ${this.stats.copyStatementsGenerated}`);
|
||||
console.log(` ⏱️ Processing time: ${this.stats.totalProcessingTime.toFixed(2)}ms`);
|
||||
|
||||
console.log('\n🔄 Current MERGE Approach:');
|
||||
console.log(` 🔧 MERGE statements: ${mergeStats.totalStatements}`);
|
||||
console.log(` ⏱️ Processing time: ${mergeStats.processingTime.toFixed(2)}ms`);
|
||||
|
||||
console.log('\n📈 Performance Analysis:');
|
||||
const speedupRatio = mergeStats.processingTime / this.stats.totalProcessingTime;
|
||||
console.log(` 🚀 COPY is ${speedupRatio.toFixed(1)}x faster at data preparation`);
|
||||
console.log(` 📊 Reduced operations: ${mergeStats.totalStatements} → ${this.stats.copyStatementsGenerated} (${((1 - this.stats.copyStatementsGenerated / mergeStats.totalStatements) * 100).toFixed(1)}% reduction)`);
|
||||
console.log(` 💾 Database load: Bulk operations vs individual statements`);
|
||||
console.log(` 🎯 Memory efficiency: Stream processing vs string concatenation`);
|
||||
|
||||
console.log('\n✅ RECOMMENDATION: COPY approach is superior for GitNexus bulk loading!');
|
||||
}
|
||||
|
||||
/**
|
||||
* Print statistics
|
||||
*/
|
||||
printStats() {
|
||||
console.log('\n📊 Final Statistics:');
|
||||
console.log(` Nodes processed: ${sampleNodes.length}`);
|
||||
console.log(` Relationships processed: ${sampleRelationships.length}`);
|
||||
console.log(` CSV files generated: ${this.stats.csvFilesGenerated}`);
|
||||
console.log(` Total CSV size: ${this.stats.csvBytesWritten} bytes`);
|
||||
console.log(` COPY statements: ${this.stats.copyStatementsGenerated}`);
|
||||
console.log(` Total processing time: ${this.stats.totalProcessingTime.toFixed(2)}ms`);
|
||||
}
|
||||
}
|
||||
|
||||
// Run the simulation
|
||||
async function runSimulation() {
|
||||
console.log('🚀 KuzuDB COPY Approach Simulation for GitNexus');
|
||||
console.log('='.repeat(60));
|
||||
console.log('This simulation demonstrates how COPY would work in GitNexus\n');
|
||||
|
||||
const simulator = new KuzuCopySimulator();
|
||||
|
||||
try {
|
||||
// Setup
|
||||
simulator.initTempDir();
|
||||
|
||||
console.log('📋 Sample GitNexus Data:');
|
||||
console.log(` 📊 Nodes: ${sampleNodes.length} (${[...new Set(sampleNodes.map(n => n.label))].join(', ')})`);
|
||||
console.log(` 🔗 Relationships: ${sampleRelationships.length} (${[...new Set(sampleRelationships.map(r => r.type))].join(', ')})`);
|
||||
console.log('');
|
||||
|
||||
// Test COPY approach
|
||||
console.log('🎯 TESTING COPY APPROACH');
|
||||
console.log('-'.repeat(40));
|
||||
const nodeCopyStatements = await simulator.processNodesCopy(sampleNodes);
|
||||
const relCopyStatements = await simulator.processRelationshipsCopy(sampleRelationships);
|
||||
|
||||
console.log('\n📥 Generated COPY Statements:');
|
||||
[...nodeCopyStatements, ...relCopyStatements].forEach(stmt => {
|
||||
console.log(` ${stmt}`);
|
||||
});
|
||||
|
||||
// Test current MERGE approach for comparison
|
||||
console.log('\n🔄 TESTING CURRENT MERGE APPROACH (for comparison)');
|
||||
console.log('-'.repeat(40));
|
||||
const mergeStats = await simulator.processNodesMerge(sampleNodes);
|
||||
|
||||
// Print comparison
|
||||
simulator.printComparison(mergeStats);
|
||||
|
||||
// Show generated CSV samples
|
||||
console.log('\n📄 Generated CSV Samples:');
|
||||
console.log('-'.repeat(40));
|
||||
|
||||
try {
|
||||
const projectCSV = readFileSync(join(simulator.tempDir, 'project_nodes.csv'), 'utf8');
|
||||
console.log('Project nodes CSV:');
|
||||
console.log(projectCSV);
|
||||
console.log('');
|
||||
|
||||
const containsCSV = readFileSync(join(simulator.tempDir, 'contains_rels.csv'), 'utf8');
|
||||
console.log('Contains relationships CSV:');
|
||||
console.log(containsCSV);
|
||||
console.log('');
|
||||
} catch (error) {
|
||||
console.warn('Could not read sample CSV files:', error.message);
|
||||
}
|
||||
|
||||
simulator.printStats();
|
||||
|
||||
console.log('\n🎉 Simulation completed successfully!');
|
||||
console.log('\n💡 Key Takeaways for GitNexus:');
|
||||
console.log(' • CSV generation is fast and memory-efficient');
|
||||
console.log(' • COPY statements reduce database operations significantly');
|
||||
console.log(' • Bulk loading scales better than individual MERGE statements');
|
||||
console.log(' • FS.writeFile + COPY is the optimal approach for large datasets');
|
||||
console.log(' • Fallback to MERGE batching ensures compatibility');
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Simulation failed:', error);
|
||||
} finally {
|
||||
// Cleanup
|
||||
simulator.cleanup();
|
||||
}
|
||||
}
|
||||
|
||||
// Run the simulation
|
||||
runSimulation()
|
||||
.then(() => process.exit(0))
|
||||
.catch(error => {
|
||||
console.error('Simulation error:', error);
|
||||
process.exit(1);
|
||||
});
|
||||
@@ -1,317 +0,0 @@
|
||||
/**
|
||||
* KuzuDB COPY Verification Test
|
||||
*
|
||||
* This test actually verifies that:
|
||||
* 1. We can write CSV data to KuzuDB WASM filesystem
|
||||
* 2. COPY statements work to load the data
|
||||
* 3. We can query the data back to confirm it was loaded
|
||||
*
|
||||
* This is a REAL test that proves the approach works.
|
||||
*/
|
||||
|
||||
import { createRequire } from 'module';
|
||||
const require = createRequire(import.meta.url);
|
||||
|
||||
async function verifyKuzuCopyApproach() {
|
||||
console.log('🧪 KuzuDB COPY Approach Verification Test');
|
||||
console.log('=' .repeat(50));
|
||||
console.log('This test will ACTUALLY verify that COPY works!\n');
|
||||
|
||||
let kuzu, db, conn;
|
||||
let testsPassed = 0;
|
||||
let testsTotal = 0;
|
||||
|
||||
function test(description, condition) {
|
||||
testsTotal++;
|
||||
if (condition) {
|
||||
console.log(`✅ TEST ${testsTotal}: ${description}`);
|
||||
testsPassed++;
|
||||
return true;
|
||||
} else {
|
||||
console.log(`❌ TEST ${testsTotal}: ${description}`);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
// Step 1: Load KuzuDB module
|
||||
console.log('📦 Step 1: Loading KuzuDB module...');
|
||||
try {
|
||||
kuzu = require('kuzu-wasm');
|
||||
test('KuzuDB module loaded successfully', !!kuzu);
|
||||
console.log(` Available APIs: ${Object.keys(kuzu.default || kuzu).join(', ')}`);
|
||||
} catch (error) {
|
||||
test('KuzuDB module loaded successfully', false);
|
||||
throw new Error(`Failed to load kuzu-wasm: ${error.message}`);
|
||||
}
|
||||
|
||||
// Step 2: Initialize KuzuDB
|
||||
console.log('\n🔧 Step 2: Initializing KuzuDB...');
|
||||
try {
|
||||
// Try different initialization approaches
|
||||
if (kuzu.default && kuzu.default.init) {
|
||||
await kuzu.default.init();
|
||||
kuzu = kuzu.default; // Use the default export
|
||||
} else if (kuzu.init) {
|
||||
await kuzu.init();
|
||||
} else {
|
||||
console.log(' ⚠️ No init method found, trying direct usage...');
|
||||
}
|
||||
|
||||
test('KuzuDB initialized successfully', true);
|
||||
|
||||
// Check FS API availability
|
||||
const hasFS = !!(kuzu.FS && kuzu.FS.writeFile);
|
||||
test('FS API (writeFile) is available', hasFS);
|
||||
|
||||
if (hasFS) {
|
||||
console.log(` FS methods: ${Object.keys(kuzu.FS).join(', ')}`);
|
||||
} else {
|
||||
console.log(' ⚠️ FS API not available - will test fallback approach');
|
||||
}
|
||||
} catch (error) {
|
||||
test('KuzuDB initialized successfully', false);
|
||||
throw new Error(`KuzuDB initialization failed: ${error.message}`);
|
||||
}
|
||||
|
||||
// Step 3: Create database and connection
|
||||
console.log('\n🗃️ Step 3: Creating database and connection...');
|
||||
try {
|
||||
db = new kuzu.Database('');
|
||||
conn = new kuzu.Connection(db);
|
||||
test('Database and connection created', !!(db && conn));
|
||||
} catch (error) {
|
||||
test('Database and connection created', false);
|
||||
throw new Error(`Database creation failed: ${error.message}`);
|
||||
}
|
||||
|
||||
// Step 4: Create schema
|
||||
console.log('\n📋 Step 4: Creating test schema...');
|
||||
try {
|
||||
await conn.query('CREATE NODE TABLE TestUser(name STRING, age INT64, PRIMARY KEY (name))');
|
||||
await conn.query('CREATE NODE TABLE TestCity(name STRING, population INT64, PRIMARY KEY (name))');
|
||||
await conn.query('CREATE REL TABLE TestFollows(FROM TestUser TO TestUser, since INT64)');
|
||||
test('Schema created successfully', true);
|
||||
} catch (error) {
|
||||
test('Schema created successfully', false);
|
||||
throw new Error(`Schema creation failed: ${error.message}`);
|
||||
}
|
||||
|
||||
// Step 5: Test COPY approach (if FS is available)
|
||||
if (kuzu.FS && kuzu.FS.writeFile) {
|
||||
console.log('\n💾 Step 5: Testing COPY approach with FS.writeFile...');
|
||||
|
||||
try {
|
||||
// Prepare test data
|
||||
const userCSV = `Alice,25
|
||||
Bob,30
|
||||
Charlie,35
|
||||
Diana,28`;
|
||||
|
||||
const cityCSV = `NewYork,8000000
|
||||
London,9000000
|
||||
Tokyo,14000000`;
|
||||
|
||||
// Write CSV to WASM filesystem
|
||||
console.log(' 📝 Writing CSV files to WASM filesystem...');
|
||||
await kuzu.FS.writeFile('/test_users.csv', userCSV);
|
||||
await kuzu.FS.writeFile('/test_cities.csv', cityCSV);
|
||||
test('CSV files written to WASM filesystem', true);
|
||||
|
||||
// Execute COPY statements
|
||||
console.log(' 📥 Executing COPY statements...');
|
||||
const userCopyResult = await conn.query("COPY TestUser FROM '/test_users.csv'");
|
||||
await userCopyResult.close();
|
||||
|
||||
const cityCopyResult = await conn.query("COPY TestCity FROM '/test_cities.csv'");
|
||||
await cityCopyResult.close();
|
||||
|
||||
test('COPY statements executed successfully', true);
|
||||
|
||||
} catch (error) {
|
||||
test('COPY statements executed successfully', false);
|
||||
console.log(` ❌ COPY approach failed: ${error.message}`);
|
||||
console.log(' 🔄 Falling back to INSERT statements...');
|
||||
|
||||
// Fallback to INSERT
|
||||
await conn.query("CREATE (u:TestUser {name: 'Alice', age: 25})");
|
||||
await conn.query("CREATE (u:TestUser {name: 'Bob', age: 30})");
|
||||
await conn.query("CREATE (u:TestUser {name: 'Charlie', age: 35})");
|
||||
await conn.query("CREATE (c:TestCity {name: 'NewYork', population: 8000000})");
|
||||
await conn.query("CREATE (c:TestCity {name: 'London', population: 9000000})");
|
||||
test('Fallback INSERT statements executed', true);
|
||||
}
|
||||
|
||||
} else {
|
||||
console.log('\n🔄 Step 5: FS not available, using INSERT statements...');
|
||||
try {
|
||||
await conn.query("CREATE (u:TestUser {name: 'Alice', age: 25})");
|
||||
await conn.query("CREATE (u:TestUser {name: 'Bob', age: 30})");
|
||||
await conn.query("CREATE (u:TestUser {name: 'Charlie', age: 35})");
|
||||
await conn.query("CREATE (c:TestCity {name: 'NewYork', population: 8000000})");
|
||||
await conn.query("CREATE (c:TestCity {name: 'London', population: 9000000})");
|
||||
test('INSERT statements executed successfully', true);
|
||||
} catch (error) {
|
||||
test('INSERT statements executed successfully', false);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
// Step 6: VERIFY DATA WAS LOADED - This is the crucial part!
|
||||
console.log('\n🔍 Step 6: VERIFYING data was actually loaded...');
|
||||
|
||||
try {
|
||||
// Count users
|
||||
console.log(' 📊 Counting users...');
|
||||
const userCountResult = await conn.query('MATCH (u:TestUser) RETURN count(u) as userCount');
|
||||
const userRows = await userCountResult.getAllObjects();
|
||||
const userCount = userRows[0]?.userCount || 0;
|
||||
await userCountResult.close();
|
||||
|
||||
test(`Users loaded correctly (expected: 3-4, got: ${userCount})`, userCount >= 3);
|
||||
console.log(` Found ${userCount} users`);
|
||||
|
||||
// Count cities
|
||||
console.log(' 🏙️ Counting cities...');
|
||||
const cityCountResult = await conn.query('MATCH (c:TestCity) RETURN count(c) as cityCount');
|
||||
const cityRows = await cityCountResult.getAllObjects();
|
||||
const cityCount = cityRows[0]?.cityCount || 0;
|
||||
await cityCountResult.close();
|
||||
|
||||
test(`Cities loaded correctly (expected: 2-3, got: ${cityCount})`, cityCount >= 2);
|
||||
console.log(` Found ${cityCount} cities`);
|
||||
|
||||
// Get actual user data
|
||||
console.log(' 👥 Retrieving user data...');
|
||||
const usersResult = await conn.query('MATCH (u:TestUser) RETURN u.name, u.age ORDER BY u.name');
|
||||
const users = await usersResult.getAllObjects();
|
||||
await usersResult.close();
|
||||
|
||||
test('User data retrieved successfully', users.length > 0);
|
||||
console.log(' Users found:');
|
||||
users.forEach(user => {
|
||||
console.log(` - ${user['u.name']}: ${user['u.age']} years old`);
|
||||
});
|
||||
|
||||
// Get actual city data
|
||||
console.log(' 🏙️ Retrieving city data...');
|
||||
const citiesResult = await conn.query('MATCH (c:TestCity) RETURN c.name, c.population ORDER BY c.population DESC');
|
||||
const cities = await citiesResult.getAllObjects();
|
||||
await citiesResult.close();
|
||||
|
||||
test('City data retrieved successfully', cities.length > 0);
|
||||
console.log(' Cities found:');
|
||||
cities.forEach(city => {
|
||||
console.log(` - ${city['c.name']}: ${city['c.population'].toLocaleString()} population`);
|
||||
});
|
||||
|
||||
// Test a complex query to make sure relationships work
|
||||
console.log(' 🔗 Testing relationship creation...');
|
||||
await conn.query("MATCH (u1:TestUser {name: 'Alice'}), (u2:TestUser {name: 'Bob'}) CREATE (u1)-[:TestFollows {since: 2023}]->(u2)");
|
||||
|
||||
const relResult = await conn.query('MATCH (u1:TestUser)-[f:TestFollows]->(u2:TestUser) RETURN u1.name, u2.name, f.since');
|
||||
const relationships = await relResult.getAllObjects();
|
||||
await relResult.close();
|
||||
|
||||
test('Relationships created and queried successfully', relationships.length > 0);
|
||||
relationships.forEach(rel => {
|
||||
console.log(` - ${rel['u1.name']} follows ${rel['u2.name']} since ${rel['f.since']}`);
|
||||
});
|
||||
|
||||
} catch (error) {
|
||||
test('Data verification completed', false);
|
||||
throw new Error(`Data verification failed: ${error.message}`);
|
||||
}
|
||||
|
||||
// Step 7: Final verification with complex query
|
||||
console.log('\n🎯 Step 7: Final complex query test...');
|
||||
try {
|
||||
const complexResult = await conn.query(`
|
||||
MATCH (u:TestUser), (c:TestCity)
|
||||
WHERE u.age > 25 AND c.population > 8000000
|
||||
RETURN u.name as user, u.age, c.name as city, c.population
|
||||
ORDER BY u.age DESC, c.population DESC
|
||||
`);
|
||||
|
||||
const complexRows = await complexResult.getAllObjects();
|
||||
await complexResult.close();
|
||||
|
||||
test('Complex query executed successfully', true);
|
||||
console.log(` Complex query returned ${complexRows.length} rows:`);
|
||||
complexRows.forEach(row => {
|
||||
console.log(` - ${row.user} (${row['u.age']}) × ${row.city} (${row['c.population'].toLocaleString()})`);
|
||||
});
|
||||
|
||||
} catch (error) {
|
||||
test('Complex query executed successfully', false);
|
||||
console.log(` ❌ Complex query failed: ${error.message}`);
|
||||
}
|
||||
|
||||
} catch (error) {
|
||||
console.error(`\n💥 Test failed with error: ${error.message}`);
|
||||
if (error.stack) {
|
||||
console.error('Stack trace:', error.stack);
|
||||
}
|
||||
} finally {
|
||||
// Cleanup
|
||||
if (conn) {
|
||||
try {
|
||||
await conn.close();
|
||||
console.log('\n🔒 Connection closed');
|
||||
} catch (e) {
|
||||
console.warn('Warning: Failed to close connection:', e.message);
|
||||
}
|
||||
}
|
||||
|
||||
if (db) {
|
||||
try {
|
||||
await db.close();
|
||||
console.log('🔒 Database closed');
|
||||
} catch (e) {
|
||||
console.warn('Warning: Failed to close database:', e.message);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Final results
|
||||
console.log('\n' + '='.repeat(50));
|
||||
console.log('📊 FINAL TEST RESULTS');
|
||||
console.log('='.repeat(50));
|
||||
console.log(`✅ Tests passed: ${testsPassed}/${testsTotal}`);
|
||||
console.log(`📊 Success rate: ${((testsPassed / testsTotal) * 100).toFixed(1)}%`);
|
||||
|
||||
if (testsPassed === testsTotal) {
|
||||
console.log('\n🎉 ALL TESTS PASSED!');
|
||||
console.log('✅ KuzuDB COPY approach is VERIFIED and ready for GitNexus!');
|
||||
|
||||
console.log('\n💡 Key findings:');
|
||||
if (kuzu.FS && kuzu.FS.writeFile) {
|
||||
console.log(' • FS.writeFile works in KuzuDB WASM');
|
||||
console.log(' • COPY statements successfully load data from CSV');
|
||||
console.log(' • Data can be queried back correctly');
|
||||
console.log(' • Complex queries work as expected');
|
||||
console.log(' • READY FOR GITNEXUS INTEGRATION! 🚀');
|
||||
} else {
|
||||
console.log(' • FS API not available in this environment');
|
||||
console.log(' • INSERT statements work as fallback');
|
||||
console.log(' • Data can be queried back correctly');
|
||||
console.log(' • Consider environment detection in GitNexus');
|
||||
}
|
||||
} else {
|
||||
console.log('\n❌ SOME TESTS FAILED');
|
||||
console.log('⚠️ COPY approach needs further investigation');
|
||||
console.log('🔄 Consider fallback to current MERGE approach');
|
||||
}
|
||||
|
||||
return testsPassed === testsTotal;
|
||||
}
|
||||
|
||||
// Run the verification
|
||||
verifyKuzuCopyApproach()
|
||||
.then(success => {
|
||||
process.exit(success ? 0 : 1);
|
||||
})
|
||||
.catch(error => {
|
||||
console.error('Verification failed:', error);
|
||||
process.exit(1);
|
||||
});
|
||||
-550
@@ -1,550 +0,0 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||||
<title>KuzuDB FS COPY Test v2</title>
|
||||
<meta http-equiv="Cache-Control" content="no-cache, no-store, must-revalidate">
|
||||
<meta http-equiv="Pragma" content="no-cache">
|
||||
<meta http-equiv="Expires" content="0">
|
||||
<style>
|
||||
body {
|
||||
font-family: 'Courier New', monospace;
|
||||
max-width: 1200px;
|
||||
margin: 0 auto;
|
||||
padding: 20px;
|
||||
background-color: #1a1a1a;
|
||||
color: #00ff00;
|
||||
}
|
||||
.output {
|
||||
background-color: #000;
|
||||
padding: 20px;
|
||||
border-radius: 8px;
|
||||
border: 1px solid #333;
|
||||
white-space: pre-wrap;
|
||||
font-family: 'Courier New', monospace;
|
||||
font-size: 14px;
|
||||
max-height: 600px;
|
||||
overflow-y: auto;
|
||||
margin-bottom: 20px;
|
||||
}
|
||||
.success { color: #00ff00; }
|
||||
.error { color: #ff4444; }
|
||||
.warning { color: #ffaa00; }
|
||||
.info { color: #44aaff; }
|
||||
button {
|
||||
background-color: #333;
|
||||
color: #fff;
|
||||
border: 1px solid #555;
|
||||
padding: 10px 20px;
|
||||
border-radius: 4px;
|
||||
cursor: pointer;
|
||||
margin-right: 10px;
|
||||
margin-bottom: 10px;
|
||||
}
|
||||
button:hover {
|
||||
background-color: #555;
|
||||
}
|
||||
button:disabled {
|
||||
background-color: #222;
|
||||
color: #666;
|
||||
cursor: not-allowed;
|
||||
}
|
||||
.status {
|
||||
padding: 10px;
|
||||
border-radius: 4px;
|
||||
margin-bottom: 20px;
|
||||
font-weight: bold;
|
||||
}
|
||||
.status.loading {
|
||||
background-color: #333;
|
||||
color: #ffaa00;
|
||||
}
|
||||
.status.success {
|
||||
background-color: #004400;
|
||||
color: #00ff00;
|
||||
}
|
||||
.status.error {
|
||||
background-color: #440000;
|
||||
color: #ff4444;
|
||||
}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<h1>🚀 KuzuDB WASM FS & COPY Test</h1>
|
||||
<p>This test verifies that we can use COPY statements with FS.writeFile in KuzuDB WASM for GitNexus.</p>
|
||||
|
||||
<div id="status" class="status loading">
|
||||
🔄 Initializing KuzuDB WASM...
|
||||
</div>
|
||||
|
||||
<div>
|
||||
<button id="runTest" disabled>Run Full Test</button>
|
||||
<button id="clearOutput">Clear Output</button>
|
||||
<button id="runFSTest" disabled>Test FS Only</button>
|
||||
<button id="runCopyTest" disabled>Test COPY Only</button>
|
||||
</div>
|
||||
|
||||
<div id="output" class="output">Initializing...</div>
|
||||
|
||||
<script type="module">
|
||||
// Import kuzu-wasm from the local node_modules
|
||||
// Note: In a real browser environment, this would be served properly
|
||||
import kuzu from './node_modules/kuzu-wasm/index.js';
|
||||
|
||||
let db = null;
|
||||
let conn = null;
|
||||
let isInitialized = false;
|
||||
|
||||
const output = document.getElementById('output');
|
||||
const status = document.getElementById('status');
|
||||
const runTestBtn = document.getElementById('runTest');
|
||||
const clearOutputBtn = document.getElementById('clearOutput');
|
||||
const runFSTestBtn = document.getElementById('runFSTest');
|
||||
const runCopyTestBtn = document.getElementById('runCopyTest');
|
||||
|
||||
function log(message, type = 'info') {
|
||||
const timestamp = new Date().toLocaleTimeString();
|
||||
const className = type === 'error' ? 'error' : type === 'warning' ? 'warning' : type === 'success' ? 'success' : 'info';
|
||||
output.innerHTML += `<span class="${className}">[${timestamp}] ${message}</span>\n`;
|
||||
output.scrollTop = output.scrollHeight;
|
||||
console.log(message);
|
||||
}
|
||||
|
||||
function setStatus(message, type = 'loading') {
|
||||
status.textContent = message;
|
||||
status.className = `status ${type}`;
|
||||
}
|
||||
|
||||
function enableButtons() {
|
||||
runTestBtn.disabled = false;
|
||||
runFSTestBtn.disabled = false;
|
||||
runCopyTestBtn.disabled = false;
|
||||
}
|
||||
|
||||
function disableButtons() {
|
||||
runTestBtn.disabled = true;
|
||||
runFSTestBtn.disabled = true;
|
||||
runCopyTestBtn.disabled = true;
|
||||
}
|
||||
|
||||
// Initialize KuzuDB
|
||||
async function initializeKuzu() {
|
||||
try {
|
||||
log('🚀 Starting KuzuDB WASM initialization...', 'info');
|
||||
|
||||
// Initialize kuzu
|
||||
await kuzu.init();
|
||||
log('✅ KuzuDB WASM initialized successfully', 'success');
|
||||
|
||||
// Check available APIs
|
||||
const apis = Object.keys(kuzu);
|
||||
log(`📋 Available APIs: ${apis.join(', ')}`, 'info');
|
||||
|
||||
// Check FS API
|
||||
if (kuzu.FS) {
|
||||
const fsMethods = Object.keys(kuzu.FS);
|
||||
log(`💾 FS API available with methods: ${fsMethods.join(', ')}`, 'success');
|
||||
} else {
|
||||
log('⚠️ FS API not available', 'warning');
|
||||
}
|
||||
|
||||
// Create database
|
||||
log('🗃️ Creating in-memory database...', 'info');
|
||||
db = new kuzu.Database('');
|
||||
log('✅ Database created', 'success');
|
||||
|
||||
// Create connection
|
||||
log('🔗 Creating connection...', 'info');
|
||||
conn = new kuzu.Connection(db);
|
||||
log('✅ Connection established', 'success');
|
||||
|
||||
isInitialized = true;
|
||||
setStatus('✅ KuzuDB Ready - Click "Run Full Test" to start', 'success');
|
||||
enableButtons();
|
||||
|
||||
} catch (error) {
|
||||
log(`❌ Initialization failed: ${error.message}`, 'error');
|
||||
setStatus(`❌ Initialization failed: ${error.message}`, 'error');
|
||||
console.error('Full error:', error);
|
||||
}
|
||||
}
|
||||
|
||||
// Test FS functionality
|
||||
async function testFS() {
|
||||
if (!isInitialized) {
|
||||
log('❌ KuzuDB not initialized', 'error');
|
||||
return false;
|
||||
}
|
||||
|
||||
try {
|
||||
log('🧪 Testing FS functionality...', 'info');
|
||||
|
||||
if (!kuzu.FS || !kuzu.FS.writeFile) {
|
||||
log('⚠️ FS.writeFile not available, skipping FS test', 'warning');
|
||||
return false;
|
||||
}
|
||||
|
||||
// Test data
|
||||
const testData = `name,age
|
||||
Alice,25
|
||||
Bob,30
|
||||
Charlie,35`;
|
||||
|
||||
log('📝 Writing test CSV to WASM filesystem...', 'info');
|
||||
await kuzu.FS.writeFile('/test.csv', testData);
|
||||
log('✅ Successfully wrote test.csv to WASM filesystem', 'success');
|
||||
|
||||
// Try to read it back
|
||||
if (kuzu.FS.readFile) {
|
||||
log('📖 Reading test CSV back from WASM filesystem...', 'info');
|
||||
const readData = await kuzu.FS.readFile('/test.csv');
|
||||
|
||||
// Handle different data types (Buffer, Uint8Array, string)
|
||||
let dataStr;
|
||||
if (typeof readData === 'string') {
|
||||
dataStr = readData;
|
||||
} else if (readData instanceof Uint8Array || readData instanceof ArrayBuffer) {
|
||||
dataStr = new TextDecoder().decode(readData);
|
||||
} else if (readData && readData.toString) {
|
||||
dataStr = readData.toString();
|
||||
} else {
|
||||
dataStr = String(readData);
|
||||
}
|
||||
|
||||
log(`✅ Read back data: ${dataStr.length} bytes`, 'success');
|
||||
log(`📄 Content preview: ${dataStr.substring(0, 50)}...`, 'info');
|
||||
}
|
||||
|
||||
return true;
|
||||
|
||||
} catch (error) {
|
||||
log(`❌ FS test failed: ${error.message}`, 'error');
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Test COPY functionality
|
||||
async function testCopy() {
|
||||
if (!isInitialized) {
|
||||
log('❌ KuzuDB not initialized', 'error');
|
||||
return false;
|
||||
}
|
||||
|
||||
try {
|
||||
log('🧪 Testing COPY functionality...', 'info');
|
||||
|
||||
// Create schema first (with conflict handling)
|
||||
log('📋 Creating test schema...', 'info');
|
||||
|
||||
try {
|
||||
await conn.query('DROP TABLE TestUser IF EXISTS');
|
||||
await conn.query('DROP TABLE TestCity IF EXISTS');
|
||||
await conn.query('DROP TABLE TestFollows IF EXISTS');
|
||||
} catch (dropError) {
|
||||
// Tables might not exist, that's OK
|
||||
log(' (Cleaning up any existing tables...)', 'info');
|
||||
}
|
||||
|
||||
await conn.query('CREATE NODE TABLE TestUser(name STRING, age INT64, PRIMARY KEY (name))');
|
||||
log('✅ Created TestUser table', 'success');
|
||||
|
||||
// Prepare CSV data
|
||||
const csvData = `Alice,25
|
||||
Bob,30
|
||||
Charlie,35
|
||||
Diana,28
|
||||
Eve,32`;
|
||||
|
||||
if (kuzu.FS && kuzu.FS.writeFile) {
|
||||
// Method 1: Use FS.writeFile + COPY
|
||||
log('📝 Method 1: Using FS.writeFile + COPY...', 'info');
|
||||
|
||||
await kuzu.FS.writeFile('/users.csv', csvData);
|
||||
log('✅ Written users.csv to WASM filesystem', 'success');
|
||||
|
||||
log('📥 Executing COPY statement...', 'info');
|
||||
const copyResult = await conn.query("COPY TestUser FROM '/users.csv'");
|
||||
log(`✅ COPY executed: ${copyResult.toString()}`, 'success');
|
||||
await copyResult.close();
|
||||
|
||||
} else {
|
||||
// Method 2: Fallback to individual INSERTs
|
||||
log('🔄 Method 2: Using individual INSERT statements...', 'info');
|
||||
|
||||
const users = [
|
||||
['Alice', 25],
|
||||
['Bob', 30],
|
||||
['Charlie', 35],
|
||||
['Diana', 28],
|
||||
['Eve', 32]
|
||||
];
|
||||
|
||||
for (const [name, age] of users) {
|
||||
const result = await conn.query(`CREATE (u:TestUser {name: '${name}', age: ${age}})`);
|
||||
await result.close();
|
||||
}
|
||||
log('✅ Inserted users individually', 'success');
|
||||
}
|
||||
|
||||
// Verify data was loaded
|
||||
log('🔍 Verifying data was loaded...', 'info');
|
||||
const result = await conn.query('MATCH (u:TestUser) RETURN u.name, u.age ORDER BY u.name');
|
||||
const rows = await result.getAllObjects();
|
||||
|
||||
log(`✅ Found ${rows.length} users:`, 'success');
|
||||
rows.forEach(row => {
|
||||
log(` 👤 ${row['u.name']}: ${row['u.age']} years old`, 'info');
|
||||
});
|
||||
|
||||
await result.close();
|
||||
return true;
|
||||
|
||||
} catch (error) {
|
||||
log(`❌ COPY test failed: ${error.message}`, 'error');
|
||||
console.error('Full error:', error);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Full comprehensive test
|
||||
async function runFullTest() {
|
||||
disableButtons();
|
||||
setStatus('🧪 Running comprehensive test...', 'loading');
|
||||
|
||||
try {
|
||||
log('🎯 Starting comprehensive KuzuDB FS + COPY test...', 'info');
|
||||
log('=' .repeat(60), 'info');
|
||||
|
||||
// Test 1: FS functionality
|
||||
log('\n📁 TEST 1: FS Functionality', 'info');
|
||||
log('-'.repeat(30), 'info');
|
||||
const fsSuccess = await testFS();
|
||||
|
||||
// Test 2: Schema creation
|
||||
log('\n📋 TEST 2: Schema Creation', 'info');
|
||||
log('-'.repeat(30), 'info');
|
||||
|
||||
// Clean up any existing tables first
|
||||
try {
|
||||
await conn.query('DROP TABLE User IF EXISTS');
|
||||
await conn.query('DROP TABLE City IF EXISTS');
|
||||
await conn.query('DROP TABLE Follows IF EXISTS');
|
||||
await conn.query('DROP TABLE LivesIn IF EXISTS');
|
||||
log('✓ Cleaned up existing tables', 'info');
|
||||
} catch (cleanupError) {
|
||||
// Tables might not exist, that's OK
|
||||
}
|
||||
|
||||
await conn.query('CREATE NODE TABLE User(name STRING, age INT64, PRIMARY KEY (name))');
|
||||
log('✅ Created User table', 'success');
|
||||
|
||||
await conn.query('CREATE NODE TABLE City(name STRING, population INT64, PRIMARY KEY (name))');
|
||||
log('✅ Created City table', 'success');
|
||||
|
||||
await conn.query('CREATE REL TABLE Follows(FROM User TO User, since INT64)');
|
||||
log('✅ Created Follows relationship table', 'success');
|
||||
|
||||
await conn.query('CREATE REL TABLE LivesIn(FROM User TO City)');
|
||||
log('✅ Created LivesIn relationship table', 'success');
|
||||
|
||||
// Test 3: Data loading (COPY vs INSERT)
|
||||
log('\n📥 TEST 3: Data Loading', 'info');
|
||||
log('-'.repeat(30), 'info');
|
||||
|
||||
const userData = `Adam,30
|
||||
Karissa,40
|
||||
Zhang,50
|
||||
Noura,25
|
||||
TestUser1,35`;
|
||||
|
||||
const cityData = `Waterloo,150000
|
||||
Kitchener,200000
|
||||
Guelph,75000
|
||||
Toronto,2930000`;
|
||||
|
||||
if (kuzu.FS && kuzu.FS.writeFile) {
|
||||
// Use COPY method - Let's actually test it!
|
||||
log('📝 Using COPY method for bulk loading...', 'info');
|
||||
|
||||
try {
|
||||
await kuzu.FS.writeFile('/users.csv', userData);
|
||||
log('✓ Written users.csv to WASM filesystem', 'success');
|
||||
|
||||
await kuzu.FS.writeFile('/cities.csv', cityData);
|
||||
log('✓ Written cities.csv to WASM filesystem', 'success');
|
||||
|
||||
log('📥 Attempting COPY statements...', 'info');
|
||||
|
||||
const copyResult1 = await conn.query("COPY User FROM '/users.csv'");
|
||||
log(`✅ Users COPY executed: ${copyResult1.toString()}`, 'success');
|
||||
await copyResult1.close();
|
||||
|
||||
const copyResult2 = await conn.query("COPY City FROM '/cities.csv'");
|
||||
log(`✅ Cities COPY executed: ${copyResult2.toString()}`, 'success');
|
||||
await copyResult2.close();
|
||||
|
||||
log('🎉 COPY METHOD SUCCESSFUL!', 'success');
|
||||
|
||||
} catch (copyError) {
|
||||
log(`❌ COPY method failed: ${copyError.message}`, 'error');
|
||||
log('🔄 Falling back to INSERT method...', 'warning');
|
||||
|
||||
// Fallback to INSERT method
|
||||
const users = [
|
||||
['Adam', 30], ['Karissa', 40], ['Zhang', 50], ['Noura', 25], ['TestUser1', 35]
|
||||
];
|
||||
|
||||
const cities = [
|
||||
['Waterloo', 150000], ['Kitchener', 200000], ['Guelph', 75000], ['Toronto', 2930000]
|
||||
];
|
||||
|
||||
for (const [name, age] of users) {
|
||||
const result = await conn.query(`CREATE (u:User {name: '${name}', age: ${age}})`);
|
||||
await result.close();
|
||||
}
|
||||
|
||||
for (const [name, pop] of cities) {
|
||||
const result = await conn.query(`CREATE (c:City {name: '${name}', population: ${pop}})`);
|
||||
await result.close();
|
||||
}
|
||||
|
||||
log('✅ Fallback INSERT method completed', 'success');
|
||||
}
|
||||
|
||||
} else {
|
||||
// Use INSERT method
|
||||
log('🔄 Using INSERT method for data loading...', 'info');
|
||||
|
||||
const users = [
|
||||
['Adam', 30], ['Karissa', 40], ['Zhang', 50], ['Noura', 25], ['TestUser1', 35]
|
||||
];
|
||||
|
||||
const cities = [
|
||||
['Waterloo', 150000], ['Kitchener', 200000], ['Guelph', 75000], ['Toronto', 2930000]
|
||||
];
|
||||
|
||||
for (const [name, age] of users) {
|
||||
const result = await conn.query(`CREATE (u:User {name: '${name}', age: ${age}})`);
|
||||
await result.close();
|
||||
}
|
||||
|
||||
for (const [name, pop] of cities) {
|
||||
const result = await conn.query(`CREATE (c:City {name: '${name}', population: ${pop}})`);
|
||||
await result.close();
|
||||
}
|
||||
|
||||
log('✅ Data loaded via INSERT statements', 'success');
|
||||
}
|
||||
|
||||
// Test 4: Relationship creation
|
||||
log('\n🔗 TEST 4: Relationship Creation', 'info');
|
||||
log('-'.repeat(30), 'info');
|
||||
|
||||
const relationshipQueries = [
|
||||
"MATCH (u1:User {name: 'Adam'}), (u2:User {name: 'Karissa'}) CREATE (u1)-[:Follows {since: 2020}]->(u2)",
|
||||
"MATCH (u1:User {name: 'Adam'}), (u2:User {name: 'Zhang'}) CREATE (u1)-[:Follows {since: 2020}]->(u2)",
|
||||
"MATCH (u:User {name: 'Adam'}), (c:City {name: 'Waterloo'}) CREATE (u)-[:LivesIn]->(c)",
|
||||
"MATCH (u:User {name: 'Karissa'}), (c:City {name: 'Waterloo'}) CREATE (u)-[:LivesIn]->(c)",
|
||||
"MATCH (u:User {name: 'Zhang'}), (c:City {name: 'Kitchener'}) CREATE (u)-[:LivesIn]->(c)"
|
||||
];
|
||||
|
||||
for (const query of relationshipQueries) {
|
||||
const result = await conn.query(query);
|
||||
await result.close();
|
||||
}
|
||||
log('✅ Relationships created successfully', 'success');
|
||||
|
||||
// Test 5: Verification queries
|
||||
log('\n🔍 TEST 5: Data Verification', 'info');
|
||||
log('-'.repeat(30), 'info');
|
||||
|
||||
// Count queries
|
||||
let result = await conn.query('MATCH (u:User) RETURN count(u) as userCount');
|
||||
let rows = await result.getAllObjects();
|
||||
log(`👥 Total users: ${rows[0]?.userCount || 0}`, 'success');
|
||||
await result.close();
|
||||
|
||||
result = await conn.query('MATCH (c:City) RETURN count(c) as cityCount');
|
||||
rows = await result.getAllObjects();
|
||||
log(`🏙️ Total cities: ${rows[0]?.cityCount || 0}`, 'success');
|
||||
await result.close();
|
||||
|
||||
result = await conn.query('MATCH ()-[f:Follows]->() RETURN count(f) as followsCount');
|
||||
rows = await result.getAllObjects();
|
||||
log(`👥 Follow relationships: ${rows[0]?.followsCount || 0}`, 'success');
|
||||
await result.close();
|
||||
|
||||
result = await conn.query('MATCH ()-[l:LivesIn]->() RETURN count(l) as livesInCount');
|
||||
rows = await result.getAllObjects();
|
||||
log(`🏠 LivesIn relationships: ${rows[0]?.livesInCount || 0}`, 'success');
|
||||
await result.close();
|
||||
|
||||
// Complex query
|
||||
log('\n📊 Sample complex query results:', 'info');
|
||||
result = await conn.query(`
|
||||
MATCH (u:User)-[l:LivesIn]->(c:City)
|
||||
RETURN u.name as person, u.age, c.name as city, c.population
|
||||
ORDER BY c.population DESC, u.name
|
||||
`);
|
||||
rows = await result.getAllObjects();
|
||||
rows.forEach(row => {
|
||||
log(` ${row.person} (${row['u.age']}) lives in ${row.city} (pop: ${row['c.population']})`, 'info');
|
||||
});
|
||||
await result.close();
|
||||
|
||||
// Test summary
|
||||
log('\n' + '='.repeat(60), 'info');
|
||||
log('🎉 COMPREHENSIVE TEST COMPLETED SUCCESSFULLY!', 'success');
|
||||
log('', 'info');
|
||||
log('📊 Test Results Summary:', 'info');
|
||||
log(` ✅ KuzuDB initialization: SUCCESS`, 'success');
|
||||
log(` ${fsSuccess ? '✅' : '⚠️'} FS API functionality: ${fsSuccess ? 'SUCCESS' : 'NOT AVAILABLE'}`, fsSuccess ? 'success' : 'warning');
|
||||
log(` ✅ Schema creation: SUCCESS`, 'success');
|
||||
log(` ✅ Data loading: SUCCESS (${fsSuccess ? 'COPY method' : 'INSERT method'})`, 'success');
|
||||
log(` ✅ Relationship creation: SUCCESS`, 'success');
|
||||
log(` ✅ Query execution: SUCCESS`, 'success');
|
||||
log('', 'info');
|
||||
log('🚀 READY FOR GITNEXUS INTEGRATION!', 'success');
|
||||
|
||||
if (fsSuccess) {
|
||||
log('', 'info');
|
||||
log('💡 Key findings for GitNexus:', 'info');
|
||||
log(' • FS.writeFile works for creating CSV files in WASM filesystem', 'info');
|
||||
log(' • COPY statements work for bulk loading from CSV files', 'info');
|
||||
log(' • This approach will be much faster than individual MERGE statements', 'info');
|
||||
log(' • Perfect for GitNexus batch operations!', 'success');
|
||||
} else {
|
||||
log('', 'info');
|
||||
log('⚠️ Note for GitNexus:', 'warning');
|
||||
log(' • FS API might not be available in all environments', 'warning');
|
||||
log(' • Fallback to current MERGE batch approach recommended', 'warning');
|
||||
log(' • Consider environment detection for optimal method selection', 'info');
|
||||
}
|
||||
|
||||
setStatus('✅ Test completed successfully!', 'success');
|
||||
|
||||
} catch (error) {
|
||||
log(`❌ Full test failed: ${error.message}`, 'error');
|
||||
setStatus(`❌ Test failed: ${error.message}`, 'error');
|
||||
console.error('Full error:', error);
|
||||
} finally {
|
||||
enableButtons();
|
||||
}
|
||||
}
|
||||
|
||||
// Event listeners
|
||||
clearOutputBtn.addEventListener('click', () => {
|
||||
output.innerHTML = '';
|
||||
});
|
||||
|
||||
runTestBtn.addEventListener('click', runFullTest);
|
||||
runFSTestBtn.addEventListener('click', testFS);
|
||||
runCopyTestBtn.addEventListener('click', testCopy);
|
||||
|
||||
// Initialize on page load
|
||||
window.addEventListener('load', initializeKuzu);
|
||||
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
-255
@@ -1,255 +0,0 @@
|
||||
/**
|
||||
* KuzuDB WASM Filesystem COPY Test
|
||||
*
|
||||
* This test verifies that we can:
|
||||
* 1. Initialize KuzuDB WASM in Node.js environment
|
||||
* 2. Write CSV data to the WASM filesystem using FS.writeFile
|
||||
* 3. Use COPY statements to bulk load data
|
||||
* 4. Query the data to verify it was loaded correctly
|
||||
*/
|
||||
|
||||
const path = require('path');
|
||||
|
||||
async function testKuzuFSCopy() {
|
||||
let kuzu;
|
||||
let db;
|
||||
let conn;
|
||||
|
||||
try {
|
||||
console.log('🚀 Starting KuzuDB WASM FS COPY Test...\n');
|
||||
|
||||
// Step 1: Initialize KuzuDB WASM
|
||||
console.log('📦 Loading kuzu-wasm package...');
|
||||
kuzu = require('kuzu-wasm');
|
||||
|
||||
// Use Node.js version for testing
|
||||
console.log('🔧 Initializing KuzuDB (Node.js mode)...');
|
||||
const kuzuNodejs = require('kuzu-wasm/nodejs');
|
||||
|
||||
// Create database and connection
|
||||
console.log('🗃️ Creating in-memory database...');
|
||||
db = new kuzuNodejs.Database(':memory:');
|
||||
conn = new kuzuNodejs.Connection(db);
|
||||
|
||||
console.log('✅ KuzuDB initialized successfully\n');
|
||||
|
||||
// Step 2: Create schema
|
||||
console.log('📋 Creating database schema...');
|
||||
|
||||
// Create node tables
|
||||
await conn.query('CREATE NODE TABLE User(name STRING, age INT64, PRIMARY KEY (name))');
|
||||
console.log('✓ Created User node table');
|
||||
|
||||
await conn.query('CREATE NODE TABLE City(name STRING, population INT64, PRIMARY KEY (name))');
|
||||
console.log('✓ Created City node table');
|
||||
|
||||
// Create relationship tables
|
||||
await conn.query('CREATE REL TABLE Follows(FROM User TO User, since INT64)');
|
||||
console.log('✓ Created Follows relationship table');
|
||||
|
||||
await conn.query('CREATE REL TABLE LivesIn(FROM User TO City)');
|
||||
console.log('✓ Created LivesIn relationship table');
|
||||
|
||||
console.log('✅ Schema created successfully\n');
|
||||
|
||||
// Step 3: Prepare CSV data (simulating GitNexus graph data)
|
||||
console.log('📝 Preparing CSV data...');
|
||||
|
||||
const userCSV = `Adam,30
|
||||
Karissa,40
|
||||
Zhang,50
|
||||
Noura,25
|
||||
TestUser1,35
|
||||
TestUser2,28`;
|
||||
|
||||
const cityCSV = `Waterloo,150000
|
||||
Kitchener,200000
|
||||
Guelph,75000
|
||||
Toronto,2930000`;
|
||||
|
||||
const followsCSV = `Adam,Karissa,2020
|
||||
Adam,Zhang,2020
|
||||
Karissa,Zhang,2021
|
||||
Zhang,Noura,2022
|
||||
TestUser1,TestUser2,2023`;
|
||||
|
||||
const livesInCSV = `Adam,Waterloo
|
||||
Karissa,Waterloo
|
||||
Zhang,Kitchener
|
||||
Noura,Guelph
|
||||
TestUser1,Toronto
|
||||
TestUser2,Toronto`;
|
||||
|
||||
console.log('✓ CSV data prepared');
|
||||
|
||||
// Step 4: Write CSV files to WASM filesystem
|
||||
console.log('💾 Writing CSV files to WASM filesystem...');
|
||||
|
||||
// Note: For Node.js version, we might need to write actual files
|
||||
// Let's try both approaches
|
||||
|
||||
const fs = require('fs').promises;
|
||||
const tmpDir = './temp_kuzu_test';
|
||||
|
||||
// Create temp directory
|
||||
try {
|
||||
await fs.mkdir(tmpDir, { recursive: true });
|
||||
} catch (e) {
|
||||
// Directory might already exist
|
||||
}
|
||||
|
||||
// Write CSV files
|
||||
await fs.writeFile(path.join(tmpDir, 'user.csv'), userCSV);
|
||||
await fs.writeFile(path.join(tmpDir, 'city.csv'), cityCSV);
|
||||
await fs.writeFile(path.join(tmpDir, 'follows.csv'), followsCSV);
|
||||
await fs.writeFile(path.join(tmpDir, 'lives-in.csv'), livesInCSV);
|
||||
|
||||
console.log('✓ CSV files written to filesystem');
|
||||
|
||||
// Step 5: Use COPY statements to load data
|
||||
console.log('📥 Loading data using COPY statements...');
|
||||
|
||||
const copyQueries = [
|
||||
`COPY User FROM '${path.join(tmpDir, 'user.csv')}'`,
|
||||
`COPY City FROM '${path.join(tmpDir, 'city.csv')}'`,
|
||||
`COPY Follows FROM '${path.join(tmpDir, 'follows.csv')}'`,
|
||||
`COPY LivesIn FROM '${path.join(tmpDir, 'lives-in.csv')}'`
|
||||
];
|
||||
|
||||
for (const query of copyQueries) {
|
||||
console.log(` Executing: ${query}`);
|
||||
const result = await conn.query(query);
|
||||
console.log(` ✓ Result: ${result.toString()}`);
|
||||
await result.close();
|
||||
}
|
||||
|
||||
console.log('✅ Data loaded successfully using COPY statements\n');
|
||||
|
||||
// Step 6: Verify data was loaded by querying
|
||||
console.log('🔍 Verifying data was loaded correctly...');
|
||||
|
||||
// Query 1: Count nodes
|
||||
console.log('\n📊 Node counts:');
|
||||
let result = await conn.query('MATCH (u:User) RETURN count(u) as userCount');
|
||||
let rows = await result.getAllObjects();
|
||||
console.log(` Users: ${rows[0]?.userCount || 0}`);
|
||||
await result.close();
|
||||
|
||||
result = await conn.query('MATCH (c:City) RETURN count(c) as cityCount');
|
||||
rows = await result.getAllObjects();
|
||||
console.log(` Cities: ${rows[0]?.cityCount || 0}`);
|
||||
await result.close();
|
||||
|
||||
// Query 2: Sample user data
|
||||
console.log('\n👥 Sample user data:');
|
||||
result = await conn.query('MATCH (u:User) RETURN u.name, u.age ORDER BY u.name LIMIT 5');
|
||||
rows = await result.getAllObjects();
|
||||
rows.forEach(row => {
|
||||
console.log(` ${row['u.name']}: ${row['u.age']} years old`);
|
||||
});
|
||||
await result.close();
|
||||
|
||||
// Query 3: Sample city data
|
||||
console.log('\n🏙️ Sample city data:');
|
||||
result = await conn.query('MATCH (c:City) RETURN c.name, c.population ORDER BY c.population DESC LIMIT 5');
|
||||
rows = await result.getAllObjects();
|
||||
rows.forEach(row => {
|
||||
console.log(` ${row['c.name']}: ${row['c.population']} population`);
|
||||
});
|
||||
await result.close();
|
||||
|
||||
// Query 4: Relationship counts
|
||||
console.log('\n🔗 Relationship counts:');
|
||||
result = await conn.query('MATCH ()-[f:Follows]->() RETURN count(f) as followsCount');
|
||||
rows = await result.getAllObjects();
|
||||
console.log(` Follows relationships: ${rows[0]?.followsCount || 0}`);
|
||||
await result.close();
|
||||
|
||||
result = await conn.query('MATCH ()-[l:LivesIn]->() RETURN count(l) as livesInCount');
|
||||
rows = await result.getAllObjects();
|
||||
console.log(` LivesIn relationships: ${rows[0]?.livesInCount || 0}`);
|
||||
await result.close();
|
||||
|
||||
// Query 5: Complex join query
|
||||
console.log('\n🎯 Complex query - Who follows whom:');
|
||||
result = await conn.query(`
|
||||
MATCH (u1:User)-[f:Follows]->(u2:User)
|
||||
RETURN u1.name as follower, u2.name as following, f.since
|
||||
ORDER BY f.since DESC
|
||||
`);
|
||||
rows = await result.getAllObjects();
|
||||
rows.forEach(row => {
|
||||
console.log(` ${row.follower} follows ${row.following} since ${row.since}`);
|
||||
});
|
||||
await result.close();
|
||||
|
||||
// Query 6: Another complex query - Where do people live
|
||||
console.log('\n🏠 Complex query - Where people live:');
|
||||
result = await conn.query(`
|
||||
MATCH (u:User)-[l:LivesIn]->(c:City)
|
||||
RETURN u.name as person, c.name as city, c.population
|
||||
ORDER BY c.population DESC, u.name
|
||||
`);
|
||||
rows = await result.getAllObjects();
|
||||
rows.forEach(row => {
|
||||
console.log(` ${row.person} lives in ${row.city} (pop: ${row.population})`);
|
||||
});
|
||||
await result.close();
|
||||
|
||||
console.log('\n✅ All verification queries completed successfully!');
|
||||
console.log('\n🎉 COPY approach is working perfectly!');
|
||||
|
||||
// Performance comparison note
|
||||
console.log('\n📈 Performance Notes:');
|
||||
console.log(' - COPY statements loaded all data in bulk operations');
|
||||
console.log(' - Much faster than individual INSERT/MERGE statements');
|
||||
console.log(' - Suitable for GitNexus bulk data loading');
|
||||
|
||||
// Cleanup temp files
|
||||
console.log('\n🧹 Cleaning up temporary files...');
|
||||
await fs.rm(tmpDir, { recursive: true, force: true });
|
||||
console.log('✓ Cleanup completed');
|
||||
|
||||
} catch (error) {
|
||||
console.error('\n❌ Test failed:', error);
|
||||
console.error('Error details:', error.message);
|
||||
if (error.stack) {
|
||||
console.error('Stack trace:', error.stack);
|
||||
}
|
||||
process.exit(1);
|
||||
} finally {
|
||||
// Cleanup database connections
|
||||
if (conn) {
|
||||
try {
|
||||
await conn.close();
|
||||
console.log('🔒 Database connection closed');
|
||||
} catch (e) {
|
||||
console.warn('Warning: Failed to close connection:', e.message);
|
||||
}
|
||||
}
|
||||
|
||||
if (db) {
|
||||
try {
|
||||
await db.close();
|
||||
console.log('🔒 Database closed');
|
||||
} catch (e) {
|
||||
console.warn('Warning: Failed to close database:', e.message);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Run the test
|
||||
if (require.main === module) {
|
||||
testKuzuFSCopy()
|
||||
.then(() => {
|
||||
console.log('\n🎊 Test completed successfully!');
|
||||
process.exit(0);
|
||||
})
|
||||
.catch((error) => {
|
||||
console.error('\n💥 Test failed with error:', error);
|
||||
process.exit(1);
|
||||
});
|
||||
}
|
||||
|
||||
module.exports = { testKuzuFSCopy };
|
||||
@@ -1 +0,0 @@
|
||||
|
||||
Generated
+4796
-4379
File diff suppressed because it is too large
Load Diff
+42
-37
@@ -1,58 +1,63 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"private": true,
|
||||
"version": "1.0.0",
|
||||
"version": "0.0.0",
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"dev": "npm run compile-queries && vite",
|
||||
"build": "npm run compile-queries && tsc -b && vite build",
|
||||
"compile-queries": "node scripts/compile-queries.js",
|
||||
"lint": "eslint .",
|
||||
"preview": "vite preview",
|
||||
"test": "jest",
|
||||
"test:watch": "jest --watch",
|
||||
"test:coverage": "jest --coverage",
|
||||
"test:ci": "jest --ci --coverage --watchAll=false"
|
||||
"dev": "vite",
|
||||
"build": "tsc -b && vite build",
|
||||
"preview": "vite preview"
|
||||
},
|
||||
"dependencies": {
|
||||
"@langchain/anthropic": "^0.1.21",
|
||||
"@langchain/core": "^0.3.66",
|
||||
"@langchain/google-genai": "^0.2.16",
|
||||
"@langchain/langgraph": "^0.0.26",
|
||||
"@langchain/openai": "^0.0.28",
|
||||
"@types/d3": "^7.4.3",
|
||||
"@types/jszip": "^3.4.0",
|
||||
"axios": "^1.6.0",
|
||||
"comlink": "^4.4.1",
|
||||
"@huggingface/transformers": "^3.0.0",
|
||||
"@isomorphic-git/lightning-fs": "^4.6.2",
|
||||
"@langchain/anthropic": "^0.3.34",
|
||||
"@langchain/core": "^0.3.0",
|
||||
"@langchain/google-genai": "^0.1.0",
|
||||
"@langchain/langgraph": "^0.2.74",
|
||||
"@langchain/openai": "^0.3.0",
|
||||
"@sigma/edge-curve": "^3.1.0",
|
||||
"@tailwindcss/vite": "^4.1.18",
|
||||
"axios": "^1.13.2",
|
||||
"buffer": "^6.0.3",
|
||||
"comlink": "^4.4.2",
|
||||
"d3": "^7.9.0",
|
||||
"graphology": "^0.26.0",
|
||||
"graphology-layout-force": "^0.2.4",
|
||||
"graphology-layout-forceatlas2": "^0.10.1",
|
||||
"graphology-layout-noverlap": "^0.4.2",
|
||||
"isomorphic-git": "^1.36.1",
|
||||
"jszip": "^3.10.1",
|
||||
"kuzu-wasm": "^0.11.1",
|
||||
"lru-cache": "^11.1.0",
|
||||
"langchain": "^0.3.37",
|
||||
"lru-cache": "^11.2.4",
|
||||
"lucide-react": "^0.562.0",
|
||||
"mermaid": "^11.12.2",
|
||||
"minisearch": "^7.2.0",
|
||||
"react": "^18.3.1",
|
||||
"react-dom": "^18.3.1",
|
||||
"react-markdown": "^10.1.0",
|
||||
"rehype-highlight": "^7.0.2",
|
||||
"react-syntax-highlighter": "^16.1.0",
|
||||
"remark-gfm": "^4.0.1",
|
||||
"uuid": "^11.1.0",
|
||||
"sigma": "^3.0.2",
|
||||
"tailwindcss": "^4.1.18",
|
||||
"uuid": "^13.0.0",
|
||||
"vite-plugin-top-level-await": "^1.6.0",
|
||||
"vite-plugin-wasm": "^3.5.0",
|
||||
"web-tree-sitter": "^0.20.8",
|
||||
"zod": "^3.25.76"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@eslint/js": "^9.11.1",
|
||||
"@types/jest": "^29.5.12",
|
||||
"@types/lru-cache": "^7.10.9",
|
||||
"@types/node": "^20.12.12",
|
||||
"@types/react": "^18.3.10",
|
||||
"@babel/types": "^7.28.5",
|
||||
"@types/jszip": "^3.4.0",
|
||||
"@types/node": "^24.10.1",
|
||||
"@types/react": "^18.3.5",
|
||||
"@types/react-dom": "^18.3.0",
|
||||
"@vitejs/plugin-react": "^4.3.2",
|
||||
"eslint": "^9.11.1",
|
||||
"eslint-plugin-react-hooks": "^5.1.0-rc.0",
|
||||
"eslint-plugin-react-refresh": "^0.4.12",
|
||||
"globals": "^15.9.0",
|
||||
"jest": "^29.7.0",
|
||||
"ts-jest": "^29.1.2",
|
||||
"typescript": "^5.5.3",
|
||||
"typescript-eslint": "^8.7.0",
|
||||
"vite": "^5.4.8"
|
||||
"@types/react-syntax-highlighter": "^15.5.13",
|
||||
"@vercel/node": "^5.5.16",
|
||||
"@vitejs/plugin-react": "^5.1.0",
|
||||
"typescript": "^5.4.5",
|
||||
"vite": "^5.2.0",
|
||||
"vite-plugin-static-copy": "^3.1.4"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,118 +0,0 @@
|
||||
# GitNexus Project Guide
|
||||
|
||||
This document provides a comprehensive technical overview of the GitNexus application, intended to give an LLM full context of the project's purpose, architecture, and implementation details.
|
||||
|
||||
## 1. Project Overview
|
||||
|
||||
GitNexus is a **client-side source code analysis tool** that runs entirely in the browser. It transforms a given codebase (from a public GitHub repository or an uploaded ZIP file) into an interactive **knowledge graph**.
|
||||
|
||||
The primary goal is to allow users to visually explore and understand complex codebases through two main interfaces:
|
||||
1. **A Graph Visualizer**: Displays the code structure as a network of nodes (files, classes, functions) and relationships (imports, calls, inheritance).
|
||||
2. **An AI Chat Interface**: A Retrieval-Augmented Generation (RAG) system that allows users to ask natural language questions about the code, which are answered by querying the knowledge graph.
|
||||
|
||||
Because it's fully client-side, no code is ever sent to a server, ensuring privacy and security.
|
||||
|
||||
## 2. Technology Stack
|
||||
|
||||
The project is built with a modern web technology stack:
|
||||
|
||||
- **Frontend Framework**: **React 18** with **TypeScript**.
|
||||
- **Build Tool**: **Vite** for fast development and optimized builds.
|
||||
- **Code Parsing**: **Tree-sitter** compiled to **WebAssembly (WASM)**. This allows for fast and accurate Abstract Syntax Tree (AST) parsing directly in the browser.
|
||||
- **AI & RAG**: **LangChain.js** is used to orchestrate the AI agent, supporting multiple LLM providers (OpenAI, Anthropic, Google Gemini).
|
||||
- **Concurrency**: **Web Workers** are used to run the entire code ingestion process in a background thread, preventing the UI from freezing. **Comlink** is used for simplifying communication with the worker.
|
||||
- **Graph Visualization**: The UI uses D3.js to render the interactive graph (as seen in dependencies and implementation).
|
||||
- **Styling**: A custom CSS-in-JS solution implemented directly within the `HomePage.tsx` component.
|
||||
|
||||
## 3. Architecture
|
||||
|
||||
The project follows a clean, modular architecture that separates concerns into distinct layers.
|
||||
|
||||
### `src/ui` - The Frontend Layer
|
||||
- **Purpose**: Contains all React components, hooks, and pages.
|
||||
- **Key Files**:
|
||||
- `pages/HomePage.tsx`: The main component that manages the application's state and orchestrates all user interactions. It handles user input, triggers the ingestion process, and displays the results.
|
||||
- `components/graph/GraphExplorer.tsx`: The React component responsible for rendering the interactive knowledge graph.
|
||||
- `components/chat/ChatInterface.tsx`: The component for the AI-powered chat.
|
||||
- `components/ErrorBoundary.tsx`: A crucial component for catching and gracefully handling runtime errors in the UI.
|
||||
|
||||
### `src/core` - The Core Logic Layer
|
||||
- **Purpose**: This is the engine of the application where the knowledge graph is built.
|
||||
- **Key Files**:
|
||||
- `ingestion/pipeline.ts`: Defines the `GraphPipeline`, which executes the multi-pass ingestion process.
|
||||
- `ingestion/structure-processor.ts`: **Pass 1**: Analyzes the file and directory structure.
|
||||
- `ingestion/parsing-processor.ts`: **Pass 2**: Uses Tree-sitter to parse source files into ASTs and extracts definitions (classes, functions, etc.).
|
||||
- `ingestion/import-processor.ts`: **Pass 3**: Resolves import statements between files.
|
||||
- `ingestion/call-processor.ts`: **Pass 4**: Resolves function and method calls between definitions.
|
||||
- `graph/graph.ts`: Defines the `SimpleKnowledgeGraph` class, the core data structure for the graph.
|
||||
- `tree-sitter/parser-loader.ts`: Manages the loading of the various language-specific Tree-sitter WASM parsers.
|
||||
|
||||
### `src/workers` - The Concurrency Layer
|
||||
- **Purpose**: Offloads the intensive ingestion process from the main UI thread.
|
||||
- **Key Files**:
|
||||
- `ingestion.worker.ts`: The entry point for the Web Worker. It receives file data from the UI, runs the `GraphPipeline`, and posts the resulting knowledge graph back to the main thread.
|
||||
|
||||
### `src/services` - The External Services Layer
|
||||
- **Purpose**: Handles communication with external sources.
|
||||
- **Key Files**:
|
||||
- `ingestion.service.ts`: Acts as a bridge between the UI (`HomePage.tsx`) and the `IngestionWorker`.
|
||||
- `github.ts`: Contains logic for fetching repository contents from the GitHub API.
|
||||
- `zip.ts`: Contains logic for reading and extracting files from an uploaded ZIP archive.
|
||||
|
||||
### `src/ai` - The Artificial Intelligence Layer
|
||||
- **Purpose**: Manages the LLM interactions for the Graph RAG chat.
|
||||
- **Key Files**:
|
||||
- `llm-service.ts`: A client that handles communication with the different supported LLM providers (OpenAI, etc.).
|
||||
- `langchain-orchestrator.ts`: Implements the ReAct agent logic using LangChain, defining the tools the agent can use (e.g., querying the graph).
|
||||
- `cypher-generator.ts`: Translates natural language questions into Cypher-like queries to be executed against the knowledge graph.
|
||||
|
||||
## 4. Core Concepts & Execution Flow
|
||||
|
||||
### The Knowledge Graph Data Model
|
||||
The entire application revolves around the `KnowledgeGraph` object defined in `src/core/graph/graph.ts`. It's a simple structure containing two arrays:
|
||||
- `nodes`: Represent code entities like `File`, `Folder`, `Class`, `Function`, etc.
|
||||
- `relationships`: Represent the connections between nodes, such as `CONTAINS`, `IMPORTS`, `CALLS`.
|
||||
|
||||
### The Ingestion Pipeline
|
||||
This is the central process for understanding code. When a user provides a repository, the following happens:
|
||||
1. **File Gathering**: The `IngestionService` fetches all file paths and their text content, either from GitHub or a ZIP file.
|
||||
2. **Worker Invocation**: The file data is passed to the `IngestionWorker`.
|
||||
3. **Pipeline Execution**: The worker runs the `GraphPipeline`, which executes its 4 passes sequentially on the `SimpleKnowledgeGraph` instance:
|
||||
- **Pass 1 (Structure)**: Creates `Project`, `Folder`, and `File` nodes, and links them with `CONTAINS` relationships.
|
||||
- **Pass 2 (Parsing)**: Parses the AST of each source file, creating `Function`, `Class`, and other code-level nodes. It links these nodes to their parent `File` node with `DEFINES` relationships.
|
||||
- **Pass 3 (Imports)**: Analyzes `import` statements and creates `IMPORTS` relationships between code entities.
|
||||
- **Pass 4 (Calls)**: Analyzes the code to find function and method calls, creating `CALLS` relationships between them.
|
||||
4. **Return to UI**: The completed graph is returned to the `HomePage.tsx` component, which updates its state and renders the visual graph.
|
||||
|
||||
This multi-pass approach ensures that the graph is built layer by layer, with each pass adding more detail and context.
|
||||
|
||||
## 5. Potential Issues & Areas for Improvement
|
||||
|
||||
This section details findings from a deep analysis of the codebase, highlighting areas for refactoring and potential bugs.
|
||||
|
||||
### 1. State Management in `HomePage.tsx`
|
||||
- **Issue**: The component uses a single, large `useState` hook to manage the entire application state (`AppState`). Any small update, such as user input in a text field, triggers a re-render of the entire component and all its children.
|
||||
- **Impact**: This can lead to a sluggish UI and poor performance, especially as the application grows in complexity.
|
||||
- **Recommendation**: Refactor the state management. Use more granular `useState` hooks for simple, independent state. For state that needs to be shared across many components, consider using React's Context API or a lightweight state management library to prevent unnecessary re-renders.
|
||||
|
||||
### 2. Redundant `generateId` Function
|
||||
- **Issue**: The file `src/core/ingestion/parsing-processor.ts` contains a local `generateId` function that uses a simple (and potentially collision-prone) hashing algorithm. A more robust, UUID-based `generateId` function already exists in `src/lib/utils.ts`.
|
||||
- **Impact**: Code duplication and the risk of using an inferior ID generation method, which could lead to node ID collisions in the graph.
|
||||
- **Recommendation**: Remove the local `generateId` function from `parsing-processor.ts` and update the file to import and use the centralized version from `src/lib/utils.ts`.
|
||||
|
||||
### 3. Inefficient Graph Integrity Checks
|
||||
- **Issue**: The `validateGraphIntegrity` method in `src/core/ingestion/pipeline.ts` performs several distinct traversals (`filter`, `some`, `find`) over the entire graph to find issues like orphaned nodes or files without definitions.
|
||||
- **Impact**: On large codebases, this can significantly slow down the final phase of the ingestion process.
|
||||
- **Recommendation**: Optimize these validation checks. Many of them can be combined into a single pass over the graph's nodes and relationships. Using a `Map` or `Set` for lookups within the pass would be much more performant than repeated array iterations.
|
||||
|
||||
### 4. Hardcoded Filtering Logic
|
||||
- **Issue**: The `CallProcessor` in `src/core/ingestion/call-processor.ts` defines its own large, hardcoded `Set` of Python built-in functions to ignore during call resolution. This logic is disconnected from the centralized language configurations.
|
||||
- **Impact**: This makes the list difficult to maintain and extend. It also violates the principle of single-source-of-truth, as language-specific knowledge should be centralized.
|
||||
- **Recommendation**: Refactor the `CallProcessor` to use the `builtinFunctions` set from the `language-config.ts` file. This centralizes language-specific data and makes the processor more modular.
|
||||
|
||||
### 5. Mock/Incomplete Query Engine
|
||||
- **Issue**: The `GraphQueryEngine` in `src/core/graph/query-engine.ts` is a mock implementation that uses regular expressions to parse Cypher-like queries. This approach is not robust and only supports a very limited subset of valid queries.
|
||||
- **Impact**: This is a critical limitation for the AI chat feature. The `CypherGenerator` can produce complex queries that the engine cannot execute, leading to failed tool calls and inaccurate answers from the AI.
|
||||
- **Recommendation**: This is a major area for future development. The regex-based parser should be replaced with a more robust solution. Options include:
|
||||
- Implementing a proper parser for a small, well-defined subset of Cypher.
|
||||
- Integrating a lightweight, in-browser graph database library that supports Cypher queries.
|
||||
Binary file not shown.
File diff suppressed because one or more lines are too long
@@ -1 +0,0 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" class="iconify iconify--logos" width="31.88" height="32" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 257"><defs><linearGradient id="IconifyId1813088fe1fbc01fb466" x1="-.828%" x2="57.636%" y1="7.652%" y2="78.411%"><stop offset="0%" stop-color="#41D1FF"></stop><stop offset="100%" stop-color="#BD34FE"></stop></linearGradient><linearGradient id="IconifyId1813088fe1fbc01fb467" x1="43.376%" x2="50.316%" y1="2.242%" y2="89.03%"><stop offset="0%" stop-color="#FFEA83"></stop><stop offset="8.333%" stop-color="#FFDD35"></stop><stop offset="100%" stop-color="#FFA800"></stop></linearGradient></defs><path fill="url(#IconifyId1813088fe1fbc01fb466)" d="M255.153 37.938L134.897 252.976c-2.483 4.44-8.862 4.466-11.382.048L.875 37.958c-2.746-4.814 1.371-10.646 6.827-9.67l120.385 21.517a6.537 6.537 0 0 0 2.322-.004l117.867-21.483c5.438-.991 9.574 4.796 6.877 9.62Z"></path><path fill="url(#IconifyId1813088fe1fbc01fb467)" d="M185.432.063L96.44 17.501a3.268 3.268 0 0 0-2.634 3.014l-5.474 92.456a3.268 3.268 0 0 0 3.997 3.378l24.777-5.718c2.318-.535 4.413 1.507 3.936 3.838l-7.361 36.047c-.495 2.426 1.782 4.5 4.151 3.78l15.304-4.649c2.372-.72 4.652 1.36 4.15 3.788l-11.698 56.621c-.732 3.542 3.979 5.473 5.943 2.437l1.313-2.028l72.516-144.72c1.215-2.423-.88-5.186-3.54-4.672l-25.505 4.922c-2.396.462-4.435-1.77-3.759-4.114l16.646-57.705c.677-2.35-1.37-4.583-3.769-4.113Z"></path></svg>
|
||||
|
Before Width: | Height: | Size: 1.5 KiB |
Binary file not shown.
@@ -1,394 +0,0 @@
|
||||
/**
|
||||
* Tree-sitter Web Worker
|
||||
* Handles parallel parsing of source code files
|
||||
*/
|
||||
|
||||
// Import tree-sitter and compiled queries using importScripts for classic workers
|
||||
importScripts('/workers/tree-sitter.js');
|
||||
importScripts('/workers/compiled-queries.js');
|
||||
|
||||
// Function to get queries for a specific language
|
||||
function getQueriesForLanguage(language) {
|
||||
return queries[language] || null;
|
||||
}
|
||||
|
||||
// Initialize tree-sitter
|
||||
let parser = null;
|
||||
let languageParsers = new Map();
|
||||
|
||||
// Initialize the worker
|
||||
async function initializeWorker() {
|
||||
try {
|
||||
// Initialize tree-sitter (TreeSitter is available as a global from importScripts)
|
||||
await TreeSitter.init();
|
||||
parser = new TreeSitter();
|
||||
|
||||
// Load language parsers
|
||||
const languageLoaders = {
|
||||
typescript: async () => {
|
||||
const language = await TreeSitter.Language.load('/wasm/typescript/tree-sitter-typescript.wasm');
|
||||
return language;
|
||||
},
|
||||
javascript: async () => {
|
||||
const language = await TreeSitter.Language.load('/wasm/javascript/tree-sitter-javascript.wasm');
|
||||
return language;
|
||||
},
|
||||
python: async () => {
|
||||
const language = await TreeSitter.Language.load('/wasm/python/tree-sitter-python.wasm');
|
||||
return language;
|
||||
}
|
||||
};
|
||||
|
||||
for (const [lang, loader] of Object.entries(languageLoaders)) {
|
||||
try {
|
||||
const languageParser = await loader();
|
||||
languageParsers.set(lang, languageParser);
|
||||
console.log(`Worker: ${lang} parser loaded successfully`);
|
||||
} catch (error) {
|
||||
console.error(`Worker: Failed to load ${lang} parser:`, error);
|
||||
}
|
||||
}
|
||||
|
||||
console.log('Worker: Tree-sitter worker initialized successfully');
|
||||
return true;
|
||||
} catch (error) {
|
||||
console.error('Worker: Failed to initialize tree-sitter worker:', error);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// Detect language from file path
|
||||
function detectLanguage(filePath) {
|
||||
const ext = filePath.split('.').pop()?.toLowerCase();
|
||||
|
||||
switch (ext) {
|
||||
case 'ts':
|
||||
case 'tsx':
|
||||
return 'typescript';
|
||||
case 'js':
|
||||
case 'jsx':
|
||||
return 'javascript';
|
||||
case 'py':
|
||||
return 'python';
|
||||
default:
|
||||
return 'javascript'; // Default fallback
|
||||
}
|
||||
}
|
||||
|
||||
// Extract definitions from AST
|
||||
function extractDefinitions(tree, filePath) {
|
||||
const definitions = [];
|
||||
const language = detectLanguage(filePath);
|
||||
|
||||
// Get queries for the language
|
||||
const languageQueries = getQueriesForLanguage(language);
|
||||
if (!languageQueries) return definitions;
|
||||
|
||||
// Execute queries to find definitions
|
||||
for (const [queryName, queryString] of Object.entries(languageQueries)) {
|
||||
try {
|
||||
const query = parser.getLanguage().query(queryString);
|
||||
const matches = query.matches(tree.rootNode);
|
||||
|
||||
for (const match of matches) {
|
||||
const definition = processMatch(match, filePath, queryName);
|
||||
if (definition) {
|
||||
definitions.push(definition);
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
console.warn(`Worker: Error executing query ${queryName}:`, error);
|
||||
}
|
||||
}
|
||||
|
||||
return definitions;
|
||||
}
|
||||
|
||||
// Get queries for specific language - now uses imported compiled queries
|
||||
// This ensures consistency with the main thread parsing logic
|
||||
|
||||
// Helper function to map query names to definition types (EXACTLY matches main thread)
|
||||
function getDefinitionType(queryName) {
|
||||
switch (queryName) {
|
||||
case 'classes':
|
||||
case 'exportClasses': return 'class';
|
||||
case 'methods':
|
||||
case 'properties':
|
||||
case 'staticmethods':
|
||||
case 'classmethods': return 'method';
|
||||
case 'functions':
|
||||
case 'arrowFunctions':
|
||||
case 'reactComponents':
|
||||
case 'reactConstComponents':
|
||||
case 'defaultExportArrows':
|
||||
case 'variableAssignments':
|
||||
case 'objectMethods':
|
||||
case 'exportFunctions':
|
||||
case 'defaultExportFunctions':
|
||||
case 'functionExpressions': return 'function';
|
||||
case 'variables':
|
||||
case 'constDeclarations':
|
||||
case 'hookCalls':
|
||||
case 'hookDestructuring':
|
||||
case 'global_variables': return 'variable';
|
||||
case 'imports':
|
||||
case 'from_imports': return 'import';
|
||||
case 'exports':
|
||||
case 'defaultExports':
|
||||
case 'moduleExports': return 'function'; // Exports usually export functions
|
||||
case 'interfaces': return 'interface';
|
||||
case 'types': return 'type';
|
||||
case 'decorators': return 'decorator';
|
||||
case 'enums': return 'enum';
|
||||
default: return 'variable';
|
||||
}
|
||||
}
|
||||
|
||||
// Process a query match into a definition
|
||||
// EXACTLY matches main thread's extractDefinition logic
|
||||
function processMatch(match, filePath, queryName) {
|
||||
try {
|
||||
const definitions = [];
|
||||
|
||||
for (const capture of match.captures) {
|
||||
const node = capture.node;
|
||||
|
||||
// Extract name using EXACT same logic as single-threaded
|
||||
let nameNode = node.childForFieldName('name');
|
||||
let name = nameNode ? nameNode.text : null;
|
||||
|
||||
// Handle different naming patterns for different query types (EXACT match to single-threaded)
|
||||
if (!name) {
|
||||
// Try alternative naming strategies based on query type
|
||||
switch (queryName) {
|
||||
case 'variables':
|
||||
case 'constDeclarations':
|
||||
case 'global_variables':
|
||||
// For variable assignments, look for identifier in left side
|
||||
const leftChild = node.namedChildren.find(child => child.type === 'identifier');
|
||||
if (leftChild) name = leftChild.text;
|
||||
break;
|
||||
|
||||
case 'hookCalls':
|
||||
case 'hookDestructuring':
|
||||
// For React hooks, try to get the variable name
|
||||
const hookVar = node.namedChildren.find(child => child.type === 'variable_declarator');
|
||||
if (hookVar) {
|
||||
const hookName = hookVar.childForFieldName('name');
|
||||
if (hookName) {
|
||||
// Handle array destructuring for useState pattern
|
||||
if (hookName.type === 'array_pattern') {
|
||||
const elements = hookName.namedChildren.filter(child => child.type === 'identifier');
|
||||
if (elements.length > 0) {
|
||||
name = elements.map(el => el.text).join(', ');
|
||||
}
|
||||
} else {
|
||||
name = hookName.text;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
case 'reactComponents':
|
||||
case 'reactConstComponents':
|
||||
case 'defaultExportArrows':
|
||||
// For React components, get the component name
|
||||
const componentVar = node.namedChildren.find(child => child.type === 'variable_declarator');
|
||||
if (componentVar) {
|
||||
const componentName = componentVar.childForFieldName('name');
|
||||
if (componentName) name = componentName.text;
|
||||
}
|
||||
break;
|
||||
|
||||
case 'moduleExports':
|
||||
// For module.exports = something, get the property name
|
||||
const memberExpr = node.namedChildren.find(child => child.type === 'member_expression');
|
||||
if (memberExpr) {
|
||||
const property = memberExpr.childForFieldName('property');
|
||||
if (property) name = property.text;
|
||||
}
|
||||
break;
|
||||
|
||||
case 'decorators':
|
||||
// For decorators, get the decorator name
|
||||
const decoratorChild = node.namedChildren.find(child => child.type === 'identifier');
|
||||
if (decoratorChild) name = decoratorChild.text;
|
||||
break;
|
||||
|
||||
default:
|
||||
// Try to find any identifier child
|
||||
const identifierChild = node.namedChildren.find(child => child.type === 'identifier');
|
||||
if (identifierChild) name = identifierChild.text;
|
||||
}
|
||||
}
|
||||
|
||||
// Skip anonymous definitions - they're usually from compiled/minified code (EXACT match to single-threaded)
|
||||
if (!name || name === 'anonymous' || name.trim().length === 0) {
|
||||
continue; // Skip this definition
|
||||
}
|
||||
|
||||
// Skip very short names that are likely noise (but keep single-letter variables like 'i', 'x')
|
||||
if (name.length === 1 && queryName !== 'variables' && queryName !== 'constDeclarations') {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Skip common noise patterns
|
||||
const noisePatterns = ['_', '__', '___', 'temp', 'tmp'];
|
||||
if (noisePatterns.includes(name.toLowerCase())) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const definition = {
|
||||
name,
|
||||
type: getDefinitionType(queryName),
|
||||
startLine: node.startPosition.row + 1,
|
||||
endLine: node.endPosition.row + 1,
|
||||
};
|
||||
|
||||
// Extract additional metadata based on definition type (EXACT match to single-threaded)
|
||||
if (definition.type === 'function' || definition.type === 'method') {
|
||||
// Try to extract parameters
|
||||
const parametersNode = node.childForFieldName('parameters');
|
||||
if (parametersNode) {
|
||||
const params = [];
|
||||
for (const param of parametersNode.namedChildren) {
|
||||
if (param.type === 'identifier' || param.type === 'formal_parameter') {
|
||||
params.push(param.text);
|
||||
}
|
||||
}
|
||||
if (params.length > 0) {
|
||||
definition.parameters = params;
|
||||
}
|
||||
}
|
||||
|
||||
// Mark React components
|
||||
if (queryName === 'reactComponents' || queryName === 'reactConstComponents') {
|
||||
definition.isAsync = false; // React components are not async by default
|
||||
definition.exportType = 'default'; // Most React components are default exports
|
||||
}
|
||||
}
|
||||
|
||||
if (definition.type === 'class') {
|
||||
// Try to extract inheritance information
|
||||
const superclassNode = node.childForFieldName('superclass');
|
||||
if (superclassNode) {
|
||||
definition.extends = [superclassNode.text];
|
||||
}
|
||||
}
|
||||
|
||||
// Handle variable types with additional context
|
||||
if (definition.type === 'variable') {
|
||||
if (queryName === 'hookCalls' || queryName === 'hookDestructuring') {
|
||||
definition.exportType = 'named'; // React hooks are typically named exports
|
||||
|
||||
// Try to extract hook type from call expression
|
||||
const callExpr = node.descendantsOfType('call_expression')[0];
|
||||
if (callExpr) {
|
||||
const funcNode = callExpr.childForFieldName('function');
|
||||
if (funcNode && funcNode.type === 'identifier') {
|
||||
definition.returnType = funcNode.text; // Store hook function name
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
definitions.push(definition);
|
||||
}
|
||||
|
||||
// Return the first definition (main thread processes one capture at a time)
|
||||
return definitions.length > 0 ? definitions[0] : null;
|
||||
} catch (error) {
|
||||
console.warn('Worker: Error processing match:', error);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
// Parse a single file
|
||||
async function parseFile(filePath, content) {
|
||||
try {
|
||||
const language = detectLanguage(filePath);
|
||||
const languageParser = languageParsers.get(language);
|
||||
|
||||
if (!languageParser) {
|
||||
throw new Error(`No parser available for language: ${language}`);
|
||||
}
|
||||
|
||||
// Set the language
|
||||
parser.setLanguage(languageParser);
|
||||
|
||||
// Parse the content
|
||||
const tree = parser.parse(content);
|
||||
|
||||
// Extract definitions
|
||||
const definitions = extractDefinitions(tree, filePath);
|
||||
|
||||
return {
|
||||
filePath,
|
||||
definitions,
|
||||
ast: {
|
||||
tree: {
|
||||
rootNode: {
|
||||
startPosition: tree.rootNode.startPosition,
|
||||
endPosition: tree.rootNode.endPosition,
|
||||
type: tree.rootNode.type,
|
||||
text: tree.rootNode.text
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
} catch (error) {
|
||||
console.error(`Worker: Error parsing file ${filePath}:`, error);
|
||||
|
||||
// Return a result with error information instead of throwing
|
||||
return {
|
||||
filePath,
|
||||
definitions: [],
|
||||
error: error.message || 'Unknown parsing error',
|
||||
ast: null
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Handle messages from main thread
|
||||
self.onmessage = async function(event) {
|
||||
const { taskId, input } = event.data;
|
||||
|
||||
try {
|
||||
// Initialize worker if not already done
|
||||
if (!parser) {
|
||||
const initialized = await initializeWorker();
|
||||
if (!initialized) {
|
||||
throw new Error('Failed to initialize tree-sitter worker');
|
||||
}
|
||||
}
|
||||
|
||||
const { filePath, content } = input;
|
||||
|
||||
// Parse the file
|
||||
const result = await parseFile(filePath, content);
|
||||
|
||||
// Send result back to main thread
|
||||
self.postMessage({
|
||||
taskId,
|
||||
result
|
||||
});
|
||||
|
||||
} catch (error) {
|
||||
// Send error back to main thread
|
||||
self.postMessage({
|
||||
taskId,
|
||||
error: error.message || 'Unknown error in tree-sitter worker'
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
// Handle worker errors
|
||||
self.onerror = function(error) {
|
||||
console.error('Worker: Unhandled error:', error);
|
||||
self.postMessage({
|
||||
taskId: 'error',
|
||||
error: error.message || 'Unhandled worker error'
|
||||
});
|
||||
};
|
||||
|
||||
console.log('Worker: Tree-sitter worker script loaded');
|
||||
File diff suppressed because one or more lines are too long
Binary file not shown.
@@ -1,98 +0,0 @@
|
||||
#!/usr/bin/env node
|
||||
|
||||
/**
|
||||
* Build-time script to compile TypeScript queries into JavaScript for Web Workers
|
||||
* This allows workers to import the same query definitions as the main thread
|
||||
*/
|
||||
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import { fileURLToPath } from 'url';
|
||||
|
||||
const __filename = fileURLToPath(import.meta.url);
|
||||
const __dirname = path.dirname(__filename);
|
||||
|
||||
// Path to the TypeScript queries file
|
||||
const queriesPath = path.join(__dirname, '../src/core/ingestion/tree-sitter-queries.ts');
|
||||
// Output path for compiled JavaScript queries
|
||||
const outputPath = path.join(__dirname, '../public/workers/compiled-queries.js');
|
||||
|
||||
function compileQueries() {
|
||||
try {
|
||||
console.log('🔨 Compiling Tree-sitter queries for Web Workers...');
|
||||
|
||||
// Read the TypeScript queries file
|
||||
const queriesContent = fs.readFileSync(queriesPath, 'utf8');
|
||||
|
||||
// Extract the query objects using simple regex (since they're just object literals)
|
||||
const typescriptMatch = queriesContent.match(/export const TYPESCRIPT_QUERIES = ({[\s\S]*?});/);
|
||||
const javascriptMatch = queriesContent.match(/export const JAVASCRIPT_QUERIES = ({[\s\S]*?});/);
|
||||
const pythonMatch = queriesContent.match(/export const PYTHON_QUERIES = ({[\s\S]*?});/);
|
||||
const javaMatch = queriesContent.match(/export const JAVA_QUERIES = ({[\s\S]*?});/);
|
||||
|
||||
if (!typescriptMatch || !javascriptMatch || !pythonMatch || !javaMatch) {
|
||||
throw new Error('Could not extract queries from TypeScript file');
|
||||
}
|
||||
|
||||
// Create JavaScript content for Web Workers (no exports, global variables only)
|
||||
const jsContent = `/**
|
||||
* AUTO-GENERATED FILE - DO NOT EDIT MANUALLY
|
||||
* Generated from src/core/ingestion/tree-sitter-queries.ts
|
||||
* Run 'npm run compile-queries' to regenerate
|
||||
*
|
||||
* IMPORTANT: This file is loaded via importScripts() in a classic Web Worker.
|
||||
* DO NOT use ES6 export statements! Use global variables (const) instead.
|
||||
* importScripts() cannot load files with 'export' statements.
|
||||
*/
|
||||
|
||||
const TYPESCRIPT_QUERIES = ${typescriptMatch[1]};
|
||||
|
||||
const JAVASCRIPT_QUERIES = ${javascriptMatch[1]};
|
||||
|
||||
const PYTHON_QUERIES = ${pythonMatch[1]};
|
||||
|
||||
const JAVA_QUERIES = ${javaMatch[1]};
|
||||
|
||||
// Helper function to get queries for a specific language
|
||||
function getQueriesForLanguage(language) {
|
||||
switch (language) {
|
||||
case 'typescript':
|
||||
return TYPESCRIPT_QUERIES;
|
||||
case 'javascript':
|
||||
return JAVASCRIPT_QUERIES;
|
||||
case 'python':
|
||||
return PYTHON_QUERIES;
|
||||
case 'java':
|
||||
return JAVA_QUERIES;
|
||||
default:
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
// Export individual query sets for backward compatibility
|
||||
const queries = {
|
||||
typescript: TYPESCRIPT_QUERIES,
|
||||
javascript: JAVASCRIPT_QUERIES,
|
||||
python: PYTHON_QUERIES,
|
||||
java: JAVA_QUERIES
|
||||
};
|
||||
`;
|
||||
|
||||
// Ensure output directory exists
|
||||
const outputDir = path.dirname(outputPath);
|
||||
if (!fs.existsSync(outputDir)) {
|
||||
fs.mkdirSync(outputDir, { recursive: true });
|
||||
}
|
||||
|
||||
// Write the compiled JavaScript file
|
||||
fs.writeFileSync(outputPath, jsContent, 'utf8');
|
||||
|
||||
console.log(`✅ Queries compiled successfully to: ${outputPath}`);
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Failed to compile queries:', error.message);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
compileQueries();
|
||||
@@ -1,135 +0,0 @@
|
||||
#!/usr/bin/env tsx
|
||||
/**
|
||||
* Configuration Migration Script
|
||||
*
|
||||
* Helps migrate from .env variables to the new gitnexus.config.ts system.
|
||||
* This script analyzes your current .env file and suggests config updates.
|
||||
*/
|
||||
|
||||
import { readFileSync, existsSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
|
||||
interface EnvMapping {
|
||||
envKey: string;
|
||||
configPath: string;
|
||||
transform?: (value: string) => any;
|
||||
}
|
||||
|
||||
const ENV_MAPPINGS: EnvMapping[] = [
|
||||
// Engine Configuration
|
||||
{ envKey: 'ENGINE_DEFAULT', configPath: 'engine.default' },
|
||||
{ envKey: 'ENGINE_LEGACY_ENABLED', configPath: 'engine.legacy.enabled', transform: (v) => v === 'true' },
|
||||
{ envKey: 'ENGINE_LEGACY_MEMORY_LIMIT_MB', configPath: 'engine.legacy.memoryLimitMB', transform: parseInt },
|
||||
{ envKey: 'ENGINE_LEGACY_BATCH_SIZE', configPath: 'engine.legacy.batchSize', transform: parseInt },
|
||||
|
||||
// Processing Configuration
|
||||
{ envKey: 'VITE_PARSING_MODE', configPath: 'processing.mode' },
|
||||
{ envKey: 'PARALLEL_MAX_WORKERS', configPath: 'processing.parallel.maxWorkers', transform: parseInt },
|
||||
{ envKey: 'PARALLEL_BATCH_SIZE', configPath: 'processing.parallel.batchSize', transform: parseInt },
|
||||
{ envKey: 'MEMORY_MAX_MB', configPath: 'processing.memory.maxMB', transform: parseInt },
|
||||
|
||||
// KuzuDB Configuration
|
||||
{ envKey: 'VITE_KUZU_ENABLED', configPath: 'kuzu.enabled', transform: (v) => v === 'true' },
|
||||
|
||||
// Logging Configuration
|
||||
{ envKey: 'LOG_LEVEL', configPath: 'logging.level' },
|
||||
{ envKey: 'LOG_ENABLE_METRICS', configPath: 'logging.enableMetrics', transform: (v) => v === 'true' },
|
||||
|
||||
// GitHub Configuration
|
||||
{ envKey: 'GITHUB_TOKEN', configPath: 'github.token' },
|
||||
{ envKey: 'GITHUB_API_URL', configPath: 'github.apiUrl' }
|
||||
];
|
||||
|
||||
function parseEnvFile(filePath: string): Record<string, string> {
|
||||
if (!existsSync(filePath)) {
|
||||
console.log(`❌ .env file not found at ${filePath}`);
|
||||
return {};
|
||||
}
|
||||
|
||||
const content = readFileSync(filePath, 'utf-8');
|
||||
const env: Record<string, string> = {};
|
||||
|
||||
content.split('\n').forEach(line => {
|
||||
line = line.trim();
|
||||
if (line && !line.startsWith('#')) {
|
||||
const [key, ...valueParts] = line.split('=');
|
||||
if (key && valueParts.length > 0) {
|
||||
env[key.trim()] = valueParts.join('=').trim();
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
return env;
|
||||
}
|
||||
|
||||
function generateConfigSuggestions(env: Record<string, string>): string {
|
||||
const suggestions: string[] = [];
|
||||
const foundMappings: EnvMapping[] = [];
|
||||
|
||||
// Find matching environment variables
|
||||
ENV_MAPPINGS.forEach(mapping => {
|
||||
if (env[mapping.envKey]) {
|
||||
foundMappings.push(mapping);
|
||||
}
|
||||
});
|
||||
|
||||
if (foundMappings.length === 0) {
|
||||
return 'No matching environment variables found for migration.';
|
||||
}
|
||||
|
||||
suggestions.push('// Suggested updates for gitnexus.config.ts based on your .env file:\n');
|
||||
|
||||
foundMappings.forEach(mapping => {
|
||||
const envValue = env[mapping.envKey];
|
||||
const transformedValue = mapping.transform ? mapping.transform(envValue) : `'${envValue}'`;
|
||||
|
||||
suggestions.push(`// ${mapping.envKey}=${envValue}`);
|
||||
suggestions.push(`${mapping.configPath}: ${transformedValue},\n`);
|
||||
});
|
||||
|
||||
return suggestions.join('\n');
|
||||
}
|
||||
|
||||
function analyzeCurrentConfig(): void {
|
||||
console.log('🔍 GitNexus Configuration Migration Analysis\n');
|
||||
|
||||
// Check for .env file
|
||||
const envPath = join(process.cwd(), '.env');
|
||||
const env = parseEnvFile(envPath);
|
||||
|
||||
if (Object.keys(env).length === 0) {
|
||||
console.log('No .env file found or it\'s empty.');
|
||||
return;
|
||||
}
|
||||
|
||||
console.log(`📋 Found ${Object.keys(env).length} environment variables in .env\n`);
|
||||
|
||||
// Generate suggestions
|
||||
const suggestions = generateConfigSuggestions(env);
|
||||
console.log('💡 Configuration Suggestions:');
|
||||
console.log('=' .repeat(50));
|
||||
console.log(suggestions);
|
||||
console.log('=' .repeat(50));
|
||||
|
||||
// Check if gitnexus.config.ts exists
|
||||
const configPath = join(process.cwd(), 'gitnexus.config.ts');
|
||||
if (existsSync(configPath)) {
|
||||
console.log('\n✅ gitnexus.config.ts already exists');
|
||||
console.log('You can update it with the suggestions above.');
|
||||
} else {
|
||||
console.log('\n⚠️ gitnexus.config.ts not found');
|
||||
console.log('Please create it first, then apply the suggestions above.');
|
||||
}
|
||||
|
||||
// Show next steps
|
||||
console.log('\n📝 Next Steps:');
|
||||
console.log('1. Update gitnexus.config.ts with the suggested values');
|
||||
console.log('2. Test your application to ensure everything works');
|
||||
console.log('3. Consider removing the corresponding .env variables');
|
||||
console.log('4. Update your deployment scripts to use the new config system');
|
||||
}
|
||||
|
||||
// Run the analysis
|
||||
if (require.main === module) {
|
||||
analyzeCurrentConfig();
|
||||
}
|
||||
-42
@@ -1,42 +0,0 @@
|
||||
#root {
|
||||
max-width: 1280px;
|
||||
margin: 0 auto;
|
||||
padding: 2rem;
|
||||
text-align: center;
|
||||
}
|
||||
|
||||
.logo {
|
||||
height: 6em;
|
||||
padding: 1.5em;
|
||||
will-change: filter;
|
||||
transition: filter 300ms;
|
||||
}
|
||||
.logo:hover {
|
||||
filter: drop-shadow(0 0 2em #646cffaa);
|
||||
}
|
||||
.logo.react:hover {
|
||||
filter: drop-shadow(0 0 2em #61dafbaa);
|
||||
}
|
||||
|
||||
@keyframes logo-spin {
|
||||
from {
|
||||
transform: rotate(0deg);
|
||||
}
|
||||
to {
|
||||
transform: rotate(360deg);
|
||||
}
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: no-preference) {
|
||||
a:nth-of-type(2) .logo {
|
||||
animation: logo-spin infinite 20s linear;
|
||||
}
|
||||
}
|
||||
|
||||
.card {
|
||||
padding: 2em;
|
||||
}
|
||||
|
||||
.read-the-docs {
|
||||
color: #888;
|
||||
}
|
||||
+192
-6
@@ -1,12 +1,198 @@
|
||||
import React from 'react';
|
||||
import { HomePage, ErrorBoundary } from './ui/index.ts';
|
||||
import { useCallback, useRef } from 'react';
|
||||
import { AppStateProvider, useAppState } from './hooks/useAppState';
|
||||
import { DropZone } from './components/DropZone';
|
||||
import { LoadingOverlay } from './components/LoadingOverlay';
|
||||
import { Header } from './components/Header';
|
||||
import { GraphCanvas, GraphCanvasHandle } from './components/GraphCanvas';
|
||||
import { RightPanel } from './components/RightPanel';
|
||||
import { SettingsPanel } from './components/SettingsPanel';
|
||||
import { StatusBar } from './components/StatusBar';
|
||||
import { FileTreePanel } from './components/FileTreePanel';
|
||||
import { CodeReferencesPanel } from './components/CodeReferencesPanel';
|
||||
import { FileEntry } from './services/zip';
|
||||
import { getActiveProviderConfig } from './core/llm/settings-service';
|
||||
|
||||
const App: React.FC = () => {
|
||||
const AppContent = () => {
|
||||
const {
|
||||
viewMode,
|
||||
setViewMode,
|
||||
setGraph,
|
||||
setFileContents,
|
||||
setProgress,
|
||||
setProjectName,
|
||||
progress,
|
||||
isRightPanelOpen,
|
||||
runPipeline,
|
||||
runPipelineFromFiles,
|
||||
isSettingsPanelOpen,
|
||||
setSettingsPanelOpen,
|
||||
refreshLLMSettings,
|
||||
initializeAgent,
|
||||
startEmbeddings,
|
||||
embeddingStatus,
|
||||
codeReferences,
|
||||
selectedNode,
|
||||
isCodePanelOpen,
|
||||
} = useAppState();
|
||||
|
||||
const graphCanvasRef = useRef<GraphCanvasHandle>(null);
|
||||
|
||||
const handleFileSelect = useCallback(async (file: File) => {
|
||||
const projectName = file.name.replace('.zip', '');
|
||||
setProjectName(projectName);
|
||||
setViewMode('loading');
|
||||
|
||||
try {
|
||||
const result = await runPipeline(file, (progress) => {
|
||||
setProgress(progress);
|
||||
});
|
||||
|
||||
setGraph(result.graph);
|
||||
setFileContents(result.fileContents);
|
||||
setViewMode('exploring');
|
||||
|
||||
// Initialize (or re-initialize) the agent AFTER a repo loads so it captures
|
||||
// the current codebase context (file contents + graph tools) in the worker.
|
||||
if (getActiveProviderConfig()) {
|
||||
initializeAgent();
|
||||
}
|
||||
|
||||
// Auto-start embeddings pipeline in background
|
||||
// Uses WebGPU if available, falls back to WASM
|
||||
startEmbeddings().catch((err) => {
|
||||
// WebGPU not available - try WASM fallback silently
|
||||
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
|
||||
startEmbeddings('wasm').catch(console.warn);
|
||||
} else {
|
||||
console.warn('Embeddings auto-start failed:', err);
|
||||
}
|
||||
});
|
||||
} catch (error) {
|
||||
console.error('Pipeline error:', error);
|
||||
setProgress({
|
||||
phase: 'error',
|
||||
percent: 0,
|
||||
message: 'Error processing file',
|
||||
detail: error instanceof Error ? error.message : 'Unknown error',
|
||||
});
|
||||
setTimeout(() => {
|
||||
setViewMode('onboarding');
|
||||
setProgress(null);
|
||||
}, 3000);
|
||||
}
|
||||
}, [setViewMode, setGraph, setFileContents, setProgress, setProjectName, runPipeline, startEmbeddings, initializeAgent]);
|
||||
|
||||
const handleGitClone = useCallback(async (files: FileEntry[]) => {
|
||||
// Extract project name from first file path (e.g., "owner-repo-123/src/..." -> "owner-repo")
|
||||
const firstPath = files[0]?.path || 'repository';
|
||||
const projectName = firstPath.split('/')[0].replace(/-\d+$/, '') || 'repository';
|
||||
|
||||
setProjectName(projectName);
|
||||
setViewMode('loading');
|
||||
|
||||
try {
|
||||
const result = await runPipelineFromFiles(files, (progress) => {
|
||||
setProgress(progress);
|
||||
});
|
||||
|
||||
setGraph(result.graph);
|
||||
setFileContents(result.fileContents);
|
||||
setViewMode('exploring');
|
||||
|
||||
// Initialize (or re-initialize) the agent AFTER a repo loads so it captures
|
||||
// the current codebase context (file contents + graph tools) in the worker.
|
||||
if (getActiveProviderConfig()) {
|
||||
initializeAgent();
|
||||
}
|
||||
|
||||
// Auto-start embeddings pipeline in background
|
||||
// Uses WebGPU if available, falls back to WASM
|
||||
startEmbeddings().catch((err) => {
|
||||
// WebGPU not available - try WASM fallback silently
|
||||
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
|
||||
startEmbeddings('wasm').catch(console.warn);
|
||||
} else {
|
||||
console.warn('Embeddings auto-start failed:', err);
|
||||
}
|
||||
});
|
||||
} catch (error) {
|
||||
console.error('Pipeline error:', error);
|
||||
setProgress({
|
||||
phase: 'error',
|
||||
percent: 0,
|
||||
message: 'Error processing repository',
|
||||
detail: error instanceof Error ? error.message : 'Unknown error',
|
||||
});
|
||||
setTimeout(() => {
|
||||
setViewMode('onboarding');
|
||||
setProgress(null);
|
||||
}, 3000);
|
||||
}
|
||||
}, [setViewMode, setGraph, setFileContents, setProgress, setProjectName, runPipelineFromFiles, startEmbeddings, initializeAgent]);
|
||||
|
||||
const handleFocusNode = useCallback((nodeId: string) => {
|
||||
graphCanvasRef.current?.focusNode(nodeId);
|
||||
}, []);
|
||||
|
||||
// Handle settings saved - refresh and reinitialize agent
|
||||
// NOTE: Must be defined BEFORE any conditional returns (React hooks rule)
|
||||
const handleSettingsSaved = useCallback(() => {
|
||||
refreshLLMSettings();
|
||||
initializeAgent();
|
||||
}, [refreshLLMSettings, initializeAgent]);
|
||||
|
||||
// Render based on view mode
|
||||
if (viewMode === 'onboarding') {
|
||||
return <DropZone onFileSelect={handleFileSelect} onGitClone={handleGitClone} />;
|
||||
}
|
||||
|
||||
if (viewMode === 'loading' && progress) {
|
||||
return <LoadingOverlay progress={progress} />;
|
||||
}
|
||||
|
||||
// Exploring view
|
||||
return (
|
||||
<ErrorBoundary>
|
||||
<HomePage />
|
||||
</ErrorBoundary>
|
||||
<div className="flex flex-col h-screen bg-void overflow-hidden">
|
||||
<Header onFocusNode={handleFocusNode} />
|
||||
|
||||
<main className="flex-1 flex min-h-0">
|
||||
{/* Left Panel - File Tree */}
|
||||
<FileTreePanel onFocusNode={handleFocusNode} />
|
||||
|
||||
{/* Graph area - takes remaining space */}
|
||||
<div className="flex-1 relative min-w-0">
|
||||
<GraphCanvas ref={graphCanvasRef} />
|
||||
|
||||
{/* Code References Panel (overlay) - does NOT resize the graph, it overlaps on top */}
|
||||
{isCodePanelOpen && (codeReferences.length > 0 || !!selectedNode) && (
|
||||
<div className="absolute inset-y-0 left-0 z-30 pointer-events-auto">
|
||||
<CodeReferencesPanel onFocusNode={handleFocusNode} />
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Right Panel - Code & Chat (tabbed) */}
|
||||
{isRightPanelOpen && <RightPanel />}
|
||||
</main>
|
||||
|
||||
<StatusBar />
|
||||
|
||||
{/* Settings Panel (modal) */}
|
||||
<SettingsPanel
|
||||
isOpen={isSettingsPanelOpen}
|
||||
onClose={() => setSettingsPanelOpen(false)}
|
||||
onSettingsSaved={handleSettingsSaved}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
function App() {
|
||||
return (
|
||||
<AppStateProvider>
|
||||
<AppContent />
|
||||
</AppStateProvider>
|
||||
);
|
||||
}
|
||||
|
||||
export default App;
|
||||
|
||||
@@ -1,291 +0,0 @@
|
||||
/**
|
||||
* KuzuDB COPY Implementation Tests
|
||||
*
|
||||
* Tests for the new COPY-based bulk loading functionality
|
||||
*/
|
||||
|
||||
import { GitNexusCSVGenerator, CSVUtils } from '../core/kuzu/csv-generator';
|
||||
import type { GraphNode, GraphRelationship, NodeLabel, RelationshipType } from '../core/graph/types';
|
||||
|
||||
describe('GitNexusCSVGenerator', () => {
|
||||
describe('generateNodeCSV', () => {
|
||||
it('should generate valid CSV for Function nodes', () => {
|
||||
const nodes: GraphNode[] = [
|
||||
{
|
||||
id: 'func_1',
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: 'testFunction',
|
||||
filePath: '/src/test.ts',
|
||||
startLine: 10,
|
||||
endLine: 20,
|
||||
parameters: ['param1', 'param2'],
|
||||
returnType: 'string',
|
||||
isAsync: true,
|
||||
isStatic: false
|
||||
}
|
||||
},
|
||||
{
|
||||
id: 'func_2',
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: 'anotherFunction',
|
||||
filePath: '/src/another.ts',
|
||||
startLine: 30,
|
||||
endLine: 40,
|
||||
parameters: [],
|
||||
returnType: 'void'
|
||||
}
|
||||
}
|
||||
];
|
||||
|
||||
const csv = GitNexusCSVGenerator.generateNodeCSV(nodes, 'Function');
|
||||
|
||||
// Check header
|
||||
expect(csv).toContain('id,');
|
||||
expect(csv).toContain('name,');
|
||||
expect(csv).toContain('filePath,');
|
||||
|
||||
// Check data rows
|
||||
expect(csv).toContain('func_1,testFunction');
|
||||
expect(csv).toContain('func_2,anotherFunction');
|
||||
|
||||
// Check array handling
|
||||
expect(csv).toContain('"[""param1"",""param2""]"'); // Escaped JSON array
|
||||
|
||||
// Check boolean handling
|
||||
expect(csv).toContain('true');
|
||||
expect(csv).toContain('false');
|
||||
});
|
||||
|
||||
it('should handle special characters in CSV', () => {
|
||||
const nodes: GraphNode[] = [
|
||||
{
|
||||
id: 'test_1',
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: 'function,with"quotes',
|
||||
docstring: 'Multi\nline\ndocstring',
|
||||
filePath: '/src/test.ts'
|
||||
}
|
||||
}
|
||||
];
|
||||
|
||||
const csv = GitNexusCSVGenerator.generateNodeCSV(nodes, 'Function');
|
||||
|
||||
// Check proper CSV escaping
|
||||
expect(csv).toContain('"function,with""quotes"');
|
||||
expect(csv).toContain('"Multi\nline\ndocstring"');
|
||||
});
|
||||
|
||||
it('should handle empty node list', () => {
|
||||
const csv = GitNexusCSVGenerator.generateNodeCSV([], 'Function');
|
||||
expect(csv).toBe('');
|
||||
});
|
||||
|
||||
it('should filter nodes by label', () => {
|
||||
const nodes: GraphNode[] = [
|
||||
{
|
||||
id: 'func_1',
|
||||
label: 'Function',
|
||||
properties: { name: 'testFunction' }
|
||||
},
|
||||
{
|
||||
id: 'class_1',
|
||||
label: 'Class',
|
||||
properties: { name: 'TestClass' }
|
||||
}
|
||||
];
|
||||
|
||||
const csv = GitNexusCSVGenerator.generateNodeCSV(nodes, 'Function');
|
||||
|
||||
expect(csv).toContain('func_1');
|
||||
expect(csv).not.toContain('class_1');
|
||||
});
|
||||
});
|
||||
|
||||
describe('generateRelationshipCSV', () => {
|
||||
it('should generate valid CSV for CALLS relationships', () => {
|
||||
const relationships: GraphRelationship[] = [
|
||||
{
|
||||
id: 'call_1',
|
||||
type: 'CALLS',
|
||||
source: 'func_1',
|
||||
target: 'func_2',
|
||||
properties: {
|
||||
callType: 'direct',
|
||||
startLine: 15,
|
||||
functionName: 'testFunction'
|
||||
}
|
||||
},
|
||||
{
|
||||
id: 'call_2',
|
||||
type: 'CALLS',
|
||||
source: 'func_2',
|
||||
target: 'func_3',
|
||||
properties: {
|
||||
callType: 'indirect',
|
||||
startLine: 25
|
||||
}
|
||||
}
|
||||
];
|
||||
|
||||
const csv = GitNexusCSVGenerator.generateRelationshipCSV(relationships, 'CALLS');
|
||||
|
||||
// Check header
|
||||
expect(csv).toContain('source,target');
|
||||
expect(csv).toContain('callType');
|
||||
expect(csv).toContain('startLine');
|
||||
|
||||
// Check data
|
||||
expect(csv).toContain('func_1,func_2');
|
||||
expect(csv).toContain('func_2,func_3');
|
||||
expect(csv).toContain('direct');
|
||||
expect(csv).toContain('indirect');
|
||||
});
|
||||
|
||||
it('should handle empty relationship list', () => {
|
||||
const csv = GitNexusCSVGenerator.generateRelationshipCSV([], 'CALLS');
|
||||
expect(csv).toBe('');
|
||||
});
|
||||
|
||||
it('should filter relationships by type', () => {
|
||||
const relationships: GraphRelationship[] = [
|
||||
{
|
||||
id: 'call_1',
|
||||
type: 'CALLS',
|
||||
source: 'func_1',
|
||||
target: 'func_2',
|
||||
properties: {}
|
||||
},
|
||||
{
|
||||
id: 'import_1',
|
||||
type: 'IMPORTS',
|
||||
source: 'file_1',
|
||||
target: 'file_2',
|
||||
properties: {}
|
||||
}
|
||||
];
|
||||
|
||||
const csv = GitNexusCSVGenerator.generateRelationshipCSV(relationships, 'CALLS');
|
||||
|
||||
expect(csv).toContain('call_1');
|
||||
expect(csv).not.toContain('import_1');
|
||||
});
|
||||
});
|
||||
|
||||
describe('CSV validation', () => {
|
||||
it('should validate correct CSV format', () => {
|
||||
const validCSV = `name,age,city
|
||||
"John Doe",30,New York
|
||||
"Jane Smith",25,Boston`;
|
||||
|
||||
const result = GitNexusCSVGenerator.validateCSVFormat(validCSV);
|
||||
|
||||
expect(result.isValid).toBe(true);
|
||||
expect(result.errors).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('should detect invalid CSV format', () => {
|
||||
const invalidCSV = `name,age,city
|
||||
"John Doe",30,New York,Extra
|
||||
"Jane Smith",25`;
|
||||
|
||||
const result = GitNexusCSVGenerator.validateCSVFormat(invalidCSV);
|
||||
|
||||
expect(result.isValid).toBe(false);
|
||||
expect(result.errors.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('should handle empty CSV', () => {
|
||||
const result = GitNexusCSVGenerator.validateCSVFormat('');
|
||||
|
||||
expect(result.isValid).toBe(false);
|
||||
expect(result.errors).toContain('CSV data is empty');
|
||||
});
|
||||
});
|
||||
|
||||
describe('CSVUtils', () => {
|
||||
it('should calculate optimal batch size', () => {
|
||||
const nodes: GraphNode[] = Array.from({ length: 100 }, (_, i) => ({
|
||||
id: `node_${i}`,
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: `function_${i}`,
|
||||
filePath: `/src/file_${i}.ts`,
|
||||
startLine: i * 10,
|
||||
endLine: i * 10 + 5
|
||||
}
|
||||
}));
|
||||
|
||||
const batchSize = CSVUtils.calculateOptimalBatchSize(nodes);
|
||||
|
||||
expect(batchSize).toBeGreaterThan(0);
|
||||
expect(batchSize).toBeLessThanOrEqual(2000);
|
||||
});
|
||||
|
||||
it('should recommend chunked processing for large datasets', () => {
|
||||
expect(CSVUtils.shouldUseChunkedProcessing(500)).toBe(false);
|
||||
expect(CSVUtils.shouldUseChunkedProcessing(1500)).toBe(true);
|
||||
});
|
||||
|
||||
it('should provide memory-safe chunk sizes', () => {
|
||||
expect(CSVUtils.getMemorySafeChunkSize(50)).toBe(50);
|
||||
expect(CSVUtils.getMemorySafeChunkSize(500)).toBe(500);
|
||||
expect(CSVUtils.getMemorySafeChunkSize(2000)).toBe(1000);
|
||||
expect(CSVUtils.getMemorySafeChunkSize(10000)).toBe(1500);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Chunked CSV generation', () => {
|
||||
it('should generate CSV in chunks for large datasets', () => {
|
||||
const nodes: GraphNode[] = Array.from({ length: 2500 }, (_, i) => ({
|
||||
id: `node_${i}`,
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: `function_${i}`,
|
||||
filePath: `/src/file_${i}.ts`
|
||||
}
|
||||
}));
|
||||
|
||||
const csv = GitNexusCSVGenerator.generateNodeCSVInChunks(nodes, 'Function', 1000);
|
||||
|
||||
// Should contain all nodes
|
||||
expect(csv.split('\n').length).toBeGreaterThan(2500); // +1 for header
|
||||
expect(csv).toContain('node_0');
|
||||
expect(csv).toContain('node_2499');
|
||||
|
||||
// Should have only one header
|
||||
const headerCount = (csv.match(/^id,/gm) || []).length;
|
||||
expect(headerCount).toBe(1);
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
// Mock data generators for testing
|
||||
function generateTestNodes(count: number, label: NodeLabel): GraphNode[] {
|
||||
return Array.from({ length: count }, (_, i) => ({
|
||||
id: `${label.toLowerCase()}_${i}`,
|
||||
label,
|
||||
properties: {
|
||||
name: `test${label}${i}`,
|
||||
filePath: `/src/test${i}.ts`,
|
||||
startLine: i * 10,
|
||||
endLine: i * 10 + 5
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
function generateTestRelationships(count: number, type: RelationshipType): GraphRelationship[] {
|
||||
return Array.from({ length: count }, (_, i) => ({
|
||||
id: `${type.toLowerCase()}_${i}`,
|
||||
type,
|
||||
source: `source_${i}`,
|
||||
target: `target_${i}`,
|
||||
properties: {
|
||||
line: i * 5
|
||||
}
|
||||
}));
|
||||
}
|
||||
|
||||
export { generateTestNodes, generateTestRelationships };
|
||||
@@ -1,636 +0,0 @@
|
||||
/**
|
||||
* Comprehensive Test Suite for KuzuDB Implementation
|
||||
*
|
||||
* Tests all components of the KuzuDB integration including:
|
||||
* - WASM loader
|
||||
* - Query engine
|
||||
* - Knowledge graph implementation
|
||||
* - Schema management
|
||||
* - Data migration
|
||||
*/
|
||||
|
||||
import { describe, it, expect, beforeEach, afterEach, jest } from '@jest/globals';
|
||||
import { KnowledgeGraph, GraphNode, GraphRelationship } from '../core/graph/types.js';
|
||||
import { initKuzuDB, isKuzuDBSupported } from '../core/kuzu/kuzu-loader.js';
|
||||
import { KuzuQueryEngine } from '../core/graph/kuzu-query-engine.js';
|
||||
import { KuzuKnowledgeGraph } from '../core/graph/kuzu-knowledge-graph.js';
|
||||
import { KuzuSchemaManager, NODE_TABLE_SCHEMAS, RELATIONSHIP_TABLE_SCHEMAS } from '../core/kuzu/kuzu-schema.js';
|
||||
import { generateId } from '../lib/utils.js';
|
||||
|
||||
// Mock feature flags for testing
|
||||
jest.mock('../config/feature-flags.js', () => ({
|
||||
isKuzuDBEnabled: () => true,
|
||||
isKuzuDBPersistenceEnabled: () => true,
|
||||
isPerformanceMonitoringEnabled: () => true
|
||||
}));
|
||||
|
||||
describe('KuzuDB WASM Loader', () => {
|
||||
it('should check WebAssembly support', () => {
|
||||
expect(typeof isKuzuDBSupported).toBe('function');
|
||||
// Note: In test environment, WebAssembly might not be available
|
||||
});
|
||||
|
||||
it('should initialize KuzuDB instance', async () => {
|
||||
// This test will be skipped if WASM is not available in test environment
|
||||
if (!isKuzuDBSupported()) {
|
||||
console.log('⚠️ Skipping WASM test - WebAssembly not supported in test environment');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
const instance = await initKuzuDB();
|
||||
expect(instance).toBeDefined();
|
||||
expect(instance.isReady).toBeDefined();
|
||||
expect(instance.createDatabase).toBeDefined();
|
||||
expect(instance.executeQuery).toBeDefined();
|
||||
} catch (error) {
|
||||
// WASM loading might fail in test environment
|
||||
console.log('⚠️ WASM loading failed in test environment:', error);
|
||||
}
|
||||
});
|
||||
|
||||
it('should handle WASM loading failures gracefully', async () => {
|
||||
// Mock fetch to simulate WASM loading failure
|
||||
const originalFetch = global.fetch;
|
||||
global.fetch = jest.fn().mockRejectedValue(new Error('WASM not found'));
|
||||
|
||||
try {
|
||||
await initKuzuDB();
|
||||
expect(false).toBe(true); // Should not reach here
|
||||
} catch (error) {
|
||||
expect(error).toBeInstanceOf(Error);
|
||||
expect(error.message).toContain('WASM loading failed');
|
||||
} finally {
|
||||
global.fetch = originalFetch;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('KuzuQueryEngine', () => {
|
||||
let queryEngine;
|
||||
|
||||
beforeEach(() => {
|
||||
queryEngine = new KuzuQueryEngine({
|
||||
databasePath: '/test_db',
|
||||
enableCache: true,
|
||||
cacheSize: 100,
|
||||
cacheTTL: 1000
|
||||
});
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
if (queryEngine.isReady()) {
|
||||
await queryEngine.close();
|
||||
}
|
||||
});
|
||||
|
||||
it('should initialize with default options', () => {
|
||||
const defaultEngine = new KuzuQueryEngine();
|
||||
expect(defaultEngine).toBeDefined();
|
||||
expect(defaultEngine.isReady()).toBe(false);
|
||||
});
|
||||
|
||||
it('should handle initialization failure', async () => {
|
||||
// Mock initKuzuDB to fail
|
||||
jest.doMock('../core/kuzu/kuzu-loader.js', () => ({
|
||||
initKuzuDB: jest.fn().mockRejectedValue(new Error('Init failed'))
|
||||
}));
|
||||
|
||||
try {
|
||||
await queryEngine.initialize();
|
||||
expect(false).toBe(true); // Should not reach here
|
||||
} catch (error) {
|
||||
expect(error).toBeInstanceOf(Error);
|
||||
expect(error.message).toContain('initialization failed');
|
||||
}
|
||||
});
|
||||
|
||||
it('should provide statistics', () => {
|
||||
const stats = queryEngine.getStatistics();
|
||||
expect(stats).toHaveProperty('isInitialized');
|
||||
expect(stats).toHaveProperty('queryCount');
|
||||
expect(stats).toHaveProperty('averageExecutionTime');
|
||||
expect(stats).toHaveProperty('cacheHitRate');
|
||||
expect(stats).toHaveProperty('cacheSize');
|
||||
|
||||
expect(stats.isInitialized).toBe(false);
|
||||
expect(stats.queryCount).toBe(0);
|
||||
expect(stats.cacheSize).toBe(0);
|
||||
});
|
||||
|
||||
it('should clear cache', () => {
|
||||
queryEngine.clearCache();
|
||||
const stats = queryEngine.getStatistics();
|
||||
expect(stats.cacheSize).toBe(0);
|
||||
});
|
||||
|
||||
it('should reject queries when not initialized', async () => {
|
||||
try {
|
||||
await queryEngine.executeQuery('MATCH (n) RETURN n');
|
||||
expect(false).toBe(true); // Should not reach here
|
||||
} catch (error) {
|
||||
expect(error).toBeInstanceOf(Error);
|
||||
expect(error.message).toContain('not initialized');
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('KuzuKnowledgeGraph', () => {
|
||||
let mockQueryEngine;
|
||||
let kuzuGraph;
|
||||
|
||||
beforeEach(() => {
|
||||
// Create mock query engine
|
||||
mockQueryEngine = {
|
||||
initialize: jest.fn().mockResolvedValue(undefined),
|
||||
executeQuery: jest.fn().mockResolvedValue({
|
||||
columns: [],
|
||||
rows: [],
|
||||
rowCount: 0,
|
||||
resultCount: 0,
|
||||
executionTime: 0
|
||||
}),
|
||||
executeGraphQuery: jest.fn().mockResolvedValue({
|
||||
nodes: [],
|
||||
relationships: [],
|
||||
executionTime: 0
|
||||
}),
|
||||
importGraph: jest.fn().mockResolvedValue(undefined),
|
||||
isReady: jest.fn().mockReturnValue(true),
|
||||
close: jest.fn().mockResolvedValue(undefined),
|
||||
getStatistics: jest.fn().mockReturnValue({
|
||||
isInitialized: true,
|
||||
queryCount: 0,
|
||||
averageExecutionTime: 0,
|
||||
cacheHitRate: 0,
|
||||
cacheSize: 0
|
||||
}),
|
||||
clearCache: jest.fn()
|
||||
};
|
||||
|
||||
kuzuGraph = new KuzuKnowledgeGraph(mockQueryEngine, {
|
||||
enableCache: true,
|
||||
batchSize: 10,
|
||||
autoCommit: false
|
||||
});
|
||||
});
|
||||
|
||||
it('should implement KnowledgeGraph interface', () => {
|
||||
expect(kuzuGraph.addNode).toBeDefined();
|
||||
expect(kuzuGraph.addRelationship).toBeDefined();
|
||||
expect(kuzuGraph.nodes).toBeDefined();
|
||||
expect(kuzuGraph.relationships).toBeDefined();
|
||||
});
|
||||
|
||||
it('should add nodes to pending batch', () => {
|
||||
const node = {
|
||||
id: generateId('test'),
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: 'testFunction',
|
||||
filePath: '/test.ts'
|
||||
}
|
||||
};
|
||||
|
||||
kuzuGraph.addNode(node);
|
||||
const stats = kuzuGraph.getCacheStatistics();
|
||||
expect(stats.pendingNodes).toBe(1);
|
||||
expect(stats.nodesCached).toBe(1);
|
||||
});
|
||||
|
||||
it('should add relationships to pending batch', () => {
|
||||
const relationship = {
|
||||
id: generateId('rel'),
|
||||
type: 'CALLS',
|
||||
source: 'func1',
|
||||
target: 'func2',
|
||||
properties: {
|
||||
confidence: 0.9
|
||||
}
|
||||
};
|
||||
|
||||
kuzuGraph.addRelationship(relationship);
|
||||
const stats = kuzuGraph.getCacheStatistics();
|
||||
expect(stats.pendingRelationships).toBe(1);
|
||||
expect(stats.relationshipsCached).toBe(1);
|
||||
});
|
||||
|
||||
it('should commit nodes to database', async () => {
|
||||
const node = {
|
||||
id: generateId('test'),
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: 'testFunction',
|
||||
filePath: '/test.ts'
|
||||
}
|
||||
};
|
||||
|
||||
kuzuGraph.addNode(node);
|
||||
await kuzuGraph.commitNodes();
|
||||
|
||||
expect(mockQueryEngine.executeQuery).toHaveBeenCalled();
|
||||
const stats = kuzuGraph.getCacheStatistics();
|
||||
expect(stats.pendingNodes).toBe(0);
|
||||
});
|
||||
|
||||
it('should commit relationships to database', async () => {
|
||||
const relationship = {
|
||||
id: generateId('rel'),
|
||||
type: 'CALLS',
|
||||
source: 'func1',
|
||||
target: 'func2',
|
||||
properties: {
|
||||
confidence: 0.9
|
||||
}
|
||||
};
|
||||
|
||||
kuzuGraph.addRelationship(relationship);
|
||||
await kuzuGraph.commitRelationships();
|
||||
|
||||
expect(mockQueryEngine.executeQuery).toHaveBeenCalled();
|
||||
const stats = kuzuGraph.getCacheStatistics();
|
||||
expect(stats.pendingRelationships).toBe(0);
|
||||
});
|
||||
|
||||
it('should find nodes by label', async () => {
|
||||
const mockNodes = [
|
||||
{
|
||||
id: 'func1',
|
||||
label: 'Function' as const,
|
||||
properties: { name: 'test1' }
|
||||
}
|
||||
];
|
||||
|
||||
mockQueryEngine.executeQuery.mockResolvedValue({
|
||||
columns: ['n'],
|
||||
rows: [mockNodes],
|
||||
rowCount: 1,
|
||||
resultCount: 1,
|
||||
executionTime: 5
|
||||
});
|
||||
|
||||
const nodes = await kuzuGraph.findNodesByLabel('Function');
|
||||
expect(mockQueryEngine.executeQuery).toHaveBeenCalledWith(
|
||||
'MATCH (n:Function) RETURN n'
|
||||
);
|
||||
});
|
||||
|
||||
it('should find node by ID', async () => {
|
||||
const mockNode = {
|
||||
id: 'func1',
|
||||
label: 'Function' as const,
|
||||
properties: { name: 'test1' }
|
||||
};
|
||||
|
||||
mockQueryEngine.executeQuery.mockResolvedValue({
|
||||
columns: ['n'],
|
||||
rows: [[mockNode]],
|
||||
rowCount: 1,
|
||||
resultCount: 1,
|
||||
executionTime: 5
|
||||
});
|
||||
|
||||
const node = await kuzuGraph.findNodeById('func1');
|
||||
expect(mockQueryEngine.executeQuery).toHaveBeenCalledWith(
|
||||
"MATCH (n) WHERE n.id = 'func1' RETURN n LIMIT 1"
|
||||
);
|
||||
});
|
||||
|
||||
it('should get connected nodes', async () => {
|
||||
mockQueryEngine.executeQuery
|
||||
.mockResolvedValueOnce({
|
||||
columns: ['target'],
|
||||
rows: [],
|
||||
rowCount: 0,
|
||||
resultCount: 0,
|
||||
executionTime: 5
|
||||
})
|
||||
.mockResolvedValueOnce({
|
||||
columns: ['source'],
|
||||
rows: [],
|
||||
rowCount: 0,
|
||||
resultCount: 0,
|
||||
executionTime: 5
|
||||
});
|
||||
|
||||
const connected = await kuzuGraph.getConnectedNodes('func1', 'CALLS');
|
||||
expect(connected).toHaveProperty('outgoing');
|
||||
expect(connected).toHaveProperty('incoming');
|
||||
expect(Array.isArray(connected.outgoing)).toBe(true);
|
||||
expect(Array.isArray(connected.incoming)).toBe(true);
|
||||
});
|
||||
|
||||
it('should get graph statistics', async () => {
|
||||
mockQueryEngine.executeQuery
|
||||
.mockResolvedValueOnce({ columns: ['count'], rows: [[100]], rowCount: 1, resultCount: 1, executionTime: 5 })
|
||||
.mockResolvedValueOnce({ columns: ['count'], rows: [[50]], rowCount: 1, resultCount: 1, executionTime: 5 })
|
||||
.mockResolvedValueOnce({ columns: ['label', 'count'], rows: [['Function', 80], ['Class', 20]], rowCount: 2, resultCount: 2, executionTime: 5 })
|
||||
.mockResolvedValueOnce({ columns: ['type', 'count'], rows: [['CALLS', 30], ['CONTAINS', 20]], rowCount: 2, resultCount: 2, executionTime: 5 });
|
||||
|
||||
const stats = await kuzuGraph.getStatistics();
|
||||
expect(stats.nodeCount).toBe(100);
|
||||
expect(stats.relationshipCount).toBe(50);
|
||||
expect(stats.nodesByLabel).toEqual({ Function: 80, Class: 20 });
|
||||
expect(stats.relationshipsByType).toEqual({ CALLS: 30, CONTAINS: 20 });
|
||||
});
|
||||
|
||||
it('should clear cache', () => {
|
||||
const node = {
|
||||
id: generateId('test'),
|
||||
label: 'Function',
|
||||
properties: { name: 'test' }
|
||||
};
|
||||
|
||||
kuzuGraph.addNode(node);
|
||||
expect(kuzuGraph.getCacheStatistics().nodesCached).toBe(1);
|
||||
|
||||
kuzuGraph.clearCache();
|
||||
expect(kuzuGraph.getCacheStatistics().nodesCached).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('KuzuSchemaManager', () => {
|
||||
let mockKuzuInstance;
|
||||
let schemaManager;
|
||||
|
||||
beforeEach(() => {
|
||||
mockKuzuInstance = {
|
||||
createNodeTable: jest.fn().mockResolvedValue(undefined),
|
||||
createRelTable: jest.fn().mockResolvedValue(undefined),
|
||||
executeQuery: jest.fn().mockResolvedValue({
|
||||
columns: ['name', 'type'],
|
||||
rows: [['Project', 'NODE'], ['Function', 'NODE'], ['CONTAINS', 'REL']],
|
||||
rowCount: 3
|
||||
})
|
||||
};
|
||||
|
||||
schemaManager = new KuzuSchemaManager(mockKuzuInstance);
|
||||
});
|
||||
|
||||
it('should initialize complete schema', async () => {
|
||||
await schemaManager.initializeSchema();
|
||||
|
||||
// Should create all node tables
|
||||
expect(mockKuzuInstance.createNodeTable).toHaveBeenCalledTimes(
|
||||
Object.keys(NODE_TABLE_SCHEMAS).length
|
||||
);
|
||||
|
||||
// Should create all relationship tables
|
||||
expect(mockKuzuInstance.createRelTable).toHaveBeenCalledTimes(
|
||||
RELATIONSHIP_TABLE_SCHEMAS.length
|
||||
);
|
||||
});
|
||||
|
||||
it('should handle table creation errors gracefully', async () => {
|
||||
mockKuzuInstance.createNodeTable.mockRejectedValue(new Error('Table exists'));
|
||||
|
||||
// Should not throw error, just log warning
|
||||
await expect(schemaManager.initializeSchema()).resolves.not.toThrow();
|
||||
});
|
||||
|
||||
it('should validate schema', async () => {
|
||||
const validation = await schemaManager.validateSchema();
|
||||
|
||||
expect(validation).toHaveProperty('isValid');
|
||||
expect(validation).toHaveProperty('missingTables');
|
||||
expect(validation).toHaveProperty('errors');
|
||||
expect(Array.isArray(validation.missingTables)).toBe(true);
|
||||
expect(Array.isArray(validation.errors)).toBe(true);
|
||||
});
|
||||
|
||||
it('should get schema information', async () => {
|
||||
const schemaInfo = await schemaManager.getSchemaInfo();
|
||||
|
||||
expect(schemaInfo).toHaveProperty('nodeTables');
|
||||
expect(schemaInfo).toHaveProperty('relationshipTables');
|
||||
expect(schemaInfo).toHaveProperty('totalTables');
|
||||
expect(Array.isArray(schemaInfo.nodeTables)).toBe(true);
|
||||
expect(Array.isArray(schemaInfo.relationshipTables)).toBe(true);
|
||||
expect(typeof schemaInfo.totalTables).toBe('number');
|
||||
});
|
||||
|
||||
it('should drop all tables', async () => {
|
||||
mockKuzuInstance.executeQuery.mockResolvedValue({
|
||||
columns: [],
|
||||
rows: [],
|
||||
rowCount: 0
|
||||
});
|
||||
|
||||
await schemaManager.dropAllTables();
|
||||
|
||||
// Should call DROP TABLE for each table
|
||||
const expectedDrops = Object.keys(NODE_TABLE_SCHEMAS).length + RELATIONSHIP_TABLE_SCHEMAS.length;
|
||||
expect(mockKuzuInstance.executeQuery).toHaveBeenCalledTimes(expectedDrops);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Data Migration Integration', () => {
|
||||
let mockQueryEngine;
|
||||
let kuzuGraph;
|
||||
|
||||
beforeEach(() => {
|
||||
mockQueryEngine = {
|
||||
initialize: jest.fn().mockResolvedValue(undefined),
|
||||
executeQuery: jest.fn().mockResolvedValue({
|
||||
columns: [],
|
||||
rows: [],
|
||||
rowCount: 0,
|
||||
resultCount: 0,
|
||||
executionTime: 0
|
||||
}),
|
||||
importGraph: jest.fn().mockResolvedValue(undefined),
|
||||
isReady: jest.fn().mockReturnValue(true),
|
||||
close: jest.fn().mockResolvedValue(undefined),
|
||||
getStatistics: jest.fn().mockReturnValue({
|
||||
isInitialized: true,
|
||||
queryCount: 0,
|
||||
averageExecutionTime: 0,
|
||||
cacheHitRate: 0,
|
||||
cacheSize: 0
|
||||
}),
|
||||
clearCache: jest.fn()
|
||||
};
|
||||
|
||||
kuzuGraph = new KuzuKnowledgeGraph(mockQueryEngine);
|
||||
});
|
||||
|
||||
it('should migrate a complete knowledge graph', async () => {
|
||||
// Create a sample knowledge graph
|
||||
const sampleGraph = {
|
||||
nodes: [
|
||||
{
|
||||
id: 'project1',
|
||||
label: 'Project',
|
||||
properties: {
|
||||
name: 'TestProject',
|
||||
path: '/test/project'
|
||||
}
|
||||
},
|
||||
{
|
||||
id: 'file1',
|
||||
label: 'File',
|
||||
properties: {
|
||||
name: 'main.ts',
|
||||
filePath: '/test/project/main.ts',
|
||||
language: 'typescript'
|
||||
}
|
||||
},
|
||||
{
|
||||
id: 'func1',
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: 'main',
|
||||
filePath: '/test/project/main.ts',
|
||||
startLine: 10,
|
||||
endLine: 20
|
||||
}
|
||||
}
|
||||
],
|
||||
relationships: [
|
||||
{
|
||||
id: 'contains1',
|
||||
type: 'CONTAINS',
|
||||
source: 'project1',
|
||||
target: 'file1',
|
||||
properties: {}
|
||||
},
|
||||
{
|
||||
id: 'contains2',
|
||||
type: 'CONTAINS',
|
||||
source: 'file1',
|
||||
target: 'func1',
|
||||
properties: {}
|
||||
}
|
||||
],
|
||||
addNode: function(node: GraphNode): void {
|
||||
this.nodes.push(node);
|
||||
},
|
||||
addRelationship: function(relationship: GraphRelationship): void {
|
||||
this.relationships.push(relationship);
|
||||
}
|
||||
};
|
||||
|
||||
// Import the graph
|
||||
await mockQueryEngine.importGraph(sampleGraph);
|
||||
|
||||
expect(mockQueryEngine.importGraph).toHaveBeenCalledWith(sampleGraph);
|
||||
});
|
||||
|
||||
it('should handle migration errors gracefully', async () => {
|
||||
mockQueryEngine.importGraph.mockRejectedValue(new Error('Import failed'));
|
||||
|
||||
const sampleGraph = {
|
||||
nodes: [],
|
||||
relationships: [],
|
||||
addNode: function(node: GraphNode): void {
|
||||
this.nodes.push(node);
|
||||
},
|
||||
addRelationship: function(relationship: GraphRelationship): void {
|
||||
this.relationships.push(relationship);
|
||||
}
|
||||
};
|
||||
|
||||
await expect(mockQueryEngine.importGraph(sampleGraph)).rejects.toThrow('Import failed');
|
||||
});
|
||||
});
|
||||
|
||||
describe('Performance and Memory Management', () => {
|
||||
let kuzuGraph;
|
||||
let mockQueryEngine;
|
||||
|
||||
beforeEach(() => {
|
||||
mockQueryEngine = {
|
||||
executeQuery: jest.fn().mockResolvedValue({
|
||||
columns: [],
|
||||
rows: [],
|
||||
rowCount: 0,
|
||||
resultCount: 0,
|
||||
executionTime: 0
|
||||
})
|
||||
};
|
||||
|
||||
kuzuGraph = new KuzuKnowledgeGraph(mockQueryEngine, {
|
||||
batchSize: 5,
|
||||
autoCommit: true
|
||||
});
|
||||
});
|
||||
|
||||
it('should handle large batches efficiently', () => {
|
||||
// Add many nodes
|
||||
for (let i = 0; i < 100; i++) {
|
||||
const node = {
|
||||
id: `node${i}`,
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: `function${i}`,
|
||||
filePath: `/test/file${i}.ts`
|
||||
}
|
||||
};
|
||||
kuzuGraph.addNode(node);
|
||||
}
|
||||
|
||||
const stats = kuzuGraph.getCacheStatistics();
|
||||
expect(stats.nodesCached).toBe(100);
|
||||
});
|
||||
|
||||
it('should manage memory usage with cache limits', () => {
|
||||
const smallCacheGraph = new KuzuKnowledgeGraph(mockQueryEngine, {
|
||||
enableCache: true,
|
||||
batchSize: 1000
|
||||
});
|
||||
|
||||
// Add nodes beyond cache limit
|
||||
for (let i = 0; i < 50; i++) {
|
||||
const node = {
|
||||
id: `node${i}`,
|
||||
label: 'Function',
|
||||
properties: { name: `function${i}` }
|
||||
};
|
||||
smallCacheGraph.addNode(node);
|
||||
}
|
||||
|
||||
// Cache should still work
|
||||
const stats = smallCacheGraph.getCacheStatistics();
|
||||
expect(stats.nodesCached).toBe(50);
|
||||
});
|
||||
});
|
||||
|
||||
// Integration test that requires manual verification
|
||||
describe('Manual Integration Tests', () => {
|
||||
it.skip('should perform end-to-end test with real KuzuDB', async () => {
|
||||
// This test is skipped by default as it requires actual KuzuDB WASM
|
||||
// To run this test:
|
||||
// 1. Ensure KuzuDB WASM is properly loaded
|
||||
// 2. Remove .skip from the test
|
||||
// 3. Run with proper test environment
|
||||
|
||||
try {
|
||||
const queryEngine = new KuzuQueryEngine();
|
||||
await queryEngine.initialize();
|
||||
|
||||
const kuzuGraph = new KuzuKnowledgeGraph(queryEngine);
|
||||
|
||||
// Add sample data
|
||||
const node = {
|
||||
id: 'test1',
|
||||
label: 'Function',
|
||||
properties: {
|
||||
name: 'testFunction',
|
||||
filePath: '/test.ts'
|
||||
}
|
||||
};
|
||||
|
||||
kuzuGraph.addNode(node);
|
||||
await kuzuGraph.commitAll();
|
||||
|
||||
// Query the data
|
||||
const nodes = await kuzuGraph.findNodesByLabel('Function');
|
||||
expect(nodes.length).toBeGreaterThan(0);
|
||||
|
||||
await queryEngine.close();
|
||||
} catch (error) {
|
||||
console.log('End-to-end test requires real KuzuDB WASM environment');
|
||||
throw error;
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -1,147 +0,0 @@
|
||||
/**
|
||||
* Parallel Processing Integration Test
|
||||
* Tests the new parallel processing implementation
|
||||
*/
|
||||
|
||||
import { GraphPipeline } from '../core/ingestion/pipeline';
|
||||
import type { PipelineInput } from '../core/ingestion/pipeline';
|
||||
|
||||
describe('Parallel Processing Integration', () => {
|
||||
let pipeline: GraphPipeline;
|
||||
|
||||
beforeEach(() => {
|
||||
pipeline = new GraphPipeline();
|
||||
});
|
||||
|
||||
it('should process files using parallel processing', async () => {
|
||||
// Mock file data with TypeScript and JavaScript files
|
||||
const mockFileContents = new Map<string, string>([
|
||||
['src/test.ts', `
|
||||
function testFunction(param: string): string {
|
||||
return "test: " + param;
|
||||
}
|
||||
|
||||
export class TestClass {
|
||||
private value: number = 42;
|
||||
|
||||
public getValue(): number {
|
||||
return this.value;
|
||||
}
|
||||
}
|
||||
`],
|
||||
['src/helper.js', `
|
||||
function helperFunction() {
|
||||
return "help";
|
||||
}
|
||||
|
||||
const constantValue = 100;
|
||||
|
||||
module.exports = { helperFunction, constantValue };
|
||||
`],
|
||||
['src/utils.ts', `
|
||||
export interface Config {
|
||||
name: string;
|
||||
version: number;
|
||||
}
|
||||
|
||||
export function processConfig(config: Config): boolean {
|
||||
return config.name.length > 0 && config.version > 0;
|
||||
}
|
||||
`]
|
||||
]);
|
||||
|
||||
const mockFilePaths = Array.from(mockFileContents.keys());
|
||||
|
||||
const input: PipelineInput = {
|
||||
projectRoot: '/test-project',
|
||||
projectName: 'parallel-test',
|
||||
filePaths: mockFilePaths,
|
||||
fileContents: mockFileContents,
|
||||
options: {
|
||||
directoryFilter: 'src',
|
||||
fileExtensions: '.ts,.js'
|
||||
}
|
||||
};
|
||||
|
||||
// Process with parallel pipeline
|
||||
console.log('🧪 Testing parallel processing pipeline...');
|
||||
const startTime = Date.now();
|
||||
|
||||
const graph = await pipeline.run(input);
|
||||
|
||||
const endTime = Date.now();
|
||||
const processingTime = endTime - startTime;
|
||||
|
||||
console.log(`⏱️ Parallel processing completed in ${processingTime}ms`);
|
||||
|
||||
// Validate results
|
||||
expect(graph).toBeDefined();
|
||||
expect(graph.nodes).toBeDefined();
|
||||
expect(graph.relationships).toBeDefined();
|
||||
|
||||
// Should have nodes for files
|
||||
const fileNodes = graph.nodes.filter(node => node.label === 'File');
|
||||
expect(fileNodes.length).toBe(3); // 3 files
|
||||
|
||||
// Should have nodes for functions
|
||||
const functionNodes = graph.nodes.filter(node => node.label === 'Function');
|
||||
expect(functionNodes.length).toBeGreaterThan(0);
|
||||
|
||||
// Should have nodes for classes
|
||||
const classNodes = graph.nodes.filter(node => node.label === 'Class');
|
||||
expect(classNodes.length).toBeGreaterThan(0);
|
||||
|
||||
// Should have relationships
|
||||
expect(graph.relationships.length).toBeGreaterThan(0);
|
||||
|
||||
// Log statistics
|
||||
const nodeStats = graph.nodes.reduce((acc, node) => {
|
||||
acc[node.label] = (acc[node.label] || 0) + 1;
|
||||
return acc;
|
||||
}, {} as Record<string, number>);
|
||||
|
||||
const relationshipStats = graph.relationships.reduce((acc, rel) => {
|
||||
acc[rel.type] = (acc[rel.type] || 0) + 1;
|
||||
return acc;
|
||||
}, {} as Record<string, number>);
|
||||
|
||||
console.log('📊 Parallel Processing Test Results:');
|
||||
console.log('Node types:', nodeStats);
|
||||
console.log('Relationship types:', relationshipStats);
|
||||
console.log(`Total nodes: ${graph.nodes.length}`);
|
||||
console.log(`Total relationships: ${graph.relationships.length}`);
|
||||
|
||||
// Basic validation that we got meaningful results
|
||||
expect(graph.nodes.length).toBeGreaterThan(5); // Should have files + definitions
|
||||
expect(Object.keys(nodeStats)).toContain('File');
|
||||
expect(Object.keys(nodeStats)).toContain('Function');
|
||||
}, 30000); // 30 second timeout for worker initialization
|
||||
|
||||
it('should handle worker pool initialization', async () => {
|
||||
const mockFileContents = new Map<string, string>([
|
||||
['simple.ts', 'const value = 42;']
|
||||
]);
|
||||
|
||||
const input: PipelineInput = {
|
||||
projectRoot: '/simple-test',
|
||||
projectName: 'worker-test',
|
||||
filePaths: ['simple.ts'],
|
||||
fileContents: mockFileContents
|
||||
};
|
||||
|
||||
// This should initialize the worker pool without errors
|
||||
const graph = await pipeline.run(input);
|
||||
|
||||
expect(graph).toBeDefined();
|
||||
expect(graph.nodes.length).toBeGreaterThan(0);
|
||||
}, 30000);
|
||||
|
||||
it('should provide diagnostic information', () => {
|
||||
const diagnostics = pipeline.getParsingDiagnostics();
|
||||
|
||||
expect(diagnostics).toBeDefined();
|
||||
expect(typeof diagnostics.processedFiles).toBe('number');
|
||||
expect(typeof diagnostics.skippedFiles).toBe('number');
|
||||
expect(typeof diagnostics.totalDefinitions).toBe('number');
|
||||
});
|
||||
});
|
||||
@@ -1,437 +0,0 @@
|
||||
import { HumanMessage, SystemMessage } from '@langchain/core/messages';
|
||||
import { z } from 'zod';
|
||||
import { StructuredOutputParser } from '@langchain/core/output_parsers';
|
||||
import type { LLMService, LLMConfig } from './llm-service.ts';
|
||||
import type { KnowledgeGraph } from '../core/graph/types.ts';
|
||||
|
||||
export interface CypherQuery {
|
||||
cypher: string;
|
||||
explanation: string;
|
||||
confidence: number;
|
||||
warnings?: string[];
|
||||
}
|
||||
|
||||
export interface CypherGenerationOptions {
|
||||
maxRetries?: number;
|
||||
includeExamples?: boolean;
|
||||
strictMode?: boolean;
|
||||
}
|
||||
|
||||
// Define Zod schema for structured output
|
||||
const CypherQuerySchema = z.object({
|
||||
cypher: z.string().describe("The Cypher query to execute"),
|
||||
explanation: z.string().describe("Brief explanation of what the query does and why this pattern was chosen"),
|
||||
confidence: z.number().min(0).max(1).describe("Confidence level between 0 and 1")
|
||||
});
|
||||
|
||||
export class CypherGenerator {
|
||||
private llmService: LLMService;
|
||||
private graphSchema: string = '';
|
||||
private outputParser: StructuredOutputParser<typeof CypherQuerySchema>;
|
||||
|
||||
// Common Cypher patterns and examples (Updated for Polymorphic Schema)
|
||||
private static readonly CYPHER_EXAMPLES = [
|
||||
{
|
||||
question: "What functions are in the main.py file?",
|
||||
cypher: "MATCH (f:CodeElement {elementType: 'File', name: 'main.py'})-[r:CodeRelationship {relationshipType: 'CONTAINS'}]->(func:CodeElement {elementType: 'Function'}) RETURN func.name, func.startLine"
|
||||
},
|
||||
{
|
||||
question: "Which functions call the authenticate function?",
|
||||
cypher: "MATCH (caller:CodeElement)-[r:CodeRelationship {relationshipType: 'CALLS'}]->(target:CodeElement {elementType: 'Function', name: 'authenticate'}) RETURN caller.name, caller.filePath"
|
||||
},
|
||||
{
|
||||
question: "Show me all classes in the project",
|
||||
cypher: "MATCH (c:CodeElement {elementType: 'Class'}) RETURN c.name, c.filePath"
|
||||
},
|
||||
{
|
||||
question: "What classes inherit from BaseService?",
|
||||
cypher: "MATCH (child:CodeElement {elementType: 'Class'})-[r:CodeRelationship {relationshipType: 'INHERITS'}]->(parent:CodeElement {elementType: 'Class', name: 'BaseService'}) RETURN child.name, child.filePath"
|
||||
},
|
||||
{
|
||||
question: "Find all methods in the UserService class",
|
||||
cypher: "MATCH (c:CodeElement {elementType: 'Class', name: 'UserService'})-[r:CodeRelationship {relationshipType: 'CONTAINS'}]->(m:CodeElement {elementType: 'Method'}) RETURN m.name, m.startLine"
|
||||
},
|
||||
{
|
||||
question: "Which methods override the save method?",
|
||||
cypher: "MATCH (child:CodeElement {elementType: 'Method'})-[r:CodeRelationship {relationshipType: 'OVERRIDES'}]->(parent:CodeElement {elementType: 'Method', name: 'save'}) RETURN child.name, child.parentClass"
|
||||
},
|
||||
{
|
||||
question: "Show all interfaces and the classes that implement them",
|
||||
cypher: "MATCH (c:CodeElement {elementType: 'Class'})-[r:CodeRelationship {relationshipType: 'IMPLEMENTS'}]->(i:CodeElement {elementType: 'Interface'}) RETURN i.name, c.name"
|
||||
},
|
||||
{
|
||||
question: "Find functions decorated with @app.route",
|
||||
cypher: "MATCH (d:CodeElement {elementType: 'Decorator', name: 'app.route'})-[r:CodeRelationship {relationshipType: 'DECORATES'}]->(f:CodeElement {elementType: 'Function'}) RETURN f.name, f.filePath"
|
||||
},
|
||||
{
|
||||
question: "What files import the requests module?",
|
||||
cypher: "MATCH (f:CodeElement {elementType: 'File'})-[r:CodeRelationship {relationshipType: 'IMPORTS'}]->(target:CodeElement) WHERE target.name CONTAINS 'requests' RETURN f.name"
|
||||
},
|
||||
{
|
||||
question: "Show the call chain from main to database functions",
|
||||
cypher: "MATCH (main:Function {name: 'main'})-[:CALLS*1..3]->(db:Function) WHERE db.name CONTAINS 'db' OR db.name CONTAINS 'database' RETURN main.name, db.name"
|
||||
},
|
||||
{
|
||||
question: "Find all functions containing 'user' in their name",
|
||||
cypher: "MATCH (f:Function) WHERE f.name CONTAINS 'user' RETURN f.name, f.filePath"
|
||||
},
|
||||
{
|
||||
question: "What functions are called through a chain of 2-4 calls from the main function?",
|
||||
cypher: "MATCH (main:Function {name: 'main'})-[:CALLS*2..4]->(target:Function) RETURN main.name, target.name"
|
||||
},
|
||||
{
|
||||
question: "How many classes are in each file?",
|
||||
cypher: "MATCH (f:File)-[:CONTAINS]->(c:Class) RETURN f.name, COUNT(c)"
|
||||
},
|
||||
{
|
||||
question: "Count all functions in the project",
|
||||
cypher: "MATCH (f:Function) RETURN COUNT(f)"
|
||||
},
|
||||
{
|
||||
question: "List all function names in alphabetical order",
|
||||
cypher: "MATCH (f:Function) RETURN COLLECT(f.name)"
|
||||
},
|
||||
{
|
||||
question: "Find files that contain both classes and functions",
|
||||
cypher: "MATCH (f:File)-[:CONTAINS]->(c:Class) WHERE EXISTS((f)-[:CONTAINS]->(:Function)) RETURN f.name"
|
||||
},
|
||||
{
|
||||
question: "Show methods that start with 'get'",
|
||||
cypher: "MATCH (m:Method) WHERE m.name CONTAINS 'get' RETURN m.name, m.filePath"
|
||||
},
|
||||
{
|
||||
question: "Find all indirect dependencies (functions that call functions that call a target)",
|
||||
cypher: "MATCH (caller:Function)-[:CALLS*2..2]->(target:Function {name: 'database_query'}) RETURN caller.name, target.name"
|
||||
}
|
||||
];
|
||||
|
||||
constructor(llmService: LLMService) {
|
||||
this.llmService = llmService;
|
||||
this.outputParser = StructuredOutputParser.fromZodSchema(CypherQuerySchema);
|
||||
}
|
||||
|
||||
/**
|
||||
* Update the graph schema for better query generation
|
||||
*/
|
||||
public updateSchema(graph: KnowledgeGraph): void {
|
||||
this.graphSchema = this.generateSchemaDescription(graph);
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a Cypher query from natural language using structured output parsing
|
||||
*/
|
||||
public async generateQuery(
|
||||
question: string,
|
||||
llmConfig: LLMConfig,
|
||||
options: CypherGenerationOptions = {}
|
||||
): Promise<CypherQuery> {
|
||||
const { maxRetries = 2, includeExamples = true, strictMode = false } = options;
|
||||
|
||||
let lastError: string | null = null;
|
||||
|
||||
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
||||
try {
|
||||
// Get the format instructions for structured output
|
||||
const formatInstructions = this.outputParser.getFormatInstructions();
|
||||
|
||||
const systemPrompt = this.buildSystemPrompt(includeExamples, strictMode, lastError, formatInstructions);
|
||||
const userPrompt = this.buildUserPrompt(question);
|
||||
|
||||
const messages = [
|
||||
new SystemMessage(systemPrompt),
|
||||
new HumanMessage(userPrompt)
|
||||
];
|
||||
|
||||
// Get the model from the service
|
||||
const model = this.llmService.getModel(llmConfig);
|
||||
if (!model) {
|
||||
throw new Error('Failed to get LLM model');
|
||||
}
|
||||
|
||||
// Invoke the model directly first
|
||||
const response = await model.invoke(messages);
|
||||
|
||||
// Parse the response content
|
||||
let result;
|
||||
try {
|
||||
// Extract the content from the response
|
||||
const content = typeof response.content === 'string'
|
||||
? response.content
|
||||
: JSON.stringify(response.content);
|
||||
|
||||
// Try to parse the structured output
|
||||
result = await this.outputParser.parse(content);
|
||||
} catch (parseError) {
|
||||
// If parsing fails, try to extract the query manually
|
||||
console.warn('Failed to parse structured output, attempting manual extraction:', parseError);
|
||||
|
||||
const content = typeof response.content === 'string'
|
||||
? response.content
|
||||
: JSON.stringify(response.content);
|
||||
|
||||
// Try to extract Cypher query from the response
|
||||
const cypherMatch = content.match(/```(?:cypher|sql)?\n?(.*?)\n?```/s) ||
|
||||
content.match(/MATCH.*?(?:RETURN|$)/si);
|
||||
|
||||
const cypherQuery = cypherMatch ? cypherMatch[1] || cypherMatch[0] : content.trim();
|
||||
|
||||
result = {
|
||||
cypher: cypherQuery,
|
||||
explanation: 'Generated query from natural language',
|
||||
confidence: 0.7
|
||||
};
|
||||
}
|
||||
|
||||
// Validate the generated query
|
||||
const validation = this.validateQuery(result.cypher);
|
||||
if (!validation.isValid) {
|
||||
lastError = validation.error!;
|
||||
if (attempt < maxRetries) {
|
||||
console.warn(`Query validation failed (attempt ${attempt + 1}): ${validation.error}`);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
...result,
|
||||
warnings: validation.warnings
|
||||
};
|
||||
|
||||
} catch (error) {
|
||||
lastError = error instanceof Error ? error.message : 'Unknown error';
|
||||
if (attempt < maxRetries) {
|
||||
console.warn(`Query generation failed (attempt ${attempt + 1}): ${lastError}`);
|
||||
continue;
|
||||
}
|
||||
|
||||
throw new Error(`Failed to generate Cypher query after ${maxRetries + 1} attempts: ${lastError}`);
|
||||
}
|
||||
}
|
||||
|
||||
throw new Error('Unexpected error in query generation');
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the system prompt with schema and examples
|
||||
*/
|
||||
private buildSystemPrompt(
|
||||
includeExamples: boolean,
|
||||
strictMode: boolean,
|
||||
lastError?: string | null,
|
||||
formatInstructions?: string
|
||||
): string {
|
||||
let prompt = `You are a Cypher query expert for a code knowledge graph using KuzuDB (a high-performance graph database). Your task is to convert natural language questions into valid Cypher queries optimized for KuzuDB.
|
||||
|
||||
IMPORTANT: This codebase uses a POLYMORPHIC SCHEMA for optimal performance:
|
||||
|
||||
POLYMORPHIC SCHEMA:
|
||||
- All nodes are stored in a single CodeElement table with an 'elementType' discriminator
|
||||
- All relationships are stored in a single CodeRelationship table with a 'relationshipType' discriminator
|
||||
|
||||
GRAPH SCHEMA:
|
||||
${this.graphSchema}
|
||||
|
||||
NODE STRUCTURE:
|
||||
- Single node type: CodeElement
|
||||
- Discriminator property: elementType
|
||||
- Element types: 'Project', 'Folder', 'File', 'Module', 'Class', 'Function', 'Method', 'Variable', 'Interface', 'Type', 'Import'
|
||||
|
||||
RELATIONSHIP STRUCTURE:
|
||||
- Single relationship type: CodeRelationship
|
||||
- Discriminator property: relationshipType
|
||||
- Relationship types: 'CONTAINS', 'CALLS', 'INHERITS', 'IMPORTS', 'OVERRIDES', 'IMPLEMENTS', 'DECORATES', 'DEFINES', 'USES', 'ACCESSES', 'EXTENDS'
|
||||
|
||||
CRITICAL QUERY PATTERNS:
|
||||
- Nodes: MATCH (n:CodeElement {elementType: 'Function'})
|
||||
- Relationships: MATCH ()-[r:CodeRelationship {relationshipType: 'CALLS'}]->()
|
||||
- Combined: MATCH (f:CodeElement {elementType: 'File'})-[r:CodeRelationship {relationshipType: 'CONTAINS'}]->(func:CodeElement {elementType: 'Function'})
|
||||
|
||||
KUZUDB OPTIMIZATION GUIDELINES:
|
||||
|
||||
1. PERFORMANCE: KuzuDB excels at complex graph traversals and pattern matching
|
||||
- Use variable-length paths (*1..5) for call chains and dependency analysis
|
||||
- Leverage WHERE clauses for efficient filtering
|
||||
- Use aggregation functions (COUNT, COLLECT) for statistics
|
||||
|
||||
2. POLYMORPHIC QUERY PATTERNS:
|
||||
|
||||
SIMPLE MATCH: Find nodes by elementType and properties
|
||||
MATCH (f:CodeElement {elementType: 'Function', name: 'main'}) RETURN f
|
||||
|
||||
RELATIONSHIP TRAVERSAL: Follow relationships using relationshipType
|
||||
MATCH (caller:CodeElement)-[r:CodeRelationship {relationshipType: 'CALLS'}]->(target:CodeElement {elementType: 'Function'}) RETURN caller.name, target.name
|
||||
|
||||
VARIABLE-LENGTH PATHS: Find chains of relationships
|
||||
MATCH (start:CodeElement {elementType: 'Function'})-[r:CodeRelationship {relationshipType: 'CALLS'}*1..3]->(end:CodeElement {elementType: 'Function'}) RETURN start.name, end.name
|
||||
|
||||
AGGREGATION: Count and collect results
|
||||
MATCH (f:CodeElement {elementType: 'File'})-[r:CodeRelationship {relationshipType: 'CONTAINS'}]->(func:CodeElement {elementType: 'Function'}) RETURN f.name, COUNT(func)
|
||||
|
||||
PATTERN MATCHING: Use WHERE clauses for filtering
|
||||
MATCH (f:CodeElement {elementType: 'Function'}) WHERE f.name CONTAINS 'user' RETURN f.name, f.filePath
|
||||
|
||||
3. POLYMORPHIC COMMON PATTERNS:
|
||||
|
||||
FIND FUNCTIONS IN FILE:
|
||||
MATCH (f:CodeElement {elementType: 'File', name: 'filename.py'})-[r:CodeRelationship {relationshipType: 'CONTAINS'}]->(func:CodeElement {elementType: 'Function'}) RETURN func.name
|
||||
|
||||
FIND CALLERS OF FUNCTION:
|
||||
MATCH (caller:CodeElement)-[r:CodeRelationship {relationshipType: 'CALLS'}]->(target:CodeElement {elementType: 'Function', name: 'functionName'}) RETURN caller.name
|
||||
|
||||
FIND INHERITANCE CHAIN:
|
||||
MATCH (child:CodeElement {elementType: 'Class'})-[r:CodeRelationship {relationshipType: 'INHERITS'}*1..5]->(parent:CodeElement {elementType: 'Class'}) RETURN child.name, parent.name
|
||||
|
||||
FIND IMPORTS:
|
||||
MATCH (f:CodeElement {elementType: 'File'})-[r:CodeRelationship {relationshipType: 'IMPORTS'}]->(module:CodeElement) WHERE module.name CONTAINS 'requests' RETURN f.name
|
||||
|
||||
COUNT ENTITIES:
|
||||
MATCH (f:CodeElement {elementType: 'Function'}) RETURN COUNT(f) as functionCount
|
||||
|
||||
COMPLEX DEPENDENCY ANALYSIS:
|
||||
MATCH (start:CodeElement {elementType: 'Function'})-[r:CodeRelationship {relationshipType: 'CALLS'}*1..5]->(target:CodeElement {elementType: 'Function'})
|
||||
WHERE start.name = 'main' AND target.name CONTAINS 'db'
|
||||
RETURN start.name, target.name, LENGTH(shortestPath((start)-[r2:CodeRelationship {relationshipType: 'CALLS'}*]->(target))) as depth`;
|
||||
|
||||
if (includeExamples) {
|
||||
prompt += `\n\nEXAMPLE QUERIES:\n`;
|
||||
CypherGenerator.CYPHER_EXAMPLES.forEach((example, index) => {
|
||||
prompt += `${index + 1}. Question: "${example.question}"\n Cypher: ${example.cypher}\n\n`;
|
||||
});
|
||||
}
|
||||
|
||||
if (strictMode) {
|
||||
prompt += `\n\nSTRICT MODE: Only generate queries that exactly match the schema. Do not make assumptions about node properties that aren't explicitly defined.`;
|
||||
}
|
||||
|
||||
if (lastError) {
|
||||
prompt += `\n\nPREVIOUS ERROR: The last query attempt failed with: "${lastError}". Please fix this issue in your new query.`;
|
||||
}
|
||||
|
||||
if (formatInstructions) {
|
||||
prompt += `\n\n${formatInstructions}`;
|
||||
}
|
||||
|
||||
return prompt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the user prompt with the question
|
||||
*/
|
||||
private buildUserPrompt(question: string): string {
|
||||
return `Please convert this question to a Cypher query: "${question}"`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate the generated Cypher query
|
||||
*/
|
||||
private validateQuery(cypher: string): { isValid: boolean; error?: string; warnings?: string[] } {
|
||||
const warnings: string[] = [];
|
||||
|
||||
if (!cypher || cypher.trim().length === 0) {
|
||||
return { isValid: false, error: 'Empty query generated' };
|
||||
}
|
||||
|
||||
// Basic syntax checks
|
||||
const upperCypher = cypher.toUpperCase();
|
||||
|
||||
// Must have MATCH or CREATE or other valid starting keywords
|
||||
if (!upperCypher.match(/^\s*(MATCH|CREATE|MERGE|WITH|RETURN|CALL|SHOW)/)) {
|
||||
return { isValid: false, error: 'Query must start with a valid Cypher keyword (MATCH, CREATE, etc.)' };
|
||||
}
|
||||
|
||||
// Check for balanced parentheses
|
||||
const openParens = (cypher.match(/\(/g) || []).length;
|
||||
const closeParens = (cypher.match(/\)/g) || []).length;
|
||||
if (openParens !== closeParens) {
|
||||
return { isValid: false, error: 'Unbalanced parentheses in query' };
|
||||
}
|
||||
|
||||
// Check for balanced brackets
|
||||
const openBrackets = (cypher.match(/\[/g) || []).length;
|
||||
const closeBrackets = (cypher.match(/\]/g) || []).length;
|
||||
if (openBrackets !== closeBrackets) {
|
||||
return { isValid: false, error: 'Unbalanced brackets in query' };
|
||||
}
|
||||
|
||||
// Check for balanced braces
|
||||
const openBraces = (cypher.match(/\{/g) || []).length;
|
||||
const closeBraces = (cypher.match(/\}/g) || []).length;
|
||||
if (openBraces !== closeBraces) {
|
||||
return { isValid: false, error: 'Unbalanced braces in query' };
|
||||
}
|
||||
|
||||
// Warn about potentially expensive operations
|
||||
if (upperCypher.includes('MATCH ()') || upperCypher.includes('MATCH (*)')) {
|
||||
warnings.push('Query matches all nodes - this could be expensive');
|
||||
}
|
||||
|
||||
if (!upperCypher.includes('RETURN') && !upperCypher.includes('DELETE') && !upperCypher.includes('SET')) {
|
||||
warnings.push('Query does not return results');
|
||||
}
|
||||
|
||||
return { isValid: true, warnings };
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a schema description from the knowledge graph
|
||||
*/
|
||||
private generateSchemaDescription(graph: KnowledgeGraph): string {
|
||||
const nodeTypes = new Set<string>();
|
||||
const relationshipTypes = new Set<string>();
|
||||
const nodeProperties = new Map<string, Set<string>>();
|
||||
|
||||
// Analyze nodes
|
||||
graph.nodes.forEach(node => {
|
||||
nodeTypes.add(node.label);
|
||||
|
||||
if (!nodeProperties.has(node.label)) {
|
||||
nodeProperties.set(node.label, new Set());
|
||||
}
|
||||
|
||||
Object.keys(node.properties).forEach(prop => {
|
||||
nodeProperties.get(node.label)!.add(prop);
|
||||
});
|
||||
});
|
||||
|
||||
// Analyze relationships
|
||||
graph.relationships.forEach(rel => {
|
||||
relationshipTypes.add(rel.type);
|
||||
});
|
||||
|
||||
let schema = `NODES (${graph.nodes.length} total):\n`;
|
||||
for (const nodeType of Array.from(nodeTypes).sort()) {
|
||||
const props = nodeProperties.get(nodeType);
|
||||
const propList = props ? Array.from(props).sort().join(', ') : 'none';
|
||||
const count = graph.nodes.filter(n => n.label === nodeType).length;
|
||||
schema += `- ${nodeType} (${count}): ${propList}\n`;
|
||||
}
|
||||
|
||||
schema += `\nRELATIONSHIPS (${graph.relationships.length} total):\n`;
|
||||
for (const relType of Array.from(relationshipTypes).sort()) {
|
||||
const count = graph.relationships.filter(r => r.type === relType).length;
|
||||
schema += `- ${relType} (${count})\n`;
|
||||
}
|
||||
|
||||
return schema;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the current schema description
|
||||
*/
|
||||
public getSchema(): string {
|
||||
return this.graphSchema;
|
||||
}
|
||||
|
||||
/**
|
||||
* Clean and format a Cypher query
|
||||
*/
|
||||
public cleanQuery(cypher: string): string {
|
||||
return cypher
|
||||
.trim()
|
||||
.replace(/\s+/g, ' ')
|
||||
.replace(/\s*([(),[\]{}])\s*/g, '$1')
|
||||
.replace(/\s*([=<>!]+)\s*/g, ' $1 ')
|
||||
.replace(/\s+/g, ' ')
|
||||
.trim();
|
||||
}
|
||||
}
|
||||
@@ -1,7 +0,0 @@
|
||||
export * from './llm-service.js';
|
||||
export * from './cypher-generator.js';
|
||||
export * from './kuzu-rag-orchestrator.js';
|
||||
export * from './react-agent.js';
|
||||
|
||||
// Export KuzuDB performance prompts
|
||||
export * from './prompts/kuzu-performance-prompts.js';
|
||||
@@ -1,503 +0,0 @@
|
||||
import { HumanMessage, SystemMessage } from '@langchain/core/messages';
|
||||
import { z } from 'zod';
|
||||
import { tool } from '@langchain/core/tools';
|
||||
import type { LLMService, LLMConfig } from './llm-service.ts';
|
||||
import type { KuzuQueryEngine, KuzuQueryResult } from '../core/graph/kuzu-query-engine.ts';
|
||||
import type { KnowledgeGraph } from '../core/graph/types.ts';
|
||||
|
||||
import { isKuzuDBEnabled } from '../config/features.ts';
|
||||
|
||||
export interface KuzuRAGContext {
|
||||
graph: KnowledgeGraph;
|
||||
fileContents: Map<string, string>;
|
||||
projectName: string;
|
||||
}
|
||||
|
||||
export interface KuzuQueryResponse {
|
||||
nodes: any[];
|
||||
relationships: any[];
|
||||
executionTime: number;
|
||||
resultCount: number;
|
||||
}
|
||||
|
||||
export interface KuzuToolResult {
|
||||
toolName: string;
|
||||
input: string;
|
||||
output: string;
|
||||
success: boolean;
|
||||
executionTime?: number;
|
||||
resultCount?: number;
|
||||
error?: string;
|
||||
}
|
||||
|
||||
export interface KuzuRAGResult {
|
||||
answer: string;
|
||||
reasoning: any[];
|
||||
cypherQueries: any[];
|
||||
confidence: number;
|
||||
sources: string[];
|
||||
performance: {
|
||||
totalExecutionTime: number;
|
||||
queryExecutionTimes: number[];
|
||||
kuzuQueryCount: number;
|
||||
};
|
||||
}
|
||||
|
||||
// Define tool schemas using Zod
|
||||
const QueryGraphSchema = z.object({
|
||||
query: z.string().describe("The Cypher query to execute on the knowledge graph")
|
||||
});
|
||||
|
||||
const GetCodeSchema = z.object({
|
||||
filePath: z.string().describe("The file path to retrieve code from")
|
||||
});
|
||||
|
||||
const SearchFilesSchema = z.object({
|
||||
pattern: z.string().describe("The search pattern to find files")
|
||||
});
|
||||
|
||||
const FinalAnswerSchema = z.object({
|
||||
answer: z.string().describe("The final answer to the user's question")
|
||||
});
|
||||
|
||||
export class KuzuRAGOrchestrator {
|
||||
private llmService: LLMService;
|
||||
private kuzuQueryEngine: KuzuQueryEngine;
|
||||
private context: KuzuRAGContext | null = null;
|
||||
|
||||
constructor(
|
||||
llmService: LLMService,
|
||||
kuzuQueryEngine: KuzuQueryEngine
|
||||
) {
|
||||
this.llmService = llmService;
|
||||
this.kuzuQueryEngine = kuzuQueryEngine;
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize the orchestrator with KuzuDB
|
||||
*/
|
||||
async initialize(): Promise<void> {
|
||||
if (isKuzuDBEnabled()) {
|
||||
await this.kuzuQueryEngine.initialize();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Set the context for RAG operations
|
||||
*/
|
||||
public setContext(context: KuzuRAGContext): void {
|
||||
this.context = context;
|
||||
}
|
||||
|
||||
/**
|
||||
* Process a question using KuzuDB-enhanced RAG with tool calling
|
||||
*/
|
||||
public async processQuestion(
|
||||
question: string,
|
||||
llmConfig: LLMConfig,
|
||||
options: {
|
||||
maxReasoningSteps?: number;
|
||||
temperature?: number;
|
||||
strictMode?: boolean;
|
||||
useKuzuDB?: boolean;
|
||||
includeReasoning?: boolean;
|
||||
queryTimeout?: number;
|
||||
maxResults?: number;
|
||||
} = {}
|
||||
): Promise<KuzuRAGResult> {
|
||||
const {
|
||||
strictMode = false,
|
||||
useKuzuDB = true,
|
||||
includeReasoning = true,
|
||||
queryTimeout = 30000,
|
||||
maxResults = 100
|
||||
} = options;
|
||||
|
||||
if (!this.context) {
|
||||
throw new Error('Context not set. Call setContext() before processing questions.');
|
||||
}
|
||||
|
||||
// Define tools using LangChain's tool calling
|
||||
const queryGraphTool = tool(
|
||||
async ({ query }: { query: string }) => {
|
||||
if (useKuzuDB && this.kuzuQueryEngine.isReady()) {
|
||||
const kuzuResult = await this.kuzuQueryEngine.executeQuery(query, {
|
||||
timeout: queryTimeout,
|
||||
maxResults,
|
||||
includeExecutionTime: true
|
||||
});
|
||||
return this.formatKuzuQueryResult(kuzuResult);
|
||||
} else {
|
||||
const result = await this.executeGraphQuery();
|
||||
return JSON.stringify(result, null, 2);
|
||||
}
|
||||
},
|
||||
{
|
||||
name: "query_graph",
|
||||
description: "Execute Cypher queries on the knowledge graph for code analysis",
|
||||
schema: QueryGraphSchema
|
||||
}
|
||||
);
|
||||
|
||||
const getCodeTool = tool(
|
||||
async ({ filePath }: { filePath: string }) => {
|
||||
return await this.getCodeSnippet(filePath);
|
||||
},
|
||||
{
|
||||
name: "get_code",
|
||||
description: "Retrieve specific code snippets from files",
|
||||
schema: GetCodeSchema
|
||||
}
|
||||
);
|
||||
|
||||
const searchFilesTool = tool(
|
||||
async ({ pattern }: { pattern: string }) => {
|
||||
const result = await this.searchFiles(pattern);
|
||||
return JSON.stringify(result, null, 2);
|
||||
},
|
||||
{
|
||||
name: "search_files",
|
||||
description: "Find files by name or content patterns",
|
||||
schema: SearchFilesSchema
|
||||
}
|
||||
);
|
||||
|
||||
const finalAnswerTool = tool(
|
||||
async ({ answer }: { answer: string }) => {
|
||||
return answer;
|
||||
},
|
||||
{
|
||||
name: "final_answer",
|
||||
description: "Provide the final answer to the user's question",
|
||||
schema: FinalAnswerSchema
|
||||
}
|
||||
);
|
||||
|
||||
// Bind tools to the LLM
|
||||
const model = this.llmService.getModel(llmConfig);
|
||||
if (!model) {
|
||||
throw new Error('Failed to get LLM model');
|
||||
}
|
||||
|
||||
// Check if bindTools method exists
|
||||
if (typeof model.bindTools !== 'function') {
|
||||
throw new Error('LLM model does not support tool binding');
|
||||
}
|
||||
|
||||
const modelWithTools = model.bindTools([
|
||||
queryGraphTool,
|
||||
getCodeTool,
|
||||
searchFilesTool,
|
||||
finalAnswerTool
|
||||
]);
|
||||
|
||||
// Build system prompt
|
||||
const systemPrompt = this.buildKuzuReActSystemPrompt(strictMode, useKuzuDB);
|
||||
const conversation = [new SystemMessage(systemPrompt), new HumanMessage(question)];
|
||||
|
||||
// Execute with tool calling
|
||||
const response = await modelWithTools.invoke(conversation);
|
||||
|
||||
// Process tool calls
|
||||
const toolResults: KuzuToolResult[] = [];
|
||||
const cypherQueries: any[] = [];
|
||||
const reasoning: any[] = [];
|
||||
const sources: string[] = [];
|
||||
const queryExecutionTimes: number[] = [];
|
||||
let kuzuQueryCount = 0;
|
||||
|
||||
if (response.tool_calls && response.tool_calls.length > 0) {
|
||||
for (const toolCall of response.tool_calls) {
|
||||
const toolResult = await this.executeToolCall(toolCall, {
|
||||
useKuzuDB,
|
||||
queryTimeout,
|
||||
maxResults,
|
||||
strictMode
|
||||
});
|
||||
|
||||
if (toolResult) {
|
||||
toolResults.push(toolResult);
|
||||
|
||||
if (toolCall.name === 'query_graph') {
|
||||
cypherQueries.push({ cypher: toolCall.args.query, explanation: 'Generated via tool calling' });
|
||||
if (toolResult.executionTime) {
|
||||
queryExecutionTimes.push(toolResult.executionTime);
|
||||
kuzuQueryCount++;
|
||||
}
|
||||
}
|
||||
|
||||
if (toolResult.output) {
|
||||
sources.push(toolResult.output);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Convert response content to string
|
||||
const answer = typeof response.content === 'string'
|
||||
? response.content
|
||||
: Array.isArray(response.content)
|
||||
? response.content.map(item => typeof item === 'string' ? item : JSON.stringify(item)).join(' ')
|
||||
: JSON.stringify(response.content);
|
||||
|
||||
return {
|
||||
answer,
|
||||
reasoning: includeReasoning ? reasoning : [],
|
||||
cypherQueries,
|
||||
confidence: 0.8, // Could be calculated based on tool results
|
||||
sources,
|
||||
performance: {
|
||||
totalExecutionTime: 0, // Will be calculated
|
||||
queryExecutionTimes,
|
||||
kuzuQueryCount
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute a tool call and return the result
|
||||
*/
|
||||
private async executeToolCall(
|
||||
toolCall: any,
|
||||
options: {
|
||||
useKuzuDB: boolean;
|
||||
queryTimeout: number;
|
||||
maxResults: number;
|
||||
strictMode: boolean;
|
||||
}
|
||||
): Promise<KuzuToolResult | null> {
|
||||
try {
|
||||
switch (toolCall.name) {
|
||||
case 'query_graph':
|
||||
if (options.useKuzuDB && this.kuzuQueryEngine.isReady()) {
|
||||
const kuzuResult = await this.kuzuQueryEngine.executeQuery(toolCall.args.query, {
|
||||
timeout: options.queryTimeout,
|
||||
maxResults: options.maxResults,
|
||||
includeExecutionTime: true
|
||||
});
|
||||
|
||||
return {
|
||||
toolName: 'kuzu_query_graph',
|
||||
input: toolCall.args.query,
|
||||
output: this.formatKuzuQueryResult(kuzuResult),
|
||||
success: true,
|
||||
executionTime: kuzuResult.executionTime,
|
||||
resultCount: kuzuResult.resultCount
|
||||
};
|
||||
} else {
|
||||
const result = await this.executeGraphQuery();
|
||||
return {
|
||||
toolName: 'query_graph',
|
||||
input: toolCall.args.query,
|
||||
output: JSON.stringify(result, null, 2),
|
||||
success: true
|
||||
};
|
||||
}
|
||||
|
||||
case 'get_code':
|
||||
const codeResult = await this.getCodeSnippet(toolCall.args.filePath);
|
||||
return {
|
||||
toolName: 'get_code',
|
||||
input: toolCall.args.filePath,
|
||||
output: codeResult,
|
||||
success: true
|
||||
};
|
||||
|
||||
case 'search_files':
|
||||
const searchResult = await this.searchFiles(toolCall.args.pattern);
|
||||
return {
|
||||
toolName: 'search_files',
|
||||
input: toolCall.args.pattern,
|
||||
output: JSON.stringify(searchResult, null, 2),
|
||||
success: true
|
||||
};
|
||||
|
||||
case 'final_answer':
|
||||
return {
|
||||
toolName: 'final_answer',
|
||||
input: toolCall.args.answer,
|
||||
output: toolCall.args.answer,
|
||||
success: true
|
||||
};
|
||||
|
||||
default:
|
||||
return {
|
||||
toolName: 'unknown',
|
||||
input: JSON.stringify(toolCall.args),
|
||||
output: `Unknown tool: ${toolCall.name}`,
|
||||
success: false,
|
||||
error: `Unknown tool: ${toolCall.name}`
|
||||
};
|
||||
}
|
||||
} catch (error) {
|
||||
return {
|
||||
toolName: toolCall.name,
|
||||
input: JSON.stringify(toolCall.args),
|
||||
output: `Error executing tool: ${error instanceof Error ? error.message : 'Unknown error'}`,
|
||||
success: false,
|
||||
error: error instanceof Error ? error.message : 'Unknown error'
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build system prompt for KuzuDB-enhanced ReAct
|
||||
*/
|
||||
private buildKuzuReActSystemPrompt(strictMode: boolean, useKuzuDB: boolean): string {
|
||||
const basePrompt = `You are an AI assistant that helps analyze codebases using a knowledge graph powered by KuzuDB (a high-performance graph database) with a POLYMORPHIC SCHEMA. You have access to sophisticated graph querying capabilities for fast and accurate code analysis.
|
||||
|
||||
CRITICAL: This database uses a polymorphic schema for optimal performance:
|
||||
- All nodes: CodeElement with elementType discriminator ('Function', 'Class', 'Method', 'File', etc.)
|
||||
- All relationships: CodeRelationship with relationshipType discriminator ('CALLS', 'CONTAINS', 'IMPORTS', etc.)
|
||||
|
||||
Available tools:
|
||||
1. query_graph - Execute Cypher queries on the knowledge graph (${useKuzuDB ? 'using KuzuDB for enhanced performance' : 'using in-memory graph'})
|
||||
2. get_code - Retrieve specific code snippets
|
||||
3. search_files - Find files by name or content patterns
|
||||
4. final_answer - Provide the final answer
|
||||
|
||||
${useKuzuDB ? `
|
||||
KUZUDB CAPABILITIES:
|
||||
- High-performance graph queries with execution time tracking
|
||||
- Complex dependency analysis and call chain traversal
|
||||
- Persistent storage across sessions
|
||||
- Optimized for large-scale codebases with polymorphic schema
|
||||
- Real-time performance monitoring
|
||||
- Single-table operations for maximum speed
|
||||
|
||||
PERFORMANCE FEATURES:
|
||||
- Query execution time is automatically tracked
|
||||
- Results include performance metrics
|
||||
- Database operations are optimized for polymorphic queries
|
||||
- Support for complex graph traversals on unified tables` : 'Using in-memory graph for queries.'}
|
||||
|
||||
${strictMode ? 'STRICT MODE: Only use exact matches and precise queries.' : 'FLEXIBLE MODE: Use heuristic matching when exact matches fail.'}
|
||||
|
||||
POLYMORPHIC QUERY OPTIMIZATION GUIDELINES:
|
||||
- ALWAYS use CodeElement with elementType filters: MATCH (n:CodeElement {elementType: 'Function'})
|
||||
- ALWAYS use CodeRelationship with relationshipType filters: MATCH ()-[r:CodeRelationship {relationshipType: 'CALLS'}]->()
|
||||
- Leverage variable-length paths: -[r:CodeRelationship {relationshipType: 'CALLS'}*1..5]->
|
||||
- Use aggregation functions for statistics on CodeElement nodes
|
||||
- Prefer polymorphic patterns over traditional node types
|
||||
- Take advantage of unified table structure for complex traversals
|
||||
|
||||
Always follow this format:
|
||||
Thought: I need to think about what information I need
|
||||
Action: tool_name
|
||||
Action Input: the input to the tool
|
||||
Observation: the result of the action
|
||||
... (repeat if needed)
|
||||
Thought: I now have enough information to answer
|
||||
Action: final_answer
|
||||
Action Input: the final answer to the user's question
|
||||
|
||||
When using query_graph, focus on POLYMORPHIC queries:
|
||||
- Complex dependency analysis using CodeRelationship patterns
|
||||
- Call chain traversal with relationshipType filters
|
||||
- Pattern matching across CodeElement nodes with elementType
|
||||
- Statistical analysis using aggregation on polymorphic tables
|
||||
- Relationship exploration between CodeElement nodes with relationshipType filters
|
||||
|
||||
EXAMPLE POLYMORPHIC QUERIES:
|
||||
- Find functions: MATCH (f:CodeElement {elementType: 'Function'}) RETURN f.name
|
||||
- Find callers: MATCH (caller:CodeElement)-[r:CodeRelationship {relationshipType: 'CALLS'}]->(target:CodeElement {elementType: 'Function', name: 'myFunc'})
|
||||
- Call chains: MATCH (start:CodeElement {elementType: 'Function'})-[r:CodeRelationship {relationshipType: 'CALLS'}*1..3]->(end:CodeElement {elementType: 'Function'})`;
|
||||
|
||||
return basePrompt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format KuzuDB query result for observation
|
||||
*/
|
||||
private formatKuzuQueryResult(result: KuzuQueryResult): string {
|
||||
const nodes = result.nodes || [];
|
||||
const relationships = result.relationships || [];
|
||||
const summary = `Found ${nodes.length} nodes and ${relationships.length} relationships (execution time: ${result.executionTime.toFixed(2)}ms)`;
|
||||
|
||||
if (nodes.length === 0 && relationships.length === 0) {
|
||||
return `${summary}. No results found.`;
|
||||
}
|
||||
|
||||
const nodeSummary = nodes.length > 0
|
||||
? `\nNodes: ${nodes.slice(0, 5).map(n => `${n.label}:${n.properties.name || n.id}`).join(', ')}${nodes.length > 5 ? '...' : ''}`
|
||||
: '';
|
||||
|
||||
const relSummary = relationships.length > 0
|
||||
? `\nRelationships: ${relationships.slice(0, 5).map(r => `${r.type}:${r.source}->${r.target}`).join(', ')}${relationships.length > 5 ? '...' : ''}`
|
||||
: '';
|
||||
|
||||
return `${summary}${nodeSummary}${relSummary}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute graph query (fallback to in-memory)
|
||||
*/
|
||||
private async executeGraphQuery(): Promise<any> {
|
||||
if (!this.context) {
|
||||
throw new Error('Context not set');
|
||||
}
|
||||
|
||||
// Simple in-memory query execution as fallback
|
||||
// This would be replaced with actual graph query logic
|
||||
return { results: [], count: 0 };
|
||||
}
|
||||
|
||||
/**
|
||||
* Get code snippet
|
||||
*/
|
||||
private async getCodeSnippet(filePath: string): Promise<string> {
|
||||
if (!this.context) {
|
||||
throw new Error('Context not set');
|
||||
}
|
||||
|
||||
const content = this.context.fileContents.get(filePath);
|
||||
return content || 'File not found';
|
||||
}
|
||||
|
||||
/**
|
||||
* Search files
|
||||
*/
|
||||
private async searchFiles(pattern: string): Promise<string[]> {
|
||||
if (!this.context) {
|
||||
throw new Error('Context not set');
|
||||
}
|
||||
|
||||
const matchingFiles: string[] = [];
|
||||
for (const [filePath, content] of this.context.fileContents.entries()) {
|
||||
if (filePath.includes(pattern) || content.includes(pattern)) {
|
||||
matchingFiles.push(filePath);
|
||||
}
|
||||
}
|
||||
|
||||
return matchingFiles.slice(0, 10); // Limit results
|
||||
}
|
||||
|
||||
/**
|
||||
* Get performance statistics
|
||||
*/
|
||||
async getPerformanceStats(): Promise<any> {
|
||||
if (!this.kuzuQueryEngine.isReady()) {
|
||||
return { status: 'KuzuDB not initialized' };
|
||||
}
|
||||
|
||||
// Get basic database stats
|
||||
const dbStats = {
|
||||
status: 'ready',
|
||||
initialized: this.kuzuQueryEngine.isReady(),
|
||||
timestamp: new Date().toISOString()
|
||||
};
|
||||
|
||||
return {
|
||||
kuzuDBStatus: 'ready',
|
||||
databaseStats: dbStats,
|
||||
queryEngineReady: this.kuzuQueryEngine.isReady()
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Close the orchestrator
|
||||
*/
|
||||
async close(): Promise<void> {
|
||||
await this.kuzuQueryEngine.close();
|
||||
}
|
||||
}
|
||||
@@ -1,328 +0,0 @@
|
||||
import { ChatOpenAI } from '@langchain/openai';
|
||||
import { AzureChatOpenAI } from '@langchain/openai';
|
||||
import { ChatAnthropic } from '@langchain/anthropic';
|
||||
import { ChatGoogleGenerativeAI } from '@langchain/google-genai';
|
||||
import type { BaseMessage } from '@langchain/core/messages';
|
||||
import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
|
||||
import { HumanMessage } from '@langchain/core/messages';
|
||||
|
||||
export type LLMProvider = 'openai' | 'azure-openai' | 'anthropic' | 'gemini';
|
||||
|
||||
export interface LLMConfig {
|
||||
provider: LLMProvider;
|
||||
apiKey: string;
|
||||
model?: string;
|
||||
temperature?: number;
|
||||
maxTokens?: number;
|
||||
maxRetries?: number;
|
||||
// Azure OpenAI specific fields
|
||||
azureOpenAIEndpoint?: string;
|
||||
azureOpenAIApiVersion?: string;
|
||||
azureOpenAIDeploymentName?: string;
|
||||
}
|
||||
|
||||
export interface ChatResponse {
|
||||
content: string;
|
||||
usage?: {
|
||||
promptTokens: number;
|
||||
completionTokens: number;
|
||||
totalTokens: number;
|
||||
};
|
||||
model?: string;
|
||||
finishReason?: string;
|
||||
}
|
||||
|
||||
export class LLMService {
|
||||
private models: Map<string, BaseChatModel> = new Map();
|
||||
private defaultConfig: Partial<LLMConfig> = {
|
||||
temperature: 0.1,
|
||||
maxTokens: 4000,
|
||||
maxRetries: 3,
|
||||
azureOpenAIApiVersion: '2024-02-01' // Default Azure OpenAI API version
|
||||
};
|
||||
|
||||
// Default models for each provider
|
||||
private static readonly DEFAULT_MODELS: Record<LLMProvider, string> = {
|
||||
openai: 'gpt-4o-mini',
|
||||
'azure-openai': 'gpt-4o-mini',
|
||||
anthropic: 'claude-3-haiku-20240307',
|
||||
gemini: 'gemini-2.5-flash' // Use the latest 2.5 Flash model as default (2025)
|
||||
};
|
||||
|
||||
constructor() {}
|
||||
|
||||
/**
|
||||
* Get the chat model instance for tool binding
|
||||
*/
|
||||
public getModel(config: LLMConfig): BaseChatModel {
|
||||
return this.getChatModel(config);
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize or get a chat model for the specified provider
|
||||
*/
|
||||
public getChatModel(config: LLMConfig): any {
|
||||
const cacheKey = this.getCacheKey(config);
|
||||
|
||||
if (this.models.has(cacheKey)) {
|
||||
return this.models.get(cacheKey)!;
|
||||
}
|
||||
|
||||
const model = this.createChatModel(config);
|
||||
this.models.set(cacheKey, model);
|
||||
return model;
|
||||
}
|
||||
|
||||
/**
|
||||
* Send a chat message and get a response
|
||||
*/
|
||||
public async chat(
|
||||
config: LLMConfig,
|
||||
messages: BaseMessage[],
|
||||
options?: { stream?: boolean }
|
||||
): Promise<ChatResponse> {
|
||||
try {
|
||||
const model = this.getChatModel(config);
|
||||
|
||||
if (options?.stream) {
|
||||
// For streaming, we'd need to handle this differently
|
||||
// For now, we'll just use regular invoke
|
||||
console.warn('Streaming not implemented yet, falling back to regular invoke');
|
||||
}
|
||||
|
||||
const response = await model.invoke(messages);
|
||||
|
||||
return {
|
||||
content: response.content as string,
|
||||
usage: response.response_metadata?.usage ? {
|
||||
promptTokens: response.response_metadata.usage.prompt_tokens || 0,
|
||||
completionTokens: response.response_metadata.usage.completion_tokens || 0,
|
||||
totalTokens: response.response_metadata.usage.total_tokens || 0
|
||||
} : undefined,
|
||||
model: config.model || LLMService.DEFAULT_MODELS[config.provider],
|
||||
finishReason: response.response_metadata?.finish_reason
|
||||
};
|
||||
} catch (error) {
|
||||
throw new Error(`LLM chat failed: ${error instanceof Error ? error.message : 'Unknown error'}`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate API key format for different providers
|
||||
*/
|
||||
public validateApiKey(provider: LLMProvider, apiKey: string): boolean {
|
||||
if (!apiKey || apiKey.trim().length === 0) {
|
||||
return false;
|
||||
}
|
||||
|
||||
switch (provider) {
|
||||
case 'openai':
|
||||
return apiKey.startsWith('sk-') && apiKey.length > 20;
|
||||
case 'azure-openai':
|
||||
// Azure OpenAI keys are typically 32 characters long and don't have a specific prefix
|
||||
return apiKey.length >= 20; // More flexible validation for Azure keys
|
||||
case 'anthropic':
|
||||
return apiKey.startsWith('sk-ant-') && apiKey.length > 20;
|
||||
case 'gemini':
|
||||
return apiKey.length > 20; // Google API keys don't have a consistent prefix
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate Azure OpenAI configuration
|
||||
*/
|
||||
public validateAzureOpenAIConfig(config: LLMConfig): { valid: boolean; error?: string } {
|
||||
if (config.provider !== 'azure-openai') {
|
||||
return { valid: false, error: 'Provider must be azure-openai' };
|
||||
}
|
||||
|
||||
if (!config.azureOpenAIEndpoint) {
|
||||
return { valid: false, error: 'Azure OpenAI endpoint is required' };
|
||||
}
|
||||
|
||||
if (!config.azureOpenAIEndpoint.includes('openai.azure.com')) {
|
||||
return { valid: false, error: 'Invalid Azure OpenAI endpoint format. Should contain "openai.azure.com"' };
|
||||
}
|
||||
|
||||
if (!config.azureOpenAIDeploymentName) {
|
||||
return { valid: false, error: 'Azure OpenAI deployment name is required' };
|
||||
}
|
||||
|
||||
return { valid: true };
|
||||
}
|
||||
|
||||
/**
|
||||
* Get available models for a provider
|
||||
*/
|
||||
public getAvailableModels(provider: LLMProvider): string[] {
|
||||
switch (provider) {
|
||||
case 'openai':
|
||||
return [
|
||||
'gpt-4o',
|
||||
'gpt-4o-mini',
|
||||
'gpt-4-turbo',
|
||||
'gpt-4',
|
||||
'gpt-3.5-turbo'
|
||||
];
|
||||
case 'azure-openai':
|
||||
return [
|
||||
'gpt-4o',
|
||||
'gpt-4o-mini',
|
||||
'gpt-4.1-mini-v2', // Common deployment name
|
||||
'gpt-4-turbo',
|
||||
'gpt-4',
|
||||
'gpt-35-turbo', // Note: Azure uses gpt-35-turbo instead of gpt-3.5-turbo
|
||||
'gpt-4-32k'
|
||||
];
|
||||
case 'anthropic':
|
||||
return [
|
||||
'claude-3-5-sonnet-20241022',
|
||||
'claude-3-5-haiku-20241022',
|
||||
'claude-3-opus-20240229',
|
||||
'claude-3-sonnet-20240229',
|
||||
'claude-3-haiku-20240307'
|
||||
];
|
||||
case 'gemini':
|
||||
return [
|
||||
'gemini-2.5-flash', // Latest and fastest (2025) - NEW DEFAULT
|
||||
'gemini-2.5-pro', // Latest pro model (2025) - PREMIUM
|
||||
'gemini-1.5-flash', // Most stable and widely available
|
||||
'gemini-1.5-pro', // Stable pro model
|
||||
'gemini-1.0-pro', // Legacy but very stable
|
||||
'gemini-1.5-flash-8b', // Smaller, efficient version
|
||||
'gemini-2.0-flash', // Newer model (may not be available to all users)
|
||||
'gemini-2.0-flash-lite' // Lightweight version
|
||||
];
|
||||
default:
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get provider display name
|
||||
*/
|
||||
public getProviderDisplayName(provider: LLMProvider): string {
|
||||
switch (provider) {
|
||||
case 'openai':
|
||||
return 'OpenAI';
|
||||
case 'azure-openai':
|
||||
return 'Azure OpenAI';
|
||||
case 'anthropic':
|
||||
return 'Anthropic';
|
||||
case 'gemini':
|
||||
return 'Google Gemini';
|
||||
default:
|
||||
return provider;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Test connection with the provider
|
||||
*/
|
||||
public async testConnection(config: LLMConfig): Promise<{ success: boolean; error?: string }> {
|
||||
try {
|
||||
// Validate Azure OpenAI config if needed
|
||||
if (config.provider === 'azure-openai') {
|
||||
const validation = this.validateAzureOpenAIConfig(config);
|
||||
if (!validation.valid) {
|
||||
return {
|
||||
success: false,
|
||||
error: validation.error
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
const model = this.createChatModel(config);
|
||||
|
||||
// Send a simple test message
|
||||
const testMessages = [
|
||||
new HumanMessage("Test connection")
|
||||
];
|
||||
|
||||
await model.invoke(testMessages);
|
||||
return { success: true };
|
||||
} catch (error) {
|
||||
return {
|
||||
success: false,
|
||||
error: error instanceof Error ? error.message : 'Connection test failed'
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Clear cached models (useful for updating API keys)
|
||||
*/
|
||||
public clearCache(): void {
|
||||
this.models.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a chat model instance based on the provider
|
||||
*/
|
||||
private createChatModel(config: LLMConfig): any {
|
||||
const mergedConfig = { ...this.defaultConfig, ...config };
|
||||
const model = mergedConfig.model || LLMService.DEFAULT_MODELS[config.provider];
|
||||
|
||||
switch (config.provider) {
|
||||
case 'openai':
|
||||
return new ChatOpenAI({
|
||||
apiKey: config.apiKey,
|
||||
model,
|
||||
temperature: mergedConfig.temperature,
|
||||
maxTokens: mergedConfig.maxTokens,
|
||||
maxRetries: mergedConfig.maxRetries,
|
||||
timeout: 30000
|
||||
});
|
||||
|
||||
case 'azure-openai':
|
||||
return new AzureChatOpenAI({
|
||||
azureOpenAIApiKey: config.apiKey,
|
||||
model: config.azureOpenAIDeploymentName, // Use deployment name as model
|
||||
temperature: mergedConfig.temperature,
|
||||
maxTokens: mergedConfig.maxTokens,
|
||||
maxRetries: mergedConfig.maxRetries,
|
||||
timeout: 30000,
|
||||
azureOpenAIApiInstanceName: config.azureOpenAIEndpoint?.replace('https://', '').split('.')[0],
|
||||
azureOpenAIApiVersion: config.azureOpenAIApiVersion,
|
||||
azureOpenAIApiDeploymentName: config.azureOpenAIDeploymentName
|
||||
});
|
||||
|
||||
case 'anthropic':
|
||||
return new ChatAnthropic({
|
||||
model: config.model || 'claude-3-sonnet-20240229',
|
||||
anthropicApiKey: config.apiKey,
|
||||
maxTokens: config.maxTokens || 4096,
|
||||
temperature: config.temperature || 0.7
|
||||
});
|
||||
|
||||
case 'gemini':
|
||||
return new ChatGoogleGenerativeAI({
|
||||
apiKey: config.apiKey,
|
||||
model,
|
||||
temperature: mergedConfig.temperature,
|
||||
maxOutputTokens: mergedConfig.maxTokens,
|
||||
maxRetries: mergedConfig.maxRetries
|
||||
});
|
||||
|
||||
default:
|
||||
throw new Error(`Unsupported provider: ${config.provider}`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate cache key for model instances
|
||||
*/
|
||||
private getCacheKey(config: LLMConfig): string {
|
||||
const model = config.model || LLMService.DEFAULT_MODELS[config.provider];
|
||||
let baseKey = `${config.provider}:${model}:${config.apiKey.slice(-8)}:${config.temperature}:${config.maxTokens}`;
|
||||
|
||||
// Add Azure OpenAI specific fields to cache key
|
||||
if (config.provider === 'azure-openai') {
|
||||
baseKey += `:${config.azureOpenAIEndpoint}:${config.azureOpenAIDeploymentName}:${config.azureOpenAIApiVersion}`;
|
||||
}
|
||||
|
||||
return baseKey;
|
||||
}
|
||||
}
|
||||
@@ -1,170 +0,0 @@
|
||||
/**
|
||||
* KuzuDB Performance-Optimized Prompts
|
||||
* Specialized prompts for leveraging KuzuDB's strengths
|
||||
*/
|
||||
|
||||
export const KUZU_PERFORMANCE_PROMPTS = {
|
||||
/**
|
||||
* System prompt for performance-focused queries
|
||||
*/
|
||||
PERFORMANCE_SYSTEM: `You are a Cypher query expert specializing in high-performance graph database queries using KuzuDB with a POLYMORPHIC SCHEMA. Your goal is to generate optimized queries that leverage KuzuDB's strengths.
|
||||
|
||||
CRITICAL: This database uses a polymorphic schema:
|
||||
- All nodes: CodeElement with elementType discriminator
|
||||
- All relationships: CodeRelationship with relationshipType discriminator
|
||||
|
||||
KUZUDB OPTIMIZATION PRINCIPLES:
|
||||
1. POLYMORPHIC QUERIES: Always use elementType and relationshipType filters
|
||||
2. COMPLEX TRAVERSALS: KuzuDB excels at variable-length path queries
|
||||
3. PATTERN MATCHING: Use sophisticated WHERE clauses for filtering
|
||||
4. AGGREGATION: Leverage COUNT, COLLECT, and other aggregation functions
|
||||
5. INDEXING: Prefer queries that can use elementType indexes
|
||||
6. BATCHING: Structure queries to minimize round trips
|
||||
|
||||
POLYMORPHIC PERFORMANCE PATTERNS:
|
||||
- Use (start:CodeElement {elementType: 'Function'})-[r:CodeRelationship {relationshipType: 'CALLS'}*1..5]->(end:CodeElement {elementType: 'Function'}) for dependency chains
|
||||
- Leverage WHERE clauses with CONTAINS for text search on CodeElement properties
|
||||
- Use aggregation for statistics: COUNT, COLLECT, AVG on CodeElement nodes
|
||||
- Always specify elementType for better index utilization
|
||||
- Use LIMIT clauses to control result size
|
||||
|
||||
OPTIMIZED QUERY TYPES:
|
||||
1. Dependency Analysis: MATCH (target:CodeElement {elementType: 'Function'})<-[r:CodeRelationship {relationshipType: 'CALLS'}*1..5]-(caller:CodeElement)
|
||||
2. Call Chain Traversal: MATCH (start:CodeElement {elementType: 'Function'})-[r:CodeRelationship {relationshipType: 'CALLS'}*]->(end:CodeElement {elementType: 'Function'})
|
||||
3. Pattern Matching: MATCH (n:CodeElement) WHERE n.elementType IN ['Function', 'Method'] AND n.name CONTAINS 'pattern'
|
||||
4. Statistical Analysis: MATCH (f:CodeElement {elementType: 'File'})-[r:CodeRelationship {relationshipType: 'CONTAINS'}]->(func:CodeElement {elementType: 'Function'}) RETURN f.name, COUNT(func)
|
||||
5. Relationship Exploration: MATCH (a:CodeElement)-[r:CodeRelationship]->(b:CodeElement) WHERE r.relationshipType IN ['CALLS', 'IMPORTS']
|
||||
|
||||
Always use polymorphic patterns and consider execution time and result relevance when generating queries.`,
|
||||
|
||||
/**
|
||||
* Prompt for complex dependency analysis
|
||||
*/
|
||||
DEPENDENCY_ANALYSIS: `Generate a Cypher query for dependency analysis that leverages KuzuDB's strength in variable-length path traversal.
|
||||
|
||||
Focus on:
|
||||
- Finding all dependencies of a specific function/class
|
||||
- Identifying call chains and dependency trees
|
||||
- Discovering indirect dependencies (2+ hops away)
|
||||
- Analyzing dependency depth and complexity
|
||||
|
||||
Use patterns like:
|
||||
- (start)-[:CALLS*1..5]->(end) for call chains
|
||||
- (start)-[:IMPORTS*1..3]->(end) for import dependencies
|
||||
- WHERE clauses to filter by specific criteria
|
||||
- Aggregation to summarize dependency statistics`,
|
||||
|
||||
/**
|
||||
* Prompt for performance monitoring queries
|
||||
*/
|
||||
PERFORMANCE_MONITORING: `Generate Cypher queries for monitoring and analyzing codebase performance metrics.
|
||||
|
||||
Focus on:
|
||||
- Counting entities by type (functions, classes, methods)
|
||||
- Analyzing code complexity through relationship density
|
||||
- Identifying performance bottlenecks in call chains
|
||||
- Measuring code coupling and cohesion
|
||||
|
||||
Use aggregation functions:
|
||||
- COUNT() for entity counting
|
||||
- COLLECT() for gathering lists
|
||||
- AVG() for average metrics
|
||||
- MAX()/MIN() for range analysis
|
||||
|
||||
Structure queries to provide actionable performance insights.`,
|
||||
|
||||
/**
|
||||
* Prompt for code pattern discovery
|
||||
*/
|
||||
PATTERN_DISCOVERY: `Generate Cypher queries for discovering code patterns and architectural insights.
|
||||
|
||||
Focus on:
|
||||
- Finding similar code structures
|
||||
- Identifying design patterns
|
||||
- Discovering architectural relationships
|
||||
- Analyzing code organization
|
||||
|
||||
Use patterns like:
|
||||
- Pattern matching with WHERE clauses
|
||||
- Relationship traversal for structural analysis
|
||||
- Aggregation for pattern frequency
|
||||
- Variable-length paths for complex relationships
|
||||
|
||||
Aim to reveal hidden patterns and architectural insights.`
|
||||
};
|
||||
|
||||
/**
|
||||
* Performance-focused query examples
|
||||
*/
|
||||
export const PERFORMANCE_QUERY_EXAMPLES = [
|
||||
{
|
||||
question: "Find all functions that are called through a chain of 3-5 function calls from the main function",
|
||||
cypher: "MATCH (main:Function {name: 'main'})-[:CALLS*3..5]->(target:Function) RETURN main.name, target.name, target.filePath",
|
||||
explanation: "Uses variable-length path to find functions 3-5 calls away from main"
|
||||
},
|
||||
{
|
||||
question: "Count how many functions each class contains and show the most complex classes",
|
||||
cypher: "MATCH (c:Class)-[:CONTAINS]->(f:Function) RETURN c.name, COUNT(f) as functionCount ORDER BY functionCount DESC LIMIT 10",
|
||||
explanation: "Uses aggregation to count functions per class and orders by complexity"
|
||||
},
|
||||
{
|
||||
question: "Find all functions that are called by more than 5 other functions",
|
||||
cypher: "MATCH (caller:Function)-[:CALLS]->(target:Function) WITH target, COUNT(caller) as callCount WHERE callCount > 5 RETURN target.name, callCount ORDER BY callCount DESC",
|
||||
explanation: "Uses aggregation to find frequently called functions"
|
||||
},
|
||||
{
|
||||
question: "Show the dependency chain from authentication functions to database functions",
|
||||
cypher: "MATCH (auth:Function)-[:CALLS*1..5]->(db:Function) WHERE auth.name CONTAINS 'auth' AND db.name CONTAINS 'db' RETURN auth.name, db.name",
|
||||
explanation: "Uses variable-length path to trace authentication to database calls"
|
||||
},
|
||||
{
|
||||
question: "Find all classes that implement more than 2 interfaces",
|
||||
cypher: "MATCH (c:Class)-[:IMPLEMENTS]->(i:Interface) WITH c, COUNT(i) as interfaceCount WHERE interfaceCount > 2 RETURN c.name, interfaceCount",
|
||||
explanation: "Uses aggregation to find classes with multiple interface implementations"
|
||||
}
|
||||
];
|
||||
|
||||
/**
|
||||
* Performance monitoring query templates
|
||||
*/
|
||||
export const PERFORMANCE_TEMPLATES = {
|
||||
// Code complexity analysis
|
||||
COMPLEXITY_ANALYSIS: `
|
||||
MATCH (f:Function)
|
||||
OPTIONAL MATCH (f)-[:CALLS]->(called:Function)
|
||||
WITH f, COUNT(called) as outgoingCalls
|
||||
OPTIONAL MATCH (caller:Function)-[:CALLS]->(f)
|
||||
WITH f, outgoingCalls, COUNT(caller) as incomingCalls
|
||||
RETURN f.name, f.filePath, outgoingCalls, incomingCalls, (outgoingCalls + incomingCalls) as totalComplexity
|
||||
ORDER BY totalComplexity DESC
|
||||
LIMIT 20
|
||||
`,
|
||||
|
||||
// Dependency depth analysis
|
||||
DEPENDENCY_DEPTH: `
|
||||
MATCH (start:Function {name: $functionName})-[:CALLS*1..10]->(target:Function)
|
||||
WITH target, LENGTH(shortestPath((start)-[:CALLS*]->(target))) as depth
|
||||
RETURN target.name, target.filePath, depth
|
||||
ORDER BY depth
|
||||
`,
|
||||
|
||||
// Code coupling analysis
|
||||
COUPLING_ANALYSIS: `
|
||||
MATCH (f1:Function)-[:CALLS]->(f2:Function)
|
||||
WHERE f1.filePath <> f2.filePath
|
||||
WITH f1.filePath as file1, f2.filePath as file2, COUNT(*) as coupling
|
||||
WHERE coupling > 5
|
||||
RETURN file1, file2, coupling
|
||||
ORDER BY coupling DESC
|
||||
`,
|
||||
|
||||
// Architecture pattern detection
|
||||
PATTERN_DETECTION: `
|
||||
MATCH (c:Class)-[:CONTAINS]->(m:Method)
|
||||
WHERE m.name CONTAINS 'get' OR m.name CONTAINS 'set'
|
||||
WITH c, COUNT(m) as accessorCount
|
||||
WHERE accessorCount > 3
|
||||
RETURN c.name, c.filePath, accessorCount
|
||||
ORDER BY accessorCount DESC
|
||||
`
|
||||
};
|
||||
@@ -1,47 +0,0 @@
|
||||
import { HumanMessage, SystemMessage, AIMessage, BaseMessage } from '@langchain/core/messages';
|
||||
import { z } from 'zod';
|
||||
import type { LLMService, LLMConfig } from './llm-service.ts';
|
||||
import type { CypherGenerator } from './cypher-generator.ts';
|
||||
import type { KnowledgeGraph } from '../core/graph/types.ts';
|
||||
import type { LocalStorageChatHistory } from '../lib/chat-history.ts';
|
||||
|
||||
// Define Zod schema for ReAct step
|
||||
const ReActStepSchema = z.object({
|
||||
thought: z.string().describe("The reasoning process - what you're thinking about"),
|
||||
action: z.enum(['query_graph', 'get_code', 'search_files', 'final_answer']).describe("The action to take - must be one of: query_graph, get_code, search_files, or final_answer"),
|
||||
actionInput: z.string().describe("Input for the action - the query, file path, search pattern, or final answer")
|
||||
});
|
||||
|
||||
export async function debugStructuredOutput(llmService: LLMService, llmConfig: LLMConfig) {
|
||||
console.log('=== DEBUGGING STRUCTURED OUTPUT ===');
|
||||
|
||||
try {
|
||||
const model = llmService.getModel(llmConfig);
|
||||
console.log('Model type:', model.constructor.name);
|
||||
console.log('Model supports withStructuredOutput:', typeof model.withStructuredOutput === 'function');
|
||||
|
||||
if (typeof model.withStructuredOutput === 'function') {
|
||||
console.log('Attempting to create structured model...');
|
||||
const structuredModel = model.withStructuredOutput(ReActStepSchema);
|
||||
console.log('Structured model created successfully');
|
||||
|
||||
// Test with a simple prompt
|
||||
const testMessages = [
|
||||
new SystemMessage('You are a helpful assistant. Respond with a structured output.'),
|
||||
new HumanMessage('Think about searching for files and respond with the appropriate action.')
|
||||
];
|
||||
|
||||
console.log('Testing structured output...');
|
||||
const response = await structuredModel.invoke(testMessages);
|
||||
console.log('Structured output response:', response);
|
||||
|
||||
return { success: true, response };
|
||||
} else {
|
||||
console.log('Model does not support structured output');
|
||||
return { success: false, reason: 'Model does not support withStructuredOutput' };
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('Error in structured output:', error);
|
||||
return { success: false, error: error instanceof Error ? error.message : 'Unknown error' };
|
||||
}
|
||||
}
|
||||
@@ -1,175 +0,0 @@
|
||||
import { ReActAgent } from './react-agent';
|
||||
import { KuzuQueryEngine } from '../core/graph/kuzu-query-engine';
|
||||
|
||||
// Mock implementations
|
||||
const mockLLMService = {
|
||||
chat: async () => ({
|
||||
content: `Thought: I need to query the graph to find information about functions
|
||||
Action: query_graph
|
||||
Action Input: Find all functions in the codebase`,
|
||||
usage: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }
|
||||
}),
|
||||
getChatModel: () => ({
|
||||
invoke: async () => ({ content: 'Mock response' })
|
||||
}),
|
||||
getModel: function() {
|
||||
return this.getChatModel();
|
||||
}
|
||||
};
|
||||
|
||||
const mockCypherGenerator = {
|
||||
generateQuery: async () => ({
|
||||
cypher: 'MATCH (f:Function) RETURN f.name, f.filePath LIMIT 10',
|
||||
explanation: 'Query to find all functions in the codebase',
|
||||
confidence: 0.9
|
||||
}),
|
||||
updateSchema: () => {}
|
||||
};
|
||||
|
||||
// Mock KuzuQueryEngine
|
||||
const mockKuzuQueryEngine = {
|
||||
initialize: jest.fn().mockResolvedValue(undefined),
|
||||
importGraph: jest.fn().mockResolvedValue(undefined),
|
||||
executeQuery: jest.fn().mockResolvedValue({
|
||||
nodes: [
|
||||
{ id: 'func1', label: 'Function', properties: { name: 'testFunction', filePath: '/test.ts' } }
|
||||
],
|
||||
relationships: [],
|
||||
resultCount: 1,
|
||||
executionTime: 5.2
|
||||
}),
|
||||
isReady: jest.fn().mockReturnValue(true)
|
||||
};
|
||||
|
||||
describe('ReActAgent - KuzuDB Integration', () => {
|
||||
let reactAgent: ReActAgent;
|
||||
|
||||
beforeEach(() => {
|
||||
reactAgent = new ReActAgent(mockLLMService as any, mockCypherGenerator as any, mockKuzuQueryEngine);
|
||||
});
|
||||
|
||||
test('should initialize KuzuQueryEngine', async () => {
|
||||
await reactAgent.initialize();
|
||||
|
||||
expect(mockKuzuQueryEngine.initialize).toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test('should import graph data into KuzuDB', async () => {
|
||||
const mockGraph = {
|
||||
nodes: [
|
||||
{ id: 'func1', label: 'Function', properties: { name: 'testFunction' } }
|
||||
],
|
||||
relationships: []
|
||||
};
|
||||
|
||||
const mockFileContents = new Map<string, string>();
|
||||
mockFileContents.set('test.ts', 'console.log("test");');
|
||||
|
||||
await reactAgent.setContext({
|
||||
graph: mockGraph,
|
||||
fileContents: mockFileContents
|
||||
}, {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
expect(mockKuzuQueryEngine.importGraph).toHaveBeenCalledWith(mockGraph);
|
||||
});
|
||||
|
||||
test('should execute Cypher queries using KuzuDB', async () => {
|
||||
// Set up context
|
||||
const mockGraph = {
|
||||
nodes: [
|
||||
{ id: 'func1', label: 'Function', properties: { name: 'testFunction' } }
|
||||
],
|
||||
relationships: []
|
||||
};
|
||||
|
||||
const mockFileContents = new Map<string, string>();
|
||||
mockFileContents.set('test.ts', 'console.log("test");');
|
||||
|
||||
await reactAgent.setContext({
|
||||
graph: mockGraph,
|
||||
fileContents: mockFileContents
|
||||
}, {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
// Process a question that should trigger a graph query
|
||||
const result = await reactAgent.processQuestion('Find all functions in the codebase', {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
// Verify that KuzuDB was used for query execution
|
||||
expect(mockKuzuQueryEngine.executeQuery).toHaveBeenCalledWith(
|
||||
'MATCH (f:Function) RETURN f.name, f.filePath LIMIT 10',
|
||||
{ includeExecutionTime: true }
|
||||
);
|
||||
|
||||
// Verify that the result contains the expected data
|
||||
expect(result.cypherQueries).toHaveLength(1);
|
||||
expect(result.cypherQueries[0].cypher).toBe('MATCH (f:Function) RETURN f.name, f.filePath LIMIT 10');
|
||||
expect(result.cypherQueries[0].confidence).toBe(0.9);
|
||||
});
|
||||
|
||||
test('should handle KuzuDB query failures gracefully', async () => {
|
||||
// Mock KuzuDB to throw an error
|
||||
const mockKuzuQueryEngineWithError = {
|
||||
...mockKuzuQueryEngine,
|
||||
executeQuery: jest.fn().mockRejectedValue(new Error('KuzuDB connection failed'))
|
||||
};
|
||||
|
||||
const testAgent = new ReActAgent(mockLLMService as any, mockCypherGenerator as any, mockKuzuQueryEngineWithError);
|
||||
|
||||
// Set up context
|
||||
const mockGraph = { nodes: [], relationships: [] };
|
||||
const mockFileContents = new Map<string, string>();
|
||||
mockFileContents.set('test.ts', 'console.log("test");');
|
||||
|
||||
await testAgent.setContext({
|
||||
graph: mockGraph,
|
||||
fileContents: mockFileContents
|
||||
}, {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
// Should handle the error gracefully
|
||||
const result = await testAgent.processQuestion('Find all functions', {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
expect(result.answer).toBeDefined();
|
||||
expect(result.answer).not.toContain('Unknown action');
|
||||
});
|
||||
|
||||
test('should fallback to placeholder when KuzuDB is not available', async () => {
|
||||
// Create agent without KuzuDB
|
||||
const testAgent = new ReActAgent(mockLLMService as any, mockCypherGenerator as any);
|
||||
|
||||
// Set up context
|
||||
const mockGraph = { nodes: [], relationships: [] };
|
||||
const mockFileContents = new Map<string, string>();
|
||||
mockFileContents.set('test.ts', 'console.log("test");');
|
||||
|
||||
await testAgent.setContext({
|
||||
graph: mockGraph,
|
||||
fileContents: mockFileContents
|
||||
}, {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
// Should still work with placeholder
|
||||
const result = await testAgent.processQuestion('Find all functions', {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
expect(result.answer).toBeDefined();
|
||||
expect(result.answer).not.toContain('Unknown action');
|
||||
});
|
||||
});
|
||||
@@ -1,139 +0,0 @@
|
||||
import { ReActAgent } from './react-agent';
|
||||
|
||||
// Simple mock implementations
|
||||
const mockLLMService = {
|
||||
chat: async () => ({
|
||||
content: `Thought: I need to search for files related to the user's question
|
||||
Action: search_files
|
||||
Action Input: react agent`,
|
||||
usage: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }
|
||||
}),
|
||||
getChatModel: () => ({
|
||||
invoke: async () => ({ content: 'Mock response' })
|
||||
}),
|
||||
getModel: function() {
|
||||
return this.getChatModel();
|
||||
}
|
||||
};
|
||||
|
||||
const mockCypherGenerator = {
|
||||
generateQuery: async () => ({
|
||||
cypher: 'MATCH (n) RETURN n LIMIT 10',
|
||||
explanation: 'Mock query for testing',
|
||||
confidence: 0.8
|
||||
}),
|
||||
updateSchema: () => {}
|
||||
};
|
||||
|
||||
describe('ReActAgent - Simple Tests', () => {
|
||||
let reactAgent: ReActAgent;
|
||||
|
||||
beforeEach(() => {
|
||||
reactAgent = new ReActAgent(mockLLMService as any, mockCypherGenerator as any);
|
||||
});
|
||||
|
||||
it('should initialize correctly', () => {
|
||||
expect(reactAgent).toBeInstanceOf(ReActAgent);
|
||||
});
|
||||
|
||||
it('should set context correctly', () => {
|
||||
const mockGraph = {
|
||||
nodes: [],
|
||||
relationships: []
|
||||
};
|
||||
|
||||
const mockFileContents = new Map<string, string>();
|
||||
mockFileContents.set('test.ts', 'console.log("test");');
|
||||
|
||||
expect(async () => {
|
||||
await reactAgent.setContext({
|
||||
graph: mockGraph,
|
||||
fileContents: mockFileContents
|
||||
}, {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
}).not.toThrow();
|
||||
});
|
||||
|
||||
it('should throw error when processing question without context', async () => {
|
||||
const llmConfig = {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
};
|
||||
|
||||
await expect(
|
||||
reactAgent.processQuestion('test question', llmConfig)
|
||||
).rejects.toThrow('Context not set');
|
||||
});
|
||||
|
||||
it('should process question with context', async () => {
|
||||
const mockGraph = {
|
||||
nodes: [],
|
||||
relationships: []
|
||||
};
|
||||
|
||||
const mockFileContents = new Map<string, string>();
|
||||
mockFileContents.set('test.ts', 'console.log("test");');
|
||||
|
||||
await reactAgent.setContext({
|
||||
graph: mockGraph,
|
||||
fileContents: mockFileContents
|
||||
}, {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
const llmConfig = {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
};
|
||||
|
||||
const result = await reactAgent.processQuestion('test question', llmConfig);
|
||||
|
||||
expect(result).toHaveProperty('answer');
|
||||
expect(result).toHaveProperty('reasoning');
|
||||
expect(result).toHaveProperty('confidence');
|
||||
expect(result).toHaveProperty('sources');
|
||||
expect(Array.isArray(result.reasoning)).toBe(true);
|
||||
expect(Array.isArray(result.sources)).toBe(true);
|
||||
expect(typeof result.confidence).toBe('number');
|
||||
});
|
||||
|
||||
it('should handle ReAct reasoning steps correctly', async () => {
|
||||
const mockGraph = {
|
||||
nodes: [],
|
||||
relationships: []
|
||||
};
|
||||
|
||||
const mockFileContents = new Map<string, string>();
|
||||
mockFileContents.set('test.ts', 'console.log("test");');
|
||||
|
||||
await reactAgent.setContext({
|
||||
graph: mockGraph,
|
||||
fileContents: mockFileContents
|
||||
}, {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
const llmConfig = {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
};
|
||||
|
||||
const result = await reactAgent.processQuestion('test question', llmConfig);
|
||||
|
||||
// Should have at least one reasoning step
|
||||
expect(result.reasoning.length).toBeGreaterThan(0);
|
||||
|
||||
// Check structure of reasoning steps
|
||||
const firstStep = result.reasoning[0];
|
||||
expect(firstStep).toHaveProperty('step');
|
||||
expect(firstStep).toHaveProperty('thought');
|
||||
expect(firstStep).toHaveProperty('action');
|
||||
expect(typeof firstStep.step).toBe('number');
|
||||
expect(typeof firstStep.thought).toBe('string');
|
||||
expect(typeof firstStep.action).toBe('string');
|
||||
});
|
||||
});
|
||||
@@ -1,131 +0,0 @@
|
||||
import { ReActAgent } from './react-agent';
|
||||
|
||||
// Mock implementations
|
||||
const mockLLMService = {
|
||||
chat: async () => ({
|
||||
content: `Thought: I need to help the user with their question
|
||||
Action: unknown_action
|
||||
Action Input: some input`,
|
||||
usage: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }
|
||||
}),
|
||||
getChatModel: () => ({
|
||||
invoke: async () => ({ content: 'Mock response' })
|
||||
}),
|
||||
getModel: function() {
|
||||
return this.getChatModel();
|
||||
}
|
||||
};
|
||||
|
||||
const mockCypherGenerator = {
|
||||
generateQuery: async () => ({
|
||||
cypher: 'MATCH (n) RETURN n LIMIT 10',
|
||||
explanation: 'Mock query for testing',
|
||||
confidence: 0.8
|
||||
}),
|
||||
updateSchema: () => {}
|
||||
};
|
||||
|
||||
describe('ReActAgent - Unknown Action Fixes', () => {
|
||||
let reactAgent: ReActAgent;
|
||||
|
||||
beforeEach(() => {
|
||||
reactAgent = new ReActAgent(mockLLMService as any, mockCypherGenerator as any);
|
||||
});
|
||||
|
||||
test('should handle unknown actions gracefully', async () => {
|
||||
// Set up context
|
||||
const mockGraph = {
|
||||
nodes: [],
|
||||
relationships: []
|
||||
};
|
||||
|
||||
const mockFileContents = new Map<string, string>();
|
||||
mockFileContents.set('test.ts', 'console.log("test");');
|
||||
|
||||
await reactAgent.setContext({
|
||||
graph: mockGraph,
|
||||
fileContents: mockFileContents
|
||||
}, {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
// Test with an unknown action
|
||||
const result = await reactAgent.processQuestion('test question', {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
// Should not throw an error and should provide a helpful response
|
||||
expect(result.answer).toBeDefined();
|
||||
expect(result.answer).not.toContain('Unknown action');
|
||||
expect(result.answer).not.toContain('Error');
|
||||
});
|
||||
|
||||
test('should normalize action variations', () => {
|
||||
const testCases = [
|
||||
{ input: 'querygraph', expected: 'query_graph' },
|
||||
{ input: 'getcode', expected: 'get_code' },
|
||||
{ input: 'searchfiles', expected: 'search_files' },
|
||||
{ input: 'finalanswer', expected: 'final_answer' },
|
||||
{ input: 'query', expected: 'query_graph' },
|
||||
{ input: 'search', expected: 'search_files' },
|
||||
{ input: 'answer', expected: 'final_answer' }
|
||||
];
|
||||
|
||||
testCases.forEach(({ input, expected }) => {
|
||||
// This tests the action normalization logic
|
||||
const normalized = input.toLowerCase().replace(/[^a-z_]/g, '');
|
||||
const actionMap: Record<string, string> = {
|
||||
'querygraph': 'query_graph',
|
||||
'query': 'query_graph',
|
||||
'getcode': 'get_code',
|
||||
'searchfiles': 'search_files',
|
||||
'search': 'search_files',
|
||||
'finalanswer': 'final_answer',
|
||||
'answer': 'final_answer'
|
||||
};
|
||||
|
||||
const result = actionMap[normalized] || normalized;
|
||||
expect(result).toBe(expected);
|
||||
});
|
||||
});
|
||||
|
||||
test('should provide helpful responses for unexpected actions', async () => {
|
||||
// Mock the LLM to return an unexpected action
|
||||
const mockLLMWithUnexpectedAction = {
|
||||
...mockLLMService,
|
||||
chat: async () => ({
|
||||
content: `Thought: I need to help the user
|
||||
Action: unexpected_action
|
||||
Action Input: help me find something`,
|
||||
usage: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }
|
||||
})
|
||||
};
|
||||
|
||||
const testAgent = new ReActAgent(mockLLMWithUnexpectedAction as any, mockCypherGenerator as any);
|
||||
|
||||
// Set up context
|
||||
const mockGraph = { nodes: [], relationships: [] };
|
||||
const mockFileContents = new Map<string, string>();
|
||||
mockFileContents.set('test.ts', 'console.log("test");');
|
||||
|
||||
await testAgent.setContext({
|
||||
graph: mockGraph,
|
||||
fileContents: mockFileContents
|
||||
}, {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
const result = await testAgent.processQuestion('help me find something', {
|
||||
provider: 'openai' as const,
|
||||
apiKey: 'test-key'
|
||||
});
|
||||
|
||||
// Should provide a helpful response instead of an error
|
||||
expect(result.answer).toBeDefined();
|
||||
expect(result.answer).not.toContain('Unknown action');
|
||||
expect(result.answer).toContain('help you');
|
||||
});
|
||||
});
|
||||
@@ -1,90 +0,0 @@
|
||||
import { ReActAgent } from './react-agent';
|
||||
import { LLMService } from './llm-service';
|
||||
import { CypherGenerator } from './cypher-generator';
|
||||
|
||||
// Mock the LLM service for testing
|
||||
jest.mock('./llm-service');
|
||||
jest.mock('./cypher-generator');
|
||||
|
||||
describe('ReActAgent Structured Output', () => {
|
||||
let reactAgent: ReActAgent;
|
||||
let mockLLMService: jest.Mocked<LLMService>;
|
||||
let mockCypherGenerator: jest.Mocked<CypherGenerator>;
|
||||
|
||||
beforeEach(() => {
|
||||
mockLLMService = new LLMService() as jest.Mocked<LLMService>;
|
||||
mockCypherGenerator = new CypherGenerator(mockLLMService) as jest.Mocked<CypherGenerator>;
|
||||
reactAgent = new ReActAgent(mockLLMService, mockCypherGenerator);
|
||||
});
|
||||
|
||||
test('should use structured output when available', async () => {
|
||||
// Mock the model to support structured output
|
||||
const mockModel = {
|
||||
withStructuredOutput: jest.fn().mockReturnValue({
|
||||
invoke: jest.fn().mockResolvedValue({
|
||||
thought: 'I need to search for files',
|
||||
action: 'search_files',
|
||||
actionInput: 'test'
|
||||
})
|
||||
})
|
||||
};
|
||||
|
||||
mockLLMService.getModel = jest.fn().mockReturnValue(mockModel);
|
||||
|
||||
// Mock context
|
||||
const mockContext = {
|
||||
graph: { nodes: [], relationships: [] },
|
||||
fileContents: new Map([['test.ts', 'console.log("test")']])
|
||||
};
|
||||
|
||||
await reactAgent.setContext(mockContext, {
|
||||
provider: 'openai',
|
||||
apiKey: 'test-key',
|
||||
model: 'gpt-4o-mini'
|
||||
});
|
||||
|
||||
// Test the process
|
||||
const result = await reactAgent.processQuestion('Find test files', {
|
||||
provider: 'openai',
|
||||
apiKey: 'test-key',
|
||||
model: 'gpt-4o-mini'
|
||||
});
|
||||
|
||||
expect(mockModel.withStructuredOutput).toHaveBeenCalled();
|
||||
expect(result.reasoning.length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
test('should fallback to regex parsing when structured output fails', async () => {
|
||||
// Mock the model to not support structured output
|
||||
const mockModel = {
|
||||
withStructuredOutput: undefined
|
||||
};
|
||||
|
||||
mockLLMService.getModel = jest.fn().mockReturnValue(mockModel);
|
||||
mockLLMService.chat = jest.fn().mockResolvedValue({
|
||||
content: 'Thought: I need to search\nAction: search_files\nAction Input: test'
|
||||
});
|
||||
|
||||
// Mock context
|
||||
const mockContext = {
|
||||
graph: { nodes: [], relationships: [] },
|
||||
fileContents: new Map([['test.ts', 'console.log("test")']])
|
||||
};
|
||||
|
||||
await reactAgent.setContext(mockContext, {
|
||||
provider: 'openai',
|
||||
apiKey: 'test-key',
|
||||
model: 'gpt-4o-mini'
|
||||
});
|
||||
|
||||
// Test the process
|
||||
const result = await reactAgent.processQuestion('Find test files', {
|
||||
provider: 'openai',
|
||||
apiKey: 'test-key',
|
||||
model: 'gpt-4o-mini'
|
||||
});
|
||||
|
||||
expect(mockLLMService.chat).toHaveBeenCalled();
|
||||
expect(result.reasoning.length).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
@@ -1,580 +0,0 @@
|
||||
import { HumanMessage, SystemMessage, AIMessage, BaseMessage } from '@langchain/core/messages';
|
||||
import { z } from 'zod';
|
||||
import type { LLMService, LLMConfig } from './llm-service.ts';
|
||||
import type { CypherGenerator } from './cypher-generator.ts';
|
||||
import type { KnowledgeGraph } from '../core/graph/types.ts';
|
||||
import type { LocalStorageChatHistory } from '../lib/chat-history.ts';
|
||||
import { configLoader } from '../config/config-loader.ts';
|
||||
|
||||
export interface ReActContext {
|
||||
graph: KnowledgeGraph;
|
||||
fileContents: Map<string, string>;
|
||||
projectName?: string;
|
||||
sessionId?: string;
|
||||
}
|
||||
|
||||
export interface ReActToolResult {
|
||||
toolName: string;
|
||||
input: string;
|
||||
output: string;
|
||||
success: boolean;
|
||||
error?: string;
|
||||
}
|
||||
|
||||
export interface ReActStep {
|
||||
step: number;
|
||||
thought: string;
|
||||
action: string;
|
||||
actionInput?: string;
|
||||
observation?: string;
|
||||
toolResult?: ReActToolResult;
|
||||
}
|
||||
|
||||
export interface ReActResult {
|
||||
answer: string;
|
||||
reasoning: ReActStep[];
|
||||
confidence: number;
|
||||
sources: string[];
|
||||
cypherQueries: Array<{
|
||||
cypher: string;
|
||||
explanation: string;
|
||||
confidence: number;
|
||||
}>;
|
||||
}
|
||||
|
||||
export interface ReActOptions {
|
||||
maxIterations?: number;
|
||||
temperature?: number;
|
||||
strictMode?: boolean;
|
||||
includeReasoning?: boolean;
|
||||
enableQueryCaching?: boolean;
|
||||
similarityThreshold?: number;
|
||||
}
|
||||
|
||||
// Define Zod schema for ReAct step
|
||||
const ReActStepSchema = z.object({
|
||||
thought: z.string().describe("The reasoning process - what you're thinking about"),
|
||||
action: z.enum(['query_graph', 'get_code', 'search_files', 'final_answer']).describe("The action to take - must be one of: query_graph, get_code, search_files, or final_answer"),
|
||||
actionInput: z.string().describe("Input for the action - the query, file path, search pattern, or final answer")
|
||||
});
|
||||
|
||||
export class ReActAgent {
|
||||
private llmService: LLMService;
|
||||
private cypherGenerator: CypherGenerator;
|
||||
private context: ReActContext | null = null;
|
||||
private chatHistory: LocalStorageChatHistory | null = null;
|
||||
private graph?: KnowledgeGraph;
|
||||
|
||||
constructor(llmService: LLMService, cypherGenerator: CypherGenerator, graph?: KnowledgeGraph) {
|
||||
this.llmService = llmService;
|
||||
this.cypherGenerator = cypherGenerator;
|
||||
this.graph = graph; // Store graph reference for KuzuDB access
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize the ReAct agent
|
||||
*/
|
||||
public async initialize(): Promise<void> {
|
||||
console.log('ReActAgent initialized');
|
||||
}
|
||||
|
||||
/**
|
||||
* Set the context for ReAct operations
|
||||
*/
|
||||
public async setContext(context: ReActContext & { projectName?: string; sessionId?: string }, _llmConfig: LLMConfig): Promise<void> {
|
||||
this.context = {
|
||||
graph: context.graph,
|
||||
fileContents: context.fileContents,
|
||||
projectName: context.projectName,
|
||||
sessionId: context.sessionId
|
||||
};
|
||||
|
||||
// Update graph reference for KuzuDB access
|
||||
this.graph = context.graph;
|
||||
|
||||
this.cypherGenerator.updateSchema(context.graph);
|
||||
|
||||
// Test KuzuDB connectivity
|
||||
await this.testKuzuDBConnection();
|
||||
}
|
||||
|
||||
/**
|
||||
* Test KuzuDB connection and log results
|
||||
*/
|
||||
private async testKuzuDBConnection(): Promise<void> {
|
||||
try {
|
||||
console.log('🧪 Testing KuzuDB connection...');
|
||||
const testResult = await this.executeGraphQuery('MATCH (n) RETURN COUNT(n) as nodeCount LIMIT 1');
|
||||
|
||||
if (testResult.success && testResult.source === 'KuzuDB') {
|
||||
console.log('✅ KuzuDB connection test successful!');
|
||||
console.log(`📊 Total nodes in KuzuDB: ${testResult.rows[0]?.[0] || 'unknown'}`);
|
||||
} else {
|
||||
console.log('⚠️ KuzuDB connection test failed, using fallback');
|
||||
}
|
||||
} catch (error) {
|
||||
console.log('❌ KuzuDB connection test error:', error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Set chat history for conversation context
|
||||
*/
|
||||
public setChatHistory(chatHistory: LocalStorageChatHistory): void {
|
||||
this.chatHistory = chatHistory;
|
||||
}
|
||||
|
||||
/**
|
||||
* Process a question using ReAct pattern with chat history
|
||||
*/
|
||||
public async processQuestion(
|
||||
question: string,
|
||||
llmConfig: LLMConfig,
|
||||
options: ReActOptions = {}
|
||||
): Promise<ReActResult> {
|
||||
if (!this.context) {
|
||||
throw new Error('Context not set. Call setContext() first.');
|
||||
}
|
||||
|
||||
const {
|
||||
maxIterations = 5,
|
||||
temperature = 0.1,
|
||||
strictMode = false,
|
||||
includeReasoning = true
|
||||
} = options;
|
||||
|
||||
const reasoning: ReActStep[] = [];
|
||||
const sources: string[] = [];
|
||||
const cypherQueries: Array<{ cypher: string; explanation: string; confidence: number }> = [];
|
||||
|
||||
// Enhanced LLM config for reasoning
|
||||
const reasoningConfig: LLMConfig = {
|
||||
...llmConfig,
|
||||
temperature: temperature
|
||||
};
|
||||
|
||||
let currentStep = 1;
|
||||
let finalAnswer = '';
|
||||
let confidence = 0.5;
|
||||
|
||||
try {
|
||||
// Build conversation with chat history
|
||||
const conversation: BaseMessage[] = [];
|
||||
|
||||
// Add system prompt
|
||||
const systemPrompt = this.buildReActSystemPrompt(strictMode);
|
||||
conversation.push(new SystemMessage(systemPrompt));
|
||||
|
||||
// Add chat history if available
|
||||
if (this.chatHistory) {
|
||||
try {
|
||||
const historyMessages = await this.chatHistory.getMessages();
|
||||
// Add recent history (last 10 messages to avoid context overflow)
|
||||
const recentHistory = historyMessages.slice(-10);
|
||||
conversation.push(...recentHistory);
|
||||
} catch (error) {
|
||||
console.warn('Failed to load chat history:', error);
|
||||
}
|
||||
}
|
||||
|
||||
// Add the current user question
|
||||
conversation.push(new HumanMessage(`Question: ${question}`));
|
||||
|
||||
while (currentStep <= maxIterations) {
|
||||
let reasoning_step: ReActStep;
|
||||
|
||||
try {
|
||||
// Try using structured output first
|
||||
const model = this.llmService.getModel(reasoningConfig);
|
||||
console.log('Attempting structured output with model:', model.constructor.name);
|
||||
|
||||
if (model && typeof model.withStructuredOutput === 'function') {
|
||||
console.log('Model supports structured output, attempting to use it...');
|
||||
const structuredModel = model.withStructuredOutput(ReActStepSchema);
|
||||
const structuredResponse = await structuredModel.invoke(conversation);
|
||||
|
||||
console.log('Structured output successful:', structuredResponse);
|
||||
|
||||
// Validate the structured response
|
||||
const validActions = ['query_graph', 'get_code', 'search_files', 'final_answer'];
|
||||
if (!validActions.includes(structuredResponse.action)) {
|
||||
console.warn(`Invalid action from structured output: ${structuredResponse.action}, falling back to regex parsing`);
|
||||
const response = await this.llmService.chat(reasoningConfig, conversation);
|
||||
reasoning_step = this.parseReasoningStep(String(response.content || ''), currentStep);
|
||||
} else {
|
||||
reasoning_step = {
|
||||
step: currentStep,
|
||||
thought: structuredResponse.thought,
|
||||
action: structuredResponse.action,
|
||||
actionInput: structuredResponse.actionInput
|
||||
};
|
||||
}
|
||||
} else {
|
||||
console.warn('Model does not support structured output, falling back to regex parsing');
|
||||
// Fallback to regular chat + regex parsing
|
||||
const response = await this.llmService.chat(reasoningConfig, conversation);
|
||||
reasoning_step = this.parseReasoningStep(String(response.content || ''), currentStep);
|
||||
}
|
||||
} catch (error) {
|
||||
// Fallback to regex parsing if structured output fails
|
||||
console.warn('Structured output failed, falling back to regex parsing:', error);
|
||||
const response = await this.llmService.chat(reasoningConfig, conversation);
|
||||
reasoning_step = this.parseReasoningStep(String(response.content || ''), currentStep);
|
||||
}
|
||||
|
||||
reasoning.push(reasoning_step);
|
||||
|
||||
// Check if we have a final answer
|
||||
if (reasoning_step.action === 'final_answer') {
|
||||
finalAnswer = reasoning_step.actionInput || '';
|
||||
confidence = Math.min(0.9, confidence + 0.2);
|
||||
break;
|
||||
}
|
||||
|
||||
// Execute the action
|
||||
const toolResult = await this.executeAction(reasoning_step.action, reasoning_step.actionInput || '', llmConfig);
|
||||
reasoning_step.toolResult = toolResult;
|
||||
reasoning_step.observation = toolResult.output;
|
||||
|
||||
// Track Cypher queries
|
||||
if (reasoning_step.action === 'query_graph' && toolResult.success) {
|
||||
cypherQueries.push({
|
||||
cypher: reasoning_step.actionInput || '',
|
||||
explanation: 'Generated via ReAct reasoning',
|
||||
confidence: confidence
|
||||
});
|
||||
}
|
||||
|
||||
// Add sources if successful
|
||||
if (toolResult.success && toolResult.output) {
|
||||
sources.push(`${reasoning_step.action}: ${reasoning_step.actionInput}`);
|
||||
}
|
||||
|
||||
// Add the tool result to conversation
|
||||
conversation.push(new AIMessage(`Thought: ${reasoning_step.thought}\nAction: ${reasoning_step.action}\nAction Input: ${reasoning_step.actionInput}`));
|
||||
conversation.push(new HumanMessage(`Observation: ${toolResult.output}`));
|
||||
|
||||
currentStep++;
|
||||
}
|
||||
|
||||
// If we didn't get a final answer, generate one based on the reasoning
|
||||
if (!finalAnswer && reasoning.length > 0) {
|
||||
const summaryPrompt = this.buildSummaryPrompt(question, reasoning);
|
||||
conversation.push(new HumanMessage(summaryPrompt));
|
||||
|
||||
const summaryResponse = await this.llmService.chat(reasoningConfig, conversation);
|
||||
finalAnswer = String(summaryResponse.content || '');
|
||||
confidence = Math.max(0.3, confidence - 0.2); // Lower confidence for incomplete reasoning
|
||||
}
|
||||
|
||||
return {
|
||||
answer: finalAnswer || 'I was unable to find a complete answer to your question.',
|
||||
reasoning: includeReasoning ? reasoning : [],
|
||||
confidence,
|
||||
sources: Array.from(new Set(sources)), // Remove duplicates
|
||||
cypherQueries
|
||||
};
|
||||
|
||||
} catch (error) {
|
||||
throw new Error(`ReAct processing failed: ${error instanceof Error ? error.message : 'Unknown error'}`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the ReAct system prompt with chat history context
|
||||
*/
|
||||
private buildReActSystemPrompt(strictMode: boolean): string {
|
||||
const prompt = `You are an expert code analyst using a ReAct (Reasoning + Acting) approach to answer questions about a codebase.
|
||||
|
||||
You have access to the following tools:
|
||||
1. query_graph: Query the code knowledge graph using natural language
|
||||
2. get_code: Retrieve the source code content of a specific file
|
||||
3. search_files: Search for files matching a pattern or containing specific text
|
||||
4. final_answer: Provide the final answer to the user's question
|
||||
|
||||
IMPORTANT INSTRUCTIONS:
|
||||
- You have access to the conversation history above, so you can reference previous questions and answers
|
||||
- Think step by step and provide your reasoning in the "thought" field
|
||||
- Choose the appropriate action from the available tools (ONLY: query_graph, get_code, search_files, or final_answer)
|
||||
- Provide the necessary input for the chosen action
|
||||
- Use the tools to gather information before providing final answers
|
||||
- Be precise and thorough in your analysis
|
||||
- Cite specific files and code snippets when possible
|
||||
- If the user refers to something from the conversation history, use that context
|
||||
|
||||
${strictMode ? 'STRICT MODE: Only use exact matches and precise queries.' : 'FLEXIBLE MODE: Use heuristic matching when exact matches fail.'}
|
||||
|
||||
You must respond with a structured output containing:
|
||||
- thought: Your reasoning process for this step
|
||||
- action: The tool you want to use (MUST be one of: query_graph, get_code, search_files, or final_answer)
|
||||
- actionInput: The input for the chosen tool
|
||||
|
||||
When providing a final_answer, make sure to give a complete, comprehensive response in the actionInput field.
|
||||
|
||||
CRITICAL: The action field must be exactly one of these four values: query_graph, get_code, search_files, or final_answer.`;
|
||||
|
||||
return prompt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a reasoning step from LLM response
|
||||
*/
|
||||
private parseReasoningStep(response: string, stepNumber: number): ReActStep {
|
||||
// Normalize the response to handle different line endings and whitespace
|
||||
const normalizedResponse = response.replace(/\r\n/g, '\n').trim();
|
||||
|
||||
// More robust regex patterns that handle various formats
|
||||
const thoughtMatch = normalizedResponse.match(/Thought:\s*(.*?)(?=\nAction:|$)/);
|
||||
const actionMatch = normalizedResponse.match(/Action:\s*(.*?)(?=\nAction Input:|$)/);
|
||||
const actionInputMatch = normalizedResponse.match(/Action Input:\s*([\s\S]*?)(?=\nThought:|$)/);
|
||||
|
||||
let action = actionMatch ? actionMatch[1].trim() : '';
|
||||
|
||||
// Normalize action names to handle variations
|
||||
if (action) {
|
||||
action = action.toLowerCase().replace(/[^a-z_]/g, '');
|
||||
|
||||
// Map common variations to expected actions
|
||||
const actionMap: Record<string, string> = {
|
||||
'querygraph': 'query_graph',
|
||||
'query_graph': 'query_graph',
|
||||
'getcode': 'get_code',
|
||||
'get_code': 'get_code',
|
||||
'searchfiles': 'search_files',
|
||||
'search_files': 'search_files',
|
||||
'finalanswer': 'final_answer',
|
||||
'final_answer': 'final_answer',
|
||||
'answer': 'final_answer',
|
||||
'respond': 'final_answer'
|
||||
};
|
||||
|
||||
action = actionMap[action] || action;
|
||||
}
|
||||
|
||||
return {
|
||||
step: stepNumber,
|
||||
thought: thoughtMatch ? thoughtMatch[1].trim() : 'No thought provided',
|
||||
action: action || 'unknown',
|
||||
actionInput: actionInputMatch ? actionInputMatch[1].trim() : ''
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute an action and return the result
|
||||
*/
|
||||
private async executeAction(action: string, input: string, llmConfig: LLMConfig): Promise<ReActToolResult> {
|
||||
try {
|
||||
let output = '';
|
||||
let success = false;
|
||||
|
||||
// Normalize action name for case-insensitive matching
|
||||
const normalizedAction = action.toLowerCase().trim();
|
||||
|
||||
switch (normalizedAction) {
|
||||
case 'query_graph':
|
||||
if (!this.context) {
|
||||
throw new Error('Context not set');
|
||||
}
|
||||
|
||||
try {
|
||||
const cypherQuery = await this.cypherGenerator.generateQuery(input, llmConfig, {
|
||||
maxRetries: 3
|
||||
});
|
||||
|
||||
// Get config once for efficiency
|
||||
const config = await configLoader.loadConfig();
|
||||
|
||||
// Add LIMIT if not present and limiting is enabled
|
||||
let finalCypher = cypherQuery.cypher;
|
||||
if (config.ai.cypher.enableLimiting && !finalCypher.toLowerCase().includes('limit')) {
|
||||
const defaultLimit = config.ai.cypher.defaultLimit;
|
||||
finalCypher += ` LIMIT ${defaultLimit}`;
|
||||
}
|
||||
|
||||
// Execute the query
|
||||
const results = await this.executeGraphQuery(finalCypher);
|
||||
|
||||
// Truncate large responses if truncation is enabled
|
||||
if (config.ai.cypher.enableTruncation && results.rows && results.rows.length > config.ai.cypher.maxLimit) {
|
||||
const maxLimit = config.ai.cypher.maxLimit;
|
||||
results.rows = results.rows.slice(0, maxLimit);
|
||||
results.rowCount = maxLimit;
|
||||
results.truncated = true;
|
||||
results.summary += ` (showing first ${maxLimit} results)`;
|
||||
}
|
||||
|
||||
output = JSON.stringify(results, null, 2);
|
||||
success = true;
|
||||
} catch (error) {
|
||||
output = `Error generating or executing query: ${error instanceof Error ? error.message : 'Unknown error'}`;
|
||||
success = false;
|
||||
}
|
||||
break;
|
||||
|
||||
case 'get_code':
|
||||
if (!this.context) {
|
||||
throw new Error('Context not set');
|
||||
}
|
||||
|
||||
const content = this.context.fileContents.get(input);
|
||||
if (content) {
|
||||
output = content;
|
||||
success = true;
|
||||
} else {
|
||||
output = `File not found: ${input}`;
|
||||
success = false;
|
||||
}
|
||||
break;
|
||||
|
||||
case 'search_files':
|
||||
if (!this.context) {
|
||||
throw new Error('Context not set');
|
||||
}
|
||||
|
||||
const matchingFiles = Array.from(this.context.fileContents.keys())
|
||||
.filter(file => file.toLowerCase().includes(input.toLowerCase()))
|
||||
.slice(0, 10); // Limit results
|
||||
|
||||
output = JSON.stringify(matchingFiles, null, 2);
|
||||
success = true;
|
||||
break;
|
||||
|
||||
case 'final_answer':
|
||||
output = input;
|
||||
success = true;
|
||||
break;
|
||||
|
||||
case 'unknown':
|
||||
case '':
|
||||
output = `No action specified. Available actions: query_graph, get_code, search_files, final_answer`;
|
||||
success = false;
|
||||
break;
|
||||
|
||||
default:
|
||||
output = `Unknown action: "${action}". Available actions: query_graph, get_code, search_files, final_answer`;
|
||||
success = false;
|
||||
}
|
||||
|
||||
return {
|
||||
toolName: action,
|
||||
input,
|
||||
output,
|
||||
success
|
||||
};
|
||||
} catch (error) {
|
||||
return {
|
||||
toolName: action,
|
||||
input,
|
||||
output: `Error executing action: ${error instanceof Error ? error.message : 'Unknown error'}`,
|
||||
success: false,
|
||||
error: error instanceof Error ? error.message : 'Unknown error'
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute a graph query using KuzuDB if available, fallback to JSON graph
|
||||
*/
|
||||
private async executeGraphQuery(cypher: string): Promise<any> {
|
||||
console.log('🔍 ReActAgent executing Cypher query:', cypher);
|
||||
|
||||
// Try to get KuzuQueryEngine from DualWriteKnowledgeGraph
|
||||
if (this.graph && 'getKuzuGraph' in this.graph) {
|
||||
const kuzuGraph = (this.graph as any).getKuzuGraph();
|
||||
if (kuzuGraph && 'executeQuery' in kuzuGraph) {
|
||||
try {
|
||||
console.log('🚀 Using KuzuDB for query execution');
|
||||
const result = await kuzuGraph.executeQuery(cypher);
|
||||
|
||||
// Format result for AI consumption
|
||||
const formattedResult = {
|
||||
success: true,
|
||||
source: 'KuzuDB',
|
||||
columns: result.columns || [],
|
||||
rows: result.rows || [],
|
||||
rowCount: result.rowCount || result.rows?.length || 0,
|
||||
executionTime: result.executionTime || 0,
|
||||
// Add human-readable summary
|
||||
summary: `Found ${result.rowCount || result.rows?.length || 0} results in ${result.executionTime || 0}ms`
|
||||
};
|
||||
|
||||
console.log(`✅ KuzuDB query successful: ${formattedResult.rowCount} rows returned`);
|
||||
return formattedResult;
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ KuzuDB query failed:', error);
|
||||
console.log('🔄 Falling back to JSON graph query');
|
||||
|
||||
// Return error info for AI to understand what went wrong
|
||||
return {
|
||||
success: false,
|
||||
source: 'KuzuDB',
|
||||
error: error instanceof Error ? error.message : 'Unknown error',
|
||||
fallback: 'Attempting JSON graph query...'
|
||||
};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback to JSON graph query
|
||||
console.log('📊 Using JSON graph fallback');
|
||||
return this.fallbackGraphQuery(cypher);
|
||||
}
|
||||
|
||||
/**
|
||||
* Fallback graph query using the JSON graph and GraphQueryEngine
|
||||
*/
|
||||
private async fallbackGraphQuery(cypher: string): Promise<any> {
|
||||
if (!this.context?.graph) {
|
||||
return {
|
||||
nodes: [],
|
||||
relationships: [],
|
||||
message: 'No graph context available',
|
||||
success: false,
|
||||
source: 'none'
|
||||
};
|
||||
}
|
||||
|
||||
try {
|
||||
// Use existing GraphQueryEngine for fallback
|
||||
const { GraphQueryEngine } = await import('../core/graph/query-engine.ts');
|
||||
const queryEngine = new GraphQueryEngine(this.context.graph);
|
||||
|
||||
console.log('📄 Using JSON graph fallback for query execution');
|
||||
const result = queryEngine.executeQuery(cypher);
|
||||
return {
|
||||
nodes: result.nodes,
|
||||
relationships: result.relationships,
|
||||
data: result.data,
|
||||
success: true,
|
||||
source: 'JSON'
|
||||
};
|
||||
} catch (error) {
|
||||
console.error('❌ Fallback query failed:', error);
|
||||
return {
|
||||
nodes: [],
|
||||
relationships: [],
|
||||
message: 'Query execution failed',
|
||||
success: false,
|
||||
error: error instanceof Error ? error.message : 'Unknown error',
|
||||
source: 'error'
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build summary prompt for incomplete reasoning
|
||||
*/
|
||||
private buildSummaryPrompt(question: string, reasoning: ReActStep[]): string {
|
||||
const reasoningText = reasoning.map(step =>
|
||||
`Step ${step.step}: ${step.thought}\nAction: ${step.action}\nResult: ${step.observation || 'No result'}`
|
||||
).join('\n\n');
|
||||
|
||||
return `Based on the following reasoning steps, provide a comprehensive answer to the question: "${question}"
|
||||
|
||||
Reasoning steps:
|
||||
${reasoningText}
|
||||
|
||||
Please provide a complete answer based on the information gathered.`;
|
||||
}
|
||||
}
|
||||
@@ -1 +0,0 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" class="iconify iconify--logos" width="35.93" height="32" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 228"><path fill="#00D8FF" d="M210.483 73.824a171.49 171.49 0 0 0-8.24-2.597c.465-1.9.893-3.777 1.273-5.621c6.238-30.281 2.16-54.676-11.769-62.708c-13.355-7.7-35.196.329-57.254 19.526a171.23 171.23 0 0 0-6.375 5.848a155.866 155.866 0 0 0-4.241-3.917C100.759 3.829 77.587-4.822 63.673 3.233C50.33 10.957 46.379 33.89 51.995 62.588a170.974 170.974 0 0 0 1.892 8.48c-3.28.932-6.445 1.924-9.474 2.98C17.309 83.498 0 98.307 0 113.668c0 15.865 18.582 31.778 46.812 41.427a145.52 145.52 0 0 0 6.921 2.165a167.467 167.467 0 0 0-2.01 9.138c-5.354 28.2-1.173 50.591 12.134 58.266c13.744 7.926 36.812-.22 59.273-19.855a145.567 145.567 0 0 0 5.342-4.923a168.064 168.064 0 0 0 6.92 6.314c21.758 18.722 43.246 26.282 56.54 18.586c13.731-7.949 18.194-32.003 12.4-61.268a145.016 145.016 0 0 0-1.535-6.842c1.62-.48 3.21-.974 4.76-1.488c29.348-9.723 48.443-25.443 48.443-41.52c0-15.417-17.868-30.326-45.517-39.844Zm-6.365 70.984c-1.4.463-2.836.91-4.3 1.345c-3.24-10.257-7.612-21.163-12.963-32.432c5.106-11 9.31-21.767 12.459-31.957c2.619.758 5.16 1.557 7.61 2.4c23.69 8.156 38.14 20.213 38.14 29.504c0 9.896-15.606 22.743-40.946 31.14Zm-10.514 20.834c2.562 12.94 2.927 24.64 1.23 33.787c-1.524 8.219-4.59 13.698-8.382 15.893c-8.067 4.67-25.32-1.4-43.927-17.412a156.726 156.726 0 0 1-6.437-5.87c7.214-7.889 14.423-17.06 21.459-27.246c12.376-1.098 24.068-2.894 34.671-5.345a134.17 134.17 0 0 1 1.386 6.193ZM87.276 214.515c-7.882 2.783-14.16 2.863-17.955.675c-8.075-4.657-11.432-22.636-6.853-46.752a156.923 156.923 0 0 1 1.869-8.499c10.486 2.32 22.093 3.988 34.498 4.994c7.084 9.967 14.501 19.128 21.976 27.15a134.668 134.668 0 0 1-4.877 4.492c-9.933 8.682-19.886 14.842-28.658 17.94ZM50.35 144.747c-12.483-4.267-22.792-9.812-29.858-15.863c-6.35-5.437-9.555-10.836-9.555-15.216c0-9.322 13.897-21.212 37.076-29.293c2.813-.98 5.757-1.905 8.812-2.773c3.204 10.42 7.406 21.315 12.477 32.332c-5.137 11.18-9.399 22.249-12.634 32.792a134.718 134.718 0 0 1-6.318-1.979Zm12.378-84.26c-4.811-24.587-1.616-43.134 6.425-47.789c8.564-4.958 27.502 2.111 47.463 19.835a144.318 144.318 0 0 1 3.841 3.545c-7.438 7.987-14.787 17.08-21.808 26.988c-12.04 1.116-23.565 2.908-34.161 5.309a160.342 160.342 0 0 1-1.76-7.887Zm110.427 27.268a347.8 347.8 0 0 0-7.785-12.803c8.168 1.033 15.994 2.404 23.343 4.08c-2.206 7.072-4.956 14.465-8.193 22.045a381.151 381.151 0 0 0-7.365-13.322Zm-45.032-43.861c5.044 5.465 10.096 11.566 15.065 18.186a322.04 322.04 0 0 0-30.257-.006c4.974-6.559 10.069-12.652 15.192-18.18ZM82.802 87.83a323.167 323.167 0 0 0-7.227 13.238c-3.184-7.553-5.909-14.98-8.134-22.152c7.304-1.634 15.093-2.97 23.209-3.984a321.524 321.524 0 0 0-7.848 12.897Zm8.081 65.352c-8.385-.936-16.291-2.203-23.593-3.793c2.26-7.3 5.045-14.885 8.298-22.6a321.187 321.187 0 0 0 7.257 13.246c2.594 4.48 5.28 8.868 8.038 13.147Zm37.542 31.03c-5.184-5.592-10.354-11.779-15.403-18.433c4.902.192 9.899.29 14.978.29c5.218 0 10.376-.117 15.453-.343c-4.985 6.774-10.018 12.97-15.028 18.486Zm52.198-57.817c3.422 7.8 6.306 15.345 8.596 22.52c-7.422 1.694-15.436 3.058-23.88 4.071a382.417 382.417 0 0 0 7.859-13.026a347.403 347.403 0 0 0 7.425-13.565Zm-16.898 8.101a358.557 358.557 0 0 1-12.281 19.815a329.4 329.4 0 0 1-23.444.823c-7.967 0-15.716-.248-23.178-.732a310.202 310.202 0 0 1-12.513-19.846h.001a307.41 307.41 0 0 1-10.923-20.627a310.278 310.278 0 0 1 10.89-20.637l-.001.001a307.318 307.318 0 0 1 12.413-19.761c7.613-.576 15.42-.876 23.31-.876H128c7.926 0 15.743.303 23.354.883a329.357 329.357 0 0 1 12.335 19.695a358.489 358.489 0 0 1 11.036 20.54a329.472 329.472 0 0 1-11 20.722Zm22.56-122.124c8.572 4.944 11.906 24.881 6.52 51.026c-.344 1.668-.73 3.367-1.15 5.09c-10.622-2.452-22.155-4.275-34.23-5.408c-7.034-10.017-14.323-19.124-21.64-27.008a160.789 160.789 0 0 1 5.888-5.4c18.9-16.447 36.564-22.941 44.612-18.3ZM128 90.808c12.625 0 22.86 10.235 22.86 22.86s-10.235 22.86-22.86 22.86s-22.86-10.235-22.86-22.86s10.235-22.86 22.86-22.86Z"></path></svg>
|
||||
|
Before Width: | Height: | Size: 4.0 KiB |
@@ -0,0 +1,459 @@
|
||||
import { useCallback, useEffect, useMemo, useRef, useState } from 'react';
|
||||
import { Code, PanelLeftClose, PanelLeft, Trash2, X, Target, FileCode, Sparkles, MousePointerClick } from 'lucide-react';
|
||||
import { Prism as SyntaxHighlighter } from 'react-syntax-highlighter';
|
||||
import { vscDarkPlus } from 'react-syntax-highlighter/dist/esm/styles/prism';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { NODE_COLORS } from '../lib/constants';
|
||||
|
||||
// Match the code theme used elsewhere in the app
|
||||
const customTheme = {
|
||||
...vscDarkPlus,
|
||||
'pre[class*="language-"]': {
|
||||
...vscDarkPlus['pre[class*="language-"]'],
|
||||
background: '#0a0a10',
|
||||
margin: 0,
|
||||
padding: '12px 0',
|
||||
fontSize: '13px',
|
||||
lineHeight: '1.6',
|
||||
},
|
||||
'code[class*="language-"]': {
|
||||
...vscDarkPlus['code[class*="language-"]'],
|
||||
background: 'transparent',
|
||||
fontFamily: '"JetBrains Mono", "Fira Code", monospace',
|
||||
},
|
||||
};
|
||||
|
||||
export interface CodeReferencesPanelProps {
|
||||
onFocusNode: (nodeId: string) => void;
|
||||
}
|
||||
|
||||
export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) => {
|
||||
const {
|
||||
graph,
|
||||
fileContents,
|
||||
selectedNode,
|
||||
codeReferences,
|
||||
removeCodeReference,
|
||||
clearCodeReferences,
|
||||
setSelectedNode,
|
||||
codeReferenceFocus,
|
||||
} = useAppState();
|
||||
|
||||
const [isCollapsed, setIsCollapsed] = useState(false);
|
||||
const [glowRefId, setGlowRefId] = useState<string | null>(null);
|
||||
const panelRef = useRef<HTMLElement | null>(null);
|
||||
const resizeRef = useRef<{ startX: number; startWidth: number } | null>(null);
|
||||
const refCardEls = useRef<Map<string, HTMLDivElement | null>>(new Map());
|
||||
const glowTimerRef = useRef<number | null>(null);
|
||||
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
if (glowTimerRef.current) {
|
||||
window.clearTimeout(glowTimerRef.current);
|
||||
glowTimerRef.current = null;
|
||||
}
|
||||
};
|
||||
}, []);
|
||||
|
||||
const [panelWidth, setPanelWidth] = useState<number>(() => {
|
||||
try {
|
||||
const saved = window.localStorage.getItem('gitnexus.codePanelWidth');
|
||||
const parsed = saved ? parseInt(saved, 10) : NaN;
|
||||
if (!Number.isFinite(parsed)) return 560; // increased default
|
||||
return Math.max(420, Math.min(parsed, 900));
|
||||
} catch {
|
||||
return 560;
|
||||
}
|
||||
});
|
||||
|
||||
useEffect(() => {
|
||||
try {
|
||||
window.localStorage.setItem('gitnexus.codePanelWidth', String(panelWidth));
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
}, [panelWidth]);
|
||||
|
||||
const startResize = useCallback((e: React.MouseEvent) => {
|
||||
e.preventDefault();
|
||||
e.stopPropagation();
|
||||
resizeRef.current = { startX: e.clientX, startWidth: panelWidth };
|
||||
document.body.style.cursor = 'col-resize';
|
||||
document.body.style.userSelect = 'none';
|
||||
|
||||
const onMove = (ev: MouseEvent) => {
|
||||
const state = resizeRef.current;
|
||||
if (!state) return;
|
||||
const delta = ev.clientX - state.startX;
|
||||
const next = Math.max(420, Math.min(state.startWidth + delta, 900));
|
||||
setPanelWidth(next);
|
||||
};
|
||||
|
||||
const onUp = () => {
|
||||
resizeRef.current = null;
|
||||
document.body.style.cursor = '';
|
||||
document.body.style.userSelect = '';
|
||||
window.removeEventListener('mousemove', onMove);
|
||||
window.removeEventListener('mouseup', onUp);
|
||||
};
|
||||
|
||||
window.addEventListener('mousemove', onMove);
|
||||
window.addEventListener('mouseup', onUp);
|
||||
}, [panelWidth]);
|
||||
|
||||
const aiReferences = useMemo(() => codeReferences.filter(r => r.source === 'ai'), [codeReferences]);
|
||||
|
||||
// When the user clicks a citation badge in chat, focus the corresponding snippet card:
|
||||
// - expand the panel if collapsed
|
||||
// - smooth-scroll the card into view
|
||||
// - briefly glow it for discoverability
|
||||
useEffect(() => {
|
||||
if (!codeReferenceFocus) return;
|
||||
|
||||
// Ensure panel is expanded
|
||||
setIsCollapsed(false);
|
||||
|
||||
const { filePath, startLine, endLine } = codeReferenceFocus;
|
||||
const target =
|
||||
aiReferences.find(r =>
|
||||
r.filePath === filePath &&
|
||||
r.startLine === startLine &&
|
||||
r.endLine === endLine
|
||||
) ??
|
||||
aiReferences.find(r => r.filePath === filePath);
|
||||
|
||||
if (!target) return;
|
||||
|
||||
// Double rAF: wait for collapse state + list DOM to render.
|
||||
requestAnimationFrame(() => {
|
||||
requestAnimationFrame(() => {
|
||||
const el = refCardEls.current.get(target.id);
|
||||
if (!el) return;
|
||||
|
||||
el.scrollIntoView({ behavior: 'smooth', block: 'center' });
|
||||
setGlowRefId(target.id);
|
||||
|
||||
if (glowTimerRef.current) {
|
||||
window.clearTimeout(glowTimerRef.current);
|
||||
}
|
||||
glowTimerRef.current = window.setTimeout(() => {
|
||||
setGlowRefId((prev) => (prev === target.id ? null : prev));
|
||||
glowTimerRef.current = null;
|
||||
}, 1200);
|
||||
});
|
||||
});
|
||||
}, [codeReferenceFocus?.ts, aiReferences]);
|
||||
|
||||
const refsWithSnippets = useMemo(() => {
|
||||
return aiReferences.map((ref) => {
|
||||
const content = fileContents.get(ref.filePath);
|
||||
if (!content) {
|
||||
return { ref, content: null as string | null, start: 0, end: 0, highlightStart: 0, highlightEnd: 0, totalLines: 0 };
|
||||
}
|
||||
|
||||
const lines = content.split('\n');
|
||||
const totalLines = lines.length;
|
||||
|
||||
const startLine = ref.startLine ?? 0;
|
||||
const endLine = ref.endLine ?? startLine;
|
||||
|
||||
const contextBefore = 3;
|
||||
const contextAfter = 20;
|
||||
const start = Math.max(0, startLine - contextBefore);
|
||||
const end = Math.min(totalLines - 1, endLine + contextAfter);
|
||||
|
||||
return {
|
||||
ref,
|
||||
content: lines.slice(start, end + 1).join('\n'),
|
||||
start,
|
||||
end,
|
||||
highlightStart: Math.max(0, startLine - start),
|
||||
highlightEnd: Math.max(0, endLine - start),
|
||||
totalLines,
|
||||
};
|
||||
});
|
||||
}, [aiReferences, fileContents]);
|
||||
|
||||
const selectedFilePath = selectedNode?.properties?.filePath;
|
||||
const selectedFileContent = selectedFilePath ? fileContents.get(selectedFilePath) : undefined;
|
||||
const selectedIsFile = selectedNode?.label === 'File' && !!selectedFilePath;
|
||||
const showSelectedViewer = !!selectedNode && !!selectedFilePath;
|
||||
const showCitations = aiReferences.length > 0;
|
||||
|
||||
if (isCollapsed) {
|
||||
return (
|
||||
<aside className="h-full w-12 bg-surface border-r border-border-subtle flex flex-col items-center py-3 gap-2 flex-shrink-0">
|
||||
<button
|
||||
onClick={() => setIsCollapsed(false)}
|
||||
className="p-2 text-text-secondary hover:text-cyan-400 hover:bg-cyan-500/10 rounded transition-colors"
|
||||
title="Expand Code Panel"
|
||||
>
|
||||
<PanelLeft className="w-5 h-5" />
|
||||
</button>
|
||||
<div className="w-6 h-px bg-border-subtle my-1" />
|
||||
{showSelectedViewer && (
|
||||
<div className="text-[9px] text-amber-400 rotate-90 whitespace-nowrap font-medium tracking-wide">
|
||||
SELECTED
|
||||
</div>
|
||||
)}
|
||||
{showCitations && (
|
||||
<div className="text-[9px] text-cyan-400 rotate-90 whitespace-nowrap font-medium tracking-wide mt-4">
|
||||
AI • {aiReferences.length}
|
||||
</div>
|
||||
)}
|
||||
</aside>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<aside
|
||||
ref={(el) => { panelRef.current = el; }}
|
||||
className="h-full bg-surface/95 backdrop-blur-md border-r border-border-subtle flex flex-col animate-slide-in relative shadow-2xl"
|
||||
style={{ width: panelWidth }}
|
||||
>
|
||||
{/* Resize handle */}
|
||||
<div
|
||||
onMouseDown={startResize}
|
||||
className="absolute top-0 right-0 h-full w-2 cursor-col-resize bg-transparent hover:bg-cyan-500/25 transition-colors"
|
||||
title="Drag to resize"
|
||||
/>
|
||||
{/* Header */}
|
||||
<div className="flex items-center justify-between px-3 py-2.5 border-b border-border-subtle bg-gradient-to-r from-elevated/60 to-surface/60">
|
||||
<div className="flex items-center gap-2">
|
||||
<Code className="w-4 h-4 text-cyan-400" />
|
||||
<span className="text-sm font-semibold text-text-primary">Code Inspector</span>
|
||||
</div>
|
||||
<div className="flex items-center gap-1.5">
|
||||
{showCitations && (
|
||||
<button
|
||||
onClick={() => clearCodeReferences()}
|
||||
className="p-1.5 text-text-muted hover:text-red-400 hover:bg-red-500/10 rounded transition-colors"
|
||||
title="Clear AI citations"
|
||||
>
|
||||
<Trash2 className="w-4 h-4" />
|
||||
</button>
|
||||
)}
|
||||
<button
|
||||
onClick={() => setIsCollapsed(true)}
|
||||
className="p-1.5 text-text-muted hover:text-text-primary hover:bg-hover rounded transition-colors"
|
||||
title="Collapse Panel"
|
||||
>
|
||||
<PanelLeftClose className="w-4 h-4" />
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="flex-1 min-h-0 flex flex-col">
|
||||
{/* Top: Selected file viewer (when a node is selected) */}
|
||||
{showSelectedViewer && (
|
||||
<div className={`${showCitations ? 'h-[42%]' : 'flex-1'} min-h-0 flex flex-col`}>
|
||||
<div className="px-3 py-2 bg-gradient-to-r from-amber-500/8 to-orange-500/5 border-b border-amber-500/20 flex items-center gap-2">
|
||||
<div className="flex items-center gap-1.5 px-2 py-0.5 bg-amber-500/15 rounded-md border border-amber-500/25">
|
||||
<MousePointerClick className="w-3 h-3 text-amber-400" />
|
||||
<span className="text-[10px] text-amber-300 font-semibold uppercase tracking-wide">Selected</span>
|
||||
</div>
|
||||
<FileCode className="w-3.5 h-3.5 text-amber-400/70 ml-1" />
|
||||
<span className="text-xs text-text-primary font-mono truncate flex-1">
|
||||
{selectedNode?.properties?.filePath?.split('/').pop() ?? selectedNode?.properties?.name}
|
||||
</span>
|
||||
<button
|
||||
onClick={() => setSelectedNode(null)}
|
||||
className="p-1 text-text-muted hover:text-amber-400 hover:bg-amber-500/10 rounded transition-colors"
|
||||
title="Clear selection"
|
||||
>
|
||||
<X className="w-4 h-4" />
|
||||
</button>
|
||||
</div>
|
||||
<div className="flex-1 min-h-0 overflow-auto scrollbar-thin">
|
||||
{selectedFileContent ? (
|
||||
<SyntaxHighlighter
|
||||
language={
|
||||
selectedFilePath?.endsWith('.py') ? 'python' :
|
||||
selectedFilePath?.endsWith('.js') || selectedFilePath?.endsWith('.jsx') ? 'javascript' :
|
||||
selectedFilePath?.endsWith('.ts') || selectedFilePath?.endsWith('.tsx') ? 'typescript' :
|
||||
'text'
|
||||
}
|
||||
style={customTheme as any}
|
||||
showLineNumbers
|
||||
startingLineNumber={1}
|
||||
lineNumberStyle={{
|
||||
minWidth: '3em',
|
||||
paddingRight: '1em',
|
||||
color: '#5a5a70',
|
||||
textAlign: 'right',
|
||||
userSelect: 'none',
|
||||
}}
|
||||
lineProps={(lineNumber) => {
|
||||
const startLine = selectedNode?.properties?.startLine;
|
||||
const endLine = selectedNode?.properties?.endLine ?? startLine;
|
||||
const isHighlighted =
|
||||
typeof startLine === 'number' &&
|
||||
lineNumber >= startLine + 1 &&
|
||||
lineNumber <= (endLine ?? startLine) + 1;
|
||||
return {
|
||||
style: {
|
||||
display: 'block',
|
||||
backgroundColor: isHighlighted ? 'rgba(6, 182, 212, 0.14)' : 'transparent',
|
||||
borderLeft: isHighlighted ? '3px solid #06b6d4' : '3px solid transparent',
|
||||
paddingLeft: '12px',
|
||||
paddingRight: '16px',
|
||||
},
|
||||
};
|
||||
}}
|
||||
wrapLines
|
||||
>
|
||||
{selectedFileContent}
|
||||
</SyntaxHighlighter>
|
||||
) : (
|
||||
<div className="px-3 py-3 text-sm text-text-muted">
|
||||
{selectedIsFile ? (
|
||||
<>Code not available in memory for <span className="font-mono">{selectedFilePath}</span></>
|
||||
) : (
|
||||
<>Select a file node to preview its contents.</>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Divider between Selected viewer and AI refs (more visible) */}
|
||||
{showSelectedViewer && showCitations && (
|
||||
<div className="h-1.5 bg-gradient-to-r from-transparent via-border-subtle to-transparent" />
|
||||
)}
|
||||
|
||||
{/* Bottom: AI citations list */}
|
||||
{showCitations && (
|
||||
<div className="flex-1 min-h-0 flex flex-col">
|
||||
{/* AI Citations Section Header */}
|
||||
<div className="px-3 py-2 bg-gradient-to-r from-cyan-500/8 to-teal-500/5 border-b border-cyan-500/20 flex items-center gap-2">
|
||||
<div className="flex items-center gap-1.5 px-2 py-0.5 bg-cyan-500/15 rounded-md border border-cyan-500/25">
|
||||
<Sparkles className="w-3 h-3 text-cyan-400" />
|
||||
<span className="text-[10px] text-cyan-300 font-semibold uppercase tracking-wide">AI Citations</span>
|
||||
</div>
|
||||
<span className="text-xs text-text-muted ml-1">{aiReferences.length} reference{aiReferences.length !== 1 ? 's' : ''}</span>
|
||||
</div>
|
||||
<div className="flex-1 min-h-0 overflow-y-auto scrollbar-thin p-3 space-y-3">
|
||||
{refsWithSnippets.map(({ ref, content, start, highlightStart, highlightEnd, totalLines }) => {
|
||||
const nodeColor = ref.label ? (NODE_COLORS as any)[ref.label] || '#6b7280' : '#6b7280';
|
||||
const hasRange = typeof ref.startLine === 'number';
|
||||
const startDisplay = hasRange ? (ref.startLine ?? 0) + 1 : undefined;
|
||||
const endDisplay = hasRange ? (ref.endLine ?? ref.startLine ?? 0) + 1 : undefined;
|
||||
const language =
|
||||
ref.filePath.endsWith('.py') ? 'python' :
|
||||
ref.filePath.endsWith('.js') || ref.filePath.endsWith('.jsx') ? 'javascript' :
|
||||
ref.filePath.endsWith('.ts') || ref.filePath.endsWith('.tsx') ? 'typescript' :
|
||||
'text';
|
||||
|
||||
const isGlowing = glowRefId === ref.id;
|
||||
|
||||
return (
|
||||
<div
|
||||
key={ref.id}
|
||||
ref={(el) => { refCardEls.current.set(ref.id, el); }}
|
||||
className={[
|
||||
'bg-elevated border border-border-subtle rounded-xl overflow-hidden transition-all',
|
||||
isGlowing ? 'ring-2 ring-cyan-300/70 shadow-[0_0_0_6px_rgba(34,211,238,0.14)] animate-pulse' : '',
|
||||
].join(' ')}
|
||||
>
|
||||
<div className="px-3 py-2 border-b border-border-subtle bg-surface/40 flex items-start gap-2">
|
||||
<span
|
||||
className="mt-0.5 px-2 py-0.5 rounded text-[10px] font-semibold uppercase tracking-wide flex-shrink-0"
|
||||
style={{ backgroundColor: nodeColor, color: '#06060a' }}
|
||||
title={ref.label ?? 'Code'}
|
||||
>
|
||||
{ref.label ?? 'Code'}
|
||||
</span>
|
||||
<div className="min-w-0 flex-1">
|
||||
<div className="text-xs text-text-primary font-medium truncate">
|
||||
{ref.name ?? ref.filePath.split('/').pop() ?? ref.filePath}
|
||||
</div>
|
||||
<div className="text-[11px] text-text-muted font-mono truncate">
|
||||
{ref.filePath}
|
||||
{startDisplay !== undefined && (
|
||||
<span className="text-text-secondary">
|
||||
{' '}
|
||||
• L{startDisplay}
|
||||
{endDisplay !== startDisplay ? `–${endDisplay}` : ''}
|
||||
</span>
|
||||
)}
|
||||
{totalLines > 0 && <span className="text-text-muted"> • {totalLines} lines</span>}
|
||||
</div>
|
||||
</div>
|
||||
<div className="flex items-center gap-1">
|
||||
{ref.nodeId && (
|
||||
<button
|
||||
onClick={() => {
|
||||
const nodeId = ref.nodeId!;
|
||||
// Sync selection + focus graph
|
||||
if (graph) {
|
||||
const node = graph.nodes.find((n) => n.id === nodeId);
|
||||
if (node) setSelectedNode(node);
|
||||
}
|
||||
onFocusNode(nodeId);
|
||||
}}
|
||||
className="p-1.5 text-text-muted hover:text-text-primary hover:bg-hover rounded transition-colors"
|
||||
title="Focus in graph"
|
||||
>
|
||||
<Target className="w-4 h-4" />
|
||||
</button>
|
||||
)}
|
||||
<button
|
||||
onClick={() => removeCodeReference(ref.id)}
|
||||
className="p-1.5 text-text-muted hover:text-text-primary hover:bg-hover rounded transition-colors"
|
||||
title="Remove"
|
||||
>
|
||||
<X className="w-4 h-4" />
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="overflow-x-auto">
|
||||
{content ? (
|
||||
<SyntaxHighlighter
|
||||
language={language}
|
||||
style={customTheme as any}
|
||||
showLineNumbers
|
||||
startingLineNumber={start + 1}
|
||||
lineNumberStyle={{
|
||||
minWidth: '3em',
|
||||
paddingRight: '1em',
|
||||
color: '#5a5a70',
|
||||
textAlign: 'right',
|
||||
userSelect: 'none',
|
||||
}}
|
||||
lineProps={(lineNumber) => {
|
||||
const isHighlighted =
|
||||
hasRange &&
|
||||
lineNumber >= start + highlightStart + 1 &&
|
||||
lineNumber <= start + highlightEnd + 1;
|
||||
return {
|
||||
style: {
|
||||
display: 'block',
|
||||
backgroundColor: isHighlighted ? 'rgba(6, 182, 212, 0.14)' : 'transparent',
|
||||
borderLeft: isHighlighted ? '3px solid #06b6d4' : '3px solid transparent',
|
||||
paddingLeft: '12px',
|
||||
paddingRight: '16px',
|
||||
},
|
||||
};
|
||||
}}
|
||||
wrapLines
|
||||
>
|
||||
{content}
|
||||
</SyntaxHighlighter>
|
||||
) : (
|
||||
<div className="px-3 py-3 text-sm text-text-muted">
|
||||
Code not available in memory for <span className="font-mono">{ref.filePath}</span>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</aside>
|
||||
);
|
||||
};
|
||||
@@ -0,0 +1,354 @@
|
||||
import { useState, useCallback, DragEvent } from 'react';
|
||||
import { Upload, FileArchive, Github, Loader2, ArrowRight, Key, Eye, EyeOff } from 'lucide-react';
|
||||
import { cloneRepository, parseGitHubUrl } from '../services/git-clone';
|
||||
import { FileEntry } from '../services/zip';
|
||||
|
||||
interface DropZoneProps {
|
||||
onFileSelect: (file: File) => void;
|
||||
onGitClone?: (files: FileEntry[]) => void;
|
||||
}
|
||||
|
||||
export const DropZone = ({ onFileSelect, onGitClone }: DropZoneProps) => {
|
||||
const [isDragging, setIsDragging] = useState(false);
|
||||
const [activeTab, setActiveTab] = useState<'zip' | 'github'>('zip');
|
||||
const [githubUrl, setGithubUrl] = useState('');
|
||||
const [githubToken, setGithubToken] = useState('');
|
||||
const [showToken, setShowToken] = useState(false);
|
||||
const [isCloning, setIsCloning] = useState(false);
|
||||
const [cloneProgress, setCloneProgress] = useState({ phase: '', percent: 0 });
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
|
||||
const handleDragOver = useCallback((e: DragEvent<HTMLDivElement>) => {
|
||||
e.preventDefault();
|
||||
e.stopPropagation();
|
||||
setIsDragging(true);
|
||||
}, []);
|
||||
|
||||
const handleDragLeave = useCallback((e: DragEvent<HTMLDivElement>) => {
|
||||
e.preventDefault();
|
||||
e.stopPropagation();
|
||||
setIsDragging(false);
|
||||
}, []);
|
||||
|
||||
const handleDrop = useCallback((e: DragEvent<HTMLDivElement>) => {
|
||||
e.preventDefault();
|
||||
e.stopPropagation();
|
||||
setIsDragging(false);
|
||||
|
||||
const files = e.dataTransfer.files;
|
||||
if (files.length > 0) {
|
||||
const file = files[0];
|
||||
if (file.name.endsWith('.zip')) {
|
||||
onFileSelect(file);
|
||||
} else {
|
||||
setError('Please drop a .zip file');
|
||||
}
|
||||
}
|
||||
}, [onFileSelect]);
|
||||
|
||||
const handleFileInput = useCallback((e: React.ChangeEvent<HTMLInputElement>) => {
|
||||
const files = e.target.files;
|
||||
if (files && files.length > 0) {
|
||||
const file = files[0];
|
||||
if (file.name.endsWith('.zip')) {
|
||||
onFileSelect(file);
|
||||
} else {
|
||||
setError('Please select a .zip file');
|
||||
}
|
||||
}
|
||||
}, [onFileSelect]);
|
||||
|
||||
const handleGitClone = async () => {
|
||||
if (!githubUrl.trim()) {
|
||||
setError('Please enter a GitHub URL');
|
||||
return;
|
||||
}
|
||||
|
||||
const parsed = parseGitHubUrl(githubUrl);
|
||||
if (!parsed) {
|
||||
setError('Invalid GitHub URL. Use format: https://github.com/owner/repo');
|
||||
return;
|
||||
}
|
||||
|
||||
setError(null);
|
||||
setIsCloning(true);
|
||||
setCloneProgress({ phase: 'starting', percent: 0 });
|
||||
|
||||
try {
|
||||
const files = await cloneRepository(
|
||||
githubUrl,
|
||||
(phase, percent) => setCloneProgress({ phase, percent }),
|
||||
githubToken || undefined // Pass token if provided
|
||||
);
|
||||
|
||||
// Clear token from memory after successful clone
|
||||
setGithubToken('');
|
||||
|
||||
if (onGitClone) {
|
||||
onGitClone(files);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('Clone failed:', err);
|
||||
const message = err instanceof Error ? err.message : 'Failed to clone repository';
|
||||
// Provide helpful error for auth failures
|
||||
if (message.includes('401') || message.includes('403') || message.includes('Authentication')) {
|
||||
if (!githubToken) {
|
||||
setError('🔒 This looks like a private repo. Add a GitHub PAT (Personal Access Token) to access it.');
|
||||
} else {
|
||||
setError('🔑 Authentication failed. Check your token permissions (needs repo access).');
|
||||
}
|
||||
} else if (message.includes('404') || message.includes('not found')) {
|
||||
setError('Repository not found. Check the URL or it might be private (needs PAT).');
|
||||
} else {
|
||||
setError(message);
|
||||
}
|
||||
} finally {
|
||||
setIsCloning(false);
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="flex items-center justify-center min-h-screen p-8 bg-void">
|
||||
{/* Background gradient effects */}
|
||||
<div className="fixed inset-0 pointer-events-none">
|
||||
<div className="absolute top-1/4 left-1/4 w-96 h-96 bg-accent/10 rounded-full blur-3xl" />
|
||||
<div className="absolute bottom-1/4 right-1/4 w-96 h-96 bg-node-interface/10 rounded-full blur-3xl" />
|
||||
</div>
|
||||
|
||||
<div className="relative w-full max-w-lg">
|
||||
{/* Tab Switcher */}
|
||||
<div className="flex mb-4 bg-surface border border-border-default rounded-xl p-1">
|
||||
<button
|
||||
onClick={() => { setActiveTab('zip'); setError(null); }}
|
||||
className={`
|
||||
flex-1 flex items-center justify-center gap-2 py-2.5 px-4 rounded-lg
|
||||
text-sm font-medium transition-all duration-200
|
||||
${activeTab === 'zip'
|
||||
? 'bg-accent text-white shadow-md'
|
||||
: 'text-text-secondary hover:text-text-primary hover:bg-elevated'
|
||||
}
|
||||
`}
|
||||
>
|
||||
<FileArchive className="w-4 h-4" />
|
||||
ZIP Upload
|
||||
</button>
|
||||
<button
|
||||
onClick={() => { setActiveTab('github'); setError(null); }}
|
||||
className={`
|
||||
flex-1 flex items-center justify-center gap-2 py-2.5 px-4 rounded-lg
|
||||
text-sm font-medium transition-all duration-200
|
||||
${activeTab === 'github'
|
||||
? 'bg-accent text-white shadow-md'
|
||||
: 'text-text-secondary hover:text-text-primary hover:bg-elevated'
|
||||
}
|
||||
`}
|
||||
>
|
||||
<Github className="w-4 h-4" />
|
||||
GitHub URL
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Error Message */}
|
||||
{error && (
|
||||
<div className="mb-4 p-3 bg-red-500/10 border border-red-500/30 rounded-xl text-red-400 text-sm text-center">
|
||||
{error}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* ZIP Upload Tab */}
|
||||
{activeTab === 'zip' && (
|
||||
<div
|
||||
className={`
|
||||
relative p-16
|
||||
bg-surface border-2 border-dashed rounded-3xl
|
||||
transition-all duration-300 cursor-pointer
|
||||
${isDragging
|
||||
? 'border-accent bg-elevated scale-105 shadow-glow'
|
||||
: 'border-border-default hover:border-accent/50 hover:bg-elevated/50 animate-breathe'
|
||||
}
|
||||
`}
|
||||
onDragOver={handleDragOver}
|
||||
onDragLeave={handleDragLeave}
|
||||
onDrop={handleDrop}
|
||||
onClick={() => document.getElementById('file-input')?.click()}
|
||||
>
|
||||
<input
|
||||
id="file-input"
|
||||
type="file"
|
||||
accept=".zip"
|
||||
className="hidden"
|
||||
onChange={handleFileInput}
|
||||
/>
|
||||
|
||||
{/* Icon */}
|
||||
<div className={`
|
||||
mx-auto w-20 h-20 mb-6
|
||||
flex items-center justify-center
|
||||
bg-gradient-to-br from-accent to-node-interface
|
||||
rounded-2xl shadow-glow
|
||||
transition-transform duration-300
|
||||
${isDragging ? 'scale-110' : ''}
|
||||
`}>
|
||||
{isDragging ? (
|
||||
<Upload className="w-10 h-10 text-white" />
|
||||
) : (
|
||||
<FileArchive className="w-10 h-10 text-white" />
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Text */}
|
||||
<h2 className="text-xl font-semibold text-text-primary text-center mb-2">
|
||||
{isDragging ? 'Drop it here!' : 'Drop your codebase'}
|
||||
</h2>
|
||||
<p className="text-sm text-text-secondary text-center mb-6">
|
||||
Drag & drop a .zip file to generate a knowledge graph
|
||||
</p>
|
||||
|
||||
{/* Hints */}
|
||||
<div className="flex items-center justify-center gap-3 text-xs text-text-muted">
|
||||
<span className="px-3 py-1.5 bg-elevated border border-border-subtle rounded-md">
|
||||
.zip
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* GitHub URL Tab */}
|
||||
{activeTab === 'github' && (
|
||||
<div className="p-8 bg-surface border border-border-default rounded-3xl">
|
||||
{/* Icon */}
|
||||
<div className="mx-auto w-20 h-20 mb-6 flex items-center justify-center bg-gradient-to-br from-[#333] to-[#24292e] rounded-2xl shadow-lg">
|
||||
<Github className="w-10 h-10 text-white" />
|
||||
</div>
|
||||
|
||||
{/* Text */}
|
||||
<h2 className="text-xl font-semibold text-text-primary text-center mb-2">
|
||||
Clone from GitHub
|
||||
</h2>
|
||||
<p className="text-sm text-text-secondary text-center mb-6">
|
||||
Enter a repository URL to clone directly
|
||||
</p>
|
||||
|
||||
{/* Inputs - wrapped in div to prevent form autofill */}
|
||||
<div className="space-y-3" data-form-type="other">
|
||||
<input
|
||||
type="url"
|
||||
name="github-repo-url-input"
|
||||
value={githubUrl}
|
||||
onChange={(e) => setGithubUrl(e.target.value)}
|
||||
onKeyDown={(e) => e.key === 'Enter' && !isCloning && handleGitClone()}
|
||||
placeholder="https://github.com/owner/repo"
|
||||
disabled={isCloning}
|
||||
autoComplete="off"
|
||||
data-lpignore="true"
|
||||
data-1p-ignore="true"
|
||||
data-form-type="other"
|
||||
className="
|
||||
w-full px-4 py-3
|
||||
bg-elevated border border-border-default rounded-xl
|
||||
text-text-primary placeholder-text-muted
|
||||
focus:outline-none focus:border-accent focus:ring-1 focus:ring-accent
|
||||
disabled:opacity-50 disabled:cursor-not-allowed
|
||||
transition-all duration-200
|
||||
"
|
||||
/>
|
||||
|
||||
{/* Token input for private repos */}
|
||||
<div className="relative">
|
||||
<div className="absolute left-3 top-1/2 -translate-y-1/2 text-text-muted">
|
||||
<Key className="w-4 h-4" />
|
||||
</div>
|
||||
<input
|
||||
type={showToken ? 'text' : 'password'}
|
||||
name="github-pat-token-input"
|
||||
value={githubToken}
|
||||
onChange={(e) => setGithubToken(e.target.value)}
|
||||
placeholder="GitHub PAT (optional, for private repos)"
|
||||
disabled={isCloning}
|
||||
autoComplete="new-password"
|
||||
data-lpignore="true"
|
||||
data-1p-ignore="true"
|
||||
data-form-type="other"
|
||||
className="
|
||||
w-full pl-10 pr-10 py-3
|
||||
bg-elevated border border-border-default rounded-xl
|
||||
text-text-primary placeholder-text-muted
|
||||
focus:outline-none focus:border-accent focus:ring-1 focus:ring-accent
|
||||
disabled:opacity-50 disabled:cursor-not-allowed
|
||||
transition-all duration-200
|
||||
"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => setShowToken(!showToken)}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 text-text-muted hover:text-text-secondary transition-colors"
|
||||
>
|
||||
{showToken ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
|
||||
<button
|
||||
onClick={handleGitClone}
|
||||
disabled={isCloning || !githubUrl.trim()}
|
||||
className="
|
||||
w-full flex items-center justify-center gap-2
|
||||
px-4 py-3
|
||||
bg-accent hover:bg-accent/90
|
||||
text-white font-medium rounded-xl
|
||||
disabled:opacity-50 disabled:cursor-not-allowed
|
||||
transition-all duration-200
|
||||
"
|
||||
>
|
||||
{isCloning ? (
|
||||
<>
|
||||
<Loader2 className="w-5 h-5 animate-spin" />
|
||||
{cloneProgress.phase === 'cloning'
|
||||
? `Cloning... ${cloneProgress.percent}%`
|
||||
: cloneProgress.phase === 'reading'
|
||||
? 'Reading files...'
|
||||
: 'Starting...'
|
||||
}
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
Clone Repository
|
||||
<ArrowRight className="w-5 h-5" />
|
||||
</>
|
||||
)}
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Progress bar */}
|
||||
{isCloning && (
|
||||
<div className="mt-4">
|
||||
<div className="h-2 bg-elevated rounded-full overflow-hidden">
|
||||
<div
|
||||
className="h-full bg-accent transition-all duration-300 ease-out"
|
||||
style={{ width: `${cloneProgress.percent}%` }}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Security note */}
|
||||
{githubToken && (
|
||||
<p className="mt-3 text-xs text-text-muted text-center">
|
||||
🔒 Token stays in your browser only, never sent to any server
|
||||
</p>
|
||||
)}
|
||||
|
||||
{/* Hints */}
|
||||
<div className="mt-4 flex items-center justify-center gap-3 text-xs text-text-muted">
|
||||
<span className="px-3 py-1.5 bg-elevated border border-border-subtle rounded-md">
|
||||
{githubToken ? 'Private + Public' : 'Public repos'}
|
||||
</span>
|
||||
<span className="px-3 py-1.5 bg-elevated border border-border-subtle rounded-md">
|
||||
Shallow clone
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
@@ -0,0 +1,195 @@
|
||||
import { Brain, Loader2, Check, AlertCircle, Zap, FlaskConical } from 'lucide-react';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { useState } from 'react';
|
||||
import { WebGPUFallbackDialog } from './WebGPUFallbackDialog';
|
||||
|
||||
/**
|
||||
* Embedding status indicator and trigger button
|
||||
* Shows in header when graph is loaded
|
||||
*/
|
||||
export const EmbeddingStatus = () => {
|
||||
const {
|
||||
embeddingStatus,
|
||||
embeddingProgress,
|
||||
startEmbeddings,
|
||||
graph,
|
||||
viewMode,
|
||||
testArrayParams,
|
||||
} = useAppState();
|
||||
|
||||
const [testResult, setTestResult] = useState<string | null>(null);
|
||||
const [showFallbackDialog, setShowFallbackDialog] = useState(false);
|
||||
|
||||
// Only show when exploring a loaded graph
|
||||
if (viewMode !== 'exploring' || !graph) return null;
|
||||
|
||||
const nodeCount = graph.nodes.length;
|
||||
|
||||
const handleStartEmbeddings = async (forceDevice?: 'webgpu' | 'wasm') => {
|
||||
try {
|
||||
await startEmbeddings(forceDevice);
|
||||
} catch (error: any) {
|
||||
// Check if it's a WebGPU not available error
|
||||
if (error?.name === 'WebGPUNotAvailableError' ||
|
||||
error?.message?.includes('WebGPU not available')) {
|
||||
setShowFallbackDialog(true);
|
||||
} else {
|
||||
console.error('Embedding failed:', error);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const handleUseCPU = () => {
|
||||
setShowFallbackDialog(false);
|
||||
handleStartEmbeddings('wasm');
|
||||
};
|
||||
|
||||
const handleSkipEmbeddings = () => {
|
||||
setShowFallbackDialog(false);
|
||||
// Just close - user can try again later if they want
|
||||
};
|
||||
|
||||
const handleTestArrayParams = async () => {
|
||||
setTestResult('Testing...');
|
||||
const result = await testArrayParams();
|
||||
if (result.success) {
|
||||
setTestResult('✅ Array params WORK!');
|
||||
console.log('✅ Array params test passed!');
|
||||
} else {
|
||||
setTestResult(`❌ ${result.error}`);
|
||||
console.error('❌ Array params test failed:', result.error);
|
||||
}
|
||||
};
|
||||
|
||||
// WebGPU fallback dialog - rendered independently of state
|
||||
const fallbackDialog = (
|
||||
<WebGPUFallbackDialog
|
||||
isOpen={showFallbackDialog}
|
||||
onClose={() => setShowFallbackDialog(false)}
|
||||
onUseCPU={handleUseCPU}
|
||||
onSkip={handleSkipEmbeddings}
|
||||
nodeCount={nodeCount}
|
||||
/>
|
||||
);
|
||||
|
||||
// Idle state - show button to start
|
||||
if (embeddingStatus === 'idle') {
|
||||
return (
|
||||
<>
|
||||
<div className="flex items-center gap-2">
|
||||
{/* Test button (dev only) */}
|
||||
{import.meta.env.DEV && (
|
||||
<button
|
||||
onClick={handleTestArrayParams}
|
||||
className="flex items-center gap-1 px-2 py-1.5 bg-surface border border-border-subtle rounded-lg text-xs text-text-muted hover:bg-hover hover:text-text-secondary transition-all"
|
||||
title="Test if KuzuDB supports array params"
|
||||
>
|
||||
<FlaskConical className="w-3 h-3" />
|
||||
{testResult || 'Test'}
|
||||
</button>
|
||||
)}
|
||||
|
||||
<button
|
||||
onClick={() => handleStartEmbeddings()}
|
||||
className="flex items-center gap-2 px-3 py-1.5 bg-surface border border-border-subtle rounded-lg text-sm text-text-secondary hover:bg-hover hover:text-text-primary hover:border-accent/50 transition-all group"
|
||||
title="Generate embeddings for semantic search"
|
||||
>
|
||||
<Brain className="w-4 h-4 text-node-interface group-hover:text-accent transition-colors" />
|
||||
<span className="hidden sm:inline">Enable Semantic Search</span>
|
||||
<Zap className="w-3 h-3 text-text-muted" />
|
||||
</button>
|
||||
</div>
|
||||
{fallbackDialog}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
// Loading model
|
||||
if (embeddingStatus === 'loading') {
|
||||
const downloadPercent = embeddingProgress?.modelDownloadPercent ?? 0;
|
||||
return (
|
||||
<>
|
||||
<div className="flex items-center gap-2.5 px-3 py-1.5 bg-surface border border-accent/30 rounded-lg text-sm">
|
||||
<Loader2 className="w-4 h-4 text-accent animate-spin" />
|
||||
<div className="flex flex-col gap-0.5">
|
||||
<span className="text-text-secondary text-xs">Loading AI model...</span>
|
||||
<div className="w-24 h-1 bg-elevated rounded-full overflow-hidden">
|
||||
<div
|
||||
className="h-full bg-gradient-to-r from-accent to-node-interface rounded-full transition-all duration-300"
|
||||
style={{ width: `${downloadPercent}%` }}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
{fallbackDialog}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
// Embedding in progress
|
||||
if (embeddingStatus === 'embedding') {
|
||||
const processed = embeddingProgress?.nodesProcessed ?? 0;
|
||||
const total = embeddingProgress?.totalNodes ?? 0;
|
||||
const percent = embeddingProgress?.percent ?? 0;
|
||||
|
||||
return (
|
||||
<div className="flex items-center gap-2.5 px-3 py-1.5 bg-surface border border-node-function/30 rounded-lg text-sm">
|
||||
<Loader2 className="w-4 h-4 text-node-function animate-spin" />
|
||||
<div className="flex flex-col gap-0.5">
|
||||
<span className="text-text-secondary text-xs">
|
||||
Embedding {processed}/{total} nodes
|
||||
</span>
|
||||
<div className="w-24 h-1 bg-elevated rounded-full overflow-hidden">
|
||||
<div
|
||||
className="h-full bg-gradient-to-r from-node-function to-accent rounded-full transition-all duration-300"
|
||||
style={{ width: `${percent}%` }}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// Indexing
|
||||
if (embeddingStatus === 'indexing') {
|
||||
return (
|
||||
<div className="flex items-center gap-2 px-3 py-1.5 bg-surface border border-node-interface/30 rounded-lg text-sm text-text-secondary">
|
||||
<Loader2 className="w-4 h-4 text-node-interface animate-spin" />
|
||||
<span className="text-xs">Creating vector index...</span>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// Ready
|
||||
if (embeddingStatus === 'ready') {
|
||||
return (
|
||||
<div
|
||||
className="flex items-center gap-2 px-3 py-1.5 bg-node-function/10 border border-node-function/30 rounded-lg text-sm text-node-function"
|
||||
title="Semantic search is ready! Use natural language in the AI chat."
|
||||
>
|
||||
<Check className="w-4 h-4" />
|
||||
<span className="text-xs font-medium">Semantic Ready</span>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
// Error
|
||||
if (embeddingStatus === 'error') {
|
||||
return (
|
||||
<>
|
||||
<button
|
||||
onClick={() => handleStartEmbeddings()}
|
||||
className="flex items-center gap-2 px-3 py-1.5 bg-red-500/10 border border-red-500/30 rounded-lg text-sm text-red-400 hover:bg-red-500/20 transition-colors"
|
||||
title={embeddingProgress?.error || 'Embedding failed. Click to retry.'}
|
||||
>
|
||||
<AlertCircle className="w-4 h-4" />
|
||||
<span className="text-xs">Failed - Retry</span>
|
||||
</button>
|
||||
{fallbackDialog}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
return null;
|
||||
};
|
||||
|
||||
@@ -0,0 +1,490 @@
|
||||
import { useState, useMemo, useCallback, useEffect } from 'react';
|
||||
import {
|
||||
ChevronRight,
|
||||
ChevronDown,
|
||||
Folder,
|
||||
FolderOpen,
|
||||
FileCode,
|
||||
Search,
|
||||
Filter,
|
||||
PanelLeftClose,
|
||||
PanelLeft,
|
||||
Box,
|
||||
Braces,
|
||||
Variable,
|
||||
Hash,
|
||||
Target,
|
||||
} from 'lucide-react';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { FILTERABLE_LABELS, NODE_COLORS } from '../lib/constants';
|
||||
import { GraphNode, NodeLabel } from '../core/graph/types';
|
||||
|
||||
// Tree node structure
|
||||
interface TreeNode {
|
||||
id: string;
|
||||
name: string;
|
||||
type: 'folder' | 'file';
|
||||
path: string;
|
||||
children: TreeNode[];
|
||||
graphNode?: GraphNode;
|
||||
}
|
||||
|
||||
// Build tree from graph nodes
|
||||
const buildFileTree = (nodes: GraphNode[]): TreeNode[] => {
|
||||
const root: TreeNode[] = [];
|
||||
const pathMap = new Map<string, TreeNode>();
|
||||
|
||||
// Filter to only folders and files
|
||||
const fileNodes = nodes.filter(n => n.label === 'Folder' || n.label === 'File');
|
||||
|
||||
// Sort by path to ensure parents come before children
|
||||
fileNodes.sort((a, b) => a.properties.filePath.localeCompare(b.properties.filePath));
|
||||
|
||||
fileNodes.forEach(node => {
|
||||
const parts = node.properties.filePath.split('/').filter(Boolean);
|
||||
let currentPath = '';
|
||||
let currentLevel = root;
|
||||
|
||||
parts.forEach((part, index) => {
|
||||
currentPath = currentPath ? `${currentPath}/${part}` : part;
|
||||
|
||||
let existing = pathMap.get(currentPath);
|
||||
|
||||
if (!existing) {
|
||||
const isLastPart = index === parts.length - 1;
|
||||
const isFile = isLastPart && node.label === 'File';
|
||||
|
||||
existing = {
|
||||
id: isLastPart ? node.id : currentPath,
|
||||
name: part,
|
||||
type: isFile ? 'file' : 'folder',
|
||||
path: currentPath,
|
||||
children: [],
|
||||
graphNode: isLastPart ? node : undefined,
|
||||
};
|
||||
|
||||
pathMap.set(currentPath, existing);
|
||||
currentLevel.push(existing);
|
||||
}
|
||||
|
||||
currentLevel = existing.children;
|
||||
});
|
||||
});
|
||||
|
||||
return root;
|
||||
};
|
||||
|
||||
// Tree item component
|
||||
interface TreeItemProps {
|
||||
node: TreeNode;
|
||||
depth: number;
|
||||
searchQuery: string;
|
||||
onNodeClick: (node: TreeNode) => void;
|
||||
expandedPaths: Set<string>;
|
||||
toggleExpanded: (path: string) => void;
|
||||
selectedPath: string | null;
|
||||
}
|
||||
|
||||
const TreeItem = ({
|
||||
node,
|
||||
depth,
|
||||
searchQuery,
|
||||
onNodeClick,
|
||||
expandedPaths,
|
||||
toggleExpanded,
|
||||
selectedPath,
|
||||
}: TreeItemProps) => {
|
||||
const isExpanded = expandedPaths.has(node.path);
|
||||
const isSelected = selectedPath === node.path;
|
||||
const hasChildren = node.children.length > 0;
|
||||
|
||||
// Filter children based on search
|
||||
const filteredChildren = useMemo(() => {
|
||||
if (!searchQuery) return node.children;
|
||||
return node.children.filter(child =>
|
||||
child.name.toLowerCase().includes(searchQuery.toLowerCase()) ||
|
||||
child.children.some(c => c.name.toLowerCase().includes(searchQuery.toLowerCase()))
|
||||
);
|
||||
}, [node.children, searchQuery]);
|
||||
|
||||
// Check if this node matches search
|
||||
const matchesSearch = searchQuery && node.name.toLowerCase().includes(searchQuery.toLowerCase());
|
||||
|
||||
const handleClick = () => {
|
||||
if (hasChildren) {
|
||||
toggleExpanded(node.path);
|
||||
}
|
||||
onNodeClick(node);
|
||||
};
|
||||
|
||||
return (
|
||||
<div>
|
||||
<button
|
||||
onClick={handleClick}
|
||||
className={`
|
||||
w-full flex items-center gap-1.5 px-2 py-1 text-left text-sm
|
||||
hover:bg-hover transition-colors rounded relative
|
||||
${isSelected ? 'bg-amber-500/15 text-amber-300 border-l-2 border-amber-400' : 'text-text-secondary hover:text-text-primary border-l-2 border-transparent'}
|
||||
${matchesSearch ? 'bg-accent/10' : ''}
|
||||
`}
|
||||
style={{ paddingLeft: `${depth * 12 + 8}px` }}
|
||||
>
|
||||
{/* Expand/collapse icon */}
|
||||
{hasChildren ? (
|
||||
isExpanded ? (
|
||||
<ChevronDown className="w-3.5 h-3.5 shrink-0 text-text-muted" />
|
||||
) : (
|
||||
<ChevronRight className="w-3.5 h-3.5 shrink-0 text-text-muted" />
|
||||
)
|
||||
) : (
|
||||
<span className="w-3.5" />
|
||||
)}
|
||||
|
||||
{/* Node icon */}
|
||||
{node.type === 'folder' ? (
|
||||
isExpanded ? (
|
||||
<FolderOpen className="w-4 h-4 shrink-0" style={{ color: NODE_COLORS.Folder }} />
|
||||
) : (
|
||||
<Folder className="w-4 h-4 shrink-0" style={{ color: NODE_COLORS.Folder }} />
|
||||
)
|
||||
) : (
|
||||
<FileCode className="w-4 h-4 shrink-0" style={{ color: NODE_COLORS.File }} />
|
||||
)}
|
||||
|
||||
{/* Name */}
|
||||
<span className="truncate font-mono text-xs">{node.name}</span>
|
||||
</button>
|
||||
|
||||
{/* Children */}
|
||||
{isExpanded && filteredChildren.length > 0 && (
|
||||
<div>
|
||||
{filteredChildren.map(child => (
|
||||
<TreeItem
|
||||
key={child.id}
|
||||
node={child}
|
||||
depth={depth + 1}
|
||||
searchQuery={searchQuery}
|
||||
onNodeClick={onNodeClick}
|
||||
expandedPaths={expandedPaths}
|
||||
toggleExpanded={toggleExpanded}
|
||||
selectedPath={selectedPath}
|
||||
/>
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
// Icon for node types
|
||||
const getNodeTypeIcon = (label: NodeLabel) => {
|
||||
switch (label) {
|
||||
case 'Folder': return Folder;
|
||||
case 'File': return FileCode;
|
||||
case 'Class': return Box;
|
||||
case 'Function': return Braces;
|
||||
case 'Method': return Braces;
|
||||
case 'Interface': return Hash;
|
||||
case 'Import': return FileCode;
|
||||
default: return Variable;
|
||||
}
|
||||
};
|
||||
|
||||
interface FileTreePanelProps {
|
||||
onFocusNode: (nodeId: string) => void;
|
||||
}
|
||||
|
||||
export const FileTreePanel = ({ onFocusNode }: FileTreePanelProps) => {
|
||||
const { graph, visibleLabels, toggleLabelVisibility, selectedNode, setSelectedNode, openCodePanel, depthFilter, setDepthFilter } = useAppState();
|
||||
|
||||
const [isCollapsed, setIsCollapsed] = useState(false);
|
||||
const [searchQuery, setSearchQuery] = useState('');
|
||||
const [expandedPaths, setExpandedPaths] = useState<Set<string>>(new Set());
|
||||
const [activeTab, setActiveTab] = useState<'files' | 'filters'>('files');
|
||||
|
||||
// Build file tree from graph
|
||||
const fileTree = useMemo(() => {
|
||||
if (!graph) return [];
|
||||
return buildFileTree(graph.nodes);
|
||||
}, [graph]);
|
||||
|
||||
// Auto-expand first level on initial load
|
||||
useEffect(() => {
|
||||
if (fileTree.length > 0 && expandedPaths.size === 0) {
|
||||
const firstLevel = new Set(fileTree.map(n => n.path));
|
||||
setExpandedPaths(firstLevel);
|
||||
}
|
||||
}, [fileTree.length]); // Only run when tree first loads
|
||||
|
||||
// Auto-expand to selected file when selectedNode changes (e.g., from graph click)
|
||||
useEffect(() => {
|
||||
const path = selectedNode?.properties?.filePath;
|
||||
if (!path) return;
|
||||
|
||||
// Expand all parent folders leading to this file
|
||||
const parts = path.split('/').filter(Boolean);
|
||||
const pathsToExpand: string[] = [];
|
||||
let currentPath = '';
|
||||
|
||||
// Build all parent paths (exclude the last part if it's a file)
|
||||
for (let i = 0; i < parts.length - 1; i++) {
|
||||
currentPath = currentPath ? `${currentPath}/${parts[i]}` : parts[i];
|
||||
pathsToExpand.push(currentPath);
|
||||
}
|
||||
|
||||
if (pathsToExpand.length > 0) {
|
||||
setExpandedPaths(prev => {
|
||||
const next = new Set(prev);
|
||||
pathsToExpand.forEach(p => next.add(p));
|
||||
return next;
|
||||
});
|
||||
}
|
||||
}, [selectedNode?.id]); // Trigger when selected node changes
|
||||
|
||||
const toggleExpanded = useCallback((path: string) => {
|
||||
setExpandedPaths(prev => {
|
||||
const next = new Set(prev);
|
||||
if (next.has(path)) {
|
||||
next.delete(path);
|
||||
} else {
|
||||
next.add(path);
|
||||
}
|
||||
return next;
|
||||
});
|
||||
}, []);
|
||||
|
||||
const handleNodeClick = useCallback((treeNode: TreeNode) => {
|
||||
if (treeNode.graphNode) {
|
||||
// Only focus if selecting a different node
|
||||
const isSameNode = selectedNode?.id === treeNode.graphNode.id;
|
||||
setSelectedNode(treeNode.graphNode);
|
||||
openCodePanel();
|
||||
if (!isSameNode) {
|
||||
onFocusNode(treeNode.graphNode.id);
|
||||
}
|
||||
}
|
||||
}, [setSelectedNode, openCodePanel, onFocusNode, selectedNode]);
|
||||
|
||||
const selectedPath = selectedNode?.properties.filePath || null;
|
||||
|
||||
if (isCollapsed) {
|
||||
return (
|
||||
<div className="h-full w-12 bg-surface border-r border-border-subtle flex flex-col items-center py-3 gap-2">
|
||||
<button
|
||||
onClick={() => setIsCollapsed(false)}
|
||||
className="p-2 text-text-secondary hover:text-text-primary hover:bg-hover rounded transition-colors"
|
||||
title="Expand Panel"
|
||||
>
|
||||
<PanelLeft className="w-5 h-5" />
|
||||
</button>
|
||||
<div className="w-6 h-px bg-border-subtle my-1" />
|
||||
<button
|
||||
onClick={() => { setIsCollapsed(false); setActiveTab('files'); }}
|
||||
className={`p-2 rounded transition-colors ${activeTab === 'files' ? 'text-accent bg-accent/10' : 'text-text-secondary hover:text-text-primary hover:bg-hover'}`}
|
||||
title="File Explorer"
|
||||
>
|
||||
<Folder className="w-5 h-5" />
|
||||
</button>
|
||||
<button
|
||||
onClick={() => { setIsCollapsed(false); setActiveTab('filters'); }}
|
||||
className={`p-2 rounded transition-colors ${activeTab === 'filters' ? 'text-accent bg-accent/10' : 'text-text-secondary hover:text-text-primary hover:bg-hover'}`}
|
||||
title="Filters"
|
||||
>
|
||||
<Filter className="w-5 h-5" />
|
||||
</button>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="h-full w-64 bg-surface border-r border-border-subtle flex flex-col animate-slide-in">
|
||||
{/* Header */}
|
||||
<div className="flex items-center justify-between px-3 py-2 border-b border-border-subtle">
|
||||
<div className="flex items-center gap-1">
|
||||
<button
|
||||
onClick={() => setActiveTab('files')}
|
||||
className={`px-2 py-1 text-xs rounded transition-colors ${
|
||||
activeTab === 'files'
|
||||
? 'bg-accent/20 text-accent'
|
||||
: 'text-text-secondary hover:text-text-primary hover:bg-hover'
|
||||
}`}
|
||||
>
|
||||
Explorer
|
||||
</button>
|
||||
<button
|
||||
onClick={() => setActiveTab('filters')}
|
||||
className={`px-2 py-1 text-xs rounded transition-colors ${
|
||||
activeTab === 'filters'
|
||||
? 'bg-accent/20 text-accent'
|
||||
: 'text-text-secondary hover:text-text-primary hover:bg-hover'
|
||||
}`}
|
||||
>
|
||||
Filters
|
||||
</button>
|
||||
</div>
|
||||
<button
|
||||
onClick={() => setIsCollapsed(true)}
|
||||
className="p-1 text-text-muted hover:text-text-primary hover:bg-hover rounded transition-colors"
|
||||
title="Collapse Panel"
|
||||
>
|
||||
<PanelLeftClose className="w-4 h-4" />
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{activeTab === 'files' && (
|
||||
<>
|
||||
{/* Search */}
|
||||
<div className="px-3 py-2 border-b border-border-subtle">
|
||||
<div className="relative">
|
||||
<Search className="absolute left-2.5 top-1/2 -translate-y-1/2 w-3.5 h-3.5 text-text-muted" />
|
||||
<input
|
||||
type="text"
|
||||
placeholder="Search files..."
|
||||
value={searchQuery}
|
||||
onChange={(e) => setSearchQuery(e.target.value)}
|
||||
className="w-full pl-8 pr-3 py-1.5 bg-elevated border border-border-subtle rounded text-xs text-text-primary placeholder:text-text-muted focus:outline-none focus:border-accent"
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* File tree */}
|
||||
<div className="flex-1 overflow-y-auto scrollbar-thin py-2">
|
||||
{fileTree.length === 0 ? (
|
||||
<div className="px-3 py-4 text-center text-text-muted text-xs">
|
||||
No files loaded
|
||||
</div>
|
||||
) : (
|
||||
fileTree.map(node => (
|
||||
<TreeItem
|
||||
key={node.id}
|
||||
node={node}
|
||||
depth={0}
|
||||
searchQuery={searchQuery}
|
||||
onNodeClick={handleNodeClick}
|
||||
expandedPaths={expandedPaths}
|
||||
toggleExpanded={toggleExpanded}
|
||||
selectedPath={selectedPath}
|
||||
/>
|
||||
))
|
||||
)}
|
||||
</div>
|
||||
</>
|
||||
)}
|
||||
|
||||
{activeTab === 'filters' && (
|
||||
<div className="flex-1 overflow-y-auto scrollbar-thin p-3">
|
||||
<div className="mb-3">
|
||||
<h3 className="text-xs font-medium text-text-secondary uppercase tracking-wide mb-2">
|
||||
Node Types
|
||||
</h3>
|
||||
<p className="text-[11px] text-text-muted mb-3">
|
||||
Toggle visibility of node types in the graph
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div className="flex flex-col gap-1">
|
||||
{FILTERABLE_LABELS.map((label) => {
|
||||
const Icon = getNodeTypeIcon(label);
|
||||
const isVisible = visibleLabels.includes(label);
|
||||
|
||||
return (
|
||||
<button
|
||||
key={label}
|
||||
onClick={() => toggleLabelVisibility(label)}
|
||||
className={`
|
||||
flex items-center gap-2.5 px-2 py-1.5 rounded text-left transition-colors
|
||||
${isVisible
|
||||
? 'bg-elevated text-text-primary'
|
||||
: 'text-text-muted hover:bg-hover hover:text-text-secondary'
|
||||
}
|
||||
`}
|
||||
>
|
||||
<div
|
||||
className={`w-5 h-5 rounded flex items-center justify-center ${isVisible ? '' : 'opacity-40'}`}
|
||||
style={{ backgroundColor: `${NODE_COLORS[label]}20` }}
|
||||
>
|
||||
<Icon className="w-3 h-3" style={{ color: NODE_COLORS[label] }} />
|
||||
</div>
|
||||
<span className="text-xs flex-1">{label}</span>
|
||||
<div
|
||||
className={`w-2 h-2 rounded-full transition-colors ${isVisible ? 'bg-accent' : 'bg-border-subtle'}`}
|
||||
/>
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
|
||||
{/* Depth Filter */}
|
||||
<div className="mt-6 pt-4 border-t border-border-subtle">
|
||||
<h3 className="text-xs font-medium text-text-secondary uppercase tracking-wide mb-2">
|
||||
<Target className="w-3 h-3 inline mr-1.5" />
|
||||
Focus Depth
|
||||
</h3>
|
||||
<p className="text-[11px] text-text-muted mb-3">
|
||||
Show nodes within N hops of selection
|
||||
</p>
|
||||
|
||||
<div className="flex flex-wrap gap-1.5">
|
||||
{[
|
||||
{ value: null, label: 'All' },
|
||||
{ value: 1, label: '1 hop' },
|
||||
{ value: 2, label: '2 hops' },
|
||||
{ value: 3, label: '3 hops' },
|
||||
{ value: 5, label: '5 hops' },
|
||||
].map(({ value, label }) => (
|
||||
<button
|
||||
key={label}
|
||||
onClick={() => setDepthFilter(value)}
|
||||
className={`
|
||||
px-2 py-1 text-xs rounded transition-colors
|
||||
${depthFilter === value
|
||||
? 'bg-accent text-white'
|
||||
: 'bg-elevated text-text-secondary hover:bg-hover hover:text-text-primary'
|
||||
}
|
||||
`}
|
||||
>
|
||||
{label}
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
|
||||
{depthFilter !== null && !selectedNode && (
|
||||
<p className="mt-2 text-[10px] text-amber-400">
|
||||
Select a node to apply depth filter
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Legend */}
|
||||
<div className="mt-6 pt-4 border-t border-border-subtle">
|
||||
<h3 className="text-xs font-medium text-text-secondary uppercase tracking-wide mb-3">
|
||||
Color Legend
|
||||
</h3>
|
||||
<div className="grid grid-cols-2 gap-2">
|
||||
{(['Folder', 'File', 'Class', 'Function', 'Interface', 'Method'] as NodeLabel[]).map(label => (
|
||||
<div key={label} className="flex items-center gap-1.5">
|
||||
<div
|
||||
className="w-2.5 h-2.5 rounded-full"
|
||||
style={{ backgroundColor: NODE_COLORS[label] }}
|
||||
/>
|
||||
<span className="text-[10px] text-text-muted">{label}</span>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Stats footer */}
|
||||
{graph && (
|
||||
<div className="px-3 py-2 border-t border-border-subtle bg-elevated/50">
|
||||
<div className="flex items-center justify-between text-[10px] text-text-muted">
|
||||
<span>{graph.nodes.length} nodes</span>
|
||||
<span>{graph.relationships.length} edges</span>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,295 @@
|
||||
import { useEffect, useCallback, useMemo, useState, forwardRef, useImperativeHandle } from 'react';
|
||||
import { ZoomIn, ZoomOut, Maximize2, Focus, RotateCcw, Play, Pause, Lightbulb, LightbulbOff } from 'lucide-react';
|
||||
import { useSigma } from '../hooks/useSigma';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { knowledgeGraphToGraphology, filterGraphByDepth, SigmaNodeAttributes, SigmaEdgeAttributes } from '../lib/graph-adapter';
|
||||
import { QueryFAB } from './QueryFAB';
|
||||
import Graph from 'graphology';
|
||||
|
||||
export interface GraphCanvasHandle {
|
||||
focusNode: (nodeId: string) => void;
|
||||
}
|
||||
|
||||
export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
const {
|
||||
graph,
|
||||
setSelectedNode,
|
||||
selectedNode: appSelectedNode,
|
||||
visibleLabels,
|
||||
openCodePanel,
|
||||
depthFilter,
|
||||
highlightedNodeIds,
|
||||
aiCitationHighlightedNodeIds,
|
||||
aiToolHighlightedNodeIds,
|
||||
blastRadiusNodeIds,
|
||||
isAIHighlightsEnabled,
|
||||
toggleAIHighlights,
|
||||
} = useAppState();
|
||||
const [hoveredNodeName, setHoveredNodeName] = useState<string | null>(null);
|
||||
|
||||
const effectiveHighlightedNodeIds = useMemo(() => {
|
||||
if (!isAIHighlightsEnabled) return highlightedNodeIds;
|
||||
const next = new Set(highlightedNodeIds);
|
||||
for (const id of aiCitationHighlightedNodeIds) next.add(id);
|
||||
for (const id of aiToolHighlightedNodeIds) next.add(id);
|
||||
// Note: blast radius nodes are handled separately with red color
|
||||
return next;
|
||||
}, [highlightedNodeIds, aiCitationHighlightedNodeIds, aiToolHighlightedNodeIds, isAIHighlightsEnabled]);
|
||||
|
||||
// Blast radius nodes (only when AI highlights enabled)
|
||||
const effectiveBlastRadiusNodeIds = useMemo(() => {
|
||||
if (!isAIHighlightsEnabled) return new Set<string>();
|
||||
return blastRadiusNodeIds;
|
||||
}, [blastRadiusNodeIds, isAIHighlightsEnabled]);
|
||||
|
||||
const handleNodeClick = useCallback((nodeId: string) => {
|
||||
if (!graph) return;
|
||||
const node = graph.nodes.find(n => n.id === nodeId);
|
||||
if (node) {
|
||||
setSelectedNode(node);
|
||||
openCodePanel();
|
||||
}
|
||||
}, [graph, setSelectedNode, openCodePanel]);
|
||||
|
||||
const handleNodeHover = useCallback((nodeId: string | null) => {
|
||||
if (!nodeId || !graph) {
|
||||
setHoveredNodeName(null);
|
||||
return;
|
||||
}
|
||||
const node = graph.nodes.find(n => n.id === nodeId);
|
||||
if (node) {
|
||||
setHoveredNodeName(node.properties.name);
|
||||
}
|
||||
}, [graph]);
|
||||
|
||||
const handleStageClick = useCallback(() => {
|
||||
setSelectedNode(null);
|
||||
}, [setSelectedNode]);
|
||||
|
||||
const {
|
||||
containerRef,
|
||||
sigmaRef,
|
||||
setGraph: setSigmaGraph,
|
||||
zoomIn,
|
||||
zoomOut,
|
||||
resetZoom,
|
||||
focusNode,
|
||||
isLayoutRunning,
|
||||
startLayout,
|
||||
stopLayout,
|
||||
selectedNode: sigmaSelectedNode,
|
||||
setSelectedNode: setSigmaSelectedNode,
|
||||
} = useSigma({
|
||||
onNodeClick: handleNodeClick,
|
||||
onNodeHover: handleNodeHover,
|
||||
onStageClick: handleStageClick,
|
||||
highlightedNodeIds: effectiveHighlightedNodeIds,
|
||||
blastRadiusNodeIds: effectiveBlastRadiusNodeIds,
|
||||
});
|
||||
|
||||
// Expose focusNode to parent via ref
|
||||
useImperativeHandle(ref, () => ({
|
||||
focusNode: (nodeId: string) => {
|
||||
// Also update app state so the selection syncs properly
|
||||
if (graph) {
|
||||
const node = graph.nodes.find(n => n.id === nodeId);
|
||||
if (node) {
|
||||
setSelectedNode(node);
|
||||
openCodePanel();
|
||||
}
|
||||
}
|
||||
focusNode(nodeId);
|
||||
}
|
||||
}), [focusNode, graph, setSelectedNode, openCodePanel]);
|
||||
|
||||
// Update Sigma graph when KnowledgeGraph changes
|
||||
useEffect(() => {
|
||||
if (!graph) return;
|
||||
const sigmaGraph = knowledgeGraphToGraphology(graph);
|
||||
setSigmaGraph(sigmaGraph);
|
||||
}, [graph, setSigmaGraph]);
|
||||
|
||||
// Update node visibility when filters change
|
||||
useEffect(() => {
|
||||
const sigma = sigmaRef.current;
|
||||
if (!sigma) return;
|
||||
|
||||
const sigmaGraph = sigma.getGraph() as Graph<SigmaNodeAttributes, SigmaEdgeAttributes>;
|
||||
if (sigmaGraph.order === 0) return; // Don't filter empty graph
|
||||
|
||||
filterGraphByDepth(sigmaGraph, appSelectedNode?.id || null, depthFilter, visibleLabels);
|
||||
sigma.refresh();
|
||||
}, [visibleLabels, depthFilter, appSelectedNode, sigmaRef]);
|
||||
|
||||
// Sync app selected node with sigma
|
||||
useEffect(() => {
|
||||
if (appSelectedNode) {
|
||||
setSigmaSelectedNode(appSelectedNode.id);
|
||||
} else {
|
||||
setSigmaSelectedNode(null);
|
||||
}
|
||||
}, [appSelectedNode, setSigmaSelectedNode]);
|
||||
|
||||
// Focus on selected node
|
||||
const handleFocusSelected = useCallback(() => {
|
||||
if (appSelectedNode) {
|
||||
focusNode(appSelectedNode.id);
|
||||
}
|
||||
}, [appSelectedNode, focusNode]);
|
||||
|
||||
// Clear selection
|
||||
const handleClearSelection = useCallback(() => {
|
||||
setSelectedNode(null);
|
||||
setSigmaSelectedNode(null);
|
||||
resetZoom();
|
||||
}, [setSelectedNode, setSigmaSelectedNode, resetZoom]);
|
||||
|
||||
return (
|
||||
<div className="relative w-full h-full bg-void">
|
||||
{/* Background gradient */}
|
||||
<div className="absolute inset-0 pointer-events-none">
|
||||
<div
|
||||
className="absolute inset-0"
|
||||
style={{
|
||||
background: `
|
||||
radial-gradient(circle at 50% 50%, rgba(124, 58, 237, 0.03) 0%, transparent 70%),
|
||||
linear-gradient(to bottom, #06060a, #0a0a10)
|
||||
`
|
||||
}}
|
||||
/>
|
||||
</div>
|
||||
|
||||
{/* Sigma container */}
|
||||
<div
|
||||
ref={containerRef}
|
||||
className="sigma-container w-full h-full cursor-grab active:cursor-grabbing"
|
||||
/>
|
||||
|
||||
{/* Hovered node tooltip - only show when NOT selected */}
|
||||
{hoveredNodeName && !sigmaSelectedNode && (
|
||||
<div className="absolute top-4 left-1/2 -translate-x-1/2 px-3 py-1.5 bg-elevated/95 border border-border-subtle rounded-lg backdrop-blur-sm z-20 pointer-events-none animate-fade-in">
|
||||
<span className="font-mono text-sm text-text-primary">{hoveredNodeName}</span>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Selection info bar */}
|
||||
{sigmaSelectedNode && appSelectedNode && (
|
||||
<div className="absolute top-4 left-1/2 -translate-x-1/2 flex items-center gap-2 px-4 py-2 bg-accent/20 border border-accent/30 rounded-xl backdrop-blur-sm z-20 animate-slide-up">
|
||||
<div className="w-2 h-2 bg-accent rounded-full animate-pulse" />
|
||||
<span className="font-mono text-sm text-text-primary">
|
||||
{appSelectedNode.properties.name}
|
||||
</span>
|
||||
<span className="text-xs text-text-muted">
|
||||
({appSelectedNode.label})
|
||||
</span>
|
||||
<button
|
||||
onClick={handleClearSelection}
|
||||
className="ml-2 px-2 py-0.5 text-xs text-text-secondary hover:text-text-primary hover:bg-white/10 rounded transition-colors"
|
||||
>
|
||||
Clear
|
||||
</button>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Graph Controls - Bottom Right */}
|
||||
<div className="absolute bottom-4 right-4 flex flex-col gap-1 z-10">
|
||||
<button
|
||||
onClick={zoomIn}
|
||||
className="w-9 h-9 flex items-center justify-center bg-elevated border border-border-subtle rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors"
|
||||
title="Zoom In"
|
||||
>
|
||||
<ZoomIn className="w-4 h-4" />
|
||||
</button>
|
||||
<button
|
||||
onClick={zoomOut}
|
||||
className="w-9 h-9 flex items-center justify-center bg-elevated border border-border-subtle rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors"
|
||||
title="Zoom Out"
|
||||
>
|
||||
<ZoomOut className="w-4 h-4" />
|
||||
</button>
|
||||
<button
|
||||
onClick={resetZoom}
|
||||
className="w-9 h-9 flex items-center justify-center bg-elevated border border-border-subtle rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors"
|
||||
title="Fit to Screen"
|
||||
>
|
||||
<Maximize2 className="w-4 h-4" />
|
||||
</button>
|
||||
|
||||
{/* Divider */}
|
||||
<div className="h-px bg-border-subtle my-1" />
|
||||
|
||||
{/* Focus on selected */}
|
||||
{appSelectedNode && (
|
||||
<button
|
||||
onClick={handleFocusSelected}
|
||||
className="w-9 h-9 flex items-center justify-center bg-accent/20 border border-accent/30 rounded-md text-accent hover:bg-accent/30 transition-colors"
|
||||
title="Focus on Selected Node"
|
||||
>
|
||||
<Focus className="w-4 h-4" />
|
||||
</button>
|
||||
)}
|
||||
|
||||
{/* Clear selection */}
|
||||
{sigmaSelectedNode && (
|
||||
<button
|
||||
onClick={handleClearSelection}
|
||||
className="w-9 h-9 flex items-center justify-center bg-elevated border border-border-subtle rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors"
|
||||
title="Clear Selection"
|
||||
>
|
||||
<RotateCcw className="w-4 h-4" />
|
||||
</button>
|
||||
)}
|
||||
|
||||
{/* Divider */}
|
||||
<div className="h-px bg-border-subtle my-1" />
|
||||
|
||||
{/* Layout control */}
|
||||
<button
|
||||
onClick={isLayoutRunning ? stopLayout : startLayout}
|
||||
className={`
|
||||
w-9 h-9 flex items-center justify-center border rounded-md transition-all
|
||||
${isLayoutRunning
|
||||
? 'bg-accent border-accent text-white shadow-glow animate-pulse'
|
||||
: 'bg-elevated border-border-subtle text-text-secondary hover:bg-hover hover:text-text-primary'
|
||||
}
|
||||
`}
|
||||
title={isLayoutRunning ? 'Stop Layout' : 'Run Layout Again'}
|
||||
>
|
||||
{isLayoutRunning ? (
|
||||
<Pause className="w-4 h-4" />
|
||||
) : (
|
||||
<Play className="w-4 h-4" />
|
||||
)}
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Layout running indicator */}
|
||||
{isLayoutRunning && (
|
||||
<div className="absolute bottom-4 left-1/2 -translate-x-1/2 flex items-center gap-2 px-3 py-1.5 bg-emerald-500/20 border border-emerald-500/30 rounded-full backdrop-blur-sm z-10 animate-fade-in">
|
||||
<div className="w-2 h-2 bg-emerald-400 rounded-full animate-ping" />
|
||||
<span className="text-xs text-emerald-400 font-medium">Layout optimizing...</span>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Query FAB */}
|
||||
<QueryFAB />
|
||||
|
||||
{/* AI Highlights toggle - Top Right */}
|
||||
<div className="absolute top-4 right-4 z-20">
|
||||
<button
|
||||
onClick={toggleAIHighlights}
|
||||
className={
|
||||
isAIHighlightsEnabled
|
||||
? 'w-10 h-10 flex items-center justify-center bg-cyan-500/15 border border-cyan-400/40 rounded-lg text-cyan-200 hover:bg-cyan-500/20 hover:border-cyan-300/60 transition-colors'
|
||||
: 'w-10 h-10 flex items-center justify-center bg-elevated border border-border-subtle rounded-lg text-text-muted hover:bg-hover hover:text-text-primary transition-colors'
|
||||
}
|
||||
title={isAIHighlightsEnabled ? 'Turn off AI highlights' : 'Turn on AI highlights'}
|
||||
>
|
||||
{isAIHighlightsEnabled ? <Lightbulb className="w-4 h-4" /> : <LightbulbOff className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
});
|
||||
|
||||
GraphCanvas.displayName = 'GraphCanvas';
|
||||
@@ -0,0 +1,239 @@
|
||||
import { Search, Settings, HelpCircle, Sparkles, Github, Star } from 'lucide-react';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { useState, useMemo, useRef, useEffect } from 'react';
|
||||
import { GraphNode } from '../core/graph/types';
|
||||
import { EmbeddingStatus } from './EmbeddingStatus';
|
||||
|
||||
// Color mapping for node types in search results
|
||||
const NODE_TYPE_COLORS: Record<string, string> = {
|
||||
Folder: '#6366f1',
|
||||
File: '#3b82f6',
|
||||
Function: '#10b981',
|
||||
Class: '#f59e0b',
|
||||
Method: '#14b8a6',
|
||||
Interface: '#ec4899',
|
||||
Variable: '#64748b',
|
||||
Import: '#475569',
|
||||
Type: '#a78bfa',
|
||||
};
|
||||
|
||||
interface HeaderProps {
|
||||
onFocusNode?: (nodeId: string) => void;
|
||||
}
|
||||
|
||||
export const Header = ({ onFocusNode }: HeaderProps) => {
|
||||
const { projectName, graph, openChatPanel, isRightPanelOpen, rightPanelTab, setSettingsPanelOpen } = useAppState();
|
||||
const [searchQuery, setSearchQuery] = useState('');
|
||||
const [isSearchOpen, setIsSearchOpen] = useState(false);
|
||||
const [selectedIndex, setSelectedIndex] = useState(0);
|
||||
const searchRef = useRef<HTMLDivElement>(null);
|
||||
const inputRef = useRef<HTMLInputElement>(null);
|
||||
|
||||
const nodeCount = graph?.nodes.length ?? 0;
|
||||
const edgeCount = graph?.relationships.length ?? 0;
|
||||
|
||||
// Search results - filter nodes by name
|
||||
const searchResults = useMemo(() => {
|
||||
if (!graph || !searchQuery.trim()) return [];
|
||||
|
||||
const query = searchQuery.toLowerCase();
|
||||
return graph.nodes
|
||||
.filter(node => node.properties.name.toLowerCase().includes(query))
|
||||
.slice(0, 10); // Limit to 10 results
|
||||
}, [graph, searchQuery]);
|
||||
|
||||
// Handle clicking outside to close dropdown
|
||||
useEffect(() => {
|
||||
const handleClickOutside = (e: MouseEvent) => {
|
||||
if (searchRef.current && !searchRef.current.contains(e.target as Node)) {
|
||||
setIsSearchOpen(false);
|
||||
}
|
||||
};
|
||||
document.addEventListener('mousedown', handleClickOutside);
|
||||
return () => document.removeEventListener('mousedown', handleClickOutside);
|
||||
}, []);
|
||||
|
||||
// Keyboard shortcut (Cmd+K / Ctrl+K)
|
||||
useEffect(() => {
|
||||
const handleKeyDown = (e: KeyboardEvent) => {
|
||||
if ((e.metaKey || e.ctrlKey) && e.key === 'k') {
|
||||
e.preventDefault();
|
||||
inputRef.current?.focus();
|
||||
setIsSearchOpen(true);
|
||||
}
|
||||
if (e.key === 'Escape') {
|
||||
setIsSearchOpen(false);
|
||||
inputRef.current?.blur();
|
||||
}
|
||||
};
|
||||
document.addEventListener('keydown', handleKeyDown);
|
||||
return () => document.removeEventListener('keydown', handleKeyDown);
|
||||
}, []);
|
||||
|
||||
// Handle keyboard navigation in results
|
||||
const handleKeyDown = (e: React.KeyboardEvent) => {
|
||||
if (!isSearchOpen || searchResults.length === 0) return;
|
||||
|
||||
if (e.key === 'ArrowDown') {
|
||||
e.preventDefault();
|
||||
setSelectedIndex(i => Math.min(i + 1, searchResults.length - 1));
|
||||
} else if (e.key === 'ArrowUp') {
|
||||
e.preventDefault();
|
||||
setSelectedIndex(i => Math.max(i - 1, 0));
|
||||
} else if (e.key === 'Enter') {
|
||||
e.preventDefault();
|
||||
const selected = searchResults[selectedIndex];
|
||||
if (selected) {
|
||||
handleSelectNode(selected);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
const handleSelectNode = (node: GraphNode) => {
|
||||
// onFocusNode handles both camera focus AND selection in useSigma
|
||||
onFocusNode?.(node.id);
|
||||
setSearchQuery('');
|
||||
setIsSearchOpen(false);
|
||||
setSelectedIndex(0);
|
||||
};
|
||||
|
||||
return (
|
||||
<header className="flex items-center justify-between px-5 py-3 bg-deep border-b border-dashed border-border-subtle">
|
||||
{/* Left section */}
|
||||
<div className="flex items-center gap-4">
|
||||
{/* Logo */}
|
||||
<div className="flex items-center gap-2.5">
|
||||
<div className="w-7 h-7 flex items-center justify-center bg-gradient-to-br from-accent to-node-interface rounded-md shadow-glow text-white text-sm font-bold">
|
||||
◇
|
||||
</div>
|
||||
<span className="font-semibold text-[15px] tracking-tight">GitNexus</span>
|
||||
</div>
|
||||
|
||||
{/* Project badge */}
|
||||
{projectName && (
|
||||
<div className="flex items-center gap-2 px-3 py-1.5 bg-surface border border-border-subtle rounded-lg text-sm text-text-secondary">
|
||||
<span className="w-1.5 h-1.5 bg-node-function rounded-full animate-pulse" />
|
||||
<span className="truncate max-w-[200px]">{projectName}</span>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Center - Search */}
|
||||
<div className="flex-1 max-w-md mx-6 relative" ref={searchRef}>
|
||||
<div className="flex items-center gap-2.5 px-3.5 py-2 bg-surface border border-border-subtle rounded-lg transition-all focus-within:border-accent focus-within:ring-2 focus-within:ring-accent/20">
|
||||
<Search className="w-4 h-4 text-text-muted flex-shrink-0" />
|
||||
<input
|
||||
ref={inputRef}
|
||||
type="text"
|
||||
placeholder="Search nodes..."
|
||||
value={searchQuery}
|
||||
onChange={(e) => {
|
||||
setSearchQuery(e.target.value);
|
||||
setIsSearchOpen(true);
|
||||
setSelectedIndex(0);
|
||||
}}
|
||||
onFocus={() => setIsSearchOpen(true)}
|
||||
onKeyDown={handleKeyDown}
|
||||
className="flex-1 bg-transparent border-none outline-none text-sm text-text-primary placeholder:text-text-muted"
|
||||
/>
|
||||
<kbd className="px-1.5 py-0.5 bg-elevated border border-border-subtle rounded text-[10px] text-text-muted font-mono">
|
||||
⌘K
|
||||
</kbd>
|
||||
</div>
|
||||
|
||||
{/* Search Results Dropdown */}
|
||||
{isSearchOpen && searchQuery.trim() && (
|
||||
<div className="absolute top-full left-0 right-0 mt-1 bg-surface border border-border-subtle rounded-lg shadow-xl overflow-hidden z-50">
|
||||
{searchResults.length === 0 ? (
|
||||
<div className="px-4 py-3 text-sm text-text-muted">
|
||||
No nodes found for "{searchQuery}"
|
||||
</div>
|
||||
) : (
|
||||
<div className="max-h-80 overflow-y-auto">
|
||||
{searchResults.map((node, index) => (
|
||||
<button
|
||||
key={node.id}
|
||||
onClick={() => handleSelectNode(node)}
|
||||
className={`w-full px-4 py-2.5 flex items-center gap-3 text-left transition-colors ${index === selectedIndex
|
||||
? 'bg-accent/20 text-text-primary'
|
||||
: 'hover:bg-hover text-text-secondary'
|
||||
}`}
|
||||
>
|
||||
{/* Node type indicator */}
|
||||
<span
|
||||
className="w-2.5 h-2.5 rounded-full flex-shrink-0"
|
||||
style={{ backgroundColor: NODE_TYPE_COLORS[node.label] || '#6b7280' }}
|
||||
/>
|
||||
{/* Node name */}
|
||||
<span className="flex-1 truncate text-sm font-medium">
|
||||
{node.properties.name}
|
||||
</span>
|
||||
{/* Node type badge */}
|
||||
<span className="text-xs text-text-muted px-2 py-0.5 bg-elevated rounded">
|
||||
{node.label}
|
||||
</span>
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Right section */}
|
||||
<div className="flex items-center gap-2">
|
||||
{/* GitHub Star Button */}
|
||||
<a
|
||||
href="https://github.com/abhigyanpatwari/GitNexus"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="flex items-center gap-2 px-3.5 py-2 bg-gradient-to-r from-purple-600 to-pink-600 hover:from-purple-500 hover:to-pink-500 rounded-lg text-white text-sm font-medium shadow-lg hover:shadow-xl hover:-translate-y-0.5 transition-all duration-200 group"
|
||||
>
|
||||
<Github className="w-4 h-4" />
|
||||
<span className="hidden sm:inline">Star if cool</span>
|
||||
<Star className="w-3.5 h-3.5 group-hover:fill-yellow-300 group-hover:text-yellow-300 transition-all" />
|
||||
<span className="hidden sm:inline">✨</span>
|
||||
</a>
|
||||
|
||||
{/* Stats */}
|
||||
{graph && (
|
||||
<div className="flex items-center gap-4 mr-2 text-xs text-text-muted">
|
||||
<span>{nodeCount} nodes</span>
|
||||
<span>{edgeCount} edges</span>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Embedding Status */}
|
||||
<EmbeddingStatus />
|
||||
|
||||
{/* Icon buttons */}
|
||||
<button
|
||||
onClick={() => setSettingsPanelOpen(true)}
|
||||
className="w-9 h-9 flex items-center justify-center rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors"
|
||||
title="AI Settings"
|
||||
>
|
||||
<Settings className="w-[18px] h-[18px]" />
|
||||
</button>
|
||||
<button className="w-9 h-9 flex items-center justify-center rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors">
|
||||
<HelpCircle className="w-[18px] h-[18px]" />
|
||||
</button>
|
||||
|
||||
{/* AI Button */}
|
||||
<button
|
||||
onClick={openChatPanel}
|
||||
className={`
|
||||
flex items-center gap-1.5 px-3.5 py-2 rounded-lg text-sm font-medium transition-all
|
||||
${isRightPanelOpen && rightPanelTab === 'chat'
|
||||
? 'bg-accent text-white shadow-glow'
|
||||
: 'bg-gradient-to-r from-accent to-accent-dim text-white shadow-glow hover:shadow-lg hover:-translate-y-0.5'
|
||||
}
|
||||
`}
|
||||
>
|
||||
<Sparkles className="w-4 h-4" />
|
||||
<span>Nexus AI</span>
|
||||
</button>
|
||||
</div>
|
||||
</header>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
import { PipelineProgress } from '../types/pipeline';
|
||||
|
||||
interface LoadingOverlayProps {
|
||||
progress: PipelineProgress;
|
||||
}
|
||||
|
||||
export const LoadingOverlay = ({ progress }: LoadingOverlayProps) => {
|
||||
return (
|
||||
<div className="fixed inset-0 flex flex-col items-center justify-center bg-void z-50">
|
||||
{/* Background gradient effects */}
|
||||
<div className="absolute inset-0 pointer-events-none">
|
||||
<div className="absolute top-1/3 left-1/3 w-96 h-96 bg-accent/10 rounded-full blur-3xl animate-pulse" />
|
||||
<div className="absolute bottom-1/3 right-1/3 w-96 h-96 bg-node-interface/10 rounded-full blur-3xl animate-pulse" />
|
||||
</div>
|
||||
|
||||
{/* Pulsing orb */}
|
||||
<div className="relative mb-10">
|
||||
<div className="w-28 h-28 bg-gradient-to-br from-accent to-node-interface rounded-full animate-pulse-glow" />
|
||||
<div className="absolute inset-0 w-28 h-28 bg-gradient-to-br from-accent to-node-interface rounded-full blur-xl opacity-50" />
|
||||
</div>
|
||||
|
||||
{/* Progress bar */}
|
||||
<div className="w-80 mb-4">
|
||||
<div className="h-1.5 bg-elevated rounded-full overflow-hidden">
|
||||
<div
|
||||
className="h-full bg-gradient-to-r from-accent to-node-interface rounded-full transition-all duration-300 ease-out"
|
||||
style={{ width: `${progress.percent}%` }}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Status text */}
|
||||
<div className="text-center">
|
||||
<p className="font-mono text-sm text-text-secondary mb-1">
|
||||
{progress.message}
|
||||
<span className="animate-pulse">|</span>
|
||||
</p>
|
||||
{progress.detail && (
|
||||
<p className="font-mono text-xs text-text-muted truncate max-w-md">
|
||||
{progress.detail}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Stats */}
|
||||
{progress.stats && (
|
||||
<div className="mt-8 flex items-center gap-6 text-xs text-text-muted">
|
||||
<div className="flex items-center gap-2">
|
||||
<span className="w-2 h-2 bg-node-file rounded-full" />
|
||||
<span>{progress.stats.filesProcessed} / {progress.stats.totalFiles} files</span>
|
||||
</div>
|
||||
<div className="flex items-center gap-2">
|
||||
<span className="w-2 h-2 bg-node-function rounded-full" />
|
||||
<span>{progress.stats.nodesCreated} nodes</span>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Percent */}
|
||||
<p className="mt-4 font-mono text-3xl font-semibold text-text-primary">
|
||||
{progress.percent}%
|
||||
</p>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
import { useEffect, useRef, useState } from 'react';
|
||||
import mermaid from 'mermaid';
|
||||
import { AlertTriangle, Maximize2, Minimize2 } from 'lucide-react';
|
||||
|
||||
// Initialize mermaid with dark theme
|
||||
mermaid.initialize({
|
||||
startOnLoad: false,
|
||||
theme: 'dark',
|
||||
themeVariables: {
|
||||
primaryColor: '#06b6d4',
|
||||
primaryTextColor: '#e4e4ed',
|
||||
primaryBorderColor: '#1e1e2a',
|
||||
lineColor: '#3b3b54',
|
||||
secondaryColor: '#1e1e2a',
|
||||
tertiaryColor: '#0a0a10',
|
||||
background: '#0a0a10',
|
||||
mainBkg: '#0f0f18',
|
||||
nodeBorder: '#3b3b54',
|
||||
clusterBkg: '#1e1e2a',
|
||||
titleColor: '#e4e4ed',
|
||||
edgeLabelBackground: '#0f0f18',
|
||||
nodeTextColor: '#e4e4ed',
|
||||
},
|
||||
flowchart: {
|
||||
curve: 'basis',
|
||||
padding: 15,
|
||||
nodeSpacing: 50,
|
||||
rankSpacing: 50,
|
||||
},
|
||||
sequence: {
|
||||
actorMargin: 50,
|
||||
boxMargin: 10,
|
||||
boxTextMargin: 5,
|
||||
noteMargin: 10,
|
||||
messageMargin: 35,
|
||||
},
|
||||
fontFamily: '"JetBrains Mono", "Fira Code", monospace',
|
||||
fontSize: 13,
|
||||
});
|
||||
|
||||
interface MermaidDiagramProps {
|
||||
code: string;
|
||||
}
|
||||
|
||||
export const MermaidDiagram = ({ code }: MermaidDiagramProps) => {
|
||||
const containerRef = useRef<HTMLDivElement>(null);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
const [isExpanded, setIsExpanded] = useState(false);
|
||||
const [svg, setSvg] = useState<string>('');
|
||||
|
||||
useEffect(() => {
|
||||
const renderDiagram = async () => {
|
||||
if (!containerRef.current) return;
|
||||
|
||||
try {
|
||||
// Generate unique ID for this diagram
|
||||
const id = `mermaid-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`;
|
||||
|
||||
// Render the diagram
|
||||
const { svg: renderedSvg } = await mermaid.render(id, code.trim());
|
||||
setSvg(renderedSvg);
|
||||
setError(null);
|
||||
} catch (err) {
|
||||
console.error('Mermaid render error:', err);
|
||||
setError(err instanceof Error ? err.message : 'Failed to render diagram');
|
||||
setSvg('');
|
||||
}
|
||||
};
|
||||
|
||||
renderDiagram();
|
||||
}, [code]);
|
||||
|
||||
if (error) {
|
||||
return (
|
||||
<div className="my-3 p-4 bg-rose-500/10 border border-rose-500/30 rounded-lg">
|
||||
<div className="flex items-center gap-2 text-rose-300 text-sm mb-2">
|
||||
<AlertTriangle className="w-4 h-4" />
|
||||
<span className="font-medium">Diagram Error</span>
|
||||
</div>
|
||||
<pre className="text-xs text-rose-200/70 font-mono whitespace-pre-wrap">{error}</pre>
|
||||
<details className="mt-2">
|
||||
<summary className="text-xs text-text-muted cursor-pointer hover:text-text-secondary">
|
||||
Show source
|
||||
</summary>
|
||||
<pre className="mt-2 p-2 bg-surface rounded text-xs text-text-muted overflow-x-auto">
|
||||
{code}
|
||||
</pre>
|
||||
</details>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className={`my-3 relative group ${isExpanded ? 'fixed inset-4 z-50' : ''}`}>
|
||||
{/* Backdrop for expanded view */}
|
||||
{isExpanded && (
|
||||
<div
|
||||
className="absolute inset-0 -m-4 bg-deep/95 backdrop-blur-sm"
|
||||
onClick={() => setIsExpanded(false)}
|
||||
/>
|
||||
)}
|
||||
|
||||
<div className={`
|
||||
relative bg-gradient-to-b from-surface to-elevated
|
||||
border border-border-subtle rounded-xl overflow-hidden
|
||||
${isExpanded ? 'h-full' : ''}
|
||||
`}>
|
||||
{/* Header */}
|
||||
<div className="flex items-center justify-between px-3 py-2 bg-surface/60 border-b border-border-subtle">
|
||||
<span className="text-[10px] text-text-muted uppercase tracking-wider font-medium">
|
||||
Diagram
|
||||
</span>
|
||||
<button
|
||||
onClick={() => setIsExpanded(!isExpanded)}
|
||||
className="p-1 text-text-muted hover:text-text-primary hover:bg-hover rounded transition-colors"
|
||||
title={isExpanded ? 'Minimize' : 'Expand'}
|
||||
>
|
||||
{isExpanded ? (
|
||||
<Minimize2 className="w-3.5 h-3.5" />
|
||||
) : (
|
||||
<Maximize2 className="w-3.5 h-3.5" />
|
||||
)}
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Diagram container */}
|
||||
<div
|
||||
ref={containerRef}
|
||||
className={`
|
||||
flex items-center justify-center p-4 overflow-auto
|
||||
${isExpanded ? 'h-[calc(100%-40px)]' : 'max-h-[400px]'}
|
||||
`}
|
||||
dangerouslySetInnerHTML={{ __html: svg }}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,405 @@
|
||||
import { useState, useRef, useEffect, useCallback } from 'react';
|
||||
import { Terminal, Play, X, ChevronDown, ChevronUp, Loader2, Sparkles, Table } from 'lucide-react';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
|
||||
const EXAMPLE_QUERIES = [
|
||||
{
|
||||
label: 'All Functions',
|
||||
query: `MATCH (n:Function) RETURN n.id AS id, n.name AS name, n.filePath AS path LIMIT 50`,
|
||||
},
|
||||
{
|
||||
label: 'All Classes',
|
||||
query: `MATCH (n:Class) RETURN n.id AS id, n.name AS name, n.filePath AS path LIMIT 50`,
|
||||
},
|
||||
{
|
||||
label: 'All Interfaces',
|
||||
query: `MATCH (n:Interface) RETURN n.id AS id, n.name AS name, n.filePath AS path LIMIT 50`,
|
||||
},
|
||||
{
|
||||
label: 'Function Calls',
|
||||
query: `MATCH (a:File)-[r:CodeRelation {type: 'CALLS'}]->(b:Function) RETURN a.id AS id, a.name AS caller, b.name AS callee LIMIT 50`,
|
||||
},
|
||||
{
|
||||
label: 'Import Dependencies',
|
||||
query: `MATCH (a:File)-[r:CodeRelation {type: 'IMPORTS'}]->(b:File) RETURN a.id AS id, a.name AS from, b.name AS imports LIMIT 50`,
|
||||
},
|
||||
];
|
||||
|
||||
export const QueryFAB = () => {
|
||||
const { setHighlightedNodeIds, setQueryResult, queryResult, clearQueryHighlights, graph, runQuery, isDatabaseReady } = useAppState();
|
||||
|
||||
const [isExpanded, setIsExpanded] = useState(false);
|
||||
const [query, setQuery] = useState('');
|
||||
const [isRunning, setIsRunning] = useState(false);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
const [showExamples, setShowExamples] = useState(false);
|
||||
const [showResults, setShowResults] = useState(true);
|
||||
|
||||
const textareaRef = useRef<HTMLTextAreaElement>(null);
|
||||
const panelRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
useEffect(() => {
|
||||
if (isExpanded && textareaRef.current) {
|
||||
textareaRef.current.focus();
|
||||
}
|
||||
}, [isExpanded]);
|
||||
|
||||
useEffect(() => {
|
||||
const handleClickOutside = (e: MouseEvent) => {
|
||||
if (panelRef.current && !panelRef.current.contains(e.target as Node)) {
|
||||
setShowExamples(false);
|
||||
}
|
||||
};
|
||||
document.addEventListener('mousedown', handleClickOutside);
|
||||
return () => document.removeEventListener('mousedown', handleClickOutside);
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
const handleKeyDown = (e: KeyboardEvent) => {
|
||||
if (e.key === 'Escape' && isExpanded) {
|
||||
setIsExpanded(false);
|
||||
setShowExamples(false);
|
||||
}
|
||||
};
|
||||
document.addEventListener('keydown', handleKeyDown);
|
||||
return () => document.removeEventListener('keydown', handleKeyDown);
|
||||
}, [isExpanded]);
|
||||
|
||||
const handleRunQuery = useCallback(async () => {
|
||||
if (!query.trim() || isRunning) return;
|
||||
|
||||
if (!graph) {
|
||||
setError('No project loaded. Load a project first.');
|
||||
return;
|
||||
}
|
||||
|
||||
const ready = await isDatabaseReady();
|
||||
if (!ready) {
|
||||
setError('Database not ready. Please wait for loading to complete.');
|
||||
return;
|
||||
}
|
||||
|
||||
setIsRunning(true);
|
||||
setError(null);
|
||||
|
||||
const startTime = performance.now();
|
||||
|
||||
try {
|
||||
const rows = await runQuery(query);
|
||||
const executionTime = performance.now() - startTime;
|
||||
|
||||
// Extract node IDs from results - handles various formats
|
||||
// 1. Array format: first element if it looks like a node ID
|
||||
// 2. Object format: any field ending with 'id' (case-insensitive)
|
||||
// 3. Values matching node ID pattern: Label:path:name
|
||||
const nodeIdPattern = /^(File|Function|Class|Method|Interface|Folder|CodeElement):/;
|
||||
|
||||
const nodeIds = rows
|
||||
.flatMap(row => {
|
||||
const ids: string[] = [];
|
||||
|
||||
if (Array.isArray(row)) {
|
||||
// Array format - check all elements for node ID patterns
|
||||
row.forEach(val => {
|
||||
if (typeof val === 'string' && (nodeIdPattern.test(val) || val.includes(':'))) {
|
||||
ids.push(val);
|
||||
}
|
||||
});
|
||||
} else if (typeof row === 'object' && row !== null) {
|
||||
// Object format - check fields ending with 'id' and values matching patterns
|
||||
Object.entries(row).forEach(([key, val]) => {
|
||||
const keyLower = key.toLowerCase();
|
||||
if (typeof val === 'string') {
|
||||
// Field name contains 'id'
|
||||
if (keyLower.includes('id') || keyLower === 'id') {
|
||||
ids.push(val);
|
||||
}
|
||||
// Value matches node ID pattern
|
||||
else if (nodeIdPattern.test(val)) {
|
||||
ids.push(val);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
return ids;
|
||||
})
|
||||
.filter(Boolean)
|
||||
.filter((id, index, arr) => arr.indexOf(id) === index);
|
||||
|
||||
setQueryResult({ rows, nodeIds, executionTime });
|
||||
setHighlightedNodeIds(new Set(nodeIds));
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : 'Query execution failed');
|
||||
setQueryResult(null);
|
||||
setHighlightedNodeIds(new Set());
|
||||
} finally {
|
||||
setIsRunning(false);
|
||||
}
|
||||
}, [query, isRunning, graph, isDatabaseReady, runQuery, setHighlightedNodeIds, setQueryResult]);
|
||||
|
||||
const handleKeyDown = (e: React.KeyboardEvent) => {
|
||||
if (e.key === 'Enter' && (e.ctrlKey || e.metaKey)) {
|
||||
e.preventDefault();
|
||||
handleRunQuery();
|
||||
}
|
||||
};
|
||||
|
||||
const handleSelectExample = (exampleQuery: string) => {
|
||||
setQuery(exampleQuery);
|
||||
setShowExamples(false);
|
||||
textareaRef.current?.focus();
|
||||
};
|
||||
|
||||
const handleClose = () => {
|
||||
setIsExpanded(false);
|
||||
setShowExamples(false);
|
||||
clearQueryHighlights();
|
||||
setError(null);
|
||||
};
|
||||
|
||||
const handleClear = () => {
|
||||
setQuery('');
|
||||
clearQueryHighlights();
|
||||
setError(null);
|
||||
textareaRef.current?.focus();
|
||||
};
|
||||
|
||||
if (!isExpanded) {
|
||||
return (
|
||||
<button
|
||||
onClick={() => setIsExpanded(true)}
|
||||
className="
|
||||
group absolute bottom-4 left-4 z-20
|
||||
flex items-center gap-2 px-4 py-2.5
|
||||
bg-gradient-to-r from-cyan-500 to-teal-500
|
||||
rounded-xl text-white font-medium text-sm
|
||||
shadow-[0_0_20px_rgba(6,182,212,0.4)]
|
||||
hover:shadow-[0_0_30px_rgba(6,182,212,0.6)]
|
||||
hover:-translate-y-0.5
|
||||
transition-all duration-200
|
||||
"
|
||||
>
|
||||
<Terminal className="w-4 h-4" />
|
||||
<span>Query</span>
|
||||
{queryResult && queryResult.nodeIds.length > 0 && (
|
||||
<span className="
|
||||
px-1.5 py-0.5 ml-1
|
||||
bg-white/20 rounded-md
|
||||
text-xs font-semibold
|
||||
">
|
||||
{queryResult.nodeIds.length}
|
||||
</span>
|
||||
)}
|
||||
</button>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div
|
||||
ref={panelRef}
|
||||
className="
|
||||
absolute bottom-4 left-4 z-20
|
||||
w-[480px] max-w-[calc(100%-2rem)]
|
||||
bg-deep/95 backdrop-blur-md
|
||||
border border-cyan-500/30
|
||||
rounded-xl
|
||||
shadow-[0_0_40px_rgba(6,182,212,0.2)]
|
||||
animate-fade-in
|
||||
"
|
||||
>
|
||||
<div className="flex items-center justify-between px-4 py-3 border-b border-border-subtle">
|
||||
<div className="flex items-center gap-2">
|
||||
<div className="w-7 h-7 flex items-center justify-center bg-gradient-to-br from-cyan-500 to-teal-500 rounded-lg">
|
||||
<Terminal className="w-4 h-4 text-white" />
|
||||
</div>
|
||||
<span className="font-medium text-sm">Cypher Query</span>
|
||||
</div>
|
||||
<button
|
||||
onClick={handleClose}
|
||||
className="p-1.5 text-text-muted hover:text-text-primary hover:bg-hover rounded-md transition-colors"
|
||||
>
|
||||
<X className="w-4 h-4" />
|
||||
</button>
|
||||
</div>
|
||||
|
||||
<div className="p-3">
|
||||
<div className="relative">
|
||||
<textarea
|
||||
ref={textareaRef}
|
||||
value={query}
|
||||
onChange={(e) => setQuery(e.target.value)}
|
||||
onKeyDown={handleKeyDown}
|
||||
placeholder="MATCH (n:Function) RETURN n.name, n.filePath LIMIT 10"
|
||||
rows={3}
|
||||
className="
|
||||
w-full px-3 py-2.5
|
||||
bg-surface border border-border-subtle rounded-lg
|
||||
text-sm font-mono text-text-primary
|
||||
placeholder:text-text-muted
|
||||
focus:border-cyan-500/50 focus:ring-2 focus:ring-cyan-500/20
|
||||
outline-none resize-none
|
||||
transition-all
|
||||
"
|
||||
/>
|
||||
</div>
|
||||
|
||||
<div className="flex items-center justify-between mt-3">
|
||||
<div className="relative">
|
||||
<button
|
||||
onClick={() => setShowExamples(!showExamples)}
|
||||
className="
|
||||
flex items-center gap-1.5 px-3 py-1.5
|
||||
text-xs text-text-secondary
|
||||
hover:text-text-primary hover:bg-hover
|
||||
rounded-md transition-colors
|
||||
"
|
||||
>
|
||||
<Sparkles className="w-3.5 h-3.5" />
|
||||
<span>Examples</span>
|
||||
<ChevronDown className={`w-3.5 h-3.5 transition-transform ${showExamples ? 'rotate-180' : ''}`} />
|
||||
</button>
|
||||
|
||||
{showExamples && (
|
||||
<div className="
|
||||
absolute bottom-full left-0 mb-2
|
||||
w-64 py-1
|
||||
bg-surface border border-border-subtle rounded-lg
|
||||
shadow-xl
|
||||
animate-fade-in
|
||||
">
|
||||
{EXAMPLE_QUERIES.map((example) => (
|
||||
<button
|
||||
key={example.label}
|
||||
onClick={() => handleSelectExample(example.query)}
|
||||
className="
|
||||
w-full px-3 py-2 text-left
|
||||
text-sm text-text-secondary
|
||||
hover:bg-hover hover:text-text-primary
|
||||
transition-colors
|
||||
"
|
||||
>
|
||||
{example.label}
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className="flex items-center gap-2">
|
||||
{query && (
|
||||
<button
|
||||
onClick={handleClear}
|
||||
className="
|
||||
px-3 py-1.5
|
||||
text-xs text-text-secondary
|
||||
hover:text-text-primary hover:bg-hover
|
||||
rounded-md transition-colors
|
||||
"
|
||||
>
|
||||
Clear
|
||||
</button>
|
||||
)}
|
||||
<button
|
||||
onClick={handleRunQuery}
|
||||
disabled={!query.trim() || isRunning}
|
||||
className="
|
||||
flex items-center gap-1.5 px-4 py-1.5
|
||||
bg-gradient-to-r from-cyan-500 to-teal-500
|
||||
rounded-md text-white text-sm font-medium
|
||||
shadow-[0_0_15px_rgba(6,182,212,0.3)]
|
||||
hover:shadow-[0_0_20px_rgba(6,182,212,0.5)]
|
||||
disabled:opacity-50 disabled:cursor-not-allowed disabled:shadow-none
|
||||
transition-all
|
||||
"
|
||||
>
|
||||
{isRunning ? (
|
||||
<Loader2 className="w-3.5 h-3.5 animate-spin" />
|
||||
) : (
|
||||
<Play className="w-3.5 h-3.5" />
|
||||
)}
|
||||
<span>Run</span>
|
||||
<kbd className="ml-1 px-1 py-0.5 bg-white/20 rounded text-[10px]">⌘↵</kbd>
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{error && (
|
||||
<div className="px-4 py-2 bg-red-500/10 border-t border-red-500/20">
|
||||
<p className="text-xs text-red-400 font-mono">{error}</p>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{queryResult && !error && (
|
||||
<div className="border-t border-cyan-500/20">
|
||||
<div className="px-4 py-2.5 bg-cyan-500/5 flex items-center justify-between">
|
||||
<div className="flex items-center gap-3 text-xs">
|
||||
<span className="text-text-secondary">
|
||||
<span className="text-cyan-400 font-semibold">{queryResult.rows.length}</span> rows
|
||||
</span>
|
||||
{queryResult.nodeIds.length > 0 && (
|
||||
<span className="text-text-secondary">
|
||||
<span className="text-cyan-400 font-semibold">{queryResult.nodeIds.length}</span> highlighted
|
||||
</span>
|
||||
)}
|
||||
<span className="text-text-muted">
|
||||
{queryResult.executionTime.toFixed(1)}ms
|
||||
</span>
|
||||
</div>
|
||||
<div className="flex items-center gap-2">
|
||||
{queryResult.nodeIds.length > 0 && (
|
||||
<button
|
||||
onClick={clearQueryHighlights}
|
||||
className="text-xs text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
Clear
|
||||
</button>
|
||||
)}
|
||||
<button
|
||||
onClick={() => setShowResults(!showResults)}
|
||||
className="flex items-center gap-1 text-xs text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
<Table className="w-3 h-3" />
|
||||
{showResults ? <ChevronDown className="w-3 h-3" /> : <ChevronUp className="w-3 h-3" />}
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{showResults && queryResult.rows.length > 0 && (
|
||||
<div className="max-h-48 overflow-auto scrollbar-thin border-t border-border-subtle">
|
||||
<table className="w-full text-xs">
|
||||
<thead className="bg-surface sticky top-0">
|
||||
<tr>
|
||||
{Object.keys(queryResult.rows[0]).map((key) => (
|
||||
<th key={key} className="px-3 py-2 text-left text-text-muted font-medium border-b border-border-subtle">
|
||||
{key}
|
||||
</th>
|
||||
))}
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{queryResult.rows.slice(0, 50).map((row, i) => (
|
||||
<tr key={i} className="hover:bg-hover/50 transition-colors">
|
||||
{Object.values(row).map((val, j) => (
|
||||
<td key={j} className="px-3 py-1.5 text-text-secondary border-b border-border-subtle/50 font-mono truncate max-w-[200px]">
|
||||
{typeof val === 'object' ? JSON.stringify(val) : String(val ?? '')}
|
||||
</td>
|
||||
))}
|
||||
</tr>
|
||||
))}
|
||||
</tbody>
|
||||
</table>
|
||||
{queryResult.rows.length > 50 && (
|
||||
<div className="px-3 py-2 text-xs text-text-muted bg-surface border-t border-border-subtle">
|
||||
Showing 50 of {queryResult.rows.length} rows
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,795 @@
|
||||
import { useState, useRef, useEffect, useCallback } from 'react';
|
||||
import {
|
||||
Send, Sparkles, User,
|
||||
PanelRightClose, Loader2, AlertTriangle
|
||||
} from 'lucide-react';
|
||||
import { Prism as SyntaxHighlighter } from 'react-syntax-highlighter';
|
||||
import { vscDarkPlus } from 'react-syntax-highlighter/dist/esm/styles/prism';
|
||||
import ReactMarkdown from 'react-markdown';
|
||||
import remarkGfm from 'remark-gfm';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { ToolCallCard } from './ToolCallCard';
|
||||
import { isProviderConfigured } from '../core/llm/settings-service';
|
||||
|
||||
// Custom syntax theme
|
||||
const customTheme = {
|
||||
...vscDarkPlus,
|
||||
'pre[class*="language-"]': {
|
||||
...vscDarkPlus['pre[class*="language-"]'],
|
||||
background: '#0a0a10',
|
||||
margin: 0,
|
||||
padding: '16px 0',
|
||||
fontSize: '13px',
|
||||
lineHeight: '1.6',
|
||||
},
|
||||
'code[class*="language-"]': {
|
||||
...vscDarkPlus['code[class*="language-"]'],
|
||||
background: 'transparent',
|
||||
fontFamily: '"JetBrains Mono", "Fira Code", monospace',
|
||||
},
|
||||
};
|
||||
|
||||
export const RightPanel = () => {
|
||||
const {
|
||||
isRightPanelOpen,
|
||||
setRightPanelOpen,
|
||||
fileContents,
|
||||
graph,
|
||||
addCodeReference,
|
||||
// LLM / chat state
|
||||
chatMessages,
|
||||
isChatLoading,
|
||||
currentToolCalls,
|
||||
agentError,
|
||||
isAgentReady,
|
||||
isAgentInitializing,
|
||||
sendChatMessage,
|
||||
clearChat,
|
||||
} = useAppState();
|
||||
|
||||
const [chatInput, setChatInput] = useState('');
|
||||
const textareaRef = useRef<HTMLTextAreaElement>(null);
|
||||
const messagesEndRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
// Auto-scroll to bottom when messages update or while streaming
|
||||
useEffect(() => {
|
||||
if (messagesEndRef.current) {
|
||||
messagesEndRef.current.scrollIntoView({ behavior: 'smooth' });
|
||||
}
|
||||
}, [chatMessages, isChatLoading]);
|
||||
|
||||
const resolveFilePathForUI = useCallback((requestedPath: string): string | null => {
|
||||
const req = requestedPath.replace(/\\/g, '/').replace(/^\.?\//, '').toLowerCase();
|
||||
if (!req) return null;
|
||||
|
||||
// Exact match first (case-insensitive)
|
||||
for (const key of fileContents.keys()) {
|
||||
const norm = key.replace(/\\/g, '/').replace(/^\.?\//, '').toLowerCase();
|
||||
if (norm === req) return key;
|
||||
}
|
||||
|
||||
// Ends-with match (best for partial paths)
|
||||
let best: { path: string; score: number } | null = null;
|
||||
for (const key of fileContents.keys()) {
|
||||
const norm = key.replace(/\\/g, '/').replace(/^\.?\//, '').toLowerCase();
|
||||
if (norm.endsWith(req)) {
|
||||
const score = 1000 - norm.length;
|
||||
if (!best || score > best.score) best = { path: key, score };
|
||||
}
|
||||
}
|
||||
return best?.path ?? null;
|
||||
}, [fileContents]);
|
||||
|
||||
const findFileNodeIdForUI = useCallback((filePath: string): string | undefined => {
|
||||
if (!graph) return undefined;
|
||||
const target = filePath.replace(/\\/g, '/').replace(/^\.?\//, '');
|
||||
const node = graph.nodes.find(
|
||||
(n) => n.label === 'File' && n.properties.filePath.replace(/\\/g, '/').replace(/^\.?\//, '') === target
|
||||
);
|
||||
return node?.id;
|
||||
}, [graph]);
|
||||
|
||||
const handleGroundingClick = useCallback((inner: string) => {
|
||||
const raw = inner.trim();
|
||||
if (!raw) return;
|
||||
|
||||
let rawPath = raw;
|
||||
let startLine1: number | undefined;
|
||||
let endLine1: number | undefined;
|
||||
|
||||
// Match line:num or line:num-num (supports both hyphen - and en dash –)
|
||||
const lineMatch = raw.match(/^(.*):(\d+)(?:[-–](\d+))?$/);
|
||||
if (lineMatch) {
|
||||
rawPath = lineMatch[1].trim();
|
||||
startLine1 = parseInt(lineMatch[2], 10);
|
||||
endLine1 = parseInt(lineMatch[3] || lineMatch[2], 10);
|
||||
}
|
||||
|
||||
const resolvedPath = resolveFilePathForUI(rawPath);
|
||||
if (!resolvedPath) return;
|
||||
|
||||
const nodeId = findFileNodeIdForUI(resolvedPath);
|
||||
|
||||
addCodeReference({
|
||||
filePath: resolvedPath,
|
||||
startLine: startLine1 ? Math.max(0, startLine1 - 1) : undefined,
|
||||
endLine: endLine1 ? Math.max(0, endLine1 - 1) : (startLine1 ? Math.max(0, startLine1 - 1) : undefined),
|
||||
nodeId,
|
||||
label: 'File',
|
||||
name: resolvedPath.split('/').pop() ?? resolvedPath,
|
||||
source: 'ai',
|
||||
});
|
||||
}, [addCodeReference, findFileNodeIdForUI, resolveFilePathForUI]);
|
||||
|
||||
// Handler for node grounding: [[Class:View]], [[Function:trigger]], etc.
|
||||
const handleNodeGroundingClick = useCallback((nodeTypeAndName: string) => {
|
||||
const raw = nodeTypeAndName.trim();
|
||||
if (!raw || !graph) return;
|
||||
|
||||
// Parse Type:Name format
|
||||
const match = raw.match(/^(Class|Function|Method|Interface|File|Folder|Variable|Enum|Type|CodeElement):(.+)$/);
|
||||
if (!match) return;
|
||||
|
||||
const [, nodeType, nodeName] = match;
|
||||
const trimmedName = nodeName.trim();
|
||||
|
||||
// Find node in graph by type + name
|
||||
const node = graph.nodes.find(n =>
|
||||
n.label === nodeType &&
|
||||
n.properties.name === trimmedName
|
||||
);
|
||||
|
||||
if (!node) {
|
||||
console.warn(`Node not found: ${nodeType}:${trimmedName}`);
|
||||
return;
|
||||
}
|
||||
|
||||
// 1. Highlight in graph (add to AI citation highlights)
|
||||
// Note: This requires accessing the state setter from parent context
|
||||
// For now, we'll add to code references which triggers the highlight
|
||||
|
||||
// 2. Add to Code Panel (if node has file/line info)
|
||||
if (node.properties.filePath) {
|
||||
const resolvedPath = resolveFilePathForUI(node.properties.filePath);
|
||||
if (resolvedPath) {
|
||||
addCodeReference({
|
||||
filePath: resolvedPath,
|
||||
startLine: node.properties.startLine ? node.properties.startLine - 1 : undefined,
|
||||
endLine: node.properties.endLine ? node.properties.endLine - 1 : undefined,
|
||||
nodeId: node.id,
|
||||
label: node.label,
|
||||
name: node.properties.name,
|
||||
source: 'ai',
|
||||
});
|
||||
}
|
||||
}
|
||||
}, [graph, resolveFilePathForUI, addCodeReference]);
|
||||
|
||||
const formatMarkdownForDisplay = useCallback((md: string) => {
|
||||
// Avoid rewriting inside fenced code blocks.
|
||||
const parts = md.split('```');
|
||||
for (let i = 0; i < parts.length; i += 2) {
|
||||
// Pattern 1: File grounding - [[file.ext]] or [[file.ext:line]] or [[file.ext:line-line]]
|
||||
// Line numbers are optional
|
||||
parts[i] = parts[i].replace(
|
||||
/\[\[([a-zA-Z0-9_\-./\\]+\.[a-zA-Z0-9]+(?::\d+(?:[-–]\d+)?)?)\]\]/g,
|
||||
(_m, inner: string) => {
|
||||
const trimmed = inner.trim();
|
||||
const href = `code-ref:${encodeURIComponent(trimmed)}`;
|
||||
return `[${trimmed}](${href})`;
|
||||
}
|
||||
);
|
||||
|
||||
// Pattern 2: Node grounding - [[Type:Name]] or [[graph:Type:Name]]
|
||||
// Valid types: Class, Function, Method, Interface, File, Folder, Variable, Enum, Type, CodeElement
|
||||
parts[i] = parts[i].replace(
|
||||
/\[\[(?:graph:)?(Class|Function|Method|Interface|File|Folder|Variable|Enum|Type|CodeElement):([^\]]+)\]\]/g,
|
||||
(_m, nodeType: string, nodeName: string) => {
|
||||
const trimmed = `${nodeType}:${nodeName.trim()}`;
|
||||
const href = `node-ref:${encodeURIComponent(trimmed)}`;
|
||||
return `[${trimmed}](${href})`;
|
||||
}
|
||||
);
|
||||
}
|
||||
return parts.join('```');
|
||||
}, []);
|
||||
|
||||
const formatRefChipLabel = useCallback((ref: string): string => {
|
||||
const raw = ref.trim();
|
||||
if (!raw) return '';
|
||||
|
||||
// Drop any scheme prefix we might have accidentally passed through
|
||||
const withoutScheme = raw.startsWith('code-ref:') ? raw.slice('code-ref:'.length) : raw;
|
||||
|
||||
// Strip query/hash
|
||||
const cleaned = withoutScheme.split('#')[0].split('?')[0];
|
||||
|
||||
// Support both hyphen - and en dash – for line ranges
|
||||
const m = cleaned.match(/^(.*):(\d+)(?:[-–](\d+))?$/);
|
||||
const path = (m ? m[1] : cleaned).replace(/\\/g, '/');
|
||||
const base = path.split('/').pop() ?? path;
|
||||
|
||||
if (!m) return base;
|
||||
|
||||
const start = m[2];
|
||||
const end = m[3] ?? m[2];
|
||||
return `${base} ${start}–${end}`;
|
||||
}, []);
|
||||
|
||||
const isLikelyFileRefHref = useCallback((href: string): boolean => {
|
||||
const h = href.trim();
|
||||
if (!h) return false;
|
||||
if (h.startsWith('code-ref:')) return true;
|
||||
if (/^(https?:|mailto:|tel:|#)/i.test(h)) return false;
|
||||
if (h.includes('://')) return false;
|
||||
|
||||
// Strip query/hash
|
||||
const cleaned = h.split('#')[0].split('?')[0];
|
||||
|
||||
// Looks like: path/to/file.ext or path\to\file.ext:12-34 (supports both hyphen and en dash)
|
||||
return /[A-Za-z0-9_\-./\\]+\.[A-Za-z0-9]+(?::\d+(?:[-–]\d+)?)?$/.test(cleaned);
|
||||
}, []);
|
||||
|
||||
const extractTextFromChildren = useCallback((children: any): string => {
|
||||
if (children == null) return '';
|
||||
if (typeof children === 'string' || typeof children === 'number') return String(children);
|
||||
if (Array.isArray(children)) return children.map(extractTextFromChildren).join('');
|
||||
// React element or other objects
|
||||
return '';
|
||||
}, []);
|
||||
|
||||
const getInternalRefFromLink = useCallback((href: string | undefined, children: any): string | null => {
|
||||
const hrefStr = (href ?? '').trim();
|
||||
const textStr = extractTextFromChildren(children).trim();
|
||||
|
||||
if (hrefStr && isLikelyFileRefHref(hrefStr)) return hrefStr;
|
||||
if (textStr && isLikelyFileRefHref(textStr)) return textStr;
|
||||
|
||||
return null;
|
||||
}, [extractTextFromChildren, isLikelyFileRefHref]);
|
||||
|
||||
// Auto-resize textarea as user types
|
||||
const adjustTextareaHeight = useCallback(() => {
|
||||
const textarea = textareaRef.current;
|
||||
if (!textarea) return;
|
||||
|
||||
// Reset height to get accurate scrollHeight
|
||||
textarea.style.height = 'auto';
|
||||
// Set to scrollHeight, capped at max
|
||||
const maxHeight = 160; // ~6 lines
|
||||
const newHeight = Math.min(textarea.scrollHeight, maxHeight);
|
||||
textarea.style.height = `${newHeight}px`;
|
||||
// Show scrollbar if content exceeds max
|
||||
textarea.style.overflowY = textarea.scrollHeight > maxHeight ? 'auto' : 'hidden';
|
||||
}, []);
|
||||
|
||||
// Adjust height when input changes
|
||||
useEffect(() => {
|
||||
adjustTextareaHeight();
|
||||
}, [chatInput, adjustTextareaHeight]);
|
||||
|
||||
// Chat handlers
|
||||
const handleSendMessage = async () => {
|
||||
if (!chatInput.trim()) return;
|
||||
const text = chatInput.trim();
|
||||
setChatInput('');
|
||||
// Reset textarea height after sending
|
||||
if (textareaRef.current) {
|
||||
textareaRef.current.style.height = '36px';
|
||||
textareaRef.current.style.overflowY = 'hidden';
|
||||
}
|
||||
await sendChatMessage(text);
|
||||
};
|
||||
|
||||
const handleKeyDown = (e: React.KeyboardEvent) => {
|
||||
if (e.key === 'Enter' && !e.shiftKey) {
|
||||
e.preventDefault();
|
||||
handleSendMessage();
|
||||
}
|
||||
};
|
||||
|
||||
const chatSuggestions = [
|
||||
'Explain the project architecture',
|
||||
'What does this project do?',
|
||||
'Show me the most important files',
|
||||
'Find all API handlers',
|
||||
];
|
||||
|
||||
if (!isRightPanelOpen) return null;
|
||||
|
||||
return (
|
||||
<aside className="w-[40%] min-w-[400px] max-w-[600px] flex flex-col bg-deep border-l border-border-subtle animate-slide-in relative z-30 flex-shrink-0">
|
||||
{/* Header */}
|
||||
<div className="flex items-center justify-between px-4 py-2 bg-surface border-b border-border-subtle">
|
||||
<div className="flex items-center gap-2.5">
|
||||
<Sparkles className="w-4 h-4 text-accent" />
|
||||
<span className="font-medium text-sm">Nexus AI</span>
|
||||
<span className="text-xs text-text-muted">• Ask about the codebase</span>
|
||||
</div>
|
||||
|
||||
{/* Close button */}
|
||||
<button
|
||||
onClick={() => setRightPanelOpen(false)}
|
||||
className="p-1.5 text-text-muted hover:text-text-primary hover:bg-hover rounded transition-colors"
|
||||
title="Close Panel"
|
||||
>
|
||||
<PanelRightClose className="w-4 h-4" />
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Chat Content */}
|
||||
<div className="flex-1 flex flex-col overflow-hidden">
|
||||
{/* Status bar */}
|
||||
<div className="flex items-center gap-2.5 px-4 py-3 bg-elevated/50 border-b border-border-subtle">
|
||||
<div className="ml-auto flex items-center gap-2">
|
||||
{!isAgentReady && (
|
||||
<span className="text-[11px] px-2 py-1 rounded-full bg-amber-500/15 text-amber-300 border border-amber-500/30">
|
||||
Configure AI
|
||||
</span>
|
||||
)}
|
||||
{isAgentInitializing && (
|
||||
<span className="text-[11px] px-2 py-1 rounded-full bg-surface border border-border-subtle flex items-center gap-1 text-text-muted">
|
||||
<Loader2 className="w-3 h-3 animate-spin" /> Connecting
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Status / errors */}
|
||||
{agentError && (
|
||||
<div className="px-4 py-3 bg-rose-500/10 border-b border-rose-500/30 text-rose-100 text-sm flex items-center gap-2">
|
||||
<AlertTriangle className="w-4 h-4" />
|
||||
<span>{agentError}</span>
|
||||
</div>
|
||||
)}
|
||||
|
||||
|
||||
|
||||
{/* Messages */}
|
||||
<div className="flex-1 overflow-y-auto p-4 scrollbar-thin">
|
||||
{chatMessages.length === 0 ? (
|
||||
<div className="flex flex-col items-center justify-center h-full text-center px-4">
|
||||
<div className="w-14 h-14 mb-4 flex items-center justify-center bg-gradient-to-br from-accent to-node-interface rounded-xl shadow-glow text-2xl">
|
||||
🧠
|
||||
</div>
|
||||
<h3 className="text-base font-medium mb-2">
|
||||
Ask me anything
|
||||
</h3>
|
||||
<p className="text-sm text-text-secondary leading-relaxed mb-5">
|
||||
I can help you understand the architecture, find functions, or explain connections.
|
||||
</p>
|
||||
<div className="flex flex-wrap gap-2 justify-center">
|
||||
{chatSuggestions.map((suggestion) => (
|
||||
<button
|
||||
key={suggestion}
|
||||
onClick={() => setChatInput(suggestion)}
|
||||
className="px-3 py-1.5 bg-elevated border border-border-subtle rounded-full text-xs text-text-secondary hover:border-accent hover:text-text-primary transition-colors"
|
||||
>
|
||||
{suggestion}
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
</div>
|
||||
) : (
|
||||
<div className="flex flex-col gap-6">
|
||||
{chatMessages.map((message) => (
|
||||
<div
|
||||
key={message.id}
|
||||
className="animate-fade-in"
|
||||
>
|
||||
{/* User message - compact label style */}
|
||||
{message.role === 'user' && (
|
||||
<div className="mb-4">
|
||||
<div className="flex items-center gap-2 mb-2">
|
||||
<User className="w-4 h-4 text-text-muted" />
|
||||
<span className="text-xs font-medium text-text-muted uppercase tracking-wide">You</span>
|
||||
</div>
|
||||
<div className="pl-6 text-sm text-text-primary">
|
||||
{message.content}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Assistant message - copilot style */}
|
||||
{message.role === 'assistant' && (
|
||||
<div>
|
||||
<div className="flex items-center gap-2 mb-3">
|
||||
<Sparkles className="w-4 h-4 text-accent" />
|
||||
<span className="text-xs font-medium text-text-muted uppercase tracking-wide">Nexus AI</span>
|
||||
{isChatLoading && message === chatMessages[chatMessages.length - 1] && (
|
||||
<Loader2 className="w-3 h-3 animate-spin text-accent" />
|
||||
)}
|
||||
</div>
|
||||
<div className="pl-6 chat-prose">
|
||||
{/* Render steps in order (reasoning, tool calls, content interleaved) */}
|
||||
{message.steps && message.steps.length > 0 ? (
|
||||
<div className="space-y-4">
|
||||
{message.steps.map((step) => (
|
||||
<div key={step.id}>
|
||||
{step.type === 'reasoning' && step.content && (
|
||||
<div className="text-text-secondary text-sm italic border-l-2 border-text-muted/30 pl-3 mb-3">
|
||||
<ReactMarkdown
|
||||
remarkPlugins={[remarkGfm]}
|
||||
components={{
|
||||
a: ({ href, children, ...props }) => {
|
||||
if (href && href.startsWith('code-ref:')) {
|
||||
const inner = decodeURIComponent(href.slice('code-ref:'.length));
|
||||
const label = formatRefChipLabel(inner);
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
handleGroundingClick(inner);
|
||||
}}
|
||||
className="code-ref-btn inline-flex items-center px-2 py-0.5 rounded-md border border-cyan-300/55 bg-cyan-400/10 !text-cyan-200 visited:!text-cyan-200 font-mono text-[12px] !no-underline hover:!no-underline hover:bg-cyan-400/15 hover:border-cyan-200/70 transition-colors"
|
||||
title={`Open in Code panel • ${inner}`}
|
||||
{...props}
|
||||
>
|
||||
<span className="text-inherit">{label || children}</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
// Handle node grounding: node-ref:Type:Name
|
||||
if (href && href.startsWith('node-ref:')) {
|
||||
const inner = decodeURIComponent(href.slice('node-ref:'.length));
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
handleNodeGroundingClick(inner);
|
||||
}}
|
||||
className="code-ref-btn inline-flex items-center px-2 py-0.5 rounded-md border border-amber-300/55 bg-amber-400/10 !text-amber-200 visited:!text-amber-200 font-mono text-[12px] !no-underline hover:!no-underline hover:bg-amber-400/15 hover:border-amber-200/70 transition-colors"
|
||||
title={`View ${inner} in Code panel`}
|
||||
{...props}
|
||||
>
|
||||
<span className="text-inherit">{children}</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
const internalRef = getInternalRefFromLink(href, children);
|
||||
if (internalRef) {
|
||||
const label = formatRefChipLabel(internalRef);
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
handleGroundingClick(internalRef);
|
||||
}}
|
||||
className="code-ref-btn inline-flex items-center px-2 py-0.5 rounded-md border border-cyan-300/55 bg-cyan-400/10 !text-cyan-200 visited:!text-cyan-200 font-mono text-[12px] !no-underline hover:!no-underline hover:bg-cyan-400/15 hover:border-cyan-200/70 transition-colors"
|
||||
title={`Open in Code panel • ${internalRef}`}
|
||||
{...props}
|
||||
>
|
||||
<span className="text-inherit">{label || children}</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
// Non-citation link - just render as plain text (not a useless external link)
|
||||
return <span {...props}>{children}</span>;
|
||||
},
|
||||
code: ({ className, children, ...props }) => {
|
||||
const match = /language-(\w+)/.exec(className || '');
|
||||
const isInline = !className && !match;
|
||||
const codeContent = String(children).replace(/\n$/, '');
|
||||
|
||||
if (isInline) {
|
||||
return <code {...props}>{children}</code>;
|
||||
}
|
||||
|
||||
const language = match ? match[1] : 'text';
|
||||
return (
|
||||
<SyntaxHighlighter
|
||||
style={customTheme}
|
||||
language={language}
|
||||
PreTag="div"
|
||||
customStyle={{
|
||||
margin: 0,
|
||||
padding: '14px 16px',
|
||||
borderRadius: '8px',
|
||||
fontSize: '13px',
|
||||
background: '#0a0a10',
|
||||
border: '1px solid #1e1e2a',
|
||||
}}
|
||||
>
|
||||
{codeContent}
|
||||
</SyntaxHighlighter>
|
||||
);
|
||||
},
|
||||
pre: ({ children }) => <>{children}</>,
|
||||
}}
|
||||
>
|
||||
{formatMarkdownForDisplay(step.content)}
|
||||
</ReactMarkdown>
|
||||
</div>
|
||||
)}
|
||||
{step.type === 'tool_call' && step.toolCall && (
|
||||
<div className="mb-3">
|
||||
<ToolCallCard toolCall={step.toolCall} defaultExpanded={false} />
|
||||
</div>
|
||||
)}
|
||||
{step.type === 'content' && step.content && (
|
||||
<div className="text-text-primary text-sm">
|
||||
<ReactMarkdown
|
||||
remarkPlugins={[remarkGfm]}
|
||||
components={{
|
||||
a: ({ href, children, ...props }) => {
|
||||
if (href && href.startsWith('code-ref:')) {
|
||||
const inner = decodeURIComponent(href.slice('code-ref:'.length));
|
||||
const label = formatRefChipLabel(inner);
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
handleGroundingClick(inner);
|
||||
}}
|
||||
className="code-ref-btn inline-flex items-center px-2 py-0.5 rounded-md border border-cyan-300/55 bg-cyan-400/10 !text-cyan-200 visited:!text-cyan-200 font-mono text-[12px] !no-underline hover:!no-underline hover:bg-cyan-400/15 hover:border-cyan-200/70 transition-colors"
|
||||
title={`Open in Code panel • ${inner}`}
|
||||
{...props}
|
||||
>
|
||||
<span className="text-inherit">{label || children}</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
// Handle node grounding: node-ref:Type:Name
|
||||
if (href && href.startsWith('node-ref:')) {
|
||||
const inner = decodeURIComponent(href.slice('node-ref:'.length));
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
handleNodeGroundingClick(inner);
|
||||
}}
|
||||
className="code-ref-btn inline-flex items-center px-2 py-0.5 rounded-md border border-amber-300/55 bg-amber-400/10 !text-amber-200 visited:!text-amber-200 font-mono text-[12px] !no-underline hover:!no-underline hover:bg-amber-400/15 hover:border-amber-200/70 transition-colors"
|
||||
title={`View ${inner} in Code panel`}
|
||||
{...props}
|
||||
>
|
||||
<span className="text-inherit">{children}</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
const internalRef = getInternalRefFromLink(href, children);
|
||||
if (internalRef) {
|
||||
const label = formatRefChipLabel(internalRef);
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
handleGroundingClick(internalRef);
|
||||
}}
|
||||
className="code-ref-btn inline-flex items-center px-2 py-0.5 rounded-md border border-cyan-300/55 bg-cyan-400/10 !text-cyan-200 visited:!text-cyan-200 font-mono text-[12px] !no-underline hover:!no-underline hover:bg-cyan-400/15 hover:border-cyan-200/70 transition-colors"
|
||||
title={`Open in Code panel • ${internalRef}`}
|
||||
{...props}
|
||||
>
|
||||
<span className="text-inherit">{label || children}</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
className="text-accent underline underline-offset-2 hover:text-purple-300"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
{...props}
|
||||
>
|
||||
{children}
|
||||
</a>
|
||||
);
|
||||
},
|
||||
code: ({ className, children, ...props }) => {
|
||||
const match = /language-(\w+)/.exec(className || '');
|
||||
const isInline = !className && !match;
|
||||
const codeContent = String(children).replace(/\n$/, '');
|
||||
|
||||
if (isInline) {
|
||||
return <code {...props}>{children}</code>;
|
||||
}
|
||||
|
||||
const language = match ? match[1] : 'text';
|
||||
return (
|
||||
<SyntaxHighlighter
|
||||
style={customTheme}
|
||||
language={language}
|
||||
PreTag="div"
|
||||
customStyle={{
|
||||
margin: 0,
|
||||
padding: '14px 16px',
|
||||
borderRadius: '8px',
|
||||
fontSize: '13px',
|
||||
background: '#0a0a10',
|
||||
border: '1px solid #1e1e2a',
|
||||
}}
|
||||
>
|
||||
{codeContent}
|
||||
</SyntaxHighlighter>
|
||||
);
|
||||
},
|
||||
pre: ({ children }) => <>{children}</>,
|
||||
}}
|
||||
>
|
||||
{formatMarkdownForDisplay(step.content)}
|
||||
</ReactMarkdown>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
) : (
|
||||
// Fallback: render content + toolCalls separately (old format)
|
||||
<div className="text-text-primary text-sm">
|
||||
<ReactMarkdown
|
||||
remarkPlugins={[remarkGfm]}
|
||||
components={{
|
||||
a: ({ href, children, ...props }) => {
|
||||
if (href && href.startsWith('code-ref:')) {
|
||||
const inner = decodeURIComponent(href.slice('code-ref:'.length));
|
||||
const label = formatRefChipLabel(inner);
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
handleGroundingClick(inner);
|
||||
}}
|
||||
className="code-ref-btn inline-flex items-center px-2 py-0.5 rounded-md border border-cyan-300/55 bg-cyan-400/10 !text-cyan-200 visited:!text-cyan-200 font-mono text-[12px] !no-underline hover:!no-underline hover:bg-cyan-400/15 hover:border-cyan-200/70 transition-colors"
|
||||
title={`Open in Code panel • ${inner}`}
|
||||
{...props}
|
||||
>
|
||||
<span className="text-inherit">{label || children}</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
// Handle node grounding: node-ref:Type:Name
|
||||
if (href && href.startsWith('node-ref:')) {
|
||||
const inner = decodeURIComponent(href.slice('node-ref:'.length));
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
handleNodeGroundingClick(inner);
|
||||
}}
|
||||
className="code-ref-btn inline-flex items-center px-2 py-0.5 rounded-md border border-amber-300/55 bg-amber-400/10 !text-amber-200 visited:!text-amber-200 font-mono text-[12px] !no-underline hover:!no-underline hover:bg-amber-400/15 hover:border-amber-200/70 transition-colors"
|
||||
title={`View ${inner} in Code panel`}
|
||||
{...props}
|
||||
>
|
||||
<span className="text-inherit">{children}</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
const internalRef = getInternalRefFromLink(href, children);
|
||||
if (internalRef) {
|
||||
const label = formatRefChipLabel(internalRef);
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
handleGroundingClick(internalRef);
|
||||
}}
|
||||
className="code-ref-btn inline-flex items-center px-2 py-0.5 rounded-md border border-cyan-300/55 bg-cyan-400/10 !text-cyan-200 visited:!text-cyan-200 font-mono text-[12px] !no-underline hover:!no-underline hover:bg-cyan-400/15 hover:border-cyan-200/70 transition-colors"
|
||||
title={`Open in Code panel • ${internalRef}`}
|
||||
{...props}
|
||||
>
|
||||
<span className="text-inherit">{label || children}</span>
|
||||
</a>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<a
|
||||
href={href}
|
||||
className="text-accent underline underline-offset-2 hover:text-purple-300"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
{...props}
|
||||
>
|
||||
{children}
|
||||
</a>
|
||||
);
|
||||
},
|
||||
code: ({ className, children, ...props }) => {
|
||||
const match = /language-(\w+)/.exec(className || '');
|
||||
const isInline = !className && !match;
|
||||
const codeContent = String(children).replace(/\n$/, '');
|
||||
|
||||
if (isInline) {
|
||||
return <code {...props}>{children}</code>;
|
||||
}
|
||||
|
||||
const language = match ? match[1] : 'text';
|
||||
return (
|
||||
<SyntaxHighlighter
|
||||
style={customTheme}
|
||||
language={language}
|
||||
PreTag="div"
|
||||
customStyle={{
|
||||
margin: 0,
|
||||
padding: '14px 16px',
|
||||
borderRadius: '8px',
|
||||
fontSize: '13px',
|
||||
background: '#0a0a10',
|
||||
border: '1px solid #1e1e2a',
|
||||
}}
|
||||
>
|
||||
{codeContent}
|
||||
</SyntaxHighlighter>
|
||||
);
|
||||
},
|
||||
pre: ({ children }) => <>{children}</>,
|
||||
}}
|
||||
>
|
||||
{formatMarkdownForDisplay(message.content)}
|
||||
</ReactMarkdown>
|
||||
{message.toolCalls && message.toolCalls.length > 0 && (
|
||||
<div className="mt-3 space-y-2">
|
||||
{message.toolCalls.map(tc => (
|
||||
<ToolCallCard key={tc.id} toolCall={tc} defaultExpanded={false} />
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
))}
|
||||
|
||||
|
||||
</div>
|
||||
)}
|
||||
{/* Scroll anchor for auto-scroll */}
|
||||
<div ref={messagesEndRef} />
|
||||
</div>
|
||||
|
||||
{/* Input */}
|
||||
<div className="p-3 bg-surface border-t border-border-subtle">
|
||||
<div className="flex items-end gap-2 px-3 py-2 bg-elevated border border-border-subtle rounded-xl transition-all focus-within:border-accent focus-within:ring-2 focus-within:ring-accent/20">
|
||||
<textarea
|
||||
ref={textareaRef}
|
||||
value={chatInput}
|
||||
onChange={(e) => setChatInput(e.target.value)}
|
||||
onKeyDown={handleKeyDown}
|
||||
placeholder="Ask about the codebase..."
|
||||
rows={1}
|
||||
className="flex-1 bg-transparent border-none outline-none text-sm text-text-primary placeholder:text-text-muted resize-none min-h-[36px] scrollbar-thin"
|
||||
style={{ height: '36px', overflowY: 'hidden' }}
|
||||
/>
|
||||
<button
|
||||
onClick={clearChat}
|
||||
className="px-2 py-1 text-xs text-text-muted hover:text-text-primary transition-colors"
|
||||
title="Clear chat"
|
||||
>
|
||||
Clear
|
||||
</button>
|
||||
<button
|
||||
onClick={handleSendMessage}
|
||||
disabled={!chatInput.trim() || isChatLoading || isAgentInitializing}
|
||||
className="w-9 h-9 flex items-center justify-center bg-accent rounded-md text-white transition-all hover:bg-accent-dim disabled:opacity-50 disabled:cursor-not-allowed"
|
||||
>
|
||||
{isChatLoading ? <Loader2 className="w-4 h-4 animate-spin" /> : <Send className="w-3.5 h-3.5" />}
|
||||
</button>
|
||||
</div>
|
||||
{!isAgentReady && !isAgentInitializing && (
|
||||
<div className="mt-2 text-xs text-amber-200 flex items-center gap-2">
|
||||
<AlertTriangle className="w-3.5 h-3.5" />
|
||||
<span>
|
||||
{isProviderConfigured()
|
||||
? 'Initializing AI agent...'
|
||||
: 'Configure an LLM provider to enable chat.'}
|
||||
</span>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
</aside>
|
||||
);
|
||||
};
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,672 @@
|
||||
import { useState, useEffect, useCallback } from 'react';
|
||||
import { X, Key, Server, Brain, Check, AlertCircle, Eye, EyeOff, RefreshCw, Loader2 } from 'lucide-react';
|
||||
import {
|
||||
loadSettings,
|
||||
saveSettings,
|
||||
getProviderDisplayName,
|
||||
getAvailableModels,
|
||||
} from '../core/llm/settings-service';
|
||||
import type { LLMSettings, LLMProvider } from '../core/llm/types';
|
||||
|
||||
interface SettingsPanelProps {
|
||||
isOpen: boolean;
|
||||
onClose: () => void;
|
||||
onSettingsSaved?: () => void;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch available Gemini models from the API
|
||||
*/
|
||||
const fetchGeminiModels = async (apiKey: string): Promise<string[]> => {
|
||||
try {
|
||||
const response = await fetch(
|
||||
`https://generativelanguage.googleapis.com/v1beta/models?key=${apiKey}`
|
||||
);
|
||||
|
||||
if (!response.ok) {
|
||||
throw new Error(`API error: ${response.status}`);
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
|
||||
// Filter for chat-capable models and extract model names
|
||||
const models = (data.models || [])
|
||||
.filter((model: any) => {
|
||||
const methods = model.supportedGenerationMethods || [];
|
||||
return methods.includes('generateContent');
|
||||
})
|
||||
.map((model: any) => {
|
||||
const name = model.name || '';
|
||||
return name.replace('models/', '');
|
||||
})
|
||||
.filter((name: string) => name.length > 0)
|
||||
.sort((a: string, b: string) => {
|
||||
const score = (n: string) => {
|
||||
if (n.includes('gemini-2')) return 0;
|
||||
if (n.includes('gemini-1.5')) return 1;
|
||||
if (n.includes('gemini-1')) return 2;
|
||||
return 3;
|
||||
};
|
||||
return score(a) - score(b) || a.localeCompare(b);
|
||||
});
|
||||
|
||||
return models;
|
||||
} catch (error) {
|
||||
console.warn('Failed to fetch Gemini models:', error);
|
||||
return [];
|
||||
}
|
||||
};
|
||||
|
||||
export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved }: SettingsPanelProps) => {
|
||||
const [settings, setSettings] = useState<LLMSettings>(loadSettings);
|
||||
const [showApiKey, setShowApiKey] = useState<Record<string, boolean>>({});
|
||||
const [saveStatus, setSaveStatus] = useState<'idle' | 'saved' | 'error'>('idle');
|
||||
|
||||
// Gemini model fetching state
|
||||
const [geminiModels, setGeminiModels] = useState<string[]>([]);
|
||||
const [isLoadingModels, setIsLoadingModels] = useState(false);
|
||||
const [modelFetchError, setModelFetchError] = useState<string | null>(null);
|
||||
const [useCustomModel, setUseCustomModel] = useState(false);
|
||||
const [useCustomAnthropicModel, setUseCustomAnthropicModel] = useState(false);
|
||||
|
||||
// Load settings when panel opens
|
||||
useEffect(() => {
|
||||
if (isOpen) {
|
||||
setSettings(loadSettings());
|
||||
setSaveStatus('idle');
|
||||
setGeminiModels([]);
|
||||
setModelFetchError(null);
|
||||
setUseCustomModel(false);
|
||||
setUseCustomAnthropicModel(false);
|
||||
}
|
||||
}, [isOpen]);
|
||||
|
||||
// Auto-fetch models when Gemini API key changes
|
||||
const fetchModels = useCallback(async (apiKey: string) => {
|
||||
if (!apiKey || apiKey.length < 10) {
|
||||
setGeminiModels([]);
|
||||
return;
|
||||
}
|
||||
|
||||
setIsLoadingModels(true);
|
||||
setModelFetchError(null);
|
||||
|
||||
const models = await fetchGeminiModels(apiKey);
|
||||
|
||||
setIsLoadingModels(false);
|
||||
|
||||
if (models.length > 0) {
|
||||
setGeminiModels(models);
|
||||
setModelFetchError(null);
|
||||
} else {
|
||||
setGeminiModels([]);
|
||||
setModelFetchError('Could not fetch models. Check your API key or enter model manually.');
|
||||
}
|
||||
}, []);
|
||||
|
||||
// Fetch models when API key is entered (debounced)
|
||||
useEffect(() => {
|
||||
if (settings.activeProvider === 'gemini' && settings.gemini?.apiKey) {
|
||||
const apiKey = settings.gemini.apiKey ?? '';
|
||||
const timer = setTimeout(() => {
|
||||
fetchModels(apiKey);
|
||||
}, 500);
|
||||
return () => clearTimeout(timer);
|
||||
}
|
||||
}, [settings.gemini?.apiKey, settings.activeProvider, fetchModels]);
|
||||
|
||||
const handleProviderChange = (provider: LLMProvider) => {
|
||||
setSettings(prev => ({ ...prev, activeProvider: provider }));
|
||||
};
|
||||
|
||||
const handleSave = () => {
|
||||
try {
|
||||
saveSettings(settings);
|
||||
setSaveStatus('saved');
|
||||
onSettingsSaved?.();
|
||||
setTimeout(() => setSaveStatus('idle'), 2000);
|
||||
} catch {
|
||||
setSaveStatus('error');
|
||||
}
|
||||
};
|
||||
|
||||
const toggleApiKeyVisibility = (key: string) => {
|
||||
setShowApiKey(prev => ({ ...prev, [key]: !prev[key] }));
|
||||
};
|
||||
|
||||
if (!isOpen) return null;
|
||||
|
||||
const providers: LLMProvider[] = ['openai', 'gemini', 'anthropic', 'azure-openai'];
|
||||
|
||||
const availableGeminiModels = geminiModels.length > 0
|
||||
? geminiModels
|
||||
: getAvailableModels('gemini');
|
||||
const currentGeminiModel = settings.gemini?.model ?? 'gemini-2.0-flash';
|
||||
const isCustomModelSelected = !availableGeminiModels.includes(currentGeminiModel);
|
||||
|
||||
const availableAnthropicModels = getAvailableModels('anthropic');
|
||||
const currentAnthropicModel = settings.anthropic?.model ?? 'claude-sonnet-4-20250514';
|
||||
const isCustomAnthropicModelSelected = !availableAnthropicModels.includes(currentAnthropicModel);
|
||||
|
||||
return (
|
||||
<div className="fixed inset-0 z-50 flex items-center justify-center">
|
||||
{/* Backdrop */}
|
||||
<div
|
||||
className="absolute inset-0 bg-black/60 backdrop-blur-sm"
|
||||
onClick={onClose}
|
||||
/>
|
||||
|
||||
{/* Panel */}
|
||||
<div className="relative bg-surface border border-border-subtle rounded-2xl shadow-2xl max-w-lg w-full mx-4 overflow-hidden max-h-[90vh] flex flex-col">
|
||||
{/* Header */}
|
||||
<div className="flex items-center justify-between px-6 py-4 border-b border-border-subtle bg-elevated/50">
|
||||
<div className="flex items-center gap-3">
|
||||
<div className="w-10 h-10 flex items-center justify-center bg-accent/20 rounded-xl">
|
||||
<Brain className="w-5 h-5 text-accent" />
|
||||
</div>
|
||||
<div>
|
||||
<h2 className="text-lg font-semibold text-text-primary">AI Settings</h2>
|
||||
<p className="text-xs text-text-muted">Configure your LLM provider</p>
|
||||
</div>
|
||||
</div>
|
||||
<button
|
||||
onClick={onClose}
|
||||
className="p-2 text-text-muted hover:text-text-primary hover:bg-hover rounded-lg transition-colors"
|
||||
>
|
||||
<X className="w-5 h-5" />
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Content */}
|
||||
<div className="flex-1 overflow-y-auto p-6 space-y-6">
|
||||
{/* Provider Selection */}
|
||||
<div className="space-y-3">
|
||||
<label className="block text-sm font-medium text-text-secondary">
|
||||
Provider
|
||||
</label>
|
||||
<div className="grid grid-cols-1 sm:grid-cols-3 gap-3">
|
||||
{providers.map(provider => (
|
||||
<button
|
||||
key={provider}
|
||||
onClick={() => handleProviderChange(provider)}
|
||||
className={`
|
||||
flex items-center gap-3 p-4 rounded-xl border-2 transition-all
|
||||
${settings.activeProvider === provider
|
||||
? 'border-accent bg-accent/10 text-text-primary'
|
||||
: 'border-border-subtle bg-elevated hover:border-accent/50 text-text-secondary'
|
||||
}
|
||||
`}
|
||||
>
|
||||
<div className={`
|
||||
w-8 h-8 rounded-lg flex items-center justify-center text-lg
|
||||
${settings.activeProvider === provider ? 'bg-accent/20' : 'bg-surface'}
|
||||
`}>
|
||||
{provider === 'openai' ? '🤖' : provider === 'gemini' ? '💎' : provider === 'anthropic' ? '🧠' : '☁️'}
|
||||
</div>
|
||||
<span className="font-medium">{getProviderDisplayName(provider)}</span>
|
||||
</button>
|
||||
))}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* OpenAI Settings */}
|
||||
{settings.activeProvider === 'openai' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['openai'] ? 'text' : 'password'}
|
||||
value={settings.openai?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
openai: { ...prev.openai!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your OpenAI API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('openai')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['openai'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
Get your API key from{' '}
|
||||
<a
|
||||
href="https://platform.openai.com/api-keys"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
OpenAI Platform
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
<select
|
||||
value={settings.openai?.model ?? 'gpt-4o'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
openai: { ...prev.openai!, model: e.target.value }
|
||||
}))}
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all appearance-none cursor-pointer"
|
||||
>
|
||||
{getAvailableModels('openai').map(model => (
|
||||
<option key={model} value={model}>{model}</option>
|
||||
))}
|
||||
</select>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Server className="w-4 h-4" />
|
||||
Base URL <span className="text-text-muted font-normal">(optional)</span>
|
||||
</label>
|
||||
<input
|
||||
type="url"
|
||||
value={settings.openai?.baseUrl ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
openai: { ...prev.openai!, baseUrl: e.target.value }
|
||||
}))}
|
||||
placeholder="https://api.openai.com/v1 (default)"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<p className="text-xs text-text-muted">
|
||||
Leave empty to use the default OpenAI API. Set a custom URL for proxies or compatible APIs.
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Gemini Settings */}
|
||||
{settings.activeProvider === 'gemini' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['gemini'] ? 'text' : 'password'}
|
||||
value={settings.gemini?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
gemini: { ...prev.gemini!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your Google AI API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('gemini')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['gemini'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
Get your API key from{' '}
|
||||
<a
|
||||
href="https://aistudio.google.com/app/apikey"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
Google AI Studio
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<div className="flex items-center justify-between">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
<div className="flex items-center gap-2">
|
||||
{isLoadingModels && (
|
||||
<span className="flex items-center gap-1 text-xs text-text-muted">
|
||||
<Loader2 className="w-3 h-3 animate-spin" />
|
||||
Fetching models...
|
||||
</span>
|
||||
)}
|
||||
{geminiModels.length > 0 && (
|
||||
<span className="text-xs text-green-400">
|
||||
{geminiModels.length} models available
|
||||
</span>
|
||||
)}
|
||||
{settings.gemini?.apiKey && !isLoadingModels && (
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => fetchModels(settings.gemini?.apiKey ?? '')}
|
||||
className="p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
title="Refresh models"
|
||||
>
|
||||
<RefreshCw className="w-3.5 h-3.5" />
|
||||
</button>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Model selector or manual input */}
|
||||
{(useCustomModel || isCustomModelSelected) ? (
|
||||
<div className="space-y-2">
|
||||
<input
|
||||
type="text"
|
||||
value={settings.gemini?.model ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
gemini: { ...prev.gemini!, model: e.target.value }
|
||||
}))}
|
||||
placeholder="e.g., gemini-2.0-flash-exp"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => {
|
||||
setUseCustomModel(false);
|
||||
if (availableGeminiModels.length > 0) {
|
||||
setSettings(prev => ({
|
||||
...prev,
|
||||
gemini: { ...prev.gemini!, model: availableGeminiModels[0] }
|
||||
}));
|
||||
}
|
||||
}}
|
||||
className="text-xs text-accent hover:underline"
|
||||
>
|
||||
← Back to model list
|
||||
</button>
|
||||
</div>
|
||||
) : (
|
||||
<div className="space-y-2">
|
||||
<select
|
||||
value={settings.gemini?.model ?? 'gemini-2.0-flash'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
gemini: { ...prev.gemini!, model: e.target.value }
|
||||
}))}
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all appearance-none cursor-pointer"
|
||||
>
|
||||
{availableGeminiModels.map(model => (
|
||||
<option key={model} value={model}>{model}</option>
|
||||
))}
|
||||
</select>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => setUseCustomModel(true)}
|
||||
className="text-xs text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
Enter model name manually →
|
||||
</button>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{modelFetchError && !useCustomModel && (
|
||||
<p className="text-xs text-amber-400 flex items-center gap-1">
|
||||
<AlertCircle className="w-3 h-3" />
|
||||
{modelFetchError}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Anthropic Settings */}
|
||||
{settings.activeProvider === 'anthropic' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['anthropic'] ? 'text' : 'password'}
|
||||
value={settings.anthropic?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
anthropic: { ...prev.anthropic!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your Anthropic API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('anthropic')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['anthropic'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
Get your API key from{' '}
|
||||
<a
|
||||
href="https://console.anthropic.com/settings/keys"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
Anthropic Console
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
|
||||
{(useCustomAnthropicModel || isCustomAnthropicModelSelected) ? (
|
||||
<div className="space-y-2">
|
||||
<input
|
||||
type="text"
|
||||
value={settings.anthropic?.model ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
anthropic: { ...prev.anthropic!, model: e.target.value }
|
||||
}))}
|
||||
placeholder="e.g., claude-sonnet-4-20250514"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => {
|
||||
setUseCustomAnthropicModel(false);
|
||||
if (availableAnthropicModels.length > 0) {
|
||||
setSettings(prev => ({
|
||||
...prev,
|
||||
anthropic: { ...prev.anthropic!, model: availableAnthropicModels[0] }
|
||||
}));
|
||||
}
|
||||
}}
|
||||
className="text-xs text-accent hover:underline"
|
||||
>
|
||||
← Back to model list
|
||||
</button>
|
||||
</div>
|
||||
) : (
|
||||
<div className="space-y-2">
|
||||
<select
|
||||
value={currentAnthropicModel}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
anthropic: { ...prev.anthropic!, model: e.target.value }
|
||||
}))}
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all appearance-none cursor-pointer"
|
||||
>
|
||||
{availableAnthropicModels.map(model => (
|
||||
<option key={model} value={model}>{model}</option>
|
||||
))}
|
||||
</select>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => setUseCustomAnthropicModel(true)}
|
||||
className="text-xs text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
Enter model name manually →
|
||||
</button>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Azure OpenAI Settings */}
|
||||
{settings.activeProvider === 'azure-openai' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['azure'] ? 'text' : 'password'}
|
||||
value={settings.azureOpenAI?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
azureOpenAI: { ...prev.azureOpenAI!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your Azure OpenAI API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('azure')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['azure'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Server className="w-4 h-4" />
|
||||
Endpoint
|
||||
</label>
|
||||
<input
|
||||
type="url"
|
||||
value={settings.azureOpenAI?.endpoint ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
azureOpenAI: { ...prev.azureOpenAI!, endpoint: e.target.value }
|
||||
}))}
|
||||
placeholder="https://your-resource.openai.azure.com"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Deployment Name</label>
|
||||
<input
|
||||
type="text"
|
||||
value={settings.azureOpenAI?.deploymentName ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
azureOpenAI: { ...prev.azureOpenAI!, deploymentName: e.target.value }
|
||||
}))}
|
||||
placeholder="e.g., gpt-4o-deployment"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
</div>
|
||||
|
||||
<div className="grid grid-cols-2 gap-4">
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
<input
|
||||
type="text"
|
||||
value={settings.azureOpenAI?.model ?? 'gpt-4o'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
azureOpenAI: { ...prev.azureOpenAI!, model: e.target.value }
|
||||
}))}
|
||||
placeholder="gpt-4o"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">API Version</label>
|
||||
<input
|
||||
type="text"
|
||||
value={settings.azureOpenAI?.apiVersion ?? '2024-08-01-preview'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
azureOpenAI: { ...prev.azureOpenAI!, apiVersion: e.target.value }
|
||||
}))}
|
||||
placeholder="2024-08-01-preview"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p className="text-xs text-text-muted">
|
||||
Configure your Azure OpenAI service in the{' '}
|
||||
<a
|
||||
href="https://portal.azure.com/#view/Microsoft_Azure_ProjectOxford/CognitiveServicesHub/~/OpenAI"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
Azure Portal
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Privacy Note */}
|
||||
<div className="p-4 bg-elevated/50 border border-border-subtle rounded-xl">
|
||||
<div className="flex gap-3">
|
||||
<div className="w-8 h-8 flex items-center justify-center bg-green-500/20 rounded-lg text-green-400 flex-shrink-0">
|
||||
🔒
|
||||
</div>
|
||||
<div className="text-xs text-text-muted leading-relaxed">
|
||||
<span className="text-text-secondary font-medium">Privacy:</span> Your API keys are stored only in your browser's local storage.
|
||||
They're sent directly to the LLM provider when you chat. Your code never leaves your machine.
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Footer */}
|
||||
<div className="flex items-center justify-between px-6 py-4 border-t border-border-subtle bg-elevated/30">
|
||||
<div className="flex items-center gap-2 text-sm">
|
||||
{saveStatus === 'saved' && (
|
||||
<span className="flex items-center gap-1.5 text-green-400 animate-fade-in">
|
||||
<Check className="w-4 h-4" />
|
||||
Settings saved
|
||||
</span>
|
||||
)}
|
||||
{saveStatus === 'error' && (
|
||||
<span className="flex items-center gap-1.5 text-red-400 animate-fade-in">
|
||||
<AlertCircle className="w-4 h-4" />
|
||||
Failed to save
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
<div className="flex items-center gap-3">
|
||||
<button
|
||||
onClick={onClose}
|
||||
className="px-4 py-2 text-sm text-text-secondary hover:text-text-primary transition-colors"
|
||||
>
|
||||
Cancel
|
||||
</button>
|
||||
<button
|
||||
onClick={handleSave}
|
||||
className="px-5 py-2 bg-accent text-white text-sm font-medium rounded-lg hover:bg-accent-dim transition-colors"
|
||||
>
|
||||
Save Settings
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
|
||||
export const StatusBar = () => {
|
||||
const { graph, progress } = useAppState();
|
||||
|
||||
const nodeCount = graph?.nodes.length ?? 0;
|
||||
const edgeCount = graph?.relationships.length ?? 0;
|
||||
|
||||
// Detect primary language
|
||||
const primaryLanguage = (() => {
|
||||
if (!graph) return null;
|
||||
const languages = graph.nodes
|
||||
.map(n => n.properties.language)
|
||||
.filter(Boolean);
|
||||
if (languages.length === 0) return null;
|
||||
|
||||
const counts = languages.reduce((acc, lang) => {
|
||||
acc[lang!] = (acc[lang!] || 0) + 1;
|
||||
return acc;
|
||||
}, {} as Record<string, number>);
|
||||
|
||||
return Object.entries(counts).sort((a, b) => b[1] - a[1])[0]?.[0];
|
||||
})();
|
||||
|
||||
return (
|
||||
<footer className="flex items-center justify-between px-5 py-2 bg-deep border-t border-dashed border-border-subtle text-[11px] text-text-muted">
|
||||
{/* Left - Status */}
|
||||
<div className="flex items-center gap-4">
|
||||
{progress && progress.phase !== 'complete' ? (
|
||||
<>
|
||||
<div className="w-28 h-1 bg-elevated rounded-full overflow-hidden">
|
||||
<div
|
||||
className="h-full bg-gradient-to-r from-accent to-node-interface rounded-full transition-all duration-300"
|
||||
style={{ width: `${progress.percent}%` }}
|
||||
/>
|
||||
</div>
|
||||
<span>{progress.message}</span>
|
||||
</>
|
||||
) : (
|
||||
<div className="flex items-center gap-1.5">
|
||||
<span className="w-1.5 h-1.5 bg-node-function rounded-full" />
|
||||
<span>Ready</span>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Right - Stats */}
|
||||
<div className="flex items-center gap-3">
|
||||
{graph && (
|
||||
<>
|
||||
<span>{nodeCount} nodes</span>
|
||||
<span className="text-border-default">•</span>
|
||||
<span>{edgeCount} edges</span>
|
||||
{primaryLanguage && (
|
||||
<>
|
||||
<span className="text-border-default">•</span>
|
||||
<span>{primaryLanguage}</span>
|
||||
</>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
</footer>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,258 @@
|
||||
/**
|
||||
* ToolCallCard Component
|
||||
*
|
||||
* Displays a tool call with expand/collapse functionality.
|
||||
* Shows the tool name, status, and when expanded, the query/args and result.
|
||||
*/
|
||||
|
||||
import { useState, useCallback, useMemo } from 'react';
|
||||
import { ChevronDown, ChevronRight, Sparkles, Check, Loader2, AlertCircle, Eye, EyeOff } from 'lucide-react';
|
||||
import type { ToolCallInfo } from '../core/llm/types';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
|
||||
interface ToolCallCardProps {
|
||||
toolCall: ToolCallInfo;
|
||||
/** Start expanded (useful for in-progress calls) */
|
||||
defaultExpanded?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format tool arguments for display
|
||||
*/
|
||||
const formatArgs = (args: Record<string, unknown>): string => {
|
||||
if (!args || Object.keys(args).length === 0) {
|
||||
return '';
|
||||
}
|
||||
|
||||
// Special handling for Cypher queries
|
||||
if ('query' in args && typeof args.query === 'string') {
|
||||
return args.query;
|
||||
}
|
||||
if ('cypher' in args && typeof args.cypher === 'string') {
|
||||
// For execute_vector_cypher, show both the natural language query and cypher
|
||||
let result = '';
|
||||
if ('query' in args) {
|
||||
result += `Search: "${args.query}"\n\n`;
|
||||
}
|
||||
result += args.cypher;
|
||||
return result;
|
||||
}
|
||||
|
||||
// For other tools, show as formatted JSON
|
||||
return JSON.stringify(args, null, 2);
|
||||
};
|
||||
|
||||
/**
|
||||
* Get status icon and color
|
||||
*/
|
||||
const getStatusDisplay = (status: ToolCallInfo['status']) => {
|
||||
switch (status) {
|
||||
case 'running':
|
||||
return {
|
||||
icon: <Loader2 className="w-3.5 h-3.5 animate-spin" />,
|
||||
color: 'text-amber-400',
|
||||
bgColor: 'bg-amber-500/10',
|
||||
borderColor: 'border-amber-500/30',
|
||||
};
|
||||
case 'completed':
|
||||
return {
|
||||
icon: <Check className="w-3.5 h-3.5" />,
|
||||
color: 'text-emerald-400',
|
||||
bgColor: 'bg-emerald-500/10',
|
||||
borderColor: 'border-emerald-500/30',
|
||||
};
|
||||
case 'error':
|
||||
return {
|
||||
icon: <AlertCircle className="w-3.5 h-3.5" />,
|
||||
color: 'text-rose-400',
|
||||
bgColor: 'bg-rose-500/10',
|
||||
borderColor: 'border-rose-500/30',
|
||||
};
|
||||
default:
|
||||
return {
|
||||
icon: <Sparkles className="w-3.5 h-3.5" />,
|
||||
color: 'text-text-muted',
|
||||
bgColor: 'bg-surface',
|
||||
borderColor: 'border-border-subtle',
|
||||
};
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Get a friendly display name for the tool
|
||||
*/
|
||||
const getToolDisplayName = (name: string): string => {
|
||||
const names: Record<string, string> = {
|
||||
// New consolidated tools
|
||||
'search': '🔍 Search Code',
|
||||
'cypher': '🔍 Cypher Query',
|
||||
'grep': '🔎 Pattern Search',
|
||||
'read': '📄 Read File',
|
||||
'highlight': '✨ Highlight in Graph',
|
||||
// Legacy names (for backwards compatibility)
|
||||
'execute_cypher': '🔍 Cypher Query',
|
||||
'execute_vector_cypher': '🧠 Semantic + Graph Query',
|
||||
'highlight_in_graph': '✨ Highlight in Graph',
|
||||
'grep_code': '🔎 Pattern Search',
|
||||
'read_file': '📄 Read File',
|
||||
};
|
||||
return names[name] || name;
|
||||
};
|
||||
|
||||
/**
|
||||
* Extract node IDs from highlight tool result
|
||||
*/
|
||||
const extractHighlightNodeIds = (result: string | undefined): string[] => {
|
||||
if (!result) return [];
|
||||
const match = result.match(/\[HIGHLIGHT_NODES:([^\]]+)\]/);
|
||||
if (match) {
|
||||
return match[1].split(',').map(id => id.trim()).filter(Boolean);
|
||||
}
|
||||
return [];
|
||||
};
|
||||
|
||||
export const ToolCallCard = ({ toolCall, defaultExpanded = false }: ToolCallCardProps) => {
|
||||
const [isExpanded, setIsExpanded] = useState(defaultExpanded);
|
||||
const { highlightedNodeIds, setHighlightedNodeIds, graph } = useAppState();
|
||||
const status = getStatusDisplay(toolCall.status);
|
||||
const formattedArgs = formatArgs(toolCall.args);
|
||||
|
||||
// Check if this is a highlight tool and extract node IDs
|
||||
const isHighlightTool = toolCall.name === 'highlight_in_graph' || toolCall.name === 'highlight';
|
||||
const rawHighlightNodeIds = isHighlightTool ? extractHighlightNodeIds(toolCall.result) : [];
|
||||
|
||||
// Resolve raw IDs to actual graph node IDs (handles partial ID matching)
|
||||
const resolvedNodeIds = useMemo(() => {
|
||||
if (rawHighlightNodeIds.length === 0 || !graph) return rawHighlightNodeIds;
|
||||
|
||||
const graphNodeIds = graph.nodes.map(n => n.id);
|
||||
const resolved: string[] = [];
|
||||
|
||||
for (const rawId of rawHighlightNodeIds) {
|
||||
if (graphNodeIds.includes(rawId)) {
|
||||
resolved.push(rawId);
|
||||
} else {
|
||||
// Try partial match - find node whose ID ends with the raw ID
|
||||
const found = graphNodeIds.find(gid =>
|
||||
gid.endsWith(rawId) || gid.endsWith(':' + rawId)
|
||||
);
|
||||
if (found) resolved.push(found);
|
||||
}
|
||||
}
|
||||
return resolved;
|
||||
}, [rawHighlightNodeIds, graph]);
|
||||
|
||||
// Check if these specific nodes are currently highlighted
|
||||
const isHighlightActive = resolvedNodeIds.length > 0 &&
|
||||
resolvedNodeIds.some(id => highlightedNodeIds.has(id));
|
||||
|
||||
// Toggle highlight on/off
|
||||
const toggleHighlight = useCallback((e: React.MouseEvent) => {
|
||||
e.stopPropagation(); // Don't trigger expand/collapse
|
||||
if (isHighlightActive) {
|
||||
// Turn off - clear highlights
|
||||
setHighlightedNodeIds(new Set());
|
||||
} else {
|
||||
// Turn on - set these nodes as highlighted
|
||||
setHighlightedNodeIds(new Set(resolvedNodeIds));
|
||||
}
|
||||
}, [isHighlightActive, resolvedNodeIds, setHighlightedNodeIds]);
|
||||
|
||||
return (
|
||||
<div className={`rounded-lg border ${status.borderColor} ${status.bgColor} overflow-hidden transition-all`}>
|
||||
{/* Header - always visible */}
|
||||
<div
|
||||
role="button"
|
||||
tabIndex={0}
|
||||
onClick={() => setIsExpanded(!isExpanded)}
|
||||
onKeyDown={(e) => { if (e.key === 'Enter' || e.key === ' ') { e.preventDefault(); setIsExpanded(!isExpanded); } }}
|
||||
className="w-full flex items-center gap-2 px-3 py-2 text-left hover:bg-white/5 transition-colors cursor-pointer select-none"
|
||||
>
|
||||
{/* Expand/collapse icon */}
|
||||
<span className="text-text-muted">
|
||||
{isExpanded ? <ChevronDown className="w-4 h-4" /> : <ChevronRight className="w-4 h-4" />}
|
||||
</span>
|
||||
|
||||
{/* Tool name */}
|
||||
<span className="flex-1 text-sm font-medium text-text-primary">
|
||||
{getToolDisplayName(toolCall.name)}
|
||||
</span>
|
||||
|
||||
{/* Highlight toggle button - only for highlight_in_graph tool with results */}
|
||||
{isHighlightTool && resolvedNodeIds.length > 0 && (
|
||||
<button
|
||||
onClick={toggleHighlight}
|
||||
className={`flex items-center gap-1 px-2 py-0.5 rounded text-xs transition-colors ${isHighlightActive
|
||||
? 'bg-accent/20 text-accent hover:bg-accent/30'
|
||||
: 'bg-surface/50 text-text-muted hover:bg-surface hover:text-text-primary'
|
||||
}`}
|
||||
title={isHighlightActive ? 'Turn off highlight' : 'Turn on highlight'}
|
||||
>
|
||||
{isHighlightActive ? (
|
||||
<>
|
||||
<Eye className="w-3 h-3" />
|
||||
<span>On</span>
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
<EyeOff className="w-3 h-3" />
|
||||
<span>Off</span>
|
||||
</>
|
||||
)}
|
||||
</button>
|
||||
)}
|
||||
|
||||
{/* Status indicator */}
|
||||
<span className={`flex items-center gap-1 text-xs ${status.color}`}>
|
||||
{status.icon}
|
||||
<span className="capitalize">{toolCall.status}</span>
|
||||
</span>
|
||||
</div>
|
||||
|
||||
{/* Expanded content */}
|
||||
{isExpanded && (
|
||||
<div className="border-t border-border-subtle/50">
|
||||
{/* Arguments/Query */}
|
||||
{formattedArgs && (
|
||||
<div className="px-3 py-2 border-b border-border-subtle/50">
|
||||
<div className="text-[10px] uppercase tracking-wider text-text-muted mb-1.5">
|
||||
{toolCall.name.includes('cypher') ? 'Query' : 'Input'}
|
||||
</div>
|
||||
<pre className="text-xs text-text-secondary bg-surface/50 rounded p-2 overflow-x-auto whitespace-pre-wrap font-mono">
|
||||
{formattedArgs}
|
||||
</pre>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Result */}
|
||||
{toolCall.result && (
|
||||
<div className="px-3 py-2">
|
||||
<div className="text-[10px] uppercase tracking-wider text-text-muted mb-1.5">
|
||||
Result
|
||||
</div>
|
||||
<div className="max-h-[400px] overflow-y-auto bg-surface/50 rounded">
|
||||
<pre className="text-xs text-text-secondary p-2 whitespace-pre-wrap font-mono">
|
||||
{toolCall.result.length > 3000
|
||||
? toolCall.result.slice(0, 3000) + '\n\n... (truncated)'
|
||||
: toolCall.result
|
||||
}
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Loading state for in-progress */}
|
||||
{toolCall.status === 'running' && !toolCall.result && (
|
||||
<div className="px-3 py-3 flex items-center gap-2 text-xs text-text-muted">
|
||||
<Loader2 className="w-3 h-3 animate-spin" />
|
||||
<span>Executing...</span>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
export default ToolCallCard;
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
import { useState, useEffect } from 'react';
|
||||
import { X, Snail, Rocket, SkipForward } from 'lucide-react';
|
||||
|
||||
interface WebGPUFallbackDialogProps {
|
||||
isOpen: boolean;
|
||||
onClose: () => void;
|
||||
onUseCPU: () => void;
|
||||
onSkip: () => void;
|
||||
nodeCount: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fun dialog shown when WebGPU isn't available
|
||||
* Lets user choose: CPU fallback (slow) or skip embeddings
|
||||
*/
|
||||
export const WebGPUFallbackDialog = ({
|
||||
isOpen,
|
||||
onClose,
|
||||
onUseCPU,
|
||||
onSkip,
|
||||
nodeCount,
|
||||
}: WebGPUFallbackDialogProps) => {
|
||||
const [isAnimating, setIsAnimating] = useState(true);
|
||||
const [isVisible, setIsVisible] = useState(false);
|
||||
|
||||
useEffect(() => {
|
||||
if (isOpen) {
|
||||
// Trigger animation after mount
|
||||
requestAnimationFrame(() => setIsVisible(true));
|
||||
} else {
|
||||
setIsVisible(false);
|
||||
}
|
||||
}, [isOpen]);
|
||||
|
||||
if (!isOpen) return null;
|
||||
|
||||
// Estimate time based on node count (rough: ~50ms per node on CPU)
|
||||
const estimatedMinutes = Math.ceil((nodeCount * 50) / 60000);
|
||||
const isSmallCodebase = nodeCount < 200;
|
||||
|
||||
return (
|
||||
<div className="fixed inset-0 z-50 flex items-center justify-center">
|
||||
{/* Backdrop */}
|
||||
<div
|
||||
className={`absolute inset-0 bg-black/60 backdrop-blur-sm transition-opacity duration-200 ${isVisible ? 'opacity-100' : 'opacity-0'}`}
|
||||
onClick={onClose}
|
||||
/>
|
||||
|
||||
{/* Dialog */}
|
||||
<div
|
||||
className={`relative bg-surface border border-border-subtle rounded-2xl shadow-2xl max-w-md w-full mx-4 overflow-hidden transition-all duration-200 ${isVisible ? 'opacity-100 scale-100' : 'opacity-0 scale-95'}`}
|
||||
>
|
||||
{/* Header with scratching emoji */}
|
||||
<div className="relative bg-gradient-to-r from-amber-500/20 to-orange-500/20 px-6 py-5 border-b border-border-subtle">
|
||||
<button
|
||||
onClick={onClose}
|
||||
className="absolute top-4 right-4 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
<X className="w-5 h-5" />
|
||||
</button>
|
||||
|
||||
<div className="flex items-center gap-4">
|
||||
{/* Animated emoji */}
|
||||
<div
|
||||
className={`text-5xl ${isAnimating ? 'animate-bounce' : ''}`}
|
||||
onAnimationEnd={() => setIsAnimating(false)}
|
||||
onClick={() => setIsAnimating(true)}
|
||||
>
|
||||
🤔
|
||||
</div>
|
||||
<div>
|
||||
<h2 className="text-lg font-semibold text-text-primary">
|
||||
WebGPU said "nope"
|
||||
</h2>
|
||||
<p className="text-sm text-text-muted mt-0.5">
|
||||
Your browser doesn't support GPU acceleration
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Content */}
|
||||
<div className="px-6 py-5 space-y-4">
|
||||
<p className="text-sm text-text-secondary leading-relaxed">
|
||||
Couldn't create embeddings with WebGPU, so semantic search (Graph RAG)
|
||||
won't be as smart. The graph still works fine though!
|
||||
</p>
|
||||
|
||||
<div className="bg-elevated/50 rounded-lg p-4 border border-border-subtle">
|
||||
<p className="text-sm text-text-secondary">
|
||||
<span className="font-medium text-text-primary">Your options:</span>
|
||||
</p>
|
||||
<ul className="mt-2 space-y-1.5 text-sm text-text-muted">
|
||||
<li className="flex items-start gap-2">
|
||||
<Snail className="w-4 h-4 mt-0.5 text-amber-400 flex-shrink-0" />
|
||||
<span>
|
||||
<strong className="text-text-secondary">Use CPU</strong> — Works but {isSmallCodebase ? 'a bit' : 'way'} slower
|
||||
{nodeCount > 0 && (
|
||||
<span className="text-text-muted"> (~{estimatedMinutes} min for {nodeCount} nodes)</span>
|
||||
)}
|
||||
</span>
|
||||
</li>
|
||||
<li className="flex items-start gap-2">
|
||||
<SkipForward className="w-4 h-4 mt-0.5 text-blue-400 flex-shrink-0" />
|
||||
<span>
|
||||
<strong className="text-text-secondary">Skip it</strong> — Graph works, just no AI semantic search
|
||||
</span>
|
||||
</li>
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
{isSmallCodebase && (
|
||||
<p className="text-xs text-node-function flex items-center gap-1.5 bg-node-function/10 px-3 py-2 rounded-lg">
|
||||
<Rocket className="w-3.5 h-3.5" />
|
||||
Small codebase detected! CPU should be fine.
|
||||
</p>
|
||||
)}
|
||||
|
||||
<p className="text-xs text-text-muted">
|
||||
💡 Tip: Try Chrome or Edge for WebGPU support
|
||||
</p>
|
||||
</div>
|
||||
|
||||
{/* Actions */}
|
||||
<div className="px-6 py-4 bg-elevated/30 border-t border-border-subtle flex gap-3">
|
||||
<button
|
||||
onClick={onSkip}
|
||||
className="flex-1 px-4 py-2.5 text-sm font-medium text-text-secondary bg-surface border border-border-subtle rounded-lg hover:bg-hover hover:text-text-primary transition-all flex items-center justify-center gap-2"
|
||||
>
|
||||
<SkipForward className="w-4 h-4" />
|
||||
Skip Embeddings
|
||||
</button>
|
||||
<button
|
||||
onClick={onUseCPU}
|
||||
className={`flex-1 px-4 py-2.5 text-sm font-medium rounded-lg transition-all flex items-center justify-center gap-2 ${
|
||||
isSmallCodebase
|
||||
? 'bg-node-function text-white hover:bg-node-function/90'
|
||||
: 'bg-amber-500/20 text-amber-300 border border-amber-500/30 hover:bg-amber-500/30'
|
||||
}`}
|
||||
>
|
||||
<Snail className="w-4 h-4" />
|
||||
Use CPU {isSmallCodebase ? '(Recommended)' : '(Slow)'}
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -1,346 +0,0 @@
|
||||
/**
|
||||
* Configuration Loader
|
||||
*
|
||||
* Loads and validates configuration from the root gitnexus.config.ts file
|
||||
* with Zod schema validation and type safety.
|
||||
*/
|
||||
|
||||
import { z } from 'zod';
|
||||
|
||||
// Zod schemas for validation
|
||||
|
||||
const ProcessingConfigSchema = z.object({
|
||||
mode: z.enum(['parallel', 'single']),
|
||||
workers: z.object({
|
||||
mode: z.enum(['auto', 'manual']),
|
||||
auto: z.object({
|
||||
enabled: z.boolean(),
|
||||
maxWorkers: z.number().min(1).max(32),
|
||||
memoryPerWorkerMB: z.number().min(20).max(200),
|
||||
cpuMultiplier: z.number().min(0.1).max(2.0)
|
||||
}),
|
||||
manual: z.object({
|
||||
count: z.number().min(1).max(32)
|
||||
})
|
||||
}),
|
||||
parallel: z.object({
|
||||
batchSize: z.number().min(1).max(100),
|
||||
workerTimeoutMs: z.number().min(10000).max(300000)
|
||||
}),
|
||||
memory: z.object({
|
||||
maxMB: z.number().min(100).max(4096),
|
||||
cleanupThresholdMB: z.number().min(50).max(2048),
|
||||
gcIntervalMs: z.number().min(5000).max(60000),
|
||||
maxFileSizeMB: z.number().min(1).max(100),
|
||||
maxFilesInMemory: z.number().min(100).max(10000)
|
||||
}),
|
||||
fileExtensions: z.array(z.string()),
|
||||
performanceMonitoring: z.boolean()
|
||||
});
|
||||
|
||||
const KuzuConfigSchema = z.object({
|
||||
enabled: z.boolean(),
|
||||
persistence: z.boolean(),
|
||||
dualWrite: z.boolean(),
|
||||
fallbackToJson: z.boolean(),
|
||||
performance: z.object({
|
||||
enableCache: z.boolean(),
|
||||
cacheSize: z.number().min(100).max(10000),
|
||||
queryTimeout: z.number().min(1000).max(60000)
|
||||
})
|
||||
});
|
||||
|
||||
const AIConfigSchema = z.object({
|
||||
cypher: z.object({
|
||||
defaultLimit: z.number().min(1).max(1000),
|
||||
maxLimit: z.number().min(10).max(1000),
|
||||
timeoutMs: z.number().min(1000).max(60000),
|
||||
enableValidation: z.boolean(),
|
||||
enableLimiting: z.boolean(),
|
||||
enableTruncation: z.boolean()
|
||||
}),
|
||||
llm: z.object({
|
||||
defaultProvider: z.enum(['openai', 'azure', 'anthropic', 'gemini']),
|
||||
providers: z.object({
|
||||
openai: z.object({
|
||||
apiKey: z.string().optional(),
|
||||
model: z.string(),
|
||||
maxTokens: z.number().min(100).max(10000),
|
||||
temperature: z.number().min(0).max(2)
|
||||
}).optional(),
|
||||
azure: z.object({
|
||||
apiKey: z.string().optional(),
|
||||
endpoint: z.string().optional(),
|
||||
deployment: z.string().optional(),
|
||||
maxTokens: z.number().min(100).max(10000),
|
||||
temperature: z.number().min(0).max(2)
|
||||
}).optional(),
|
||||
anthropic: z.object({
|
||||
apiKey: z.string().optional(),
|
||||
model: z.string(),
|
||||
maxTokens: z.number().min(100).max(10000),
|
||||
temperature: z.number().min(0).max(2)
|
||||
}).optional(),
|
||||
gemini: z.object({
|
||||
apiKey: z.string().optional(),
|
||||
model: z.string(),
|
||||
maxTokens: z.number().min(100).max(10000),
|
||||
temperature: z.number().min(0).max(2)
|
||||
}).optional()
|
||||
})
|
||||
})
|
||||
});
|
||||
|
||||
const IgnoreConfigSchema = z.object({
|
||||
enabled: z.boolean(),
|
||||
patterns: z.array(z.string()),
|
||||
suffixes: z.array(z.string()),
|
||||
fileExtensions: z.array(z.string()),
|
||||
customPatterns: z.array(z.string())
|
||||
});
|
||||
|
||||
const LoggingConfigSchema = z.object({
|
||||
level: z.enum(['debug', 'info', 'warn', 'error']),
|
||||
enableMetrics: z.boolean(),
|
||||
enablePerformance: z.boolean(),
|
||||
maxEntries: z.number().min(100).max(10000),
|
||||
monitoringIntervalMs: z.number().min(5000).max(60000)
|
||||
});
|
||||
|
||||
const GitHubConfigSchema = z.object({
|
||||
token: z.string().optional(),
|
||||
apiUrl: z.string().url(),
|
||||
rateLimit: z.object({
|
||||
maxRequests: z.number().min(1).max(5000),
|
||||
windowMs: z.number().min(1000).max(3600000)
|
||||
}),
|
||||
retry: z.object({
|
||||
maxRetries: z.number().min(0).max(10),
|
||||
backoffMs: z.number().min(100).max(10000)
|
||||
})
|
||||
});
|
||||
|
||||
const FeaturesConfigSchema = z.object({
|
||||
// AI Features
|
||||
enableAdvancedRAG: z.boolean(),
|
||||
enableReActReasoning: z.boolean(),
|
||||
enableMultiLLM: z.boolean(),
|
||||
|
||||
// Performance Features
|
||||
enableWebWorkers: z.boolean(),
|
||||
enableBatchProcessing: z.boolean(),
|
||||
enableKuzuCopy: z.boolean(),
|
||||
enablePolymorphicNodes: z.boolean(),
|
||||
enableCaching: z.boolean(),
|
||||
enableWorkerPool: z.boolean(),
|
||||
enableParallelParsing: z.boolean(),
|
||||
enableParallelProcessing: z.boolean(),
|
||||
|
||||
// Debug Features
|
||||
enableDebugMode: z.boolean(),
|
||||
enablePerformanceLogging: z.boolean(),
|
||||
enableQueryLogging: z.boolean()
|
||||
});
|
||||
|
||||
const GitNexusConfigSchema = z.object({
|
||||
processing: ProcessingConfigSchema,
|
||||
kuzu: KuzuConfigSchema,
|
||||
ai: AIConfigSchema,
|
||||
ignore: IgnoreConfigSchema,
|
||||
logging: LoggingConfigSchema,
|
||||
github: GitHubConfigSchema,
|
||||
features: FeaturesConfigSchema,
|
||||
environment: z.enum(['development', 'staging', 'production'])
|
||||
});
|
||||
|
||||
export type ValidatedGitNexusConfig = z.infer<typeof GitNexusConfigSchema>;
|
||||
|
||||
/**
|
||||
* Configuration Loader Service
|
||||
*/
|
||||
export class ConfigLoader {
|
||||
private static instance: ConfigLoader;
|
||||
private config: ValidatedGitNexusConfig | null = null;
|
||||
private validationErrors: string[] = [];
|
||||
|
||||
private constructor() {}
|
||||
|
||||
public static getInstance(): ConfigLoader {
|
||||
if (!ConfigLoader.instance) {
|
||||
ConfigLoader.instance = new ConfigLoader();
|
||||
}
|
||||
return ConfigLoader.instance;
|
||||
}
|
||||
|
||||
/**
|
||||
* Load and validate configuration from gitnexus.config.ts
|
||||
*/
|
||||
public async loadConfig(): Promise<ValidatedGitNexusConfig> {
|
||||
if (this.config) {
|
||||
return this.config;
|
||||
}
|
||||
|
||||
try {
|
||||
// Import the config file
|
||||
const configModule = await import('../../gitnexus.config.ts');
|
||||
const rawConfig = configModule.default;
|
||||
|
||||
// Validate with Zod
|
||||
const result = GitNexusConfigSchema.safeParse(rawConfig);
|
||||
|
||||
if (!result.success) {
|
||||
this.validationErrors = result.error.errors.map(e =>
|
||||
`${e.path.join('.')}: ${e.message}`
|
||||
);
|
||||
console.error('❌ Configuration validation errors:', this.validationErrors);
|
||||
|
||||
// Return a minimal valid config as fallback
|
||||
this.config = this.getMinimalConfig();
|
||||
} else {
|
||||
this.config = result.data;
|
||||
console.log('✅ Configuration loaded and validated successfully');
|
||||
}
|
||||
|
||||
return this.config;
|
||||
} catch (error) {
|
||||
console.error('❌ Failed to load gitnexus.config.ts:', error);
|
||||
this.config = this.getMinimalConfig();
|
||||
return this.config;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get minimal fallback configuration
|
||||
*/
|
||||
private getMinimalConfig(): ValidatedGitNexusConfig {
|
||||
return {
|
||||
processing: {
|
||||
mode: 'parallel',
|
||||
workers: {
|
||||
mode: 'auto',
|
||||
auto: {
|
||||
enabled: true,
|
||||
maxWorkers: 16,
|
||||
memoryPerWorkerMB: 60,
|
||||
cpuMultiplier: 0.75
|
||||
},
|
||||
manual: {
|
||||
count: 4
|
||||
}
|
||||
},
|
||||
parallel: {
|
||||
batchSize: 20,
|
||||
workerTimeoutMs: 60000
|
||||
},
|
||||
memory: {
|
||||
maxMB: 512,
|
||||
cleanupThresholdMB: 400,
|
||||
gcIntervalMs: 30000,
|
||||
maxFileSizeMB: 10,
|
||||
maxFilesInMemory: 1000
|
||||
},
|
||||
fileExtensions: ['.js', '.ts', '.jsx', '.tsx', '.py'],
|
||||
performanceMonitoring: true
|
||||
},
|
||||
kuzu: {
|
||||
enabled: true,
|
||||
persistence: true,
|
||||
dualWrite: true,
|
||||
fallbackToJson: true,
|
||||
performance: {
|
||||
enableCache: true,
|
||||
cacheSize: 1000,
|
||||
queryTimeout: 30000
|
||||
}
|
||||
},
|
||||
ai: {
|
||||
cypher: {
|
||||
defaultLimit: 20,
|
||||
maxLimit: 100,
|
||||
timeoutMs: 30000,
|
||||
enableValidation: true,
|
||||
enableLimiting: true,
|
||||
enableTruncation: true
|
||||
},
|
||||
llm: {
|
||||
defaultProvider: 'openai',
|
||||
providers: {}
|
||||
}
|
||||
},
|
||||
ignore: {
|
||||
enabled: true,
|
||||
patterns: ['node_modules', '.git', 'build', 'dist'],
|
||||
suffixes: ['.tmp', '~'],
|
||||
fileExtensions: ['.pyc', '.zip', '.jpg'],
|
||||
customPatterns: []
|
||||
},
|
||||
logging: {
|
||||
level: 'info',
|
||||
enableMetrics: true,
|
||||
enablePerformance: true,
|
||||
maxEntries: 1000,
|
||||
monitoringIntervalMs: 30000
|
||||
},
|
||||
github: {
|
||||
token: '',
|
||||
apiUrl: 'https://api.github.com',
|
||||
rateLimit: {
|
||||
maxRequests: 60,
|
||||
windowMs: 60000
|
||||
},
|
||||
retry: {
|
||||
maxRetries: 3,
|
||||
backoffMs: 1000
|
||||
}
|
||||
},
|
||||
features: {
|
||||
// AI Features
|
||||
enableAdvancedRAG: true,
|
||||
enableReActReasoning: true,
|
||||
enableMultiLLM: true,
|
||||
|
||||
// Performance Features
|
||||
enableWebWorkers: true,
|
||||
enableBatchProcessing: true,
|
||||
enableKuzuCopy: false,
|
||||
enablePolymorphicNodes: false,
|
||||
enableCaching: true,
|
||||
enableWorkerPool: true,
|
||||
enableParallelParsing: true,
|
||||
enableParallelProcessing: true,
|
||||
|
||||
// Debug Features
|
||||
enableDebugMode: false,
|
||||
enablePerformanceLogging: true,
|
||||
enableQueryLogging: false
|
||||
},
|
||||
environment: 'development'
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Get current configuration
|
||||
*/
|
||||
public getConfig(): ValidatedGitNexusConfig | null {
|
||||
return this.config;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get validation errors
|
||||
*/
|
||||
public getValidationErrors(): string[] {
|
||||
return [...this.validationErrors];
|
||||
}
|
||||
|
||||
/**
|
||||
* Reload configuration (useful for development)
|
||||
*/
|
||||
public async reloadConfig(): Promise<ValidatedGitNexusConfig> {
|
||||
this.config = null;
|
||||
this.validationErrors = [];
|
||||
return this.loadConfig();
|
||||
}
|
||||
}
|
||||
|
||||
// Export singleton instance
|
||||
export const configLoader = ConfigLoader.getInstance();
|
||||
@@ -1,347 +0,0 @@
|
||||
/**
|
||||
* Legacy Configuration System (DEPRECATED)
|
||||
*
|
||||
* This file is being replaced by the new centralized config system:
|
||||
* - Root config: gitnexus.config.ts
|
||||
* - Config loader: src/config/config-loader.ts
|
||||
* - Ignore service: src/config/ignore-service.ts
|
||||
*
|
||||
* TODO: Migrate remaining usage to the new system
|
||||
*/
|
||||
|
||||
import { z } from 'zod';
|
||||
import { configLoader, type ValidatedGitNexusConfig } from './config-loader.ts';
|
||||
|
||||
// Configuration schemas with validation
|
||||
const MemoryConfigSchema = z.object({
|
||||
maxMemoryMB: z.number().min(100).max(2048).default(512),
|
||||
cleanupThresholdMB: z.number().min(50).max(1024).default(400),
|
||||
gcIntervalMs: z.number().min(5000).max(60000).default(30000),
|
||||
maxFileSizeMB: z.number().min(1).max(50).default(10),
|
||||
maxFilesInMemory: z.number().min(100).max(10000).default(1000)
|
||||
});
|
||||
|
||||
const GitHubConfigSchema = z.object({
|
||||
apiUrl: z.string().url().default('https://api.github.com'),
|
||||
token: z.string().optional(),
|
||||
rateLimit: z.object({
|
||||
maxRequests: z.number().min(1).max(5000).default(60),
|
||||
windowMs: z.number().min(1000).max(3600000).default(60000)
|
||||
}),
|
||||
retry: z.object({
|
||||
maxRetries: z.number().min(0).max(5).default(3),
|
||||
backoffMs: z.number().min(100).max(10000).default(1000)
|
||||
})
|
||||
});
|
||||
|
||||
const LLMConfigSchema = z.object({
|
||||
providers: z.object({
|
||||
openai: z.object({
|
||||
apiKey: z.string().optional(),
|
||||
model: z.string().default('gpt-4'),
|
||||
maxTokens: z.number().min(100).max(10000).default(2000),
|
||||
temperature: z.number().min(0).max(2).default(0.7)
|
||||
}).optional(),
|
||||
azure: z.object({
|
||||
apiKey: z.string().optional(),
|
||||
endpoint: z.string().url().optional(),
|
||||
deployment: z.string().optional(),
|
||||
maxTokens: z.number().min(100).max(10000).default(2000),
|
||||
temperature: z.number().min(0).max(2).default(0.7)
|
||||
}).optional(),
|
||||
anthropic: z.object({
|
||||
apiKey: z.string().optional(),
|
||||
model: z.string().default('claude-3-sonnet-20240229'),
|
||||
maxTokens: z.number().min(100).max(10000).default(2000),
|
||||
temperature: z.number().min(0).max(2).default(0.7)
|
||||
}).optional(),
|
||||
gemini: z.object({
|
||||
apiKey: z.string().optional(),
|
||||
model: z.string().default('gemini-pro'),
|
||||
maxTokens: z.number().min(100).max(10000).default(2000),
|
||||
temperature: z.number().min(0).max(2).default(0.7)
|
||||
}).optional()
|
||||
}),
|
||||
defaultProvider: z.enum(['openai', 'azure', 'anthropic', 'gemini']).default('openai')
|
||||
});
|
||||
|
||||
const ProcessingConfigSchema = z.object({
|
||||
batchSize: z.number().min(1).max(100).default(10),
|
||||
maxConcurrentRequests: z.number().min(1).max(50).default(5),
|
||||
timeoutMs: z.number().min(1000).max(300000).default(30000),
|
||||
retry: z.object({
|
||||
maxRetries: z.number().min(0).max(5).default(3),
|
||||
backoffMs: z.number().min(100).max(10000).default(1000)
|
||||
}),
|
||||
fileExtensions: z.array(z.string()).default([
|
||||
'.js', '.ts', '.jsx', '.tsx', '.py', '.java', '.cpp', '.c', '.h', '.hpp',
|
||||
'.cs', '.php', '.rb', '.go', '.rs', '.swift', '.kt', '.scala', '.dart',
|
||||
'.json', '.yaml', '.yml', '.xml', '.toml', '.ini', '.cfg', '.properties'
|
||||
])
|
||||
});
|
||||
|
||||
const LoggingConfigSchema = z.object({
|
||||
level: z.enum(['debug', 'info', 'warn', 'error']).default('info'),
|
||||
enableMetrics: z.boolean().default(true),
|
||||
enablePerformanceTracking: z.boolean().default(true),
|
||||
maxLogEntries: z.number().min(100).max(10000).default(1000),
|
||||
monitoringIntervalMs: z.number().min(5000).max(60000).default(30000)
|
||||
});
|
||||
|
||||
// Main configuration schema
|
||||
const AppConfigSchema = z.object({
|
||||
memory: MemoryConfigSchema,
|
||||
github: GitHubConfigSchema,
|
||||
llm: LLMConfigSchema,
|
||||
processing: ProcessingConfigSchema,
|
||||
logging: LoggingConfigSchema,
|
||||
environment: z.enum(['development', 'staging', 'production']).default('development')
|
||||
});
|
||||
|
||||
export type AppConfig = z.infer<typeof AppConfigSchema>;
|
||||
export type MemoryConfig = z.infer<typeof MemoryConfigSchema>;
|
||||
export type GitHubConfig = z.infer<typeof GitHubConfigSchema>;
|
||||
export type LLMConfig = z.infer<typeof LLMConfigSchema>;
|
||||
export type ProcessingConfig = z.infer<typeof ProcessingConfigSchema>;
|
||||
export type LoggingConfig = z.infer<typeof LoggingConfigSchema>;
|
||||
|
||||
/**
|
||||
* Configuration service with environment validation (LEGACY)
|
||||
*
|
||||
* This is a compatibility layer that bridges to the new centralized config system.
|
||||
* New code should use configLoader directly.
|
||||
*/
|
||||
export class ConfigService {
|
||||
private static instance: ConfigService;
|
||||
private config: AppConfig;
|
||||
private validationErrors: string[] = [];
|
||||
private newConfig: ValidatedGitNexusConfig | null = null;
|
||||
|
||||
private constructor() {
|
||||
this.config = this.loadConfiguration();
|
||||
this.validateEnvironment();
|
||||
this.loadNewConfig();
|
||||
}
|
||||
|
||||
/**
|
||||
* Load the new centralized configuration
|
||||
*/
|
||||
private async loadNewConfig(): Promise<void> {
|
||||
try {
|
||||
this.newConfig = await configLoader.loadConfig();
|
||||
} catch (error) {
|
||||
console.warn('Failed to load new config system, using legacy config:', error);
|
||||
}
|
||||
}
|
||||
|
||||
public static getInstance(): ConfigService {
|
||||
if (!ConfigService.instance) {
|
||||
ConfigService.instance = new ConfigService();
|
||||
}
|
||||
return ConfigService.instance;
|
||||
}
|
||||
|
||||
/**
|
||||
* Load configuration from environment variables and defaults
|
||||
*/
|
||||
private loadConfiguration(): AppConfig {
|
||||
try {
|
||||
const config: AppConfig = {
|
||||
memory: {
|
||||
maxMemoryMB: this.getEnvNumber('MEMORY_MAX_MB', 512),
|
||||
cleanupThresholdMB: this.getEnvNumber('MEMORY_CLEANUP_THRESHOLD_MB', 400),
|
||||
gcIntervalMs: this.getEnvNumber('MEMORY_GC_INTERVAL_MS', 30000),
|
||||
maxFileSizeMB: this.getEnvNumber('MEMORY_MAX_FILE_SIZE_MB', 10),
|
||||
maxFilesInMemory: this.getEnvNumber('MEMORY_MAX_FILES', 1000)
|
||||
},
|
||||
github: {
|
||||
apiUrl: this.getEnvString('GITHUB_API_URL', 'https://api.github.com') ?? 'https://api.github.com',
|
||||
token: this.getEnvString('GITHUB_TOKEN'),
|
||||
rateLimit: {
|
||||
maxRequests: this.getEnvNumber('GITHUB_RATE_LIMIT_MAX', 60),
|
||||
windowMs: this.getEnvNumber('GITHUB_RATE_LIMIT_WINDOW_MS', 60000)
|
||||
},
|
||||
retry: {
|
||||
maxRetries: this.getEnvNumber('GITHUB_RETRY_MAX', 3),
|
||||
backoffMs: this.getEnvNumber('GITHUB_RETRY_BACKOFF_MS', 1000)
|
||||
}
|
||||
},
|
||||
llm: {
|
||||
providers: {
|
||||
openai: this.getLLMProviderConfig('OPENAI'),
|
||||
azure: this.getLLMProviderConfig('AZURE'),
|
||||
anthropic: this.getLLMProviderConfig('ANTHROPIC'),
|
||||
gemini: this.getLLMProviderConfig('GEMINI')
|
||||
},
|
||||
defaultProvider: (this.getEnvString('LLM_DEFAULT_PROVIDER', 'openai') as 'openai' | 'azure' | 'anthropic' | 'gemini') ?? 'openai'
|
||||
},
|
||||
processing: {
|
||||
batchSize: this.getEnvNumber('PROCESSING_BATCH_SIZE', 10),
|
||||
maxConcurrentRequests: this.getEnvNumber('PROCESSING_MAX_CONCURRENT', 5),
|
||||
timeoutMs: this.getEnvNumber('PROCESSING_TIMEOUT_MS', 30000),
|
||||
retry: {
|
||||
maxRetries: this.getEnvNumber('PROCESSING_RETRY_MAX', 3),
|
||||
backoffMs: this.getEnvNumber('PROCESSING_RETRY_BACKOFF_MS', 1000)
|
||||
},
|
||||
fileExtensions: this.getEnvArray('PROCESSING_FILE_EXTENSIONS', [
|
||||
'.js', '.ts', '.jsx', '.tsx', '.py', '.java', '.cpp', '.c', '.h', '.hpp',
|
||||
'.cs', '.php', '.rb', '.go', '.rs', '.swift', '.kt', '.scala', '.dart',
|
||||
'.json', '.yaml', '.yml', '.xml', '.toml', '.ini', '.cfg', '.properties'
|
||||
])
|
||||
},
|
||||
logging: {
|
||||
level: (this.getEnvString('LOG_LEVEL', 'info') as 'debug' | 'info' | 'warn' | 'error') ?? 'info',
|
||||
enableMetrics: this.getEnvBoolean('LOG_ENABLE_METRICS', true),
|
||||
enablePerformanceTracking: this.getEnvBoolean('LOG_ENABLE_PERFORMANCE', true),
|
||||
maxLogEntries: this.getEnvNumber('LOG_MAX_ENTRIES', 1000),
|
||||
monitoringIntervalMs: this.getEnvNumber('LOG_MONITORING_INTERVAL_MS', 30000)
|
||||
},
|
||||
environment: (this.getEnvString('NODE_ENV', 'development') as 'development' | 'staging' | 'production') ?? 'development'
|
||||
};
|
||||
|
||||
// Validate with Zod
|
||||
const result = AppConfigSchema.safeParse(config);
|
||||
if (!result.success) {
|
||||
this.validationErrors = result.error.errors.map(e => `${e.path.join('.')}: ${e.message}`);
|
||||
console.warn('Configuration validation errors:', this.validationErrors);
|
||||
}
|
||||
|
||||
return result.success ? result.data : AppConfigSchema.parse({
|
||||
memory: {},
|
||||
github: { rateLimit: {}, retry: {} },
|
||||
llm: { providers: {} },
|
||||
processing: { retry: {} },
|
||||
logging: {},
|
||||
});
|
||||
} catch (error) {
|
||||
console.error('Failed to load configuration:', error);
|
||||
return AppConfigSchema.parse({});
|
||||
}
|
||||
}
|
||||
|
||||
private getLLMProviderConfig(prefix: string) {
|
||||
const apiKey = this.getEnvString(`${prefix}_API_KEY`);
|
||||
if (!apiKey) return undefined;
|
||||
|
||||
return {
|
||||
apiKey,
|
||||
model: this.getEnvString(`${prefix}_MODEL`) ?? 'gpt-4',
|
||||
maxTokens: this.getEnvNumber(`${prefix}_MAX_TOKENS`, 2000),
|
||||
temperature: this.getEnvNumber(`${prefix}_TEMPERATURE`, 0.7)
|
||||
};
|
||||
}
|
||||
|
||||
private getEnvString(key: string, defaultValue?: string): string | undefined {
|
||||
return typeof process !== 'undefined' ? process.env[key] : defaultValue;
|
||||
}
|
||||
|
||||
private getEnvNumber(key: string, defaultValue: number): number {
|
||||
const value = this.getEnvString(key);
|
||||
return value ? parseInt(value, 10) || defaultValue : defaultValue;
|
||||
}
|
||||
|
||||
private getEnvBoolean(key: string, defaultValue: boolean): boolean {
|
||||
const value = this.getEnvString(key);
|
||||
return value ? value.toLowerCase() === 'true' : defaultValue;
|
||||
}
|
||||
|
||||
private getEnvArray(key: string, defaultValue: string[]): string[] {
|
||||
const value = this.getEnvString(key);
|
||||
return value ? value.split(',').map(s => s.trim()) : defaultValue;
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate environment configuration
|
||||
*/
|
||||
private validateEnvironment(): void {
|
||||
const warnings: string[] = [];
|
||||
|
||||
// Check for required API keys
|
||||
if (!this.config.github.token) {
|
||||
warnings.push('GitHub token not provided - rate limits will be lower');
|
||||
}
|
||||
|
||||
const hasAnyLLMProvider = Object.values(this.config.llm.providers)
|
||||
.some(provider => provider?.apiKey);
|
||||
|
||||
if (!hasAnyLLMProvider) {
|
||||
warnings.push('No LLM provider API keys configured - AI features disabled');
|
||||
}
|
||||
|
||||
// Check memory limits
|
||||
if (this.config.memory.maxMemoryMB < 256) {
|
||||
warnings.push('Memory limit is very low - may cause performance issues');
|
||||
}
|
||||
|
||||
if (warnings.length > 0) {
|
||||
console.warn('Configuration warnings:', warnings);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get configuration values
|
||||
*/
|
||||
public getConfiguration(): AppConfig {
|
||||
return this.config;
|
||||
}
|
||||
|
||||
public get memory(): MemoryConfig {
|
||||
return this.config.memory;
|
||||
}
|
||||
|
||||
public get github(): GitHubConfig {
|
||||
return this.config.github;
|
||||
}
|
||||
|
||||
public get llm(): LLMConfig {
|
||||
return this.config.llm;
|
||||
}
|
||||
|
||||
public get processing(): ProcessingConfig {
|
||||
return this.config.processing;
|
||||
}
|
||||
|
||||
public get logging(): LoggingConfig {
|
||||
return this.config.logging;
|
||||
}
|
||||
|
||||
public get environment(): string {
|
||||
return this.config.environment;
|
||||
}
|
||||
|
||||
public get isDevelopment(): boolean {
|
||||
return this.config.environment === 'development';
|
||||
}
|
||||
|
||||
public get isProduction(): boolean {
|
||||
return this.config.environment === 'production';
|
||||
}
|
||||
|
||||
/**
|
||||
* Get validation errors
|
||||
*/
|
||||
public getValidationErrors(): string[] {
|
||||
return [...this.validationErrors];
|
||||
}
|
||||
|
||||
/**
|
||||
* Update configuration at runtime
|
||||
*/
|
||||
public updateConfig(updates: Partial<AppConfig>): void {
|
||||
try {
|
||||
const newConfig = { ...this.config, ...updates };
|
||||
const result = AppConfigSchema.safeParse(newConfig);
|
||||
if (result.success) {
|
||||
this.config = result.data;
|
||||
} else {
|
||||
throw new Error(`Invalid configuration: ${result.error.errors.map(e => e.message).join(', ')}`);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('Failed to update configuration:', error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Export singleton instance
|
||||
export const config = ConfigService.getInstance();
|
||||
@@ -1,169 +0,0 @@
|
||||
/**
|
||||
* Feature Flag Utilities
|
||||
*
|
||||
* Centralized feature flag access using GitNexus configuration.
|
||||
* This replaces the old feature-flags.ts system.
|
||||
*/
|
||||
|
||||
import { ConfigLoader } from './config-loader.ts';
|
||||
import type { ValidatedGitNexusConfig } from './config-loader.ts';
|
||||
|
||||
let cachedConfig: ValidatedGitNexusConfig | null = null;
|
||||
|
||||
/**
|
||||
* Get the current configuration (with caching)
|
||||
*/
|
||||
async function getConfig(): Promise<ValidatedGitNexusConfig> {
|
||||
if (!cachedConfig) {
|
||||
cachedConfig = await ConfigLoader.getInstance().loadConfig();
|
||||
}
|
||||
return cachedConfig;
|
||||
}
|
||||
|
||||
/**
|
||||
* Synchronous feature flag checks (uses cached config)
|
||||
* These functions maintain compatibility with the old feature-flags.ts API
|
||||
*/
|
||||
|
||||
// KuzuDB Features
|
||||
export function isKuzuDBEnabled(): boolean {
|
||||
return cachedConfig?.kuzu.enabled ?? true;
|
||||
}
|
||||
|
||||
export function isKuzuDBPersistenceEnabled(): boolean {
|
||||
return cachedConfig?.kuzu.persistence ?? true;
|
||||
}
|
||||
|
||||
export function isKuzuDBPerformanceMonitoringEnabled(): boolean {
|
||||
return cachedConfig?.features.enablePerformanceLogging ?? true;
|
||||
}
|
||||
|
||||
export function isKuzuCopyEnabled(): boolean {
|
||||
return cachedConfig?.features.enableKuzuCopy ?? false;
|
||||
}
|
||||
|
||||
export function isPolymorphicNodesEnabled(): boolean {
|
||||
return cachedConfig?.features.enablePolymorphicNodes ?? false;
|
||||
}
|
||||
|
||||
// Processing Features
|
||||
export function isParallelParsingEnabled(): boolean {
|
||||
return cachedConfig?.features.enableParallelParsing ?? true;
|
||||
}
|
||||
|
||||
export function isParallelProcessingEnabled(): boolean {
|
||||
return cachedConfig?.processing.mode === 'parallel' &&
|
||||
(cachedConfig?.features.enableParallelProcessing ?? true);
|
||||
}
|
||||
|
||||
export function isWebWorkersEnabled(): boolean {
|
||||
return cachedConfig?.features.enableWebWorkers ?? true;
|
||||
}
|
||||
|
||||
export function isWorkerPoolEnabled(): boolean {
|
||||
return cachedConfig?.features.enableWorkerPool ?? true;
|
||||
}
|
||||
|
||||
// Performance Features
|
||||
export function isCachingEnabled(): boolean {
|
||||
return cachedConfig?.features.enableCaching ?? true;
|
||||
}
|
||||
|
||||
export function isBatchProcessingEnabled(): boolean {
|
||||
return cachedConfig?.features.enableBatchProcessing ?? true;
|
||||
}
|
||||
|
||||
export function isPerformanceMonitoringEnabled(): boolean {
|
||||
return cachedConfig?.features.enablePerformanceLogging ?? true;
|
||||
}
|
||||
|
||||
// AI Features
|
||||
export function isAdvancedRAGEnabled(): boolean {
|
||||
return cachedConfig?.features.enableAdvancedRAG ?? true;
|
||||
}
|
||||
|
||||
export function isReActReasoningEnabled(): boolean {
|
||||
return cachedConfig?.features.enableReActReasoning ?? true;
|
||||
}
|
||||
|
||||
export function isMultiLLMEnabled(): boolean {
|
||||
return cachedConfig?.features.enableMultiLLM ?? true;
|
||||
}
|
||||
|
||||
// Debug Features
|
||||
export function isDebugModeEnabled(): boolean {
|
||||
return cachedConfig?.features.enableDebugMode ?? false;
|
||||
}
|
||||
|
||||
export function isQueryLoggingEnabled(): boolean {
|
||||
return cachedConfig?.features.enableQueryLogging ?? false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get all feature flags as an object
|
||||
*/
|
||||
export function getFeatureFlags() {
|
||||
const config = cachedConfig;
|
||||
if (!config) {
|
||||
// Return defaults if config not loaded yet
|
||||
return {
|
||||
enableAdvancedRAG: true,
|
||||
enableReActReasoning: true,
|
||||
enableMultiLLM: true,
|
||||
enableWebWorkers: true,
|
||||
enableBatchProcessing: true,
|
||||
enableCaching: true,
|
||||
enableWorkerPool: true,
|
||||
enableParallelParsing: true,
|
||||
enableParallelProcessing: true,
|
||||
enableKuzuDB: true,
|
||||
enableKuzuDBPersistence: true,
|
||||
enableKuzuDBPerformanceMonitoring: true,
|
||||
enableKuzuCopy: false,
|
||||
enablePolymorphicNodes: false,
|
||||
enableDebugMode: false,
|
||||
enablePerformanceLogging: true,
|
||||
enableQueryLogging: false
|
||||
};
|
||||
}
|
||||
|
||||
return {
|
||||
...config.features,
|
||||
// Add KuzuDB flags from kuzu config section
|
||||
enableKuzuDB: config.kuzu.enabled,
|
||||
enableKuzuDBPersistence: config.kuzu.persistence,
|
||||
enableKuzuDBPerformanceMonitoring: config.features.enablePerformanceLogging
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize the feature system (loads config into cache)
|
||||
*/
|
||||
export async function initializeFeatures(): Promise<void> {
|
||||
try {
|
||||
cachedConfig = await getConfig();
|
||||
console.log('✅ Features initialized from GitNexus config');
|
||||
} catch (error) {
|
||||
console.warn('⚠️ Failed to initialize features, using defaults:', error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Legacy compatibility - matches old FeatureFlagManager API
|
||||
*/
|
||||
export const featureFlagManager = {
|
||||
getFlag: (flagName: string): boolean => {
|
||||
const flags = getFeatureFlags();
|
||||
return (flags as any)[flagName] ?? false;
|
||||
},
|
||||
|
||||
setFlag: (flagName: string, value: boolean): void => {
|
||||
console.warn('⚠️ setFlag is not supported in consolidated config system. Update gitnexus.config.ts instead.');
|
||||
},
|
||||
|
||||
isKuzuDBEnabled,
|
||||
isDebugModeEnabled
|
||||
};
|
||||
|
||||
// Auto-initialize when module is imported
|
||||
initializeFeatures().catch(console.error);
|
||||
+227
-272
@@ -1,284 +1,239 @@
|
||||
/**
|
||||
* Centralized Ignore Service
|
||||
*
|
||||
* Provides robust file and directory filtering based on centralized configuration.
|
||||
* Implements the same logic as the Python example with directory-component matching.
|
||||
*/
|
||||
|
||||
import { configLoader, type ValidatedGitNexusConfig } from './config-loader.ts';
|
||||
|
||||
export class IgnoreService {
|
||||
private static instance: IgnoreService;
|
||||
private config: ValidatedGitNexusConfig | null = null;
|
||||
private patternsSet: Set<string> = new Set();
|
||||
private suffixesSet: Set<string> = new Set();
|
||||
private extensionsSet: Set<string> = new Set();
|
||||
private customRegexes: RegExp[] = [];
|
||||
|
||||
private constructor() {}
|
||||
|
||||
public static getInstance(): IgnoreService {
|
||||
if (!IgnoreService.instance) {
|
||||
IgnoreService.instance = new IgnoreService();
|
||||
}
|
||||
return IgnoreService.instance;
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize the ignore service with configuration
|
||||
*/
|
||||
public async initialize(): Promise<void> {
|
||||
this.config = await configLoader.loadConfig();
|
||||
this.updatePatterns();
|
||||
}
|
||||
|
||||
/**
|
||||
* Update internal pattern sets from configuration
|
||||
*/
|
||||
private updatePatterns(): void {
|
||||
if (!this.config) return;
|
||||
|
||||
const { ignore } = this.config;
|
||||
|
||||
// Convert patterns to lowercase Set for O(1) lookup
|
||||
this.patternsSet = new Set(ignore.patterns.map(p => p.toLowerCase()));
|
||||
this.suffixesSet = new Set(ignore.suffixes.map(s => s.toLowerCase()));
|
||||
this.extensionsSet = new Set(ignore.fileExtensions.map(e => e.toLowerCase()));
|
||||
|
||||
// Compile custom regex patterns
|
||||
this.customRegexes = ignore.customPatterns
|
||||
.map(pattern => {
|
||||
try {
|
||||
return new RegExp(pattern, 'i');
|
||||
} catch (error) {
|
||||
console.warn(`Invalid regex pattern: ${pattern}`, error);
|
||||
return null;
|
||||
}
|
||||
})
|
||||
.filter((regex): regex is RegExp => regex !== null);
|
||||
|
||||
console.log(`🔧 IgnoreService initialized with ${this.patternsSet.size} patterns, ${this.suffixesSet.size} suffixes, ${this.extensionsSet.size} extensions`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a file path should be ignored
|
||||
*
|
||||
* Uses directory-component matching like the Python example:
|
||||
* - Splits path into components
|
||||
* - Checks each component against ignore patterns
|
||||
* - Prevents false positives (e.g., "my_node_modules_notes.txt")
|
||||
*/
|
||||
public shouldIgnorePath(filePath: string): boolean {
|
||||
if (!this.config?.ignore.enabled) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Normalize path separators and remove leading/trailing slashes
|
||||
const normalizedPath = filePath.replace(/\\/g, '/').replace(/^\/+|\/+$/g, '');
|
||||
const DEFAULT_IGNORE_LIST = new Set([
|
||||
// Version Control
|
||||
'.git',
|
||||
'.svn',
|
||||
'.hg',
|
||||
'.bzr',
|
||||
|
||||
// Split path into components for directory-part matching
|
||||
const pathComponents = normalizedPath.split('/').filter(Boolean);
|
||||
|
||||
// Check each component against ignore patterns (case-insensitive)
|
||||
for (const component of pathComponents) {
|
||||
const lowerComponent = component.toLowerCase();
|
||||
|
||||
if (this.patternsSet.has(lowerComponent)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check file suffixes
|
||||
const lowerPath = normalizedPath.toLowerCase();
|
||||
for (const suffix of this.suffixesSet) {
|
||||
if (lowerPath.endsWith(suffix)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check file extensions
|
||||
const fileName = pathComponents[pathComponents.length - 1] || '';
|
||||
const lowerFileName = fileName.toLowerCase();
|
||||
// IDEs & Editors
|
||||
'.idea',
|
||||
'.vscode',
|
||||
'.vs',
|
||||
'.eclipse',
|
||||
'.settings',
|
||||
'.DS_Store',
|
||||
'Thumbs.db',
|
||||
|
||||
// Dependencies
|
||||
'node_modules',
|
||||
'bower_components',
|
||||
'jspm_packages',
|
||||
'vendor', // PHP/Go
|
||||
// 'packages' removed - commonly used for monorepo source code (lerna, pnpm, yarn workspaces)
|
||||
'venv',
|
||||
'.venv',
|
||||
'env',
|
||||
'.env',
|
||||
'__pycache__',
|
||||
'.pytest_cache',
|
||||
'.mypy_cache',
|
||||
'site-packages',
|
||||
'.tox',
|
||||
'eggs',
|
||||
'.eggs',
|
||||
'lib64',
|
||||
'parts',
|
||||
'sdist',
|
||||
'wheels',
|
||||
|
||||
// Build Outputs
|
||||
'dist',
|
||||
'build',
|
||||
'out',
|
||||
'output',
|
||||
'bin',
|
||||
'obj',
|
||||
'target', // Java/Rust
|
||||
'.next',
|
||||
'.nuxt',
|
||||
'.output',
|
||||
'.vercel',
|
||||
'.netlify',
|
||||
'.serverless',
|
||||
'_build',
|
||||
'public/build',
|
||||
'.parcel-cache',
|
||||
'.turbo',
|
||||
'.svelte-kit',
|
||||
|
||||
// Test & Coverage
|
||||
'coverage',
|
||||
'.nyc_output',
|
||||
'htmlcov',
|
||||
'.coverage',
|
||||
'__tests__', // Often just test files
|
||||
'__mocks__',
|
||||
'.jest',
|
||||
|
||||
for (const ext of this.extensionsSet) {
|
||||
if (lowerFileName.endsWith(ext)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check custom regex patterns
|
||||
for (const regex of this.customRegexes) {
|
||||
if (regex.test(normalizedPath)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Filter an array of paths, removing ignored ones
|
||||
*
|
||||
* Optimized for batch processing with early returns
|
||||
*/
|
||||
public filterPaths(paths: string[]): string[] {
|
||||
if (!this.config?.ignore.enabled) {
|
||||
return paths;
|
||||
}
|
||||
|
||||
const startTime = performance.now();
|
||||
const filtered = paths.filter(path => !this.shouldIgnorePath(path));
|
||||
const endTime = performance.now();
|
||||
|
||||
const filteredCount = paths.length - filtered.length;
|
||||
if (filteredCount > 0) {
|
||||
console.log(`🚫 IgnoreService filtered ${filteredCount}/${paths.length} paths in ${(endTime - startTime).toFixed(2)}ms`);
|
||||
}
|
||||
|
||||
return filtered;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a directory should be ignored
|
||||
*
|
||||
* Specialized method for directory filtering during traversal
|
||||
*/
|
||||
public shouldIgnoreDirectory(dirPath: string): boolean {
|
||||
if (!this.config?.ignore.enabled) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// For directories, we only check the directory name itself
|
||||
const dirName = dirPath.split('/').pop()?.toLowerCase() || '';
|
||||
return this.patternsSet.has(dirName);
|
||||
}
|
||||
|
||||
/**
|
||||
* Get current ignore statistics
|
||||
*/
|
||||
public getStats(): {
|
||||
enabled: boolean;
|
||||
patterns: number;
|
||||
suffixes: number;
|
||||
extensions: number;
|
||||
customPatterns: number;
|
||||
} {
|
||||
return {
|
||||
enabled: this.config?.ignore.enabled ?? false,
|
||||
patterns: this.patternsSet.size,
|
||||
suffixes: this.suffixesSet.size,
|
||||
extensions: this.extensionsSet.size,
|
||||
customPatterns: this.customRegexes.length
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Update ignore patterns at runtime
|
||||
*/
|
||||
public updateIgnorePatterns(updates: Partial<ValidatedGitNexusConfig['ignore']>): void {
|
||||
if (!this.config) return;
|
||||
|
||||
// Update configuration
|
||||
this.config.ignore = { ...this.config.ignore, ...updates };
|
||||
// Logs & Temp
|
||||
'logs',
|
||||
'log',
|
||||
'tmp',
|
||||
'temp',
|
||||
'cache',
|
||||
'.cache',
|
||||
'.tmp',
|
||||
'.temp',
|
||||
|
||||
// Rebuild pattern sets
|
||||
this.updatePatterns();
|
||||
// Generated/Compiled
|
||||
'.generated',
|
||||
'generated',
|
||||
'auto-generated',
|
||||
'.terraform',
|
||||
'.serverless',
|
||||
|
||||
console.log('🔄 IgnoreService patterns updated');
|
||||
// Documentation (optional - might want to keep)
|
||||
// 'docs',
|
||||
// 'documentation',
|
||||
|
||||
// Misc
|
||||
'.husky',
|
||||
'.github', // GitHub config, not code
|
||||
'.circleci',
|
||||
'.gitlab',
|
||||
'fixtures', // Test fixtures
|
||||
'snapshots', // Jest snapshots
|
||||
'__snapshots__',
|
||||
]);
|
||||
|
||||
const IGNORED_EXTENSIONS = new Set([
|
||||
// Images
|
||||
'.png', '.jpg', '.jpeg', '.gif', '.svg', '.ico', '.webp', '.bmp', '.tiff', '.tif',
|
||||
'.psd', '.ai', '.sketch', '.fig', '.xd',
|
||||
|
||||
// Archives
|
||||
'.zip', '.tar', '.gz', '.rar', '.7z', '.bz2', '.xz', '.tgz',
|
||||
|
||||
// Binary/Compiled
|
||||
'.exe', '.dll', '.so', '.dylib', '.a', '.lib', '.o', '.obj',
|
||||
'.class', '.jar', '.war', '.ear',
|
||||
'.pyc', '.pyo', '.pyd',
|
||||
'.beam', // Erlang
|
||||
'.wasm', // WebAssembly - important!
|
||||
'.node', // Native Node addons
|
||||
|
||||
// Documents
|
||||
'.pdf', '.doc', '.docx', '.xls', '.xlsx', '.ppt', '.pptx',
|
||||
'.odt', '.ods', '.odp',
|
||||
|
||||
// Media
|
||||
'.mp4', '.mp3', '.wav', '.mov', '.avi', '.mkv', '.flv', '.wmv',
|
||||
'.ogg', '.webm', '.flac', '.aac', '.m4a',
|
||||
|
||||
// Fonts
|
||||
'.woff', '.woff2', '.ttf', '.eot', '.otf',
|
||||
|
||||
// Databases
|
||||
'.db', '.sqlite', '.sqlite3', '.mdb', '.accdb',
|
||||
|
||||
// Minified/Bundled files
|
||||
'.min.js', '.min.css', '.bundle.js', '.chunk.js',
|
||||
|
||||
// Source maps (debug files, not source)
|
||||
'.map',
|
||||
|
||||
// Lock files (handled separately, but also here)
|
||||
'.lock',
|
||||
|
||||
// Certificates & Keys (security - don't index!)
|
||||
'.pem', '.key', '.crt', '.cer', '.p12', '.pfx',
|
||||
|
||||
// Data files (often large/binary)
|
||||
'.csv', '.tsv', '.parquet', '.avro', '.feather',
|
||||
'.npy', '.npz', '.pkl', '.pickle', '.h5', '.hdf5',
|
||||
|
||||
// Misc binary
|
||||
'.bin', '.dat', '.data', '.raw',
|
||||
'.iso', '.img', '.dmg',
|
||||
]);
|
||||
|
||||
// Files to ignore by exact name
|
||||
const IGNORED_FILES = new Set([
|
||||
'package-lock.json',
|
||||
'yarn.lock',
|
||||
'pnpm-lock.yaml',
|
||||
'composer.lock',
|
||||
'Gemfile.lock',
|
||||
'poetry.lock',
|
||||
'Cargo.lock',
|
||||
'go.sum',
|
||||
'.gitignore',
|
||||
'.gitattributes',
|
||||
'.npmrc',
|
||||
'.yarnrc',
|
||||
'.editorconfig',
|
||||
'.prettierrc',
|
||||
'.prettierignore',
|
||||
'.eslintignore',
|
||||
'.dockerignore',
|
||||
'Thumbs.db',
|
||||
'.DS_Store',
|
||||
'LICENSE',
|
||||
'LICENSE.md',
|
||||
'LICENSE.txt',
|
||||
'CHANGELOG.md',
|
||||
'CHANGELOG',
|
||||
'CONTRIBUTING.md',
|
||||
'CODE_OF_CONDUCT.md',
|
||||
'SECURITY.md',
|
||||
'.env',
|
||||
'.env.local',
|
||||
'.env.development',
|
||||
'.env.production',
|
||||
'.env.test',
|
||||
'.env.example',
|
||||
]);
|
||||
|
||||
|
||||
|
||||
export const shouldIgnorePath = (filePath: string): boolean => {
|
||||
const normalizedPath = filePath.replace(/\\/g, '/');
|
||||
const parts = normalizedPath.split('/');
|
||||
const fileName = parts[parts.length - 1];
|
||||
const fileNameLower = fileName.toLowerCase();
|
||||
|
||||
// Check if any path segment is in ignore list
|
||||
for (const part of parts) {
|
||||
if (DEFAULT_IGNORE_LIST.has(part)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Add custom patterns at runtime
|
||||
*/
|
||||
public addCustomPatterns(patterns: string[], type: 'patterns' | 'suffixes' | 'extensions' | 'regex' = 'patterns'): void {
|
||||
if (!this.config) return;
|
||||
|
||||
switch (type) {
|
||||
case 'patterns':
|
||||
this.config.ignore.patterns.push(...patterns);
|
||||
break;
|
||||
case 'suffixes':
|
||||
this.config.ignore.suffixes.push(...patterns);
|
||||
break;
|
||||
case 'extensions':
|
||||
this.config.ignore.fileExtensions.push(...patterns);
|
||||
break;
|
||||
case 'regex':
|
||||
this.config.ignore.customPatterns.push(...patterns);
|
||||
break;
|
||||
}
|
||||
|
||||
this.updatePatterns();
|
||||
console.log(`➕ Added ${patterns.length} ${type} to IgnoreService`);
|
||||
// Check exact filename matches
|
||||
if (IGNORED_FILES.has(fileName) || IGNORED_FILES.has(fileNameLower)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test a path against ignore patterns (for debugging)
|
||||
*/
|
||||
public testPath(filePath: string): {
|
||||
ignored: boolean;
|
||||
reason?: string;
|
||||
matchedPattern?: string;
|
||||
} {
|
||||
if (!this.config?.ignore.enabled) {
|
||||
return { ignored: false };
|
||||
// Check extension
|
||||
const lastDotIndex = fileNameLower.lastIndexOf('.');
|
||||
if (lastDotIndex !== -1) {
|
||||
const ext = fileNameLower.substring(lastDotIndex);
|
||||
if (IGNORED_EXTENSIONS.has(ext)) return true;
|
||||
|
||||
// Handle compound extensions like .min.js, .bundle.js
|
||||
const secondLastDot = fileNameLower.lastIndexOf('.', lastDotIndex - 1);
|
||||
if (secondLastDot !== -1) {
|
||||
const compoundExt = fileNameLower.substring(secondLastDot);
|
||||
if (IGNORED_EXTENSIONS.has(compoundExt)) return true;
|
||||
}
|
||||
|
||||
const normalizedPath = filePath.replace(/\\/g, '/').replace(/^\/+|\/+$/g, '');
|
||||
const pathComponents = normalizedPath.split('/').filter(Boolean);
|
||||
|
||||
// Check directory components
|
||||
for (const component of pathComponents) {
|
||||
const lowerComponent = component.toLowerCase();
|
||||
if (this.patternsSet.has(lowerComponent)) {
|
||||
return {
|
||||
ignored: true,
|
||||
reason: 'directory pattern',
|
||||
matchedPattern: component
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Check suffixes
|
||||
const lowerPath = normalizedPath.toLowerCase();
|
||||
for (const suffix of this.suffixesSet) {
|
||||
if (lowerPath.endsWith(suffix)) {
|
||||
return {
|
||||
ignored: true,
|
||||
reason: 'file suffix',
|
||||
matchedPattern: suffix
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Check extensions
|
||||
const fileName = pathComponents[pathComponents.length - 1] || '';
|
||||
const lowerFileName = fileName.toLowerCase();
|
||||
for (const ext of this.extensionsSet) {
|
||||
if (lowerFileName.endsWith(ext)) {
|
||||
return {
|
||||
ignored: true,
|
||||
reason: 'file extension',
|
||||
matchedPattern: ext
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
// Check regex patterns
|
||||
for (let i = 0; i < this.customRegexes.length; i++) {
|
||||
const regex = this.customRegexes[i];
|
||||
if (regex.test(normalizedPath)) {
|
||||
return {
|
||||
ignored: true,
|
||||
reason: 'custom regex',
|
||||
matchedPattern: this.config.ignore.customPatterns[i]
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
return { ignored: false };
|
||||
}
|
||||
|
||||
// Ignore hidden files (starting with .)
|
||||
if (fileName.startsWith('.') && fileName !== '.') {
|
||||
// But allow some important config files
|
||||
const allowedDotFiles = ['.env', '.gitignore']; // Already in IGNORED_FILES, so this is redundant
|
||||
// Actually, let's NOT ignore all dot files - many are important configs
|
||||
// Just rely on the explicit lists above
|
||||
}
|
||||
|
||||
// Ignore files that look like generated/bundled code
|
||||
if (fileNameLower.includes('.bundle.') ||
|
||||
fileNameLower.includes('.chunk.') ||
|
||||
fileNameLower.includes('.generated.') ||
|
||||
fileNameLower.endsWith('.d.ts')) { // TypeScript declaration files
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
// Export singleton instance
|
||||
export const ignoreService = IgnoreService.getInstance();
|
||||
|
||||
@@ -1,328 +0,0 @@
|
||||
/**
|
||||
* Language-specific configuration and built-ins
|
||||
* Centralized configuration to eliminate hardcoded values
|
||||
*/
|
||||
|
||||
export interface LanguageConfig {
|
||||
name: string;
|
||||
extensions: string[];
|
||||
builtinFunctions: Set<string>;
|
||||
builtinTypes: Set<string>;
|
||||
commentPatterns: {
|
||||
singleLine: string[];
|
||||
multiLineStart: string[];
|
||||
multiLineEnd: string[];
|
||||
};
|
||||
importPatterns: {
|
||||
import: RegExp[];
|
||||
fromImport: RegExp[];
|
||||
require: RegExp[];
|
||||
};
|
||||
|
||||
}
|
||||
|
||||
// Python built-in functions and types
|
||||
const PYTHON_BUILTINS = new Set([
|
||||
// Core built-ins
|
||||
'int', 'str', 'float', 'bool', 'list', 'dict', 'set', 'tuple',
|
||||
'len', 'range', 'enumerate', 'zip', 'map', 'filter', 'sorted',
|
||||
'sum', 'min', 'max', 'abs', 'round', 'all', 'any', 'hasattr',
|
||||
'getattr', 'setattr', 'isinstance', 'issubclass', 'type',
|
||||
'print', 'input', 'open', 'format', 'join', 'split', 'strip',
|
||||
'replace', 'upper', 'lower', 'append', 'extend', 'insert',
|
||||
'remove', 'pop', 'clear', 'copy', 'update', 'keys', 'values',
|
||||
'items', 'get', 'add', 'discard', 'union', 'intersection',
|
||||
'difference', 'locals', 'globals', 'vars', 'dir', 'help', 'id', 'hash',
|
||||
'ord', 'chr', 'bin', 'oct', 'hex', 'divmod', 'pow', 'exec',
|
||||
'eval', 'compile', 'next', 'iter', 'reversed', 'slice',
|
||||
|
||||
// String methods
|
||||
'endswith', 'startswith', 'find', 'rfind', 'index', 'rindex',
|
||||
'count', 'encode', 'decode', 'capitalize', 'title', 'swapcase',
|
||||
'center', 'ljust', 'rjust', 'zfill', 'expandtabs', 'splitlines',
|
||||
'partition', 'rpartition', 'translate', 'maketrans', 'casefold',
|
||||
'isalnum', 'isalpha', 'isascii', 'isdecimal', 'isdigit', 'isidentifier',
|
||||
'islower', 'isnumeric', 'isprintable', 'isspace', 'istitle', 'isupper',
|
||||
'lstrip', 'rstrip', 'removeprefix', 'removesuffix',
|
||||
|
||||
// List/sequence methods
|
||||
'sort', 'reverse', 'count', 'index',
|
||||
|
||||
// Dictionary methods
|
||||
'setdefault', 'popitem', 'fromkeys',
|
||||
|
||||
// Set methods
|
||||
'difference_update', 'intersection_update', 'symmetric_difference',
|
||||
'symmetric_difference_update', 'isdisjoint', 'issubset', 'issuperset',
|
||||
|
||||
// Common exceptions
|
||||
'ValueError', 'TypeError', 'KeyError', 'IndexError', 'AttributeError',
|
||||
'ImportError', 'ModuleNotFoundError', 'FileNotFoundError',
|
||||
'ConnectionError', 'HTTPException', 'RuntimeError', 'OSError',
|
||||
'Exception', 'BaseException', 'StopIteration', 'GeneratorExit'
|
||||
]);
|
||||
|
||||
const PYTHON_LIBRARY_FUNCTIONS = new Set([
|
||||
// Date/time methods
|
||||
'now', 'today', 'fromisoformat', 'isoformat', 'astimezone',
|
||||
'strftime', 'strptime', 'timestamp', 'weekday', 'isoweekday',
|
||||
'date', 'time', 'timetz', 'utctimetuple', 'timetuple',
|
||||
|
||||
// Random
|
||||
'random', 'choice', 'randint', 'shuffle',
|
||||
|
||||
// Logging methods
|
||||
'debug', 'info', 'warning', 'error', 'critical', 'exception',
|
||||
'getLogger', 'basicConfig', 'StreamHandler',
|
||||
|
||||
// Environment
|
||||
'load_dotenv', 'getenv', 'dirname', 'abspath', 'join', 'exists', 'run',
|
||||
|
||||
// Database/ORM methods
|
||||
'find', 'find_one', 'update_one', 'insert_one', 'delete_one',
|
||||
'aggregate', 'bulk_write', 'to_list', 'sort', 'limit', 'close',
|
||||
'ObjectId', 'UpdateOne', 'AsyncIOMotorClient', 'command',
|
||||
|
||||
// Pydantic/FastAPI
|
||||
'Field', 'validator', 'field_validator', 'model_dump', 'model_dump_json',
|
||||
'FastAPI', 'HTTPException', 'add_middleware', 'include_router',
|
||||
|
||||
// Threading/async
|
||||
'Lock', 'RLock', 'Semaphore', 'Event', 'Condition', 'Barrier',
|
||||
'sleep', 'gather', 'create_task', 'run_until_complete',
|
||||
|
||||
// Collections
|
||||
'defaultdict', 'Counter', 'OrderedDict', 'deque', 'namedtuple',
|
||||
|
||||
// Math/statistics
|
||||
'mean', 'median', 'mode', 'stdev', 'variance', 'sqrt', 'pow',
|
||||
'sin', 'cos', 'tan', 'log', 'exp', 'ceil', 'floor',
|
||||
|
||||
// UUID
|
||||
'uuid4', 'uuid1', 'uuid3', 'uuid5',
|
||||
|
||||
// URL/HTTP
|
||||
'quote', 'unquote', 'quote_plus', 'unquote_plus', 'urlencode',
|
||||
|
||||
// JSON
|
||||
'loads', 'dumps', 'load', 'dump',
|
||||
|
||||
// Regex
|
||||
'match', 'search', 'findall', 'finditer', 'sub', 'subn', 'compile',
|
||||
|
||||
// AI/ML libraries
|
||||
'AsyncAzureOpenAI', 'AzureOpenAI', 'OpenAI', 'wrap_openai', 'create'
|
||||
]);
|
||||
|
||||
// JavaScript built-in functions and types
|
||||
const JAVASCRIPT_BUILTINS = new Set([
|
||||
// Global functions
|
||||
'parseInt', 'parseFloat', 'isNaN', 'isFinite', 'decodeURI', 'decodeURIComponent',
|
||||
'encodeURI', 'encodeURIComponent', 'eval', 'setTimeout', 'setInterval',
|
||||
'clearTimeout', 'clearInterval', 'console', 'alert', 'confirm', 'prompt',
|
||||
|
||||
// Object methods
|
||||
'toString', 'valueOf', 'hasOwnProperty', 'isPrototypeOf', 'propertyIsEnumerable',
|
||||
|
||||
// Array methods
|
||||
'push', 'pop', 'shift', 'unshift', 'slice', 'splice', 'concat', 'join',
|
||||
'reverse', 'sort', 'indexOf', 'lastIndexOf', 'forEach', 'map', 'filter',
|
||||
'reduce', 'reduceRight', 'every', 'some', 'find', 'findIndex', 'includes',
|
||||
|
||||
// String methods
|
||||
'charAt', 'charCodeAt', 'concat', 'indexOf', 'lastIndexOf', 'localeCompare',
|
||||
'match', 'replace', 'search', 'slice', 'split', 'substring', 'toLowerCase',
|
||||
'toUpperCase', 'trim', 'padStart', 'padEnd',
|
||||
|
||||
// Math
|
||||
'abs', 'ceil', 'floor', 'round', 'max', 'min', 'pow', 'sqrt', 'random',
|
||||
|
||||
// Date
|
||||
'getTime', 'getFullYear', 'getMonth', 'getDate', 'getDay', 'getHours',
|
||||
'getMinutes', 'getSeconds', 'getMilliseconds', 'toISOString', 'toDateString',
|
||||
|
||||
// Promise/async
|
||||
'then', 'catch', 'finally', 'resolve', 'reject', 'all', 'race',
|
||||
|
||||
// JSON
|
||||
'parse', 'stringify',
|
||||
|
||||
// Types
|
||||
'Object', 'Array', 'String', 'Number', 'Boolean', 'Function', 'Date',
|
||||
'RegExp', 'Error', 'Promise', 'Map', 'Set', 'WeakMap', 'WeakSet'
|
||||
]);
|
||||
|
||||
// TypeScript built-ins (extends JavaScript)
|
||||
const TYPESCRIPT_BUILTINS = new Set([
|
||||
...JAVASCRIPT_BUILTINS,
|
||||
// TypeScript specific
|
||||
'Partial', 'Required', 'Readonly', 'Pick', 'Omit', 'Exclude', 'Extract',
|
||||
'Record', 'Parameters', 'ConstructorParameters', 'ReturnType',
|
||||
'InstanceType', 'ThisParameterType', 'OmitThisParameter', 'ThisType'
|
||||
]);
|
||||
|
||||
// Note: Ignore patterns have been moved to the centralized IgnoreService
|
||||
// See src/config/ignore-service.ts and gitnexus.config.ts
|
||||
|
||||
// Language configurations
|
||||
export const LANGUAGE_CONFIGS: Record<string, LanguageConfig> = {
|
||||
python: {
|
||||
name: 'Python',
|
||||
extensions: ['.py', '.pyx', '.pyi'],
|
||||
builtinFunctions: new Set([...PYTHON_BUILTINS, ...PYTHON_LIBRARY_FUNCTIONS]),
|
||||
builtinTypes: new Set(['int', 'str', 'float', 'bool', 'list', 'dict', 'set', 'tuple']),
|
||||
commentPatterns: {
|
||||
singleLine: ['#'],
|
||||
multiLineStart: ['"""', "'''"],
|
||||
multiLineEnd: ['"""', "'''"]
|
||||
},
|
||||
importPatterns: {
|
||||
import: [/^import\s+(.+)$/],
|
||||
fromImport: [/^from\s+(.+)\s+import\s+(.+)$/],
|
||||
require: []
|
||||
}
|
||||
},
|
||||
|
||||
javascript: {
|
||||
name: 'JavaScript',
|
||||
extensions: ['.js', '.mjs', '.cjs', '.jsx'],
|
||||
builtinFunctions: JAVASCRIPT_BUILTINS,
|
||||
builtinTypes: new Set(['Object', 'Array', 'String', 'Number', 'Boolean', 'Function']),
|
||||
commentPatterns: {
|
||||
singleLine: ['//'],
|
||||
multiLineStart: ['/*'],
|
||||
multiLineEnd: ['*/']
|
||||
},
|
||||
importPatterns: {
|
||||
import: [/^import\s+.*\s+from\s+['"'](.+)['"]$/],
|
||||
fromImport: [],
|
||||
require: [/require\s*\(\s*['"'](.+)['"]\s*\)/]
|
||||
}
|
||||
},
|
||||
|
||||
typescript: {
|
||||
name: 'TypeScript',
|
||||
extensions: ['.ts', '.tsx'],
|
||||
builtinFunctions: TYPESCRIPT_BUILTINS,
|
||||
builtinTypes: new Set(['Object', 'Array', 'String', 'Number', 'Boolean', 'Function']),
|
||||
commentPatterns: {
|
||||
singleLine: ['//'],
|
||||
multiLineStart: ['/*'],
|
||||
multiLineEnd: ['*/']
|
||||
},
|
||||
importPatterns: {
|
||||
import: [/^import\s+.*\s+from\s+['"'](.+)['"]$/],
|
||||
fromImport: [],
|
||||
require: [/require\s*\(\s*['"'](.+)['"]\s*\)/]
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Language abstraction interface
|
||||
export interface ParsedDefinition {
|
||||
name: string;
|
||||
type: 'function' | 'class' | 'method' | 'interface' | 'enum' | 'decorator' | 'variable';
|
||||
startLine: number;
|
||||
endLine: number;
|
||||
parentClass?: string;
|
||||
decorators?: string[];
|
||||
baseClasses?: string[];
|
||||
parameters?: string[];
|
||||
returnType?: string;
|
||||
}
|
||||
|
||||
export interface ImportInfo {
|
||||
localName: string;
|
||||
importedFrom: string;
|
||||
exportedName: string;
|
||||
importType: 'default' | 'named' | 'namespace' | 'dynamic';
|
||||
}
|
||||
|
||||
export interface LanguageProcessor {
|
||||
name: string;
|
||||
extensions: string[];
|
||||
isBuiltinFunction(name: string): boolean;
|
||||
isBuiltinType(name: string): boolean;
|
||||
parseDefinitions(content: string, filePath: string): ParsedDefinition[];
|
||||
extractImports(content: string): ImportInfo[];
|
||||
}
|
||||
|
||||
// Base language processor implementation
|
||||
export abstract class BaseLanguageProcessor implements LanguageProcessor {
|
||||
protected config: LanguageConfig;
|
||||
|
||||
constructor(config: LanguageConfig) {
|
||||
this.config = config;
|
||||
}
|
||||
|
||||
get name(): string {
|
||||
return this.config.name;
|
||||
}
|
||||
|
||||
get extensions(): string[] {
|
||||
return this.config.extensions;
|
||||
}
|
||||
|
||||
isBuiltinFunction(name: string): boolean {
|
||||
return this.config.builtinFunctions.has(name);
|
||||
}
|
||||
|
||||
isBuiltinType(name: string): boolean {
|
||||
return this.config.builtinTypes.has(name);
|
||||
}
|
||||
|
||||
abstract parseDefinitions(content: string, filePath: string): ParsedDefinition[];
|
||||
abstract extractImports(content: string): ImportInfo[];
|
||||
}
|
||||
|
||||
// Language processor factory
|
||||
export class LanguageProcessorFactory {
|
||||
private static processors: Map<string, () => LanguageProcessor> = new Map();
|
||||
|
||||
static register(language: string, factory: () => LanguageProcessor): void {
|
||||
this.processors.set(language, factory);
|
||||
}
|
||||
|
||||
static create(language: string): LanguageProcessor | null {
|
||||
const factory = this.processors.get(language);
|
||||
return factory ? factory() : null;
|
||||
}
|
||||
|
||||
static getConfig(language: string): LanguageConfig | null {
|
||||
return LANGUAGE_CONFIGS[language] || null;
|
||||
}
|
||||
|
||||
static getSupportedLanguages(): string[] {
|
||||
return Object.keys(LANGUAGE_CONFIGS);
|
||||
}
|
||||
}
|
||||
|
||||
// Utility functions for language detection
|
||||
export const languageDetection = {
|
||||
detectFromExtension(filePath: string): string | null {
|
||||
const extension = filePath.substring(filePath.lastIndexOf('.')).toLowerCase();
|
||||
|
||||
for (const [lang, config] of Object.entries(LANGUAGE_CONFIGS)) {
|
||||
if (config.extensions.includes(extension)) {
|
||||
return lang;
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
},
|
||||
|
||||
detectFromContent(content: string): string | null {
|
||||
// Simple content-based detection
|
||||
if (content.includes('def ') && content.includes('import ')) {
|
||||
return 'python';
|
||||
}
|
||||
if (content.includes('function ') || content.includes('const ') || content.includes('let ')) {
|
||||
if (content.includes('interface ') || content.includes(': string')) {
|
||||
return 'typescript';
|
||||
}
|
||||
return 'javascript';
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
};
|
||||
@@ -0,0 +1,14 @@
|
||||
export enum SupportedLanguages {
|
||||
JavaScript = 'javascript',
|
||||
TypeScript = 'typescript',
|
||||
Python = 'python',
|
||||
// Java = 'java',
|
||||
// C = 'c',
|
||||
// CPlusPlus = 'cpp',
|
||||
// CSharp = 'csharp',
|
||||
// Go = 'go',
|
||||
// Rust = 'rust',
|
||||
// PHP = 'php',
|
||||
// Ruby = 'ruby',
|
||||
// Swift = 'swift',
|
||||
}
|
||||
@@ -0,0 +1,302 @@
|
||||
/**
|
||||
* Embedder Module
|
||||
*
|
||||
* Singleton factory for transformers.js embedding pipeline.
|
||||
* Handles model loading, caching, and both single and batch embedding operations.
|
||||
*
|
||||
* Uses snowflake-arctic-embed-xs by default (22M params, 384 dims, ~90MB)
|
||||
*/
|
||||
|
||||
import { pipeline, env, type FeatureExtractionPipeline } from '@huggingface/transformers';
|
||||
import { DEFAULT_EMBEDDING_CONFIG, type EmbeddingConfig, type ModelProgress } from './types';
|
||||
|
||||
// Module-level state for singleton pattern
|
||||
let embedderInstance: FeatureExtractionPipeline | null = null;
|
||||
let isInitializing = false;
|
||||
let initPromise: Promise<FeatureExtractionPipeline> | null = null;
|
||||
let currentDevice: 'webgpu' | 'wasm' | null = null;
|
||||
|
||||
/**
|
||||
* Progress callback type for model loading
|
||||
*/
|
||||
export type ModelProgressCallback = (progress: ModelProgress) => void;
|
||||
|
||||
/**
|
||||
* Custom error thrown when WebGPU is not available
|
||||
* Allows UI to prompt user for fallback choice
|
||||
*/
|
||||
export class WebGPUNotAvailableError extends Error {
|
||||
constructor(originalError?: Error) {
|
||||
super('WebGPU not available in this browser');
|
||||
this.name = 'WebGPUNotAvailableError';
|
||||
this.cause = originalError;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if WebGPU is available in this browser
|
||||
* Quick check without loading the model
|
||||
*/
|
||||
export const checkWebGPUAvailability = async (): Promise<boolean> => {
|
||||
try {
|
||||
// Cast to any to avoid WebGPU types not being available in all TS configs
|
||||
const nav = navigator as any;
|
||||
if (!nav.gpu) {
|
||||
return false;
|
||||
}
|
||||
const adapter = await nav.gpu.requestAdapter();
|
||||
if (!adapter) {
|
||||
return false;
|
||||
}
|
||||
// Try to get a device - this is where it usually fails
|
||||
const device = await adapter.requestDevice();
|
||||
device.destroy(); // Clean up
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Get the current device being used for inference
|
||||
*/
|
||||
export const getCurrentDevice = (): 'webgpu' | 'wasm' | null => currentDevice;
|
||||
|
||||
/**
|
||||
* Initialize the embedding model
|
||||
* Uses singleton pattern - only loads once, subsequent calls return cached instance
|
||||
*
|
||||
* @param onProgress - Optional callback for model download progress
|
||||
* @param config - Optional configuration override
|
||||
* @param forceDevice - Force a specific device (bypasses WebGPU check)
|
||||
* @returns Promise resolving to the embedder pipeline
|
||||
* @throws WebGPUNotAvailableError if WebGPU is requested but unavailable
|
||||
*/
|
||||
export const initEmbedder = async (
|
||||
onProgress?: ModelProgressCallback,
|
||||
config: Partial<EmbeddingConfig> = {},
|
||||
forceDevice?: 'webgpu' | 'wasm'
|
||||
): Promise<FeatureExtractionPipeline> => {
|
||||
// Return existing instance if available
|
||||
if (embedderInstance) {
|
||||
return embedderInstance;
|
||||
}
|
||||
|
||||
// If already initializing, wait for that promise
|
||||
if (isInitializing && initPromise) {
|
||||
return initPromise;
|
||||
}
|
||||
|
||||
isInitializing = true;
|
||||
|
||||
const finalConfig = { ...DEFAULT_EMBEDDING_CONFIG, ...config };
|
||||
const requestedDevice = forceDevice || finalConfig.device;
|
||||
|
||||
initPromise = (async () => {
|
||||
try {
|
||||
// Configure transformers.js environment
|
||||
env.allowLocalModels = false;
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`🧠 Loading embedding model: ${finalConfig.modelId}`);
|
||||
}
|
||||
|
||||
const progressCallback = onProgress ? (data: any) => {
|
||||
const progress: ModelProgress = {
|
||||
status: data.status || 'progress',
|
||||
file: data.file,
|
||||
progress: data.progress,
|
||||
loaded: data.loaded,
|
||||
total: data.total,
|
||||
};
|
||||
onProgress(progress);
|
||||
} : undefined;
|
||||
|
||||
// If WebGPU is requested (default), check availability first
|
||||
if (requestedDevice === 'webgpu') {
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('🔧 Checking WebGPU availability...');
|
||||
}
|
||||
|
||||
const webgpuAvailable = await checkWebGPUAvailability();
|
||||
|
||||
if (!webgpuAvailable) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn('⚠️ WebGPU not available');
|
||||
}
|
||||
isInitializing = false;
|
||||
initPromise = null;
|
||||
throw new WebGPUNotAvailableError();
|
||||
}
|
||||
|
||||
// Try WebGPU
|
||||
try {
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('🔧 Initializing WebGPU backend...');
|
||||
}
|
||||
|
||||
// Type assertion needed due to complex union types in transformers.js
|
||||
embedderInstance = await (pipeline as any)(
|
||||
'feature-extraction',
|
||||
finalConfig.modelId,
|
||||
{
|
||||
device: 'webgpu',
|
||||
dtype: 'fp32',
|
||||
progress_callback: progressCallback,
|
||||
}
|
||||
);
|
||||
currentDevice = 'webgpu';
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('✅ Using WebGPU backend');
|
||||
}
|
||||
} catch (err) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn('⚠️ WebGPU initialization failed:', err);
|
||||
}
|
||||
isInitializing = false;
|
||||
initPromise = null;
|
||||
embedderInstance = null;
|
||||
throw new WebGPUNotAvailableError(err as Error);
|
||||
}
|
||||
} else {
|
||||
// WASM mode requested (user chose fallback)
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('🔧 Initializing WASM backend (this will be slower)...');
|
||||
}
|
||||
|
||||
// Type assertion needed due to complex union types in transformers.js
|
||||
embedderInstance = await (pipeline as any)(
|
||||
'feature-extraction',
|
||||
finalConfig.modelId,
|
||||
{
|
||||
device: 'wasm', // WASM-based CPU execution
|
||||
dtype: 'fp32',
|
||||
progress_callback: progressCallback,
|
||||
}
|
||||
);
|
||||
currentDevice = 'wasm';
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('✅ Using WASM backend');
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('✅ Embedding model loaded successfully');
|
||||
}
|
||||
|
||||
return embedderInstance!;
|
||||
} catch (error) {
|
||||
// Re-throw WebGPUNotAvailableError as-is
|
||||
if (error instanceof WebGPUNotAvailableError) {
|
||||
throw error;
|
||||
}
|
||||
isInitializing = false;
|
||||
initPromise = null;
|
||||
embedderInstance = null;
|
||||
throw error;
|
||||
} finally {
|
||||
isInitializing = false;
|
||||
}
|
||||
})();
|
||||
|
||||
return initPromise;
|
||||
};
|
||||
|
||||
/**
|
||||
* Check if the embedder is initialized and ready
|
||||
*/
|
||||
export const isEmbedderReady = (): boolean => {
|
||||
return embedderInstance !== null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Get the embedder instance (throws if not initialized)
|
||||
*/
|
||||
export const getEmbedder = (): FeatureExtractionPipeline => {
|
||||
if (!embedderInstance) {
|
||||
throw new Error('Embedder not initialized. Call initEmbedder() first.');
|
||||
}
|
||||
return embedderInstance;
|
||||
};
|
||||
|
||||
/**
|
||||
* Embed a single text string
|
||||
*
|
||||
* @param text - Text to embed
|
||||
* @returns Float32Array of embedding vector (384 dimensions)
|
||||
*/
|
||||
export const embedText = async (text: string): Promise<Float32Array> => {
|
||||
const embedder = getEmbedder();
|
||||
|
||||
const result = await embedder(text, {
|
||||
pooling: 'mean',
|
||||
normalize: true,
|
||||
});
|
||||
|
||||
// Result is a Tensor, convert to Float32Array
|
||||
return new Float32Array(result.data as ArrayLike<number>);
|
||||
};
|
||||
|
||||
/**
|
||||
* Embed multiple texts in a single batch
|
||||
* More efficient than calling embedText multiple times
|
||||
*
|
||||
* @param texts - Array of texts to embed
|
||||
* @returns Array of Float32Array embedding vectors
|
||||
*/
|
||||
export const embedBatch = async (texts: string[]): Promise<Float32Array[]> => {
|
||||
if (texts.length === 0) {
|
||||
return [];
|
||||
}
|
||||
|
||||
const embedder = getEmbedder();
|
||||
|
||||
// Process batch
|
||||
const result = await embedder(texts, {
|
||||
pooling: 'mean',
|
||||
normalize: true,
|
||||
});
|
||||
|
||||
// Result shape is [batch_size, dimensions]
|
||||
// Need to split into individual vectors
|
||||
const data = result.data as ArrayLike<number>;
|
||||
const dimensions = DEFAULT_EMBEDDING_CONFIG.dimensions;
|
||||
const embeddings: Float32Array[] = [];
|
||||
|
||||
for (let i = 0; i < texts.length; i++) {
|
||||
const start = i * dimensions;
|
||||
const end = start + dimensions;
|
||||
embeddings.push(new Float32Array(Array.prototype.slice.call(data, start, end)));
|
||||
}
|
||||
|
||||
return embeddings;
|
||||
};
|
||||
|
||||
/**
|
||||
* Convert Float32Array to regular number array (for KuzuDB storage)
|
||||
*/
|
||||
export const embeddingToArray = (embedding: Float32Array): number[] => {
|
||||
return Array.from(embedding);
|
||||
};
|
||||
|
||||
/**
|
||||
* Cleanup the embedder (free memory)
|
||||
* Call this when done with embeddings
|
||||
*/
|
||||
export const disposeEmbedder = async (): Promise<void> => {
|
||||
if (embedderInstance) {
|
||||
// transformers.js pipelines may have a dispose method
|
||||
try {
|
||||
if ('dispose' in embedderInstance && typeof embedderInstance.dispose === 'function') {
|
||||
await embedderInstance.dispose();
|
||||
}
|
||||
} catch {
|
||||
// Ignore disposal errors
|
||||
}
|
||||
embedderInstance = null;
|
||||
initPromise = null;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -0,0 +1,399 @@
|
||||
/**
|
||||
* Embedding Pipeline Module
|
||||
*
|
||||
* Orchestrates the background embedding process:
|
||||
* 1. Query embeddable nodes from KuzuDB
|
||||
* 2. Generate text representations
|
||||
* 3. Batch embed using transformers.js
|
||||
* 4. Update KuzuDB with embeddings
|
||||
* 5. Create vector index for semantic search
|
||||
*/
|
||||
|
||||
import { initEmbedder, embedBatch, embedText, embeddingToArray, isEmbedderReady } from './embedder';
|
||||
import { generateBatchEmbeddingTexts, generateEmbeddingText } from './text-generator';
|
||||
import {
|
||||
type EmbeddingProgress,
|
||||
type EmbeddingConfig,
|
||||
type EmbeddableNode,
|
||||
type SemanticSearchResult,
|
||||
type ModelProgress,
|
||||
DEFAULT_EMBEDDING_CONFIG,
|
||||
EMBEDDABLE_LABELS,
|
||||
} from './types';
|
||||
|
||||
/**
|
||||
* Progress callback type
|
||||
*/
|
||||
export type EmbeddingProgressCallback = (progress: EmbeddingProgress) => void;
|
||||
|
||||
/**
|
||||
* Query all embeddable nodes from KuzuDB
|
||||
* Uses table-specific queries (File has different schema than code elements)
|
||||
*/
|
||||
const queryEmbeddableNodes = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>
|
||||
): Promise<EmbeddableNode[]> => {
|
||||
const allNodes: EmbeddableNode[] = [];
|
||||
|
||||
// Query each embeddable table with table-specific columns
|
||||
for (const label of EMBEDDABLE_LABELS) {
|
||||
try {
|
||||
let query: string;
|
||||
|
||||
if (label === 'File') {
|
||||
// File nodes don't have startLine/endLine
|
||||
query = `
|
||||
MATCH (n:File)
|
||||
RETURN n.id AS id, n.name AS name, 'File' AS label,
|
||||
n.filePath AS filePath, n.content AS content
|
||||
`;
|
||||
} else {
|
||||
// Code elements have startLine/endLine
|
||||
query = `
|
||||
MATCH (n:${label})
|
||||
RETURN n.id AS id, n.name AS name, '${label}' AS label,
|
||||
n.filePath AS filePath, n.content AS content,
|
||||
n.startLine AS startLine, n.endLine AS endLine
|
||||
`;
|
||||
}
|
||||
|
||||
const rows = await executeQuery(query);
|
||||
for (const row of rows) {
|
||||
allNodes.push({
|
||||
id: row.id ?? row[0],
|
||||
name: row.name ?? row[1],
|
||||
label: row.label ?? row[2],
|
||||
filePath: row.filePath ?? row[3],
|
||||
content: row.content ?? row[4] ?? '',
|
||||
startLine: row.startLine ?? row[5],
|
||||
endLine: row.endLine ?? row[6],
|
||||
});
|
||||
}
|
||||
} catch (error) {
|
||||
// Table might not exist or be empty, continue
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn(`Query for ${label} nodes failed:`, error);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return allNodes;
|
||||
};
|
||||
|
||||
/**
|
||||
* Batch INSERT embeddings into separate CodeEmbedding table
|
||||
* Using a separate lightweight table avoids copy-on-write overhead
|
||||
* that occurs when UPDATEing nodes with large content fields
|
||||
*/
|
||||
const batchInsertEmbeddings = async (
|
||||
executeWithReusedStatement: (
|
||||
cypher: string,
|
||||
paramsList: Array<Record<string, any>>
|
||||
) => Promise<void>,
|
||||
updates: Array<{ id: string; embedding: number[] }>
|
||||
): Promise<void> => {
|
||||
// INSERT into separate embedding table - much more memory efficient!
|
||||
const cypher = `CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`;
|
||||
const paramsList = updates.map(u => ({ nodeId: u.id, embedding: u.embedding }));
|
||||
await executeWithReusedStatement(cypher, paramsList);
|
||||
};
|
||||
|
||||
/**
|
||||
* Create the vector index for semantic search
|
||||
* Now indexes the separate CodeEmbedding table
|
||||
*/
|
||||
const createVectorIndex = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>
|
||||
): Promise<void> => {
|
||||
const cypher = `
|
||||
CALL CREATE_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx', 'embedding', metric := 'cosine')
|
||||
`;
|
||||
|
||||
try {
|
||||
await executeQuery(cypher);
|
||||
} catch (error) {
|
||||
// Index might already exist
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn('Vector index creation warning:', error);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Run the embedding pipeline
|
||||
*
|
||||
* @param executeQuery - Function to execute Cypher queries against KuzuDB
|
||||
* @param executeWithReusedStatement - Function to execute with reused prepared statement
|
||||
* @param onProgress - Callback for progress updates
|
||||
* @param config - Optional configuration override
|
||||
*/
|
||||
export const runEmbeddingPipeline = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>,
|
||||
executeWithReusedStatement: (cypher: string, paramsList: Array<Record<string, any>>) => Promise<void>,
|
||||
onProgress: EmbeddingProgressCallback,
|
||||
config: Partial<EmbeddingConfig> = {}
|
||||
): Promise<void> => {
|
||||
const finalConfig = { ...DEFAULT_EMBEDDING_CONFIG, ...config };
|
||||
|
||||
try {
|
||||
// Phase 1: Load embedding model
|
||||
onProgress({
|
||||
phase: 'loading-model',
|
||||
percent: 0,
|
||||
modelDownloadPercent: 0,
|
||||
});
|
||||
|
||||
await initEmbedder((modelProgress: ModelProgress) => {
|
||||
// Report model download progress
|
||||
const downloadPercent = modelProgress.progress ?? 0;
|
||||
onProgress({
|
||||
phase: 'loading-model',
|
||||
percent: Math.round(downloadPercent * 0.2), // 0-20% for model loading
|
||||
modelDownloadPercent: downloadPercent,
|
||||
});
|
||||
}, finalConfig);
|
||||
|
||||
onProgress({
|
||||
phase: 'loading-model',
|
||||
percent: 20,
|
||||
modelDownloadPercent: 100,
|
||||
});
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('🔍 Querying embeddable nodes...');
|
||||
}
|
||||
|
||||
// Phase 2: Query embeddable nodes
|
||||
const nodes = await queryEmbeddableNodes(executeQuery);
|
||||
const totalNodes = nodes.length;
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`📊 Found ${totalNodes} embeddable nodes`);
|
||||
}
|
||||
|
||||
if (totalNodes === 0) {
|
||||
onProgress({
|
||||
phase: 'ready',
|
||||
percent: 100,
|
||||
nodesProcessed: 0,
|
||||
totalNodes: 0,
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
// Phase 3: Batch embed nodes
|
||||
const batchSize = finalConfig.batchSize;
|
||||
const totalBatches = Math.ceil(totalNodes / batchSize);
|
||||
let processedNodes = 0;
|
||||
|
||||
onProgress({
|
||||
phase: 'embedding',
|
||||
percent: 20,
|
||||
nodesProcessed: 0,
|
||||
totalNodes,
|
||||
currentBatch: 0,
|
||||
totalBatches,
|
||||
});
|
||||
|
||||
for (let batchIndex = 0; batchIndex < totalBatches; batchIndex++) {
|
||||
const start = batchIndex * batchSize;
|
||||
const end = Math.min(start + batchSize, totalNodes);
|
||||
const batch = nodes.slice(start, end);
|
||||
|
||||
// Generate texts for this batch
|
||||
const texts = generateBatchEmbeddingTexts(batch, finalConfig);
|
||||
|
||||
// Embed the batch
|
||||
const embeddings = await embedBatch(texts);
|
||||
|
||||
// Update KuzuDB with embeddings
|
||||
const updates = batch.map((node, i) => ({
|
||||
id: node.id,
|
||||
embedding: embeddingToArray(embeddings[i]),
|
||||
}));
|
||||
|
||||
await batchInsertEmbeddings(executeWithReusedStatement, updates);
|
||||
|
||||
processedNodes += batch.length;
|
||||
|
||||
// Report progress (20-90% for embedding phase)
|
||||
const embeddingProgress = 20 + ((processedNodes / totalNodes) * 70);
|
||||
onProgress({
|
||||
phase: 'embedding',
|
||||
percent: Math.round(embeddingProgress),
|
||||
nodesProcessed: processedNodes,
|
||||
totalNodes,
|
||||
currentBatch: batchIndex + 1,
|
||||
totalBatches,
|
||||
});
|
||||
}
|
||||
|
||||
// Phase 4: Create vector index
|
||||
onProgress({
|
||||
phase: 'indexing',
|
||||
percent: 90,
|
||||
nodesProcessed: totalNodes,
|
||||
totalNodes,
|
||||
});
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('📇 Creating vector index...');
|
||||
}
|
||||
|
||||
await createVectorIndex(executeQuery);
|
||||
|
||||
// Complete
|
||||
onProgress({
|
||||
phase: 'ready',
|
||||
percent: 100,
|
||||
nodesProcessed: totalNodes,
|
||||
totalNodes,
|
||||
});
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('✅ Embedding pipeline complete!');
|
||||
}
|
||||
} catch (error) {
|
||||
const errorMessage = error instanceof Error ? error.message : 'Unknown error';
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.error('❌ Embedding pipeline error:', error);
|
||||
}
|
||||
|
||||
onProgress({
|
||||
phase: 'error',
|
||||
percent: 0,
|
||||
error: errorMessage,
|
||||
});
|
||||
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Perform semantic search using the vector index
|
||||
*
|
||||
* Uses CodeEmbedding table and queries each node table to get metadata
|
||||
*
|
||||
* @param executeQuery - Function to execute Cypher queries
|
||||
* @param query - Search query text
|
||||
* @param k - Number of results to return (default: 10)
|
||||
* @param maxDistance - Maximum distance threshold (default: 0.5)
|
||||
* @returns Array of search results ordered by relevance
|
||||
*/
|
||||
export const semanticSearch = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>,
|
||||
query: string,
|
||||
k: number = 10,
|
||||
maxDistance: number = 0.5
|
||||
): Promise<SemanticSearchResult[]> => {
|
||||
if (!isEmbedderReady()) {
|
||||
throw new Error('Embedding model not initialized. Run embedding pipeline first.');
|
||||
}
|
||||
|
||||
// Embed the query
|
||||
const queryEmbedding = await embedText(query);
|
||||
const queryVec = embeddingToArray(queryEmbedding);
|
||||
const queryVecStr = `[${queryVec.join(',')}]`;
|
||||
|
||||
// Query the vector index on CodeEmbedding to get nodeIds and distances
|
||||
const vectorQuery = `
|
||||
CALL QUERY_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx',
|
||||
CAST(${queryVecStr} AS FLOAT[384]), ${k})
|
||||
YIELD node AS emb, distance
|
||||
WITH emb, distance
|
||||
WHERE distance < ${maxDistance}
|
||||
RETURN emb.nodeId AS nodeId, distance
|
||||
ORDER BY distance
|
||||
`;
|
||||
|
||||
const embResults = await executeQuery(vectorQuery);
|
||||
|
||||
if (embResults.length === 0) {
|
||||
return [];
|
||||
}
|
||||
|
||||
// Get metadata for each result by querying each node table
|
||||
const results: SemanticSearchResult[] = [];
|
||||
|
||||
for (const embRow of embResults) {
|
||||
const nodeId = embRow.nodeId ?? embRow[0];
|
||||
const distance = embRow.distance ?? embRow[1];
|
||||
|
||||
// Extract label from node ID (format: Label:path:name)
|
||||
const labelEndIdx = nodeId.indexOf(':');
|
||||
const label = labelEndIdx > 0 ? nodeId.substring(0, labelEndIdx) : 'Unknown';
|
||||
|
||||
// Query the specific table for this node
|
||||
// File nodes don't have startLine/endLine
|
||||
try {
|
||||
let nodeQuery: string;
|
||||
if (label === 'File') {
|
||||
nodeQuery = `
|
||||
MATCH (n:File {id: '${nodeId.replace(/'/g, "''")}'})
|
||||
RETURN n.name AS name, n.filePath AS filePath
|
||||
`;
|
||||
} else {
|
||||
nodeQuery = `
|
||||
MATCH (n:${label} {id: '${nodeId.replace(/'/g, "''")}'})
|
||||
RETURN n.name AS name, n.filePath AS filePath,
|
||||
n.startLine AS startLine, n.endLine AS endLine
|
||||
`;
|
||||
}
|
||||
const nodeRows = await executeQuery(nodeQuery);
|
||||
if (nodeRows.length > 0) {
|
||||
const nodeRow = nodeRows[0];
|
||||
results.push({
|
||||
nodeId,
|
||||
name: nodeRow.name ?? nodeRow[0] ?? '',
|
||||
label,
|
||||
filePath: nodeRow.filePath ?? nodeRow[1] ?? '',
|
||||
distance,
|
||||
startLine: label !== 'File' ? (nodeRow.startLine ?? nodeRow[2]) : undefined,
|
||||
endLine: label !== 'File' ? (nodeRow.endLine ?? nodeRow[3]) : undefined,
|
||||
});
|
||||
}
|
||||
} catch {
|
||||
// Table might not exist, skip
|
||||
}
|
||||
}
|
||||
|
||||
return results;
|
||||
};
|
||||
|
||||
/**
|
||||
* Semantic search with graph expansion (flattened results)
|
||||
*
|
||||
* Note: With multi-table schema, graph traversal is simplified.
|
||||
* Returns semantic matches with their metadata.
|
||||
* For full graph traversal, use execute_vector_cypher tool directly.
|
||||
*
|
||||
* @param executeQuery - Function to execute Cypher queries
|
||||
* @param query - Search query text
|
||||
* @param k - Number of initial semantic matches (default: 5)
|
||||
* @param _hops - Unused (kept for API compatibility).
|
||||
* @returns Semantic matches with metadata
|
||||
*/
|
||||
export const semanticSearchWithContext = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>,
|
||||
query: string,
|
||||
k: number = 5,
|
||||
_hops: number = 1
|
||||
): Promise<any[]> => {
|
||||
// For multi-table schema, just return semantic search results
|
||||
// Graph traversal is complex with separate tables - use execute_vector_cypher instead
|
||||
const results = await semanticSearch(executeQuery, query, k, 0.5);
|
||||
|
||||
return results.map(r => ({
|
||||
matchId: r.nodeId,
|
||||
matchName: r.name,
|
||||
matchLabel: r.label,
|
||||
matchPath: r.filePath,
|
||||
distance: r.distance,
|
||||
connectedId: null,
|
||||
connectedName: null,
|
||||
connectedLabel: null,
|
||||
relationType: null,
|
||||
}));
|
||||
};
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
/**
|
||||
* Embeddings Module
|
||||
*
|
||||
* Re-exports for the embedding pipeline system.
|
||||
*/
|
||||
|
||||
export * from './types';
|
||||
export * from './embedder';
|
||||
export * from './text-generator';
|
||||
export * from './embedding-pipeline';
|
||||
|
||||
@@ -0,0 +1,235 @@
|
||||
/**
|
||||
* Text Generator Module
|
||||
*
|
||||
* Pure functions to generate embedding text from code nodes.
|
||||
* Combines node metadata with code snippets for semantic matching.
|
||||
*/
|
||||
|
||||
import type { EmbeddableNode, EmbeddingConfig } from './types';
|
||||
import { DEFAULT_EMBEDDING_CONFIG } from './types';
|
||||
|
||||
/**
|
||||
* Extract the filename from a file path
|
||||
*/
|
||||
const getFileName = (filePath: string): string => {
|
||||
const parts = filePath.split('/');
|
||||
return parts[parts.length - 1] || filePath;
|
||||
};
|
||||
|
||||
/**
|
||||
* Extract the directory path from a file path
|
||||
*/
|
||||
const getDirectory = (filePath: string): string => {
|
||||
const parts = filePath.split('/');
|
||||
parts.pop();
|
||||
return parts.join('/') || '';
|
||||
};
|
||||
|
||||
/**
|
||||
* Truncate content to max length, preserving word boundaries
|
||||
*/
|
||||
const truncateContent = (content: string, maxLength: number): string => {
|
||||
if (content.length <= maxLength) {
|
||||
return content;
|
||||
}
|
||||
|
||||
// Find last space before maxLength to avoid cutting words
|
||||
const truncated = content.slice(0, maxLength);
|
||||
const lastSpace = truncated.lastIndexOf(' ');
|
||||
|
||||
if (lastSpace > maxLength * 0.8) {
|
||||
return truncated.slice(0, lastSpace) + '...';
|
||||
}
|
||||
|
||||
return truncated + '...';
|
||||
};
|
||||
|
||||
/**
|
||||
* Clean code content for embedding
|
||||
* Removes excessive whitespace while preserving structure
|
||||
*/
|
||||
const cleanContent = (content: string): string => {
|
||||
return content
|
||||
// Normalize line endings
|
||||
.replace(/\r\n/g, '\n')
|
||||
// Remove excessive blank lines (more than 2)
|
||||
.replace(/\n{3,}/g, '\n\n')
|
||||
// Trim each line
|
||||
.split('\n')
|
||||
.map(line => line.trimEnd())
|
||||
.join('\n')
|
||||
.trim();
|
||||
};
|
||||
|
||||
/**
|
||||
* Generate embedding text for a Function node
|
||||
*/
|
||||
const generateFunctionText = (
|
||||
node: EmbeddableNode,
|
||||
maxSnippetLength: number
|
||||
): string => {
|
||||
const parts: string[] = [
|
||||
`Function: ${node.name}`,
|
||||
`File: ${getFileName(node.filePath)}`,
|
||||
];
|
||||
|
||||
const dir = getDirectory(node.filePath);
|
||||
if (dir) {
|
||||
parts.push(`Directory: ${dir}`);
|
||||
}
|
||||
|
||||
if (node.content) {
|
||||
const cleanedContent = cleanContent(node.content);
|
||||
const snippet = truncateContent(cleanedContent, maxSnippetLength);
|
||||
parts.push('', snippet);
|
||||
}
|
||||
|
||||
return parts.join('\n');
|
||||
};
|
||||
|
||||
/**
|
||||
* Generate embedding text for a Class node
|
||||
*/
|
||||
const generateClassText = (
|
||||
node: EmbeddableNode,
|
||||
maxSnippetLength: number
|
||||
): string => {
|
||||
const parts: string[] = [
|
||||
`Class: ${node.name}`,
|
||||
`File: ${getFileName(node.filePath)}`,
|
||||
];
|
||||
|
||||
const dir = getDirectory(node.filePath);
|
||||
if (dir) {
|
||||
parts.push(`Directory: ${dir}`);
|
||||
}
|
||||
|
||||
if (node.content) {
|
||||
const cleanedContent = cleanContent(node.content);
|
||||
const snippet = truncateContent(cleanedContent, maxSnippetLength);
|
||||
parts.push('', snippet);
|
||||
}
|
||||
|
||||
return parts.join('\n');
|
||||
};
|
||||
|
||||
/**
|
||||
* Generate embedding text for a Method node
|
||||
*/
|
||||
const generateMethodText = (
|
||||
node: EmbeddableNode,
|
||||
maxSnippetLength: number
|
||||
): string => {
|
||||
const parts: string[] = [
|
||||
`Method: ${node.name}`,
|
||||
`File: ${getFileName(node.filePath)}`,
|
||||
];
|
||||
|
||||
const dir = getDirectory(node.filePath);
|
||||
if (dir) {
|
||||
parts.push(`Directory: ${dir}`);
|
||||
}
|
||||
|
||||
if (node.content) {
|
||||
const cleanedContent = cleanContent(node.content);
|
||||
const snippet = truncateContent(cleanedContent, maxSnippetLength);
|
||||
parts.push('', snippet);
|
||||
}
|
||||
|
||||
return parts.join('\n');
|
||||
};
|
||||
|
||||
/**
|
||||
* Generate embedding text for an Interface node
|
||||
*/
|
||||
const generateInterfaceText = (
|
||||
node: EmbeddableNode,
|
||||
maxSnippetLength: number
|
||||
): string => {
|
||||
const parts: string[] = [
|
||||
`Interface: ${node.name}`,
|
||||
`File: ${getFileName(node.filePath)}`,
|
||||
];
|
||||
|
||||
const dir = getDirectory(node.filePath);
|
||||
if (dir) {
|
||||
parts.push(`Directory: ${dir}`);
|
||||
}
|
||||
|
||||
if (node.content) {
|
||||
const cleanedContent = cleanContent(node.content);
|
||||
const snippet = truncateContent(cleanedContent, maxSnippetLength);
|
||||
parts.push('', snippet);
|
||||
}
|
||||
|
||||
return parts.join('\n');
|
||||
};
|
||||
|
||||
/**
|
||||
* Generate embedding text for a File node
|
||||
* Uses file name and first N characters of content
|
||||
*/
|
||||
const generateFileText = (
|
||||
node: EmbeddableNode,
|
||||
maxSnippetLength: number
|
||||
): string => {
|
||||
const parts: string[] = [
|
||||
`File: ${node.name}`,
|
||||
`Path: ${node.filePath}`,
|
||||
];
|
||||
|
||||
if (node.content) {
|
||||
const cleanedContent = cleanContent(node.content);
|
||||
// For files, use a shorter snippet since they can be very long
|
||||
const snippet = truncateContent(cleanedContent, Math.min(maxSnippetLength, 300));
|
||||
parts.push('', snippet);
|
||||
}
|
||||
|
||||
return parts.join('\n');
|
||||
};
|
||||
|
||||
/**
|
||||
* Generate embedding text for any embeddable node
|
||||
* Dispatches to the appropriate generator based on node label
|
||||
*
|
||||
* @param node - The node to generate text for
|
||||
* @param config - Optional configuration for max snippet length
|
||||
* @returns Text suitable for embedding
|
||||
*/
|
||||
export const generateEmbeddingText = (
|
||||
node: EmbeddableNode,
|
||||
config: Partial<EmbeddingConfig> = {}
|
||||
): string => {
|
||||
const maxSnippetLength = config.maxSnippetLength ?? DEFAULT_EMBEDDING_CONFIG.maxSnippetLength;
|
||||
|
||||
switch (node.label) {
|
||||
case 'Function':
|
||||
return generateFunctionText(node, maxSnippetLength);
|
||||
case 'Class':
|
||||
return generateClassText(node, maxSnippetLength);
|
||||
case 'Method':
|
||||
return generateMethodText(node, maxSnippetLength);
|
||||
case 'Interface':
|
||||
return generateInterfaceText(node, maxSnippetLength);
|
||||
case 'File':
|
||||
return generateFileText(node, maxSnippetLength);
|
||||
default:
|
||||
// Fallback for any other embeddable type
|
||||
return `${node.label}: ${node.name}\nPath: ${node.filePath}`;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Generate embedding texts for a batch of nodes
|
||||
*
|
||||
* @param nodes - Array of nodes to generate text for
|
||||
* @param config - Optional configuration
|
||||
* @returns Array of texts in the same order as input nodes
|
||||
*/
|
||||
export const generateBatchEmbeddingTexts = (
|
||||
nodes: EmbeddableNode[],
|
||||
config: Partial<EmbeddingConfig> = {}
|
||||
): string[] => {
|
||||
return nodes.map(node => generateEmbeddingText(node, config));
|
||||
};
|
||||
|
||||
@@ -0,0 +1,117 @@
|
||||
/**
|
||||
* Embedding Pipeline Types
|
||||
*
|
||||
* Type definitions for the embedding generation and semantic search system.
|
||||
*/
|
||||
|
||||
/**
|
||||
* Node labels that should be embedded for semantic search
|
||||
* These are code elements that benefit from semantic matching
|
||||
*/
|
||||
export const EMBEDDABLE_LABELS = [
|
||||
'Function',
|
||||
'Class',
|
||||
'Method',
|
||||
'Interface',
|
||||
'File',
|
||||
] as const;
|
||||
|
||||
export type EmbeddableLabel = typeof EMBEDDABLE_LABELS[number];
|
||||
|
||||
/**
|
||||
* Check if a label should be embedded
|
||||
*/
|
||||
export const isEmbeddableLabel = (label: string): label is EmbeddableLabel =>
|
||||
EMBEDDABLE_LABELS.includes(label as EmbeddableLabel);
|
||||
|
||||
/**
|
||||
* Embedding pipeline phases
|
||||
*/
|
||||
export type EmbeddingPhase =
|
||||
| 'idle'
|
||||
| 'loading-model'
|
||||
| 'embedding'
|
||||
| 'indexing'
|
||||
| 'ready'
|
||||
| 'error';
|
||||
|
||||
/**
|
||||
* Progress information for the embedding pipeline
|
||||
*/
|
||||
export interface EmbeddingProgress {
|
||||
phase: EmbeddingPhase;
|
||||
percent: number;
|
||||
modelDownloadPercent?: number;
|
||||
nodesProcessed?: number;
|
||||
totalNodes?: number;
|
||||
currentBatch?: number;
|
||||
totalBatches?: number;
|
||||
error?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Configuration for the embedding pipeline
|
||||
*/
|
||||
export interface EmbeddingConfig {
|
||||
/** Model identifier for transformers.js */
|
||||
modelId: string;
|
||||
/** Number of nodes to embed in each batch */
|
||||
batchSize: number;
|
||||
/** Embedding vector dimensions */
|
||||
dimensions: number;
|
||||
/** Device to use for inference: 'webgpu' for GPU acceleration, 'wasm' for WASM-based CPU */
|
||||
device: 'webgpu' | 'wasm';
|
||||
/** Maximum characters of code snippet to include */
|
||||
maxSnippetLength: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Default embedding configuration
|
||||
* Uses snowflake-arctic-embed-xs for browser efficiency
|
||||
* Tries WebGPU first (fast), user can choose WASM fallback if unavailable
|
||||
*/
|
||||
export const DEFAULT_EMBEDDING_CONFIG: EmbeddingConfig = {
|
||||
modelId: 'Snowflake/snowflake-arctic-embed-xs',
|
||||
batchSize: 16,
|
||||
dimensions: 384,
|
||||
device: 'webgpu', // WebGPU preferred, WASM fallback available if user chooses
|
||||
maxSnippetLength: 500,
|
||||
};
|
||||
|
||||
/**
|
||||
* Result from semantic search
|
||||
*/
|
||||
export interface SemanticSearchResult {
|
||||
nodeId: string;
|
||||
name: string;
|
||||
label: string;
|
||||
filePath: string;
|
||||
distance: number;
|
||||
startLine?: number;
|
||||
endLine?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Node data for embedding (minimal structure from KuzuDB query)
|
||||
*/
|
||||
export interface EmbeddableNode {
|
||||
id: string;
|
||||
name: string;
|
||||
label: string;
|
||||
filePath: string;
|
||||
content: string;
|
||||
startLine?: number;
|
||||
endLine?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Model download progress from transformers.js
|
||||
*/
|
||||
export interface ModelProgress {
|
||||
status: 'initiate' | 'download' | 'progress' | 'done' | 'ready';
|
||||
file?: string;
|
||||
progress?: number;
|
||||
loaded?: number;
|
||||
total?: number;
|
||||
}
|
||||
|
||||
@@ -1,295 +0,0 @@
|
||||
/**
|
||||
* Dual-Write Knowledge Graph Implementation
|
||||
*
|
||||
* This class implements transparent dual-write functionality, writing data to both
|
||||
* JSON (SimpleKnowledgeGraph) and KuzuDB simultaneously. The JSON storage remains
|
||||
* the primary source of truth, while KuzuDB provides enhanced query capabilities.
|
||||
*/
|
||||
|
||||
import type { KnowledgeGraph, GraphNode, GraphRelationship } from './types.ts';
|
||||
import { SimpleKnowledgeGraph } from './graph.ts';
|
||||
import type { KuzuKnowledgeGraph } from './kuzu-knowledge-graph.ts';
|
||||
|
||||
export class DualWriteKnowledgeGraph implements KnowledgeGraph {
|
||||
private jsonGraph: SimpleKnowledgeGraph;
|
||||
private kuzuGraph: KuzuKnowledgeGraph | null;
|
||||
private enableKuzuDB: boolean;
|
||||
private dualWriteStats: {
|
||||
nodesWrittenToJSON: number;
|
||||
nodesWrittenToKuzuDB: number;
|
||||
relationshipsWrittenToJSON: number;
|
||||
relationshipsWrittenToKuzuDB: number;
|
||||
kuzuErrors: number;
|
||||
};
|
||||
|
||||
constructor(kuzuGraph?: KuzuKnowledgeGraph) {
|
||||
this.jsonGraph = new SimpleKnowledgeGraph();
|
||||
this.kuzuGraph = kuzuGraph || null;
|
||||
this.enableKuzuDB = !!kuzuGraph;
|
||||
this.dualWriteStats = {
|
||||
nodesWrittenToJSON: 0,
|
||||
nodesWrittenToKuzuDB: 0,
|
||||
relationshipsWrittenToJSON: 0,
|
||||
relationshipsWrittenToKuzuDB: 0,
|
||||
kuzuErrors: 0
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Get all nodes in the graph (from JSON primary storage)
|
||||
*/
|
||||
get nodes(): GraphNode[] {
|
||||
return this.jsonGraph.nodes;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get all relationships in the graph (from JSON primary storage)
|
||||
*/
|
||||
get relationships(): GraphRelationship[] {
|
||||
return this.jsonGraph.relationships;
|
||||
}
|
||||
|
||||
/**
|
||||
* Add node - transparent dual-write
|
||||
* Maintains exact same synchronous interface as original
|
||||
*/
|
||||
addNode(node: GraphNode): void {
|
||||
// Always write to JSON first (primary storage)
|
||||
this.jsonGraph.addNode(node);
|
||||
this.dualWriteStats.nodesWrittenToJSON++;
|
||||
|
||||
// Write to KuzuDB in background if enabled
|
||||
if (this.enableKuzuDB && this.kuzuGraph) {
|
||||
try {
|
||||
// KuzuDB addNode is synchronous (batched)
|
||||
this.kuzuGraph.addNode(node);
|
||||
this.dualWriteStats.nodesWrittenToKuzuDB++;
|
||||
} catch (error) {
|
||||
this.dualWriteStats.kuzuErrors++;
|
||||
console.warn(`❌ KuzuDB node write failed for ${node.id} (${node.label}):`, error);
|
||||
// Continue - JSON is primary, KuzuDB failure shouldn't break the process
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Add relationship - transparent dual-write
|
||||
* Maintains exact same synchronous interface as original
|
||||
*/
|
||||
addRelationship(relationship: GraphRelationship): void {
|
||||
// Always write to JSON first (primary storage)
|
||||
this.jsonGraph.addRelationship(relationship);
|
||||
this.dualWriteStats.relationshipsWrittenToJSON++;
|
||||
|
||||
// Write to KuzuDB in background if enabled
|
||||
if (this.enableKuzuDB && this.kuzuGraph) {
|
||||
try {
|
||||
// KuzuDB addRelationship is synchronous (batched)
|
||||
this.kuzuGraph.addRelationship(relationship);
|
||||
this.dualWriteStats.relationshipsWrittenToKuzuDB++;
|
||||
} catch (error) {
|
||||
this.dualWriteStats.kuzuErrors++;
|
||||
console.warn(`❌ KuzuDB relationship write failed for ${relationship.id} (${relationship.type}):`, error);
|
||||
// Continue - JSON is primary, KuzuDB failure shouldn't break the process
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get KuzuDB instance for advanced operations (optional)
|
||||
*/
|
||||
getKuzuGraph(): KuzuKnowledgeGraph | null {
|
||||
return this.kuzuGraph;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if KuzuDB is enabled and available
|
||||
*/
|
||||
isKuzuDBEnabled(): boolean {
|
||||
return this.enableKuzuDB && this.kuzuGraph !== null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get dual-write statistics
|
||||
*/
|
||||
getDualWriteStats() {
|
||||
return { ...this.dualWriteStats };
|
||||
}
|
||||
|
||||
/**
|
||||
* Log dual-write statistics
|
||||
*/
|
||||
logDualWriteStats(): void {
|
||||
console.log('📊 Dual-Write Statistics:');
|
||||
console.log(` JSON nodes written: ${this.dualWriteStats.nodesWrittenToJSON}`);
|
||||
console.log(` JSON relationships written: ${this.dualWriteStats.relationshipsWrittenToJSON}`);
|
||||
console.log(` Total JSON entities: ${this.dualWriteStats.nodesWrittenToJSON + this.dualWriteStats.relationshipsWrittenToJSON}`);
|
||||
|
||||
// Compare with actual graph counts
|
||||
const actualNodes = this.jsonGraph.nodes.length;
|
||||
const actualRels = this.jsonGraph.relationships.length;
|
||||
console.log(` Actual JSON graph: ${actualNodes} nodes, ${actualRels} relationships`);
|
||||
|
||||
if (actualNodes !== this.dualWriteStats.nodesWrittenToJSON) {
|
||||
console.warn(` ⚠️ Mismatch: Expected ${this.dualWriteStats.nodesWrittenToJSON} nodes, but graph has ${actualNodes}`);
|
||||
}
|
||||
if (actualRels !== this.dualWriteStats.relationshipsWrittenToJSON) {
|
||||
console.warn(` ⚠️ Mismatch: Expected ${this.dualWriteStats.relationshipsWrittenToJSON} relationships, but graph has ${actualRels}`);
|
||||
}
|
||||
|
||||
if (this.enableKuzuDB) {
|
||||
console.log(` KuzuDB nodes written: ${this.dualWriteStats.nodesWrittenToKuzuDB}`);
|
||||
console.log(` KuzuDB relationships written: ${this.dualWriteStats.relationshipsWrittenToKuzuDB}`);
|
||||
console.log(` KuzuDB errors: ${this.dualWriteStats.kuzuErrors}`);
|
||||
|
||||
const totalWrites = this.dualWriteStats.nodesWrittenToJSON + this.dualWriteStats.relationshipsWrittenToJSON;
|
||||
const kuzuWrites = this.dualWriteStats.nodesWrittenToKuzuDB + this.dualWriteStats.relationshipsWrittenToKuzuDB;
|
||||
const successRate = totalWrites > 0 ? (kuzuWrites / totalWrites * 100).toFixed(1) : '0';
|
||||
console.log(` KuzuDB success rate: ${successRate}%`);
|
||||
|
||||
if (this.dualWriteStats.kuzuErrors > 0) {
|
||||
console.log(` ⚠️ ${this.dualWriteStats.kuzuErrors} KuzuDB write failures detected - check logs above for details`);
|
||||
}
|
||||
} else {
|
||||
console.log(' KuzuDB: Disabled');
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Flush any pending KuzuDB operations (for cleanup)
|
||||
*/
|
||||
async flushKuzuDB(): Promise<void> {
|
||||
if (this.enableKuzuDB && this.kuzuGraph) {
|
||||
try {
|
||||
await this.kuzuGraph.commitAll();
|
||||
console.log('✅ KuzuDB operations flushed successfully');
|
||||
|
||||
// Verify KuzuDB has data by running test queries
|
||||
await this.verifyKuzuDBData();
|
||||
} catch (error) {
|
||||
console.error('❌ Failed to flush KuzuDB operations:', error);
|
||||
this.dualWriteStats.kuzuErrors++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Verify KuzuDB contains expected data by running test queries
|
||||
*/
|
||||
private async verifyKuzuDBData(): Promise<void> {
|
||||
if (!this.kuzuGraph || !('executeQuery' in this.kuzuGraph)) {
|
||||
console.log('⚠️ Cannot verify KuzuDB data - no query interface available');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
console.log('🔍 Verifying KuzuDB data...');
|
||||
|
||||
// Test query 1: Count all nodes
|
||||
const nodeCountResult = await (this.kuzuGraph as any).executeQuery('MATCH (n) RETURN COUNT(n) as nodeCount');
|
||||
const nodeCount = nodeCountResult.rows?.[0]?.[0] || 0;
|
||||
console.log(`📊 KuzuDB Verification - Total nodes: ${nodeCount}`);
|
||||
|
||||
// Test query 2: Count all relationships
|
||||
const relCountResult = await (this.kuzuGraph as any).executeQuery('MATCH ()-[r]->() RETURN COUNT(r) as relCount');
|
||||
const relCount = relCountResult.rows?.[0]?.[0] || 0;
|
||||
console.log(`📊 KuzuDB Verification - Total relationships: ${relCount}`);
|
||||
|
||||
// Test query 3: Count nodes by type (KuzuDB-compatible)
|
||||
// Query each node table separately since KuzuDB doesn't have labels() function
|
||||
// Check if polymorphic nodes are enabled
|
||||
const { isPolymorphicNodesEnabled } = await import('../../config/features.ts');
|
||||
|
||||
if (isPolymorphicNodesEnabled()) {
|
||||
// Polymorphic approach: Query by elementType
|
||||
console.log('📊 KuzuDB Verification - Node types (polymorphic):');
|
||||
const nodeTypes = ['Function', 'Class', 'Method', 'File', 'Variable', 'Interface', 'Type', 'Import', 'Project', 'Folder'];
|
||||
|
||||
for (const nodeType of nodeTypes) {
|
||||
try {
|
||||
const result = await (this.kuzuGraph as any).executeQuery(`MATCH (n:CodeElement {elementType: '${nodeType}'}) RETURN COUNT(n) as count`);
|
||||
const count = result.rows?.[0]?.[0] || 0;
|
||||
if (count > 0) {
|
||||
console.log(` ${nodeType}: ${count} nodes`);
|
||||
}
|
||||
} catch (error) {
|
||||
// Skip silently
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Traditional approach: Query individual tables
|
||||
console.log('📊 KuzuDB Verification - Node types:');
|
||||
const nodeTypes = ['Function', 'Class', 'Method', 'File', 'Variable', 'Interface', 'Type', 'Import', 'Project', 'Folder'];
|
||||
|
||||
for (const nodeType of nodeTypes) {
|
||||
try {
|
||||
const result = await (this.kuzuGraph as any).executeQuery(`MATCH (n:${nodeType}) RETURN COUNT(n) as count`);
|
||||
const count = result.rows?.[0]?.[0] || 0;
|
||||
if (count > 0) {
|
||||
console.log(` ${nodeType}: ${count} nodes`);
|
||||
}
|
||||
} catch (error) {
|
||||
// Node type might not exist in this database, skip silently
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Test query 4: Count relationships by type (KuzuDB-compatible)
|
||||
// Query each relationship table separately since KuzuDB doesn't have type() function
|
||||
const relTypes = ['CONTAINS', 'CALLS', 'INHERITS', 'IMPLEMENTS', 'OVERRIDES', 'IMPORTS', 'DEFINES', 'BELONGS_TO', 'USES', 'ACCESSES', 'EXTENDS'];
|
||||
|
||||
if (isPolymorphicNodesEnabled()) {
|
||||
// Polymorphic approach: Query by relationshipType
|
||||
console.log('📊 KuzuDB Verification - Relationship types (polymorphic):');
|
||||
|
||||
for (const relType of relTypes) {
|
||||
try {
|
||||
const result = await (this.kuzuGraph as any).executeQuery(`MATCH ()-[r:CodeRelationship {relationshipType: '${relType}'}]->() RETURN COUNT(r) as count`);
|
||||
const count = result.rows?.[0]?.[0] || 0;
|
||||
if (count > 0) {
|
||||
console.log(` ${relType}: ${count} relationships`);
|
||||
}
|
||||
} catch (error) {
|
||||
// Skip silently
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Traditional approach: Query individual tables
|
||||
console.log('📊 KuzuDB Verification - Relationship types:');
|
||||
|
||||
for (const relType of relTypes) {
|
||||
try {
|
||||
const result = await (this.kuzuGraph as any).executeQuery(`MATCH ()-[r:${relType}]->() RETURN COUNT(r) as count`);
|
||||
const count = result.rows?.[0]?.[0] || 0;
|
||||
if (count > 0) {
|
||||
console.log(` ${relType}: ${count} relationships`);
|
||||
}
|
||||
} catch (error) {
|
||||
// Relationship type might not exist in this database, skip silently
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Test query 5: Sample some actual data
|
||||
const sampleQuery = isPolymorphicNodesEnabled()
|
||||
? `MATCH (f:CodeElement {elementType: 'Function'}) RETURN f.name, f.filePath, f.startLine LIMIT 5`
|
||||
: `MATCH (f:Function) RETURN f.name, f.filePath, f.startLine LIMIT 5`;
|
||||
|
||||
const sampleResult = await (this.kuzuGraph as any).executeQuery(sampleQuery);
|
||||
|
||||
if (sampleResult.rows && sampleResult.rows.length > 0) {
|
||||
console.log('📊 KuzuDB Verification - Sample functions:');
|
||||
sampleResult.rows.forEach((row: any) => {
|
||||
console.log(` ${row[0]} (${row[1]}:${row[2]})`);
|
||||
});
|
||||
}
|
||||
|
||||
console.log('✅ KuzuDB verification completed successfully');
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ KuzuDB verification failed:', error);
|
||||
console.error('❌ This indicates the data may not be properly stored in KuzuDB');
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+35
-24
@@ -1,30 +1,41 @@
|
||||
/**
|
||||
* Graph interfaces and types
|
||||
*/
|
||||
import { GraphNode, GraphRelationship, KnowledgeGraph } from './types'
|
||||
|
||||
import { GraphNode, GraphRelationship } from './types.js';
|
||||
export const createKnowledgeGraph = (): KnowledgeGraph => {
|
||||
const nodeMap = new Map<string, GraphNode>();
|
||||
const relationshipMap = new Map<string, GraphRelationship>();
|
||||
|
||||
export interface KnowledgeGraph {
|
||||
nodes: GraphNode[];
|
||||
relationships: GraphRelationship[];
|
||||
addNode(node: GraphNode): void;
|
||||
addRelationship(relationship: GraphRelationship): void;
|
||||
}
|
||||
const addNode = (node: GraphNode) => {
|
||||
if(!nodeMap.has(node.id)) {
|
||||
nodeMap.set(node.id, node);
|
||||
}
|
||||
};
|
||||
|
||||
export interface GraphProcessor<T> {
|
||||
process(graph: KnowledgeGraph, input: T): Promise<void>;
|
||||
}
|
||||
const addRelationship = (relationship: GraphRelationship) => {
|
||||
if (!relationshipMap.has(relationship.id)) {
|
||||
relationshipMap.set(relationship.id, relationship);
|
||||
}
|
||||
};
|
||||
|
||||
// Simple implementation of KnowledgeGraph
|
||||
export class SimpleKnowledgeGraph implements KnowledgeGraph {
|
||||
nodes: GraphNode[] = [];
|
||||
relationships: GraphRelationship[] = [];
|
||||
return{
|
||||
get nodes(){
|
||||
return Array.from(nodeMap.values())
|
||||
},
|
||||
|
||||
get relationships(){
|
||||
return Array.from(relationshipMap.values())
|
||||
},
|
||||
|
||||
addNode(node: GraphNode): void {
|
||||
this.nodes.push(node);
|
||||
}
|
||||
// O(1) count getters - avoid creating arrays just for length
|
||||
get nodeCount() {
|
||||
return nodeMap.size;
|
||||
},
|
||||
|
||||
addRelationship(relationship: GraphRelationship): void {
|
||||
this.relationships.push(relationship);
|
||||
}
|
||||
}
|
||||
get relationshipCount() {
|
||||
return relationshipMap.size;
|
||||
},
|
||||
|
||||
addNode,
|
||||
addRelationship,
|
||||
|
||||
};
|
||||
};
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,683 +0,0 @@
|
||||
/**
|
||||
* KuzuDB Query Engine
|
||||
*
|
||||
* This module provides a high-level interface for executing queries
|
||||
* against KuzuDB, with support for graph data import, query optimization,
|
||||
* and performance monitoring.
|
||||
*/
|
||||
|
||||
import type { KnowledgeGraph, GraphNode, GraphRelationship } from './types.ts';
|
||||
import type { KuzuInstance, QueryResult } from '../kuzu/kuzu-loader.ts';
|
||||
import { initKuzuDB } from '../kuzu/kuzu-loader.ts';
|
||||
import { isKuzuDBEnabled, isKuzuDBPersistenceEnabled } from '../../config/features.ts';
|
||||
|
||||
export interface QueryOptions {
|
||||
timeout?: number;
|
||||
maxResults?: number;
|
||||
includeExecutionTime?: boolean;
|
||||
useCache?: boolean;
|
||||
}
|
||||
|
||||
export interface KuzuQueryResult extends QueryResult {
|
||||
nodes?: GraphNode[];
|
||||
relationships?: GraphRelationship[];
|
||||
resultCount: number;
|
||||
executionTime: number;
|
||||
fromCache?: boolean;
|
||||
}
|
||||
|
||||
export interface QueryCache {
|
||||
[queryHash: string]: {
|
||||
result: KuzuQueryResult;
|
||||
timestamp: number;
|
||||
hitCount: number;
|
||||
};
|
||||
}
|
||||
|
||||
export interface KuzuQueryEngineOptions {
|
||||
databasePath?: string;
|
||||
enableCache?: boolean;
|
||||
cacheSize?: number;
|
||||
cacheTTL?: number; // Time to live in milliseconds
|
||||
}
|
||||
|
||||
/**
|
||||
* High-level query engine for KuzuDB operations
|
||||
*/
|
||||
export class KuzuQueryEngine {
|
||||
private kuzuInstance: KuzuInstance | null = null;
|
||||
private isInitialized: boolean = false;
|
||||
private databasePath: string = '/gitnexus_db';
|
||||
private queryCache: QueryCache = {};
|
||||
private cacheEnabled: boolean = true;
|
||||
private maxCacheSize: number = 1000;
|
||||
private cacheTTL: number = 5 * 60 * 1000; // 5 minutes default
|
||||
private queryCount: number = 0;
|
||||
private totalExecutionTime: number = 0;
|
||||
|
||||
constructor(options: KuzuQueryEngineOptions = {}) {
|
||||
this.databasePath = options.databasePath || '/gitnexus_db';
|
||||
this.cacheEnabled = options.enableCache ?? true;
|
||||
this.maxCacheSize = options.cacheSize || 1000;
|
||||
this.cacheTTL = options.cacheTTL || 5 * 60 * 1000;
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize the KuzuDB query engine
|
||||
*/
|
||||
async initialize(): Promise<void> {
|
||||
if (this.isInitialized) {
|
||||
console.log('✅ KuzuQueryEngine already initialized');
|
||||
return;
|
||||
}
|
||||
|
||||
if (!isKuzuDBEnabled()) {
|
||||
console.log('⚠️ KuzuDB is disabled via feature flags');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
console.log('🚀 Initializing KuzuQueryEngine...');
|
||||
|
||||
// Initialize KuzuDB instance
|
||||
console.log('🔧 Step 1: Initializing KuzuDB instance...');
|
||||
this.kuzuInstance = await initKuzuDB();
|
||||
console.log('✅ Step 1 complete: KuzuDB instance initialized');
|
||||
|
||||
// Create database
|
||||
console.log('🔧 Step 2: Creating database...');
|
||||
await this.kuzuInstance.createDatabase(this.databasePath);
|
||||
console.log('✅ Step 2 complete: Database created');
|
||||
|
||||
// Initialize schema
|
||||
console.log('🔧 Step 3: Initializing schema...');
|
||||
await this.initializeSchema();
|
||||
console.log('✅ Step 3 complete: Schema initialized');
|
||||
|
||||
this.isInitialized = true;
|
||||
console.log('✅ KuzuQueryEngine initialized successfully');
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Failed to initialize KuzuQueryEngine:', error);
|
||||
console.error('❌ Error details:', error);
|
||||
throw new Error(`KuzuQueryEngine initialization failed: ${error instanceof Error ? error.message : 'Unknown error'}`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Import a knowledge graph into KuzuDB
|
||||
*/
|
||||
async importGraph(graph: KnowledgeGraph): Promise<void> {
|
||||
if (!this.isReady()) {
|
||||
throw new Error('KuzuQueryEngine not initialized. Call initialize() first.');
|
||||
}
|
||||
|
||||
try {
|
||||
console.log(`🔄 Importing graph with ${graph.nodes.length} nodes and ${graph.relationships.length} relationships...`);
|
||||
|
||||
const startTime = performance.now();
|
||||
|
||||
// Import nodes in batches
|
||||
await this.importNodes(graph.nodes);
|
||||
|
||||
// Import relationships in batches
|
||||
await this.importRelationships(graph.relationships);
|
||||
|
||||
const importTime = performance.now() - startTime;
|
||||
console.log(`✅ Graph import completed in ${importTime.toFixed(2)}ms`);
|
||||
|
||||
// Clear cache after import
|
||||
this.clearCache();
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Failed to import graph:', error);
|
||||
throw new Error(`Graph import failed: ${error instanceof Error ? error.message : 'Unknown error'}`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute a Cypher query with optional caching and optimization
|
||||
*/
|
||||
async executeQuery(cypher: string, options: QueryOptions = {}): Promise<KuzuQueryResult> {
|
||||
if (!this.isReady()) {
|
||||
throw new Error('KuzuQueryEngine not initialized. Call initialize() first.');
|
||||
}
|
||||
|
||||
const {
|
||||
timeout = 30000,
|
||||
maxResults = 1000,
|
||||
includeExecutionTime = true,
|
||||
useCache = this.cacheEnabled
|
||||
} = options;
|
||||
|
||||
try {
|
||||
// Check cache first
|
||||
if (useCache) {
|
||||
const cachedResult = this.getCachedResult(cypher);
|
||||
if (cachedResult) {
|
||||
return cachedResult;
|
||||
}
|
||||
}
|
||||
|
||||
// Execute query with timeout
|
||||
const queryPromise = this.executeQueryInternal(cypher, maxResults);
|
||||
const timeoutPromise = new Promise<never>((_, reject) => {
|
||||
setTimeout(() => reject(new Error('Query timeout')), timeout);
|
||||
});
|
||||
|
||||
const result = await Promise.race([queryPromise, timeoutPromise]);
|
||||
|
||||
// Cache successful results
|
||||
if (useCache && !result.error) {
|
||||
this.cacheResult(cypher, result);
|
||||
}
|
||||
|
||||
// Update statistics
|
||||
this.queryCount++;
|
||||
this.totalExecutionTime += result.executionTime;
|
||||
|
||||
return result;
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Query execution failed:', error);
|
||||
|
||||
// Re-throw the error so that calling code can handle it (e.g., auto-recovery)
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute a query and return graph nodes and relationships
|
||||
*/
|
||||
async executeGraphQuery(cypher: string, options: QueryOptions = {}): Promise<{
|
||||
nodes: GraphNode[];
|
||||
relationships: GraphRelationship[];
|
||||
executionTime: number;
|
||||
}> {
|
||||
const result = await this.executeQuery(cypher, options);
|
||||
|
||||
// Parse result into nodes and relationships
|
||||
const nodes: GraphNode[] = [];
|
||||
const relationships: GraphRelationship[] = [];
|
||||
|
||||
// This is a simplified parser - in practice, you'd need to parse
|
||||
// the actual KuzuDB result format to extract nodes and relationships
|
||||
for (const row of result.rows) {
|
||||
// TODO: Implement proper parsing based on KuzuDB result format
|
||||
// This is a placeholder implementation
|
||||
}
|
||||
|
||||
return {
|
||||
nodes,
|
||||
relationships,
|
||||
executionTime: result.executionTime
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Get query engine statistics
|
||||
*/
|
||||
getStatistics(): {
|
||||
isInitialized: boolean;
|
||||
queryCount: number;
|
||||
averageExecutionTime: number;
|
||||
cacheHitRate: number;
|
||||
cacheSize: number;
|
||||
} {
|
||||
const cacheHits = Object.values(this.queryCache).reduce((sum, entry) => sum + entry.hitCount, 0);
|
||||
const cacheHitRate = this.queryCount > 0 ? (cacheHits / this.queryCount) * 100 : 0;
|
||||
const averageExecutionTime = this.queryCount > 0 ? this.totalExecutionTime / this.queryCount : 0;
|
||||
|
||||
return {
|
||||
isInitialized: this.isInitialized,
|
||||
queryCount: this.queryCount,
|
||||
averageExecutionTime,
|
||||
cacheHitRate,
|
||||
cacheSize: Object.keys(this.queryCache).length
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Clear the query cache
|
||||
*/
|
||||
clearCache(): void {
|
||||
this.queryCache = {};
|
||||
console.log('✅ Query cache cleared');
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if the query engine is ready for operations
|
||||
*/
|
||||
isReady(): boolean {
|
||||
return this.isInitialized && this.kuzuInstance !== null && this.kuzuInstance.isReady();
|
||||
}
|
||||
|
||||
/**
|
||||
* Close the query engine and cleanup resources
|
||||
*/
|
||||
async close(): Promise<void> {
|
||||
try {
|
||||
if (this.kuzuInstance) {
|
||||
await this.kuzuInstance.closeDatabase();
|
||||
this.kuzuInstance = null;
|
||||
}
|
||||
|
||||
this.isInitialized = false;
|
||||
this.clearCache();
|
||||
|
||||
console.log('✅ KuzuQueryEngine closed successfully');
|
||||
} catch (error) {
|
||||
console.error('❌ Failed to close KuzuQueryEngine:', error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize the database schema with all required tables
|
||||
*/
|
||||
private async initializeSchema(): Promise<void> {
|
||||
if (!this.kuzuInstance) {
|
||||
throw new Error('KuzuDB instance not available');
|
||||
}
|
||||
|
||||
try {
|
||||
console.log('📋 Initializing KuzuDB schema...');
|
||||
|
||||
// Use KuzuSchemaManager for complete and correct schema
|
||||
const { KuzuSchemaManager } = await import('../kuzu/kuzu-schema.ts');
|
||||
const schemaManager = new KuzuSchemaManager(this.kuzuInstance);
|
||||
|
||||
await schemaManager.initializeSchema();
|
||||
|
||||
// NOTE: Removed outdated createNodeTables() and createRelationshipTables() methods
|
||||
// KuzuSchemaManager now handles all schema creation with complete definitions
|
||||
|
||||
console.log('✅ Schema initialization completed');
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Schema initialization failed:', error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Create all node tables
|
||||
*/
|
||||
private async createNodeTables(): Promise<void> {
|
||||
if (!this.kuzuInstance) return;
|
||||
|
||||
const nodeTables = [
|
||||
{
|
||||
name: 'Project',
|
||||
schema: {
|
||||
id: 'STRING',
|
||||
name: 'STRING',
|
||||
path: 'STRING',
|
||||
description: 'STRING',
|
||||
version: 'STRING',
|
||||
createdAt: 'STRING'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'Folder',
|
||||
schema: {
|
||||
id: 'STRING',
|
||||
name: 'STRING',
|
||||
path: 'STRING',
|
||||
fullPath: 'STRING',
|
||||
depth: 'INT64'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'File',
|
||||
schema: {
|
||||
id: 'STRING',
|
||||
name: 'STRING',
|
||||
path: 'STRING',
|
||||
filePath: 'STRING',
|
||||
extension: 'STRING',
|
||||
language: 'STRING',
|
||||
size: 'INT64',
|
||||
definitionCount: 'INT64',
|
||||
lineCount: 'INT64'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'Function',
|
||||
schema: {
|
||||
id: 'STRING',
|
||||
name: 'STRING',
|
||||
filePath: 'STRING',
|
||||
type: 'STRING',
|
||||
startLine: 'INT64',
|
||||
endLine: 'INT64',
|
||||
qualifiedName: 'STRING',
|
||||
parameters: 'STRING[]',
|
||||
returnType: 'STRING',
|
||||
accessibility: 'STRING',
|
||||
isStatic: 'BOOLEAN',
|
||||
isAsync: 'BOOLEAN',
|
||||
parentClass: 'STRING'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'Class',
|
||||
schema: {
|
||||
id: 'STRING',
|
||||
name: 'STRING',
|
||||
filePath: 'STRING',
|
||||
startLine: 'INT64',
|
||||
endLine: 'INT64',
|
||||
qualifiedName: 'STRING',
|
||||
accessibility: 'STRING',
|
||||
isAbstract: 'BOOLEAN',
|
||||
extends: 'STRING[]',
|
||||
implements: 'STRING[]'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'Method',
|
||||
schema: {
|
||||
id: 'STRING',
|
||||
name: 'STRING',
|
||||
filePath: 'STRING',
|
||||
startLine: 'INT64',
|
||||
endLine: 'INT64',
|
||||
qualifiedName: 'STRING',
|
||||
parameters: 'STRING[]',
|
||||
returnType: 'STRING',
|
||||
accessibility: 'STRING',
|
||||
isStatic: 'BOOLEAN',
|
||||
isAsync: 'BOOLEAN',
|
||||
parentClass: 'STRING'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'Variable',
|
||||
schema: {
|
||||
id: 'STRING',
|
||||
name: 'STRING',
|
||||
filePath: 'STRING',
|
||||
startLine: 'INT64',
|
||||
endLine: 'INT64',
|
||||
type: 'STRING',
|
||||
accessibility: 'STRING',
|
||||
isStatic: 'BOOLEAN'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'Interface',
|
||||
schema: {
|
||||
id: 'STRING',
|
||||
name: 'STRING',
|
||||
filePath: 'STRING',
|
||||
startLine: 'INT64',
|
||||
endLine: 'INT64',
|
||||
qualifiedName: 'STRING',
|
||||
extends: 'STRING[]'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'Type',
|
||||
schema: {
|
||||
id: 'STRING',
|
||||
name: 'STRING',
|
||||
filePath: 'STRING',
|
||||
startLine: 'INT64',
|
||||
endLine: 'INT64',
|
||||
qualifiedName: 'STRING',
|
||||
typeDefinition: 'STRING'
|
||||
}
|
||||
}
|
||||
];
|
||||
|
||||
for (const table of nodeTables) {
|
||||
try {
|
||||
await this.kuzuInstance.createNodeTable(table.name, table.schema);
|
||||
} catch (error) {
|
||||
// Table might already exist, which is fine
|
||||
console.log(`ℹ️ Node table ${table.name} might already exist`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Create all relationship tables
|
||||
*/
|
||||
private async createRelationshipTables(): Promise<void> {
|
||||
if (!this.kuzuInstance) return;
|
||||
|
||||
const relationshipTables = [
|
||||
{
|
||||
name: 'CONTAINS',
|
||||
from: 'Project',
|
||||
to: 'Folder',
|
||||
schema: {}
|
||||
},
|
||||
{
|
||||
name: 'CALLS',
|
||||
from: 'Function',
|
||||
to: 'Function',
|
||||
schema: {
|
||||
confidence: 'DOUBLE',
|
||||
callType: 'STRING',
|
||||
stage: 'STRING',
|
||||
distance: 'INT64'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'IMPORTS',
|
||||
from: 'File',
|
||||
to: 'File',
|
||||
schema: {
|
||||
importType: 'STRING',
|
||||
localName: 'STRING',
|
||||
exportedName: 'STRING'
|
||||
}
|
||||
},
|
||||
{
|
||||
name: 'INHERITS',
|
||||
from: 'Class',
|
||||
to: 'Class',
|
||||
schema: {
|
||||
inheritanceType: 'STRING'
|
||||
}
|
||||
}
|
||||
];
|
||||
|
||||
for (const table of relationshipTables) {
|
||||
try {
|
||||
await this.kuzuInstance.createRelTable(table.name, table.from, table.to, table.schema);
|
||||
} catch (error) {
|
||||
// Table might already exist, which is fine
|
||||
console.log(`ℹ️ Relationship table ${table.name} might already exist`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Import nodes in batches
|
||||
*/
|
||||
private async importNodes(nodes: GraphNode[]): Promise<void> {
|
||||
if (!this.kuzuInstance) return;
|
||||
|
||||
const batchSize = 100;
|
||||
const batches = Math.ceil(nodes.length / batchSize);
|
||||
|
||||
for (let i = 0; i < batches; i++) {
|
||||
const batch = nodes.slice(i * batchSize, (i + 1) * batchSize);
|
||||
|
||||
for (const node of batch) {
|
||||
try {
|
||||
await this.kuzuInstance.insertNode(node.label, {
|
||||
id: node.id,
|
||||
...node.properties
|
||||
});
|
||||
} catch (error) {
|
||||
console.warn(`Failed to insert node ${node.id}:`, error);
|
||||
}
|
||||
}
|
||||
|
||||
// Progress logging
|
||||
if (batches > 10 && i % Math.ceil(batches / 10) === 0) {
|
||||
console.log(`📊 Node import progress: ${Math.round((i / batches) * 100)}%`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Import relationships in batches
|
||||
*/
|
||||
private async importRelationships(relationships: GraphRelationship[]): Promise<void> {
|
||||
if (!this.kuzuInstance) return;
|
||||
|
||||
const batchSize = 100;
|
||||
const batches = Math.ceil(relationships.length / batchSize);
|
||||
|
||||
for (let i = 0; i < batches; i++) {
|
||||
const batch = relationships.slice(i * batchSize, (i + 1) * batchSize);
|
||||
|
||||
for (const rel of batch) {
|
||||
try {
|
||||
await this.kuzuInstance.insertRel(
|
||||
rel.type,
|
||||
rel.source,
|
||||
rel.target,
|
||||
rel.properties
|
||||
);
|
||||
} catch (error) {
|
||||
console.warn(`Failed to insert relationship ${rel.id}:`, error);
|
||||
}
|
||||
}
|
||||
|
||||
// Progress logging
|
||||
if (batches > 10 && i % Math.ceil(batches / 10) === 0) {
|
||||
console.log(`📊 Relationship import progress: ${Math.round((i / batches) * 100)}%`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute query internally with performance monitoring
|
||||
*/
|
||||
private async executeQueryInternal(cypher: string, maxResults: number): Promise<KuzuQueryResult> {
|
||||
if (!this.kuzuInstance) {
|
||||
throw new Error('KuzuDB instance not available');
|
||||
}
|
||||
|
||||
const startTime = performance.now();
|
||||
|
||||
try {
|
||||
const result = await this.kuzuInstance.executeQuery(cypher);
|
||||
const executionTime = performance.now() - startTime;
|
||||
|
||||
// Ensure rows is a proper array and limit results if specified
|
||||
const rowsArray = Array.isArray(result.rows) ? result.rows : Array.from(result.rows || []);
|
||||
const limitedRows = maxResults > 0 ? rowsArray.slice(0, maxResults) : rowsArray;
|
||||
|
||||
return {
|
||||
...result,
|
||||
rows: limitedRows,
|
||||
resultCount: result.rowCount,
|
||||
executionTime
|
||||
};
|
||||
} catch (error) {
|
||||
const executionTime = performance.now() - startTime;
|
||||
throw new Error(`Query execution failed after ${executionTime.toFixed(2)}ms: ${error instanceof Error ? error.message : 'Unknown error'}`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get cached query result if available and not expired
|
||||
*/
|
||||
private getCachedResult(cypher: string): KuzuQueryResult | null {
|
||||
if (!this.cacheEnabled) return null;
|
||||
|
||||
const queryHash = this.hashQuery(cypher);
|
||||
const cached = this.queryCache[queryHash];
|
||||
|
||||
if (!cached) return null;
|
||||
|
||||
// Check if cache entry has expired
|
||||
if (Date.now() - cached.timestamp > this.cacheTTL) {
|
||||
delete this.queryCache[queryHash];
|
||||
return null;
|
||||
}
|
||||
|
||||
// Update hit count
|
||||
cached.hitCount++;
|
||||
|
||||
return {
|
||||
...cached.result,
|
||||
fromCache: true
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Cache a query result
|
||||
*/
|
||||
private cacheResult(cypher: string, result: KuzuQueryResult): void {
|
||||
if (!this.cacheEnabled) return;
|
||||
|
||||
// Clean cache if it's getting too large
|
||||
if (Object.keys(this.queryCache).length >= this.maxCacheSize) {
|
||||
this.cleanCache();
|
||||
}
|
||||
|
||||
const queryHash = this.hashQuery(cypher);
|
||||
this.queryCache[queryHash] = {
|
||||
result: { ...result },
|
||||
timestamp: Date.now(),
|
||||
hitCount: 0
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Clean old cache entries
|
||||
*/
|
||||
private cleanCache(): void {
|
||||
const now = Date.now();
|
||||
const entries = Object.entries(this.queryCache);
|
||||
|
||||
// Remove expired entries
|
||||
entries.forEach(([hash, entry]) => {
|
||||
if (now - entry.timestamp > this.cacheTTL) {
|
||||
delete this.queryCache[hash];
|
||||
}
|
||||
});
|
||||
|
||||
// If still too large, remove least recently used entries
|
||||
const remainingEntries = Object.entries(this.queryCache);
|
||||
if (remainingEntries.length >= this.maxCacheSize) {
|
||||
remainingEntries
|
||||
.sort((a, b) => a[1].timestamp - b[1].timestamp)
|
||||
.slice(0, Math.floor(this.maxCacheSize * 0.2))
|
||||
.forEach(([hash]) => delete this.queryCache[hash]);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a hash for a query string
|
||||
*/
|
||||
private hashQuery(cypher: string): string {
|
||||
// Simple hash function for query caching
|
||||
let hash = 0;
|
||||
for (let i = 0; i < cypher.length; i++) {
|
||||
const char = cypher.charCodeAt(i);
|
||||
hash = ((hash << 5) - hash) + char;
|
||||
hash = hash & hash; // Convert to 32-bit integer
|
||||
}
|
||||
return hash.toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the KuzuDB instance for direct access (used by auto-recovery)
|
||||
*/
|
||||
getKuzuInstance(): KuzuInstance {
|
||||
if (!this.kuzuInstance) {
|
||||
throw new Error('KuzuDB instance not available. Call initialize() first.');
|
||||
}
|
||||
return this.kuzuInstance;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,503 +0,0 @@
|
||||
import type { KnowledgeGraph, GraphNode, GraphRelationship } from './types.ts';
|
||||
|
||||
export interface QueryResult {
|
||||
nodes: GraphNode[];
|
||||
relationships: GraphRelationship[];
|
||||
data: Record<string, any>[];
|
||||
}
|
||||
|
||||
export interface QueryOptions {
|
||||
limit?: number;
|
||||
offset?: number;
|
||||
}
|
||||
|
||||
export class GraphQueryEngine {
|
||||
constructor(private graph: KnowledgeGraph) {}
|
||||
|
||||
/**
|
||||
* Execute a simplified Cypher-like query against the knowledge graph
|
||||
*/
|
||||
public executeQuery(cypher: string, options: QueryOptions = {}): QueryResult {
|
||||
const { limit = 100, offset = 0 } = options;
|
||||
|
||||
try {
|
||||
const parsedQuery = this.parseCypher(cypher);
|
||||
|
||||
switch (parsedQuery.type) {
|
||||
case 'MATCH':
|
||||
return this.executeMatchQuery(parsedQuery, limit, offset);
|
||||
case 'MATCH_WHERE':
|
||||
return this.executeWhereQuery(parsedQuery, limit, offset);
|
||||
case 'MATCH_PATH':
|
||||
return this.executePathQuery(parsedQuery, limit, offset);
|
||||
case 'MATCH_AGGREGATION':
|
||||
return this.executeAggregationQuery(parsedQuery, limit, offset);
|
||||
case 'MATCH_RELATIONSHIP':
|
||||
return this.executeRelationshipQuery(parsedQuery, limit, offset);
|
||||
case 'COUNT_ALL':
|
||||
return this.executeCountAllQuery(parsedQuery);
|
||||
default:
|
||||
throw new Error(`Unsupported query type: ${parsedQuery.type}`);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('Query execution failed:', error);
|
||||
return { nodes: [], relationships: [], data: [] };
|
||||
}
|
||||
}
|
||||
|
||||
private parseCypher(cypher: string) {
|
||||
// Enhanced parsing with WHERE clause support
|
||||
const simpleMatchWithWherePattern = /MATCH\s+\((\w+):(\w+)(?:\s*\{([^}]*)\})?\)\s+WHERE\s+(.+?)\s+RETURN\s+(.+)/i;
|
||||
const whereMatch = cypher.match(simpleMatchWithWherePattern);
|
||||
|
||||
if (whereMatch) {
|
||||
const [, variable, label, properties, whereClause, returnClause] = whereMatch;
|
||||
return {
|
||||
type: 'MATCH_WHERE',
|
||||
variable,
|
||||
label,
|
||||
properties: this.parseProperties(properties || ''),
|
||||
whereClause: whereClause.trim(),
|
||||
returnClause: returnClause.trim()
|
||||
};
|
||||
}
|
||||
|
||||
// Variable-length relationship pattern: MATCH (a)-[:REL*1..3]->(b)
|
||||
const variableLengthPattern = /MATCH\s+\((\w+)(?::(\w+))?\)-\[:(\w+)\*(\d+)\.\.(\d+)\]->\((\w+)(?::(\w+))?\)\s+RETURN\s+(.+)/i;
|
||||
const varLenMatch = cypher.match(variableLengthPattern);
|
||||
|
||||
if (varLenMatch) {
|
||||
const [, sourceVar, sourceLabel, relType, minDepth, maxDepth, targetVar, targetLabel, returnClause] = varLenMatch;
|
||||
return {
|
||||
type: 'MATCH_PATH',
|
||||
sourceVar,
|
||||
sourceLabel,
|
||||
relationshipType: relType,
|
||||
minDepth: parseInt(minDepth),
|
||||
maxDepth: parseInt(maxDepth),
|
||||
targetVar,
|
||||
targetLabel,
|
||||
returnClause: returnClause.trim()
|
||||
};
|
||||
}
|
||||
|
||||
// Count all nodes pattern: MATCH (n) RETURN COUNT(n)
|
||||
const countAllPattern = /MATCH\s+\((\w+)\)\s+RETURN\s+COUNT\(\1\)(?:\s+as\s+(\w+))?(?:\s+LIMIT\s+\d+)?/i;
|
||||
const countAllMatch = cypher.match(countAllPattern);
|
||||
|
||||
if (countAllMatch) {
|
||||
const [, variable, alias] = countAllMatch;
|
||||
return {
|
||||
type: 'COUNT_ALL',
|
||||
variable,
|
||||
alias: alias || 'count',
|
||||
returnClause: `COUNT(${variable})`
|
||||
};
|
||||
}
|
||||
|
||||
// Aggregation pattern: MATCH (n:Label) RETURN COUNT(n)
|
||||
const aggregationPattern = /MATCH\s+\((\w+):(\w+)(?:\s*\{([^}]*)\})?\)\s+RETURN\s+(COUNT|COLLECT|AVG|SUM)\(([^)]+)\)/i;
|
||||
const aggMatch = cypher.match(aggregationPattern);
|
||||
|
||||
if (aggMatch) {
|
||||
const [, variable, label, properties, aggFunction, aggTarget] = aggMatch;
|
||||
return {
|
||||
type: 'MATCH_AGGREGATION',
|
||||
variable,
|
||||
label,
|
||||
properties: this.parseProperties(properties || ''),
|
||||
aggregationFunction: aggFunction.toUpperCase(),
|
||||
aggregationTarget: aggTarget.trim(),
|
||||
returnClause: `${aggFunction}(${aggTarget})`
|
||||
};
|
||||
}
|
||||
|
||||
// Pattern: MATCH (n:Label {property: 'value'}) RETURN n.property
|
||||
const simpleMatchPattern = /MATCH\s+\((\w+):(\w+)(?:\s*\{([^}]*)\})?\)\s+RETURN\s+(.+)/i;
|
||||
const simpleMatch = cypher.match(simpleMatchPattern);
|
||||
|
||||
if (simpleMatch) {
|
||||
const [, variable, label, properties, returnClause] = simpleMatch;
|
||||
return {
|
||||
type: 'MATCH',
|
||||
variable,
|
||||
label,
|
||||
properties: this.parseProperties(properties || ''),
|
||||
returnClause: returnClause.trim()
|
||||
};
|
||||
}
|
||||
|
||||
// Pattern: MATCH (a)-[:RELATIONSHIP]->(b:Label) RETURN a, b
|
||||
const relationshipPattern = /MATCH\s+\((\w+)(?::(\w+))?\)-\[:(\w+)\]->\((\w+)(?::(\w+))?\)\s+RETURN\s+(.+)/i;
|
||||
const relMatch = cypher.match(relationshipPattern);
|
||||
|
||||
if (relMatch) {
|
||||
const [, sourceVar, sourceLabel, relType, targetVar, targetLabel, returnClause] = relMatch;
|
||||
return {
|
||||
type: 'MATCH_RELATIONSHIP',
|
||||
sourceVar,
|
||||
sourceLabel,
|
||||
relationshipType: relType,
|
||||
targetVar,
|
||||
targetLabel,
|
||||
returnClause: returnClause.trim()
|
||||
};
|
||||
}
|
||||
|
||||
throw new Error(`Cannot parse Cypher query: ${cypher}`);
|
||||
}
|
||||
|
||||
private parseProperties(propString: string): Record<string, string> {
|
||||
const props: Record<string, string> = {};
|
||||
if (!propString.trim()) return props;
|
||||
|
||||
// Simple property parsing: name: 'value', type: 'Function'
|
||||
const matches = propString.match(/(\w+):\s*['"]([^'"]*)['"]/g);
|
||||
if (matches) {
|
||||
matches.forEach(match => {
|
||||
const [, key, value] = match.match(/(\w+):\s*['"]([^'"]*)['"]/!) || [];
|
||||
if (key && value) {
|
||||
props[key] = value;
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
return props;
|
||||
}
|
||||
|
||||
private executeMatchQuery(query: any, limit: number, offset: number): QueryResult {
|
||||
let matchingNodes = this.graph.nodes.filter(node => {
|
||||
// Match by label
|
||||
if (query.label && node.label !== query.label) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Match by properties
|
||||
for (const [key, value] of Object.entries(query.properties)) {
|
||||
if (node.properties[key] !== value) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
});
|
||||
|
||||
// Apply pagination
|
||||
matchingNodes = matchingNodes.slice(offset, offset + limit);
|
||||
|
||||
// Process return clause
|
||||
const data = matchingNodes.map(node => {
|
||||
const result: Record<string, any> = {};
|
||||
|
||||
if (query.returnClause.includes(`${query.variable}.`)) {
|
||||
// Return specific properties: n.name, n.filePath
|
||||
const propertyMatches = query.returnClause.match(new RegExp(`${query.variable}\\.(\\w+)`, 'g'));
|
||||
if (propertyMatches) {
|
||||
propertyMatches.forEach((match: string) => {
|
||||
const prop = match.split('.')[1];
|
||||
result[prop] = node.properties[prop];
|
||||
});
|
||||
}
|
||||
} else if (query.returnClause === query.variable) {
|
||||
// Return entire node
|
||||
result.node = node;
|
||||
}
|
||||
|
||||
return result;
|
||||
});
|
||||
|
||||
return {
|
||||
nodes: matchingNodes,
|
||||
relationships: [],
|
||||
data
|
||||
};
|
||||
}
|
||||
|
||||
private executeWhereQuery(query: any, limit: number, offset: number): QueryResult {
|
||||
let matchingNodes = this.graph.nodes.filter(node => {
|
||||
if (query.label && node.label !== query.label) return false;
|
||||
|
||||
for (const [key, value] of Object.entries(query.properties)) {
|
||||
if (node.properties[key] !== value) return false;
|
||||
}
|
||||
|
||||
return this.evaluateWhereClause(node, query.whereClause, query.variable);
|
||||
});
|
||||
|
||||
matchingNodes = matchingNodes.slice(offset, offset + limit);
|
||||
|
||||
const data = matchingNodes.map(node => {
|
||||
const result: Record<string, any> = {};
|
||||
if (query.returnClause.includes(`${query.variable}.`)) {
|
||||
const propertyMatches = query.returnClause.match(new RegExp(`${query.variable}\\.(\\w+)`, 'g'));
|
||||
if (propertyMatches) {
|
||||
propertyMatches.forEach((match: string) => {
|
||||
const prop = match.split('.')[1];
|
||||
result[prop] = node.properties[prop];
|
||||
});
|
||||
}
|
||||
} else if (query.returnClause === query.variable) {
|
||||
result.node = node;
|
||||
}
|
||||
return result;
|
||||
});
|
||||
|
||||
return { nodes: matchingNodes, relationships: [], data };
|
||||
}
|
||||
|
||||
private evaluateWhereClause(node: GraphNode, whereClause: string, variable: string): boolean {
|
||||
// Simple WHERE clause evaluation
|
||||
const containsPattern = new RegExp(`${variable}\\.(\\w+)\\s+CONTAINS\\s+['"]([^'"]*)['"]/i`);
|
||||
const containsMatch = whereClause.match(containsPattern);
|
||||
|
||||
if (containsMatch) {
|
||||
const [, property, value] = containsMatch;
|
||||
const nodeValue = node.properties[property];
|
||||
return typeof nodeValue === 'string' && nodeValue.toLowerCase().includes(value.toLowerCase());
|
||||
}
|
||||
|
||||
const equalsPattern = new RegExp(`${variable}\\.(\\w+)\\s*=\\s*['"]([^'"]*)['"]/i`);
|
||||
const equalsMatch = whereClause.match(equalsPattern);
|
||||
|
||||
if (equalsMatch) {
|
||||
const [, property, value] = equalsMatch;
|
||||
return node.properties[property] === value;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
private executePathQuery(query: any, limit: number, offset: number): QueryResult {
|
||||
// Use existing pathsBetween function from query.ts
|
||||
const sourceNodes = this.graph.nodes.filter(n =>
|
||||
!query.sourceLabel || n.label === query.sourceLabel
|
||||
);
|
||||
const targetNodes = this.graph.nodes.filter(n =>
|
||||
!query.targetLabel || n.label === query.targetLabel
|
||||
);
|
||||
|
||||
const allResults: { source: GraphNode; target: GraphNode; path: GraphRelationship[] }[] = [];
|
||||
|
||||
for (const source of sourceNodes.slice(0, 50)) { // Limit source nodes to avoid explosion
|
||||
for (const target of targetNodes.slice(0, 50)) {
|
||||
if (source.id === target.id) continue;
|
||||
|
||||
const paths = this.findPaths(source.id, target.id, query.relationshipType, query.minDepth, query.maxDepth);
|
||||
paths.forEach(path => {
|
||||
allResults.push({ source, target, path });
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const paginatedResults = allResults.slice(offset, offset + limit);
|
||||
|
||||
const data = paginatedResults.map(({ source, target, path }) => {
|
||||
const result: Record<string, any> = {};
|
||||
if (query.returnClause.includes(query.sourceVar)) {
|
||||
result[query.sourceVar] = source;
|
||||
}
|
||||
if (query.returnClause.includes(query.targetVar)) {
|
||||
result[query.targetVar] = target;
|
||||
}
|
||||
result.pathLength = path.length;
|
||||
return result;
|
||||
});
|
||||
|
||||
return {
|
||||
nodes: paginatedResults.flatMap(r => [r.source, r.target]),
|
||||
relationships: paginatedResults.flatMap(r => r.path),
|
||||
data
|
||||
};
|
||||
}
|
||||
|
||||
private findPaths(sourceId: string, targetId: string, relType: string, minDepth: number, maxDepth: number): GraphRelationship[][] {
|
||||
const paths: GraphRelationship[][] = [];
|
||||
const visited = new Set<string>();
|
||||
|
||||
const dfs = (currentId: string, currentPath: GraphRelationship[], depth: number) => {
|
||||
if (depth > maxDepth) return;
|
||||
if (currentId === targetId && depth >= minDepth) {
|
||||
paths.push([...currentPath]);
|
||||
return;
|
||||
}
|
||||
|
||||
visited.add(currentId);
|
||||
|
||||
const outgoingRels = this.graph.relationships.filter(r =>
|
||||
r.source === currentId && r.type === relType && !visited.has(r.target)
|
||||
);
|
||||
|
||||
for (const rel of outgoingRels) {
|
||||
dfs(rel.target, [...currentPath, rel], depth + 1);
|
||||
}
|
||||
|
||||
visited.delete(currentId);
|
||||
};
|
||||
|
||||
dfs(sourceId, [], 0);
|
||||
return paths.slice(0, 10); // Limit paths to prevent explosion
|
||||
}
|
||||
|
||||
private executeAggregationQuery(query: any, _limit: number, _offset: number): QueryResult {
|
||||
let matchingNodes = this.graph.nodes.filter(node => {
|
||||
if (query.label && node.label !== query.label) return false;
|
||||
|
||||
for (const [key, value] of Object.entries(query.properties)) {
|
||||
if (node.properties[key] !== value) return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
});
|
||||
|
||||
let aggregatedValue: any;
|
||||
|
||||
switch (query.aggregationFunction) {
|
||||
case 'COUNT':
|
||||
aggregatedValue = matchingNodes.length;
|
||||
break;
|
||||
case 'COLLECT':
|
||||
const targetProperty = query.aggregationTarget.includes('.')
|
||||
? query.aggregationTarget.split('.')[1]
|
||||
: 'name';
|
||||
aggregatedValue = matchingNodes.map(n => n.properties[targetProperty]).filter(Boolean);
|
||||
break;
|
||||
default:
|
||||
aggregatedValue = matchingNodes.length;
|
||||
}
|
||||
|
||||
return {
|
||||
nodes: [],
|
||||
relationships: [],
|
||||
data: [{ [query.returnClause]: aggregatedValue }]
|
||||
};
|
||||
}
|
||||
|
||||
private executeRelationshipQuery(query: any, limit: number, offset: number): QueryResult {
|
||||
const results: { source: GraphNode; target: GraphNode; relationship: GraphRelationship }[] = [];
|
||||
|
||||
// Find all relationships of the specified type
|
||||
const matchingRels = this.graph.relationships.filter(rel =>
|
||||
rel.type === query.relationshipType
|
||||
);
|
||||
|
||||
for (const rel of matchingRels) {
|
||||
const sourceNode = this.graph.nodes.find(n => n.id === rel.source);
|
||||
const targetNode = this.graph.nodes.find(n => n.id === rel.target);
|
||||
|
||||
if (!sourceNode || !targetNode) continue;
|
||||
|
||||
// Apply label filters
|
||||
if (query.sourceLabel && sourceNode.label !== query.sourceLabel) continue;
|
||||
if (query.targetLabel && targetNode.label !== query.targetLabel) continue;
|
||||
|
||||
results.push({ source: sourceNode, target: targetNode, relationship: rel });
|
||||
}
|
||||
|
||||
// Apply pagination
|
||||
const paginatedResults = results.slice(offset, offset + limit);
|
||||
|
||||
// Process return clause
|
||||
const data = paginatedResults.map(({ source, target }) => {
|
||||
const result: Record<string, any> = {};
|
||||
|
||||
if (query.returnClause.includes(query.sourceVar)) {
|
||||
result[query.sourceVar] = source;
|
||||
}
|
||||
if (query.returnClause.includes(query.targetVar)) {
|
||||
result[query.targetVar] = target;
|
||||
}
|
||||
|
||||
return result;
|
||||
});
|
||||
|
||||
return {
|
||||
nodes: paginatedResults.flatMap(r => [r.source, r.target]),
|
||||
relationships: paginatedResults.map(r => r.relationship),
|
||||
data
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Test the query engine with sample queries (for development/debugging)
|
||||
*/
|
||||
public testQueries(): { query: string; result: any; success: boolean }[] {
|
||||
const testCases = [
|
||||
"MATCH (f:Function) RETURN COUNT(f)",
|
||||
"MATCH (c:Class) WHERE c.name CONTAINS 'Service' RETURN c.name",
|
||||
"MATCH (a:Function)-[:CALLS*1..2]->(b:Function) RETURN a.name, b.name",
|
||||
"MATCH (f:File)-[:CONTAINS]->(c:Class) RETURN f.name, COUNT(c)",
|
||||
"MATCH (m:Method) WHERE m.name CONTAINS 'get' RETURN COLLECT(m.name)"
|
||||
];
|
||||
|
||||
return testCases.map(query => {
|
||||
try {
|
||||
const result = this.executeQuery(query, { limit: 5 });
|
||||
return { query, result, success: true };
|
||||
} catch (error) {
|
||||
return {
|
||||
query,
|
||||
result: error instanceof Error ? error.message : 'Unknown error',
|
||||
success: false
|
||||
};
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Get query statistics
|
||||
*/
|
||||
public getStats(): { nodeCount: number; relationshipCount: number; nodeTypes: string[]; relationshipTypes: string[] } {
|
||||
const nodeTypes = [...new Set(this.graph.nodes.map(n => n.label))];
|
||||
const relationshipTypes = [...new Set(this.graph.relationships.map(r => r.type))];
|
||||
|
||||
return {
|
||||
nodeCount: this.graph.nodes.length,
|
||||
relationshipCount: this.graph.relationships.length,
|
||||
nodeTypes,
|
||||
relationshipTypes
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Find nodes by text search
|
||||
*/
|
||||
public searchNodes(searchTerm: string, nodeType?: string): GraphNode[] {
|
||||
const lowerSearch = searchTerm.toLowerCase();
|
||||
|
||||
return this.graph.nodes.filter(node => {
|
||||
if (nodeType && node.label !== nodeType) return false;
|
||||
|
||||
// Search in node properties
|
||||
const searchableText = [
|
||||
node.properties.name,
|
||||
node.properties.filePath,
|
||||
node.properties.qualifiedName
|
||||
].filter(Boolean).join(' ').toLowerCase();
|
||||
|
||||
return searchableText.includes(lowerSearch);
|
||||
});
|
||||
}
|
||||
|
||||
private executeCountAllQuery(query: any): QueryResult {
|
||||
const totalCount = this.graph.nodes.length;
|
||||
|
||||
const result: Record<string, any> = {};
|
||||
result[query.alias] = totalCount;
|
||||
|
||||
return {
|
||||
nodes: [],
|
||||
relationships: [],
|
||||
data: [result]
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Get all relationships for a node
|
||||
*/
|
||||
public getNodeRelationships(nodeId: string): { incoming: GraphRelationship[]; outgoing: GraphRelationship[] } {
|
||||
const incoming = this.graph.relationships.filter(r => r.target === nodeId);
|
||||
const outgoing = this.graph.relationships.filter(r => r.source === nodeId);
|
||||
|
||||
return { incoming, outgoing };
|
||||
}
|
||||
}
|
||||
@@ -1,234 +0,0 @@
|
||||
import type { KnowledgeGraph, GraphNode, GraphRelationship } from './types.ts';
|
||||
|
||||
export type NodeFilter = {
|
||||
idEquals?: string | string[];
|
||||
labelIn?: Array<GraphNode['label']>;
|
||||
nameContains?: string;
|
||||
pathContains?: string;
|
||||
props?: Record<string, unknown>;
|
||||
};
|
||||
|
||||
export type RelFilter = {
|
||||
typeIn?: Array<GraphRelationship['type']>;
|
||||
fromIdIn?: string[];
|
||||
toIdIn?: string[];
|
||||
};
|
||||
|
||||
export type TraverseOptions = {
|
||||
depth?: number;
|
||||
direction?: 'out' | 'in' | 'both';
|
||||
relTypeIn?: Array<GraphRelationship['type']>;
|
||||
limitNodes?: number;
|
||||
};
|
||||
|
||||
export type PathOptions = {
|
||||
relTypeIn?: Array<GraphRelationship['type']>;
|
||||
maxDepth?: number;
|
||||
maxPaths?: number;
|
||||
};
|
||||
|
||||
export interface GraphIndex {
|
||||
nodeById: Map<string, GraphNode>;
|
||||
outAdjacency: Map<string, GraphRelationship[]>;
|
||||
inAdjacency: Map<string, GraphRelationship[]>;
|
||||
}
|
||||
|
||||
export function indexGraph(graph: KnowledgeGraph): GraphIndex {
|
||||
const nodeById = new Map<string, GraphNode>();
|
||||
const outAdjacency = new Map<string, GraphRelationship[]>();
|
||||
const inAdjacency = new Map<string, GraphRelationship[]>();
|
||||
|
||||
for (const node of graph.nodes) {
|
||||
nodeById.set(node.id, node);
|
||||
}
|
||||
|
||||
for (const rel of graph.relationships) {
|
||||
if (!outAdjacency.has(rel.source)) outAdjacency.set(rel.source, []);
|
||||
if (!inAdjacency.has(rel.target)) inAdjacency.set(rel.target, []);
|
||||
outAdjacency.get(rel.source)!.push(rel);
|
||||
inAdjacency.get(rel.target)!.push(rel);
|
||||
}
|
||||
|
||||
return { nodeById, outAdjacency, inAdjacency };
|
||||
}
|
||||
|
||||
export function queryNodes(graph: KnowledgeGraph, filter: NodeFilter): GraphNode[] {
|
||||
const idSet = new Set(typeof filter.idEquals === 'string' ? [filter.idEquals] : filter.idEquals || []);
|
||||
const nameNeedle = filter.nameContains?.toLowerCase();
|
||||
const pathNeedle = filter.pathContains?.toLowerCase();
|
||||
|
||||
return graph.nodes.filter((node) => {
|
||||
if (idSet.size > 0 && !idSet.has(node.id)) return false;
|
||||
if (filter.labelIn && filter.labelIn.length > 0 && !filter.labelIn.includes(node.label)) return false;
|
||||
|
||||
if (nameNeedle) {
|
||||
const name = String(node.properties?.name || '').toLowerCase();
|
||||
if (!name.includes(nameNeedle)) return false;
|
||||
}
|
||||
|
||||
if (pathNeedle) {
|
||||
const path = String(node.properties?.path || node.properties?.filePath || '').toLowerCase();
|
||||
if (!path.includes(pathNeedle)) return false;
|
||||
}
|
||||
|
||||
if (filter.props) {
|
||||
for (const [key, expected] of Object.entries(filter.props)) {
|
||||
if ((node.properties as Record<string, unknown>)?.[key] !== expected) return false;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
export function queryRelationships(graph: KnowledgeGraph, filter: RelFilter): GraphRelationship[] {
|
||||
const typeSet = new Set(filter.typeIn || []);
|
||||
const fromSet = new Set(filter.fromIdIn || []);
|
||||
const toSet = new Set(filter.toIdIn || []);
|
||||
|
||||
return graph.relationships.filter((rel) => {
|
||||
if (typeSet.size > 0 && !typeSet.has(rel.type)) return false;
|
||||
if (fromSet.size > 0 && !fromSet.has(rel.source)) return false;
|
||||
if (toSet.size > 0 && !toSet.has(rel.target)) return false;
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
export function neighborsOf(
|
||||
graph: KnowledgeGraph,
|
||||
seedNodeIds: string[],
|
||||
options: TraverseOptions = {}
|
||||
): KnowledgeGraph {
|
||||
const { depth = 1, direction = 'both', relTypeIn, limitNodes = 500 } = options;
|
||||
const { outAdjacency, inAdjacency, nodeById } = indexGraph(graph);
|
||||
|
||||
const allowedRelTypes = relTypeIn ? new Set(relTypeIn) : null;
|
||||
const visitedNodes = new Set<string>(seedNodeIds);
|
||||
const resultNodes = new Set<string>(seedNodeIds);
|
||||
const resultRels: GraphRelationship[] = [];
|
||||
|
||||
let frontier = [...seedNodeIds];
|
||||
let currentDepth = 0;
|
||||
|
||||
while (frontier.length > 0 && currentDepth < depth && resultNodes.size < limitNodes) {
|
||||
const nextFrontier: string[] = [];
|
||||
for (const nodeId of frontier) {
|
||||
if (direction === 'out' || direction === 'both') {
|
||||
const outRels = outAdjacency.get(nodeId) || [];
|
||||
for (const rel of outRels) {
|
||||
if (allowedRelTypes && !allowedRelTypes.has(rel.type)) continue;
|
||||
resultRels.push(rel);
|
||||
if (!visitedNodes.has(rel.target)) {
|
||||
visitedNodes.add(rel.target);
|
||||
resultNodes.add(rel.target);
|
||||
nextFrontier.push(rel.target);
|
||||
if (resultNodes.size >= limitNodes) break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (resultNodes.size >= limitNodes) break;
|
||||
if (direction === 'in' || direction === 'both') {
|
||||
const inRels = inAdjacency.get(nodeId) || [];
|
||||
for (const rel of inRels) {
|
||||
if (allowedRelTypes && !allowedRelTypes.has(rel.type)) continue;
|
||||
resultRels.push(rel);
|
||||
if (!visitedNodes.has(rel.source)) {
|
||||
visitedNodes.add(rel.source);
|
||||
resultNodes.add(rel.source);
|
||||
nextFrontier.push(rel.source);
|
||||
if (resultNodes.size >= limitNodes) break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (resultNodes.size >= limitNodes) break;
|
||||
}
|
||||
frontier = nextFrontier;
|
||||
currentDepth++;
|
||||
}
|
||||
|
||||
const nodes = Array.from(resultNodes).map((id) => nodeById.get(id)!).filter(Boolean);
|
||||
const relationships = dedupeRelationships(resultRels);
|
||||
return { nodes, relationships };
|
||||
}
|
||||
|
||||
export function pathsBetween(
|
||||
graph: KnowledgeGraph,
|
||||
fromId: string,
|
||||
toId: string,
|
||||
options: PathOptions = {}
|
||||
): KnowledgeGraph {
|
||||
const { maxDepth = 6, relTypeIn, maxPaths = 3 } = options;
|
||||
const { outAdjacency, nodeById } = indexGraph(graph);
|
||||
const allowedRelTypes = relTypeIn ? new Set(relTypeIn) : null;
|
||||
|
||||
const queue: Array<{ nodeId: string; path: GraphRelationship[] }> = [{ nodeId: fromId, path: [] }];
|
||||
const visited = new Set<string>([fromId]);
|
||||
const foundPaths: GraphRelationship[][] = [];
|
||||
|
||||
while (queue.length > 0 && foundPaths.length < maxPaths) {
|
||||
const { nodeId, path } = queue.shift()!;
|
||||
if (path.length > maxDepth) continue;
|
||||
if (nodeId === toId) {
|
||||
foundPaths.push(path);
|
||||
continue;
|
||||
}
|
||||
const outRels = outAdjacency.get(nodeId) || [];
|
||||
for (const rel of outRels) {
|
||||
if (allowedRelTypes && !allowedRelTypes.has(rel.type)) continue;
|
||||
const nextId = rel.target;
|
||||
if (!visited.has(nextId)) {
|
||||
visited.add(nextId);
|
||||
queue.push({ nodeId: nextId, path: [...path, rel] });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const relSet = new Set<GraphRelationship>();
|
||||
for (const p of foundPaths) for (const r of p) relSet.add(r);
|
||||
const nodeSet = new Set<string>();
|
||||
for (const r of relSet) {
|
||||
nodeSet.add(r.source);
|
||||
nodeSet.add(r.target);
|
||||
}
|
||||
|
||||
const nodes = Array.from(nodeSet).map((id) => nodeById.get(id)!).filter(Boolean);
|
||||
const relationships = dedupeRelationships(Array.from(relSet));
|
||||
return { nodes, relationships };
|
||||
}
|
||||
|
||||
export function subgraphFromNodesAndRels(graph: KnowledgeGraph, nodeIds: string[], rels: GraphRelationship[]): KnowledgeGraph {
|
||||
const nodeIdSet = new Set(nodeIds);
|
||||
const nodes = graph.nodes.filter((n) => nodeIdSet.has(n.id));
|
||||
const relationships = dedupeRelationships(rels);
|
||||
return { nodes, relationships };
|
||||
}
|
||||
|
||||
export function summarizeSubgraph(subgraph: KnowledgeGraph): string {
|
||||
const counts = subgraph.nodes.reduce<Record<string, number>>((acc, n) => {
|
||||
acc[n.label] = (acc[n.label] || 0) + 1;
|
||||
return acc;
|
||||
}, {});
|
||||
const relCounts = subgraph.relationships.reduce<Record<string, number>>((acc, r) => {
|
||||
acc[r.type] = (acc[r.type] || 0) + 1;
|
||||
return acc;
|
||||
}, {});
|
||||
const parts: string[] = [];
|
||||
parts.push(`Nodes: ${subgraph.nodes.length} (${Object.entries(counts).map(([k, v]) => `${k}:${v}`).join(', ') || 'none'})`);
|
||||
parts.push(`Relationships: ${subgraph.relationships.length} (${Object.entries(relCounts).map(([k, v]) => `${k}:${v}`).join(', ') || 'none'})`);
|
||||
const examples = subgraph.nodes.slice(0, 5).map((n) => `${n.label}:${String(n.properties?.name || n.id)}`);
|
||||
if (examples.length > 0) parts.push(`Examples: ${examples.join(', ')}`);
|
||||
return parts.join('\n');
|
||||
}
|
||||
|
||||
function dedupeRelationships(rels: GraphRelationship[]): GraphRelationship[] {
|
||||
const seen = new Set<string>();
|
||||
const out: GraphRelationship[] = [];
|
||||
for (const r of rels) {
|
||||
const key = `${r.type}|${r.source}|${r.target}`;
|
||||
if (!seen.has(key)) {
|
||||
seen.add(key);
|
||||
out.push(r);
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
@@ -1,257 +0,0 @@
|
||||
interface TrieNode {
|
||||
children: Map<string, TrieNode>;
|
||||
definitions: FunctionDefinition[];
|
||||
isEndOfWord: boolean;
|
||||
}
|
||||
|
||||
interface FunctionDefinition {
|
||||
nodeId: string;
|
||||
qualifiedName: string;
|
||||
filePath: string;
|
||||
functionName: string;
|
||||
type: 'function' | 'method' | 'class' | 'interface' | 'enum';
|
||||
startLine?: number;
|
||||
endLine?: number;
|
||||
}
|
||||
|
||||
export class FunctionRegistryTrie {
|
||||
private root: TrieNode;
|
||||
private allDefinitions: Map<string, FunctionDefinition>;
|
||||
// Performance optimization: Index for fast lookups
|
||||
private functionNameIndex: Map<string, FunctionDefinition[]>;
|
||||
private filePathIndex: Map<string, FunctionDefinition[]>;
|
||||
|
||||
constructor() {
|
||||
this.root = {
|
||||
children: new Map(),
|
||||
definitions: [],
|
||||
isEndOfWord: false
|
||||
};
|
||||
this.allDefinitions = new Map();
|
||||
this.functionNameIndex = new Map();
|
||||
this.filePathIndex = new Map();
|
||||
}
|
||||
|
||||
/**
|
||||
* Add a function definition to the trie with optimized indexing
|
||||
* @param definition The function definition to add
|
||||
*/
|
||||
addDefinition(definition: FunctionDefinition): void {
|
||||
const parts = definition.qualifiedName.split('.');
|
||||
let currentNode = this.root;
|
||||
|
||||
// Build the trie path
|
||||
for (const part of parts) {
|
||||
if (!currentNode.children.has(part)) {
|
||||
currentNode.children.set(part, {
|
||||
children: new Map(),
|
||||
definitions: [],
|
||||
isEndOfWord: false
|
||||
});
|
||||
}
|
||||
currentNode = currentNode.children.get(part)!;
|
||||
}
|
||||
|
||||
// Mark end of word and store definition
|
||||
currentNode.isEndOfWord = true;
|
||||
currentNode.definitions.push(definition);
|
||||
|
||||
// Store in flat map for quick access
|
||||
this.allDefinitions.set(definition.nodeId, definition);
|
||||
|
||||
// Update indexes for performance
|
||||
this.updateIndexes(definition);
|
||||
}
|
||||
|
||||
/**
|
||||
* Update performance indexes when adding definitions
|
||||
*/
|
||||
private updateIndexes(definition: FunctionDefinition): void {
|
||||
// Function name index
|
||||
if (!this.functionNameIndex.has(definition.functionName)) {
|
||||
this.functionNameIndex.set(definition.functionName, []);
|
||||
}
|
||||
this.functionNameIndex.get(definition.functionName)!.push(definition);
|
||||
|
||||
// File path index
|
||||
if (!this.filePathIndex.has(definition.filePath)) {
|
||||
this.filePathIndex.set(definition.filePath, []);
|
||||
}
|
||||
this.filePathIndex.get(definition.filePath)!.push(definition);
|
||||
}
|
||||
|
||||
/**
|
||||
* Find all definitions that end with the given name (OPTIMIZED)
|
||||
* This is the key method for heuristic resolution
|
||||
* @param name The function name to search for
|
||||
* @returns Array of matching definitions
|
||||
*/
|
||||
findEndingWith(name: string): FunctionDefinition[] {
|
||||
// Use index for O(1) lookup instead of O(n) tree traversal
|
||||
return this.functionNameIndex.get(name) || [];
|
||||
}
|
||||
|
||||
/**
|
||||
* Get exact definition by qualified name
|
||||
* @param qualifiedName The full qualified name
|
||||
* @returns The definition if found
|
||||
*/
|
||||
getExactMatch(qualifiedName: string): FunctionDefinition[] {
|
||||
const parts = qualifiedName.split('.');
|
||||
let currentNode = this.root;
|
||||
|
||||
for (const part of parts) {
|
||||
if (!currentNode.children.has(part)) {
|
||||
return [];
|
||||
}
|
||||
currentNode = currentNode.children.get(part)!;
|
||||
}
|
||||
|
||||
return currentNode.isEndOfWord ? currentNode.definitions : [];
|
||||
}
|
||||
|
||||
/**
|
||||
* Find definitions in the same file (OPTIMIZED)
|
||||
* @param filePath The file path to search in
|
||||
* @param functionName The function name to find
|
||||
* @returns Array of matching definitions in the same file
|
||||
*/
|
||||
findInSameFile(filePath: string, functionName: string): FunctionDefinition[] {
|
||||
// Use file path index for faster lookup
|
||||
const fileDefinitions = this.filePathIndex.get(filePath) || [];
|
||||
return fileDefinitions.filter(def => def.functionName === functionName);
|
||||
}
|
||||
|
||||
/**
|
||||
* Get all definitions in a specific file (NEW - OPTIMIZED)
|
||||
* @param filePath The file path to search in
|
||||
* @returns Array of all definitions in the file
|
||||
*/
|
||||
getDefinitionsInFile(filePath: string): FunctionDefinition[] {
|
||||
return this.filePathIndex.get(filePath) || [];
|
||||
}
|
||||
|
||||
/**
|
||||
* Find definitions by type (NEW - OPTIMIZED)
|
||||
* @param type The definition type to search for
|
||||
* @returns Array of matching definitions
|
||||
*/
|
||||
findByType(type: FunctionDefinition['type']): FunctionDefinition[] {
|
||||
return Array.from(this.allDefinitions.values()).filter(def => def.type === type);
|
||||
}
|
||||
|
||||
/**
|
||||
* Get all definitions for debugging/stats
|
||||
* @returns All stored definitions
|
||||
*/
|
||||
getAllDefinitions(): FunctionDefinition[] {
|
||||
return Array.from(this.allDefinitions.values());
|
||||
}
|
||||
|
||||
/**
|
||||
* Get statistics about the trie (NEW)
|
||||
* @returns Trie statistics
|
||||
*/
|
||||
getStatistics(): {
|
||||
totalDefinitions: number;
|
||||
definitionsByType: Record<string, number>;
|
||||
fileCount: number;
|
||||
uniqueFunctionNames: number;
|
||||
} {
|
||||
const definitionsByType: Record<string, number> = {};
|
||||
|
||||
for (const definition of this.allDefinitions.values()) {
|
||||
definitionsByType[definition.type] = (definitionsByType[definition.type] || 0) + 1;
|
||||
}
|
||||
|
||||
return {
|
||||
totalDefinitions: this.allDefinitions.size,
|
||||
definitionsByType,
|
||||
fileCount: this.filePathIndex.size,
|
||||
uniqueFunctionNames: this.functionNameIndex.size
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Calculate import distance between two file paths
|
||||
* Lower score = closer/better match
|
||||
* @param callerPath Path of the calling file
|
||||
* @param candidatePath Path of the candidate definition file
|
||||
* @returns Distance score (lower is better)
|
||||
*/
|
||||
static calculateImportDistance(callerPath: string, candidatePath: string): number {
|
||||
const callerParts = callerPath.split('/').filter(p => p !== '');
|
||||
const candidateParts = candidatePath.split('/').filter(p => p !== '');
|
||||
|
||||
// Find common prefix length
|
||||
let commonPrefixLength = 0;
|
||||
const minLength = Math.min(callerParts.length, candidateParts.length);
|
||||
|
||||
for (let i = 0; i < minLength; i++) {
|
||||
if (callerParts[i] === candidateParts[i]) {
|
||||
commonPrefixLength++;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate base distance
|
||||
const maxLength = Math.max(callerParts.length, candidateParts.length);
|
||||
let distance = maxLength - commonPrefixLength;
|
||||
|
||||
// Bonus for sibling modules (same parent directory)
|
||||
if (commonPrefixLength === Math.min(callerParts.length, candidateParts.length) - 1) {
|
||||
distance -= 1; // Sibling bonus
|
||||
}
|
||||
|
||||
return distance;
|
||||
}
|
||||
|
||||
/**
|
||||
* Clear all data
|
||||
*/
|
||||
clear(): void {
|
||||
this.root = {
|
||||
children: new Map(),
|
||||
definitions: [],
|
||||
isEndOfWord: false
|
||||
};
|
||||
this.allDefinitions.clear();
|
||||
this.functionNameIndex.clear();
|
||||
this.filePathIndex.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove definitions from a specific file (NEW - for incremental updates)
|
||||
* @param filePath The file path to remove definitions for
|
||||
*/
|
||||
removeFileDefinitions(filePath: string): void {
|
||||
const definitions = this.filePathIndex.get(filePath);
|
||||
if (!definitions) return;
|
||||
|
||||
// Remove from all indexes
|
||||
for (const definition of definitions) {
|
||||
this.allDefinitions.delete(definition.nodeId);
|
||||
|
||||
// Remove from function name index
|
||||
const functionDefs = this.functionNameIndex.get(definition.functionName);
|
||||
if (functionDefs) {
|
||||
const index = functionDefs.indexOf(definition);
|
||||
if (index > -1) {
|
||||
functionDefs.splice(index, 1);
|
||||
}
|
||||
if (functionDefs.length === 0) {
|
||||
this.functionNameIndex.delete(definition.functionName);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Remove from file path index
|
||||
this.filePathIndex.delete(filePath);
|
||||
|
||||
// TODO: Clean up the trie structure (complex operation, can be done later)
|
||||
// For now, we'll leave empty nodes in the trie for performance
|
||||
}
|
||||
}
|
||||
|
||||
export type { FunctionDefinition };
|
||||
+34
-78
@@ -1,12 +1,12 @@
|
||||
export type NodeLabel =
|
||||
| 'Project'
|
||||
| 'Package'
|
||||
| 'Module'
|
||||
| 'Folder'
|
||||
| 'File'
|
||||
| 'Class'
|
||||
| 'Function'
|
||||
| 'Method'
|
||||
export type NodeLabel =
|
||||
| 'Project'
|
||||
| 'Package'
|
||||
| 'Module'
|
||||
| 'Folder'
|
||||
| 'File'
|
||||
| 'Class'
|
||||
| 'Function'
|
||||
| 'Method'
|
||||
| 'Variable'
|
||||
| 'Interface'
|
||||
| 'Enum'
|
||||
@@ -15,10 +15,14 @@ export type NodeLabel =
|
||||
| 'Type'
|
||||
| 'CodeElement';
|
||||
|
||||
export interface GraphNode {
|
||||
id: string;
|
||||
label: NodeLabel;
|
||||
properties: NodeProperties;
|
||||
|
||||
export type NodeProperties = {
|
||||
name: string,
|
||||
filePath: string,
|
||||
startLine?: number,
|
||||
endLine?: number,
|
||||
language?: string,
|
||||
isExported?: boolean,
|
||||
}
|
||||
|
||||
export type RelationshipType =
|
||||
@@ -31,74 +35,26 @@ export type RelationshipType =
|
||||
| 'DEFINES'
|
||||
| 'DECORATES'
|
||||
| 'IMPLEMENTS'
|
||||
| 'ACCESSES'
|
||||
| 'EXTENDS'
|
||||
| 'BELONGS_TO';
|
||||
|
||||
export interface GraphNode {
|
||||
id: string,
|
||||
label: NodeLabel,
|
||||
properties: NodeProperties,
|
||||
}
|
||||
|
||||
export interface GraphRelationship {
|
||||
id: string;
|
||||
type: RelationshipType;
|
||||
source: string;
|
||||
target: string;
|
||||
properties: RelationshipProperties;
|
||||
}
|
||||
|
||||
// Type-safe property interfaces
|
||||
export interface NodeProperties {
|
||||
// Common properties
|
||||
name?: string;
|
||||
path?: string;
|
||||
filePath?: string;
|
||||
extension?: string;
|
||||
language?: string;
|
||||
size?: number;
|
||||
|
||||
// Project-specific
|
||||
description?: string;
|
||||
version?: string;
|
||||
|
||||
// File-specific
|
||||
definitionCount?: number;
|
||||
lineCount?: number;
|
||||
|
||||
// Definition-specific
|
||||
type?: string;
|
||||
startLine?: number;
|
||||
endLine?: number;
|
||||
qualifiedName?: string;
|
||||
parameters?: string[];
|
||||
returnType?: string;
|
||||
|
||||
// Relationship-specific
|
||||
relationshipType?: string;
|
||||
[key: string]: string | number | boolean | string[] | undefined;
|
||||
}
|
||||
|
||||
export interface RelationshipProperties {
|
||||
// Common properties
|
||||
strength?: number;
|
||||
confidence?: number;
|
||||
|
||||
// Import-specific
|
||||
importType?: 'default' | 'named' | 'namespace' | 'dynamic';
|
||||
alias?: string;
|
||||
|
||||
// Call-specific
|
||||
callType?: 'function' | 'method' | 'constructor';
|
||||
arguments?: string[];
|
||||
|
||||
// Dependency-specific
|
||||
dependencyType?: 'direct' | 'transitive' | 'dev';
|
||||
version?: string;
|
||||
|
||||
[key: string]: string | number | boolean | string[] | undefined;
|
||||
id: string,
|
||||
sourceId: string,
|
||||
targetId: string,
|
||||
type: RelationshipType,
|
||||
}
|
||||
|
||||
export interface KnowledgeGraph {
|
||||
nodes: GraphNode[];
|
||||
relationships: GraphRelationship[];
|
||||
|
||||
// Methods for adding nodes and relationships
|
||||
addNode(node: GraphNode): void;
|
||||
addRelationship(relationship: GraphRelationship): void;
|
||||
}
|
||||
nodes: GraphNode[],
|
||||
relationships: GraphRelationship[],
|
||||
nodeCount: number,
|
||||
relationshipCount: number,
|
||||
addNode: (node: GraphNode) => void,
|
||||
addRelationship: (relationship: GraphRelationship) => void,
|
||||
}
|
||||
@@ -0,0 +1,47 @@
|
||||
import { LRUCache } from 'lru-cache';
|
||||
import Parser from 'web-tree-sitter';
|
||||
|
||||
// Define the interface for the Cache
|
||||
export interface ASTCache {
|
||||
get: (filePath: string) => Parser.Tree | undefined;
|
||||
set: (filePath: string, tree: Parser.Tree) => void;
|
||||
clear: () => void;
|
||||
stats: () => { size: number; maxSize: number };
|
||||
}
|
||||
|
||||
export const createASTCache = (maxSize: number = 50): ASTCache => {
|
||||
// Initialize the cache with a 'dispose' handler
|
||||
// This is the magic: When an item is evicted (dropped), this runs automatically.
|
||||
const cache = new LRUCache<string, Parser.Tree>({
|
||||
max: maxSize,
|
||||
dispose: (tree) => {
|
||||
try {
|
||||
// CRITICAL: Free the WASM memory when the tree leaves the cache
|
||||
tree.delete();
|
||||
} catch (e) {
|
||||
console.warn('Failed to delete tree from WASM memory', e);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
return {
|
||||
get: (filePath: string) => {
|
||||
const tree = cache.get(filePath);
|
||||
return tree; // Returns undefined if not found
|
||||
},
|
||||
|
||||
set: (filePath: string, tree: Parser.Tree) => {
|
||||
cache.set(filePath, tree);
|
||||
},
|
||||
|
||||
clear: () => {
|
||||
cache.clear();
|
||||
},
|
||||
|
||||
stats: () => ({
|
||||
size: cache.size,
|
||||
maxSize: maxSize
|
||||
})
|
||||
};
|
||||
};
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,123 @@
|
||||
/**
|
||||
* Heritage Processor
|
||||
*
|
||||
* Extracts class inheritance relationships:
|
||||
* - EXTENDS: Class extends another Class (TS, JS, Python)
|
||||
* - IMPLEMENTS: Class implements an Interface (TS only)
|
||||
*/
|
||||
|
||||
import { KnowledgeGraph } from '../graph/types';
|
||||
import { ASTCache } from './ast-cache';
|
||||
import { SymbolTable } from './symbol-table';
|
||||
import { loadParser, loadLanguage } from '../tree-sitter/parser-loader';
|
||||
import { LANGUAGE_QUERIES } from './tree-sitter-queries';
|
||||
import { generateId } from '../../lib/utils';
|
||||
import { getLanguageFromFilename } from './utils';
|
||||
|
||||
export const processHeritage = async (
|
||||
graph: KnowledgeGraph,
|
||||
files: { path: string; content: string }[],
|
||||
astCache: ASTCache,
|
||||
symbolTable: SymbolTable,
|
||||
onProgress?: (current: number, total: number) => void
|
||||
) => {
|
||||
const parser = await loadParser();
|
||||
|
||||
for (let i = 0; i < files.length; i++) {
|
||||
const file = files[i];
|
||||
onProgress?.(i + 1, files.length);
|
||||
|
||||
// 1. Check language support
|
||||
const language = getLanguageFromFilename(file.path);
|
||||
if (!language) continue;
|
||||
|
||||
const queryStr = LANGUAGE_QUERIES[language];
|
||||
if (!queryStr) continue;
|
||||
|
||||
// 2. Load the language
|
||||
await loadLanguage(language, file.path);
|
||||
|
||||
// 3. Get AST
|
||||
let tree = astCache.get(file.path);
|
||||
let wasReparsed = false;
|
||||
|
||||
if (!tree) {
|
||||
tree = parser.parse(file.content);
|
||||
wasReparsed = true;
|
||||
}
|
||||
|
||||
let query;
|
||||
let matches;
|
||||
try {
|
||||
query = parser.getLanguage().query(queryStr);
|
||||
matches = query.matches(tree.rootNode);
|
||||
} catch (queryError) {
|
||||
console.warn(`Heritage query error for ${file.path}:`, queryError);
|
||||
if (wasReparsed) tree.delete();
|
||||
continue;
|
||||
}
|
||||
|
||||
// 4. Process heritage matches
|
||||
matches.forEach(match => {
|
||||
const captureMap: Record<string, any> = {};
|
||||
match.captures.forEach(c => {
|
||||
captureMap[c.name] = c.node;
|
||||
});
|
||||
|
||||
// EXTENDS: Class extends another Class
|
||||
if (captureMap['heritage.class'] && captureMap['heritage.extends']) {
|
||||
const className = captureMap['heritage.class'].text;
|
||||
const parentClassName = captureMap['heritage.extends'].text;
|
||||
|
||||
// Resolve both class IDs
|
||||
const childId = symbolTable.lookupExact(file.path, className) ||
|
||||
symbolTable.lookupFuzzy(className)[0]?.nodeId ||
|
||||
generateId('Class', `${file.path}:${className}`);
|
||||
|
||||
const parentId = symbolTable.lookupFuzzy(parentClassName)[0]?.nodeId ||
|
||||
generateId('Class', `${parentClassName}`);
|
||||
|
||||
if (childId && parentId && childId !== parentId) {
|
||||
const relId = generateId('EXTENDS', `${childId}->${parentId}`);
|
||||
|
||||
graph.addRelationship({
|
||||
id: relId,
|
||||
sourceId: childId,
|
||||
targetId: parentId,
|
||||
type: 'EXTENDS'
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// IMPLEMENTS: Class implements Interface (TypeScript only)
|
||||
if (captureMap['heritage.class'] && captureMap['heritage.implements']) {
|
||||
const className = captureMap['heritage.class'].text;
|
||||
const interfaceName = captureMap['heritage.implements'].text;
|
||||
|
||||
// Resolve class and interface IDs
|
||||
const classId = symbolTable.lookupExact(file.path, className) ||
|
||||
symbolTable.lookupFuzzy(className)[0]?.nodeId ||
|
||||
generateId('Class', `${file.path}:${className}`);
|
||||
|
||||
const interfaceId = symbolTable.lookupFuzzy(interfaceName)[0]?.nodeId ||
|
||||
generateId('Interface', `${interfaceName}`);
|
||||
|
||||
if (classId && interfaceId) {
|
||||
const relId = generateId('IMPLEMENTS', `${classId}->${interfaceId}`);
|
||||
|
||||
graph.addRelationship({
|
||||
id: relId,
|
||||
sourceId: classId,
|
||||
targetId: interfaceId,
|
||||
type: 'IMPLEMENTS'
|
||||
});
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
// Cleanup
|
||||
if (wasReparsed) {
|
||||
tree.delete();
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -1,711 +1,148 @@
|
||||
import type { KnowledgeGraph, GraphRelationship } from '../graph/types.ts';
|
||||
import type { ParsedAST } from './parsing-processor.ts';
|
||||
import Parser from 'web-tree-sitter';
|
||||
import { KnowledgeGraph } from '../graph/types';
|
||||
import { ASTCache } from './ast-cache';
|
||||
import { loadParser, loadLanguage } from '../tree-sitter/parser-loader';
|
||||
import { LANGUAGE_QUERIES } from './tree-sitter-queries';
|
||||
import { generateId } from '../../lib/utils';
|
||||
import { getLanguageFromFilename } from './utils';
|
||||
|
||||
// Simple path utilities for browser compatibility
|
||||
const pathUtils = {
|
||||
extname: (filePath: string): string => {
|
||||
const lastDot = filePath.lastIndexOf('.');
|
||||
return lastDot === -1 ? '' : filePath.substring(lastDot);
|
||||
},
|
||||
dirname: (filePath: string): string => {
|
||||
const lastSlash = Math.max(filePath.lastIndexOf('/'), filePath.lastIndexOf('\\'));
|
||||
return lastSlash === -1 ? '.' : filePath.substring(0, lastSlash);
|
||||
},
|
||||
resolve: (basePath: string, relativePath: string): string => {
|
||||
// Simple relative path resolution
|
||||
if (relativePath.startsWith('./')) {
|
||||
return basePath + '/' + relativePath.substring(2);
|
||||
} else if (relativePath.startsWith('../')) {
|
||||
const parts = basePath.split('/');
|
||||
const relativeParts = relativePath.split('/');
|
||||
let upCount = 0;
|
||||
for (const part of relativeParts) {
|
||||
if (part === '..') upCount++;
|
||||
else break;
|
||||
// Type: Map<FilePath, Set<ResolvedFilePath>>
|
||||
// Stores all files that a given file imports from
|
||||
export type ImportMap = Map<string, Set<string>>;
|
||||
|
||||
export const createImportMap = (): ImportMap => new Map();
|
||||
|
||||
// Helper: Resolve relative paths (e.g. "../utils" -> "src/lib/utils.ts")
|
||||
const resolveImportPath = (
|
||||
currentFile: string,
|
||||
importPath: string,
|
||||
allFiles: Set<string>
|
||||
): string | null => {
|
||||
// 1. Handle non-relative imports (libraries like 'react')
|
||||
if (!importPath.startsWith('.')) return null; // We skip node_modules for now
|
||||
|
||||
// 2. Resolve '..' and '.'
|
||||
const currentDir = currentFile.split('/').slice(0, -1);
|
||||
const parts = importPath.split('/');
|
||||
|
||||
for (const part of parts) {
|
||||
if (part === '.') continue;
|
||||
if (part === '..') {
|
||||
currentDir.pop();
|
||||
} else {
|
||||
currentDir.push(part);
|
||||
}
|
||||
}
|
||||
|
||||
const basePath = currentDir.join('/');
|
||||
|
||||
// 3. Try extensions (prioritize .tsx for React projects)
|
||||
const extensions = ['', '.tsx', '.ts', '.jsx', '.js', '/index.tsx', '/index.ts', '/index.jsx', '/index.js'];
|
||||
|
||||
for (const ext of extensions) {
|
||||
const candidate = basePath + ext;
|
||||
if (allFiles.has(candidate)) return candidate;
|
||||
}
|
||||
|
||||
return null;
|
||||
};
|
||||
|
||||
export const processImports = async (
|
||||
graph: KnowledgeGraph,
|
||||
files: { path: string; content: string }[],
|
||||
astCache: ASTCache,
|
||||
importMap: ImportMap,
|
||||
onProgress?: (current: number, total: number) => void
|
||||
) => {
|
||||
// Create a Set of all file paths for fast lookup during resolution
|
||||
const allFilePaths = new Set(files.map(f => f.path));
|
||||
const parser = await loadParser();
|
||||
|
||||
for (let i = 0; i < files.length; i++) {
|
||||
const file = files[i];
|
||||
onProgress?.(i + 1, files.length);
|
||||
|
||||
// 1. Check language support first
|
||||
const language = getLanguageFromFilename(file.path);
|
||||
if (!language) continue;
|
||||
|
||||
const queryStr = LANGUAGE_QUERIES[language];
|
||||
if (!queryStr) continue;
|
||||
|
||||
// 2. ALWAYS load the language before querying (parser is stateful)
|
||||
await loadLanguage(language, file.path);
|
||||
|
||||
// 3. Get AST (Try Cache First)
|
||||
let tree = astCache.get(file.path);
|
||||
let wasReparsed = false;
|
||||
|
||||
if (!tree) {
|
||||
// Cache Miss: Re-parse (slower, but necessary if evicted)
|
||||
tree = parser.parse(file.content);
|
||||
wasReparsed = true;
|
||||
}
|
||||
|
||||
let query;
|
||||
let matches;
|
||||
try {
|
||||
query = parser.getLanguage().query(queryStr);
|
||||
matches = query.matches(tree.rootNode);
|
||||
} catch (queryError: any) {
|
||||
// Detailed debug logging for query failures
|
||||
console.group(`🔴 Query Error: ${file.path}`);
|
||||
console.log('Language:', language);
|
||||
console.log('Query (first 200 chars):', queryStr.substring(0, 200) + '...');
|
||||
console.log('Error:', queryError?.message || queryError);
|
||||
console.log('File content (first 300 chars):', file.content.substring(0, 300));
|
||||
console.log('AST root type:', tree.rootNode?.type);
|
||||
console.log('AST has errors:', tree.rootNode?.hasError);
|
||||
console.groupEnd();
|
||||
|
||||
if (wasReparsed) tree.delete();
|
||||
continue;
|
||||
}
|
||||
|
||||
matches.forEach(match => {
|
||||
const captureMap: Record<string, any> = {};
|
||||
match.captures.forEach(c => captureMap[c.name] = c.node);
|
||||
|
||||
if (captureMap['import']) {
|
||||
const sourceNode = captureMap['import.source'];
|
||||
if (!sourceNode) return;
|
||||
|
||||
// Clean path (remove quotes)
|
||||
const rawImportPath = sourceNode.text.replace(/['"]/g, '');
|
||||
|
||||
// Resolve to actual file in the system
|
||||
const resolvedPath = resolveImportPath(file.path, rawImportPath, allFilePaths);
|
||||
|
||||
if (resolvedPath) {
|
||||
// A. Update Graph (File -> IMPORTS -> File)
|
||||
const sourceId = generateId('File', file.path);
|
||||
const targetId = generateId('File', resolvedPath);
|
||||
const relId = generateId('IMPORTS', `${file.path}->${resolvedPath}`);
|
||||
|
||||
graph.addRelationship({
|
||||
id: relId,
|
||||
sourceId,
|
||||
targetId,
|
||||
type: 'IMPORTS'
|
||||
});
|
||||
|
||||
// B. Update Import Map (For Pass 4)
|
||||
// Store all resolved import paths for this file
|
||||
if (!importMap.has(file.path)) {
|
||||
importMap.set(file.path, new Set());
|
||||
}
|
||||
importMap.get(file.path)!.add(resolvedPath);
|
||||
}
|
||||
}
|
||||
const resultParts = parts.slice(0, -upCount);
|
||||
const remainingParts = relativeParts.slice(upCount);
|
||||
return [...resultParts, ...remainingParts].join('/');
|
||||
});
|
||||
|
||||
// If re-parsed just for this, delete the tree to save memory
|
||||
if (wasReparsed) {
|
||||
tree.delete();
|
||||
}
|
||||
return basePath + '/' + relativePath;
|
||||
},
|
||||
join: (...parts: string[]): string => {
|
||||
return parts.join('/').replace(/\/+/g, '/');
|
||||
}
|
||||
};
|
||||
|
||||
interface ImportMap {
|
||||
[importingFile: string]: {
|
||||
[localName: string]: {
|
||||
targetFile: string;
|
||||
exportedName: string;
|
||||
importType: 'default' | 'named' | 'namespace' | 'dynamic';
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
interface ImportInfo {
|
||||
importingFile: string;
|
||||
localName: string;
|
||||
targetFile: string;
|
||||
exportedName: string;
|
||||
importType: 'default' | 'named' | 'namespace' | 'dynamic';
|
||||
}
|
||||
|
||||
export class ImportProcessor {
|
||||
private importMap: ImportMap = {};
|
||||
private projectFiles: Set<string> = new Set();
|
||||
|
||||
private stats = {
|
||||
nodesProcessed: 0,
|
||||
relationshipsProcessed: 0
|
||||
};
|
||||
|
||||
constructor() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Process all imports after parsing is complete
|
||||
* @param graph The knowledge graph being built
|
||||
* @param astMap Map of file paths to their parsed ASTs
|
||||
* @param fileContents Map of file contents
|
||||
* @returns Updated graph with import relationships
|
||||
*/
|
||||
async process(
|
||||
graph: KnowledgeGraph,
|
||||
astMap: Map<string, ParsedAST>,
|
||||
fileContents: Map<string, string>
|
||||
): Promise<KnowledgeGraph> {
|
||||
try {
|
||||
console.log('📦 ImportProcessor: Starting import resolution...');
|
||||
|
||||
// Reset statistics
|
||||
this.stats = { nodesProcessed: 0, relationshipsProcessed: 0 };
|
||||
|
||||
// Build set of all project files for validation
|
||||
this.projectFiles = new Set(fileContents.keys());
|
||||
|
||||
// Clear previous import map
|
||||
this.importMap = {};
|
||||
|
||||
let totalImportsFound = 0;
|
||||
let totalImportsResolved = 0;
|
||||
|
||||
// Process imports for each file
|
||||
for (const [filePath, ast] of astMap) {
|
||||
const fileImports = await this.processFileImports(filePath, ast, graph);
|
||||
totalImportsFound += fileImports.found;
|
||||
totalImportsResolved += fileImports.resolved;
|
||||
}
|
||||
|
||||
console.log('✅ ImportProcessor: Completed import resolution');
|
||||
console.log(`📊 Found ${totalImportsFound} imports, resolved ${totalImportsResolved} (${totalImportsResolved > 0 ? ((totalImportsResolved/totalImportsFound)*100).toFixed(1) : '0'}%)`);
|
||||
console.log(`📋 Built import map for ${Object.keys(this.importMap).length} files`);
|
||||
console.log(`📊 ImportProcessor: ${this.stats.nodesProcessed} nodes, ${this.stats.relationshipsProcessed} relationships`);
|
||||
|
||||
return graph;
|
||||
} catch (error) {
|
||||
console.error('❌ ImportProcessor failed:', error);
|
||||
throw error;
|
||||
} finally {
|
||||
// Cleanup resources (if any)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Process imports for a single file
|
||||
*/
|
||||
private async processFileImports(
|
||||
filePath: string,
|
||||
ast: ParsedAST,
|
||||
graph: KnowledgeGraph
|
||||
): Promise<{ found: number; resolved: number }> {
|
||||
if (!ast.tree) {
|
||||
return { found: 0, resolved: 0 };
|
||||
}
|
||||
|
||||
|
||||
|
||||
const imports = this.extractImports(ast.tree.rootNode, filePath);
|
||||
|
||||
|
||||
|
||||
if (imports.length === 0) return { found: 0, resolved: 0 };
|
||||
|
||||
// Initialize import map for this file
|
||||
this.importMap[filePath] = {};
|
||||
|
||||
let found = 0;
|
||||
let resolved = 0;
|
||||
|
||||
for (const importInfo of imports) {
|
||||
// Store in import map
|
||||
this.importMap[filePath][importInfo.localName] = {
|
||||
targetFile: importInfo.targetFile,
|
||||
exportedName: importInfo.exportedName,
|
||||
importType: importInfo.importType
|
||||
};
|
||||
|
||||
// Create IMPORTS relationship in graph
|
||||
await this.createImportRelationship(graph, importInfo);
|
||||
found++;
|
||||
if (importInfo.targetFile !== importInfo.exportedName) { // Only count as resolved if it's not a default import
|
||||
resolved++;
|
||||
}
|
||||
}
|
||||
return { found, resolved };
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract import statements from AST
|
||||
*/
|
||||
private extractImports(rootNode: Parser.SyntaxNode, filePath: string): ImportInfo[] {
|
||||
const imports: ImportInfo[] = [];
|
||||
const language = this.detectLanguage(filePath);
|
||||
|
||||
if (language === 'python') {
|
||||
this.extractPythonImports(rootNode, filePath, imports);
|
||||
} else if (language === 'javascript' || language === 'typescript') {
|
||||
this.extractJSImports(rootNode, filePath, imports);
|
||||
}
|
||||
|
||||
return imports;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract Python imports
|
||||
*/
|
||||
private extractPythonImports(
|
||||
node: Parser.SyntaxNode,
|
||||
filePath: string,
|
||||
imports: ImportInfo[]
|
||||
): void {
|
||||
if (node.type === 'import_statement') {
|
||||
// Handle: import module
|
||||
// Handle: import module as alias
|
||||
const moduleNode = node.childForFieldName('name');
|
||||
if (moduleNode) {
|
||||
const moduleName = moduleNode.text;
|
||||
const targetFile = this.resolveModulePath(moduleName, filePath, 'python');
|
||||
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: moduleName.split('.').pop() || moduleName,
|
||||
targetFile,
|
||||
exportedName: moduleName,
|
||||
importType: 'namespace'
|
||||
});
|
||||
}
|
||||
} else if (node.type === 'import_from_statement') {
|
||||
// Handle: from module import name
|
||||
// Handle: from module import name as alias
|
||||
const moduleNode = node.childForFieldName('module_name');
|
||||
const namesNode = node.childForFieldName('name');
|
||||
|
||||
if (moduleNode && namesNode) {
|
||||
const moduleName = moduleNode.text;
|
||||
const targetFile = this.resolveModulePath(moduleName, filePath, 'python');
|
||||
|
||||
// Handle multiple imports: from module import a, b, c
|
||||
if (namesNode.type === 'import_list') {
|
||||
for (let i = 0; i < namesNode.childCount; i++) {
|
||||
const nameNode = namesNode.child(i);
|
||||
if (nameNode && nameNode.type === 'import_from_statement') {
|
||||
const importName = nameNode.text;
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: importName,
|
||||
targetFile,
|
||||
exportedName: importName,
|
||||
importType: 'named'
|
||||
});
|
||||
}
|
||||
}
|
||||
} else {
|
||||
const importName = namesNode.text;
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: importName,
|
||||
targetFile,
|
||||
exportedName: importName,
|
||||
importType: 'named'
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Recursively process children
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const child = node.child(i);
|
||||
if (child) {
|
||||
this.extractPythonImports(child, filePath, imports);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract JavaScript/TypeScript imports
|
||||
*/
|
||||
private extractJSImports(
|
||||
node: Parser.SyntaxNode,
|
||||
filePath: string,
|
||||
imports: ImportInfo[]
|
||||
): void {
|
||||
|
||||
|
||||
if (node.type === 'import_statement') {
|
||||
|
||||
|
||||
// Try different approaches to find the source
|
||||
let sourceNode = node.childForFieldName('source');
|
||||
if (!sourceNode) {
|
||||
// Try finding string literal directly
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const child = node.child(i);
|
||||
if (child && (child.type === 'string' || child.type === 'string_literal')) {
|
||||
sourceNode = child;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!sourceNode) {
|
||||
return;
|
||||
}
|
||||
|
||||
const sourcePath = sourceNode.text.replace(/['"]/g, '');
|
||||
const targetFile = this.resolveModulePath(sourcePath, filePath, 'javascript');
|
||||
|
||||
|
||||
|
||||
// Handle different import patterns
|
||||
let importClauseNode: Parser.SyntaxNode | null = node.childForFieldName('import_clause');
|
||||
|
||||
// CRITICAL FIX: If field-based approach fails, search by node type
|
||||
if (!importClauseNode) {
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const child = node.child(i);
|
||||
if (child?.type === 'import_clause') {
|
||||
importClauseNode = child;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (importClauseNode) {
|
||||
this.processJSImportClause(importClauseNode, filePath, targetFile, imports);
|
||||
} else {
|
||||
|
||||
// Handle simple imports like: import 'module'
|
||||
if (node.text.trim().startsWith('import') && !node.text.includes('{') && !node.text.includes('from')) {
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: '_side_effect_',
|
||||
targetFile,
|
||||
exportedName: '*',
|
||||
importType: 'namespace'
|
||||
});
|
||||
|
||||
} else {
|
||||
// Try to extract import manually from text as fallback
|
||||
const importText = node.text.trim();
|
||||
const match = importText.match(/import\s+(.+?)\s+from\s+['"]([^'"]+)['"]/);;
|
||||
if (match) {
|
||||
const importPart = match[1].trim();
|
||||
|
||||
// Handle simple default import
|
||||
if (!importPart.includes('{') && !importPart.includes('*')) {
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: importPart,
|
||||
targetFile,
|
||||
exportedName: 'default',
|
||||
importType: 'default'
|
||||
});
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if (node.type === 'variable_declaration') {
|
||||
// Handle CommonJS: const x = require('module')
|
||||
this.processRequireStatement(node, filePath, imports);
|
||||
}
|
||||
|
||||
// Recursively process children
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const child = node.child(i);
|
||||
if (child) {
|
||||
this.extractJSImports(child, filePath, imports);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Process JS import clause (handles named, default, namespace imports)
|
||||
*/
|
||||
private processJSImportClause(
|
||||
importClauseNode: Parser.SyntaxNode,
|
||||
filePath: string,
|
||||
targetFile: string,
|
||||
imports: ImportInfo[]
|
||||
): void {
|
||||
// Track what we've processed to ensure we don't miss anything
|
||||
let processedSomething = false;
|
||||
|
||||
for (let i = 0; i < importClauseNode.childCount; i++) {
|
||||
const child = importClauseNode.child(i);
|
||||
if (!child) continue;
|
||||
if (child.type === 'identifier') {
|
||||
// Default import - this is the most common case we're missing
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: child.text,
|
||||
targetFile,
|
||||
exportedName: 'default',
|
||||
importType: 'default'
|
||||
});
|
||||
|
||||
processedSomething = true;
|
||||
} else if (child.type === 'named_imports') {
|
||||
// Process named imports: { a, b, c }
|
||||
this.processNamedImportsNode(child, filePath, targetFile, imports);
|
||||
processedSomething = true;
|
||||
} else if (child.type === 'namespace_import') {
|
||||
// Namespace import: * as name
|
||||
const nameNode = child.childForFieldName('name');
|
||||
if (nameNode) {
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: nameNode.text,
|
||||
targetFile,
|
||||
exportedName: '*',
|
||||
importType: 'namespace'
|
||||
});
|
||||
|
||||
processedSomething = true;
|
||||
}
|
||||
} else if (child.type === 'import_specifier') {
|
||||
// Direct import specifier (should be handled by named_imports, but just in case)
|
||||
const nameNode = child.childForFieldName('name') || child.child(0);
|
||||
const aliasNode = child.childForFieldName('alias');
|
||||
|
||||
if (nameNode) {
|
||||
const exportedName = nameNode.text;
|
||||
const localName = aliasNode ? aliasNode.text : exportedName;
|
||||
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName,
|
||||
targetFile,
|
||||
exportedName,
|
||||
importType: 'named'
|
||||
});
|
||||
|
||||
processedSomething = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback: If structured processing didn't work, try text parsing
|
||||
if (!processedSomething) {
|
||||
const clauseText = importClauseNode.text.trim();
|
||||
|
||||
if (clauseText.startsWith('{') && clauseText.endsWith('}')) {
|
||||
// Named imports like { foo, bar }
|
||||
const namedImports = clauseText.slice(1, -1)
|
||||
.split(',')
|
||||
.map(s => s.trim())
|
||||
.filter(s => s.length > 0);
|
||||
|
||||
namedImports.forEach(importName => {
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: importName,
|
||||
targetFile,
|
||||
exportedName: importName,
|
||||
importType: 'named'
|
||||
});
|
||||
});
|
||||
|
||||
} else if (clauseText.includes(' as ')) {
|
||||
// Namespace import like * as foo
|
||||
const namespaceMatch = clauseText.match(/\*\s+as\s+(\w+)/);
|
||||
if (namespaceMatch) {
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: namespaceMatch[1],
|
||||
targetFile,
|
||||
exportedName: '*',
|
||||
importType: 'namespace'
|
||||
});
|
||||
|
||||
}
|
||||
} else {
|
||||
// Simple default import
|
||||
const defaultName = clauseText.split(',')[0].trim(); // Handle mixed imports
|
||||
if (defaultName && !defaultName.includes('{') && !defaultName.includes('*')) {
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: defaultName,
|
||||
targetFile,
|
||||
exportedName: 'default',
|
||||
importType: 'default'
|
||||
});
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private processNamedImportsNode(
|
||||
namedImportsNode: Parser.SyntaxNode,
|
||||
filePath: string,
|
||||
targetFile: string,
|
||||
imports: ImportInfo[]
|
||||
): void {
|
||||
for (let j = 0; j < namedImportsNode.childCount; j++) {
|
||||
const namedChild = namedImportsNode.child(j);
|
||||
if (namedChild && namedChild.type === 'import_specifier') {
|
||||
|
||||
const nameNode = namedChild.childForFieldName('name') || namedChild.child(0);
|
||||
const aliasNode = namedChild.childForFieldName('alias');
|
||||
|
||||
if (nameNode) {
|
||||
const exportedName = nameNode.text;
|
||||
const localName = aliasNode ? aliasNode.text : exportedName;
|
||||
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName,
|
||||
targetFile,
|
||||
exportedName,
|
||||
importType: 'named'
|
||||
});
|
||||
|
||||
}
|
||||
} else if (namedChild && namedChild.type === 'identifier') {
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: namedChild.text,
|
||||
targetFile,
|
||||
exportedName: namedChild.text,
|
||||
importType: 'named'
|
||||
});
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Process CommonJS require statements
|
||||
*/
|
||||
private processRequireStatement(
|
||||
node: Parser.SyntaxNode,
|
||||
filePath: string,
|
||||
imports: ImportInfo[]
|
||||
): void {
|
||||
// Look for: const x = require('module')
|
||||
const declaratorNode = node.child(1); // variable_declarator
|
||||
if (!declaratorNode) return;
|
||||
|
||||
const nameNode = declaratorNode.childForFieldName('name');
|
||||
const valueNode = declaratorNode.childForFieldName('value');
|
||||
|
||||
if (nameNode && valueNode && valueNode.type === 'call_expression') {
|
||||
const functionNode = valueNode.childForFieldName('function');
|
||||
const argumentsNode = valueNode.childForFieldName('arguments');
|
||||
|
||||
if (functionNode?.text === 'require' && argumentsNode) {
|
||||
const firstArg = argumentsNode.child(1); // Skip opening paren
|
||||
if (firstArg && firstArg.type === 'string') {
|
||||
const modulePath = firstArg.text.replace(/['"]/g, '');
|
||||
const targetFile = this.resolveModulePath(modulePath, filePath, 'javascript');
|
||||
|
||||
imports.push({
|
||||
importingFile: filePath,
|
||||
localName: nameNode.text,
|
||||
targetFile,
|
||||
exportedName: 'default',
|
||||
importType: 'dynamic'
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve module path to actual file path
|
||||
*/
|
||||
private resolveModulePath(moduleName: string, importingFile: string, language: 'python' | 'javascript'): string {
|
||||
// Handle relative imports
|
||||
if (moduleName.startsWith('.')) {
|
||||
const importingDir = pathUtils.dirname(importingFile);
|
||||
const resolvedPath = pathUtils.resolve(importingDir, moduleName);
|
||||
|
||||
// Try different extensions
|
||||
const extensions = language === 'python' ? ['.py'] : ['.js', '.ts', '.tsx', '.jsx'];
|
||||
|
||||
for (const ext of extensions) {
|
||||
const candidate = resolvedPath + ext;
|
||||
if (this.projectFiles.has(candidate)) {
|
||||
return candidate;
|
||||
}
|
||||
}
|
||||
|
||||
// Try index files
|
||||
for (const ext of extensions) {
|
||||
const indexCandidate = pathUtils.join(resolvedPath, `index${ext}`);
|
||||
if (this.projectFiles.has(indexCandidate)) {
|
||||
return indexCandidate;
|
||||
}
|
||||
}
|
||||
|
||||
return resolvedPath; // Return even if not found, for external modules
|
||||
}
|
||||
|
||||
// Handle absolute/package imports for Python
|
||||
if (language === 'python') {
|
||||
// First, try to find files that match the module pattern
|
||||
const modulePatterns = [
|
||||
// Direct module.py
|
||||
moduleName.replace(/\./g, '/') + '.py',
|
||||
// Package with __init__.py
|
||||
moduleName.replace(/\./g, '/') + '/__init__.py',
|
||||
// Try within the project structure
|
||||
`src/python/${moduleName.replace(/\./g, '/')}.py`,
|
||||
`src/python/${moduleName.replace(/\./g, '/')}/__init__.py`,
|
||||
];
|
||||
|
||||
// Also try to match partial paths for complex project structures
|
||||
for (const filePath of this.projectFiles) {
|
||||
if (filePath.endsWith('.py')) {
|
||||
// Check if this file could match the module name
|
||||
const moduleSegments = moduleName.split('.');
|
||||
const pathSegments = filePath.replace('.py', '').split('/');
|
||||
|
||||
// Try to match the last few segments
|
||||
if (moduleSegments.length > 0) {
|
||||
const lastSegment = moduleSegments[moduleSegments.length - 1];
|
||||
const fileName = pathSegments[pathSegments.length - 1];
|
||||
|
||||
// If the last segment matches the filename, this could be it
|
||||
if (fileName === lastSegment) {
|
||||
// Check if the path contains the module structure
|
||||
const modulePathInFile = moduleSegments.slice(0, -1).join('/');
|
||||
if (!modulePathInFile || filePath.includes(modulePathInFile)) {
|
||||
return filePath;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Try the direct patterns
|
||||
for (const pattern of modulePatterns) {
|
||||
if (this.projectFiles.has(pattern)) {
|
||||
return pattern;
|
||||
}
|
||||
}
|
||||
|
||||
// For complex module paths, try to find any file that ends with the module name
|
||||
const lastModuleSegment = moduleName.split('.').pop();
|
||||
if (lastModuleSegment) {
|
||||
for (const filePath of this.projectFiles) {
|
||||
if (filePath.endsWith(`${lastModuleSegment}.py`)) {
|
||||
return filePath;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// For external modules or unresolved, return as-is
|
||||
return moduleName;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create IMPORTS relationship in the graph with dual-write support
|
||||
*/
|
||||
private async createImportRelationship(graph: KnowledgeGraph, importInfo: ImportInfo): Promise<void> {
|
||||
// Find source and target nodes
|
||||
const sourceNode = graph.nodes.find(n =>
|
||||
n.label === 'File' && n.properties.filePath === importInfo.importingFile
|
||||
);
|
||||
|
||||
const targetNode = graph.nodes.find(n =>
|
||||
n.label === 'File' && n.properties.filePath === importInfo.targetFile
|
||||
);
|
||||
|
||||
if (sourceNode && targetNode && sourceNode.id !== targetNode.id) {
|
||||
// Check if relationship already exists
|
||||
const existingRel = graph.relationships.find(r =>
|
||||
r.type === 'IMPORTS' &&
|
||||
r.source === sourceNode.id &&
|
||||
r.target === targetNode.id
|
||||
);
|
||||
|
||||
if (!existingRel) {
|
||||
const relationship: GraphRelationship = {
|
||||
id: `imports_${sourceNode.id}_${targetNode.id}_${Date.now()}`,
|
||||
type: 'IMPORTS',
|
||||
source: sourceNode.id,
|
||||
target: targetNode.id,
|
||||
properties: {
|
||||
importType: importInfo.importType,
|
||||
localName: importInfo.localName,
|
||||
exportedName: importInfo.exportedName
|
||||
}
|
||||
};
|
||||
|
||||
graph.addRelationship(relationship);
|
||||
this.stats.relationshipsProcessed++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect programming language from file extension
|
||||
*/
|
||||
private detectLanguage(filePath: string): 'python' | 'javascript' | 'typescript' {
|
||||
const ext = pathUtils.extname(filePath).toLowerCase();
|
||||
|
||||
if (ext === '.py') return 'python';
|
||||
if (ext === '.ts' || ext === '.tsx') return 'typescript';
|
||||
return 'javascript'; // .js, .jsx, or default
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the complete import map for use by CallProcessor
|
||||
*/
|
||||
getImportMap(): ImportMap {
|
||||
return this.importMap;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get import info for a specific file and local name
|
||||
*/
|
||||
getImportInfo(filePath: string, localName: string): ImportMap[string][string] | null {
|
||||
return this.importMap[filePath]?.[localName] || null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Clear all data
|
||||
*/
|
||||
clear(): void {
|
||||
this.importMap = {};
|
||||
this.projectFiles.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* Get processing statistics
|
||||
*/
|
||||
public getStats() {
|
||||
return {
|
||||
nodesProcessed: this.stats.nodesProcessed,
|
||||
relationshipsProcessed: this.stats.relationshipsProcessed
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
export type { ImportMap, ImportInfo };
|
||||
@@ -1,462 +0,0 @@
|
||||
/**
|
||||
* KuzuDB-Aware Processor Base Class
|
||||
*
|
||||
* This base class provides KuzuDB integration capabilities for all
|
||||
* ingestion processors, implementing the dual-write pattern where
|
||||
* data is written to both JSON (SimpleKnowledgeGraph) and KuzuDB.
|
||||
*/
|
||||
|
||||
import type { KnowledgeGraph, GraphNode, GraphRelationship } from '../graph/types.ts';
|
||||
import { KuzuQueryEngine } from '../graph/kuzu-query-engine.ts';
|
||||
import { KuzuKnowledgeGraph } from '../graph/kuzu-knowledge-graph.ts';
|
||||
import { isKuzuDBEnabled, isKuzuDBPersistenceEnabled } from '../../config/features.ts';
|
||||
|
||||
export interface KuzuProcessorOptions {
|
||||
enableKuzuDB?: boolean;
|
||||
batchSize?: number;
|
||||
autoCommit?: boolean;
|
||||
enableValidation?: boolean;
|
||||
}
|
||||
|
||||
export interface ProcessorStats {
|
||||
nodesProcessed: number;
|
||||
relationshipsProcessed: number;
|
||||
kuzuNodesWritten: number;
|
||||
kuzuRelationshipsWritten: number;
|
||||
kuzuErrors: number;
|
||||
validationErrors: number;
|
||||
processingTime: number;
|
||||
// Additional stats for CallProcessor
|
||||
totalCalls?: number;
|
||||
exactMatches?: number;
|
||||
sameFileMatches?: number;
|
||||
heuristicMatches?: number;
|
||||
failed?: number;
|
||||
callTypes?: Record<string, number>;
|
||||
failuresByCategory?: {
|
||||
externalLibraries: number;
|
||||
pythonBuiltins: number;
|
||||
actualFailures: number;
|
||||
};
|
||||
}
|
||||
|
||||
export interface TransactionState {
|
||||
isActive: boolean;
|
||||
startTime: number;
|
||||
nodesBatch: GraphNode[];
|
||||
relationshipsBatch: GraphRelationship[];
|
||||
rollbackData: {
|
||||
jsonNodes: GraphNode[];
|
||||
jsonRelationships: GraphRelationship[];
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Base class for processors that support KuzuDB dual-write pattern
|
||||
*/
|
||||
export abstract class KuzuProcessorBase {
|
||||
protected kuzuQueryEngine: KuzuQueryEngine | null = null;
|
||||
protected kuzuGraph: KuzuKnowledgeGraph | null = null;
|
||||
protected options: KuzuProcessorOptions;
|
||||
protected stats: ProcessorStats;
|
||||
protected transaction: TransactionState;
|
||||
|
||||
constructor(options: KuzuProcessorOptions = {}) {
|
||||
this.options = {
|
||||
enableKuzuDB: options.enableKuzuDB ?? isKuzuDBEnabled(),
|
||||
batchSize: options.batchSize ?? 100,
|
||||
autoCommit: options.autoCommit ?? true,
|
||||
enableValidation: options.enableValidation ?? true
|
||||
};
|
||||
|
||||
this.stats = {
|
||||
nodesProcessed: 0,
|
||||
relationshipsProcessed: 0,
|
||||
kuzuNodesWritten: 0,
|
||||
kuzuRelationshipsWritten: 0,
|
||||
kuzuErrors: 0,
|
||||
validationErrors: 0,
|
||||
processingTime: 0
|
||||
};
|
||||
|
||||
this.transaction = {
|
||||
isActive: false,
|
||||
startTime: 0,
|
||||
nodesBatch: [],
|
||||
relationshipsBatch: [],
|
||||
rollbackData: {
|
||||
jsonNodes: [],
|
||||
jsonRelationships: []
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize KuzuDB integration
|
||||
*/
|
||||
protected async initializeKuzuDB(): Promise<void> {
|
||||
if (!this.options.enableKuzuDB) {
|
||||
console.log('⚠️ KuzuDB integration disabled for this processor');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
console.log('🚀 Initializing KuzuDB integration...');
|
||||
|
||||
// Initialize query engine
|
||||
this.kuzuQueryEngine = new KuzuQueryEngine({
|
||||
enableCache: true,
|
||||
cacheSize: 1000,
|
||||
cacheTTL: 5 * 60 * 1000 // 5 minutes
|
||||
});
|
||||
|
||||
await this.kuzuQueryEngine.initialize();
|
||||
|
||||
// Create KuzuDB knowledge graph
|
||||
this.kuzuGraph = new KuzuKnowledgeGraph(this.kuzuQueryEngine, {
|
||||
enableCache: true,
|
||||
batchSize: this.options.batchSize,
|
||||
autoCommit: this.options.autoCommit
|
||||
});
|
||||
|
||||
console.log('✅ KuzuDB integration initialized successfully');
|
||||
} catch (error) {
|
||||
console.error('❌ Failed to initialize KuzuDB integration:', error);
|
||||
this.kuzuQueryEngine = null;
|
||||
this.kuzuGraph = null;
|
||||
// Continue without KuzuDB - graceful degradation
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Dual-write a node to both JSON graph and KuzuDB
|
||||
*/
|
||||
protected async addNodeDualWrite(jsonGraph: KnowledgeGraph, node: GraphNode): Promise<void> {
|
||||
const startTime = performance.now();
|
||||
|
||||
try {
|
||||
// Always write to JSON graph first (primary storage)
|
||||
jsonGraph.addNode(node);
|
||||
this.stats.nodesProcessed++;
|
||||
|
||||
// Write to KuzuDB if enabled and available
|
||||
if (this.kuzuGraph) {
|
||||
try {
|
||||
this.kuzuGraph.addNode(node);
|
||||
this.stats.kuzuNodesWritten++;
|
||||
} catch (kuzuError) {
|
||||
this.stats.kuzuErrors++;
|
||||
console.warn(`KuzuDB node write failed for ${node.id}:`, kuzuError);
|
||||
// Continue - JSON is primary, KuzuDB failure shouldn't break the process
|
||||
}
|
||||
}
|
||||
|
||||
// Data validation if enabled
|
||||
if (this.options.enableValidation && this.kuzuGraph) {
|
||||
await this.validateNodeConsistency(node);
|
||||
}
|
||||
|
||||
} catch (error) {
|
||||
console.error(`Failed to add node ${node.id}:`, error);
|
||||
throw error;
|
||||
} finally {
|
||||
this.stats.processingTime += performance.now() - startTime;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Dual-write a relationship to both JSON graph and KuzuDB
|
||||
*/
|
||||
protected async addRelationshipDualWrite(
|
||||
jsonGraph: KnowledgeGraph,
|
||||
relationship: GraphRelationship
|
||||
): Promise<void> {
|
||||
const startTime = performance.now();
|
||||
|
||||
try {
|
||||
// Always write to JSON graph first (primary storage)
|
||||
jsonGraph.addRelationship(relationship);
|
||||
this.stats.relationshipsProcessed++;
|
||||
|
||||
// Write to KuzuDB if enabled and available
|
||||
if (this.kuzuGraph) {
|
||||
try {
|
||||
this.kuzuGraph.addRelationship(relationship);
|
||||
this.stats.kuzuRelationshipsWritten++;
|
||||
} catch (kuzuError) {
|
||||
this.stats.kuzuErrors++;
|
||||
console.warn(`KuzuDB relationship write failed for ${relationship.id}:`, kuzuError);
|
||||
// Continue - JSON is primary, KuzuDB failure shouldn't break the process
|
||||
}
|
||||
}
|
||||
|
||||
// Data validation if enabled
|
||||
if (this.options.enableValidation && this.kuzuGraph) {
|
||||
await this.validateRelationshipConsistency(relationship);
|
||||
}
|
||||
|
||||
} catch (error) {
|
||||
console.error(`Failed to add relationship ${relationship.id}:`, error);
|
||||
throw error;
|
||||
} finally {
|
||||
this.stats.processingTime += performance.now() - startTime;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Commit all pending KuzuDB operations
|
||||
*/
|
||||
protected async commitKuzuDB(): Promise<void> {
|
||||
if (!this.kuzuGraph) return;
|
||||
|
||||
try {
|
||||
console.log('📝 Committing KuzuDB operations...');
|
||||
await this.kuzuGraph.commitAll();
|
||||
console.log('✅ KuzuDB operations committed successfully');
|
||||
} catch (error) {
|
||||
console.error('❌ Failed to commit KuzuDB operations:', error);
|
||||
this.stats.kuzuErrors++;
|
||||
// Don't throw - this shouldn't break the main process
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate that a node exists consistently in both storages
|
||||
*/
|
||||
private async validateNodeConsistency(node: GraphNode): Promise<void> {
|
||||
if (!this.kuzuGraph) return;
|
||||
|
||||
try {
|
||||
// This is a placeholder for validation logic
|
||||
// In a real implementation, you would query KuzuDB to verify the node exists
|
||||
// and has the same properties as the JSON version
|
||||
|
||||
// For now, we'll just increment the validation counter
|
||||
// TODO: Implement actual validation once KuzuDB queries are working
|
||||
|
||||
} catch (error) {
|
||||
this.stats.validationErrors++;
|
||||
console.warn(`Validation failed for node ${node.id}:`, error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate that a relationship exists consistently in both storages
|
||||
*/
|
||||
private async validateRelationshipConsistency(relationship: GraphRelationship): Promise<void> {
|
||||
if (!this.kuzuGraph) return;
|
||||
|
||||
try {
|
||||
// This is a placeholder for validation logic
|
||||
// In a real implementation, you would query KuzuDB to verify the relationship exists
|
||||
// and has the same properties as the JSON version
|
||||
|
||||
// For now, we'll just increment the validation counter
|
||||
// TODO: Implement actual validation once KuzuDB queries are working
|
||||
|
||||
} catch (error) {
|
||||
this.stats.validationErrors++;
|
||||
console.warn(`Validation failed for relationship ${relationship.id}:`, error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get processing statistics
|
||||
*/
|
||||
public getStats(): ProcessorStats {
|
||||
return { ...this.stats };
|
||||
}
|
||||
|
||||
/**
|
||||
* Reset processing statistics
|
||||
*/
|
||||
protected resetStats(): void {
|
||||
this.stats = {
|
||||
nodesProcessed: 0,
|
||||
relationshipsProcessed: 0,
|
||||
kuzuNodesWritten: 0,
|
||||
kuzuRelationshipsWritten: 0,
|
||||
kuzuErrors: 0,
|
||||
validationErrors: 0,
|
||||
processingTime: 0
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Log processing statistics
|
||||
*/
|
||||
protected logStats(processorName: string): void {
|
||||
const stats = this.stats;
|
||||
const kuzuEnabled = this.kuzuGraph !== null;
|
||||
|
||||
console.log(`📊 ${processorName} Statistics:`);
|
||||
console.log(` Nodes processed: ${stats.nodesProcessed}`);
|
||||
console.log(` Relationships processed: ${stats.relationshipsProcessed}`);
|
||||
console.log(` Processing time: ${stats.processingTime.toFixed(2)}ms`);
|
||||
|
||||
if (kuzuEnabled) {
|
||||
console.log(` KuzuDB nodes written: ${stats.kuzuNodesWritten}`);
|
||||
console.log(` KuzuDB relationships written: ${stats.kuzuRelationshipsWritten}`);
|
||||
console.log(` KuzuDB errors: ${stats.kuzuErrors}`);
|
||||
console.log(` Validation errors: ${stats.validationErrors}`);
|
||||
|
||||
const successRate = stats.nodesProcessed + stats.relationshipsProcessed > 0
|
||||
? ((stats.kuzuNodesWritten + stats.kuzuRelationshipsWritten) /
|
||||
(stats.nodesProcessed + stats.relationshipsProcessed) * 100).toFixed(1)
|
||||
: '0';
|
||||
console.log(` KuzuDB success rate: ${successRate}%`);
|
||||
} else {
|
||||
console.log(' KuzuDB: Disabled');
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Cleanup resources
|
||||
*/
|
||||
protected async cleanup(): Promise<void> {
|
||||
try {
|
||||
// Commit any pending operations
|
||||
await this.commitKuzuDB();
|
||||
|
||||
// Close KuzuDB connection
|
||||
if (this.kuzuQueryEngine) {
|
||||
await this.kuzuQueryEngine.close();
|
||||
this.kuzuQueryEngine = null;
|
||||
this.kuzuGraph = null;
|
||||
}
|
||||
|
||||
console.log('✅ Processor cleanup completed');
|
||||
} catch (error) {
|
||||
console.error('❌ Error during processor cleanup:', error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if KuzuDB integration is enabled and ready
|
||||
*/
|
||||
protected isKuzuDBReady(): boolean {
|
||||
return this.kuzuGraph !== null &&
|
||||
this.kuzuQueryEngine !== null &&
|
||||
this.kuzuQueryEngine.isReady();
|
||||
}
|
||||
|
||||
/**
|
||||
* Get KuzuDB integration status
|
||||
*/
|
||||
public getKuzuDBStatus(): {
|
||||
enabled: boolean;
|
||||
ready: boolean;
|
||||
queryEngine: boolean;
|
||||
graph: boolean;
|
||||
} {
|
||||
return {
|
||||
enabled: this.options.enableKuzuDB || false,
|
||||
ready: this.isKuzuDBReady(),
|
||||
queryEngine: this.kuzuQueryEngine !== null,
|
||||
graph: this.kuzuGraph !== null
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Begin a transaction for batch operations
|
||||
*/
|
||||
protected async beginTransaction(): Promise<void> {
|
||||
if (this.transaction.isActive) {
|
||||
console.warn('⚠️ Transaction already active, committing previous transaction');
|
||||
await this.commitTransaction();
|
||||
}
|
||||
|
||||
this.transaction = {
|
||||
isActive: true,
|
||||
startTime: performance.now(),
|
||||
nodesBatch: [],
|
||||
relationshipsBatch: [],
|
||||
rollbackData: {
|
||||
jsonNodes: [],
|
||||
jsonRelationships: []
|
||||
}
|
||||
};
|
||||
|
||||
console.log('🔄 Transaction started');
|
||||
}
|
||||
|
||||
/**
|
||||
* Commit the current transaction
|
||||
*/
|
||||
protected async commitTransaction(): Promise<void> {
|
||||
if (!this.transaction.isActive) {
|
||||
console.warn('⚠️ No active transaction to commit');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
// Commit to KuzuDB if available
|
||||
if (this.kuzuGraph) {
|
||||
await this.kuzuGraph.commitAll();
|
||||
}
|
||||
|
||||
const duration = performance.now() - this.transaction.startTime;
|
||||
console.log(`✅ Transaction committed successfully in ${duration.toFixed(2)}ms`);
|
||||
console.log(`📊 Committed ${this.transaction.nodesBatch.length} nodes and ${this.transaction.relationshipsBatch.length} relationships`);
|
||||
|
||||
// Reset transaction state
|
||||
this.transaction.isActive = false;
|
||||
this.transaction.nodesBatch = [];
|
||||
this.transaction.relationshipsBatch = [];
|
||||
this.transaction.rollbackData = { jsonNodes: [], jsonRelationships: [] };
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Transaction commit failed:', error);
|
||||
await this.rollbackTransaction();
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Rollback the current transaction
|
||||
*/
|
||||
protected async rollbackTransaction(): Promise<void> {
|
||||
if (!this.transaction.isActive) {
|
||||
console.warn('⚠️ No active transaction to rollback');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
console.log('🔄 Rolling back transaction...');
|
||||
|
||||
// Note: For JSON rollback, we would need to maintain the original state
|
||||
// This is a simplified implementation - in production, you'd want more sophisticated rollback logic
|
||||
|
||||
const duration = performance.now() - this.transaction.startTime;
|
||||
console.log(`🔙 Transaction rolled back in ${duration.toFixed(2)}ms`);
|
||||
console.log(`📊 Rolled back ${this.transaction.nodesBatch.length} nodes and ${this.transaction.relationshipsBatch.length} relationships`);
|
||||
|
||||
// Reset transaction state
|
||||
this.transaction.isActive = false;
|
||||
this.transaction.nodesBatch = [];
|
||||
this.transaction.relationshipsBatch = [];
|
||||
this.transaction.rollbackData = { jsonNodes: [], jsonRelationships: [] };
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Transaction rollback failed:', error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get transaction status
|
||||
*/
|
||||
protected getTransactionStatus(): {
|
||||
isActive: boolean;
|
||||
duration: number;
|
||||
nodeCount: number;
|
||||
relationshipCount: number;
|
||||
} {
|
||||
return {
|
||||
isActive: this.transaction.isActive,
|
||||
duration: this.transaction.isActive ? performance.now() - this.transaction.startTime : 0,
|
||||
nodeCount: this.transaction.nodesBatch.length,
|
||||
relationshipCount: this.transaction.relationshipsBatch.length
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -1,798 +0,0 @@
|
||||
import { GraphNode, GraphRelationship, NodeLabel, RelationshipType } from '../graph/types.js';
|
||||
import { MemoryManager } from '../../services/memory-manager.js';
|
||||
import { KnowledgeGraph, GraphProcessor } from '../graph/graph.js';
|
||||
import {
|
||||
OptimizedSet,
|
||||
DuplicateDetector,
|
||||
pathUtils
|
||||
} from '../../lib/shared-utils.js';
|
||||
import { ignoreService } from '../../config/ignore-service.js';
|
||||
import { WebWorkerPool, WebWorkerPoolUtils } from '../../lib/web-worker-pool.js';
|
||||
import { FunctionRegistryTrie, FunctionDefinition } from '../graph/trie.js';
|
||||
import { generateDeterministicId } from '../../lib/utils.ts';
|
||||
import Parser from 'web-tree-sitter';
|
||||
import { initTreeSitter, loadTypeScriptParser, loadPythonParser, loadJavaScriptParser } from '../tree-sitter/parser-loader.js';
|
||||
|
||||
export interface ParsingInput {
|
||||
filePaths: string[];
|
||||
fileContents: Map<string, string>;
|
||||
options?: { directoryFilter?: string; fileExtensions?: string };
|
||||
}
|
||||
|
||||
export interface ParsedDefinition {
|
||||
name: string;
|
||||
type: 'function' | 'class' | 'method' | 'variable' | 'import' | 'interface' | 'type' | 'decorator';
|
||||
startLine: number;
|
||||
endLine?: number;
|
||||
parameters?: string[] | undefined;
|
||||
returnType?: string | undefined;
|
||||
accessibility?: 'public' | 'private' | 'protected';
|
||||
isStatic?: boolean | undefined;
|
||||
isAsync?: boolean | undefined;
|
||||
parentClass?: string | undefined;
|
||||
decorators?: string[] | undefined;
|
||||
extends?: string[] | undefined;
|
||||
implements?: string[] | undefined;
|
||||
importPath?: string | undefined;
|
||||
exportType?: 'named' | 'default' | 'namespace';
|
||||
docstring?: string | undefined;
|
||||
}
|
||||
|
||||
export interface ParsedAST {
|
||||
tree: any;
|
||||
}
|
||||
|
||||
export interface ParallelParsingResult {
|
||||
filePath: string;
|
||||
definitions: ParsedDefinition[];
|
||||
ast: ParsedAST;
|
||||
success: boolean;
|
||||
error?: string;
|
||||
}
|
||||
|
||||
|
||||
|
||||
export class ParallelParsingProcessor implements GraphProcessor<ParsingInput> {
|
||||
private memoryManager: MemoryManager;
|
||||
private duplicateDetector = new DuplicateDetector<string>((item: string) => item);
|
||||
private processedFiles = new OptimizedSet<string>();
|
||||
private astMap: Map<string, ParsedAST> = new Map(); // Keep for compatibility with import/call processors
|
||||
private functionTrie: FunctionRegistryTrie = new FunctionRegistryTrie();
|
||||
private workerPool: WebWorkerPool;
|
||||
private isInitialized: boolean = false;
|
||||
private parser: Parser | null = null;
|
||||
private languageParsers: Map<string, Parser.Language> = new Map();
|
||||
private memoryMonitorInterval: NodeJS.Timeout | null = null;
|
||||
private readonly MAX_AST_MAP_SIZE = 1000; // Limit AST map size to prevent memory issues
|
||||
|
||||
constructor() {
|
||||
this.memoryManager = MemoryManager.getInstance();
|
||||
// Worker pool will be initialized asynchronously in initializeWorkerPool()
|
||||
this.workerPool = null as any; // Temporary until initialization
|
||||
}
|
||||
|
||||
public getASTMap(): Map<string, ParsedAST> {
|
||||
return this.astMap;
|
||||
}
|
||||
|
||||
public getFunctionRegistry(): FunctionRegistryTrie {
|
||||
return this.functionTrie;
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize the worker pool
|
||||
*/
|
||||
private async initializeWorkerPool(): Promise<void> {
|
||||
if (this.isInitialized) return;
|
||||
|
||||
try {
|
||||
console.log('ParallelParsingProcessor: Initializing worker pool...');
|
||||
|
||||
// Check if Web Workers are supported
|
||||
if (!WebWorkerPoolUtils.isSupported()) {
|
||||
throw new Error('Web Workers are not supported in this environment');
|
||||
}
|
||||
|
||||
// Create worker pool using configuration
|
||||
this.workerPool = await WebWorkerPoolUtils.createWorkerPool({
|
||||
workerScript: '/workers/tree-sitter-worker.js',
|
||||
name: 'ParallelParsingPool'
|
||||
});
|
||||
|
||||
// Set up worker pool event listeners
|
||||
this.workerPool.on('workerCreated', (data: unknown) => {
|
||||
const { workerId, totalWorkers } = data as { workerId: number, totalWorkers: number };
|
||||
console.log(`ParallelParsingProcessor: Worker ${workerId} created (${totalWorkers} total)`);
|
||||
});
|
||||
|
||||
this.workerPool.on('workerError', (data: unknown) => {
|
||||
const { workerId, error } = data as { workerId: number, error: string };
|
||||
console.warn(`ParallelParsingProcessor: Worker ${workerId} error:`, error);
|
||||
});
|
||||
|
||||
this.workerPool.on('shutdown', () => {
|
||||
console.log('ParallelParsingProcessor: Worker pool shutdown');
|
||||
});
|
||||
|
||||
// Start memory monitoring
|
||||
this.startMemoryMonitoring();
|
||||
|
||||
this.isInitialized = true;
|
||||
console.log('ParallelParsingProcessor: Worker pool initialized successfully');
|
||||
} catch (error) {
|
||||
console.error('ParallelParsingProcessor: Failed to initialize worker pool:', error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
public async process(graph: KnowledgeGraph, input: ParsingInput): Promise<void> {
|
||||
const { filePaths, fileContents, options } = input;
|
||||
|
||||
console.log(`ParallelParsingProcessor: Processing ${filePaths.length} total paths with worker pool`);
|
||||
|
||||
const memoryStats = this.memoryManager.getStats();
|
||||
console.log(`Memory status: ${memoryStats.usedMemoryMB}MB used, ${memoryStats.fileCount} files cached`);
|
||||
|
||||
// Initialize worker pool
|
||||
await this.initializeWorkerPool();
|
||||
|
||||
const filteredFiles = this.applyFiltering(filePaths, fileContents, options);
|
||||
|
||||
console.log(`ParallelParsingProcessor: After filtering: ${filteredFiles.length} files to parse`);
|
||||
|
||||
const sourceFiles = filteredFiles.filter((path: string) => this.isSourceFile(path));
|
||||
const configFiles = filteredFiles.filter((path: string) => this.isConfigFile(path));
|
||||
const allProcessableFiles = [...sourceFiles, ...configFiles];
|
||||
|
||||
console.log(`ParallelParsingProcessor: Found ${sourceFiles.length} source files and ${configFiles.length} config files`);
|
||||
|
||||
try {
|
||||
// Process files in parallel using worker pool
|
||||
const results = await this.processFilesInParallel(allProcessableFiles, fileContents);
|
||||
|
||||
// Process results and build graph
|
||||
await this.processResults(results, graph, fileContents);
|
||||
|
||||
console.log(`ParallelParsingProcessor: Successfully processed ${this.processedFiles.size} files`);
|
||||
} catch (error) {
|
||||
console.error('ParallelParsingProcessor: Error during parallel processing:', error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Process files in parallel using worker pool
|
||||
*/
|
||||
private async processFilesInParallel(
|
||||
filePaths: string[],
|
||||
fileContents: Map<string, string>
|
||||
): Promise<ParallelParsingResult[]> {
|
||||
const startTime = performance.now();
|
||||
|
||||
// Prepare tasks for worker pool
|
||||
const tasks = filePaths.map(filePath => ({
|
||||
filePath,
|
||||
content: fileContents.get(filePath) || ''
|
||||
}));
|
||||
|
||||
console.log(`ParallelParsingProcessor: Starting parallel processing of ${tasks.length} files`);
|
||||
|
||||
let workerResults: ParallelParsingResult[] = [];
|
||||
|
||||
if (tasks.length > 0) {
|
||||
try {
|
||||
// Process with progress tracking
|
||||
const results = await this.workerPool.executeWithProgress<any, ParallelParsingResult>(
|
||||
tasks,
|
||||
(completed, total) => {
|
||||
const progress = ((completed / total) * 100).toFixed(1);
|
||||
console.log(`ParallelParsingProcessor: Progress: ${progress}% (${completed}/${total})`);
|
||||
}
|
||||
);
|
||||
|
||||
const endTime = performance.now();
|
||||
const duration = endTime - startTime;
|
||||
|
||||
console.log(`ParallelParsingProcessor: Parallel processing completed in ${duration.toFixed(2)}ms`);
|
||||
if (tasks.length > 0) {
|
||||
console.log(`ParallelParsingProcessor: Average time per file: ${(duration / tasks.length).toFixed(2)}ms`);
|
||||
}
|
||||
|
||||
// Log worker pool statistics
|
||||
const stats = this.workerPool.getStats();
|
||||
console.log('ParallelParsingProcessor: Worker pool stats:', stats);
|
||||
|
||||
// Transform worker results to match expected format
|
||||
workerResults = results.map((result: any, index: number) => {
|
||||
if (result && result.filePath) {
|
||||
return {
|
||||
filePath: result.filePath,
|
||||
definitions: result.definitions || [],
|
||||
ast: result.ast || null,
|
||||
success: !result.error,
|
||||
error: result.error
|
||||
};
|
||||
} else {
|
||||
// Handle undefined/null results
|
||||
return {
|
||||
filePath: tasks[index]?.filePath || 'unknown',
|
||||
definitions: [],
|
||||
ast: null,
|
||||
success: false,
|
||||
error: 'Worker returned undefined result'
|
||||
};
|
||||
}
|
||||
});
|
||||
} catch (error) {
|
||||
console.error('ParallelParsingProcessor: Error in worker pool processing:', error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`ParallelParsingProcessor: Total results: ${workerResults.length} processed`);
|
||||
|
||||
return workerResults;
|
||||
}
|
||||
|
||||
/**
|
||||
* Process parsing results and build graph
|
||||
*/
|
||||
private async processResults(results: ParallelParsingResult[], graph: KnowledgeGraph, fileContents: Map<string, string>): Promise<void> {
|
||||
console.log(`ParallelParsingProcessor: Processing ${results.length} parsing results`);
|
||||
|
||||
let successfulFiles = 0;
|
||||
let failedFiles = 0;
|
||||
let totalDefinitions = 0;
|
||||
|
||||
// Initialize main thread parser for AST recreation (needed for import/call processors)
|
||||
await this.initializeMainThreadParser();
|
||||
|
||||
for (const result of results) {
|
||||
if (result.success) {
|
||||
successfulFiles++;
|
||||
|
||||
// Recreate full AST in main thread (workers can't serialize Tree-sitter objects)
|
||||
await this.recreateAST(result.filePath, fileContents);
|
||||
|
||||
// Process definitions
|
||||
if (result.definitions && result.definitions.length > 0) {
|
||||
await this.processDefinitions(result.filePath, result.definitions, graph);
|
||||
totalDefinitions += result.definitions.length;
|
||||
}
|
||||
|
||||
this.processedFiles.add(result.filePath);
|
||||
} else {
|
||||
failedFiles++;
|
||||
console.warn(`ParallelParsingProcessor: Failed to parse ${result.filePath}: ${result.error}`);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`ParallelParsingProcessor: Processing complete - ${successfulFiles} successful, ${failedFiles} failed`);
|
||||
console.log(`ParallelParsingProcessor: Total definitions extracted: ${totalDefinitions}`);
|
||||
|
||||
// Log final memory and AST map statistics
|
||||
const finalMemoryStats = this.memoryManager.getStats();
|
||||
console.log(`ParallelParsingProcessor: Final Memory Status - ${finalMemoryStats.usedMemoryMB}MB used, ${finalMemoryStats.fileCount} files cached`);
|
||||
console.log(`ParallelParsingProcessor: AST Map Size: ${this.astMap.size} entries`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Process definitions and add to graph
|
||||
*/
|
||||
private async processDefinitions(
|
||||
filePath: string,
|
||||
definitions: ParsedDefinition[],
|
||||
graph: KnowledgeGraph
|
||||
): Promise<void> {
|
||||
for (const definition of definitions) {
|
||||
try {
|
||||
await this.addDefinitionToGraph(filePath, definition, graph);
|
||||
} catch (error) {
|
||||
console.warn(`ParallelParsingProcessor: Error processing definition ${definition.name}:`, error);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Add a definition to the graph
|
||||
*/
|
||||
private async addDefinitionToGraph(
|
||||
filePath: string,
|
||||
definition: ParsedDefinition,
|
||||
graph: KnowledgeGraph
|
||||
): Promise<void> {
|
||||
// Generate unique ID based on file path and definition name (same as single-threaded)
|
||||
const nodeId = generateDeterministicId(definition.type, `${filePath}_${definition.name}_${definition.startLine}`);
|
||||
|
||||
if (this.duplicateDetector.checkAndMark(nodeId)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Create graph node
|
||||
const node: GraphNode = {
|
||||
id: nodeId,
|
||||
label: this.mapDefinitionTypeToNodeLabel(definition.type),
|
||||
properties: {
|
||||
name: definition.name,
|
||||
filePath,
|
||||
type: definition.type,
|
||||
startLine: definition.startLine,
|
||||
endLine: definition.endLine,
|
||||
qualifiedName: `${filePath}:${definition.name}`,
|
||||
parameters: definition.parameters,
|
||||
returnType: definition.returnType,
|
||||
accessibility: definition.accessibility,
|
||||
isStatic: definition.isStatic,
|
||||
isAsync: definition.isAsync,
|
||||
parentClass: definition.parentClass,
|
||||
docstring: definition.docstring
|
||||
}
|
||||
};
|
||||
|
||||
graph.addNode(node);
|
||||
|
||||
// Add to function registry if applicable
|
||||
if (['function', 'method', 'class', 'interface', 'enum'].includes(definition.type)) {
|
||||
const functionDef: FunctionDefinition = {
|
||||
nodeId: nodeId,
|
||||
qualifiedName: `${filePath}:${definition.name}`,
|
||||
filePath,
|
||||
functionName: definition.name,
|
||||
type: definition.type as 'function' | 'method' | 'class' | 'interface' | 'enum',
|
||||
};
|
||||
this.functionTrie.addDefinition(functionDef);
|
||||
}
|
||||
|
||||
// Find existing file node created by StructureProcessor (same as single-threaded)
|
||||
let fileNode = graph.nodes.find(node =>
|
||||
node.label === 'File' &&
|
||||
(node.properties.filePath === filePath || node.properties.path === filePath)
|
||||
);
|
||||
|
||||
// If no existing file node found, create one (fallback)
|
||||
if (!fileNode) {
|
||||
fileNode = {
|
||||
id: generateDeterministicId('file', filePath),
|
||||
label: 'File' as NodeLabel,
|
||||
properties: {
|
||||
name: pathUtils.getFileName(filePath),
|
||||
path: filePath,
|
||||
filePath: filePath,
|
||||
language: this.detectLanguage(filePath)
|
||||
}
|
||||
};
|
||||
graph.addNode(fileNode);
|
||||
}
|
||||
|
||||
// Add DEFINES relationship from file to definition (same as single-threaded)
|
||||
const definesRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('defines', `${fileNode.id}-${nodeId}`),
|
||||
type: 'DEFINES' as RelationshipType,
|
||||
source: fileNode.id,
|
||||
target: nodeId,
|
||||
properties: {
|
||||
filePath: filePath,
|
||||
line_number: definition.startLine
|
||||
}
|
||||
};
|
||||
|
||||
graph.addRelationship(definesRelationship);
|
||||
|
||||
// Add additional relationships like single-threaded version
|
||||
if (definition.extends && definition.extends.length > 0) {
|
||||
definition.extends.forEach((extendedClass) => {
|
||||
const extendsRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('extends', `${nodeId}-${extendedClass}`),
|
||||
type: 'EXTENDS' as RelationshipType,
|
||||
source: nodeId,
|
||||
target: generateDeterministicId('class', extendedClass),
|
||||
properties: {}
|
||||
};
|
||||
|
||||
graph.addRelationship(extendsRelationship);
|
||||
});
|
||||
}
|
||||
|
||||
if (definition.implements && definition.implements.length > 0) {
|
||||
definition.implements.forEach((implementedInterface) => {
|
||||
const implementsRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('implements', `${nodeId}-${implementedInterface}`),
|
||||
type: 'IMPLEMENTS' as RelationshipType,
|
||||
source: nodeId,
|
||||
target: generateDeterministicId('interface', implementedInterface),
|
||||
properties: {}
|
||||
};
|
||||
|
||||
graph.addRelationship(implementsRelationship);
|
||||
});
|
||||
}
|
||||
|
||||
if (definition.importPath) {
|
||||
const importRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('imports', `${nodeId}-${definition.importPath}`),
|
||||
type: 'IMPORTS' as RelationshipType,
|
||||
source: nodeId,
|
||||
target: generateDeterministicId('file', definition.importPath || 'unknown'),
|
||||
properties: {
|
||||
importPath: definition.importPath
|
||||
}
|
||||
};
|
||||
graph.addRelationship(importRelationship);
|
||||
}
|
||||
|
||||
if (definition.parentClass) {
|
||||
const parentRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('belongs_to', `${nodeId}-${definition.parentClass}`),
|
||||
type: 'BELONGS_TO' as RelationshipType,
|
||||
source: nodeId,
|
||||
target: generateDeterministicId('class', definition.parentClass || 'unknown'),
|
||||
properties: { parentClass: definition.parentClass }
|
||||
};
|
||||
graph.addRelationship(parentRelationship);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Map definition type to node label
|
||||
*/
|
||||
private mapDefinitionTypeToNodeLabel(type: string): NodeLabel {
|
||||
switch (type) {
|
||||
case 'function':
|
||||
return 'Function';
|
||||
case 'class':
|
||||
return 'Class';
|
||||
case 'method':
|
||||
return 'Method';
|
||||
case 'variable':
|
||||
return 'Variable';
|
||||
case 'import':
|
||||
return 'Import';
|
||||
case 'interface':
|
||||
return 'Interface';
|
||||
case 'type':
|
||||
return 'Type';
|
||||
case 'decorator':
|
||||
return 'Decorator';
|
||||
default:
|
||||
return 'CodeElement';
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
private applyFiltering(
|
||||
filePaths: string[],
|
||||
fileContents: Map<string, string>,
|
||||
options?: { directoryFilter?: string; fileExtensions?: string }): string[] {
|
||||
|
||||
let filtered = filePaths;
|
||||
|
||||
// Apply directory filter if specified
|
||||
if (options?.directoryFilter) {
|
||||
filtered = filtered.filter(path => path.includes(options.directoryFilter ?? ''));
|
||||
}
|
||||
|
||||
// Apply extension filter if specified
|
||||
if (options?.fileExtensions) {
|
||||
const extensions = options.fileExtensions.split(',').map(ext => ext.trim()).filter(ext => ext.length);
|
||||
filtered = filtered.filter(path => extensions.some(ext => path.endsWith(ext)));
|
||||
}
|
||||
|
||||
// Apply centralized ignore patterns
|
||||
filtered = ignoreService.filterPaths(filtered);
|
||||
|
||||
// Apply content filter (only exclude truly empty files)
|
||||
const emptyFiles: string[] = [];
|
||||
filtered = filtered.filter(path => {
|
||||
const content = fileContents.get(path);
|
||||
if (!content || content.trim().length === 0) {
|
||||
emptyFiles.push(path);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
|
||||
return filtered;
|
||||
}
|
||||
|
||||
private isSourceFile(filePath: string): boolean {
|
||||
// Only include actual programming language source files (EXACT MATCH TO SINGLE-THREADED)
|
||||
const sourceExtensions = [
|
||||
// JavaScript/TypeScript (core web technologies)
|
||||
'.js', '.ts', '.jsx', '.tsx',
|
||||
// Python
|
||||
'.py',
|
||||
// Java
|
||||
'.java',
|
||||
// C/C++
|
||||
'.cpp', '.c', '.cc', '.cxx', '.h', '.hpp', '.hxx',
|
||||
// C#
|
||||
'.cs',
|
||||
// Only include other languages if they're commonly used
|
||||
'.php', '.rb', '.go', '.rs'
|
||||
// Removed: .mjs, .cjs (might be build artifacts)
|
||||
// Removed: .html, .htm, .xml (markup, not source code)
|
||||
// Removed: .vue, .svelte (framework-specific)
|
||||
// Removed: .kt, .scala, .swift (less common)
|
||||
];
|
||||
return sourceExtensions.some(ext => filePath.toLowerCase().endsWith(ext));
|
||||
}
|
||||
|
||||
private isConfigFile(filePath: string): boolean {
|
||||
// Only include config files that might contain meaningful definitions (EXACT MATCH TO SINGLE-THREADED)
|
||||
const configFiles = [
|
||||
'package.json', 'tsconfig.json', 'jsconfig.json',
|
||||
'webpack.config.js', 'vite.config.ts', 'vite.config.js',
|
||||
'.eslintrc.js', '.eslintrc.json',
|
||||
'babel.config.js', 'rollup.config.js'
|
||||
// Removed: .prettierrc (formatting, no definitions)
|
||||
// Removed: pyproject.toml, setup.py (might be worth including if Python project)
|
||||
// Removed: requirements.txt (just dependencies)
|
||||
// Removed: Dockerfile, docker-compose.yml (deployment, not source)
|
||||
// Removed: .gitignore, .gitattributes (git config, no definitions)
|
||||
// Removed: README.md, LICENSE (documentation, no definitions)
|
||||
];
|
||||
const configExtensions = ['.json']; // Only JSON configs, removed .yaml, .yml, .toml, .ini, .cfg
|
||||
|
||||
return configFiles.some(name => filePath.endsWith(name)) ||
|
||||
configExtensions.some(ext => filePath.toLowerCase().endsWith(ext));
|
||||
}
|
||||
|
||||
/**
|
||||
* Shutdown the worker pool and cleanup resources
|
||||
*/
|
||||
public async shutdown(): Promise<void> {
|
||||
try {
|
||||
console.log('ParallelParsingProcessor: Starting shutdown...');
|
||||
|
||||
// Stop memory monitoring
|
||||
if (this.memoryMonitorInterval) {
|
||||
clearInterval(this.memoryMonitorInterval);
|
||||
this.memoryMonitorInterval = null;
|
||||
}
|
||||
|
||||
// Shutdown worker pool
|
||||
if (this.workerPool) {
|
||||
await this.workerPool.shutdown();
|
||||
}
|
||||
|
||||
// Clear large data structures to free memory
|
||||
this.astMap.clear();
|
||||
this.processedFiles.clear();
|
||||
this.functionTrie = new FunctionRegistryTrie(); // Reset trie
|
||||
|
||||
|
||||
// Clear language parsers
|
||||
this.languageParsers.clear();
|
||||
|
||||
// Reset parser
|
||||
if (this.parser) {
|
||||
try {
|
||||
this.parser.delete();
|
||||
} catch (error) {
|
||||
console.warn('Error deleting parser:', error);
|
||||
}
|
||||
this.parser = null;
|
||||
}
|
||||
|
||||
this.isInitialized = false;
|
||||
console.log('ParallelParsingProcessor: Shutdown complete');
|
||||
} catch (error) {
|
||||
console.error('ParallelParsingProcessor: Error during shutdown:', error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get worker pool statistics
|
||||
*/
|
||||
public getWorkerPoolStats() {
|
||||
return this.workerPool ? this.workerPool.getStats() : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Start memory monitoring to prevent memory leaks
|
||||
*/
|
||||
private startMemoryMonitoring(): void {
|
||||
// Monitor memory every 30 seconds
|
||||
this.memoryMonitorInterval = setInterval(() => {
|
||||
this.monitorMemoryUsage();
|
||||
}, 30000);
|
||||
}
|
||||
|
||||
/**
|
||||
* Monitor memory usage and trigger cleanup if needed
|
||||
*/
|
||||
private monitorMemoryUsage(): void {
|
||||
try {
|
||||
const memoryStats = this.memoryManager.getStats();
|
||||
const astMapSize = this.astMap.size;
|
||||
const processedFilesSize = this.processedFiles.size;
|
||||
|
||||
console.log(`ParallelParsingProcessor Memory Stats:
|
||||
- Memory Manager: ${memoryStats.usedMemoryMB}MB used, ${memoryStats.fileCount} files cached
|
||||
- AST Map: ${astMapSize} entries
|
||||
- Processed Files: ${processedFilesSize} entries`);
|
||||
|
||||
// If AST map is getting too large, clean up old entries
|
||||
if (astMapSize > this.MAX_AST_MAP_SIZE) {
|
||||
console.warn(`AST map size (${astMapSize}) exceeds limit (${this.MAX_AST_MAP_SIZE}), cleaning up...`);
|
||||
this.cleanupASTMap();
|
||||
}
|
||||
|
||||
// If memory usage is high, trigger memory manager cleanup
|
||||
if (memoryStats.usedMemoryMB > 500) { // 500MB threshold
|
||||
console.warn(`High memory usage detected (${memoryStats.usedMemoryMB}MB), triggering cleanup...`);
|
||||
this.memoryManager.clearCache();
|
||||
|
||||
}
|
||||
|
||||
// Monitor worker pool memory usage
|
||||
// Memory monitoring is now handled by the MemoryManager
|
||||
console.log('💾 Memory monitoring active via MemoryManager');
|
||||
|
||||
} catch (error) {
|
||||
console.warn('Error monitoring memory usage:', error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Clean up old AST map entries to prevent memory bloat
|
||||
*/
|
||||
private cleanupASTMap(): void {
|
||||
try {
|
||||
const entries = Array.from(this.astMap.entries());
|
||||
const toRemove = entries.length - Math.floor(this.MAX_AST_MAP_SIZE * 0.8); // Keep 80% of max
|
||||
|
||||
if (toRemove > 0) {
|
||||
// Remove oldest entries (assuming they're less likely to be needed)
|
||||
const keysToRemove = entries.slice(0, toRemove).map(([key]) => key);
|
||||
keysToRemove.forEach(key => this.astMap.delete(key));
|
||||
|
||||
console.log(`Cleaned up ${toRemove} AST map entries`);
|
||||
}
|
||||
} catch (error) {
|
||||
console.warn('Error cleaning up AST map:', error);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Get diagnostic information about parsing results
|
||||
*/
|
||||
public getDiagnosticInfo(): {
|
||||
processedFiles: number;
|
||||
skippedFiles: number;
|
||||
totalDefinitions: number;
|
||||
definitionsByType: Record<string, number>;
|
||||
definitionsByFile: Record<string, number>;
|
||||
processingErrors: string[];
|
||||
} {
|
||||
// For now, return basic stats - can be enhanced later
|
||||
return {
|
||||
processedFiles: this.processedFiles.size,
|
||||
skippedFiles: 0, // Would need to track this during processing
|
||||
totalDefinitions: 0, // Would need to track this during processing
|
||||
definitionsByType: {}, // Would need to track this during processing
|
||||
definitionsByFile: {}, // Would need to track this during processing
|
||||
processingErrors: [] // Would need to track errors during processing
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Method to analyze a specific file (simplified version for compatibility)
|
||||
*/
|
||||
public async analyzeFile(filePath: string, content: string): Promise<{
|
||||
language: string;
|
||||
isSourceFile: boolean;
|
||||
isConfigFile: boolean;
|
||||
isCompiled: boolean;
|
||||
contentLength: number;
|
||||
queryResults: Record<string, number>;
|
||||
extractionIssues: string[];
|
||||
}> {
|
||||
const language = this.detectLanguage(filePath);
|
||||
const isSourceFile = this.isSourceFile(filePath);
|
||||
const isConfigFile = this.isConfigFile(filePath);
|
||||
const extractionIssues: string[] = [];
|
||||
|
||||
if (!isSourceFile && !isConfigFile) {
|
||||
extractionIssues.push('File is not recognized as a source or config file');
|
||||
}
|
||||
|
||||
if (content.trim().length === 0) {
|
||||
extractionIssues.push('File is empty');
|
||||
}
|
||||
|
||||
return {
|
||||
language,
|
||||
isSourceFile,
|
||||
isConfigFile,
|
||||
isCompiled: false, // Simplified - would need proper detection
|
||||
contentLength: content.length,
|
||||
queryResults: {}, // Would need worker-based analysis for full results
|
||||
extractionIssues
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect programming language from file path
|
||||
*/
|
||||
private detectLanguage(filePath: string): string {
|
||||
const ext = filePath.split('.').pop()?.toLowerCase();
|
||||
switch (ext) {
|
||||
case 'ts':
|
||||
case 'tsx':
|
||||
return 'typescript';
|
||||
case 'js':
|
||||
case 'jsx':
|
||||
case 'mjs':
|
||||
return 'javascript';
|
||||
case 'py':
|
||||
return 'python';
|
||||
case 'java':
|
||||
return 'java';
|
||||
case 'cpp':
|
||||
case 'cc':
|
||||
case 'cxx':
|
||||
return 'cpp';
|
||||
case 'c':
|
||||
return 'c';
|
||||
case 'h':
|
||||
case 'hpp':
|
||||
return 'c'; // Treat headers as C for now
|
||||
default:
|
||||
return 'unknown';
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Initialize main thread parser for AST recreation
|
||||
*/
|
||||
private async initializeMainThreadParser(): Promise<void> {
|
||||
if (this.parser) return;
|
||||
|
||||
this.parser = await initTreeSitter();
|
||||
|
||||
const languageLoaders = {
|
||||
typescript: loadTypeScriptParser,
|
||||
javascript: loadJavaScriptParser,
|
||||
python: loadPythonParser,
|
||||
};
|
||||
|
||||
for (const [lang, loader] of Object.entries(languageLoaders)) {
|
||||
try {
|
||||
const languageParser = await loader();
|
||||
this.languageParsers.set(lang, languageParser);
|
||||
} catch (error) {
|
||||
console.error(`Failed to load ${lang} parser:`, error);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Recreate full AST in main thread (needed for import/call processors)
|
||||
*/
|
||||
private async recreateAST(filePath: string, fileContents: Map<string, string>): Promise<void> {
|
||||
const content = fileContents.get(filePath);
|
||||
if (!content || !this.parser) return;
|
||||
|
||||
const language = this.detectLanguage(filePath);
|
||||
const langParser = this.languageParsers.get(language);
|
||||
|
||||
if (!langParser) return;
|
||||
|
||||
try {
|
||||
this.parser.setLanguage(langParser);
|
||||
const tree = this.parser.parse(content);
|
||||
|
||||
// Store in AST map for import/call processors
|
||||
this.astMap.set(filePath, { tree });
|
||||
|
||||
// Clean up AST map if it gets too large
|
||||
if (this.astMap.size > this.MAX_AST_MAP_SIZE) {
|
||||
this.cleanupASTMap();
|
||||
}
|
||||
} catch (error) {
|
||||
console.error(`Failed to recreate AST for ${filePath}:`, error);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,303 +0,0 @@
|
||||
import { SimpleKnowledgeGraph } from '../graph/graph.js';
|
||||
import type { KnowledgeGraph } from '../graph/types.ts';
|
||||
import { StructureProcessor } from './structure-processor.ts';
|
||||
import { ParallelParsingProcessor } from './parallel-parsing-processor.ts';
|
||||
import { ImportProcessor } from './import-processor.ts';
|
||||
import { CallProcessor } from './call-processor.ts';
|
||||
import { WebWorkerPoolUtils } from '../../lib/web-worker-pool.js';
|
||||
import { isKuzuDBEnabled } from '../../config/features.ts';
|
||||
|
||||
export interface PipelineInput {
|
||||
projectRoot: string;
|
||||
projectName: string;
|
||||
filePaths: string[];
|
||||
fileContents: Map<string, string>;
|
||||
options?: {
|
||||
directoryFilter?: string;
|
||||
fileExtensions?: string;
|
||||
useParallelProcessing?: boolean;
|
||||
maxWorkers?: number;
|
||||
};
|
||||
}
|
||||
|
||||
export interface PipelineProgress {
|
||||
phase: 'structure' | 'parsing' | 'imports' | 'calls';
|
||||
message: string;
|
||||
progress: number;
|
||||
timestamp: number;
|
||||
}
|
||||
|
||||
export class ParallelGraphPipeline {
|
||||
private structureProcessor: StructureProcessor;
|
||||
private parsingProcessor: ParallelParsingProcessor;
|
||||
private importProcessor: ImportProcessor;
|
||||
private callProcessor!: CallProcessor;
|
||||
private progressCallback?: (progress: PipelineProgress) => void;
|
||||
|
||||
constructor() {
|
||||
this.structureProcessor = new StructureProcessor();
|
||||
this.parsingProcessor = new ParallelParsingProcessor();
|
||||
this.importProcessor = new ImportProcessor();
|
||||
}
|
||||
|
||||
/**
|
||||
* Set progress callback
|
||||
*/
|
||||
public setProgressCallback(callback: (progress: PipelineProgress) => void): void {
|
||||
this.progressCallback = callback;
|
||||
}
|
||||
|
||||
/**
|
||||
* Update progress
|
||||
*/
|
||||
private updateProgress(phase: PipelineProgress['phase'], message: string, progress: number): void {
|
||||
if (this.progressCallback) {
|
||||
this.progressCallback({
|
||||
phase,
|
||||
message,
|
||||
progress: Math.min(progress, 100),
|
||||
timestamp: Date.now()
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
public async run(input: PipelineInput): Promise<KnowledgeGraph> {
|
||||
const { projectRoot, projectName, filePaths, fileContents, options } = input;
|
||||
|
||||
// Create appropriate graph implementation based on feature flags
|
||||
const graph = await this.createGraph();
|
||||
const startTime = performance.now();
|
||||
|
||||
console.log(`🚀 Starting parallel 4-pass ingestion for project: ${projectName}`);
|
||||
console.log(`📊 Processing ${filePaths.length} files with ${options?.useParallelProcessing ? 'parallel' : 'sequential'} processing`);
|
||||
|
||||
try {
|
||||
// Pass 1: Structure Analysis (Sequential - lightweight)
|
||||
console.log('📁 Pass 1: Analyzing project structure...');
|
||||
this.updateProgress('structure', 'Analyzing project structure...', 0);
|
||||
|
||||
await this.structureProcessor.process(graph, {
|
||||
projectRoot,
|
||||
projectName,
|
||||
filePaths
|
||||
});
|
||||
|
||||
this.updateProgress('structure', 'Project structure analysis complete', 100);
|
||||
|
||||
// Pass 2: Code Parsing and Definition Extraction (Parallel - CPU intensive)
|
||||
console.log('🔍 Pass 2: Parsing code and extracting definitions (parallel)...');
|
||||
this.updateProgress('parsing', 'Initializing parallel parsing...', 0);
|
||||
|
||||
await this.parsingProcessor.process(graph, {
|
||||
filePaths,
|
||||
fileContents,
|
||||
options
|
||||
});
|
||||
|
||||
this.updateProgress('parsing', 'Parallel parsing complete', 100);
|
||||
|
||||
// Get AST map and function registry from parsing processor
|
||||
const astMap = this.parsingProcessor.getASTMap();
|
||||
const functionTrie = this.parsingProcessor.getFunctionRegistry();
|
||||
|
||||
this.callProcessor = new CallProcessor(functionTrie);
|
||||
|
||||
// Pass 3: Import Resolution (Sequential - depends on parsing results)
|
||||
console.log('🔗 Pass 3: Resolving imports and building dependency map...');
|
||||
this.updateProgress('imports', 'Resolving imports...', 0);
|
||||
|
||||
await this.importProcessor.process(graph, astMap, fileContents);
|
||||
|
||||
this.updateProgress('imports', 'Import resolution complete', 100);
|
||||
|
||||
// Pass 4: Call Resolution (Sequential - depends on import map)
|
||||
console.log('📞 Pass 4: Resolving function calls with 3-stage strategy...');
|
||||
this.updateProgress('calls', 'Resolving function calls...', 0);
|
||||
|
||||
const importMap = this.importProcessor.getImportMap();
|
||||
await this.callProcessor.process(graph, astMap, importMap);
|
||||
|
||||
this.updateProgress('calls', 'Call resolution complete', 100);
|
||||
|
||||
const endTime = performance.now();
|
||||
const totalDuration = endTime - startTime;
|
||||
|
||||
console.log(`✅ Parallel ingestion complete in ${totalDuration.toFixed(2)}ms`);
|
||||
console.log(`📊 Graph contains ${graph.nodes.length} nodes and ${graph.relationships.length} relationships`);
|
||||
|
||||
// Log performance statistics
|
||||
this.logPerformanceStats(graph, totalDuration);
|
||||
|
||||
// Log worker pool statistics if available
|
||||
const workerStats = this.parsingProcessor.getWorkerPoolStats();
|
||||
if (workerStats) {
|
||||
console.log('🔧 Worker Pool Statistics:', workerStats);
|
||||
}
|
||||
|
||||
// Handle post-processing for KuzuDB (batched mode only)
|
||||
if ('flushKuzuDB' in graph) {
|
||||
console.log('🔄 Flushing batched KuzuDB operations...');
|
||||
await (graph as any).flushKuzuDB();
|
||||
(graph as any).logDualWriteStats();
|
||||
}
|
||||
|
||||
console.log(`📈 Total entities: ${graph.nodes.length + graph.relationships.length}`);
|
||||
|
||||
return graph;
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Error in parallel pipeline:', error);
|
||||
// Ensure cleanup happens even on error
|
||||
await this.cleanup();
|
||||
throw error;
|
||||
} finally {
|
||||
// Final cleanup to ensure no resources leak
|
||||
await this.cleanup();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Log performance statistics
|
||||
*/
|
||||
private logPerformanceStats(graph: KnowledgeGraph, totalDuration: number): void {
|
||||
// Debug: Show graph structure
|
||||
const nodesByType = graph.nodes.reduce((acc, node) => {
|
||||
acc[node.label] = (acc[node.label] || 0) + 1;
|
||||
return acc;
|
||||
}, {} as Record<string, number>);
|
||||
|
||||
const relationshipsByType = graph.relationships.reduce((acc, rel) => {
|
||||
acc[rel.type] = (acc[rel.type] || 0) + 1;
|
||||
return acc;
|
||||
}, {} as Record<string, number>);
|
||||
|
||||
console.log('📊 Graph Statistics:');
|
||||
console.log('Nodes by type:', nodesByType);
|
||||
console.log('Relationships by type:', relationshipsByType);
|
||||
|
||||
// Debug: Find isolated nodes (nodes with no relationships)
|
||||
const connectedNodeIds = new Set<string>();
|
||||
graph.relationships.forEach(rel => {
|
||||
connectedNodeIds.add(rel.source);
|
||||
connectedNodeIds.add(rel.target);
|
||||
});
|
||||
|
||||
const isolatedNodes = graph.nodes.filter(node => !connectedNodeIds.has(node.id));
|
||||
if (isolatedNodes.length > 0) {
|
||||
console.warn(`⚠️ Found ${isolatedNodes.length} isolated nodes:`);
|
||||
const isolatedByType = isolatedNodes.reduce((acc, node) => {
|
||||
acc[node.label] = (acc[node.label] || 0) + 1;
|
||||
return acc;
|
||||
}, {} as Record<string, number>);
|
||||
console.log('Isolated nodes by type:', isolatedByType);
|
||||
}
|
||||
|
||||
// Performance metrics
|
||||
const totalNodes = graph.nodes.length;
|
||||
const totalRelationships = graph.relationships.length;
|
||||
const processingRate = totalDuration > 0 ? (totalNodes + totalRelationships) / (totalDuration / 1000) : 0;
|
||||
|
||||
console.log('⚡ Performance Metrics:');
|
||||
console.log(` Total processing time: ${totalDuration.toFixed(2)}ms`);
|
||||
console.log(` Processing rate: ${processingRate.toFixed(2)} entities/second`);
|
||||
console.log(` Average time per node: ${totalNodes > 0 ? (totalDuration / totalNodes).toFixed(2) : 0}ms`);
|
||||
console.log(` Average time per relationship: ${totalRelationships > 0 ? (totalDuration / totalRelationships).toFixed(2) : 0}ms`);
|
||||
}
|
||||
|
||||
/**
|
||||
* Cleanup resources
|
||||
*/
|
||||
private async cleanup(): Promise<void> {
|
||||
try {
|
||||
console.log('🧹 Cleaning up parallel pipeline resources...');
|
||||
|
||||
// Shutdown parsing processor (which includes worker pool)
|
||||
await this.parsingProcessor.shutdown();
|
||||
|
||||
// Cleanup singleton worker pool instances
|
||||
await WebWorkerPoolUtils.cleanupAllPools();
|
||||
|
||||
console.log('✅ Parallel pipeline cleanup complete');
|
||||
} catch (error) {
|
||||
console.warn('⚠️ Error during cleanup:', error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get worker pool statistics
|
||||
*/
|
||||
public getWorkerPoolStats() {
|
||||
return this.parsingProcessor.getWorkerPoolStats();
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if parallel processing is supported
|
||||
*/
|
||||
public static isParallelProcessingSupported(): boolean {
|
||||
return WebWorkerPoolUtils.isSupported();
|
||||
}
|
||||
|
||||
/**
|
||||
* Get optimal worker count for current system
|
||||
*/
|
||||
public static async getOptimalWorkerCount(): Promise<number> {
|
||||
// Worker count is now determined by configuration
|
||||
const { ConfigLoader } = await import('../../config/config-loader.ts');
|
||||
const { calculateWorkerCount } = await import('../../lib/worker-calculator.ts');
|
||||
|
||||
const config = await ConfigLoader.getInstance().loadConfig();
|
||||
const workerCalc = await calculateWorkerCount(config);
|
||||
|
||||
return workerCalc.workerCount;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get hardware concurrency
|
||||
*/
|
||||
public static getHardwareConcurrency(): number {
|
||||
return WebWorkerPoolUtils.getHardwareConcurrency();
|
||||
}
|
||||
|
||||
/**
|
||||
* Create appropriate graph implementation based on feature flags
|
||||
*/
|
||||
private async createGraph(): Promise<KnowledgeGraph> {
|
||||
console.log(`🔍 KuzuDB enabled check: ${isKuzuDBEnabled()}`);
|
||||
|
||||
if (isKuzuDBEnabled()) {
|
||||
try {
|
||||
console.log('🚀 Initializing KuzuDB integration...');
|
||||
|
||||
// Initialize KuzuDB query engine
|
||||
const { KuzuQueryEngine } = await import('../graph/kuzu-query-engine.ts');
|
||||
const queryEngine = new KuzuQueryEngine({
|
||||
enableCache: true,
|
||||
cacheSize: 1000,
|
||||
cacheTTL: 5 * 60 * 1000 // 5 minutes
|
||||
});
|
||||
|
||||
await queryEngine.initialize();
|
||||
|
||||
// Create KuzuDB knowledge graph
|
||||
const { KuzuKnowledgeGraph } = await import('../graph/kuzu-knowledge-graph.ts');
|
||||
const kuzuGraph = new KuzuKnowledgeGraph(queryEngine, {
|
||||
enableCache: true,
|
||||
batchSize: 100,
|
||||
autoCommit: false
|
||||
});
|
||||
|
||||
// Use dual-write mode (batched)
|
||||
const { DualWriteKnowledgeGraph } = await import('../graph/dual-write-knowledge-graph.ts');
|
||||
console.log('✅ KuzuDB integration initialized - using dual-write mode (batched)');
|
||||
return new DualWriteKnowledgeGraph(kuzuGraph);
|
||||
|
||||
} catch (error) {
|
||||
console.warn('❌ KuzuDB initialization failed, falling back to JSON-only mode:', error);
|
||||
return new SimpleKnowledgeGraph();
|
||||
}
|
||||
} else {
|
||||
console.log('📝 Using JSON-only storage mode');
|
||||
return new SimpleKnowledgeGraph();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,861 +1,124 @@
|
||||
import { GraphNode, GraphRelationship, NodeLabel, RelationshipType } from '../graph/types.js';
|
||||
import { MemoryManager } from '../../services/memory-manager.js';
|
||||
import { KnowledgeGraph, GraphProcessor } from '../graph/graph.js';
|
||||
import {
|
||||
pathUtils,
|
||||
OptimizedSet,
|
||||
DuplicateDetector,
|
||||
BatchProcessor
|
||||
} from '../../lib/shared-utils.js';
|
||||
import { ignoreService } from '../../config/ignore-service.js';
|
||||
import Parser from 'web-tree-sitter';
|
||||
import { TYPESCRIPT_QUERIES, JAVASCRIPT_QUERIES, PYTHON_QUERIES, JAVA_QUERIES } from './tree-sitter-queries';
|
||||
import { initTreeSitter, loadTypeScriptParser, loadPythonParser, loadJavaScriptParser } from '../tree-sitter/parser-loader.js';
|
||||
import { FunctionRegistryTrie, FunctionDefinition } from '../graph/trie.js';
|
||||
import { generateDeterministicId } from '../../lib/utils';
|
||||
import { KnowledgeGraph, GraphNode, GraphRelationship } from '../graph/types';
|
||||
import { loadParser, loadLanguage } from '../tree-sitter/parser-loader';
|
||||
import { LANGUAGE_QUERIES } from './tree-sitter-queries';
|
||||
import { generateId } from '../../lib/utils';
|
||||
import { SymbolTable } from './symbol-table';
|
||||
import { ASTCache } from './ast-cache';
|
||||
import { getLanguageFromFilename } from './utils';
|
||||
|
||||
export interface ParsingInput {
|
||||
filePaths: string[];
|
||||
fileContents: Map<string, string>;
|
||||
options?: { directoryFilter?: string; fileExtensions?: string };
|
||||
}
|
||||
export type FileProgressCallback = (current: number, total: number, filePath: string) => void;
|
||||
|
||||
export interface ParsedDefinition {
|
||||
name: string;
|
||||
type: 'function' | 'class' | 'method' | 'variable' | 'import' | 'interface' | 'type' | 'decorator';
|
||||
startLine: number;
|
||||
endLine?: number;
|
||||
parameters?: string[] | undefined;
|
||||
returnType?: string | undefined;
|
||||
accessibility?: 'public' | 'private' | 'protected';
|
||||
export const processParsing = async (
|
||||
graph: KnowledgeGraph,
|
||||
files: { path: string; content: string }[],
|
||||
symbolTable: SymbolTable,
|
||||
astCache: ASTCache,
|
||||
onFileProgress?: FileProgressCallback
|
||||
) => {
|
||||
|
||||
const parser = await loadParser();
|
||||
const total = files.length;
|
||||
|
||||
isStatic?: boolean | undefined;
|
||||
isAsync?: boolean | undefined;
|
||||
parentClass?: string | undefined;
|
||||
decorators?: string[] | undefined;
|
||||
extends?: string[] | undefined;
|
||||
implements?: string[] | undefined;
|
||||
importPath?: string | undefined;
|
||||
exportType?: 'named' | 'default' | 'namespace';
|
||||
docstring?: string | undefined;
|
||||
}
|
||||
|
||||
export interface ParsedAST {
|
||||
tree: Parser.Tree;
|
||||
}
|
||||
|
||||
|
||||
|
||||
export class ParsingProcessor implements GraphProcessor<ParsingInput> {
|
||||
private memoryManager: MemoryManager;
|
||||
private duplicateDetector = new DuplicateDetector<string>((item: string) => item);
|
||||
private processedFiles = new OptimizedSet<string>();
|
||||
private parser: Parser | null = null;
|
||||
private languageParsers: Map<string, Parser.Language> = new Map();
|
||||
private astMap: Map<string, ParsedAST> = new Map();
|
||||
private functionTrie: FunctionRegistryTrie = new FunctionRegistryTrie();
|
||||
|
||||
private stats = {
|
||||
nodesProcessed: 0,
|
||||
relationshipsProcessed: 0
|
||||
};
|
||||
|
||||
constructor() {
|
||||
this.memoryManager = MemoryManager.getInstance();
|
||||
}
|
||||
|
||||
public getASTMap(): Map<string, ParsedAST> {
|
||||
return this.astMap;
|
||||
}
|
||||
|
||||
public getFunctionRegistry(): FunctionRegistryTrie {
|
||||
return this.functionTrie;
|
||||
}
|
||||
|
||||
|
||||
public async process(graph: KnowledgeGraph, input: ParsingInput): Promise<void> {
|
||||
const { filePaths, fileContents, options } = input;
|
||||
|
||||
try {
|
||||
// Reset statistics
|
||||
this.stats = { nodesProcessed: 0, relationshipsProcessed: 0 };
|
||||
|
||||
console.log(`🔍 Starting parsing with KuzuDB dual-write for ${filePaths.length} files...`);
|
||||
|
||||
const memoryStats = this.memoryManager.getStats();
|
||||
const filteredFiles = this.applyFiltering(filePaths, fileContents, options);
|
||||
|
||||
const BATCH_SIZE = 10;
|
||||
const sourceFiles = filteredFiles.filter((path: string) => this.isSourceFile(path));
|
||||
const configFiles = filteredFiles.filter((path: string) => this.isConfigFile(path));
|
||||
const allProcessableFiles = [...sourceFiles, ...configFiles];
|
||||
|
||||
await this.initializeParser();
|
||||
|
||||
const batchProcessor = new BatchProcessor<string, void>(BATCH_SIZE, async (filePaths: string[]) => {
|
||||
for (const filePath of filePaths) {
|
||||
if (this.processedFiles.has(filePath)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const content = fileContents.get(filePath);
|
||||
if (!content) {
|
||||
continue;
|
||||
}
|
||||
|
||||
try {
|
||||
await this.parseFile(graph, filePath, content);
|
||||
this.processedFiles.add(filePath);
|
||||
} catch (error) {
|
||||
console.warn(`Failed to parse file ${filePath}:`, error);
|
||||
}
|
||||
}
|
||||
return [];
|
||||
});
|
||||
|
||||
await batchProcessor.processAll(allProcessableFiles);
|
||||
|
||||
console.log('✅ Parsing completed successfully');
|
||||
console.log(`📊 ParsingProcessor: ${this.stats.nodesProcessed} nodes, ${this.stats.relationshipsProcessed} relationships`);
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Parsing process failed:', error);
|
||||
throw error;
|
||||
} finally {
|
||||
// Cleanup resources (parser cleanup)
|
||||
if (this.parser) {
|
||||
this.parser.delete();
|
||||
this.parser = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private applyFiltering(
|
||||
filePaths: string[],
|
||||
fileContents: Map<string, string>,
|
||||
options?: { directoryFilter?: string; fileExtensions?: string }): string[] {
|
||||
|
||||
let filtered = filePaths;
|
||||
|
||||
// Apply directory filter if specified
|
||||
if (options?.directoryFilter) {
|
||||
filtered = filtered.filter(path => path.includes(options.directoryFilter ?? ''));
|
||||
}
|
||||
|
||||
// Apply extension filter if specified
|
||||
if (options?.fileExtensions) {
|
||||
const extensions = options.fileExtensions.split(',').map(ext => ext.trim()).filter(ext => ext.length);
|
||||
filtered = filtered.filter(path => extensions.some(ext => path.endsWith(ext)));
|
||||
}
|
||||
|
||||
// Apply centralized ignore patterns
|
||||
const beforeIgnoreFilter = filtered.length;
|
||||
filtered = ignoreService.filterPaths(filtered);
|
||||
|
||||
// Apply content filter (only exclude truly empty files)
|
||||
const beforeContentFilter = filtered.length;
|
||||
const emptyFiles: string[] = [];
|
||||
filtered = filtered.filter(path => {
|
||||
const content = fileContents.get(path);
|
||||
if (!content || content.trim().length === 0) {
|
||||
emptyFiles.push(path);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
|
||||
return filtered;
|
||||
}
|
||||
|
||||
private isSourceFile(filePath: string): boolean {
|
||||
// Only include actual programming language source files
|
||||
const sourceExtensions = [
|
||||
// JavaScript/TypeScript (core web technologies)
|
||||
'.js', '.ts', '.jsx', '.tsx',
|
||||
// Python
|
||||
'.py',
|
||||
// Java
|
||||
'.java',
|
||||
// C/C++
|
||||
'.cpp', '.c', '.cc', '.cxx', '.h', '.hpp', '.hxx',
|
||||
// C#
|
||||
'.cs',
|
||||
// Only include other languages if they're commonly used
|
||||
'.php', '.rb', '.go', '.rs'
|
||||
// Removed: .mjs, .cjs (might be build artifacts)
|
||||
// Removed: .html, .htm, .xml (markup, not source code)
|
||||
// Removed: .vue, .svelte (framework-specific)
|
||||
// Removed: .kt, .scala, .swift (less common)
|
||||
];
|
||||
return sourceExtensions.some(ext => filePath.toLowerCase().endsWith(ext));
|
||||
}
|
||||
|
||||
private isConfigFile(filePath: string): boolean {
|
||||
// Only include config files that might contain meaningful definitions
|
||||
const configFiles = [
|
||||
'package.json', 'tsconfig.json', 'jsconfig.json',
|
||||
'webpack.config.js', 'vite.config.ts', 'vite.config.js',
|
||||
'.eslintrc.js', '.eslintrc.json',
|
||||
'babel.config.js', 'rollup.config.js'
|
||||
// Removed: .prettierrc (formatting, no definitions)
|
||||
// Removed: pyproject.toml, setup.py (might be worth including if Python project)
|
||||
// Removed: requirements.txt (just dependencies)
|
||||
// Removed: Dockerfile, docker-compose.yml (deployment, not source)
|
||||
// Removed: .gitignore, .gitattributes (git config, no definitions)
|
||||
// Removed: README.md, LICENSE (documentation, no definitions)
|
||||
];
|
||||
const configExtensions = ['.json']; // Only JSON configs, removed .yaml, .yml, .toml, .ini, .cfg
|
||||
|
||||
return configFiles.some(name => filePath.endsWith(name)) ||
|
||||
configExtensions.some(ext => filePath.toLowerCase().endsWith(ext));
|
||||
}
|
||||
|
||||
private async initializeParser(): Promise<void> {
|
||||
if (this.parser) return;
|
||||
for (let i = 0; i < files.length; i++) {
|
||||
const file = files[i];
|
||||
|
||||
this.parser = await initTreeSitter();
|
||||
|
||||
const languageLoaders = {
|
||||
typescript: loadTypeScriptParser,
|
||||
javascript: loadJavaScriptParser,
|
||||
python: loadPythonParser,
|
||||
};
|
||||
|
||||
for (const [lang, loader] of Object.entries(languageLoaders)) {
|
||||
try {
|
||||
const languageParser = await loader();
|
||||
this.languageParsers.set(lang, languageParser);
|
||||
} catch (error) {
|
||||
console.error(`Failed to load ${lang} parser:`, error);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private async parseFile(graph: KnowledgeGraph, filePath: string, content: string): Promise<void> {
|
||||
const language = this.detectLanguage(filePath);
|
||||
const fileName = pathUtils.getFileName(filePath);
|
||||
// Report progress for each file
|
||||
onFileProgress?.(i + 1, total, file.path);
|
||||
|
||||
// Skip compiled/minified files for JavaScript
|
||||
if (language === 'javascript' && this.isCompiledOrMinified(content, filePath)) {
|
||||
await this.parseGenericFile(graph, filePath, content);
|
||||
return;
|
||||
}
|
||||
const language = getLanguageFromFilename(file.path);
|
||||
|
||||
if (!language) continue;
|
||||
|
||||
await loadLanguage(language, file.path);
|
||||
|
||||
|
||||
const langParser = this.languageParsers.get(language);
|
||||
|
||||
if (!langParser || !this.parser) {
|
||||
await this.parseGenericFile(graph, filePath, content);
|
||||
return;
|
||||
// 3. Parse the text content into an AST
|
||||
const tree = parser.parse(file.content);
|
||||
|
||||
// Store in cache immediately (this might evict an old one)
|
||||
astCache.set(file.path, tree);
|
||||
|
||||
// 4. Get the specific query string for this language
|
||||
const queryString = LANGUAGE_QUERIES[language];
|
||||
if (!queryString) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// 5. Run the query against the AST root node
|
||||
// This looks for patterns like (function_declaration)
|
||||
let query;
|
||||
let matches;
|
||||
try {
|
||||
this.parser.setLanguage(langParser);
|
||||
const tree = this.parser.parse(content);
|
||||
this.astMap.set(filePath, { tree });
|
||||
const definitions: ParsedDefinition[] = [];
|
||||
query = parser.getLanguage().query(queryString);
|
||||
matches = query.matches(tree.rootNode);
|
||||
} catch (queryError) {
|
||||
console.warn(`Query error for ${file.path}:`, queryError);
|
||||
continue;
|
||||
}
|
||||
|
||||
const queries = this.getQueriesForLanguage(language);
|
||||
if (!queries) {
|
||||
await this.parseGenericFile(graph, filePath, content);
|
||||
// 6. Process every match found
|
||||
matches.forEach(match => {
|
||||
const captureMap: Record<string, any> = {};
|
||||
|
||||
match.captures.forEach(c => {
|
||||
captureMap[c.name] = c.node;
|
||||
});
|
||||
|
||||
// Skip imports here - they are handled by import-processor.ts
|
||||
// which creates proper File -> IMPORTS -> File relationships
|
||||
if (captureMap['import']) {
|
||||
return;
|
||||
}
|
||||
|
||||
let totalMatches = 0;
|
||||
// Process queries
|
||||
for (const [queryName, queryString] of Object.entries(queries)) {
|
||||
let queryResults: Parser.QueryMatch[] = [];
|
||||
|
||||
try {
|
||||
const query = langParser.query(queryString as string);
|
||||
queryResults = query.matches(tree.rootNode);
|
||||
totalMatches += queryResults.length;
|
||||
|
||||
for (const match of queryResults) {
|
||||
for (const capture of match.captures) {
|
||||
const node = capture.node;
|
||||
const definition = this.extractDefinition(node, queryName, filePath);
|
||||
if (definition) {
|
||||
definitions.push(definition);
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (queryError) {
|
||||
// Removed verbose console log
|
||||
}
|
||||
// Skip call expressions - they are handled by call-processor.ts
|
||||
if (captureMap['call']) {
|
||||
return;
|
||||
}
|
||||
|
||||
const nameNode = captureMap['name'];
|
||||
if (!nameNode) return;
|
||||
|
||||
await this.addDefinitionsToGraph(graph, filePath, definitions);
|
||||
} catch (parseError) {
|
||||
await this.parseGenericFile(graph, filePath, content);
|
||||
}
|
||||
}
|
||||
|
||||
private extractDefinition(node: Parser.SyntaxNode, queryName: string, filePath: string): ParsedDefinition | null {
|
||||
let nameNode = node.childForFieldName('name');
|
||||
let name = nameNode ? nameNode.text : null;
|
||||
|
||||
// Handle different naming patterns for different query types
|
||||
if (!name) {
|
||||
// Try alternative naming strategies based on query type
|
||||
switch (queryName) {
|
||||
case 'variables':
|
||||
case 'constDeclarations':
|
||||
case 'global_variables':
|
||||
// For variable assignments, look for identifier in left side
|
||||
const leftChild = node.namedChildren.find(child => child.type === 'identifier');
|
||||
if (leftChild) name = leftChild.text;
|
||||
break;
|
||||
|
||||
case 'hookCalls':
|
||||
case 'hookDestructuring':
|
||||
// For React hooks, try to get the variable name
|
||||
const hookVar = node.namedChildren.find(child => child.type === 'variable_declarator');
|
||||
if (hookVar) {
|
||||
const hookName = hookVar.childForFieldName('name');
|
||||
if (hookName) {
|
||||
// Handle array destructuring for useState pattern
|
||||
if (hookName.type === 'array_pattern') {
|
||||
const elements = hookName.namedChildren.filter(child => child.type === 'identifier');
|
||||
if (elements.length > 0) {
|
||||
name = elements.map(el => el.text).join(', ');
|
||||
}
|
||||
} else {
|
||||
name = hookName.text;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
case 'reactComponents':
|
||||
case 'reactConstComponents':
|
||||
case 'defaultExportArrows':
|
||||
// For React components, get the component name
|
||||
const componentVar = node.namedChildren.find(child => child.type === 'variable_declarator');
|
||||
if (componentVar) {
|
||||
const componentName = componentVar.childForFieldName('name');
|
||||
if (componentName) name = componentName.text;
|
||||
}
|
||||
break;
|
||||
|
||||
case 'moduleExports':
|
||||
// For module.exports = something, get the property name
|
||||
const memberExpr = node.namedChildren.find(child => child.type === 'member_expression');
|
||||
if (memberExpr) {
|
||||
const property = memberExpr.childForFieldName('property');
|
||||
if (property) name = property.text;
|
||||
}
|
||||
break;
|
||||
|
||||
case 'decorators':
|
||||
// For decorators, get the decorator name
|
||||
const decoratorChild = node.namedChildren.find(child => child.type === 'identifier');
|
||||
if (decoratorChild) name = decoratorChild.text;
|
||||
break;
|
||||
|
||||
default:
|
||||
// Try to find any identifier child
|
||||
const identifierChild = node.namedChildren.find(child => child.type === 'identifier');
|
||||
if (identifierChild) name = identifierChild.text;
|
||||
}
|
||||
}
|
||||
|
||||
// Skip anonymous definitions - they're usually from compiled/minified code
|
||||
if (!name || name === 'anonymous' || name.trim().length === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Skip very short names that are likely noise (but keep single-letter variables like 'i', 'x')
|
||||
if (name.length === 1 && queryName !== 'variables' && queryName !== 'constDeclarations') {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Skip common noise patterns
|
||||
const noisePatterns = ['_', '__', '___', 'temp', 'tmp'];
|
||||
if (noisePatterns.includes(name.toLowerCase())) {
|
||||
return null;
|
||||
}
|
||||
|
||||
const definition: ParsedDefinition = {
|
||||
name,
|
||||
type: this.getDefinitionType(queryName),
|
||||
startLine: node.startPosition.row + 1,
|
||||
endLine: node.endPosition.row + 1,
|
||||
};
|
||||
|
||||
// Extract additional metadata based on definition type
|
||||
if (definition.type === 'function' || definition.type === 'method') {
|
||||
// Try to extract parameters
|
||||
const parametersNode = node.childForFieldName('parameters');
|
||||
if (parametersNode) {
|
||||
const params: string[] = [];
|
||||
for (const param of parametersNode.namedChildren) {
|
||||
if (param.type === 'identifier' || param.type === 'formal_parameter') {
|
||||
params.push(param.text);
|
||||
}
|
||||
}
|
||||
if (params.length > 0) {
|
||||
definition.parameters = params;
|
||||
}
|
||||
}
|
||||
const nodeName = nameNode.text;
|
||||
|
||||
// Check for async functions - removed async queries as they were causing Tree-sitter errors
|
||||
// if (queryName === 'async_functions' || queryName === 'async_methods') {
|
||||
// definition.isAsync = true;
|
||||
// }
|
||||
let nodeLabel = 'CodeElement';
|
||||
|
||||
// Mark React components
|
||||
if (queryName === 'reactComponents' || queryName === 'reactConstComponents') {
|
||||
definition.isAsync = false; // React components are not async by default
|
||||
definition.exportType = 'default'; // Most React components are default exports
|
||||
}
|
||||
}
|
||||
|
||||
if (definition.type === 'class') {
|
||||
// Try to extract inheritance information
|
||||
const superclassNode = node.childForFieldName('superclass');
|
||||
if (superclassNode) {
|
||||
definition.extends = [superclassNode.text];
|
||||
}
|
||||
}
|
||||
|
||||
// Handle variable types with additional context
|
||||
if (definition.type === 'variable') {
|
||||
if (queryName === 'hookCalls' || queryName === 'hookDestructuring') {
|
||||
definition.exportType = 'named'; // React hooks are typically named exports
|
||||
|
||||
// Try to extract hook type from call expression
|
||||
const callExpr = node.descendantsOfType('call_expression')[0];
|
||||
if (callExpr) {
|
||||
const funcNode = callExpr.childForFieldName('function');
|
||||
if (funcNode && funcNode.type === 'identifier') {
|
||||
definition.returnType = funcNode.text; // Store hook function name
|
||||
}
|
||||
if (captureMap['definition.function']) nodeLabel = 'Function';
|
||||
else if (captureMap['definition.class']) nodeLabel = 'Class';
|
||||
else if (captureMap['definition.interface']) nodeLabel = 'Interface';
|
||||
else if (captureMap['definition.method']) nodeLabel = 'Method';
|
||||
|
||||
const nodeId = generateId(nodeLabel, `${file.path}:${nodeName}`);
|
||||
|
||||
const node: GraphNode = {
|
||||
id: nodeId,
|
||||
label: nodeLabel as any,
|
||||
properties: {
|
||||
name: nodeName,
|
||||
filePath: file.path,
|
||||
startLine: nameNode.startPosition.row,
|
||||
endLine: nameNode.endPosition.row,
|
||||
language: language
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return definition;
|
||||
}
|
||||
};
|
||||
|
||||
private getDefinitionType(queryName: string): ParsedDefinition['type'] {
|
||||
switch (queryName) {
|
||||
case 'classes':
|
||||
case 'exportClasses': return 'class';
|
||||
case 'methods':
|
||||
case 'properties':
|
||||
case 'staticmethods':
|
||||
case 'classmethods': return 'method';
|
||||
case 'functions':
|
||||
case 'arrowFunctions':
|
||||
case 'reactComponents':
|
||||
case 'reactConstComponents':
|
||||
case 'defaultExportArrows':
|
||||
case 'variableAssignments':
|
||||
case 'objectMethods':
|
||||
case 'exportFunctions':
|
||||
case 'defaultExportFunctions':
|
||||
case 'functionExpressions': return 'function';
|
||||
case 'variables':
|
||||
case 'constDeclarations':
|
||||
case 'hookCalls':
|
||||
case 'hookDestructuring':
|
||||
case 'global_variables': return 'variable';
|
||||
case 'imports':
|
||||
case 'from_imports': return 'import';
|
||||
case 'exports':
|
||||
case 'defaultExports':
|
||||
case 'moduleExports': return 'function'; // Exports usually export functions
|
||||
case 'interfaces': return 'interface';
|
||||
case 'types': return 'type';
|
||||
case 'enums': return 'type';
|
||||
case 'decorators': return 'decorator';
|
||||
default:
|
||||
console.warn(`Unknown query type: ${queryName}, defaulting to 'function'`);
|
||||
return 'function'; // Better default than 'variable'
|
||||
}
|
||||
}
|
||||
graph.addNode(node);
|
||||
|
||||
private isCompiledOrMinified(content: string, filePath: string): boolean {
|
||||
// Check file name patterns for known compiled files
|
||||
const fileName = filePath.split('/').pop()?.toLowerCase() || '';
|
||||
if (fileName.includes('.min.') ||
|
||||
fileName.includes('.bundle.') ||
|
||||
fileName.includes('tree-sitter.js') ||
|
||||
fileName.includes('kuzu_wasm_worker.js')) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Check content characteristics for minified code
|
||||
const lines = content.split('\n');
|
||||
if (lines.length > 0) {
|
||||
const firstLine = lines[0];
|
||||
// Register in Symbol Table (only definitions, not imports)
|
||||
symbolTable.add(file.path, nodeName, nodeId, nodeLabel);
|
||||
|
||||
const fileId = generateId('File', file.path);
|
||||
|
||||
// Very long first line (typical of minified code)
|
||||
if (firstLine.length > 500) {
|
||||
return true;
|
||||
}
|
||||
const relId = generateId('DEFINES', `${fileId}->${nodeId}`);
|
||||
|
||||
// Contains typical minified patterns
|
||||
if (firstLine.includes('var Module=void 0!==Module?Module:{}') ||
|
||||
firstLine.includes('__webpack_require__') ||
|
||||
firstLine.includes('!function(') ||
|
||||
content.includes('/*! ') || // Webpack/build tool comments
|
||||
content.match(/^\s*!function\s*\(/)) { // IIFE patterns
|
||||
return true;
|
||||
}
|
||||
}
|
||||
const relationship: GraphRelationship = {
|
||||
id: relId,
|
||||
sourceId: fileId,
|
||||
targetId: nodeId,
|
||||
type: 'DEFINES'
|
||||
};
|
||||
|
||||
graph.addRelationship(relationship);
|
||||
});
|
||||
|
||||
return false;
|
||||
// Don't delete tree here - LRU cache handles cleanup when evicted
|
||||
}
|
||||
|
||||
private detectLanguage(filePath: string): string {
|
||||
const extension = pathUtils.extname(filePath).toLowerCase();
|
||||
|
||||
switch (extension) {
|
||||
case '.ts':
|
||||
case '.tsx':
|
||||
return 'typescript';
|
||||
case '.js':
|
||||
case '.jsx':
|
||||
return 'javascript';
|
||||
case '.py':
|
||||
return 'python';
|
||||
case '.java':
|
||||
return 'java';
|
||||
default:
|
||||
return 'generic';
|
||||
}
|
||||
}
|
||||
|
||||
private getQueriesForLanguage(language: string): Record<string, string> | null {
|
||||
switch (language) {
|
||||
case 'typescript':
|
||||
return TYPESCRIPT_QUERIES;
|
||||
case 'javascript':
|
||||
return JAVASCRIPT_QUERIES; // Use separate JavaScript queries
|
||||
case 'python':
|
||||
return PYTHON_QUERIES;
|
||||
case 'java':
|
||||
return JAVA_QUERIES;
|
||||
default:
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
private async parseGenericFile(graph: KnowledgeGraph, filePath: string, _content: string): Promise<void> {
|
||||
// Find existing file node created by StructureProcessor
|
||||
let fileNode = graph.nodes.find(node =>
|
||||
node.label === 'File' &&
|
||||
(node.properties.filePath === filePath || node.properties.path === filePath)
|
||||
);
|
||||
|
||||
// If no existing file node found, create one (fallback)
|
||||
if (!fileNode) {
|
||||
fileNode = {
|
||||
id: generateDeterministicId('file', filePath),
|
||||
label: 'File' as NodeLabel,
|
||||
properties: {
|
||||
name: pathUtils.getFileName(filePath),
|
||||
path: filePath,
|
||||
filePath: filePath,
|
||||
size: _content.length,
|
||||
language: this.detectLanguage(filePath)
|
||||
}
|
||||
};
|
||||
graph.addNode(fileNode);
|
||||
} else {
|
||||
// Update existing file node with additional properties
|
||||
fileNode.properties.size = _content.length;
|
||||
fileNode.properties.language = this.detectLanguage(filePath);
|
||||
}
|
||||
}
|
||||
|
||||
private async addDefinitionsToGraph(
|
||||
graph: KnowledgeGraph,
|
||||
filePath: string,
|
||||
definitions: ParsedDefinition[]
|
||||
): Promise<void> {
|
||||
// Find existing file node created by StructureProcessor
|
||||
let fileNode = graph.nodes.find(node =>
|
||||
node.label === 'File' &&
|
||||
(node.properties.filePath === filePath || node.properties.path === filePath)
|
||||
);
|
||||
|
||||
// If no existing file node found, create one (fallback) with dual-write
|
||||
if (!fileNode) {
|
||||
fileNode = {
|
||||
id: generateDeterministicId('file', filePath),
|
||||
label: 'File' as NodeLabel,
|
||||
properties: {
|
||||
name: pathUtils.getFileName(filePath),
|
||||
path: filePath,
|
||||
filePath: filePath,
|
||||
language: this.detectLanguage(filePath)
|
||||
}
|
||||
};
|
||||
graph.addNode(fileNode);
|
||||
this.stats.nodesProcessed++;
|
||||
}
|
||||
|
||||
for (const def of definitions) {
|
||||
// Generate unique ID based on file path and definition name
|
||||
const nodeId = generateDeterministicId(def.type, `${filePath}_${def.name}_${def.startLine}`);
|
||||
|
||||
if (this.duplicateDetector.checkAndMark(nodeId)) continue;
|
||||
|
||||
const node: GraphNode = {
|
||||
id: nodeId,
|
||||
label: this.getNodeLabelForType(def.type),
|
||||
properties: {
|
||||
name: def.name,
|
||||
type: def.type,
|
||||
startLine: def.startLine,
|
||||
endLine: def.endLine,
|
||||
parameters: def.parameters,
|
||||
returnType: def.returnType,
|
||||
accessibility: def.accessibility,
|
||||
isStatic: def.isStatic,
|
||||
isAsync: def.isAsync,
|
||||
parentClass: def.parentClass,
|
||||
decorators: def.decorators,
|
||||
extends: def.extends,
|
||||
implements: def.implements,
|
||||
importPath: def.importPath,
|
||||
exportType: def.exportType,
|
||||
docstring: def.docstring,
|
||||
filePath: filePath
|
||||
}
|
||||
};
|
||||
|
||||
graph.addNode(node);
|
||||
this.stats.nodesProcessed++;
|
||||
|
||||
if (def.type === 'function' || def.type === 'method' || def.type === 'class' || def.type === 'interface') {
|
||||
const functionDef: FunctionDefinition = {
|
||||
nodeId: nodeId,
|
||||
qualifiedName: `${filePath}:${def.name}`,
|
||||
filePath: filePath,
|
||||
functionName: def.name,
|
||||
type: def.type,
|
||||
startLine: def.startLine,
|
||||
endLine: def.endLine,
|
||||
};
|
||||
this.functionTrie.addDefinition(functionDef);
|
||||
}
|
||||
|
||||
const definesRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('defines', `${fileNode.id}-${node.id}`),
|
||||
type: 'DEFINES' as RelationshipType,
|
||||
source: fileNode.id,
|
||||
target: node.id,
|
||||
properties: {
|
||||
filePath: filePath,
|
||||
line_number: def.startLine
|
||||
}
|
||||
};
|
||||
|
||||
graph.addRelationship(definesRelationship);
|
||||
this.stats.relationshipsProcessed++;
|
||||
|
||||
if (def.extends && def.extends.length > 0) {
|
||||
for (const extendedClass of def.extends) {
|
||||
const extendsRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('extends', `${node.id}-${extendedClass}`),
|
||||
type: 'INHERITS' as RelationshipType,
|
||||
source: node.id,
|
||||
target: generateDeterministicId('class', extendedClass),
|
||||
properties: {}
|
||||
};
|
||||
|
||||
graph.addRelationship(extendsRelationship);
|
||||
this.stats.relationshipsProcessed++;
|
||||
}
|
||||
}
|
||||
|
||||
if (def.implements && def.implements.length > 0) {
|
||||
for (const implementedInterface of def.implements) {
|
||||
const implementsRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('implements', `${node.id}-${implementedInterface}`),
|
||||
type: 'IMPLEMENTS' as RelationshipType,
|
||||
source: node.id,
|
||||
target: generateDeterministicId('interface', implementedInterface),
|
||||
properties: {}
|
||||
};
|
||||
|
||||
graph.addRelationship(implementsRelationship);
|
||||
this.stats.relationshipsProcessed++;
|
||||
}
|
||||
}
|
||||
|
||||
if (def.importPath) {
|
||||
const importRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('imports', `${node.id}-${def.importPath}`),
|
||||
type: 'IMPORTS' as RelationshipType,
|
||||
source: node.id,
|
||||
target: generateDeterministicId('file', def.importPath || 'unknown'),
|
||||
properties: {
|
||||
importPath: def.importPath
|
||||
}
|
||||
};
|
||||
graph.addRelationship(importRelationship);
|
||||
this.stats.relationshipsProcessed++;
|
||||
}
|
||||
|
||||
if (def.parentClass) {
|
||||
const parentRelationship: GraphRelationship = {
|
||||
id: generateDeterministicId('belongs_to', `${node.id}-${def.parentClass}`),
|
||||
type: 'BELONGS_TO' as RelationshipType,
|
||||
source: node.id,
|
||||
target: generateDeterministicId('class', def.parentClass || 'unknown'),
|
||||
properties: {}
|
||||
};
|
||||
graph.addRelationship(parentRelationship);
|
||||
this.stats.relationshipsProcessed++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private getNodeLabelForType(type: string): NodeLabel {
|
||||
switch (type) {
|
||||
case 'class': return 'Class' as NodeLabel;
|
||||
case 'function': return 'Function' as NodeLabel;
|
||||
case 'method': return 'Method' as NodeLabel;
|
||||
case 'variable': return 'Variable' as NodeLabel;
|
||||
case 'import': return 'Import' as NodeLabel;
|
||||
case 'interface': return 'Interface' as NodeLabel;
|
||||
case 'type': return 'Type' as NodeLabel;
|
||||
case 'decorator': return 'Decorator' as NodeLabel;
|
||||
default: return 'CodeElement' as NodeLabel;
|
||||
}
|
||||
}
|
||||
|
||||
private generateContentHash(content: string): string {
|
||||
let hash = 0;
|
||||
for (let i = 0; i < content.length; i++) {
|
||||
const char = content.charCodeAt(i);
|
||||
hash = ((hash << 5) - hash) + char;
|
||||
hash = hash & hash;
|
||||
}
|
||||
return Math.abs(hash).toString(36);
|
||||
}
|
||||
|
||||
public reset(): void {
|
||||
this.processedFiles.clear();
|
||||
this.duplicateDetector.clear();
|
||||
this.memoryManager.clearCache();
|
||||
}
|
||||
|
||||
/**
|
||||
* Diagnostic method to analyze why files might not have definitions
|
||||
*/
|
||||
public getDiagnosticInfo(): {
|
||||
processedFiles: number;
|
||||
skippedFiles: number;
|
||||
totalDefinitions: number;
|
||||
definitionsByType: Record<string, number>;
|
||||
definitionsByFile: Record<string, number>;
|
||||
processingErrors: string[];
|
||||
} {
|
||||
const definitionsByType: Record<string, number> = {};
|
||||
const definitionsByFile: Record<string, number> = {};
|
||||
let totalDefinitions = 0;
|
||||
|
||||
// Analyze function registry
|
||||
const allDefs = this.functionTrie.getAllDefinitions();
|
||||
allDefs.forEach(def => {
|
||||
totalDefinitions++;
|
||||
definitionsByType[def.type] = (definitionsByType[def.type] || 0) + 1;
|
||||
definitionsByFile[def.filePath] = (definitionsByFile[def.filePath] || 0) + 1;
|
||||
});
|
||||
|
||||
return {
|
||||
processedFiles: this.processedFiles.size,
|
||||
skippedFiles: 0, // Would need to track this during processing
|
||||
totalDefinitions,
|
||||
definitionsByType,
|
||||
definitionsByFile,
|
||||
processingErrors: [] // Would need to track errors during processing
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Method to analyze a specific file and explain why it might not have definitions
|
||||
*/
|
||||
public async analyzeFile(filePath: string, content: string): Promise<{
|
||||
language: string;
|
||||
isSourceFile: boolean;
|
||||
isConfigFile: boolean;
|
||||
isCompiled: boolean;
|
||||
contentLength: number;
|
||||
queryResults: Record<string, number>;
|
||||
extractionIssues: string[];
|
||||
}> {
|
||||
const language = this.detectLanguage(filePath);
|
||||
const isSourceFile = this.isSourceFile(filePath);
|
||||
const isConfigFile = this.isConfigFile(filePath);
|
||||
const isCompiled = language === 'javascript' && this.isCompiledOrMinified(content, filePath);
|
||||
const extractionIssues: string[] = [];
|
||||
const queryResults: Record<string, number> = {};
|
||||
|
||||
if (!isSourceFile && !isConfigFile) {
|
||||
extractionIssues.push('File is not recognized as a source or config file');
|
||||
}
|
||||
|
||||
if (isCompiled) {
|
||||
extractionIssues.push('File appears to be compiled/minified and is skipped');
|
||||
}
|
||||
|
||||
if (content.trim().length === 0) {
|
||||
extractionIssues.push('File is empty');
|
||||
}
|
||||
|
||||
const langParser = this.languageParsers.get(language);
|
||||
if (!langParser || !this.parser) {
|
||||
extractionIssues.push(`No parser available for language: ${language}`);
|
||||
} else {
|
||||
try {
|
||||
this.parser.setLanguage(langParser);
|
||||
const tree = this.parser.parse(content);
|
||||
|
||||
const queries = this.getQueriesForLanguage(language);
|
||||
if (queries) {
|
||||
for (const [queryName, queryString] of Object.entries(queries)) {
|
||||
try {
|
||||
const query = langParser.query(queryString as string);
|
||||
const matches = query.matches(tree.rootNode);
|
||||
queryResults[queryName] = matches.length;
|
||||
} catch (queryError) {
|
||||
extractionIssues.push(`Query '${queryName}' failed: ${queryError}`);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
extractionIssues.push(`No queries available for language: ${language}`);
|
||||
}
|
||||
} catch (parseError) {
|
||||
extractionIssues.push(`Parse error: ${parseError}`);
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
language,
|
||||
isSourceFile,
|
||||
isConfigFile,
|
||||
isCompiled,
|
||||
contentLength: content.length,
|
||||
queryResults,
|
||||
extractionIssues
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Get processing statistics
|
||||
*/
|
||||
public getStats() {
|
||||
return {
|
||||
nodesProcessed: this.stats.nodesProcessed,
|
||||
relationshipsProcessed: this.stats.relationshipsProcessed
|
||||
};
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
+173
-322
@@ -1,333 +1,184 @@
|
||||
import { SimpleKnowledgeGraph } from '../graph/graph.js';
|
||||
import type { KnowledgeGraph } from '../graph/types.ts';
|
||||
import { StructureProcessor } from './structure-processor.ts';
|
||||
import { ParsingProcessor } from './parsing-processor.ts';
|
||||
import { ParallelParsingProcessor } from './parallel-parsing-processor.ts';
|
||||
import { ImportProcessor } from './import-processor.ts';
|
||||
import { CallProcessor } from './call-processor.ts';
|
||||
import { isParallelParsingEnabled, isKuzuDBEnabled } from '../../config/features.ts';
|
||||
import { createKnowledgeGraph } from '../graph/graph';
|
||||
import { extractZip, FileEntry } from '../../services/zip';
|
||||
import { processStructure } from './structure-processor';
|
||||
import { processParsing } from './parsing-processor';
|
||||
import { processImports, createImportMap } from './import-processor';
|
||||
import { processCalls } from './call-processor';
|
||||
import { processHeritage } from './heritage-processor';
|
||||
import { createSymbolTable } from './symbol-table';
|
||||
import { createASTCache } from './ast-cache';
|
||||
import { PipelineProgress, PipelineResult } from '../../types/pipeline';
|
||||
|
||||
export interface PipelineInput {
|
||||
projectRoot: string;
|
||||
projectName: string;
|
||||
filePaths: string[];
|
||||
fileContents: Map<string, string>;
|
||||
options?: {
|
||||
directoryFilter?: string;
|
||||
fileExtensions?: string;
|
||||
/**
|
||||
* Run the ingestion pipeline from a ZIP file
|
||||
*/
|
||||
export const runIngestionPipeline = async ( file: File, onProgress: (progress: PipelineProgress) => void): Promise<PipelineResult> => {
|
||||
// Phase 1: Extracting (0-15%)
|
||||
onProgress({
|
||||
phase: 'extracting',
|
||||
percent: 0,
|
||||
message: 'Extracting ZIP file...',
|
||||
});
|
||||
|
||||
// Fake progress for extraction (JSZip doesn't expose progress)
|
||||
const fakeExtractionProgress = setInterval(() => {
|
||||
onProgress({
|
||||
phase: 'extracting',
|
||||
percent: Math.min(14, Math.random() * 10 + 5),
|
||||
message: 'Extracting ZIP file...',
|
||||
});
|
||||
}, 200);
|
||||
|
||||
const files = await extractZip(file);
|
||||
clearInterval(fakeExtractionProgress);
|
||||
|
||||
// Continue with common pipeline
|
||||
return runPipelineFromFiles(files, onProgress);
|
||||
};
|
||||
|
||||
/**
|
||||
* Run the ingestion pipeline from pre-extracted files (e.g., from git clone)
|
||||
*/
|
||||
export const runPipelineFromFiles = async (
|
||||
files: FileEntry[],
|
||||
onProgress: (progress: PipelineProgress) => void
|
||||
): Promise<PipelineResult> => {
|
||||
const graph = createKnowledgeGraph();
|
||||
const fileContents = new Map<string, string>();
|
||||
const symbolTable = createSymbolTable();
|
||||
const astCache = createASTCache(50); // Keep last 50 files hot
|
||||
const importMap = createImportMap();
|
||||
|
||||
// Cleanup function for error handling
|
||||
const cleanup = () => {
|
||||
astCache.clear();
|
||||
symbolTable.clear();
|
||||
};
|
||||
}
|
||||
|
||||
export class GraphPipeline {
|
||||
private structureProcessor: StructureProcessor;
|
||||
private parsingProcessor: ParsingProcessor | ParallelParsingProcessor;
|
||||
private importProcessor: ImportProcessor;
|
||||
private callProcessor!: CallProcessor;
|
||||
|
||||
constructor() {
|
||||
this.structureProcessor = new StructureProcessor();
|
||||
|
||||
// Choose parsing processor based on feature flag
|
||||
if (isParallelParsingEnabled()) {
|
||||
console.log('🚀 Using Parallel Processing (Multi-threaded with Workers)');
|
||||
this.parsingProcessor = new ParallelParsingProcessor();
|
||||
} else {
|
||||
console.log('🔄 Using Single-threaded Processing');
|
||||
this.parsingProcessor = new ParsingProcessor();
|
||||
}
|
||||
|
||||
this.importProcessor = new ImportProcessor();
|
||||
|
||||
}
|
||||
|
||||
public async run(input: PipelineInput): Promise<KnowledgeGraph> {
|
||||
const { projectRoot, projectName, filePaths, fileContents, options } = input;
|
||||
|
||||
// Create appropriate graph implementation based on feature flags
|
||||
const graph = await this.createGraph();
|
||||
|
||||
const processingMode = isParallelParsingEnabled() ? 'parallel' : 'single-threaded';
|
||||
console.log(`🚀 Starting 4-pass ingestion for project: ${projectName} (${processingMode} processing)`);
|
||||
|
||||
// Pass 1: Structure Analysis
|
||||
console.log('📁 Pass 1: Analyzing project structure...');
|
||||
await this.structureProcessor.process(graph, {
|
||||
projectRoot,
|
||||
projectName,
|
||||
filePaths
|
||||
|
||||
try {
|
||||
// Store file contents for code panel
|
||||
files.forEach(f => fileContents.set(f.path, f.content));
|
||||
|
||||
onProgress({
|
||||
phase: 'extracting',
|
||||
percent: 15,
|
||||
message: 'ZIP extracted successfully',
|
||||
stats: { filesProcessed: 0, totalFiles: files.length, nodesCreated: 0 },
|
||||
});
|
||||
|
||||
// Phase 2: Structure (15-30%)
|
||||
onProgress({
|
||||
phase: 'structure',
|
||||
percent: 15,
|
||||
message: 'Analyzing project structure...',
|
||||
stats: { filesProcessed: 0, totalFiles: files.length, nodesCreated: 0 },
|
||||
});
|
||||
|
||||
const filePaths = files.map(f => f.path);
|
||||
processStructure(graph, filePaths);
|
||||
|
||||
onProgress({
|
||||
phase: 'structure',
|
||||
percent: 30,
|
||||
message: 'Project structure analyzed',
|
||||
stats: { filesProcessed: files.length, totalFiles: files.length, nodesCreated: graph.nodeCount },
|
||||
});
|
||||
|
||||
// Phase 3: Parsing (30-70%)
|
||||
onProgress({
|
||||
phase: 'parsing',
|
||||
percent: 30,
|
||||
message: 'Parsing code definitions...',
|
||||
stats: { filesProcessed: 0, totalFiles: files.length, nodesCreated: graph.nodeCount },
|
||||
});
|
||||
|
||||
await processParsing(graph, files, symbolTable, astCache, (current, total, filePath) => {
|
||||
const parsingProgress = 30 + ((current / total) * 40);
|
||||
onProgress({
|
||||
phase: 'parsing',
|
||||
percent: Math.round(parsingProgress),
|
||||
message: 'Parsing code definitions...',
|
||||
detail: filePath,
|
||||
stats: { filesProcessed: current, totalFiles: total, nodesCreated: graph.nodeCount },
|
||||
});
|
||||
|
||||
// Pass 2: Code Parsing and Definition Extraction (populates FunctionRegistryTrie)
|
||||
console.log(`🔍 Pass 2: Parsing code and extracting definitions (${processingMode})...`);
|
||||
await this.parsingProcessor.process(graph, {
|
||||
filePaths,
|
||||
fileContents,
|
||||
options // Pass filtering options to ParsingProcessor
|
||||
});
|
||||
|
||||
// Get AST map and function registry from parsing processor
|
||||
const astMap = this.parsingProcessor.getASTMap();
|
||||
const functionTrie = this.parsingProcessor.getFunctionRegistry();
|
||||
|
||||
this.callProcessor = new CallProcessor(functionTrie);
|
||||
|
||||
// Pass 3: Import Resolution (builds complete import map)
|
||||
console.log('🔗 Pass 3: Resolving imports and building dependency map...');
|
||||
await this.importProcessor.process(graph, astMap, fileContents);
|
||||
|
||||
// Pass 4: Call Resolution (uses import map and function trie)
|
||||
console.log('📞 Pass 4: Resolving function calls with 3-stage strategy...');
|
||||
const importMap = this.importProcessor.getImportMap();
|
||||
await this.callProcessor.process(graph, astMap, importMap);
|
||||
|
||||
console.log(`Ingestion complete. Graph contains ${graph.nodes.length} nodes and ${graph.relationships.length} relationships.`);
|
||||
|
||||
// Flush KuzuDB operations and log dual-write statistics if using DualWriteKnowledgeGraph
|
||||
const { DualWriteKnowledgeGraph } = await import('../graph/dual-write-knowledge-graph.ts');
|
||||
if (graph instanceof DualWriteKnowledgeGraph) {
|
||||
await graph.flushKuzuDB();
|
||||
graph.logDualWriteStats();
|
||||
}
|
||||
|
||||
// Debug: Show graph structure
|
||||
const nodesByType = graph.nodes.reduce((acc, node) => {
|
||||
acc[node.label] = (acc[node.label] || 0) + 1;
|
||||
return acc;
|
||||
}, {} as Record<string, number>);
|
||||
|
||||
const relationshipsByType = graph.relationships.reduce((acc, rel) => {
|
||||
acc[rel.type] = (acc[rel.type] || 0) + 1;
|
||||
return acc;
|
||||
}, {} as Record<string, number>);
|
||||
|
||||
console.log('📊 Graph Statistics:');
|
||||
console.log('Nodes by type:', nodesByType);
|
||||
console.log('Relationships by type:', relationshipsByType);
|
||||
console.log(`📈 Total entities: ${graph.nodes.length + graph.relationships.length}`);
|
||||
|
||||
// Debug: Find isolated nodes (nodes with no relationships)
|
||||
const connectedNodeIds = new Set<string>();
|
||||
graph.relationships.forEach(rel => {
|
||||
connectedNodeIds.add(rel.source);
|
||||
connectedNodeIds.add(rel.target);
|
||||
});
|
||||
|
||||
const isolatedNodes = graph.nodes.filter(node => !connectedNodeIds.has(node.id));
|
||||
if (isolatedNodes.length > 0) {
|
||||
console.warn(`⚠️ Found ${isolatedNodes.length} isolated nodes:`);
|
||||
const isolatedByType = isolatedNodes.reduce((acc, node) => {
|
||||
acc[node.label] = (acc[node.label] || 0) + 1;
|
||||
return acc;
|
||||
}, {} as Record<string, number>);
|
||||
console.warn('Isolated nodes by type:', isolatedByType);
|
||||
|
||||
// Show some examples
|
||||
console.warn('Sample isolated nodes:', isolatedNodes.slice(0, 5).map(n => ({
|
||||
type: n.label,
|
||||
name: n.properties.name || n.properties.filePath || n.id,
|
||||
properties: Object.keys(n.properties)
|
||||
})));
|
||||
}
|
||||
|
||||
// Debug: Check for files without content
|
||||
const fileNodes = graph.nodes.filter(n => n.label === 'File');
|
||||
const filesWithoutDefinitions = fileNodes.filter(fileNode => {
|
||||
const hasDefinitions = graph.relationships.some(rel =>
|
||||
rel.source === fileNode.id &&
|
||||
rel.type === 'DEFINES' &&
|
||||
graph.nodes.some(targetNode =>
|
||||
targetNode.id === rel.target &&
|
||||
['Function', 'Class', 'Method', 'Variable'].includes(targetNode.label)
|
||||
)
|
||||
);
|
||||
return !hasDefinitions;
|
||||
});
|
||||
|
||||
if (filesWithoutDefinitions.length > 0) {
|
||||
console.warn(`⚠️ Found ${filesWithoutDefinitions.length} files without definitions:`);
|
||||
console.warn('Files without content:', filesWithoutDefinitions.slice(0, 5).map(n =>
|
||||
n.properties.filePath || n.properties.name
|
||||
));
|
||||
}
|
||||
|
||||
// Validate graph integrity
|
||||
this.validateGraphIntegrity(graph);
|
||||
|
||||
return graph;
|
||||
}
|
||||
});
|
||||
|
||||
/**
|
||||
* Get detailed diagnostic information about parsing results
|
||||
*/
|
||||
public getParsingDiagnostics() {
|
||||
return this.parsingProcessor.getDiagnosticInfo();
|
||||
}
|
||||
|
||||
/**
|
||||
* Analyze a specific file to understand why it might not have definitions
|
||||
*/
|
||||
public async analyzeSpecificFile(filePath: string, content: string) {
|
||||
return await this.parsingProcessor.analyzeFile(filePath, content);
|
||||
}
|
||||
// Phase 4: Imports (70-82%)
|
||||
onProgress({
|
||||
phase: 'imports',
|
||||
percent: 70,
|
||||
message: 'Resolving imports...',
|
||||
stats: { filesProcessed: 0, totalFiles: files.length, nodesCreated: graph.nodeCount },
|
||||
});
|
||||
|
||||
/**
|
||||
* Validate graph integrity and identify potential issues
|
||||
*/
|
||||
private validateGraphIntegrity(graph: KnowledgeGraph): void {
|
||||
console.log('🔍 Validating graph integrity...');
|
||||
|
||||
const issues: string[] = [];
|
||||
|
||||
// Check 1: Orphaned relationships (references to non-existent nodes)
|
||||
const nodeIds = new Set(graph.nodes.map(n => n.id));
|
||||
const orphanedRels = graph.relationships.filter(rel =>
|
||||
!nodeIds.has(rel.source) || !nodeIds.has(rel.target)
|
||||
);
|
||||
|
||||
if (orphanedRels.length > 0) {
|
||||
issues.push(`${orphanedRels.length} relationships reference non-existent nodes`);
|
||||
}
|
||||
|
||||
// Check 2: Files without proper structure connections
|
||||
const projectNodes = graph.nodes.filter(n => n.label === 'Project');
|
||||
const folderNodes = graph.nodes.filter(n => n.label === 'Folder');
|
||||
const fileNodes = graph.nodes.filter(n => n.label === 'File');
|
||||
|
||||
const filesNotConnectedToStructure = fileNodes.filter(fileNode => {
|
||||
const hasStructuralParent = graph.relationships.some(rel =>
|
||||
rel.target === fileNode.id &&
|
||||
rel.type === 'CONTAINS' &&
|
||||
(projectNodes.some(p => p.id === rel.source) || folderNodes.some(f => f.id === rel.source))
|
||||
);
|
||||
return !hasStructuralParent;
|
||||
await processImports(graph, files, astCache, importMap, (current, total) => {
|
||||
const importProgress = 70 + ((current / total) * 12);
|
||||
onProgress({
|
||||
phase: 'imports',
|
||||
percent: Math.round(importProgress),
|
||||
message: 'Resolving imports...',
|
||||
stats: { filesProcessed: current, totalFiles: total, nodesCreated: graph.nodeCount },
|
||||
});
|
||||
|
||||
if (filesNotConnectedToStructure.length > 0) {
|
||||
issues.push(`${filesNotConnectedToStructure.length} files not connected to project structure`);
|
||||
}
|
||||
|
||||
// Check 3: Source files without any definitions
|
||||
const sourceFileExtensions = ['.js', '.ts', '.jsx', '.tsx', '.py', '.java', '.cpp', '.c', '.cs'];
|
||||
const sourceFiles = fileNodes.filter(fileNode => {
|
||||
const filePath = fileNode.properties.filePath as string || '';
|
||||
return sourceFileExtensions.some(ext => filePath.endsWith(ext));
|
||||
});
|
||||
|
||||
const sourceFilesWithoutDefinitions = sourceFiles.filter(fileNode => {
|
||||
const hasDefinitions = graph.relationships.some(rel =>
|
||||
rel.source === fileNode.id &&
|
||||
rel.type === 'DEFINES' &&
|
||||
graph.nodes.some(n =>
|
||||
n.id === rel.target &&
|
||||
['Function', 'Class', 'Method', 'Variable'].includes(n.label)
|
||||
)
|
||||
);
|
||||
return !hasDefinitions;
|
||||
});
|
||||
|
||||
if (sourceFilesWithoutDefinitions.length > 0) {
|
||||
issues.push(`${sourceFilesWithoutDefinitions.length} source files contain no parsed definitions`);
|
||||
console.warn('Source files without definitions:',
|
||||
sourceFilesWithoutDefinitions.slice(0, 3).map(n => n.properties.filePath)
|
||||
);
|
||||
}
|
||||
|
||||
// Check 4: Functions/Classes without file parents
|
||||
const definitionNodes = graph.nodes.filter(n =>
|
||||
['Function', 'Class', 'Method', 'Variable'].includes(n.label)
|
||||
);
|
||||
|
||||
const definitionsWithoutFiles = definitionNodes.filter(defNode => {
|
||||
const hasFileParent = graph.relationships.some(rel =>
|
||||
rel.target === defNode.id &&
|
||||
rel.type === 'DEFINES' &&
|
||||
graph.nodes.some(n => n.id === rel.source && n.label === 'File')
|
||||
);
|
||||
return !hasFileParent;
|
||||
});
|
||||
|
||||
if (definitionsWithoutFiles.length > 0) {
|
||||
issues.push(`${definitionsWithoutFiles.length} definitions not connected to files`);
|
||||
}
|
||||
|
||||
// Check 5: Import/Call relationship issues
|
||||
const importRels = graph.relationships.filter(r => r.type === 'IMPORTS');
|
||||
const callRels = graph.relationships.filter(r => r.type === 'CALLS');
|
||||
|
||||
if (sourceFiles.length > 1 && importRels.length === 0) {
|
||||
issues.push('No import relationships found between files');
|
||||
}
|
||||
|
||||
if (definitionNodes.length > 1 && callRels.length === 0) {
|
||||
issues.push('No function call relationships found');
|
||||
}
|
||||
|
||||
// Report results
|
||||
if (issues.length === 0) {
|
||||
console.log('✅ Graph integrity validation passed');
|
||||
} else {
|
||||
console.warn('⚠️ Graph integrity issues found:');
|
||||
issues.forEach((issue, i) => console.warn(` ${i + 1}. ${issue}`));
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
public getStats(graph: KnowledgeGraph): { nodeStats: Record<string, number>; relationshipStats: Record<string, number> } {
|
||||
const nodeStats: Record<string, number> = {};
|
||||
const relationshipStats: Record<string, number> = {};
|
||||
|
||||
for (const node of graph.nodes) {
|
||||
nodeStats[node.label] = (nodeStats[node.label] || 0) + 1;
|
||||
}
|
||||
|
||||
for (const relationship of graph.relationships) {
|
||||
relationshipStats[relationship.type] = (relationshipStats[relationship.type] || 0) + 1;
|
||||
}
|
||||
|
||||
return { nodeStats, relationshipStats };
|
||||
}
|
||||
|
||||
public getCallStats() {
|
||||
return this.callProcessor.getStats();
|
||||
}
|
||||
// Phase 5: Calls (82-98%)
|
||||
onProgress({
|
||||
phase: 'calls',
|
||||
percent: 82,
|
||||
message: 'Tracing function calls...',
|
||||
stats: { filesProcessed: 0, totalFiles: files.length, nodesCreated: graph.nodeCount },
|
||||
});
|
||||
|
||||
/**
|
||||
* Create appropriate graph implementation based on feature flags
|
||||
*/
|
||||
private async createGraph(): Promise<KnowledgeGraph> {
|
||||
if (isKuzuDBEnabled()) {
|
||||
try {
|
||||
console.log('🚀 Initializing KuzuDB integration...');
|
||||
|
||||
// Initialize KuzuDB query engine
|
||||
const { KuzuQueryEngine } = await import('../graph/kuzu-query-engine.ts');
|
||||
const queryEngine = new KuzuQueryEngine({
|
||||
enableCache: true,
|
||||
cacheSize: 1000,
|
||||
cacheTTL: 5 * 60 * 1000 // 5 minutes
|
||||
});
|
||||
|
||||
await queryEngine.initialize();
|
||||
|
||||
// Create KuzuDB knowledge graph
|
||||
const { KuzuKnowledgeGraph } = await import('../graph/kuzu-knowledge-graph.ts');
|
||||
const kuzuGraph = new KuzuKnowledgeGraph(queryEngine, {
|
||||
enableCache: true,
|
||||
batchSize: 100,
|
||||
autoCommit: false
|
||||
});
|
||||
|
||||
// Return transparent dual-write graph
|
||||
const { DualWriteKnowledgeGraph } = await import('../graph/dual-write-knowledge-graph.ts');
|
||||
console.log('✅ KuzuDB integration initialized - using dual-write mode');
|
||||
return new DualWriteKnowledgeGraph(kuzuGraph);
|
||||
|
||||
} catch (error) {
|
||||
console.warn('❌ KuzuDB initialization failed, falling back to JSON-only mode:', error);
|
||||
return new SimpleKnowledgeGraph();
|
||||
}
|
||||
} else {
|
||||
console.log('📝 Using JSON-only storage mode');
|
||||
return new SimpleKnowledgeGraph();
|
||||
}
|
||||
await processCalls(graph, files, astCache, symbolTable, importMap, (current, total) => {
|
||||
const callProgress = 82 + ((current / total) * 10);
|
||||
onProgress({
|
||||
phase: 'calls',
|
||||
percent: Math.round(callProgress),
|
||||
message: 'Tracing function calls...',
|
||||
stats: { filesProcessed: current, totalFiles: total, nodesCreated: graph.nodeCount },
|
||||
});
|
||||
});
|
||||
|
||||
// Phase 6: Heritage - Class inheritance (92-98%)
|
||||
onProgress({
|
||||
phase: 'heritage',
|
||||
percent: 92,
|
||||
message: 'Extracting class inheritance...',
|
||||
stats: { filesProcessed: 0, totalFiles: files.length, nodesCreated: graph.nodeCount },
|
||||
});
|
||||
|
||||
await processHeritage(graph, files, astCache, symbolTable, (current, total) => {
|
||||
const heritageProgress = 92 + ((current / total) * 6);
|
||||
onProgress({
|
||||
phase: 'heritage',
|
||||
percent: Math.round(heritageProgress),
|
||||
message: 'Extracting class inheritance...',
|
||||
stats: { filesProcessed: current, totalFiles: total, nodesCreated: graph.nodeCount },
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
// Phase 6: Complete (100%)
|
||||
onProgress({
|
||||
phase: 'complete',
|
||||
percent: 100,
|
||||
message: 'Graph generation complete!',
|
||||
stats: {
|
||||
filesProcessed: files.length,
|
||||
totalFiles: files.length,
|
||||
nodesCreated: graph.nodeCount
|
||||
},
|
||||
});
|
||||
|
||||
// Cleanup WASM memory before returning
|
||||
astCache.clear();
|
||||
|
||||
return { graph, fileContents };
|
||||
|
||||
} catch (error) {
|
||||
cleanup();
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1,319 +1,46 @@
|
||||
import type { KnowledgeGraph } from '../graph/graph.ts';
|
||||
import type { GraphNode, GraphRelationship } from '../graph/types.ts';
|
||||
import { generateDeterministicId } from '../../lib/utils.ts';
|
||||
import { ignoreService } from '../../config/ignore-service.js';
|
||||
import { generateId } from "@/lib/utils";
|
||||
import { KnowledgeGraph, GraphNode, GraphRelationship } from "../graph/types";
|
||||
|
||||
export interface StructureInput {
|
||||
projectRoot: string;
|
||||
projectName: string;
|
||||
filePaths: string[]; // Now includes ALL paths: files AND directories
|
||||
export const processStructure = ( graph: KnowledgeGraph, paths: string[])=>{
|
||||
paths.forEach( path => {
|
||||
const parts = path.split('/')
|
||||
let currentPath = ''
|
||||
let parentId = ''
|
||||
|
||||
parts.forEach( (part, index ) => {
|
||||
const isFile = index === parts.length - 1
|
||||
const label = isFile ? 'File' : 'Folder'
|
||||
|
||||
currentPath = currentPath ? `${currentPath}/${part}` : part
|
||||
|
||||
const nodeId=generateId(label, currentPath)
|
||||
|
||||
const node: GraphNode = {
|
||||
id: nodeId,
|
||||
label: label,
|
||||
properties: {
|
||||
name: part,
|
||||
filePath: currentPath
|
||||
}
|
||||
}
|
||||
graph.addNode(node)
|
||||
|
||||
if(parentId){
|
||||
const relId = generateId('CONTAINS', `${parentId}->${nodeId}`)
|
||||
|
||||
const relationship: GraphRelationship={
|
||||
id: relId,
|
||||
type: 'CONTAINS',
|
||||
sourceId: parentId,
|
||||
targetId: nodeId
|
||||
}
|
||||
|
||||
graph.addRelationship(relationship)
|
||||
}
|
||||
|
||||
parentId = nodeId
|
||||
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
export class StructureProcessor {
|
||||
private nodeIdMap: Map<string, string> = new Map();
|
||||
|
||||
private stats = {
|
||||
nodesProcessed: 0,
|
||||
relationshipsProcessed: 0
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* Process complete repository structure directly from discovered paths
|
||||
* This is the new robust approach that doesn't infer structure
|
||||
* Now includes KuzuDB dual-write support
|
||||
*/
|
||||
public async process(graph: KnowledgeGraph, input: StructureInput): Promise<void> {
|
||||
const { projectRoot, projectName, filePaths } = input;
|
||||
|
||||
try {
|
||||
// Reset statistics
|
||||
this.stats = { nodesProcessed: 0, relationshipsProcessed: 0 };
|
||||
|
||||
console.log(`📁 Processing structure for ${projectName} with ${filePaths.length} paths...`);
|
||||
|
||||
// Create project root node
|
||||
const projectNode = this.createProjectNode(projectName, projectRoot);
|
||||
graph.addNode(projectNode);
|
||||
this.stats.nodesProcessed++;
|
||||
|
||||
// Separate files and directories from the complete path list
|
||||
const { directories, files } = this.categorizePaths(filePaths);
|
||||
|
||||
// Filter out ignored directories from KG display (but keep for internal structure)
|
||||
const visibleDirectories = directories.filter(dir => !this.shouldHideDirectory(dir));
|
||||
const hiddenDirectoriesCount = directories.length - visibleDirectories.length;
|
||||
|
||||
// Create directory nodes only for visible directories
|
||||
const directoryNodes = this.createDirectoryNodes(visibleDirectories);
|
||||
for (const node of directoryNodes) {
|
||||
graph.addNode(node);
|
||||
this.stats.nodesProcessed++;
|
||||
}
|
||||
|
||||
// Filter out files that are inside ignored directories
|
||||
const visibleFiles = files.filter(file => !this.shouldHideFile(file));
|
||||
const hiddenFilesCount = files.length - visibleFiles.length;
|
||||
|
||||
// Create file nodes only for visible files
|
||||
const fileNodes = this.createFileNodes(visibleFiles);
|
||||
for (const node of fileNodes) {
|
||||
graph.addNode(node);
|
||||
this.stats.nodesProcessed++;
|
||||
}
|
||||
|
||||
// Establish CONTAINS relationships for visible structure only
|
||||
this.createContainsRelationships(graph, projectNode.id, visibleDirectories, visibleFiles);
|
||||
|
||||
const totalHidden = hiddenDirectoriesCount + hiddenFilesCount;
|
||||
console.log(`✅ Structure processing completed. Hidden ${totalHidden} items from display.`);
|
||||
|
||||
// Log processing statistics
|
||||
console.log(`📊 StructureProcessor: ${this.stats.nodesProcessed} nodes, ${this.stats.relationshipsProcessed} relationships`);
|
||||
|
||||
} catch (error) {
|
||||
console.error('❌ Structure processing failed:', error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
private categorizePaths(allPaths: string[]): { directories: string[], files: string[] } {
|
||||
const directories: string[] = [];
|
||||
const files: string[] = [];
|
||||
const pathSet = new Set(allPaths);
|
||||
|
||||
for (const path of allPaths) {
|
||||
// A path is a directory if:
|
||||
// 1. Other paths exist that start with this path + "/"
|
||||
// 2. OR it doesn't have a file extension and other paths are nested under it
|
||||
const isDirectory = allPaths.some(otherPath =>
|
||||
otherPath !== path && otherPath.startsWith(path + '/')
|
||||
);
|
||||
|
||||
if (isDirectory) {
|
||||
directories.push(path);
|
||||
} else {
|
||||
// It's a file if it's not identified as a directory
|
||||
files.push(path);
|
||||
}
|
||||
}
|
||||
|
||||
// Also add intermediate directories that might not be explicitly listed
|
||||
const allIntermediateDirs = new Set<string>();
|
||||
for (const path of allPaths) {
|
||||
const parts = path.split('/');
|
||||
for (let i = 1; i < parts.length; i++) {
|
||||
const intermediatePath = parts.slice(0, i).join('/');
|
||||
if (intermediatePath && !pathSet.has(intermediatePath)) {
|
||||
allIntermediateDirs.add(intermediatePath);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Add intermediate directories that weren't explicitly listed
|
||||
directories.push(...Array.from(allIntermediateDirs));
|
||||
|
||||
return {
|
||||
directories: [...new Set(directories)].sort(), // Remove duplicates and sort
|
||||
files: files.sort()
|
||||
};
|
||||
}
|
||||
|
||||
private createProjectNode(projectName: string, projectRoot: string): GraphNode {
|
||||
const id = generateDeterministicId('project', projectName);
|
||||
this.nodeIdMap.set('', id); // Empty path represents project root
|
||||
|
||||
return {
|
||||
id,
|
||||
label: 'Project',
|
||||
properties: {
|
||||
name: projectName,
|
||||
path: projectRoot,
|
||||
createdAt: new Date().toISOString()
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Create nodes for directories directly from discovered directory paths
|
||||
*/
|
||||
private createDirectoryNodes(directoryPaths: string[]): GraphNode[] {
|
||||
const nodes: GraphNode[] = [];
|
||||
|
||||
for (const dirPath of directoryPaths) {
|
||||
if (!dirPath) continue;
|
||||
|
||||
const id = generateDeterministicId('folder', dirPath);
|
||||
this.nodeIdMap.set(dirPath, id);
|
||||
|
||||
const pathParts = dirPath.split('/');
|
||||
const dirName = pathParts[pathParts.length - 1];
|
||||
|
||||
const node: GraphNode = {
|
||||
id,
|
||||
label: 'Folder',
|
||||
properties: {
|
||||
name: dirName,
|
||||
path: dirPath,
|
||||
fullPath: dirPath,
|
||||
depth: pathParts.length
|
||||
}
|
||||
};
|
||||
|
||||
nodes.push(node);
|
||||
}
|
||||
|
||||
return nodes;
|
||||
}
|
||||
|
||||
/**
|
||||
* Create nodes for files directly from discovered file paths
|
||||
*/
|
||||
private createFileNodes(filePaths: string[]): GraphNode[] {
|
||||
const nodes: GraphNode[] = [];
|
||||
|
||||
for (const filePath of filePaths) {
|
||||
if (!filePath) continue;
|
||||
|
||||
const id = generateDeterministicId('file', filePath);
|
||||
this.nodeIdMap.set(filePath, id);
|
||||
|
||||
const fileName = filePath.split('/').pop() || filePath;
|
||||
const extension = this.getFileExtension(fileName);
|
||||
|
||||
const node: GraphNode = {
|
||||
id,
|
||||
label: 'File',
|
||||
properties: {
|
||||
name: fileName,
|
||||
path: filePath,
|
||||
filePath: filePath, // For compatibility with existing code
|
||||
extension,
|
||||
// Note: definitionCount will be set later by ParsingProcessor
|
||||
// language will be determined later by ParsingProcessor
|
||||
}
|
||||
};
|
||||
|
||||
nodes.push(node);
|
||||
}
|
||||
|
||||
return nodes;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Create CONTAINS relationships
|
||||
*/
|
||||
private createContainsRelationships(
|
||||
graph: KnowledgeGraph,
|
||||
projectId: string,
|
||||
directories: string[],
|
||||
files: string[]
|
||||
): void {
|
||||
// Create relationships: directories contain subdirectories and files
|
||||
const allPaths = [...directories, ...files];
|
||||
|
||||
for (const path of allPaths) {
|
||||
const parentPath = this.getParentPath(path);
|
||||
const parentId = parentPath === '' ? projectId : this.nodeIdMap.get(parentPath);
|
||||
const childId = this.nodeIdMap.get(path);
|
||||
|
||||
// Only create relationships if both parent and child nodes exist in the graph
|
||||
if (parentId && childId && parentId !== childId) {
|
||||
const relationship: GraphRelationship = {
|
||||
id: generateDeterministicId('contains', `${parentId}-${childId}`),
|
||||
type: 'CONTAINS',
|
||||
source: parentId,
|
||||
target: childId,
|
||||
properties: {}
|
||||
};
|
||||
|
||||
graph.addRelationship(relationship);
|
||||
this.stats.relationshipsProcessed++;
|
||||
} else if (!parentId && parentPath !== '') {
|
||||
// If parent directory was hidden, connect directly to project or nearest visible parent
|
||||
const visibleParentId = this.findVisibleParent(parentPath, projectId);
|
||||
if (visibleParentId && childId && visibleParentId !== childId) {
|
||||
const relationship: GraphRelationship = {
|
||||
id: generateDeterministicId('contains', `${parentId}-${childId}`),
|
||||
type: 'CONTAINS',
|
||||
source: visibleParentId,
|
||||
target: childId,
|
||||
properties: {}
|
||||
};
|
||||
|
||||
graph.addRelationship(relationship);
|
||||
this.stats.relationshipsProcessed++;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Find the nearest visible parent directory or project root
|
||||
*/
|
||||
private findVisibleParent(path: string, projectId: string): string {
|
||||
if (path === '') return projectId;
|
||||
|
||||
const parentPath = this.getParentPath(path);
|
||||
const parentId = this.nodeIdMap.get(parentPath);
|
||||
|
||||
if (parentId) {
|
||||
return parentId; // Found visible parent
|
||||
}
|
||||
|
||||
// Recursively look for visible parent
|
||||
return this.findVisibleParent(parentPath, projectId);
|
||||
}
|
||||
|
||||
private getParentPath(path: string): string {
|
||||
if (!path || !path.includes('/')) {
|
||||
return ''; // Root level
|
||||
}
|
||||
|
||||
const lastSlashIndex = path.lastIndexOf('/');
|
||||
return path.substring(0, lastSlashIndex);
|
||||
}
|
||||
|
||||
private getFileExtension(fileName: string): string {
|
||||
const lastDotIndex = fileName.lastIndexOf('.');
|
||||
if (lastDotIndex === -1 || lastDotIndex === 0) {
|
||||
return '';
|
||||
}
|
||||
return fileName.substring(lastDotIndex);
|
||||
}
|
||||
|
||||
public getNodeId(path: string): string | undefined {
|
||||
return this.nodeIdMap.get(path);
|
||||
}
|
||||
|
||||
public clear(): void {
|
||||
this.nodeIdMap.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a directory should be hidden from the KG visualization
|
||||
* Uses centralized ignore service
|
||||
*/
|
||||
private shouldHideDirectory(dirPath: string): boolean {
|
||||
return ignoreService.shouldIgnoreDirectory(dirPath);
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a file should be hidden from the KG visualization
|
||||
* Uses centralized ignore service
|
||||
*/
|
||||
private shouldHideFile(filePath: string): boolean {
|
||||
return ignoreService.shouldIgnorePath(filePath);
|
||||
}
|
||||
|
||||
/**
|
||||
* Get processing statistics
|
||||
*/
|
||||
public getStats() {
|
||||
return {
|
||||
nodesProcessed: this.stats.nodesProcessed,
|
||||
relationshipsProcessed: this.stats.relationshipsProcessed
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
export interface SymbolDefinition {
|
||||
nodeId: string;
|
||||
filePath: string;
|
||||
type: string; // 'Function', 'Class', etc.
|
||||
}
|
||||
|
||||
export interface SymbolTable {
|
||||
/**
|
||||
* Register a new symbol definition
|
||||
*/
|
||||
add: (filePath: string, name: string, nodeId: string, type: string) => void;
|
||||
|
||||
/**
|
||||
* High Confidence: Look for a symbol specifically inside a file
|
||||
* Returns the Node ID if found
|
||||
*/
|
||||
lookupExact: (filePath: string, name: string) => string | undefined;
|
||||
|
||||
/**
|
||||
* Low Confidence: Look for a symbol anywhere in the project
|
||||
* Used when imports are missing or for framework magic
|
||||
*/
|
||||
lookupFuzzy: (name: string) => SymbolDefinition[];
|
||||
|
||||
/**
|
||||
* Debugging: See how many symbols are tracked
|
||||
*/
|
||||
getStats: () => { fileCount: number; globalSymbolCount: number };
|
||||
|
||||
/**
|
||||
* Cleanup memory
|
||||
*/
|
||||
clear: () => void;
|
||||
}
|
||||
|
||||
export const createSymbolTable = (): SymbolTable => {
|
||||
// 1. File-Specific Index (The "Good" one)
|
||||
// Structure: FilePath -> (SymbolName -> NodeID)
|
||||
const fileIndex = new Map<string, Map<string, string>>();
|
||||
|
||||
// 2. Global Reverse Index (The "Backup")
|
||||
// Structure: SymbolName -> [List of Definitions]
|
||||
const globalIndex = new Map<string, SymbolDefinition[]>();
|
||||
|
||||
const add = (filePath: string, name: string, nodeId: string, type: string) => {
|
||||
// A. Add to File Index
|
||||
if (!fileIndex.has(filePath)) {
|
||||
fileIndex.set(filePath, new Map());
|
||||
}
|
||||
fileIndex.get(filePath)!.set(name, nodeId);
|
||||
|
||||
// B. Add to Global Index
|
||||
if (!globalIndex.has(name)) {
|
||||
globalIndex.set(name, []);
|
||||
}
|
||||
globalIndex.get(name)!.push({ nodeId, filePath, type });
|
||||
};
|
||||
|
||||
const lookupExact = (filePath: string, name: string): string | undefined => {
|
||||
const fileSymbols = fileIndex.get(filePath);
|
||||
if (!fileSymbols) return undefined;
|
||||
return fileSymbols.get(name);
|
||||
};
|
||||
|
||||
const lookupFuzzy = (name: string): SymbolDefinition[] => {
|
||||
return globalIndex.get(name) || [];
|
||||
};
|
||||
|
||||
const getStats = () => ({
|
||||
fileCount: fileIndex.size,
|
||||
globalSymbolCount: globalIndex.size
|
||||
});
|
||||
|
||||
const clear = () => {
|
||||
fileIndex.clear();
|
||||
globalIndex.clear();
|
||||
};
|
||||
|
||||
return { add, lookupExact, lookupFuzzy, getStats, clear };
|
||||
};
|
||||
@@ -1,256 +1,156 @@
|
||||
import { SupportedLanguages } from '../../config/supported-languages';
|
||||
|
||||
export const TYPESCRIPT_QUERIES = {
|
||||
imports: `
|
||||
(import_statement) @import
|
||||
`,
|
||||
classes: `
|
||||
(class_declaration) @class
|
||||
`,
|
||||
methods: `
|
||||
(method_definition) @method
|
||||
`,
|
||||
functions: `
|
||||
(function_declaration) @function
|
||||
`,
|
||||
arrowFunctions: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (arrow_function))) @arrow_function
|
||||
`,
|
||||
// React functional components with type annotations (simplified)
|
||||
reactComponents: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (arrow_function
|
||||
(type_annotation)))) @react_component
|
||||
`,
|
||||
// React functional components as const declarations (simplified)
|
||||
reactConstComponents: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (as_expression
|
||||
(arrow_function)))) @react_const_component
|
||||
`,
|
||||
// Default export arrow functions (common React pattern)
|
||||
defaultExportArrows: `
|
||||
(export_statement
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (arrow_function)))) @default_export_arrow
|
||||
`,
|
||||
// React hooks (useState, useEffect, etc.)
|
||||
hookCalls: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @hook_name
|
||||
value: (call_expression
|
||||
function: (identifier) @hook_function
|
||||
(#match? @hook_function "^use[A-Z].*")))) @hook_call
|
||||
`,
|
||||
// Hook calls with array destructuring (useState pattern)
|
||||
hookDestructuring: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (array_pattern) @hook_pattern
|
||||
value: (call_expression
|
||||
function: (identifier) @hook_function
|
||||
(#match? @hook_function "^use[A-Z].*")))) @hook_destructuring
|
||||
`,
|
||||
variables: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @variable)) @var_declaration
|
||||
`,
|
||||
constDeclarations: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @const
|
||||
value: _)) @const_declaration
|
||||
`,
|
||||
// Function expressions assigned to variables
|
||||
functionExpressions: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (function_expression))) @function_expression
|
||||
`,
|
||||
exports: `
|
||||
(export_statement) @export
|
||||
`,
|
||||
exportFunctions: `
|
||||
(export_statement
|
||||
(function_declaration) @export_function)
|
||||
`,
|
||||
exportClasses: `
|
||||
(export_statement
|
||||
(class_declaration) @export_class)
|
||||
`,
|
||||
// Default exports
|
||||
defaultExports: `
|
||||
(export_statement
|
||||
(identifier) @default_export)
|
||||
`,
|
||||
// Default export functions
|
||||
defaultExportFunctions: `
|
||||
(export_statement
|
||||
(function_declaration) @default_export_function)
|
||||
`,
|
||||
interfaces: `
|
||||
(interface_declaration) @interface
|
||||
`,
|
||||
types: `
|
||||
(type_alias_declaration) @type
|
||||
`,
|
||||
enums: `
|
||||
(enum_declaration) @enum
|
||||
`,
|
||||
};
|
||||
/*
|
||||
* Tree-sitter queries for extracting code definitions.
|
||||
*
|
||||
* Note: Different grammars (typescript vs tsx vs javascript) may have
|
||||
* slightly different node types. These queries are designed to be
|
||||
* compatible with the standard tree-sitter grammars.
|
||||
*/
|
||||
|
||||
// JavaScript queries - similar to TypeScript but without TS-specific syntax
|
||||
export const JAVASCRIPT_QUERIES = {
|
||||
imports: `
|
||||
(import_statement) @import
|
||||
`,
|
||||
classes: `
|
||||
(class_declaration) @class
|
||||
`,
|
||||
methods: `
|
||||
(method_definition) @method
|
||||
`,
|
||||
functions: `
|
||||
(function_declaration) @function
|
||||
`,
|
||||
arrowFunctions: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (arrow_function))) @arrow_function
|
||||
`,
|
||||
variables: `
|
||||
(variable_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @variable)) @var_declaration
|
||||
`,
|
||||
constDeclarations: `
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @const
|
||||
value: _)) @const_declaration
|
||||
`,
|
||||
exports: `
|
||||
(export_statement) @export
|
||||
`,
|
||||
defaultExports: `
|
||||
(export_statement
|
||||
(identifier) @default_export)
|
||||
`,
|
||||
exportFunctions: `
|
||||
(export_statement
|
||||
(function_declaration) @export_function)
|
||||
`,
|
||||
exportClasses: `
|
||||
(export_statement
|
||||
(class_declaration) @export_class)
|
||||
`,
|
||||
variableAssignments: `
|
||||
(variable_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (function_expression))) @var_function
|
||||
`,
|
||||
objectMethods: `
|
||||
(assignment_expression
|
||||
left: (member_expression
|
||||
property: (property_identifier) @name)
|
||||
right: (function_expression)) @obj_method
|
||||
`,
|
||||
moduleExports: `
|
||||
(assignment_expression
|
||||
left: (member_expression
|
||||
object: (identifier) @module
|
||||
property: (property_identifier) @export_name)
|
||||
right: _) @module_export
|
||||
`,
|
||||
functionExpressions: `
|
||||
(assignment_expression
|
||||
left: (identifier) @name
|
||||
right: (function_expression)) @func_expr
|
||||
`,
|
||||
};
|
||||
// TypeScript queries - works with tree-sitter-typescript
|
||||
export const TYPESCRIPT_QUERIES = `
|
||||
(class_declaration
|
||||
name: (type_identifier) @name) @definition.class
|
||||
|
||||
export const PYTHON_QUERIES = {
|
||||
imports: `
|
||||
(import_statement) @import
|
||||
`,
|
||||
from_imports: `
|
||||
(import_from_statement) @from_import
|
||||
`,
|
||||
classes: `
|
||||
(class_definition) @class
|
||||
`,
|
||||
functions: `
|
||||
(function_definition) @function
|
||||
`,
|
||||
methods: `
|
||||
(class_definition
|
||||
body: (block
|
||||
(function_definition) @method))
|
||||
`,
|
||||
variables: `
|
||||
(assignment
|
||||
left: (identifier) @variable
|
||||
right: _) @var_assignment
|
||||
`,
|
||||
global_variables: `
|
||||
(assignment
|
||||
left: (identifier) @global_var
|
||||
right: _) @global_assignment
|
||||
`,
|
||||
decorators: `
|
||||
(decorated_definition
|
||||
(decorator) @decorator)
|
||||
`,
|
||||
properties: `
|
||||
(class_definition
|
||||
body: (block
|
||||
(decorated_definition
|
||||
(decorator
|
||||
(identifier) @property_decorator
|
||||
(#eq? @property_decorator "property"))
|
||||
(function_definition) @property)))
|
||||
`,
|
||||
staticmethods: `
|
||||
(class_definition
|
||||
body: (block
|
||||
(decorated_definition
|
||||
(decorator
|
||||
(identifier) @static_decorator
|
||||
(#eq? @static_decorator "staticmethod"))
|
||||
(function_definition) @static_method)))
|
||||
`,
|
||||
classmethods: `
|
||||
(class_definition
|
||||
body: (block
|
||||
(decorated_definition
|
||||
(decorator
|
||||
(identifier) @class_decorator
|
||||
(#eq? @class_decorator "classmethod"))
|
||||
(function_definition) @class_method)))
|
||||
`,
|
||||
};
|
||||
(interface_declaration
|
||||
name: (type_identifier) @name) @definition.interface
|
||||
|
||||
export const JAVA_QUERIES = {
|
||||
classes: `
|
||||
(class_declaration) @class
|
||||
`,
|
||||
methods: `
|
||||
(method_declaration) @method
|
||||
`,
|
||||
interfaces: `
|
||||
(interface_declaration) @interface
|
||||
`,
|
||||
(function_declaration
|
||||
name: (identifier) @name) @definition.function
|
||||
|
||||
(method_definition
|
||||
name: (property_identifier) @name) @definition.method
|
||||
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (arrow_function))) @definition.function
|
||||
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (function_expression))) @definition.function
|
||||
|
||||
(export_statement
|
||||
declaration: (lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (arrow_function)))) @definition.function
|
||||
|
||||
(export_statement
|
||||
declaration: (lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (function_expression)))) @definition.function
|
||||
|
||||
(import_statement
|
||||
source: (string) @import.source) @import
|
||||
|
||||
(call_expression
|
||||
function: (identifier) @call.name) @call
|
||||
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
property: (property_identifier) @call.name)) @call
|
||||
|
||||
; Heritage queries - class extends
|
||||
(class_declaration
|
||||
name: (type_identifier) @heritage.class
|
||||
(class_heritage
|
||||
(extends_clause
|
||||
value: (identifier) @heritage.extends))) @heritage
|
||||
|
||||
; Heritage queries - class implements interface
|
||||
(class_declaration
|
||||
name: (type_identifier) @heritage.class
|
||||
(class_heritage
|
||||
(implements_clause
|
||||
(type_identifier) @heritage.implements))) @heritage.impl
|
||||
`;
|
||||
|
||||
// JavaScript queries - works with tree-sitter-javascript
|
||||
export const JAVASCRIPT_QUERIES = `
|
||||
(class_declaration
|
||||
name: (identifier) @name) @definition.class
|
||||
|
||||
(function_declaration
|
||||
name: (identifier) @name) @definition.function
|
||||
|
||||
(method_definition
|
||||
name: (property_identifier) @name) @definition.method
|
||||
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (arrow_function))) @definition.function
|
||||
|
||||
(lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (function_expression))) @definition.function
|
||||
|
||||
(export_statement
|
||||
declaration: (lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (arrow_function)))) @definition.function
|
||||
|
||||
(export_statement
|
||||
declaration: (lexical_declaration
|
||||
(variable_declarator
|
||||
name: (identifier) @name
|
||||
value: (function_expression)))) @definition.function
|
||||
|
||||
(import_statement
|
||||
source: (string) @import.source) @import
|
||||
|
||||
(call_expression
|
||||
function: (identifier) @call.name) @call
|
||||
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
property: (property_identifier) @call.name)) @call
|
||||
|
||||
; Heritage queries - class extends (JavaScript uses different AST than TypeScript)
|
||||
; In tree-sitter-javascript, class_heritage directly contains the parent identifier
|
||||
(class_declaration
|
||||
name: (identifier) @heritage.class
|
||||
(class_heritage
|
||||
(identifier) @heritage.extends)) @heritage
|
||||
`;
|
||||
|
||||
// Python queries - works with tree-sitter-python
|
||||
export const PYTHON_QUERIES = `
|
||||
(class_definition
|
||||
name: (identifier) @name) @definition.class
|
||||
|
||||
(function_definition
|
||||
name: (identifier) @name) @definition.function
|
||||
|
||||
(import_statement
|
||||
name: (dotted_name) @import.source) @import
|
||||
|
||||
(import_from_statement
|
||||
module_name: (dotted_name) @import.source) @import
|
||||
|
||||
(call
|
||||
function: (identifier) @call.name) @call
|
||||
|
||||
(call
|
||||
function: (attribute
|
||||
attribute: (identifier) @call.name)) @call
|
||||
|
||||
; Heritage queries - Python class inheritance
|
||||
(class_definition
|
||||
name: (identifier) @heritage.class
|
||||
superclasses: (argument_list
|
||||
(identifier) @heritage.extends)) @heritage
|
||||
`;
|
||||
|
||||
export const LANGUAGE_QUERIES: Record<SupportedLanguages, string> = {
|
||||
[SupportedLanguages.TypeScript]: TYPESCRIPT_QUERIES,
|
||||
[SupportedLanguages.JavaScript]: JAVASCRIPT_QUERIES,
|
||||
[SupportedLanguages.Python]: PYTHON_QUERIES,
|
||||
};
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user