fix: Lazy grammar loading and quantized embeddings to prevent V8 WASM OOM
Fixes #54 — `codegraph init -i` crashes with "Fatal process out of memory: Zone" on large codebases because all 16 tree-sitter WASM grammar modules were compiled upfront by V8, exhausting the WASM Zone allocator. Changes: - initGrammars() now only initializes the tree-sitter WASM runtime (Parser.init()), no longer eagerly loads all grammar files - New loadGrammarsForLanguages() loads only grammars for languages actually present in the project (e.g. a Dart project loads ~2-3 grammars instead of 16) - Orchestrator detects needed languages after file scan, before parsing begins - Embedding pipeline now uses quantized model (~67MB vs ~270MB) to further reduce WASM memory pressure when embeddings are enabled
This commit is contained in:
+15
-3
@@ -18,7 +18,7 @@ import {
|
||||
} from '../types';
|
||||
import { QueryBuilder } from '../db/queries';
|
||||
import { extractFromSource } from './tree-sitter';
|
||||
import { detectLanguage, isLanguageSupported, initGrammars } from './grammars';
|
||||
import { detectLanguage, isLanguageSupported, initGrammars, loadGrammarsForLanguages } from './grammars';
|
||||
import { logDebug, logWarn } from '../errors';
|
||||
import { captureException } from '../sentry';
|
||||
import { validatePathWithinRoot, normalizePath } from '../utils';
|
||||
@@ -378,6 +378,12 @@ export class ExtractionOrchestrator {
|
||||
};
|
||||
}
|
||||
|
||||
// Load only the grammars needed for languages actually present in the project.
|
||||
// This avoids compiling all 16+ WASM grammar modules upfront, which can cause
|
||||
// V8 WASM Zone OOM on large codebases (see issue #54).
|
||||
const neededLanguages = [...new Set(files.map((f) => detectLanguage(f)))];
|
||||
await loadGrammarsForLanguages(neededLanguages);
|
||||
|
||||
// Phase 2: Parse files (read in parallel batches, parse/store sequentially)
|
||||
const total = files.length;
|
||||
let processed = 0;
|
||||
@@ -683,7 +689,7 @@ export class ExtractionOrchestrator {
|
||||
* Uses git status as a fast path when available, falling back to full scan.
|
||||
*/
|
||||
async sync(onProgress?: (progress: IndexProgress) => void): Promise<SyncResult> {
|
||||
await initGrammars();
|
||||
await initGrammars(); // Initialize WASM runtime (grammars loaded lazily below)
|
||||
const startTime = Date.now();
|
||||
let filesChecked = 0;
|
||||
let filesAdded = 0;
|
||||
@@ -794,6 +800,12 @@ export class ExtractionOrchestrator {
|
||||
}
|
||||
}
|
||||
|
||||
// Load only grammars needed for changed files
|
||||
if (filesToIndex.length > 0) {
|
||||
const neededLanguages = [...new Set(filesToIndex.map((f) => detectLanguage(f)))];
|
||||
await loadGrammarsForLanguages(neededLanguages);
|
||||
}
|
||||
|
||||
// Index changed files
|
||||
const total = filesToIndex.length;
|
||||
for (let i = 0; i < filesToIndex.length; i++) {
|
||||
@@ -920,4 +932,4 @@ export class ExtractionOrchestrator {
|
||||
|
||||
// Re-export useful types and functions
|
||||
export { extractFromSource } from './tree-sitter';
|
||||
export { detectLanguage, isLanguageSupported, getSupportedLanguages, initGrammars } from './grammars';
|
||||
export { detectLanguage, isLanguageSupported, isGrammarLoaded, getSupportedLanguages, initGrammars, loadGrammarsForLanguages, loadAllGrammars } from './grammars';
|
||||
|
||||
Reference in New Issue
Block a user