Adds expression index on lower(name) for memory-efficient case-insensitive searches, replacing in-memory caches that caused OOM on large codebases. Includes batched reference resolution, enhanced error reporting with detailed breakdown by error type, and improved CLI progress display for scanning phases.
350 lines
10 KiB
TypeScript
350 lines
10 KiB
TypeScript
import { Node, Edge, ExtractionResult, ExtractionError, UnresolvedReference } from '../types';
|
|
import { generateNodeId } from './tree-sitter-helpers';
|
|
|
|
/**
|
|
* LiquidExtractor - Extracts relationships from Liquid template files
|
|
*
|
|
* Liquid is a templating language (used by Shopify, Jekyll, etc.) that doesn't
|
|
* have traditional functions or classes. Instead, we extract:
|
|
* - Section references ({% section 'name' %})
|
|
* - Snippet references ({% render 'name' %} and {% include 'name' %})
|
|
* - Schema blocks ({% schema %}...{% endschema %})
|
|
*/
|
|
export class LiquidExtractor {
|
|
private filePath: string;
|
|
private source: string;
|
|
private nodes: Node[] = [];
|
|
private edges: Edge[] = [];
|
|
private unresolvedReferences: UnresolvedReference[] = [];
|
|
private errors: ExtractionError[] = [];
|
|
|
|
constructor(filePath: string, source: string) {
|
|
this.filePath = filePath;
|
|
this.source = source;
|
|
}
|
|
|
|
/**
|
|
* Extract from Liquid source
|
|
*/
|
|
extract(): ExtractionResult {
|
|
const startTime = Date.now();
|
|
|
|
try {
|
|
// Create file node
|
|
const fileNode = this.createFileNode();
|
|
|
|
// Extract render/include statements (snippet references)
|
|
this.extractSnippetReferences(fileNode.id);
|
|
|
|
// Extract section references
|
|
this.extractSectionReferences(fileNode.id);
|
|
|
|
// Extract schema block
|
|
this.extractSchema(fileNode.id);
|
|
|
|
// Extract assign statements as variables
|
|
this.extractAssignments(fileNode.id);
|
|
} catch (error) {
|
|
this.errors.push({
|
|
message: `Liquid extraction error: ${error instanceof Error ? error.message : String(error)}`,
|
|
severity: 'error',
|
|
code: 'parse_error',
|
|
});
|
|
}
|
|
|
|
return {
|
|
nodes: this.nodes,
|
|
edges: this.edges,
|
|
unresolvedReferences: this.unresolvedReferences,
|
|
errors: this.errors,
|
|
durationMs: Date.now() - startTime,
|
|
};
|
|
}
|
|
|
|
/**
|
|
* Create a file node for the Liquid template
|
|
*/
|
|
private createFileNode(): Node {
|
|
const lines = this.source.split('\n');
|
|
const id = generateNodeId(this.filePath, 'file', this.filePath, 1);
|
|
|
|
const fileNode: Node = {
|
|
id,
|
|
kind: 'file',
|
|
name: this.filePath.split('/').pop() || this.filePath,
|
|
qualifiedName: this.filePath,
|
|
filePath: this.filePath,
|
|
language: 'liquid',
|
|
startLine: 1,
|
|
endLine: lines.length,
|
|
startColumn: 0,
|
|
endColumn: lines[lines.length - 1]?.length || 0,
|
|
updatedAt: Date.now(),
|
|
};
|
|
|
|
this.nodes.push(fileNode);
|
|
return fileNode;
|
|
}
|
|
|
|
/**
|
|
* Extract {% render 'snippet' %} and {% include 'snippet' %} references
|
|
*/
|
|
private extractSnippetReferences(fileNodeId: string): void {
|
|
// Match {% render 'name' %} or {% include 'name' %} with optional parameters
|
|
const renderRegex = /\{%[-]?\s*(render|include)\s+['"]([^'"]+)['"]/g;
|
|
let match;
|
|
|
|
while ((match = renderRegex.exec(this.source)) !== null) {
|
|
const [fullMatch, tagType, snippetName] = match;
|
|
const line = this.getLineNumber(match.index);
|
|
|
|
// Create an import node for searchability
|
|
const importNodeId = generateNodeId(this.filePath, 'import', snippetName!, line);
|
|
const importNode: Node = {
|
|
id: importNodeId,
|
|
kind: 'import',
|
|
name: snippetName!,
|
|
qualifiedName: `${this.filePath}::import:${snippetName}`,
|
|
filePath: this.filePath,
|
|
language: 'liquid',
|
|
signature: fullMatch,
|
|
startLine: line,
|
|
endLine: line,
|
|
startColumn: match.index - this.getLineStart(line),
|
|
endColumn: match.index - this.getLineStart(line) + fullMatch.length,
|
|
updatedAt: Date.now(),
|
|
};
|
|
this.nodes.push(importNode);
|
|
|
|
// Add containment edge from file to import
|
|
this.edges.push({
|
|
source: fileNodeId,
|
|
target: importNodeId,
|
|
kind: 'contains',
|
|
});
|
|
|
|
// Create a component node for the snippet reference
|
|
const nodeId = generateNodeId(this.filePath, 'component', `${tagType}:${snippetName}`, line);
|
|
|
|
const node: Node = {
|
|
id: nodeId,
|
|
kind: 'component',
|
|
name: snippetName!,
|
|
qualifiedName: `${this.filePath}::${tagType}:${snippetName}`,
|
|
filePath: this.filePath,
|
|
language: 'liquid',
|
|
startLine: line,
|
|
endLine: line,
|
|
startColumn: match.index - this.getLineStart(line),
|
|
endColumn: match.index - this.getLineStart(line) + fullMatch.length,
|
|
updatedAt: Date.now(),
|
|
};
|
|
|
|
this.nodes.push(node);
|
|
|
|
// Add containment edge from file
|
|
this.edges.push({
|
|
source: fileNodeId,
|
|
target: nodeId,
|
|
kind: 'contains',
|
|
});
|
|
|
|
// Add unresolved reference to the snippet file
|
|
this.unresolvedReferences.push({
|
|
fromNodeId: fileNodeId,
|
|
referenceName: `snippets/${snippetName}.liquid`,
|
|
referenceKind: 'references',
|
|
line,
|
|
column: match.index - this.getLineStart(line),
|
|
});
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Extract {% section 'name' %} references
|
|
*/
|
|
private extractSectionReferences(fileNodeId: string): void {
|
|
// Match {% section 'name' %}
|
|
const sectionRegex = /\{%[-]?\s*section\s+['"]([^'"]+)['"]/g;
|
|
let match;
|
|
|
|
while ((match = sectionRegex.exec(this.source)) !== null) {
|
|
const [fullMatch, sectionName] = match;
|
|
const line = this.getLineNumber(match.index);
|
|
|
|
// Create an import node for searchability
|
|
const importNodeId = generateNodeId(this.filePath, 'import', sectionName!, line);
|
|
const importNode: Node = {
|
|
id: importNodeId,
|
|
kind: 'import',
|
|
name: sectionName!,
|
|
qualifiedName: `${this.filePath}::import:${sectionName}`,
|
|
filePath: this.filePath,
|
|
language: 'liquid',
|
|
signature: fullMatch,
|
|
startLine: line,
|
|
endLine: line,
|
|
startColumn: match.index - this.getLineStart(line),
|
|
endColumn: match.index - this.getLineStart(line) + fullMatch.length,
|
|
updatedAt: Date.now(),
|
|
};
|
|
this.nodes.push(importNode);
|
|
|
|
// Add containment edge from file to import
|
|
this.edges.push({
|
|
source: fileNodeId,
|
|
target: importNodeId,
|
|
kind: 'contains',
|
|
});
|
|
|
|
// Create a component node for the section reference
|
|
const nodeId = generateNodeId(this.filePath, 'component', `section:${sectionName}`, line);
|
|
|
|
const node: Node = {
|
|
id: nodeId,
|
|
kind: 'component',
|
|
name: sectionName!,
|
|
qualifiedName: `${this.filePath}::section:${sectionName}`,
|
|
filePath: this.filePath,
|
|
language: 'liquid',
|
|
startLine: line,
|
|
endLine: line,
|
|
startColumn: match.index - this.getLineStart(line),
|
|
endColumn: match.index - this.getLineStart(line) + fullMatch.length,
|
|
updatedAt: Date.now(),
|
|
};
|
|
|
|
this.nodes.push(node);
|
|
|
|
// Add containment edge from file
|
|
this.edges.push({
|
|
source: fileNodeId,
|
|
target: nodeId,
|
|
kind: 'contains',
|
|
});
|
|
|
|
// Add unresolved reference to the section file
|
|
this.unresolvedReferences.push({
|
|
fromNodeId: fileNodeId,
|
|
referenceName: `sections/${sectionName}.liquid`,
|
|
referenceKind: 'references',
|
|
line,
|
|
column: match.index - this.getLineStart(line),
|
|
});
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Extract {% schema %}...{% endschema %} blocks
|
|
*/
|
|
private extractSchema(fileNodeId: string): void {
|
|
// Match {% schema %}...{% endschema %}
|
|
const schemaRegex = /\{%[-]?\s*schema\s*[-]?%\}([\s\S]*?)\{%[-]?\s*endschema\s*[-]?%\}/g;
|
|
let match;
|
|
|
|
while ((match = schemaRegex.exec(this.source)) !== null) {
|
|
const [fullMatch, schemaContent] = match;
|
|
const startLine = this.getLineNumber(match.index);
|
|
const endLine = this.getLineNumber(match.index + fullMatch.length);
|
|
|
|
// Try to parse the schema JSON to get the name
|
|
let schemaName = 'schema';
|
|
try {
|
|
const schemaJson = JSON.parse(schemaContent!);
|
|
if (schemaJson.name) {
|
|
schemaName = schemaJson.name;
|
|
}
|
|
} catch {
|
|
// Schema isn't valid JSON, use default name
|
|
}
|
|
|
|
// Create a node for the schema
|
|
const nodeId = generateNodeId(this.filePath, 'constant', `schema:${schemaName}`, startLine);
|
|
|
|
const node: Node = {
|
|
id: nodeId,
|
|
kind: 'constant',
|
|
name: schemaName,
|
|
qualifiedName: `${this.filePath}::schema:${schemaName}`,
|
|
filePath: this.filePath,
|
|
language: 'liquid',
|
|
startLine,
|
|
endLine,
|
|
startColumn: match.index - this.getLineStart(startLine),
|
|
endColumn: 0,
|
|
docstring: schemaContent?.trim().substring(0, 200), // Store first 200 chars as docstring
|
|
updatedAt: Date.now(),
|
|
};
|
|
|
|
this.nodes.push(node);
|
|
|
|
// Add containment edge from file
|
|
this.edges.push({
|
|
source: fileNodeId,
|
|
target: nodeId,
|
|
kind: 'contains',
|
|
});
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Extract {% assign var = value %} statements
|
|
*/
|
|
private extractAssignments(fileNodeId: string): void {
|
|
// Match {% assign variable_name = ... %}
|
|
const assignRegex = /\{%[-]?\s*assign\s+(\w+)\s*=/g;
|
|
let match;
|
|
|
|
while ((match = assignRegex.exec(this.source)) !== null) {
|
|
const [, variableName] = match;
|
|
const line = this.getLineNumber(match.index);
|
|
|
|
// Create a variable node
|
|
const nodeId = generateNodeId(this.filePath, 'variable', variableName!, line);
|
|
|
|
const node: Node = {
|
|
id: nodeId,
|
|
kind: 'variable',
|
|
name: variableName!,
|
|
qualifiedName: `${this.filePath}::${variableName}`,
|
|
filePath: this.filePath,
|
|
language: 'liquid',
|
|
startLine: line,
|
|
endLine: line,
|
|
startColumn: match.index - this.getLineStart(line),
|
|
endColumn: match.index - this.getLineStart(line) + match[0].length,
|
|
updatedAt: Date.now(),
|
|
};
|
|
|
|
this.nodes.push(node);
|
|
|
|
// Add containment edge from file
|
|
this.edges.push({
|
|
source: fileNodeId,
|
|
target: nodeId,
|
|
kind: 'contains',
|
|
});
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Get the line number for a character index
|
|
*/
|
|
private getLineNumber(index: number): number {
|
|
const substring = this.source.substring(0, index);
|
|
return (substring.match(/\n/g) || []).length + 1;
|
|
}
|
|
|
|
/**
|
|
* Get the character index of the start of a line
|
|
*/
|
|
private getLineStart(lineNumber: number): number {
|
|
const lines = this.source.split('\n');
|
|
let index = 0;
|
|
for (let i = 0; i < lineNumber - 1 && i < lines.length; i++) {
|
|
index += lines[i]!.length + 1; // +1 for newline
|
|
}
|
|
return index;
|
|
}
|
|
}
|