feat(extraction+resolution): Astro support — frontmatter/template extraction + src/pages routes (#768) (#815)

.astro files were not indexed at all, leaving a typical Astro site mostly
invisible to search/impact/explore. New AstroExtractor (Svelte/Vue SFC
pattern): component node per file, TS frontmatter + <script> blocks
delegated to the TypeScript extractor, template {fn(...)} calls (incl. the
multiline `{posts.map((post) => (` opening line), PascalCase component-tag
references. New astroResolver: Astro global + astro:* virtual modules as
framework-provided, component resolution with the #764 ambiguity rule,
src/pages/ file-based routes ([param]→:param, [...rest]→*rest, _-prefixed
and *.config.* excluded). SFC languages now preload the TS/JS grammars
their extractors delegate to (a pure-SFC file set previously had none
loaded). Also fixes a pre-existing Svelte/Vue script-block off-by-one that
reported every script symbol one line low.

Validated per the playbook: stalux (the issue's repro) 54/54 .astro files
indexed, getIconNode found at its exact line, 14/14 routes, 93.0% fair
cross-file coverage; AstroPaper 27/27 components, 13/13 routes (underscore
dirs correctly excluded), explore connects page→Card→Datetime through the
jsx-render synthesizer; node/edge counts stable across re-syncs.

Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Colby Mchenry
2026-06-11 18:50:11 -05:00
committed by GitHub
co-authored by Claude Opus 4.8
parent 763ee9c825
commit 823ffd1c3d
15 changed files with 934 additions and 16 deletions
+365
View File
@@ -0,0 +1,365 @@
import { Node, Edge, ExtractionResult, ExtractionError, UnresolvedReference } from '../types';
import { generateNodeId } from './tree-sitter-helpers';
import { TreeSitterExtractor } from './tree-sitter';
import { isLanguageSupported } from './grammars';
/**
* Astro built-in components — compiler-provided (`<Fragment>`) or shipped by
* `astro:components` (`<Code>`, `<Debug>`), not user code.
*/
const ASTRO_BUILTIN_COMPONENTS = new Set(['Fragment', 'Code', 'Debug']);
/**
* AstroExtractor - Extracts code relationships from Astro component files
*
* Astro files are multi-language: a TypeScript frontmatter block fenced by
* `---` lines, a JSX-like HTML template, and optional <script>/<style> blocks.
* Rather than parsing a full Astro grammar, we extract the frontmatter and
* <script> contents and delegate them to the TypeScript TreeSitterExtractor
* (Astro processes both as TypeScript by default — no `lang` attr needed).
*
* Also extracts function calls from template expressions (`{fn(...)}`) and
* component usages (`<PascalCase>`) so cross-file edges are captured even
* when the only reference lives in markup.
*
* Every .astro file produces a component node (Astro components are always
* importable).
*/
export class AstroExtractor {
private filePath: string;
private source: string;
private nodes: Node[] = [];
private edges: Edge[] = [];
private unresolvedReferences: UnresolvedReference[] = [];
private errors: ExtractionError[] = [];
constructor(filePath: string, source: string) {
this.filePath = filePath;
this.source = source;
}
/**
* Extract from Astro source
*/
extract(): ExtractionResult {
const startTime = Date.now();
try {
// Create component node for the .astro file itself
const componentNode = this.createComponentNode();
// Extract and process the frontmatter block (--- fenced, TypeScript)
const frontmatter = this.extractFrontmatter();
if (frontmatter) {
this.processScriptContent(frontmatter, componentNode.id, 'frontmatter');
}
// Extract and process <script> blocks (client-side, TypeScript-capable)
for (const block of this.extractScriptBlocks()) {
this.processScriptContent(block, componentNode.id, 'script');
}
// Ranges the template scans must skip: frontmatter + <script>/<style>
const coveredRanges = this.getCoveredRanges(frontmatter);
// Extract function calls from template expressions ({fn(...)})
this.extractTemplateCalls(componentNode.id, coveredRanges);
// Extract component usages from template (<ComponentName>)
this.extractTemplateComponents(componentNode.id, coveredRanges);
} catch (error) {
this.errors.push({
message: `Astro extraction error: ${error instanceof Error ? error.message : String(error)}`,
severity: 'error',
code: 'parse_error',
});
}
return {
nodes: this.nodes,
edges: this.edges,
unresolvedReferences: this.unresolvedReferences,
errors: this.errors,
durationMs: Date.now() - startTime,
};
}
/**
* Create a component node for the .astro file
*/
private createComponentNode(): Node {
const lines = this.source.split('\n');
const fileName = this.filePath.split(/[/\\]/).pop() || this.filePath;
const componentName = fileName.replace(/\.astro$/, '');
const id = generateNodeId(this.filePath, 'component', componentName, 1);
const node: Node = {
id,
kind: 'component',
name: componentName,
qualifiedName: `${this.filePath}::${componentName}`,
filePath: this.filePath,
language: 'astro',
startLine: 1,
endLine: lines.length,
startColumn: 0,
endColumn: lines[lines.length - 1]?.length || 0,
isExported: true, // Astro components are always importable
updatedAt: Date.now(),
};
this.nodes.push(node);
return node;
}
/**
* Extract the frontmatter block: the content between the opening `---`
* fence (first non-blank line of the file) and the closing `---` fence.
* An unclosed fence is treated as "no frontmatter" rather than swallowing
* the whole template as TypeScript.
*
* Returns the content plus its 0-indexed start line, or null.
*/
private extractFrontmatter(): { content: string; startLine: number; endLine: number } | null {
const lines = this.source.split('\n');
// Opening fence must be the first non-blank line
let openIdx = -1;
for (let i = 0; i < lines.length; i++) {
const trimmed = lines[i]!.trim();
if (trimmed === '') continue;
if (trimmed === '---') openIdx = i;
break;
}
if (openIdx === -1) return null;
// Closing fence
let closeIdx = -1;
for (let i = openIdx + 1; i < lines.length; i++) {
if (lines[i]!.trim() === '---') {
closeIdx = i;
break;
}
}
if (closeIdx === -1) return null;
return {
content: lines.slice(openIdx + 1, closeIdx).join('\n'),
startLine: openIdx + 1, // 0-indexed line where content starts
endLine: closeIdx, // 0-indexed line of the closing fence
};
}
/**
* Extract <script> blocks from the template portion
*/
private extractScriptBlocks(): Array<{ content: string; startLine: number }> {
const blocks: Array<{ content: string; startLine: number }> = [];
const scriptRegex = /<script(\s[^>]*)?>(?<content>[\s\S]*?)<\/script>/g;
let match;
while ((match = scriptRegex.exec(this.source)) !== null) {
const content = match.groups?.content || match[2] || '';
// Calculate the 0-indexed line where the content begins. The content
// starts right after the opening tag's `>` — its leading `\n` is part
// of the content, so relative line 1 sits ON the tag's closing line
// (do not add 1 here; that double-counts the embedded newline).
const beforeScript = this.source.substring(0, match.index);
const scriptTagLine = (beforeScript.match(/\n/g) || []).length;
const openingTag = match[0].substring(0, match[0].indexOf('>') + 1);
const openingTagLines = (openingTag.match(/\n/g) || []).length;
const contentStartLine = scriptTagLine + openingTagLines; // 0-indexed
blocks.push({ content, startLine: contentStartLine });
}
return blocks;
}
/**
* Process frontmatter / script content by delegating to TreeSitterExtractor.
* Astro treats both as TypeScript by default.
*/
private processScriptContent(
block: { content: string; startLine: number },
componentNodeId: string,
label: 'frontmatter' | 'script'
): void {
if (!isLanguageSupported('typescript')) {
this.errors.push({
message: `Parser for typescript not available, cannot parse Astro ${label} block`,
severity: 'warning',
});
return;
}
// Delegate to TreeSitterExtractor
const extractor = new TreeSitterExtractor(this.filePath, block.content, 'typescript');
const result = extractor.extract();
// Offset line numbers from the block back to .astro file positions
for (const node of result.nodes) {
node.startLine += block.startLine;
node.endLine += block.startLine;
node.language = 'astro'; // Mark as astro, not TS
this.nodes.push(node);
// Add containment edge from component to this node
this.edges.push({
source: componentNodeId,
target: node.id,
kind: 'contains',
});
}
// Offset edges (they reference line numbers)
for (const edge of result.edges) {
if (edge.line) {
edge.line += block.startLine;
}
this.edges.push(edge);
}
// Offset unresolved references
for (const ref of result.unresolvedReferences) {
ref.line += block.startLine;
ref.filePath = this.filePath;
ref.language = 'astro';
this.unresolvedReferences.push(ref);
}
// Carry over errors
for (const error of result.errors) {
if (error.line) {
error.line += block.startLine;
}
this.errors.push(error);
}
}
/**
* Line ranges (0-indexed, inclusive) the template scans must skip:
* the frontmatter block and <script>/<style> blocks.
*/
private getCoveredRanges(
frontmatter: { startLine: number; endLine: number } | null
): Array<[number, number]> {
const coveredRanges: Array<[number, number]> = [];
if (frontmatter) {
// Cover from the opening fence line through the closing fence line
coveredRanges.push([frontmatter.startLine - 1, frontmatter.endLine]);
}
const tagRegex = /<(script|style)(\s[^>]*)?>[\s\S]*?<\/\1>/g;
let tagMatch;
while ((tagMatch = tagRegex.exec(this.source)) !== null) {
const startLine = (this.source.substring(0, tagMatch.index).match(/\n/g) || []).length;
const endLine = startLine + (tagMatch[0].match(/\n/g) || []).length;
coveredRanges.push([startLine, endLine]);
}
return coveredRanges;
}
/**
* Extract function calls from Astro template expressions.
*
* Astro templates embed JSX-like expressions (`{formatDate(post.date)}`,
* `class:list={cn(...)}`), so calls frequently live in markup rather than
* the frontmatter. We scan template lines for `{expression}` groups and
* extract call patterns from them. A `{` group left open at end-of-line
* (the pervasive `{posts.map((post) => (` pattern) contributes the calls
* on its opening line.
*/
private extractTemplateCalls(
componentNodeId: string,
coveredRanges: Array<[number, number]>
): void {
const lines = this.source.split('\n');
// Complete groups: {...} — excluding JSX comments ({/* ... */})
const exprRegex = /\{([^}/][^}]*)\}/g;
// A group opened but not closed on this line
const openExprRegex = /\{([^}/][^}]*)$/;
for (let lineIdx = 0; lineIdx < lines.length; lineIdx++) {
if (coveredRanges.some(([start, end]) => lineIdx >= start && lineIdx <= end)) continue;
const line = lines[lineIdx]!;
const exprs: Array<{ text: string; offset: number }> = [];
let exprMatch;
while ((exprMatch = exprRegex.exec(line)) !== null) {
exprs.push({ text: exprMatch[1]!, offset: exprMatch.index });
}
const openMatch = openExprRegex.exec(line.replace(exprRegex, ''));
if (openMatch) {
exprs.push({ text: openMatch[1]!, offset: line.lastIndexOf('{') });
}
for (const expr of exprs) {
// Extract function calls: identifiers followed by (
// Matches: cn(...), formatDate(...), obj.method(...)
const callRegex = /\b([a-zA-Z_$][\w$.]*)\s*\(/g;
let callMatch;
while ((callMatch = callRegex.exec(expr.text)) !== null) {
const calleeName = callMatch[1]!;
// Skip control-flow keywords valid inside expressions
if (calleeName === 'if' || calleeName === 'await' || calleeName === 'function') continue;
this.unresolvedReferences.push({
fromNodeId: componentNodeId,
referenceName: calleeName,
referenceKind: 'calls',
line: lineIdx + 1, // 1-indexed
column: expr.offset + callMatch.index,
filePath: this.filePath,
language: 'astro',
});
}
}
}
}
/**
* Extract component usages from the Astro template.
*
* PascalCase tags like <Layout>, <PostCard /> represent component
* instantiations — analogous to function calls in imperative code.
* Lowercase tags are native HTML (Astro does not register kebab-case
* components the way Vue does, so those are real custom elements and
* are skipped).
*/
private extractTemplateComponents(
componentNodeId: string,
coveredRanges: Array<[number, number]>
): void {
const lines = this.source.split('\n');
// Opening/self-closing tags (closing tags </Foo> start with </ so won't match)
const componentTagRegex = /<([A-Z][a-zA-Z0-9_$]*)\b/g;
for (let lineIdx = 0; lineIdx < lines.length; lineIdx++) {
if (coveredRanges.some(([start, end]) => lineIdx >= start && lineIdx <= end)) continue;
const line = lines[lineIdx]!;
let match;
while ((match = componentTagRegex.exec(line)) !== null) {
const componentName = match[1]!;
if (ASTRO_BUILTIN_COMPONENTS.has(componentName)) continue;
this.unresolvedReferences.push({
fromNodeId: componentNodeId,
referenceName: componentName,
referenceKind: 'references',
line: lineIdx + 1, // 1-indexed
column: match.index + 1,
filePath: this.filePath,
language: 'astro',
});
}
}
}
}
+14 -3
View File
@@ -10,7 +10,7 @@ import * as path from 'path';
import { Parser, Language as WasmLanguage } from 'web-tree-sitter';
import { Language } from '../types';
export type GrammarLanguage = Exclude<Language, 'svelte' | 'vue' | 'liquid' | 'razor' | 'yaml' | 'twig' | 'xml' | 'properties' | 'unknown'>;
export type GrammarLanguage = Exclude<Language, 'svelte' | 'vue' | 'astro' | 'liquid' | 'razor' | 'yaml' | 'twig' | 'xml' | 'properties' | 'unknown'>;
/**
* WASM filename map — maps each language to its .wasm grammar file
@@ -93,6 +93,7 @@ export const EXTENSION_MAP: Record<string, Language> = {
'.liquid': 'liquid',
'.svelte': 'svelte',
'.vue': 'vue',
'.astro': 'astro',
'.pas': 'pascal',
'.dpr': 'pascal',
'.dpk': 'pascal',
@@ -183,6 +184,14 @@ export async function loadGrammarsForLanguages(languages: Language[]): Promise<v
await initGrammars();
}
// SFC languages (svelte/vue/astro) have no grammar of their own — their
// extractors delegate <script>/frontmatter content to the TS/JS extractor,
// so those grammars must be loaded even when no plain .ts/.js file is in
// the index set (e.g. a pure-.astro content site).
if (languages.some((l) => l === 'svelte' || l === 'vue' || l === 'astro')) {
languages = [...languages, 'typescript', 'javascript'];
}
// Deduplicate and filter to languages that have WASM grammars and aren't already loaded
const toLoad = [...new Set(languages)].filter(
(lang): lang is GrammarLanguage =>
@@ -300,6 +309,7 @@ function looksLikeObjc(source: string): boolean {
export function isLanguageSupported(language: Language): boolean {
if (language === 'svelte') return true; // custom extractor (script block delegation)
if (language === 'vue') return true; // custom extractor (script block delegation)
if (language === 'astro') return true; // custom extractor (frontmatter/script block delegation)
if (language === 'liquid') return true; // custom regex extractor
if (language === 'razor') return true; // custom RazorExtractor (.cshtml/.razor markup)
if (language === 'yaml') return true; // file-level tracking only; Drupal routing extraction via framework resolver
@@ -314,7 +324,7 @@ export function isLanguageSupported(language: Language): boolean {
* Check if a grammar has been loaded and is ready for parsing.
*/
export function isGrammarLoaded(language: Language): boolean {
if (language === 'svelte' || language === 'vue' || language === 'liquid' || language === 'razor') return true;
if (language === 'svelte' || language === 'vue' || language === 'astro' || language === 'liquid' || language === 'razor') return true;
if (language === 'yaml' || language === 'twig') return true; // no WASM grammar needed
if (language === 'xml' || language === 'properties') return true; // no WASM grammar needed
return languageCache.has(language);
@@ -337,7 +347,7 @@ export function isFileLevelOnlyLanguage(language: Language): boolean {
* Get all supported languages (those with grammar definitions).
*/
export function getSupportedLanguages(): Language[] {
return [...(Object.keys(WASM_GRAMMAR_FILES) as GrammarLanguage[]), 'svelte', 'vue', 'liquid'];
return [...(Object.keys(WASM_GRAMMAR_FILES) as GrammarLanguage[]), 'svelte', 'vue', 'astro', 'liquid'];
}
/**
@@ -403,6 +413,7 @@ export function getLanguageDisplayName(language: Language): string {
dart: 'Dart',
svelte: 'Svelte',
vue: 'Vue',
astro: 'Astro',
liquid: 'Liquid',
pascal: 'Pascal / Delphi',
scala: 'Scala',
+6 -3
View File
@@ -135,13 +135,16 @@ export class SvelteExtractor {
// Detect module script
const isModule = /context\s*=\s*["']module["']/.test(attrs);
// Calculate start line of the script content (line after <script>)
// Calculate the 0-indexed line where the content begins. The content
// starts right after the opening tag's `>` — its leading `\n` is part
// of the content, so relative line 1 sits ON the tag's closing line
// (adding 1 here double-counted the embedded newline and shifted every
// script-block symbol down a line).
const beforeScript = this.source.substring(0, match.index);
const scriptTagLine = (beforeScript.match(/\n/g) || []).length;
// The content starts on the line after the opening <script> tag
const openingTag = match[0].substring(0, match[0].indexOf('>') + 1);
const openingTagLines = (openingTag.match(/\n/g) || []).length;
const contentStartLine = scriptTagLine + openingTagLines + 1; // 0-indexed line
const contentStartLine = scriptTagLine + openingTagLines; // 0-indexed line
blocks.push({
content,
+5
View File
@@ -24,6 +24,7 @@ import { EXTRACTORS } from './languages';
import { LiquidExtractor } from './liquid-extractor';
import { RazorExtractor } from './razor-extractor';
import { SvelteExtractor } from './svelte-extractor';
import { AstroExtractor } from './astro-extractor';
import { DfmExtractor } from './dfm-extractor';
import { VueExtractor } from './vue-extractor';
import { MyBatisExtractor } from './mybatis-extractor';
@@ -4752,6 +4753,10 @@ export function extractFromSource(
// Use custom extractor for Vue
const extractor = new VueExtractor(filePath, source);
result = extractor.extract();
} else if (detectedLanguage === 'astro') {
// Use custom extractor for Astro (frontmatter + template delegation)
const extractor = new AstroExtractor(filePath, source);
result = extractor.extract();
} else if (detectedLanguage === 'liquid') {
// Use custom extractor for Liquid
const extractor = new LiquidExtractor(filePath, source);
+6 -3
View File
@@ -143,13 +143,16 @@ export class VueExtractor {
// Detect <script setup>
const isSetup = /\bsetup\b/.test(attrs);
// Calculate start line of the script content (line after <script>)
// Calculate the 0-indexed line where the content begins. The content
// starts right after the opening tag's `>` — its leading `\n` is part
// of the content, so relative line 1 sits ON the tag's closing line
// (adding 1 here double-counted the embedded newline and shifted every
// script-block symbol down a line).
const beforeScript = this.source.substring(0, match.index);
const scriptTagLine = (beforeScript.match(/\n/g) || []).length;
// The content starts on the line after the opening <script> tag
const openingTag = match[0].substring(0, match[0].indexOf('>') + 1);
const openingTagLines = (openingTag.match(/\n/g) || []).length;
const contentStartLine = scriptTagLine + openingTagLines + 1; // 0-indexed line
const contentStartLine = scriptTagLine + openingTagLines; // 0-indexed line
blocks.push({
content,