fix(prompt-hook): close the segment-vocab integrity gaps (#1141, #1142, #1144, #1145, #1146) (#1150)

Five hardening fixes to the #1136 MEDIUM (graph-derived) tier:

- #1141: updateNode() now writes the segment vocabulary like insertNode()
  does — framework post-extract renames (NestJS route prefixing) left the
  new name permanently unsearchable (the old rows orphaned, the backfill
  gated on an EMPTY vocab, so even a full re-index re-created the drift).
- #1142: new CodeGraph.healSegmentVocabIfEmpty() — the hook opens the
  graph without sync, so a database migrated from pre-vocab schema kept
  the MEDIUM tier dormant until some unrelated sync ran. The hook heals
  on first use (one SELECT when populated; lock-aware, defers to a
  running sync) and records noop-vocab-empty when it can't.
- #1144: a name whose only nodes are file/import kind is skipped instead
  of falling back to surfacing an import statement as a matched symbol;
  import specifiers no longer enter the vocab at all (shared
  isSegmentableKind gate across insertNode/updateNode/rebuild page query)
  since they can never be surfaced and only inflate rarity statistics.
- #1145: plural variant folding is keyed on English plural spelling —
  bare-s plurals no longer mint a bogus -es sibling (services→servic),
  unambiguous sibilant-es plurals no longer mint a bogus -s sibling
  (classes→classe), trailing -ss singulars no longer strip (class→clas);
  genuinely ambiguous endings (caches/databases) still emit both keys.
- #1146: getSegmentCoOccurrence folds variants to their original word
  inside the SQL (CASE mapping + COUNT(DISTINCT word)) so a plural pair
  of ONE word can't tie with a genuine two-word match and crowd it past
  the pre-fold ORDER BY/LIMIT; the JS re-check stays as the honesty layer.

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Colby Mchenry
2026-07-02 17:23:54 -05:00
committed by GitHub
co-authored by Claude Fable 5
parent be55b93d02
commit 35611b92bb
7 changed files with 233 additions and 20 deletions
+50 -11
View File
@@ -319,8 +319,19 @@ export class QueryBuilder {
// before use, and a full index clears the table at its start. File nodes
// are excluded: a file's basename duplicates the symbols inside it
// (state-machine.ts / OrderStateMachine), which double-counts every
// concept and defeats the singleton-vs-cluster rarity statistics.
if (node.kind !== 'file') this.insertNameSegments(node.name);
// concept and defeats the singleton-vs-cluster rarity statistics. Import
// nodes are excluded too (#1144): they're named after module specifiers
// ("external-unindexed-pkg", "./utils/helpers"), not symbols — an
// import-only name can never be surfaced (getSegmentMatches requires a
// real definition), so its rows would only inflate the rarity statistics.
if (this.isSegmentableKind(node.kind)) this.insertNameSegments(node.name);
}
/** Which node kinds contribute their name to the segment vocabulary — the
* single gate shared by insertNode, updateNode, and the rebuild page query
* (getDistinctNodeNames), so the write paths can't drift apart. */
private isSegmentableKind(kind: string): boolean {
return kind !== 'file' && kind !== 'import';
}
/** Write `name`'s segments into name_segment_vocab (idempotent). */
@@ -412,6 +423,16 @@ export class QueryBuilder {
returnType: node.returnType ?? null,
updatedAt: node.updatedAt ?? Date.now(),
});
// updateNode is a second real write path to `nodes` — framework
// post-extract passes rewrite names through it (NestJS route prefixing),
// and a renamed node's new name must reach the segment vocabulary just
// like an inserted one's (#1141). Without this the rename left the new
// name permanently unsearchable: the old name's rows became honest-gate
// orphans and the only backfill is gated on the vocab being EMPTY.
// insertNameSegments is idempotent (in-memory set + INSERT OR IGNORE),
// so no name-changed check is needed.
if (this.isSegmentableKind(node.kind)) this.insertNameSegments(node.name);
}
/**
@@ -461,11 +482,12 @@ export class QueryBuilder {
return row === undefined;
}
/** One page of distinct non-file node names, for batched vocab rebuilds
* (file basenames are excluded from the vocab — see insertNode). */
/** One page of distinct segmentable node names, for batched vocab rebuilds
* (file basenames and import specifiers are excluded from the vocab — see
* insertNode). */
getDistinctNodeNames(limit: number, offset: number): string[] {
const rows = this.db
.prepare("SELECT DISTINCT name FROM nodes WHERE kind != 'file' ORDER BY name LIMIT ? OFFSET ?")
.prepare("SELECT DISTINCT name FROM nodes WHERE kind NOT IN ('file', 'import') ORDER BY name LIMIT ? OFFSET ?")
.all(limit, offset) as Array<{ name: string }>;
return rows.map((r) => r.name);
}
@@ -478,17 +500,29 @@ export class QueryBuilder {
}
/**
* Names whose segments cover at least `minSegments` of the given segments
* Names whose segments cover at least `minWords` distinct PROMPT WORDS
* the co-occurrence probe behind the prompt hook's medium tier: the words
* "state" and "machine" both being segments of `OrderStateMachine` is strong
* evidence the prompt names that symbol in prose. Ordered by coverage.
*
* Takes (segment variant → original word) pairs and folds variants back to
* their word INSIDE the SQL: a name matching both `service` and `services`
* counts ONE word, not two. Counting raw variants let plural-variant pairs
* of a single word tie with genuine two-word matches and — because ORDER
* BY/LIMIT run here, before any JS-side re-check — crowd a real match past
* the LIMIT on vocab-heavy repos (#1146).
*/
getSegmentCoOccurrence(segments: string[], minSegments: number, limit: number): Array<{ name: string; matches: number }> {
if (segments.length === 0) return [];
const placeholders = segments.map(() => '?').join(', ');
getSegmentCoOccurrence(
variants: Array<{ segment: string; word: string }>,
minWords: number,
limit: number,
): Array<{ name: string; matches: number }> {
if (variants.length === 0) return [];
const placeholders = variants.map(() => '?').join(', ');
const whens = variants.map(() => 'WHEN ? THEN ?').join(' ');
const rows = this.db
.prepare(
`SELECT name, COUNT(DISTINCT segment) AS matches
`SELECT name, COUNT(DISTINCT CASE segment ${whens} END) AS matches
FROM name_segment_vocab
WHERE segment IN (${placeholders})
GROUP BY name
@@ -496,7 +530,12 @@ export class QueryBuilder {
ORDER BY matches DESC, length(name) ASC
LIMIT ?`,
)
.all(...segments, minSegments, limit) as Array<{ name: string; matches: number }>;
.all(
...variants.flatMap((v) => [v.segment, v.word]),
...variants.map((v) => v.segment),
minWords,
limit,
) as Array<{ name: string; matches: number }>;
return rows;
}