feat(mcp): score-proportional byte allocation for explore, with a relative cliff (CG-12, #1500)
The explore envelope used to follow FILE SIZE, not relevance. Every admitted
file was capped at the same flat `maxCharsPerFile`, while the whole-file rule
handed anything under `maxCharsPerFile * 3` its entire contents — a 3x swing
decided by how big a file happened to be:
- self-query: `memory-budget.ts` (score 18) shipped whole and took 51.2% of
the response; `src/mcp/tools.ts` (score 41, 4x the graph mass, 3x the term
hits — it holds the allocator itself) was clipped at 3,800 and got 32.9%.
- #1500 Go fixture: two generated CRUD files shipped whole at ~4.5K each AND
consumed two of the tier's four file slots, so `BuildPayslip` — the
hand-written "calculate" half of the question — ranked #6 and never
rendered at all.
`allocateExploreBudget` now reserves each ranked file a share of the envelope
before anything renders, so the render loop spends a reservation instead of
racing for whatever the files above it left:
- weight = score x worth x (spine ? 2 : 1), where `worth` is `rankPenalty`
applied a SECOND time — ranking answers "is this file about the query",
allocation answers "will these bytes teach the agent anything", and
generated CRUD can legitimately rank while its bytes stay boilerplate;
- a relative cliff at 15% of the top weight (capped at SCORE_FLOOR_MAX, so a
god-file can't silence peers the score floor just admitted) gives a file
ZERO source — path, symbols and line numbers only — and crucially frees its
`maxFiles` slot for a file that earns its bytes;
- every admitted file gets MIN_CHARS, then the remainder splits by weight:
the floor keeps a diffuse survey question returning a spread, the remainder
concentrates a precise one;
- the flat per-file cap is retired as the primary guard, leaving a 70%-of-
envelope safety valve.
Two changes were needed to make the reservation bite: an oversize cluster now
shrinks by whole MEMBER symbol ranges (a single-cluster god-file previously
took ~40% more than allotted, and the file below it was dropped for lack of
room), and the arrival-order budget stops are gone — they cut files by the
order they were reached rather than by merit.
Measured: payroll-go answer group 25.6% -> 78.7%, generated 57.4% -> 0%, and
`func (s *Service) BuildPayslip` now delivered; self-query `tools.ts` 18.5% ->
60.6%, past the epic's >50% bar. Controls hold: cobra/gin diffuse survey
queries keep their file spread (3->3, 3->4), express's middleware query is
byte-identical, and gin's flow query moves its top file from the thin `ginS`
singleton wrapper to `routergroup.go`.
One documented exception to "no previously-unclipped file becomes clipped":
`memory-budget.ts` was unclipped-whole at 5,672 and now clusters within its
3.1K reservation. That is the epic's own diagnosis of the bug — it scored 18
against 58 and was taking the larger slice purely for being small.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
a3898cdc70
commit
5f7f5f59df
@@ -218,11 +218,13 @@ describe('#1500 — generated Go CRUD beside a hand-written payroll workflow', (
|
||||
* `cycle.go` now delivers 38.9% and the generated layer 23.5%. Four of the
|
||||
* five gates below are green and are now live regressions.
|
||||
*
|
||||
* STILL OPEN, for CG-12 (proportional byte allocation): the render loop
|
||||
* still allocates by FILE SIZE within the ranked set, and `maxFiles` is 4 at
|
||||
* this tier — so `payslip_builder.go` ranks #6 and never renders, and
|
||||
* `func (s *Service) BuildPayslip` is absent. Its `it.fails` passes ONLY
|
||||
* while that is still true. ⚠ When it goes red, remove `.fails`, keep it.
|
||||
* AFTER CG-12 (score-proportional allocation): every file's share of the
|
||||
* envelope is reserved before anything renders, and a file under 15% of the
|
||||
* top weight gets no source at all — so the two generated files cliff to
|
||||
* pointers, hand their `maxFiles` slots to the hand-written store and
|
||||
* builder, and the answer group takes ~79% with the generated layer at 0%.
|
||||
* `func (s *Service) BuildPayslip` — the "calculate" half of the question —
|
||||
* finally reaches the agent. All gates below are live regressions now.
|
||||
*/
|
||||
it('CG-10 GATE: concentrates the envelope on the hand-written workflow', () => {
|
||||
expect(answerShare()).toBeGreaterThanOrEqual(0.55);
|
||||
@@ -249,14 +251,22 @@ describe('#1500 — generated Go CRUD beside a hand-written payroll workflow', (
|
||||
expect(response).toContain('s.store.Upsert(ctx, slip)');
|
||||
});
|
||||
|
||||
it.fails('CG-12 GATE: delivers the calculation the question asks about', () => {
|
||||
// `payslip_builder.go` ranks #6; the tier's maxFiles is 4 and the render
|
||||
// loop spends by file size, so it never gets bytes. Score-proportional
|
||||
// allocation (CG-12) is what closes this.
|
||||
it('CG-12 GATE: delivers the calculation the question asks about', () => {
|
||||
// `payslip_builder.go` ranks #6 and the tier's maxFiles is 4 — it reaches
|
||||
// the response only because the two generated files cliff to pointers
|
||||
// WITHOUT consuming a slot. That slot hand-off is the CG-12 mechanism.
|
||||
expect(bytes.get('internal/usecase/payroll/payslip_builder.go') ?? 0).toBeGreaterThan(0);
|
||||
expect(response).toContain('func (s *Service) BuildPayslip');
|
||||
});
|
||||
|
||||
it('CG-12 GATE: withholds the generated CRUD bytes but still names it', () => {
|
||||
// A cliffed file costs ~100 chars instead of ~4,500, and stays one
|
||||
// follow-up explore away — withholding is only cheap if it stays nameable.
|
||||
expect(bytes.get('internal/gen/fkit/payroll/payslip.go') ?? 0).toBe(0);
|
||||
expect(response).toContain('**Not shown above — explore these names for their source**');
|
||||
expect(response).toMatch(/internal\/gen\/fkit\/payroll\/payslip\.go: \w+:\d+/);
|
||||
});
|
||||
|
||||
it('records the shape of the allocation so a regression is legible', () => {
|
||||
// Not a gate — a snapshot of the split, so a future change that shifts the
|
||||
// numbers shows up in the diff rather than silently flipping a gate.
|
||||
@@ -269,7 +279,7 @@ describe('#1500 — generated Go CRUD beside a hand-written payroll workflow', (
|
||||
}).toEqual({
|
||||
generatedWinsEnvelope: false,
|
||||
workflowFileDelivers: true,
|
||||
builderFileDelivers: false,
|
||||
builderFileDelivers: true,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -202,7 +202,9 @@ describe('codegraph_explore allocation diagnostic', () => {
|
||||
/files [\d,]+ grouped .*past low-value filter .*past score floor \(>=[\d.]+\).*in output \(maxFiles \d+\)/,
|
||||
);
|
||||
// Per-file columns.
|
||||
expect(out).toMatch(/#\s+alloc%\s+deliv%\s+bytes\s+score\s+graph\s+hits\s+pen\s+flags\s+render\s+file/);
|
||||
expect(out).toMatch(/#\s+alloc%\s+deliv%\s+bytes\s+reserved\s+score\s+graph\s+hits\s+pen\s+flags\s+render\s+file/);
|
||||
// The proportional split (CG-12): what was reserved, and where the cliff fell.
|
||||
expect(out).toMatch(/allocation [\d,]+ reserved of [\d,]+ pool · cliff at weight [\d.]+/);
|
||||
expect(out).toContain('src/session.ts');
|
||||
expect(out).toMatch(/\d+\.\d%/);
|
||||
// Kind mix — what each file's score was bought with.
|
||||
|
||||
@@ -0,0 +1,214 @@
|
||||
/**
|
||||
* Score-proportional byte allocation for codegraph_explore (CG-12 / #1500).
|
||||
*
|
||||
* `allocateExploreBudget` decides, before anything renders, how many chars of
|
||||
* source each ranked file may spend. Its contract is what stops the explore
|
||||
* envelope from following FILE SIZE — which is the bug #1500 reported: a small
|
||||
* weakly-relevant file shipped whole while the file that actually answered the
|
||||
* question was clipped at a flat per-file cap.
|
||||
*
|
||||
* These pin the allocator's invariants directly. End-to-end behaviour on the two
|
||||
* regression fixtures lives in `explore-allocation-1500.test.ts`.
|
||||
*/
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { allocateExploreBudget, getExploreOutputBudget } from '../src/mcp/tools';
|
||||
import type { ExploreAllocationCandidate } from '../src/mcp/tools';
|
||||
|
||||
/** A candidate with sane defaults — tests override only what they're about. */
|
||||
const cand = (
|
||||
path: string,
|
||||
score: number,
|
||||
extra: Partial<ExploreAllocationCandidate> = {},
|
||||
): ExploreAllocationCandidate => ({ path, score, worth: 1, spine: false, ...extra });
|
||||
|
||||
const TIER_FILE_COUNTS = [10, 100, 300, 1000, 4000, 10000, 20000, 60000];
|
||||
|
||||
describe('allocateExploreBudget — proportional split', () => {
|
||||
const budget = getExploreOutputBudget(1000); // 24,000 / 6,500 / 8 files
|
||||
|
||||
it('gives the higher-scoring file the bigger share', () => {
|
||||
const { allowances } = allocateExploreBudget(
|
||||
[cand('a.ts', 40), cand('b.ts', 10)],
|
||||
budget,
|
||||
8,
|
||||
);
|
||||
expect(allowances.get('a.ts')!).toBeGreaterThan(allowances.get('b.ts')!);
|
||||
});
|
||||
|
||||
it('scales the split with the score RATIO, not just the ordering', () => {
|
||||
// The heart of the fix. Under the old flat `maxCharsPerFile` both files got
|
||||
// the same cap and the split fell out of whichever happened to be small
|
||||
// enough to ship whole; here a 4x score buys materially more than a 1.1x one.
|
||||
const wide = allocateExploreBudget([cand('a.ts', 40), cand('b.ts', 10)], budget, 8).allowances;
|
||||
const narrow = allocateExploreBudget([cand('a.ts', 22), cand('b.ts', 20)], budget, 8).allowances;
|
||||
expect(wide.get('a.ts')! / wide.get('b.ts')!)
|
||||
.toBeGreaterThan(narrow.get('a.ts')! / narrow.get('b.ts')!);
|
||||
});
|
||||
|
||||
it('never reserves more than the envelope', () => {
|
||||
const { allowances, pool } = allocateExploreBudget(
|
||||
[cand('a.ts', 90), cand('b.ts', 40), cand('c.ts', 30), cand('d.ts', 12)],
|
||||
budget,
|
||||
8,
|
||||
);
|
||||
const reserved = [...allowances.values()].reduce((s, n) => s + n, 0);
|
||||
expect(reserved).toBeLessThanOrEqual(pool);
|
||||
expect(pool).toBeLessThanOrEqual(budget.maxOutputChars);
|
||||
});
|
||||
|
||||
it('caps any single file at the MAX_SHARE safety valve', () => {
|
||||
// The per-file cap is retired as the primary guard, but a lone dominant file
|
||||
// must still not be handed the entire response.
|
||||
const { allowances } = allocateExploreBudget([cand('god.ts', 500)], budget, 8);
|
||||
expect(allowances.get('god.ts')!).toBeLessThanOrEqual(Math.round(budget.maxOutputChars * 0.7));
|
||||
});
|
||||
|
||||
it('lets the top file exceed the old flat per-file cap when it earns it', () => {
|
||||
// The regression this task exists to fix: `maxCharsPerFile` clipped the file
|
||||
// that scored 4x its peers at exactly the same 6,500 as the noise.
|
||||
const { allowances } = allocateExploreBudget(
|
||||
[cand('answer.ts', 60), cand('noise.ts', 12)],
|
||||
budget,
|
||||
8,
|
||||
);
|
||||
expect(allowances.get('answer.ts')!).toBeGreaterThan(budget.maxCharsPerFile);
|
||||
});
|
||||
});
|
||||
|
||||
describe('allocateExploreBudget — the relative cliff', () => {
|
||||
const budget = getExploreOutputBudget(1000);
|
||||
|
||||
it('gives zero source to a file far below the top score', () => {
|
||||
const { allowances, cliffed } = allocateExploreBudget(
|
||||
[cand('answer.ts', 90), cand('incidental.ts', 3)],
|
||||
budget,
|
||||
8,
|
||||
);
|
||||
expect(cliffed).toContain('incidental.ts');
|
||||
expect(allowances.has('incidental.ts')).toBe(false);
|
||||
});
|
||||
|
||||
it('is RELATIVE — the same score survives against weaker company', () => {
|
||||
const strong = allocateExploreBudget([cand('a.ts', 90), cand('b.ts', 8)], budget, 8);
|
||||
const even = allocateExploreBudget([cand('a.ts', 12), cand('b.ts', 8)], budget, 8);
|
||||
expect(strong.cliffed).toContain('b.ts');
|
||||
expect(even.cliffed).not.toContain('b.ts');
|
||||
});
|
||||
|
||||
it('never rises above the score-floor ceiling, however dominant the top file', () => {
|
||||
// A 500-scoring god-file would otherwise put the cliff at 75 and silence
|
||||
// every peer the score floor had just deliberately admitted.
|
||||
const { cliffed } = allocateExploreBudget(
|
||||
[cand('god.ts', 500), cand('peer.ts', 13), cand('peer2.ts', 11)],
|
||||
budget,
|
||||
8,
|
||||
);
|
||||
expect(cliffed).toEqual([]);
|
||||
});
|
||||
|
||||
it('doubles the penalty on bytes that are worth less (generated / low-value)', () => {
|
||||
// `worth` is `rankPenalty` applied a second time: generated CRUD can rank on
|
||||
// name collisions while its bytes stay boilerplate. Same score, different fate.
|
||||
const { cliffed } = allocateExploreBudget(
|
||||
[cand('answer.ts', 60), cand('gen.ts', 12, { worth: 0.3 }), cand('hand.ts', 12)],
|
||||
budget,
|
||||
8,
|
||||
);
|
||||
expect(cliffed).toContain('gen.ts');
|
||||
expect(cliffed).not.toContain('hand.ts');
|
||||
});
|
||||
|
||||
it('exempts flow-spine files from the cliff', () => {
|
||||
// Clipping the spine causes the Read fallback — it IS the answer to a flow
|
||||
// question — so a spine file is never zeroed on relative score alone.
|
||||
const { allowances, cliffed } = allocateExploreBudget(
|
||||
[cand('a.ts', 400), cand('spine.ts', 2, { spine: true })],
|
||||
budget,
|
||||
8,
|
||||
);
|
||||
expect(cliffed).not.toContain('spine.ts');
|
||||
expect(allowances.get('spine.ts')!).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('never cliffs every candidate — an empty response costs a round-trip', () => {
|
||||
const { allowances, cliffed } = allocateExploreBudget([cand('only.ts', 0.5)], budget, 8);
|
||||
expect(cliffed).toEqual([]);
|
||||
expect(allowances.get('only.ts')!).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('hands a cliffed file\'s maxFiles slot to the next file down', () => {
|
||||
// The mechanism that got `BuildPayslip` into the #1500 response: cliffing is
|
||||
// not just "spend fewer bytes here", it frees the SLOT too.
|
||||
const { allowances } = allocateExploreBudget(
|
||||
[cand('a.ts', 90), cand('noise.ts', 2), cand('b.ts', 40)],
|
||||
budget,
|
||||
2,
|
||||
);
|
||||
expect([...allowances.keys()].sort()).toEqual(['a.ts', 'b.ts']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('allocateExploreBudget — the floor keeps diffuse questions useful', () => {
|
||||
const budget = getExploreOutputBudget(1000);
|
||||
|
||||
it('gives every admitted file a slice big enough for a method', () => {
|
||||
// A survey question must still return a spread. The earlier design cliffed a
|
||||
// starved file instead of flooring it, and that CASCADED: removing the
|
||||
// smallest raised everyone else so little that the next-smallest starved too,
|
||||
// eating six legitimately-ranked peers one at a time.
|
||||
const files = [cand('a.ts', 100), cand('b.ts', 90), ...Array.from({ length: 6 }, (_, i) => cand(`p${i}.ts`, 20))];
|
||||
const { allowances } = allocateExploreBudget(files, budget, 8);
|
||||
expect(allowances.size).toBe(8);
|
||||
for (const [, chars] of allowances) expect(chars).toBeGreaterThanOrEqual(700);
|
||||
});
|
||||
|
||||
it('serves fewer files well rather than many badly when the envelope cannot afford them', () => {
|
||||
const tiny = getExploreOutputBudget(10); // 13,000-char envelope
|
||||
const files = Array.from({ length: 40 }, (_, i) => cand(`f${i}.ts`, 50 - i * 0.1));
|
||||
const { allowances, cliffed } = allocateExploreBudget(files, tiny, 40);
|
||||
expect(allowances.size).toBeLessThan(40);
|
||||
expect(cliffed.length).toBeGreaterThan(0);
|
||||
for (const [, chars] of allowances) expect(chars).toBeGreaterThanOrEqual(700);
|
||||
const reserved = [...allowances.values()].reduce((s, n) => s + n, 0);
|
||||
expect(reserved).toBeLessThanOrEqual(tiny.maxOutputChars);
|
||||
});
|
||||
|
||||
it('returns an empty allocation for an empty candidate list', () => {
|
||||
const { allowances, cliffed } = allocateExploreBudget([], budget, 8);
|
||||
expect(allowances.size).toBe(0);
|
||||
expect(cliffed).toEqual([]);
|
||||
});
|
||||
|
||||
it('does not crash or over-allocate when every score is zero', () => {
|
||||
const { allowances } = allocateExploreBudget([cand('a.ts', 0), cand('b.ts', 0)], budget, 8);
|
||||
const reserved = [...allowances.values()].reduce((s, n) => s + n, 0);
|
||||
expect(reserved).toBeLessThanOrEqual(budget.maxOutputChars);
|
||||
});
|
||||
});
|
||||
|
||||
describe('allocateExploreBudget — tier invariant', () => {
|
||||
it('never gives a larger tier a smaller allowance than a smaller tier', () => {
|
||||
// The standing invariant from `getExploreOutputBudget`: a bigger project must
|
||||
// never be served LESS per file. It held for the flat cap by inspection; with
|
||||
// a proportional split it has to hold for the same candidate set across every
|
||||
// tier, which is what this walks.
|
||||
const files = [cand('a.ts', 60), cand('b.ts', 30), cand('c.ts', 15)];
|
||||
let previous: Map<string, number> | null = null;
|
||||
for (const fileCount of TIER_FILE_COUNTS) {
|
||||
const { allowances } = allocateExploreBudget(files, getExploreOutputBudget(fileCount), 8);
|
||||
if (previous) {
|
||||
for (const [path, chars] of allowances) {
|
||||
expect(chars, `${path} shrank at ${fileCount} files`).toBeGreaterThanOrEqual(previous.get(path)!);
|
||||
}
|
||||
}
|
||||
previous = allowances;
|
||||
}
|
||||
});
|
||||
|
||||
it('cliffs the same files at every tier — the cliff is relative, not sized', () => {
|
||||
const files = [cand('a.ts', 90), cand('noise.ts', 2)];
|
||||
const cliffs = TIER_FILE_COUNTS.map((n) =>
|
||||
allocateExploreBudget(files, getExploreOutputBudget(n), 8).cliffed.join(','));
|
||||
expect(new Set(cliffs).size).toBe(1);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user