mirror of
https://github.com/garrytan/gbrain.git
synced 2026-07-31 04:07:52 +00:00
feat(v0.34 pre-w0): add code-retrieval eval harness for v0.34 ship gate
Captures pre-v0.34 retrieval quality on the gbrain self-corpus before any code-intel work lands, so the v0.34 ship gate (precision@5 +10pp OR answered_rate +15pp on >=15/30 questions) measures real improvement rather than an after-the-fact retuned baseline. * src/eval/code-retrieval/harness.ts -- pure-function metrics (precision@k, recall@k, top-1 stability, gate evaluator) + EvalRunReport types stable across schema_version 1 * src/eval/code-retrieval/questions.json -- 30 questions across callers / callees / definition / references / blast_radius / execution_flow / cluster_membership kinds, expected_files captured against current gbrain layout * src/eval/code-retrieval/strategies.ts -- BaselineStrategy (hybridSearch) + WithCodeIntelStrategy stub (post-W3 fills in code_blast/code_flow/etc.) * src/commands/eval-code-retrieval.ts -- gbrain eval code-retrieval CLI with --baseline / --with-code-intel / --compare subcommands * test/code-retrieval-harness.test.ts -- 26 unit tests across metrics, loader, gate logic; no engine dependency PRE-V0.34 BASELINE WORKFLOW: gbrain eval code-retrieval --baseline --save /tmp/baseline-1.json (run 3x for noise floor) V0.34 SHIP GATE (after W3 lands): gbrain eval code-retrieval --with-code-intel --save /tmp/v034.json gbrain eval code-retrieval --compare /tmp/baseline-1.json /tmp/v034.json Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.7
parent
71ed8d0d21
commit
eb75e1b32b
@@ -0,0 +1,260 @@
|
||||
/**
|
||||
* v0.34 pre-w0 — `gbrain eval code-retrieval` CLI.
|
||||
*
|
||||
* Subcommands:
|
||||
* --baseline Capture pre-v0.34 retrieval quality. Saves the number
|
||||
* v0.34 must beat. Run 3x for noise floor.
|
||||
* --with-code-intel Run against v0.34's code-intel MCP ops. Pre-W3 this
|
||||
* returns mostly-empty results (honest baseline).
|
||||
* --compare A B Read two saved reports and compute the gate verdict.
|
||||
*
|
||||
* Output:
|
||||
* stdout — human-readable table by default; `--json` for machine.
|
||||
* `--save <path>` writes the EvalRunReport JSON to disk.
|
||||
*
|
||||
* The CLI deliberately separates capture (read-only) from compare (pure JSON
|
||||
* read) so the eval data can move between machines / CI without re-running
|
||||
* against a brain.
|
||||
*/
|
||||
|
||||
import { writeFileSync, existsSync, readFileSync } from 'fs';
|
||||
import { resolve, dirname, join } from 'path';
|
||||
import { mkdirSync } from 'fs';
|
||||
import type { BrainEngine } from '../core/engine.ts';
|
||||
import {
|
||||
loadQuestions,
|
||||
runCodeRetrievalEval,
|
||||
evaluateGate,
|
||||
DEFAULT_GATE,
|
||||
type EvalRunReport,
|
||||
} from '../eval/code-retrieval/harness.ts';
|
||||
import { BaselineStrategy, WithCodeIntelStrategy } from '../eval/code-retrieval/strategies.ts';
|
||||
import { createProgress } from '../core/progress.ts';
|
||||
import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts';
|
||||
|
||||
interface ParsedArgs {
|
||||
help: boolean;
|
||||
baseline: boolean;
|
||||
withCodeIntel: boolean;
|
||||
compare?: { a: string; b: string };
|
||||
corpus: string;
|
||||
questionsPath: string;
|
||||
source?: string;
|
||||
k: number;
|
||||
save?: string;
|
||||
json: boolean;
|
||||
}
|
||||
|
||||
const DEFAULT_QUESTIONS_PATH = 'src/eval/code-retrieval/questions.json';
|
||||
|
||||
function parseArgs(args: string[]): ParsedArgs {
|
||||
const out: ParsedArgs = {
|
||||
help: false,
|
||||
baseline: false,
|
||||
withCodeIntel: false,
|
||||
corpus: 'gbrain',
|
||||
questionsPath: DEFAULT_QUESTIONS_PATH,
|
||||
k: 5,
|
||||
json: false,
|
||||
};
|
||||
let i = 0;
|
||||
while (i < args.length) {
|
||||
const a = args[i];
|
||||
if (a === '--help' || a === '-h') out.help = true;
|
||||
else if (a === '--baseline') out.baseline = true;
|
||||
else if (a === '--with-code-intel') out.withCodeIntel = true;
|
||||
else if (a === '--compare') {
|
||||
const aPath = args[++i];
|
||||
const bPath = args[++i];
|
||||
if (!aPath || !bPath) throw new Error('--compare requires two file paths');
|
||||
out.compare = { a: aPath, b: bPath };
|
||||
} else if (a === '--corpus') out.corpus = args[++i] ?? 'gbrain';
|
||||
else if (a === '--questions') out.questionsPath = args[++i] ?? DEFAULT_QUESTIONS_PATH;
|
||||
else if (a === '--source') out.source = args[++i];
|
||||
else if (a === '--k') out.k = parseInt(args[++i] ?? '5', 10);
|
||||
else if (a === '--save') out.save = args[++i];
|
||||
else if (a === '--json') out.json = true;
|
||||
i++;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function printHelp(): void {
|
||||
process.stdout.write(`
|
||||
gbrain eval code-retrieval — v0.34 retrieval baseline + gate
|
||||
|
||||
Capture a retrieval-quality number against a curated question set. The
|
||||
baseline captured here is what v0.34 must beat (precision@5 +10pp OR
|
||||
answered_rate +15pp on >=15/30 questions).
|
||||
|
||||
USAGE:
|
||||
gbrain eval code-retrieval --baseline [--corpus NAME] [--save PATH] [--json]
|
||||
gbrain eval code-retrieval --with-code-intel [--corpus NAME] [--save PATH] [--json]
|
||||
gbrain eval code-retrieval --compare BASELINE.json WITH_CODE_INTEL.json [--json]
|
||||
|
||||
OPTIONS:
|
||||
--baseline Capture pre-v0.34 retrieval (query + search only)
|
||||
--with-code-intel Capture v0.34 mode (code_blast / code_flow / etc.)
|
||||
--compare A B Compare two saved reports; emits gate pass/fail
|
||||
--corpus NAME Brain corpus to query (default: gbrain)
|
||||
--questions PATH Question file (default: ${DEFAULT_QUESTIONS_PATH})
|
||||
--source SOURCE_ID Source to scope queries to (default: brain default)
|
||||
--k N Top-k cutoff (default: 5)
|
||||
--save PATH Write EvalRunReport JSON to disk
|
||||
--json Machine-readable JSON to stdout
|
||||
-h, --help This help
|
||||
|
||||
PRE-V0.34 BASELINE WORKFLOW (the assignment):
|
||||
# Capture baseline 3x for noise floor
|
||||
gbrain eval code-retrieval --baseline --save /tmp/baseline-1.json
|
||||
gbrain eval code-retrieval --baseline --save /tmp/baseline-2.json
|
||||
gbrain eval code-retrieval --baseline --save /tmp/baseline-3.json
|
||||
|
||||
V0.34 SHIP GATE:
|
||||
# After v0.34 ships, capture the with-code-intel run and compare
|
||||
gbrain eval code-retrieval --with-code-intel --save /tmp/v034.json
|
||||
gbrain eval code-retrieval --compare /tmp/baseline-1.json /tmp/v034.json
|
||||
`);
|
||||
}
|
||||
|
||||
export async function runEvalCodeRetrieval(engine: BrainEngine, args: string[]): Promise<void> {
|
||||
let opts: ParsedArgs;
|
||||
try {
|
||||
opts = parseArgs(args);
|
||||
} catch (err: any) {
|
||||
process.stderr.write(`error: ${err.message}\n`);
|
||||
process.exit(2);
|
||||
return;
|
||||
}
|
||||
|
||||
if (opts.help) {
|
||||
printHelp();
|
||||
return;
|
||||
}
|
||||
|
||||
// --compare mode: pure JSON read, no engine needed (engine still passed by
|
||||
// the dispatcher; we just don't touch it).
|
||||
if (opts.compare) {
|
||||
return runCompare(opts);
|
||||
}
|
||||
|
||||
if (!opts.baseline && !opts.withCodeIntel) {
|
||||
process.stderr.write('error: specify --baseline or --with-code-intel (or --compare)\n');
|
||||
printHelp();
|
||||
process.exit(2);
|
||||
return;
|
||||
}
|
||||
|
||||
const questions = loadQuestions(resolve(opts.questionsPath));
|
||||
const sourceId = opts.source ?? (await resolveDefaultSource(engine));
|
||||
|
||||
const strategy = opts.baseline
|
||||
? new BaselineStrategy(engine, sourceId)
|
||||
: new WithCodeIntelStrategy(engine, sourceId);
|
||||
|
||||
const progress = createProgress(cliOptsToProgressOptions(getCliOptions()));
|
||||
progress.start(`eval.code_retrieval.${strategy.mode}`, questions.questions.length);
|
||||
|
||||
const report = await runCodeRetrievalEval(questions.questions, strategy, {
|
||||
k: opts.k,
|
||||
corpus: opts.corpus,
|
||||
onProgress: (done, _total, qid) => progress.tick(1, qid),
|
||||
});
|
||||
|
||||
progress.finish();
|
||||
|
||||
if (opts.save) {
|
||||
const savePath = resolve(opts.save);
|
||||
mkdirSync(dirname(savePath), { recursive: true });
|
||||
writeFileSync(savePath, JSON.stringify(report, null, 2));
|
||||
process.stderr.write(`[eval] saved report to ${savePath}\n`);
|
||||
}
|
||||
|
||||
if (opts.json) {
|
||||
process.stdout.write(JSON.stringify(report, null, 2) + '\n');
|
||||
return;
|
||||
}
|
||||
|
||||
printSingleReport(report);
|
||||
}
|
||||
|
||||
function runCompare(opts: ParsedArgs): void {
|
||||
if (!opts.compare) return;
|
||||
const aPath = resolve(opts.compare.a);
|
||||
const bPath = resolve(opts.compare.b);
|
||||
for (const p of [aPath, bPath]) {
|
||||
if (!existsSync(p)) {
|
||||
process.stderr.write(`error: report not found at ${p}\n`);
|
||||
process.exit(2);
|
||||
return;
|
||||
}
|
||||
}
|
||||
const a: EvalRunReport = JSON.parse(readFileSync(aPath, 'utf8'));
|
||||
const b: EvalRunReport = JSON.parse(readFileSync(bPath, 'utf8'));
|
||||
|
||||
// Convention: the first arg is baseline, the second is with-code-intel.
|
||||
// If labels disagree, swap so the comparison is meaningful.
|
||||
let baseline = a;
|
||||
let withCodeIntel = b;
|
||||
if (a.mode === 'with-code-intel' && b.mode === 'baseline') {
|
||||
baseline = b;
|
||||
withCodeIntel = a;
|
||||
}
|
||||
|
||||
const gate = evaluateGate(baseline, withCodeIntel, DEFAULT_GATE);
|
||||
|
||||
if (opts.json) {
|
||||
process.stdout.write(
|
||||
JSON.stringify(
|
||||
{
|
||||
schema_version: 1,
|
||||
passed: gate.passed,
|
||||
precision_delta_pp: gate.precision_delta_pp,
|
||||
top_1_stability_rate: gate.top_1_stability_rate,
|
||||
questions_cleared_bar: gate.questions_cleared_bar,
|
||||
questions_total: gate.questions_total,
|
||||
summary: gate.summary,
|
||||
},
|
||||
null,
|
||||
2,
|
||||
) + '\n',
|
||||
);
|
||||
process.exit(gate.passed ? 0 : 1);
|
||||
return;
|
||||
}
|
||||
|
||||
process.stdout.write(`\n${gate.summary}\n\n`);
|
||||
process.stdout.write(`baseline: precision@${baseline.k}=${(baseline.mean_precision_at_k * 100).toFixed(1)}% answered=${(baseline.answered_rate * 100).toFixed(1)}% (commit ${baseline.commit})\n`);
|
||||
process.stdout.write(`with-code-intel: precision@${withCodeIntel.k}=${(withCodeIntel.mean_precision_at_k * 100).toFixed(1)}% answered=${(withCodeIntel.answered_rate * 100).toFixed(1)}% (commit ${withCodeIntel.commit})\n`);
|
||||
process.stdout.write(`delta: +${gate.precision_delta_pp.toFixed(1)}pp precision top-1 stability=${(gate.top_1_stability_rate * 100).toFixed(1)}%\n`);
|
||||
process.stdout.write(`cleared bar: ${gate.questions_cleared_bar}/${gate.questions_total}\n\n`);
|
||||
|
||||
process.exit(gate.passed ? 0 : 1);
|
||||
}
|
||||
|
||||
function printSingleReport(report: EvalRunReport): void {
|
||||
process.stdout.write(`\n=== code-retrieval eval (mode=${report.mode}) ===\n`);
|
||||
process.stdout.write(`corpus: ${report.corpus}\n`);
|
||||
process.stdout.write(`commit: ${report.commit}\n`);
|
||||
process.stdout.write(`captured: ${report.captured_at}\n`);
|
||||
process.stdout.write(`questions: ${report.questions.length}\n`);
|
||||
process.stdout.write(`precision@${report.k}: ${(report.mean_precision_at_k * 100).toFixed(1)}%\n`);
|
||||
process.stdout.write(`answered: ${report.questions.filter((q) => q.answered).length}/${report.questions.length} (${(report.answered_rate * 100).toFixed(1)}%)\n`);
|
||||
process.stdout.write(`latency: ${report.total_latency_ms}ms total, ${(report.total_latency_ms / Math.max(1, report.questions.length)).toFixed(0)}ms/q\n`);
|
||||
process.stdout.write(`\nper-question:\n`);
|
||||
for (const q of report.questions) {
|
||||
const status = q.answered ? '✓' : '✗';
|
||||
process.stdout.write(` ${status} ${q.id.padEnd(20)} p@${report.k}=${(q.precision_at_k * 100).toFixed(0)}% recall@${report.k}=${(q.recall_at_k * 100).toFixed(0)}% (${q.latency_ms}ms)\n`);
|
||||
}
|
||||
process.stdout.write('\n');
|
||||
}
|
||||
|
||||
async function resolveDefaultSource(engine: BrainEngine): Promise<string> {
|
||||
// Try the engine's listSources if it exists; otherwise return a sensible default.
|
||||
// Most brains have one source; this picks that one.
|
||||
const sources = await (engine as any).listSources?.();
|
||||
if (Array.isArray(sources) && sources.length > 0) {
|
||||
return sources[0].id ?? 'default';
|
||||
}
|
||||
return 'default';
|
||||
}
|
||||
@@ -45,6 +45,14 @@ export async function runEvalCommand(engine: BrainEngine, args: string[]): Promi
|
||||
const { runEvalCrossModal } = await import('./eval-cross-modal.ts');
|
||||
process.exit(await runEvalCrossModal(args.slice(1)));
|
||||
}
|
||||
if (sub === 'code-retrieval') {
|
||||
// v0.34 pre-w0 — code-retrieval baseline / gate harness. Needs a brain
|
||||
// for the baseline (BaselineStrategy calls hybridSearch); --compare
|
||||
// mode reads JSON only but the engine is already connected by this
|
||||
// dispatcher.
|
||||
const { runEvalCodeRetrieval } = await import('./eval-code-retrieval.ts');
|
||||
return runEvalCodeRetrieval(engine, args.slice(1));
|
||||
}
|
||||
|
||||
const opts = parseArgs(args);
|
||||
|
||||
|
||||
@@ -0,0 +1,390 @@
|
||||
/**
|
||||
* v0.34 pre-w0 — code-retrieval eval harness.
|
||||
*
|
||||
* The baseline this captures is what v0.34 must beat (precision@5 +10pp OR
|
||||
* top-1 stability +15pp on >=15/30 questions, above the 3-run noise floor).
|
||||
*
|
||||
* Two modes:
|
||||
* --baseline Uses only `query` + `search` (today's gbrain). The number
|
||||
* captured here is the v0.34 ship gate.
|
||||
* --with-code-intel Uses the v0.34 code-intel MCP ops (code_blast / code_flow
|
||||
* / code_def / code_refs / code_callers / code_callees /
|
||||
* code_cluster_get). Numerator of the gate comparison.
|
||||
*
|
||||
* Pure-function metrics live at the bottom of this file; the runner can be
|
||||
* stubbed for unit testing without touching the brain.
|
||||
*
|
||||
* Hermetic by design: cli.ts skips connectEngine() when the user is in
|
||||
* --help mode. Real runs connect to ~/.gbrain (the dogfood brain) by default
|
||||
* unless --corpus points elsewhere.
|
||||
*/
|
||||
|
||||
import { readFileSync, existsSync } from 'fs';
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
// Types — stable contract for downstream consumers (gbrain-evals, CI)
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
|
||||
export type CodeQuestionKind =
|
||||
| 'callers'
|
||||
| 'callees'
|
||||
| 'definition'
|
||||
| 'references'
|
||||
| 'blast_radius'
|
||||
| 'execution_flow'
|
||||
| 'cluster_membership';
|
||||
|
||||
export interface CodeQuestion {
|
||||
id: string;
|
||||
kind: CodeQuestionKind;
|
||||
/** Human-language query an agent would type. */
|
||||
query: string;
|
||||
/** Canonical symbol to look up structurally (post-v0.34). */
|
||||
symbol: string;
|
||||
/** Expected file paths that must appear in the retrieved set. */
|
||||
expected_files: string[];
|
||||
/**
|
||||
* Minimum recall@k against `expected_files` for this question to count as
|
||||
* "answered." For some post-v0.34-only questions, baseline expectations
|
||||
* are intentionally low (0.3) since today's gbrain can't answer them.
|
||||
*/
|
||||
expected_min_recall: number;
|
||||
note?: string;
|
||||
}
|
||||
|
||||
export interface CodeQuestionFile {
|
||||
version: number;
|
||||
schema: string;
|
||||
corpus: string;
|
||||
description: string;
|
||||
questions: CodeQuestion[];
|
||||
}
|
||||
|
||||
export interface QuestionResult {
|
||||
id: string;
|
||||
kind: CodeQuestionKind;
|
||||
/** Files actually returned, in rank order (top-k). */
|
||||
retrieved_files: string[];
|
||||
/** Top-1 file (the single most-confident answer). */
|
||||
top_1: string | null;
|
||||
/** precision@k = |relevant ∩ retrieved| / |retrieved|. */
|
||||
precision_at_k: number;
|
||||
/** recall@k = |relevant ∩ retrieved| / |relevant|. */
|
||||
recall_at_k: number;
|
||||
/** Whether this question's bar (`expected_min_recall`) was cleared. */
|
||||
answered: boolean;
|
||||
/** Total latency for this question's tool calls, in ms. */
|
||||
latency_ms: number;
|
||||
}
|
||||
|
||||
export interface EvalRunReport {
|
||||
mode: 'baseline' | 'with-code-intel';
|
||||
schema_version: 1;
|
||||
corpus: string;
|
||||
k: number;
|
||||
questions: QuestionResult[];
|
||||
/** Mean precision@k across all questions. */
|
||||
mean_precision_at_k: number;
|
||||
/** Fraction of questions that cleared their expected_min_recall bar. */
|
||||
answered_rate: number;
|
||||
/** Top-1 stability — set when comparing two runs; null for single-run. */
|
||||
top_1_stability_rate?: number;
|
||||
/** Aggregate run latency, ms. */
|
||||
total_latency_ms: number;
|
||||
/** ISO-8601 capture time. */
|
||||
captured_at: string;
|
||||
/** Git short-SHA at capture time. */
|
||||
commit: string;
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
// Pure metrics (no engine dependency, fully unit-testable)
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* precision@k = relevant ∩ retrieved (top-k) / retrieved (top-k)
|
||||
* Returns 0 when retrieved is empty.
|
||||
*/
|
||||
export function precisionAtK(retrieved: string[], relevant: Set<string>, k: number): number {
|
||||
const topK = retrieved.slice(0, k);
|
||||
if (topK.length === 0) return 0;
|
||||
let hits = 0;
|
||||
for (const r of topK) {
|
||||
if (relevant.has(r)) hits++;
|
||||
}
|
||||
return hits / topK.length;
|
||||
}
|
||||
|
||||
/**
|
||||
* recall@k = relevant ∩ retrieved (top-k) / relevant
|
||||
* Returns 1 when relevant is empty (degenerate case).
|
||||
*/
|
||||
export function recallAtK(retrieved: string[], relevant: Set<string>, k: number): number {
|
||||
if (relevant.size === 0) return 1;
|
||||
const topK = new Set(retrieved.slice(0, k));
|
||||
let hits = 0;
|
||||
for (const r of relevant) {
|
||||
if (topK.has(r)) hits++;
|
||||
}
|
||||
return hits / relevant.size;
|
||||
}
|
||||
|
||||
/**
|
||||
* top-1 stability rate between two runs over the same question set.
|
||||
* Computed as |{q : run1.top_1 === run2.top_1}| / total. NaN → 0.
|
||||
*/
|
||||
export function top1StabilityRate(run1: QuestionResult[], run2: QuestionResult[]): number {
|
||||
if (run1.length === 0 || run2.length === 0) return 0;
|
||||
const lookup = new Map(run2.map((q) => [q.id, q.top_1]));
|
||||
let stable = 0;
|
||||
let comparable = 0;
|
||||
for (const q of run1) {
|
||||
if (!lookup.has(q.id)) continue;
|
||||
comparable++;
|
||||
if (q.top_1 === lookup.get(q.id)) stable++;
|
||||
}
|
||||
return comparable === 0 ? 0 : stable / comparable;
|
||||
}
|
||||
|
||||
/**
|
||||
* Match `retrieved_files` against `expected_files` accounting for prefix
|
||||
* matching: a retrieved file "src/foo/bar.ts" counts as matching the
|
||||
* expected entry "src/foo/" (trailing slash = directory match).
|
||||
* Set semantics — duplicates collapse.
|
||||
*/
|
||||
export function normalizeRetrieved(retrieved_files: string[]): string[] {
|
||||
// Dedupe + drop empties; do not mutate order.
|
||||
const seen = new Set<string>();
|
||||
const out: string[] = [];
|
||||
for (const f of retrieved_files) {
|
||||
if (!f) continue;
|
||||
if (seen.has(f)) continue;
|
||||
seen.add(f);
|
||||
out.push(f);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export function expandExpectedToRelevantSet(expected: string[]): {
|
||||
exactFiles: Set<string>;
|
||||
dirPrefixes: string[];
|
||||
} {
|
||||
const exactFiles = new Set<string>();
|
||||
const dirPrefixes: string[] = [];
|
||||
for (const e of expected) {
|
||||
if (e.endsWith('/')) dirPrefixes.push(e);
|
||||
else exactFiles.add(e);
|
||||
}
|
||||
return { exactFiles, dirPrefixes };
|
||||
}
|
||||
|
||||
export function isFileRelevant(file: string, expected: ReturnType<typeof expandExpectedToRelevantSet>): boolean {
|
||||
if (expected.exactFiles.has(file)) return true;
|
||||
for (const p of expected.dirPrefixes) {
|
||||
if (file.startsWith(p)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
// Loader
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
|
||||
export function loadQuestions(path: string): CodeQuestionFile {
|
||||
if (!existsSync(path)) {
|
||||
throw new Error(`questions file not found: ${path}`);
|
||||
}
|
||||
const raw = readFileSync(path, 'utf8');
|
||||
let parsed: CodeQuestionFile;
|
||||
try {
|
||||
parsed = JSON.parse(raw);
|
||||
} catch (err: any) {
|
||||
throw new Error(`failed to parse questions JSON: ${err.message}`);
|
||||
}
|
||||
if (parsed.version !== 1) {
|
||||
throw new Error(`unsupported questions file version ${parsed.version} (expected 1)`);
|
||||
}
|
||||
if (!Array.isArray(parsed.questions) || parsed.questions.length === 0) {
|
||||
throw new Error('questions file contains no questions');
|
||||
}
|
||||
for (const q of parsed.questions) {
|
||||
if (!q.id || !q.kind || !q.query || !Array.isArray(q.expected_files)) {
|
||||
throw new Error(`malformed question entry: ${JSON.stringify(q)}`);
|
||||
}
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
// Mode dispatch — pluggable retrieval strategies
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* A retrieval strategy maps a question to a ranked list of file paths.
|
||||
* Pure abstraction so the runner is testable without engine dependencies.
|
||||
*/
|
||||
export interface RetrievalStrategy {
|
||||
readonly mode: 'baseline' | 'with-code-intel';
|
||||
retrieve(question: CodeQuestion, k: number): Promise<{ files: string[]; latency_ms: number }>;
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
// Runner
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
|
||||
export interface RunnerOpts {
|
||||
k: number;
|
||||
corpus: string;
|
||||
onProgress?: (done: number, total: number, currentQ: string) => void;
|
||||
}
|
||||
|
||||
export async function runCodeRetrievalEval(
|
||||
questions: CodeQuestion[],
|
||||
strategy: RetrievalStrategy,
|
||||
opts: RunnerOpts,
|
||||
): Promise<EvalRunReport> {
|
||||
const results: QuestionResult[] = [];
|
||||
const startedAt = Date.now();
|
||||
|
||||
for (let i = 0; i < questions.length; i++) {
|
||||
const q = questions[i];
|
||||
const t0 = Date.now();
|
||||
let retrieved: string[] = [];
|
||||
let latency_ms = 0;
|
||||
try {
|
||||
const r = await strategy.retrieve(q, opts.k);
|
||||
retrieved = normalizeRetrieved(r.files);
|
||||
latency_ms = r.latency_ms;
|
||||
} catch (err: any) {
|
||||
// Retrieval errors are part of the eval signal — record as empty result.
|
||||
retrieved = [];
|
||||
latency_ms = Date.now() - t0;
|
||||
process.stderr.write(`[eval] retrieval error on ${q.id}: ${err?.message ?? err}\n`);
|
||||
}
|
||||
|
||||
const expected = expandExpectedToRelevantSet(q.expected_files);
|
||||
const relevantSet = new Set(retrieved.filter((f) => isFileRelevant(f, expected)));
|
||||
const recall_at_k = recallAtK(Array.from(relevantSet), expected.exactFiles, opts.k);
|
||||
const precision_at_k = precisionAtK(retrieved, relevantSet, opts.k);
|
||||
const top_1 = retrieved[0] ?? null;
|
||||
const answered = recall_at_k >= q.expected_min_recall;
|
||||
|
||||
results.push({
|
||||
id: q.id,
|
||||
kind: q.kind,
|
||||
retrieved_files: retrieved,
|
||||
top_1,
|
||||
precision_at_k,
|
||||
recall_at_k,
|
||||
answered,
|
||||
latency_ms,
|
||||
});
|
||||
|
||||
opts.onProgress?.(i + 1, questions.length, q.id);
|
||||
}
|
||||
|
||||
const total_latency_ms = Date.now() - startedAt;
|
||||
const mean_precision_at_k = results.reduce((a, r) => a + r.precision_at_k, 0) / Math.max(1, results.length);
|
||||
const answered_rate = results.filter((r) => r.answered).length / Math.max(1, results.length);
|
||||
|
||||
let commit = 'unknown';
|
||||
try {
|
||||
const { execSync } = await import('child_process');
|
||||
commit = execSync('git rev-parse --short HEAD', { encoding: 'utf8' }).trim();
|
||||
} catch {
|
||||
// Not in a git repo or git unavailable — leave as 'unknown'
|
||||
}
|
||||
|
||||
return {
|
||||
mode: strategy.mode,
|
||||
schema_version: 1,
|
||||
corpus: opts.corpus,
|
||||
k: opts.k,
|
||||
questions: results,
|
||||
mean_precision_at_k,
|
||||
answered_rate,
|
||||
total_latency_ms,
|
||||
captured_at: new Date().toISOString(),
|
||||
commit,
|
||||
};
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
// Comparison — used by the v0.34 ship gate
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
|
||||
export interface GateResult {
|
||||
passed: boolean;
|
||||
precision_delta_pp: number;
|
||||
top_1_stability_rate: number;
|
||||
questions_cleared_bar: number;
|
||||
questions_total: number;
|
||||
baseline: EvalRunReport;
|
||||
with_code_intel: EvalRunReport;
|
||||
/** Free-form summary explaining pass/fail. */
|
||||
summary: string;
|
||||
}
|
||||
|
||||
export interface GateOpts {
|
||||
/** Required precision@5 delta (in percentage points) to pass. Default: 10. */
|
||||
required_precision_delta_pp: number;
|
||||
/** Required top-1 stability rate to pass (alternative criterion). Default: 0.15 lower bound. */
|
||||
required_top_1_stability_delta: number;
|
||||
/** Minimum questions that must clear `expected_min_recall` in with-code-intel mode. Default: 15. */
|
||||
min_questions_cleared: number;
|
||||
}
|
||||
|
||||
export const DEFAULT_GATE: GateOpts = {
|
||||
required_precision_delta_pp: 10,
|
||||
required_top_1_stability_delta: 0.15,
|
||||
min_questions_cleared: 15,
|
||||
};
|
||||
|
||||
export function evaluateGate(
|
||||
baseline: EvalRunReport,
|
||||
with_code_intel: EvalRunReport,
|
||||
opts: GateOpts = DEFAULT_GATE,
|
||||
): GateResult {
|
||||
const precision_delta_pp = (with_code_intel.mean_precision_at_k - baseline.mean_precision_at_k) * 100;
|
||||
const top_1_stability_rate = top1StabilityRate(baseline.questions, with_code_intel.questions);
|
||||
const questions_cleared_bar = with_code_intel.questions.filter((q) => q.answered).length;
|
||||
const questions_total = with_code_intel.questions.length;
|
||||
|
||||
const precisionPasses = precision_delta_pp >= opts.required_precision_delta_pp;
|
||||
// Stability is "how much DIFFERENT" — for an improvement gate we want
|
||||
// either precision OR a substantive reordering that lands more good answers.
|
||||
// We use a delta over baseline's own stability, treating baseline as 1.0
|
||||
// self-stability. Convention: pass when stability is lower (more changes)
|
||||
// AND a higher answered_rate. Otherwise the gate would punish v0.34 for
|
||||
// changing the retrieval order even when it lands better answers.
|
||||
const stabilityPasses =
|
||||
with_code_intel.answered_rate - baseline.answered_rate >= opts.required_top_1_stability_delta;
|
||||
const enoughCleared = questions_cleared_bar >= opts.min_questions_cleared;
|
||||
|
||||
const passed = enoughCleared && (precisionPasses || stabilityPasses);
|
||||
|
||||
const reasons: string[] = [];
|
||||
if (!enoughCleared) {
|
||||
reasons.push(
|
||||
`only ${questions_cleared_bar}/${questions_total} questions cleared expected_min_recall (need >=${opts.min_questions_cleared})`,
|
||||
);
|
||||
}
|
||||
if (precisionPasses) reasons.push(`precision@${baseline.k} +${precision_delta_pp.toFixed(1)}pp (>=${opts.required_precision_delta_pp})`);
|
||||
else reasons.push(`precision@${baseline.k} delta ${precision_delta_pp.toFixed(1)}pp (<${opts.required_precision_delta_pp})`);
|
||||
if (stabilityPasses) reasons.push(`answered_rate +${((with_code_intel.answered_rate - baseline.answered_rate) * 100).toFixed(1)}pp`);
|
||||
|
||||
const summary = passed
|
||||
? `GATE PASS — ${reasons.join('; ')}`
|
||||
: `GATE FAIL — ${reasons.join('; ')}`;
|
||||
|
||||
return {
|
||||
passed,
|
||||
precision_delta_pp,
|
||||
top_1_stability_rate,
|
||||
questions_cleared_bar,
|
||||
questions_total,
|
||||
baseline,
|
||||
with_code_intel,
|
||||
summary,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,252 @@
|
||||
{
|
||||
"version": 1,
|
||||
"schema": "1",
|
||||
"corpus": "gbrain",
|
||||
"description": "v0.34 code-retrieval baseline. 30 questions about gbrain's own codebase covering callers, callees, definitions, blast radius, and cluster membership. The 'expected' field is a SET of file:line tuples (not page slugs) for code-symbol retrieval. Match semantics: a result counts as relevant when its (file, line) matches any expected tuple OR its (file, symbol_qualified) matches when symbol_qualified is provided.\n\nTo regenerate / extend: bump version, add questions, document the diff. Pre-v0.34 baseline is captured against this exact set so it cannot be retroactively tuned.",
|
||||
"questions": [
|
||||
{
|
||||
"id": "callers-001",
|
||||
"kind": "callers",
|
||||
"query": "what calls performSync",
|
||||
"symbol": "performSync",
|
||||
"expected_files": ["src/commands/sync.ts", "src/core/cycle.ts", "src/commands/jobs.ts"],
|
||||
"expected_min_recall": 0.6
|
||||
},
|
||||
{
|
||||
"id": "callers-002",
|
||||
"kind": "callers",
|
||||
"query": "what calls hybridSearch",
|
||||
"symbol": "hybridSearch",
|
||||
"expected_files": ["src/core/operations.ts", "src/core/eval-capture.ts", "src/core/search/eval.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "callers-003",
|
||||
"kind": "callers",
|
||||
"query": "what calls runCycle",
|
||||
"symbol": "runCycle",
|
||||
"expected_files": ["src/commands/dream.ts", "src/commands/autopilot.ts", "src/commands/jobs.ts"],
|
||||
"expected_min_recall": 0.6
|
||||
},
|
||||
{
|
||||
"id": "callers-004",
|
||||
"kind": "callers",
|
||||
"query": "what calls importFromContent",
|
||||
"symbol": "importFromContent",
|
||||
"expected_files": ["src/commands/import.ts", "src/commands/sync.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "callers-005",
|
||||
"kind": "callers",
|
||||
"query": "what calls embed",
|
||||
"symbol": "embed",
|
||||
"expected_files": ["src/commands/embed.ts", "src/core/import-file.ts", "src/core/search/eval.ts"],
|
||||
"expected_min_recall": 0.4
|
||||
},
|
||||
{
|
||||
"id": "callers-006",
|
||||
"kind": "callers",
|
||||
"query": "what calls put_page operation",
|
||||
"symbol": "putPage",
|
||||
"expected_files": ["src/core/operations.ts", "src/commands/import.ts", "src/core/cycle"],
|
||||
"expected_min_recall": 0.4
|
||||
},
|
||||
{
|
||||
"id": "callees-001",
|
||||
"kind": "callees",
|
||||
"query": "what does performSync call",
|
||||
"symbol": "performSync",
|
||||
"expected_files": ["src/core/import-file.ts", "src/core/sync.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "callees-002",
|
||||
"kind": "callees",
|
||||
"query": "what does runCycle call",
|
||||
"symbol": "runCycle",
|
||||
"expected_files": ["src/core/cycle.ts", "src/commands/sync.ts", "src/commands/embed.ts", "src/commands/extract.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "callees-003",
|
||||
"kind": "callees",
|
||||
"query": "what does code_blast call from request to db",
|
||||
"symbol": "code_blast",
|
||||
"expected_files": ["src/core/operations.ts", "src/core/postgres-engine.ts"],
|
||||
"expected_min_recall": 0.3,
|
||||
"note": "Post-v0.34 expectation; pre-v0.34 baseline returns nothing meaningful — expected behavior"
|
||||
},
|
||||
{
|
||||
"id": "def-001",
|
||||
"kind": "definition",
|
||||
"query": "where is hybridSearch defined",
|
||||
"symbol": "hybridSearch",
|
||||
"expected_files": ["src/core/search/hybrid.ts"],
|
||||
"expected_min_recall": 1.0
|
||||
},
|
||||
{
|
||||
"id": "def-002",
|
||||
"kind": "definition",
|
||||
"query": "where is runCycle defined",
|
||||
"symbol": "runCycle",
|
||||
"expected_files": ["src/core/cycle.ts"],
|
||||
"expected_min_recall": 1.0
|
||||
},
|
||||
{
|
||||
"id": "def-003",
|
||||
"kind": "definition",
|
||||
"query": "where is MinionQueue class defined",
|
||||
"symbol": "MinionQueue",
|
||||
"expected_files": ["src/core/minions/queue.ts"],
|
||||
"expected_min_recall": 1.0
|
||||
},
|
||||
{
|
||||
"id": "def-004",
|
||||
"kind": "definition",
|
||||
"query": "where is PGLiteEngine connect method",
|
||||
"symbol": "PGLiteEngine.connect",
|
||||
"expected_files": ["src/core/pglite-engine.ts"],
|
||||
"expected_min_recall": 1.0
|
||||
},
|
||||
{
|
||||
"id": "def-005",
|
||||
"kind": "definition",
|
||||
"query": "where is the trust boundary check for remote contexts",
|
||||
"symbol": "OperationContext.remote",
|
||||
"expected_files": ["src/core/operations.ts", "src/mcp/dispatch.ts", "src/core/types.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "refs-001",
|
||||
"kind": "references",
|
||||
"query": "where is parseMarkdown used",
|
||||
"symbol": "parseMarkdown",
|
||||
"expected_files": ["src/core/markdown.ts", "src/core/import-file.ts"],
|
||||
"expected_min_recall": 0.4
|
||||
},
|
||||
{
|
||||
"id": "refs-002",
|
||||
"kind": "references",
|
||||
"query": "where is getStats used",
|
||||
"symbol": "getStats",
|
||||
"expected_files": ["src/core/postgres-engine.ts", "src/core/pglite-engine.ts", "src/commands/serve-http.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "blast-001",
|
||||
"kind": "blast_radius",
|
||||
"query": "what breaks if I change the signature of performSync",
|
||||
"symbol": "performSync",
|
||||
"expected_files": ["src/commands/sync.ts", "src/core/cycle.ts", "src/commands/jobs.ts", "src/commands/autopilot.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "blast-002",
|
||||
"kind": "blast_radius",
|
||||
"query": "what breaks if I change OperationContext.sourceId",
|
||||
"symbol": "OperationContext.sourceId",
|
||||
"expected_files": ["src/core/operations.ts", "src/mcp/dispatch.ts", "src/core/search/two-pass.ts"],
|
||||
"expected_min_recall": 0.4
|
||||
},
|
||||
{
|
||||
"id": "blast-003",
|
||||
"kind": "blast_radius",
|
||||
"query": "what breaks if I rename minion_jobs.max_stalled column",
|
||||
"symbol": "max_stalled",
|
||||
"expected_files": ["src/core/minions/queue.ts", "src/core/minions/types.ts", "src/core/migrate.ts", "src/core/pglite-schema.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "flow-001",
|
||||
"kind": "execution_flow",
|
||||
"query": "how does an MCP tool call flow from request to handler",
|
||||
"symbol": "dispatchToolCall",
|
||||
"expected_files": ["src/mcp/server.ts", "src/mcp/dispatch.ts", "src/core/operations.ts"],
|
||||
"expected_min_recall": 0.6
|
||||
},
|
||||
{
|
||||
"id": "flow-002",
|
||||
"kind": "execution_flow",
|
||||
"query": "how does gbrain sync flow from cli to database write",
|
||||
"symbol": "performSync",
|
||||
"expected_files": ["src/commands/sync.ts", "src/core/sync.ts", "src/core/import-file.ts", "src/core/postgres-engine.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "flow-003",
|
||||
"kind": "execution_flow",
|
||||
"query": "how does a code_blast call flow to the database",
|
||||
"symbol": "code_blast",
|
||||
"expected_files": ["src/core/operations.ts", "src/core/postgres-engine.ts"],
|
||||
"expected_min_recall": 0.3,
|
||||
"note": "Post-v0.34 expectation; pre-v0.34 baseline returns nothing — expected"
|
||||
},
|
||||
{
|
||||
"id": "cluster-001",
|
||||
"kind": "cluster_membership",
|
||||
"query": "which functions cluster with hybridSearch",
|
||||
"symbol": "hybridSearch",
|
||||
"expected_files": ["src/core/search/hybrid.ts", "src/core/search/expansion.ts", "src/core/search/intent.ts", "src/core/search/sql-ranking.ts"],
|
||||
"expected_min_recall": 0.5,
|
||||
"note": "Post-v0.34 expectation — clusters are new in v0.34"
|
||||
},
|
||||
{
|
||||
"id": "cluster-002",
|
||||
"kind": "cluster_membership",
|
||||
"query": "which functions cluster with runCycle",
|
||||
"symbol": "runCycle",
|
||||
"expected_files": ["src/core/cycle.ts", "src/core/cycle/synthesize.ts", "src/core/cycle/patterns.ts", "src/core/cycle/emotional-weight.ts"],
|
||||
"expected_min_recall": 0.5,
|
||||
"note": "Post-v0.34 expectation"
|
||||
},
|
||||
{
|
||||
"id": "cross-cutting-001",
|
||||
"kind": "callers",
|
||||
"query": "what calls applyForwardReferenceBootstrap",
|
||||
"symbol": "applyForwardReferenceBootstrap",
|
||||
"expected_files": ["src/core/pglite-engine.ts", "src/core/postgres-engine.ts"],
|
||||
"expected_min_recall": 0.8
|
||||
},
|
||||
{
|
||||
"id": "cross-cutting-002",
|
||||
"kind": "callers",
|
||||
"query": "what calls sqlQueryForEngine",
|
||||
"symbol": "sqlQueryForEngine",
|
||||
"expected_files": ["src/commands/auth.ts", "src/commands/serve-http.ts", "src/core/oauth-provider.ts", "src/commands/files.ts", "src/mcp/http-transport.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "cross-cutting-003",
|
||||
"kind": "callers",
|
||||
"query": "what calls summarizeMcpParams",
|
||||
"symbol": "summarizeMcpParams",
|
||||
"expected_files": ["src/mcp/dispatch.ts", "src/commands/serve-http.ts"],
|
||||
"expected_min_recall": 0.7
|
||||
},
|
||||
{
|
||||
"id": "edge-001",
|
||||
"kind": "callers",
|
||||
"query": "what calls extractEntityRefs",
|
||||
"symbol": "extractEntityRefs",
|
||||
"expected_files": ["src/core/link-extraction.ts", "src/commands/extract.ts", "src/commands/backlinks.ts"],
|
||||
"expected_min_recall": 0.5
|
||||
},
|
||||
{
|
||||
"id": "edge-002",
|
||||
"kind": "callers",
|
||||
"query": "what calls validateUploadPath",
|
||||
"symbol": "validateUploadPath",
|
||||
"expected_files": ["src/core/operations.ts"],
|
||||
"expected_min_recall": 0.6
|
||||
},
|
||||
{
|
||||
"id": "edge-003",
|
||||
"kind": "definition",
|
||||
"query": "where is the gbrain serve http entrypoint",
|
||||
"symbol": "runServeHttp",
|
||||
"expected_files": ["src/commands/serve-http.ts"],
|
||||
"expected_min_recall": 1.0
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,103 @@
|
||||
/**
|
||||
* v0.34 pre-w0 — retrieval strategies for the code-retrieval eval.
|
||||
*
|
||||
* Two strategies ship in the pre-w0 PR:
|
||||
* - baseline: query + search via hybridSearch (today's gbrain)
|
||||
* - with-code-intel: code_blast / code_flow / code_def / etc.
|
||||
*
|
||||
* The with-code-intel strategy is a stub until v0.34 W3 lands; the harness
|
||||
* runs against it and produces a meaningful "what does v0.34 need to beat"
|
||||
* number purely from the baseline. After W3 lands, the stub fills in and
|
||||
* the gate becomes a real comparison.
|
||||
*
|
||||
* Strategies are split into their own module so the harness module stays
|
||||
* engine-free and unit-testable.
|
||||
*/
|
||||
|
||||
import type { BrainEngine } from '../../core/engine.ts';
|
||||
import { hybridSearch } from '../../core/search/hybrid.ts';
|
||||
import type { CodeQuestion, RetrievalStrategy } from './harness.ts';
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
// Baseline: hybridSearch over the same brain
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
|
||||
export class BaselineStrategy implements RetrievalStrategy {
|
||||
readonly mode = 'baseline' as const;
|
||||
|
||||
constructor(private readonly engine: BrainEngine, private readonly sourceId: string) {}
|
||||
|
||||
async retrieve(q: CodeQuestion, k: number): Promise<{ files: string[]; latency_ms: number }> {
|
||||
const t0 = Date.now();
|
||||
// Query string is the natural-language question. hybridSearch returns
|
||||
// chunks; we collapse to unique file paths in rank order.
|
||||
const results = await hybridSearch(this.engine, q.query, {
|
||||
limit: Math.max(k * 3, 10), // overshoot to get distinct files
|
||||
strategy: 'hybrid',
|
||||
expand: false, // deterministic; no multi-query expansion
|
||||
} as any);
|
||||
const latency_ms = Date.now() - t0;
|
||||
const filesSeen = new Set<string>();
|
||||
const files: string[] = [];
|
||||
for (const r of results) {
|
||||
const path = pickFilePath(r);
|
||||
if (!path) continue;
|
||||
if (filesSeen.has(path)) continue;
|
||||
filesSeen.add(path);
|
||||
files.push(path);
|
||||
if (files.length >= k) break;
|
||||
}
|
||||
return { files, latency_ms };
|
||||
}
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
// With-code-intel: code_blast / code_flow / etc.
|
||||
//
|
||||
// Until v0.34 W3 lands, this strategy falls through to a noop that
|
||||
// returns zero results. The eval harness handles empty results as
|
||||
// "question not answered" — which is the truth pre-W3.
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
|
||||
export class WithCodeIntelStrategy implements RetrievalStrategy {
|
||||
readonly mode = 'with-code-intel' as const;
|
||||
|
||||
constructor(private readonly engine: BrainEngine, private readonly sourceId: string) {}
|
||||
|
||||
async retrieve(q: CodeQuestion, k: number): Promise<{ files: string[]; latency_ms: number }> {
|
||||
const t0 = Date.now();
|
||||
// STUB: post-v0.34 W3 ships actual op handlers; until then, this strategy
|
||||
// returns nothing. The harness records this as "answered: false" which
|
||||
// is the honest pre-W3 baseline for this mode.
|
||||
//
|
||||
// When W3 lands, this method dispatches by question kind:
|
||||
// q.kind === 'callers' / 'blast_radius' → call code_blast op
|
||||
// q.kind === 'callees' / 'execution_flow' → call code_flow op
|
||||
// q.kind === 'definition' → call code_def op (MCP-exposed in W3)
|
||||
// q.kind === 'references' → call code_refs op
|
||||
// q.kind === 'cluster_membership' → call code_cluster_get op
|
||||
//
|
||||
// For now: noop.
|
||||
const latency_ms = Date.now() - t0;
|
||||
return { files: [], latency_ms };
|
||||
}
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
// File-path extraction from a search result
|
||||
//
|
||||
// Brain pages of kind 'code' carry the file path as the slug suffix
|
||||
// (e.g. slug `code/src/core/markdown.ts`). For non-code pages, we
|
||||
// skip — the eval is about code retrieval.
|
||||
// ─────────────────────────────────────────────────────────────────
|
||||
|
||||
function pickFilePath(result: any): string | null {
|
||||
if (!result?.slug) return null;
|
||||
const slug: string = result.slug;
|
||||
if (slug.startsWith('code/')) {
|
||||
return slug.slice('code/'.length);
|
||||
}
|
||||
// Some code pages may sit under different prefixes depending on source
|
||||
// config; for now, only handle the canonical 'code/' prefix.
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,245 @@
|
||||
/**
|
||||
* v0.34 pre-w0 — unit tests for the code-retrieval eval harness.
|
||||
*
|
||||
* Pure-function metrics + loader + gate logic. No engine, no API, no fixture
|
||||
* files outside the questions.json checked in alongside the harness.
|
||||
*/
|
||||
|
||||
import { describe, test, expect } from 'bun:test';
|
||||
import {
|
||||
precisionAtK,
|
||||
recallAtK,
|
||||
top1StabilityRate,
|
||||
normalizeRetrieved,
|
||||
expandExpectedToRelevantSet,
|
||||
isFileRelevant,
|
||||
loadQuestions,
|
||||
evaluateGate,
|
||||
DEFAULT_GATE,
|
||||
type EvalRunReport,
|
||||
type QuestionResult,
|
||||
} from '../src/eval/code-retrieval/harness.ts';
|
||||
|
||||
describe('precisionAtK', () => {
|
||||
test('returns 0 for empty retrieved set', () => {
|
||||
expect(precisionAtK([], new Set(['a']), 5)).toBe(0);
|
||||
});
|
||||
|
||||
test('returns 1.0 when all top-k are relevant', () => {
|
||||
expect(precisionAtK(['a', 'b', 'c'], new Set(['a', 'b', 'c']), 5)).toBe(1);
|
||||
});
|
||||
|
||||
test('returns 0 when zero top-k are relevant', () => {
|
||||
expect(precisionAtK(['x', 'y'], new Set(['a', 'b']), 5)).toBe(0);
|
||||
});
|
||||
|
||||
test('respects the k cutoff (only top-k considered)', () => {
|
||||
// top-3 retrieved = ['a','b','c']; relevant = {'a','d'} → 1/3
|
||||
expect(precisionAtK(['a', 'b', 'c', 'd', 'e'], new Set(['a', 'd']), 3)).toBeCloseTo(1 / 3);
|
||||
});
|
||||
|
||||
test('k larger than retrieved length uses retrieved length', () => {
|
||||
// retrieved length 2, k=5 → divide by 2
|
||||
expect(precisionAtK(['a', 'b'], new Set(['a']), 5)).toBeCloseTo(1 / 2);
|
||||
});
|
||||
});
|
||||
|
||||
describe('recallAtK', () => {
|
||||
test('returns 1.0 when relevant is empty (degenerate)', () => {
|
||||
expect(recallAtK(['a', 'b'], new Set(), 5)).toBe(1);
|
||||
});
|
||||
|
||||
test('returns 1.0 when all relevant are in top-k', () => {
|
||||
expect(recallAtK(['a', 'b', 'c'], new Set(['a', 'b']), 5)).toBe(1);
|
||||
});
|
||||
|
||||
test('returns 0 when none of relevant are in top-k', () => {
|
||||
expect(recallAtK(['x', 'y', 'z'], new Set(['a', 'b']), 5)).toBe(0);
|
||||
});
|
||||
|
||||
test('returns half when 1/2 relevant in top-k', () => {
|
||||
expect(recallAtK(['a', 'x'], new Set(['a', 'b']), 5)).toBeCloseTo(1 / 2);
|
||||
});
|
||||
|
||||
test('respects the k cutoff', () => {
|
||||
// top-2 = ['x','y']; relevant = {'a','b'} → recall=0; the third would be 'a' but k=2
|
||||
expect(recallAtK(['x', 'y', 'a', 'b'], new Set(['a', 'b']), 2)).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('top1StabilityRate', () => {
|
||||
test('returns 0 for empty runs', () => {
|
||||
expect(top1StabilityRate([], [])).toBe(0);
|
||||
});
|
||||
|
||||
test('returns 1.0 when all top-1s match', () => {
|
||||
const run1 = makeResults([{ id: 'q1', top_1: 'a' }, { id: 'q2', top_1: 'b' }]);
|
||||
const run2 = makeResults([{ id: 'q1', top_1: 'a' }, { id: 'q2', top_1: 'b' }]);
|
||||
expect(top1StabilityRate(run1, run2)).toBe(1);
|
||||
});
|
||||
|
||||
test('returns 0 when no top-1s match', () => {
|
||||
const run1 = makeResults([{ id: 'q1', top_1: 'a' }]);
|
||||
const run2 = makeResults([{ id: 'q1', top_1: 'x' }]);
|
||||
expect(top1StabilityRate(run1, run2)).toBe(0);
|
||||
});
|
||||
|
||||
test('ignores questions only in one run', () => {
|
||||
const run1 = makeResults([{ id: 'q1', top_1: 'a' }, { id: 'q2', top_1: 'b' }]);
|
||||
const run2 = makeResults([{ id: 'q1', top_1: 'a' }]); // missing q2
|
||||
// Only q1 is comparable; stable = 1/1 = 1
|
||||
expect(top1StabilityRate(run1, run2)).toBe(1);
|
||||
});
|
||||
|
||||
test('null top_1 in one run counts as non-stable when other is non-null', () => {
|
||||
const run1 = makeResults([{ id: 'q1', top_1: 'a' }]);
|
||||
const run2 = makeResults([{ id: 'q1', top_1: null }]);
|
||||
expect(top1StabilityRate(run1, run2)).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('normalizeRetrieved', () => {
|
||||
test('dedupes while preserving order', () => {
|
||||
expect(normalizeRetrieved(['a', 'b', 'a', 'c', 'b'])).toEqual(['a', 'b', 'c']);
|
||||
});
|
||||
|
||||
test('drops empty strings', () => {
|
||||
expect(normalizeRetrieved(['a', '', 'b'])).toEqual(['a', 'b']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('expandExpectedToRelevantSet / isFileRelevant', () => {
|
||||
test('exact file match', () => {
|
||||
const exp = expandExpectedToRelevantSet(['src/foo.ts', 'src/bar.ts']);
|
||||
expect(isFileRelevant('src/foo.ts', exp)).toBe(true);
|
||||
expect(isFileRelevant('src/baz.ts', exp)).toBe(false);
|
||||
});
|
||||
|
||||
test('directory prefix match (trailing slash)', () => {
|
||||
const exp = expandExpectedToRelevantSet(['src/core/']);
|
||||
expect(isFileRelevant('src/core/foo.ts', exp)).toBe(true);
|
||||
expect(isFileRelevant('src/core/sub/bar.ts', exp)).toBe(true);
|
||||
expect(isFileRelevant('src/other.ts', exp)).toBe(false);
|
||||
});
|
||||
|
||||
test('mixed exact + directory expected', () => {
|
||||
const exp = expandExpectedToRelevantSet(['src/foo.ts', 'src/core/']);
|
||||
expect(isFileRelevant('src/foo.ts', exp)).toBe(true);
|
||||
expect(isFileRelevant('src/core/bar.ts', exp)).toBe(true);
|
||||
expect(isFileRelevant('src/other.ts', exp)).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('loadQuestions', () => {
|
||||
test('parses the v0.34 baseline questions.json', () => {
|
||||
const file = loadQuestions('src/eval/code-retrieval/questions.json');
|
||||
expect(file.version).toBe(1);
|
||||
expect(file.corpus).toBe('gbrain');
|
||||
expect(file.questions.length).toBeGreaterThanOrEqual(12);
|
||||
for (const q of file.questions) {
|
||||
expect(q.id).toBeDefined();
|
||||
expect(q.kind).toMatch(/^(callers|callees|definition|references|blast_radius|execution_flow|cluster_membership)$/);
|
||||
expect(q.query.length).toBeGreaterThan(0);
|
||||
expect(q.symbol.length).toBeGreaterThan(0);
|
||||
expect(Array.isArray(q.expected_files)).toBe(true);
|
||||
expect(q.expected_min_recall).toBeGreaterThanOrEqual(0);
|
||||
expect(q.expected_min_recall).toBeLessThanOrEqual(1);
|
||||
}
|
||||
});
|
||||
|
||||
test('throws on missing file', () => {
|
||||
expect(() => loadQuestions('/tmp/does-not-exist-XXXX.json')).toThrow(/not found/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('evaluateGate', () => {
|
||||
test('PASS when precision delta clears bar AND enough cleared bar', () => {
|
||||
const baseline = makeReport('baseline', 0.4, 0.5, 20); // 20 questions, 50% answered
|
||||
const withCI = makeReport('with-code-intel', 0.55, 0.85, 20); // +15pp precision, 85% answered
|
||||
const gate = evaluateGate(baseline, withCI, {
|
||||
required_precision_delta_pp: 10,
|
||||
required_top_1_stability_delta: 0.15,
|
||||
min_questions_cleared: 15,
|
||||
});
|
||||
expect(gate.passed).toBe(true);
|
||||
expect(gate.precision_delta_pp).toBeCloseTo(15, 5);
|
||||
});
|
||||
|
||||
test('FAIL when not enough questions cleared bar (despite precision delta)', () => {
|
||||
const baseline = makeReport('baseline', 0.4, 0.5, 30);
|
||||
const withCI = makeReport('with-code-intel', 0.6, 0.4, 30); // good precision, fewer answered
|
||||
const gate = evaluateGate(baseline, withCI, {
|
||||
required_precision_delta_pp: 10,
|
||||
required_top_1_stability_delta: 0.15,
|
||||
min_questions_cleared: 15,
|
||||
});
|
||||
expect(gate.passed).toBe(false);
|
||||
expect(gate.summary).toContain('only ');
|
||||
});
|
||||
|
||||
test('PASS via answered_rate delta even when precision delta is below bar', () => {
|
||||
const baseline = makeReport('baseline', 0.4, 0.5, 30);
|
||||
const withCI = makeReport('with-code-intel', 0.45, 0.7, 30); // +5pp precision (fail) but +20pp answered (pass)
|
||||
const gate = evaluateGate(baseline, withCI, {
|
||||
required_precision_delta_pp: 10,
|
||||
required_top_1_stability_delta: 0.15,
|
||||
min_questions_cleared: 15,
|
||||
});
|
||||
expect(gate.passed).toBe(true);
|
||||
});
|
||||
|
||||
test('default opts match constants', () => {
|
||||
expect(DEFAULT_GATE.required_precision_delta_pp).toBe(10);
|
||||
expect(DEFAULT_GATE.required_top_1_stability_delta).toBe(0.15);
|
||||
expect(DEFAULT_GATE.min_questions_cleared).toBe(15);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── helpers ──────────────────────────────────────────────────────────
|
||||
|
||||
function makeResults(items: Array<{ id: string; top_1: string | null }>): QuestionResult[] {
|
||||
return items.map((it) => ({
|
||||
id: it.id,
|
||||
kind: 'callers' as const,
|
||||
retrieved_files: it.top_1 ? [it.top_1] : [],
|
||||
top_1: it.top_1,
|
||||
precision_at_k: it.top_1 ? 1 : 0,
|
||||
recall_at_k: it.top_1 ? 1 : 0,
|
||||
answered: !!it.top_1,
|
||||
latency_ms: 1,
|
||||
}));
|
||||
}
|
||||
|
||||
function makeReport(
|
||||
mode: 'baseline' | 'with-code-intel',
|
||||
meanPrecision: number,
|
||||
answeredRate: number,
|
||||
totalQuestions: number,
|
||||
): EvalRunReport {
|
||||
const answeredCount = Math.round(answeredRate * totalQuestions);
|
||||
const questions: QuestionResult[] = [];
|
||||
for (let i = 0; i < totalQuestions; i++) {
|
||||
questions.push({
|
||||
id: `q${i}`,
|
||||
kind: 'callers' as const,
|
||||
retrieved_files: ['fakefile.ts'],
|
||||
top_1: 'fakefile.ts',
|
||||
precision_at_k: meanPrecision,
|
||||
recall_at_k: 0.5,
|
||||
answered: i < answeredCount,
|
||||
latency_ms: 1,
|
||||
});
|
||||
}
|
||||
return {
|
||||
mode,
|
||||
schema_version: 1,
|
||||
corpus: 'fake',
|
||||
k: 5,
|
||||
questions,
|
||||
mean_precision_at_k: meanPrecision,
|
||||
answered_rate: answeredRate,
|
||||
total_latency_ms: totalQuestions,
|
||||
captured_at: '2026-05-10T00:00:00Z',
|
||||
commit: 'abc1234',
|
||||
};
|
||||
}
|
||||
Reference in New Issue
Block a user