Files
gbrain/test/benchmark-knowledge-runtime.ts
T
Garry TanandClaude Opus 4.7 7973d0d965 feat(v0.34 STEP 0): promote OperationContext.sourceId to REQUIRED (D4)
Flip src/core/operations.ts:350 `sourceId?: string` → `sourceId: string`.
Mirrors v0.26.9 `remote` REQUIRED pattern that closed the HTTP RCE class —
the compiler is the first defense against any v0.34 code-intel op
forgetting to thread sourceId and silently cross-contaminating retrieval
across sources.

- src/mcp/dispatch.ts: buildOperationContext auto-fills 'default' when
  opts.sourceId is undefined. Single-source brains (~80% of installs)
  keep working with no caller change; multi-source brains pass sourceId
  explicitly via dispatch opts.
- src/cli.ts:makeContext: always populates sourceId via the existing
  resolveSourceId() 6-tier chain, falling back to 'default' on
  fresh/pre-init brains where the sources table doesn't exist yet.
- src/commands/book-mirror.ts, src/core/minions/tools/brain-allowlist.ts:
  Two production context-builders that previously omitted sourceId.
  Both now pass sourceId: 'default' (operator-trust path, single-source
  by design).
- 10 test/* files: every OperationContext literal now passes sourceId.

test/operation-context-sourceid-required.test.ts: paired contract test
(6 cases) pinning the type contract. @ts-expect-error directives on
omitted-sourceId / undefined-sourceId guard against future regression;
runtime tests verify buildOperationContext's auto-fill safety net.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-14 08:37:01 -07:00

394 lines
12 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Knowledge Runtime Benchmark — does the branch actually improve gbrain?
*
* Three measurable comparisons, each isolating one claim the PR makes.
* All run in-process against PGLite with mocked resolvers. Deterministic,
* no network, no API keys.
*
* 1. TIME-TO-QUERYABLE: seed pages via put_page OPERATION, immediately
* query timeline. With auto_timeline ON (branch default), timeline is
* populated at write-time; with auto_timeline OFF (master behavior),
* timeline is empty until user runs `gbrain extract timeline`.
* Metric: % of expected timeline queries that return correct answers
* immediately after ingest.
*
* 2. INTEGRITY REPAIR RATE: seed pages with bare-tweet phrases, mock the
* x_handle_to_tweet resolver with a realistic confidence distribution
* (70% high / 20% mid / 10% low), run the three-bucket repair logic.
* Metric: % auto-repaired, % sent to review, % skipped.
*
* 3. DOCTOR COMPLETENESS: seed a brain with 6 known integrity issues
* (bare tweets, dead-looking external link patterns, grandfathered
* pages), run the scanIntegrity helper doctor now invokes. Metric:
* issues-surfaced / issues-planted.
*
* Usage: bun run test/benchmark-knowledge-runtime.ts
* bun run test/benchmark-knowledge-runtime.ts --json
*/
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
import { operationsByName } from '../src/core/operations.ts';
import type { OperationContext } from '../src/core/operations.ts';
import { ResolverRegistry } from '../src/core/resolvers/registry.ts';
import type { Resolver, ResolverContext } from '../src/core/resolvers/index.ts';
import {
findBareTweetHits,
findExternalLinks,
extractXHandleFromFrontmatter,
scanIntegrity,
} from '../src/commands/integrity.ts';
const jsonMode = process.argv.includes('--json');
const log = jsonMode ? (..._args: unknown[]) => {} : console.log;
// ─── Shared helpers ─────────────────────────────────────────────
async function freshEngine(): Promise<PGLiteEngine> {
const e = new PGLiteEngine();
await e.connect({});
await e.initSchema();
return e;
}
function makeOpCtx(engine: PGLiteEngine): OperationContext {
return {
engine,
config: { engine: 'pglite' } as any,
logger: { info: () => {}, warn: () => {}, error: () => {} },
dryRun: false,
remote: false,
sourceId: 'default',
};
}
function round(n: number, digits = 2): number {
const p = 10 ** digits;
return Math.round(n * p) / p;
}
// ─── Benchmark 1: Time-to-queryable ────────────────────────────
interface SeedPage {
slug: string;
content: string;
expectedTimeline: Array<{ date: string; summary: string }>;
}
function makeTTQSeeds(): SeedPage[] {
const seeds: SeedPage[] = [];
for (let i = 0; i < 20; i++) {
const dateA = `2026-0${(i % 9) + 1}-15`;
const dateB = `2026-0${(i % 9) + 1}-28`;
seeds.push({
slug: `people/person-${i}`,
content: [
`---`,
`type: person`,
`title: Person ${i}`,
`---`,
``,
`Person ${i} is a founder.`,
``,
`## Timeline`,
``,
`- **${dateA}** | Shipped v${i}.0`,
`- **${dateB}** | Closed round ${i}`,
].join('\n'),
expectedTimeline: [
{ date: dateA, summary: `Shipped v${i}.0` },
{ date: dateB, summary: `Closed round ${i}` },
],
});
}
return seeds;
}
async function runTTQ(autoTimeline: boolean): Promise<{ expected: number; found: number; pct: number }> {
const engine = await freshEngine();
await engine.setConfig('auto_timeline', autoTimeline ? 'true' : 'false');
const ctx = makeOpCtx(engine);
const putOp = operationsByName['put_page']!;
const seeds = makeTTQSeeds();
for (const s of seeds) {
await putOp.handler(ctx, { slug: s.slug, content: s.content });
}
let expected = 0, found = 0;
const isoDate = (d: unknown): string => {
if (d instanceof Date) return d.toISOString().slice(0, 10);
return String(d).slice(0, 10);
};
for (const s of seeds) {
const entries = await engine.getTimeline(s.slug);
for (const e of s.expectedTimeline) {
expected++;
if (entries.some(row => isoDate(row.date) === e.date && row.summary === e.summary)) found++;
}
}
await engine.disconnect();
return { expected, found, pct: expected > 0 ? found / expected : 0 };
}
// ─── Benchmark 2: Integrity repair rate ────────────────────────
/** Fake resolver that returns a confidence score derived deterministically
* from the input handle so runs are reproducible. Mirrors the real
* x_handle_to_tweet resolver output shape.
*/
function makeFakeXResolver(): Resolver<{ handle: string; keywords: string }, {
url?: string; tweet_id?: string; created_at?: string;
candidates: Array<{ tweet_id: string; text: string; created_at: string; score: number; url: string }>;
}> {
return {
id: 'x_handle_to_tweet',
cost: 'free',
backend: 'local',
description: 'Fake for benchmark',
async available() { return true; },
async resolve(req) {
const h = req.input.handle;
// Deterministic distribution: 70% high conf, 20% mid, 10% low
const bucket = hashString(h) % 10;
let confidence: number;
if (bucket < 7) confidence = 0.85;
else if (bucket < 9) confidence = 0.65;
else confidence = 0.30;
const tid = String(1000000000 + (hashString(h) % 999999999));
return {
value: {
url: `https://x.com/${h}/status/${tid}`,
tweet_id: tid,
created_at: '2026-04-01T12:00:00.000Z',
candidates: [{
tweet_id: tid,
text: 'fake tweet text',
created_at: '2026-04-01T12:00:00.000Z',
score: confidence,
url: `https://x.com/${h}/status/${tid}`,
}],
},
confidence,
source: 'fake',
fetchedAt: new Date(),
};
},
};
}
function hashString(s: string): number {
let h = 0;
for (let i = 0; i < s.length; i++) h = (h * 31 + s.charCodeAt(i)) | 0;
return Math.abs(h);
}
async function runIntegrityBench(): Promise<{
pages: number;
hits: number;
bucketAuto: number;
bucketReview: number;
bucketSkip: number;
pctAuto: number;
pctReview: number;
pctSkip: number;
}> {
const engine = await freshEngine();
const ctx = makeOpCtx(engine);
const putOp = operationsByName['put_page']!;
const handles: string[] = [];
for (let i = 0; i < 50; i++) {
const handle = `handle${i}`;
handles.push(handle);
const content = [
`---`,
`type: person`,
`title: Person ${i}`,
`x_handle: ${handle}`,
`---`,
``,
`Person ${i} tweeted about AI safety this year.`,
].join('\n');
await putOp.handler(ctx, { slug: `people/person-${i}`, content });
}
// Build an isolated registry with our fake resolver
const registry = new ResolverRegistry();
registry.register(makeFakeXResolver());
const resolverCtx: ResolverContext = {
engine,
config: {},
logger: { info: () => {}, warn: () => {}, error: () => {} },
requestId: 'bench',
remote: false,
};
const confidenceThreshold = 0.8;
const reviewLower = 0.5;
let bucketAuto = 0, bucketReview = 0, bucketSkip = 0, hits = 0, pages = 0;
const slugs = [...(await engine.getAllSlugs())].sort();
for (const slug of slugs) {
const page = await engine.getPage(slug);
if (!page) continue;
pages++;
const handle = extractXHandleFromFrontmatter(page.frontmatter);
const bareHits = findBareTweetHits(page.compiled_truth, slug);
if (bareHits.length === 0 || !handle) continue;
for (const hit of bareHits) {
hits++;
const result = await registry.resolve<{ handle: string; keywords: string }, any>(
'x_handle_to_tweet',
{ handle, keywords: hit.rawLine.slice(0, 150) },
resolverCtx,
);
if (result.confidence >= confidenceThreshold) bucketAuto++;
else if (result.confidence >= reviewLower) bucketReview++;
else bucketSkip++;
}
}
await engine.disconnect();
return {
pages,
hits,
bucketAuto,
bucketReview,
bucketSkip,
pctAuto: hits > 0 ? bucketAuto / hits : 0,
pctReview: hits > 0 ? bucketReview / hits : 0,
pctSkip: hits > 0 ? bucketSkip / hits : 0,
};
}
// ─── Benchmark 3: Doctor completeness ──────────────────────────
async function runDoctorCompletenessBench(): Promise<{
planted: number;
surfaced: number;
pct: number;
breakdown: { bareTweets: number; externalLinks: number; grandfathered: number };
}> {
const engine = await freshEngine();
const ctx = makeOpCtx(engine);
const putOp = operationsByName['put_page']!;
// Plant known issues
// 3 bare-tweet phrases across 2 pages
// 3 external link citations (look like dead-link candidates)
// 1 grandfathered page (should be ignored = not counted as surfaced)
await putOp.handler(ctx, {
slug: 'people/alice',
content: `---
type: person
title: Alice
x_handle: alice
---
Alice tweeted about scaling last week. She also posted on X yesterday.
`,
});
await putOp.handler(ctx, {
slug: 'people/bob',
content: `---
type: person
title: Bob
x_handle: bob
---
Bob wrote a tweet covering the incident.
`,
});
await putOp.handler(ctx, {
slug: 'concepts/essays',
content: `---
type: concept
title: Essays
---
See [PG's essay](http://old-defunct.example/essay1) and [another](https://dead.example/x).
Also [a third reference](https://invalid.example/path).
`,
});
await putOp.handler(ctx, {
slug: 'people/legacy',
content: `---
type: person
title: Legacy
validate: false
---
Legacy tweeted about old things that should be ignored.
`,
});
const res = await scanIntegrity(engine);
const planted = 3 + 3 + 1; // 7 total, 1 grandfathered (should NOT surface)
const shouldSurface = 3 + 3; // 6
const surfaced = res.bareHits.length + res.externalHits.length;
await engine.disconnect();
return {
planted,
surfaced,
pct: surfaced / shouldSurface,
breakdown: {
bareTweets: res.bareHits.length,
externalLinks: res.externalHits.length,
grandfathered: planted - shouldSurface, // 1
},
};
}
// ─── Main runner ───────────────────────────────────────────────
async function main() {
log('# Knowledge Runtime Benchmark');
log(`Generated: ${new Date().toISOString().slice(0, 19)}`);
log('');
log('## 1. Time-to-queryable brain');
const ttqBranch = await runTTQ(true);
const ttqMaster = await runTTQ(false);
log(` branch (auto_timeline=on): ${ttqBranch.found}/${ttqBranch.expected} queryable (${round(ttqBranch.pct * 100)}%)`);
log(` master (auto_timeline=off): ${ttqMaster.found}/${ttqMaster.expected} queryable (${round(ttqMaster.pct * 100)}%)`);
log('');
log('## 2. Integrity repair rate (mocked resolver, 70/20/10 distribution)');
const intRes = await runIntegrityBench();
log(` pages scanned: ${intRes.pages}`);
log(` bare-tweet hits: ${intRes.hits}`);
log(` auto-repair (≥0.8): ${intRes.bucketAuto} (${round(intRes.pctAuto * 100)}%)`);
log(` review (0.50.8): ${intRes.bucketReview} (${round(intRes.pctReview * 100)}%)`);
log(` skip (<0.5): ${intRes.bucketSkip} (${round(intRes.pctSkip * 100)}%)`);
log('');
log('## 3. Doctor completeness');
const docRes = await runDoctorCompletenessBench();
log(` issues planted: ${docRes.planted} (6 should surface, 1 grandfathered)`);
log(` issues surfaced: ${docRes.surfaced} (${round(docRes.pct * 100)}%)`);
log(` bare tweets caught: ${docRes.breakdown.bareTweets}/3`);
log(` external links caught: ${docRes.breakdown.externalLinks}/3`);
log(` grandfathered correctly skipped: ${docRes.breakdown.grandfathered}/1`);
log('');
const report = {
ttq: { branch: ttqBranch, master: ttqMaster },
integrity: intRes,
doctor: docRes,
};
if (jsonMode) {
console.log(JSON.stringify(report, null, 2));
}
}
main().catch(e => {
console.error(e);
process.exit(1);
});