mirror of
https://github.com/garrytan/gbrain.git
synced 2026-07-31 04:07:52 +00:00
* fix(autopilot,eval): nightly quality probe enable path works end-to-end + wire conversation-parser probe Takeover of #2629 and #2630 (rebased onto master; dropped the test/engine-find-trajectory.test.ts hunk both PRs carried — master already ships the equivalent gateway-dims fix). #2629 — nightly quality probe enable path: - autopilot + doctor read the probe flag dual-plane (DB config row from 'gbrain config set' wins, ~/.gbrain/config.json fallback) via new resolveProbeEnabled/resolveProbeMaxUsd helpers - resolveRepoRoot prefers the gbrain package root where the committed fixture lives, not the brain repoPath - rate_limited skips no longer write an audit row every autopilot cycle - eval-longmemeval strips 'provider:' recipe ids before raw Anthropic SDK calls and emits the gold answer for downstream judges - cross-modal batch folds the gold answer into the judge task; probe passes QA-shaped dimensions instead of the agent-response rubric - DEFAULT_SLOTS slot A moves to openai:gpt-5.2 (gpt-4o left the recipe); new consistency test pins every default slot to its recipe #2630 — conversation-parser nightly probe wire-up: - autopilot step 4.6 invokes runConversationParserNightlyProbe (dual-plane flag + D10 tokenmax mode-gate, package-root fixtures, 24h gate, audit trail via new src/core/audit-parser-probe.ts) - doctor's conversation_parser_probe_health replaces the hardcoded 'Skipped' stub with a real pure-function check over the audit trail Co-authored-by: p3ob7o <p3ob7o@users.noreply.github.com> Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(pricing): add openai:gpt-5.2 canonical entry for the new default slot A DEFAULT_SLOTS slot A moved to openai:gpt-5.2, which had no CANONICAL_PRICING entry — estimateCost silently dropped slot A from the --max-usd pre-flight and est_cost_usd audit rows (~1/3 under-count on the default panel). Rates from the OpenAI recipe chat touchpoint (verified 2026-04-20). Also refresh the --slot-a-model help text default and pin a pricing-presence assertion in the DEFAULT_SLOTS consistency test. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: Sinabina <sinabina@Sinabinas-MacBook-Pro-4.local> Co-authored-by: p3ob7o <p3ob7o@users.noreply.github.com> Co-authored-by: Claude Fable 5 <noreply@anthropic.com> Co-authored-by: Garry Tan <garrytan@gmail.com>
78 lines
3.4 KiB
TypeScript
78 lines
3.4 KiB
TypeScript
/**
|
|
* Source-shape regression tests for the autopilot wiring of
|
|
* `runConversationParserNightlyProbe` (step 4.6).
|
|
*
|
|
* Same rationale as autopilot-nightly-probe-wiring.test.ts: the loop is
|
|
* hard to drive end-to-end, so these pin the structural protections —
|
|
* the dual-plane flag read, the D10 tokenmax mode-gate, the package-root
|
|
* fixture resolution, the audit-flood guard, and the try/catch posture.
|
|
*
|
|
* The probe's own gate/scoring logic is pinned by the module's unit
|
|
* tests; the audit trail by audit-parser-probe.serial.test.ts.
|
|
*/
|
|
|
|
import { describe, test, expect } from 'bun:test';
|
|
import { readFileSync } from 'node:fs';
|
|
import { resolve } from 'node:path';
|
|
|
|
const AUTOPILOT_SRC = resolve('src/commands/autopilot.ts');
|
|
const SOURCE = readFileSync(AUTOPILOT_SRC, 'utf-8');
|
|
|
|
describe('autopilot wiring: conversation-parser probe', () => {
|
|
test('invokes the phase module and the audit trail', () => {
|
|
expect(SOURCE).toContain(`runConversationParserNightlyProbe`);
|
|
expect(SOURCE).toContain(`conversation-parser/nightly-probe`);
|
|
expect(SOURCE).toContain(`logParserProbeEvent`);
|
|
expect(SOURCE).toContain(`audit-parser-probe`);
|
|
});
|
|
|
|
test('flag reads dual-plane: DB row (gbrain config set) wins, file plane fallback', () => {
|
|
expect(SOURCE).toContain(`getConfig('autopilot.conversation_parser_probe.enabled')`);
|
|
expect(SOURCE).toContain(`cfg?.autopilot?.conversation_parser_probe?.enabled === true`);
|
|
});
|
|
|
|
test('D10 mode-gate present: tokenmax brains run the probe by default', () => {
|
|
expect(SOURCE).toMatch(/parserEnabled \|\| searchMode === 'tokenmax'/);
|
|
});
|
|
|
|
test('fixtures resolve from the gbrain package root, NOT the brain repoPath', () => {
|
|
// The committed fixtures live in the gbrain source tree; resolving
|
|
// them against sync.repo_path would point into the user's brain repo.
|
|
expect(SOURCE).toMatch(/fileURLToPath\(new URL\('\.\.\/\.\.', import\.meta\.url\)\)/);
|
|
expect(SOURCE).toContain(`'conversation-formats', 'all.jsonl'`);
|
|
expect(SOURCE).toContain(`'conversation-formats', 'adversarial.jsonl'`);
|
|
});
|
|
|
|
test('missing fixtures skip quietly (no audit row, once-per-process stderr note)', () => {
|
|
// Compiled-binary installs carry no source tree; writing failure rows
|
|
// would flip doctor to WARN on every binary install.
|
|
expect(SOURCE).toContain(`parserProbeFixtureWarned`);
|
|
});
|
|
|
|
test('rate_limited outcomes are NOT audit-logged (flood guard)', () => {
|
|
expect(SOURCE).toMatch(/outcome !== 'rate_limited'\) logParserProbeEvent\(result\)/);
|
|
});
|
|
|
|
test('rate-limit gate delegates to the audit module, not inline event reads', () => {
|
|
expect(SOURCE).toContain(`parserProbeRanWithin(24 * 60 * 60 * 1000)`);
|
|
});
|
|
|
|
test('LLM-key gate reads gateway.isAvailable("chat") in-process', () => {
|
|
expect(SOURCE).toContain(`isAvailable('chat')`);
|
|
});
|
|
|
|
test('probe call wrapped in try/catch that does NOT bump consecutiveErrors', () => {
|
|
expect(SOURCE).toMatch(/catch[\s\S]*?autopilot\.parser_probe[\s\S]*?do NOT bump consecutiveErrors/);
|
|
});
|
|
|
|
test('DI shape: the exact 7 fields of the parser probe NightlyProbeDeps', () => {
|
|
expect(SOURCE).toContain(`isEnabled:`);
|
|
expect(SOURCE).toContain(`searchMode:`);
|
|
expect(SOURCE).toContain(`hasLlmKey:`);
|
|
expect(SOURCE).toContain(`resolveFixturePath:`);
|
|
expect(SOURCE).toContain(`resolveAdversarialPath:`);
|
|
expect(SOURCE).toContain(`shouldSkipForRateLimit:`);
|
|
expect(SOURCE).toContain(`now:`);
|
|
});
|
|
});
|