mirror of
https://github.com/garrytan/gbrain.git
synced 2026-07-27 22:15:33 +00:00
* fix: think --model fails loud on unresolvable model; never persist empty synthesis (#1698) Slash-form model ids (anthropic/claude-sonnet-4-6) silently degraded to the no-LLM stub and wrote empty synthesis pages with exit 0 (reporter saw 200). Three compounding defects, fixed: - normalizeModelId (src/core/model-id.ts): one shared provider:model normalizer replacing 4 colon-only inlines; slash→colon, bare→default, malformed leading-separator (:foo) returned unchanged so resolveRecipe throws loud. - validateModelId + probeChatModel (gateway.ts): shared id-validity + key probe; runThink hard-errors on an explicit --model it can't run (no silent degrade). - synthesisOk + persist-skip (think/index.ts): empty/malformed/empty-JSON synthesis is never persisted; --save with no synthesis exits 1. - auto-think (cycle/auto-think.ts): empty synthesis no longer counts complete or advances the cooldown (the autonomous third caller of persistSynthesis). - hasAnthropicKey consolidated into src/core/ai/anthropic-key.ts (3 copies → 1). MCP think op sets modelExplicit; saved_slug '' maps to null. * chore: bump version and changelog (v0.42.4.0) #1698 think --model fail-loud wave. Also files the P3 follow-up TODO for the provider-symmetric early gate (D1 accept-as-is). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
187 lines
8.9 KiB
TypeScript
187 lines
8.9 KiB
TypeScript
import { describe, test, expect, beforeAll, afterAll, beforeEach } from 'bun:test';
|
|
import { mkdtempSync, rmSync, existsSync } from 'node:fs';
|
|
import { join } from 'node:path';
|
|
import { tmpdir } from 'node:os';
|
|
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
|
import { runPhaseAutoThink } from '../src/core/cycle/auto-think.ts';
|
|
import { runPhaseDrift, __testing as driftTesting } from '../src/core/cycle/drift.ts';
|
|
import { _resetBudgetMeterWarningsForTest } from '../src/core/cycle/budget-meter.ts';
|
|
import type { ThinkLLMClient } from '../src/core/think/index.ts';
|
|
|
|
let engine: PGLiteEngine;
|
|
let alicePageId: number;
|
|
let tmpDir: string;
|
|
|
|
function makeStubClient(answer: string): ThinkLLMClient {
|
|
return {
|
|
create: async () => ({
|
|
id: 'msg_stub',
|
|
type: 'message',
|
|
role: 'assistant',
|
|
model: 'stub',
|
|
stop_reason: 'end_turn',
|
|
stop_sequence: null,
|
|
usage: { input_tokens: 10, output_tokens: 10, cache_creation_input_tokens: 0, cache_read_input_tokens: 0, server_tool_use: null, service_tier: null },
|
|
content: [{ type: 'text', text: JSON.stringify({ answer, citations: [], gaps: [] }) }],
|
|
}),
|
|
};
|
|
}
|
|
|
|
beforeAll(async () => {
|
|
engine = new PGLiteEngine();
|
|
await engine.connect({});
|
|
await engine.initSchema();
|
|
const alice = await engine.putPage('people/alice-example', {
|
|
title: 'Alice', type: 'person', compiled_truth: 'Alice content',
|
|
});
|
|
alicePageId = alice.id;
|
|
// Add takes spanning the soft band so drift candidates exist
|
|
await engine.addTakesBatch([
|
|
{ page_id: alicePageId, row_num: 1, claim: 'CEO of Acme', kind: 'fact', holder: 'world', weight: 1.0 },
|
|
{ page_id: alicePageId, row_num: 2, claim: 'Strong technical founder', kind: 'take', holder: 'garry', weight: 0.6 },
|
|
{ page_id: alicePageId, row_num: 3, claim: 'Will reach $50B', kind: 'bet', holder: 'garry', weight: 0.5 },
|
|
]);
|
|
// Add timeline entries to give drift candidates "recent evidence"
|
|
await engine.addTimelineEntriesBatch([
|
|
{ slug: 'people/alice-example', date: new Date().toISOString().slice(0, 10), source: 'crustdata', summary: 'Funding round closed' },
|
|
{ slug: 'people/alice-example', date: new Date().toISOString().slice(0, 10), source: 'meeting', summary: 'OH discussion' },
|
|
]);
|
|
});
|
|
|
|
afterAll(async () => {
|
|
await engine.disconnect();
|
|
});
|
|
|
|
beforeEach(() => {
|
|
_resetBudgetMeterWarningsForTest();
|
|
tmpDir = mkdtempSync(join(tmpdir(), 'auto-think-'));
|
|
});
|
|
|
|
describe('runPhaseAutoThink', () => {
|
|
test('skipped when not enabled', async () => {
|
|
const r = await runPhaseAutoThink(engine, { dryRun: false, auditPath: join(tmpDir, 'budget.jsonl') });
|
|
expect(r.status).toBe('skipped');
|
|
expect(r.detail).toContain('false');
|
|
});
|
|
|
|
test('skipped when enabled but no questions', async () => {
|
|
await engine.setConfig('dream.auto_think.enabled', 'true');
|
|
await engine.setConfig('dream.auto_think.questions', '[]');
|
|
const r = await runPhaseAutoThink(engine, { dryRun: false, auditPath: join(tmpDir, 'b1.jsonl') });
|
|
expect(r.status).toBe('skipped');
|
|
expect(r.detail).toContain('empty');
|
|
await engine.setConfig('dream.auto_think.enabled', 'false');
|
|
});
|
|
|
|
test('runs when enabled with questions, marks success on cooldown ts', async () => {
|
|
await engine.setConfig('dream.auto_think.enabled', 'true');
|
|
await engine.setConfig('dream.auto_think.questions', JSON.stringify(['What about technical founders?']));
|
|
await engine.setConfig('dream.auto_think.max_per_cycle', '1');
|
|
await engine.setConfig('dream.auto_think.budget', '10.0');
|
|
await engine.setConfig('dream.auto_think.auto_commit', 'false');
|
|
// Clear any prior cooldown
|
|
await engine.setConfig('dream.auto_think.last_completion_ts', '');
|
|
const r = await runPhaseAutoThink(engine, {
|
|
dryRun: false,
|
|
client: makeStubClient('Alice [people/alice-example#2] is a strong founder.'),
|
|
auditPath: join(tmpDir, 'b2.jsonl'),
|
|
});
|
|
expect(r.status).toBe('complete');
|
|
expect((r.totals as { synthesized?: number }).synthesized).toBe(1);
|
|
const ts = await engine.getConfig('dream.auto_think.last_completion_ts');
|
|
expect(ts).toBeTruthy();
|
|
expect(ts!.length).toBeGreaterThan(0);
|
|
await engine.setConfig('dream.auto_think.enabled', 'false');
|
|
});
|
|
|
|
// #1698 (codex #5): an empty synthesis must NOT count as complete or advance the
|
|
// cooldown — otherwise auto-think silently reports success and suppresses retry until
|
|
// the cooldown expires. An empty-answer stub drives runThink's synthesisOk=false path.
|
|
test('empty synthesis → partial, 0 synthesized, cooldown NOT advanced', async () => {
|
|
await engine.setConfig('dream.auto_think.enabled', 'true');
|
|
await engine.setConfig('dream.auto_think.questions', JSON.stringify(['Q-empty']));
|
|
await engine.setConfig('dream.auto_think.max_per_cycle', '1');
|
|
await engine.setConfig('dream.auto_think.budget', '10.0');
|
|
await engine.setConfig('dream.auto_think.auto_commit', 'false');
|
|
await engine.setConfig('dream.auto_think.cooldown_days', '30');
|
|
await engine.setConfig('dream.auto_think.last_completion_ts', '');
|
|
const r = await runPhaseAutoThink(engine, {
|
|
dryRun: false,
|
|
client: makeStubClient(''), // empty answer → synthesisOk=false
|
|
auditPath: join(tmpDir, 'b-empty.jsonl'),
|
|
});
|
|
expect(r.status).toBe('partial');
|
|
expect((r.totals as { synthesized?: number }).synthesized).toBe(0);
|
|
// Cooldown must stay empty so the next cycle retries (no silent success).
|
|
const ts = await engine.getConfig('dream.auto_think.last_completion_ts');
|
|
expect(ts ?? '').toBe('');
|
|
await engine.setConfig('dream.auto_think.enabled', 'false');
|
|
await engine.setConfig('dream.auto_think.cooldown_days', '0');
|
|
});
|
|
|
|
test('cooldown skips next run', async () => {
|
|
await engine.setConfig('dream.auto_think.enabled', 'true');
|
|
await engine.setConfig('dream.auto_think.questions', JSON.stringify(['Q1']));
|
|
await engine.setConfig('dream.auto_think.cooldown_days', '30');
|
|
// Set a recent completion ts
|
|
await engine.setConfig('dream.auto_think.last_completion_ts', new Date().toISOString());
|
|
const r = await runPhaseAutoThink(engine, { dryRun: false, auditPath: join(tmpDir, 'b3.jsonl') });
|
|
expect(r.status).toBe('skipped');
|
|
expect(r.detail).toContain('cooled down');
|
|
await engine.setConfig('dream.auto_think.enabled', 'false');
|
|
await engine.setConfig('dream.auto_think.last_completion_ts', '');
|
|
});
|
|
|
|
test('budget exhausted denies further submits, returns partial', async () => {
|
|
await engine.setConfig('dream.auto_think.enabled', 'true');
|
|
await engine.setConfig('dream.auto_think.questions', JSON.stringify(['Q1', 'Q2', 'Q3']));
|
|
await engine.setConfig('dream.auto_think.max_per_cycle', '3');
|
|
await engine.setConfig('dream.auto_think.budget', '0.001'); // tiny cap forces budget_exhausted on first submit
|
|
await engine.setConfig('dream.auto_think.cooldown_days', '0');
|
|
await engine.setConfig('dream.auto_think.last_completion_ts', '');
|
|
// Ensure a clean meter state (no warn-once leftover)
|
|
_resetBudgetMeterWarningsForTest();
|
|
const r = await runPhaseAutoThink(engine, {
|
|
dryRun: false,
|
|
client: makeStubClient('test'),
|
|
auditPath: join(tmpDir, 'b4.jsonl'),
|
|
});
|
|
// First submit denied → no syntheses → status 'partial' if any attempts, else 'skipped'.
|
|
// Our impl returns 'partial' when results.length > 0 and anyComplete=false.
|
|
expect(['partial', 'skipped']).toContain(r.status);
|
|
await engine.setConfig('dream.auto_think.enabled', 'false');
|
|
});
|
|
});
|
|
|
|
describe('runPhaseDrift', () => {
|
|
test('skipped when not enabled', async () => {
|
|
const r = await runPhaseDrift(engine, { dryRun: false, auditPath: join(tmpDir, 'd0.jsonl') });
|
|
expect(r.status).toBe('skipped');
|
|
});
|
|
|
|
test('findDriftCandidates returns soft-band takes with recent evidence', async () => {
|
|
const cands = await driftTesting.findDriftCandidates(engine, 30);
|
|
// Row 2 (weight 0.6) and row 3 (weight 0.5) qualify; row 1 (1.0) is filtered.
|
|
expect(cands.length).toBeGreaterThanOrEqual(1);
|
|
expect(cands.every(c => c.weight >= 0.3 && c.weight <= 0.85)).toBe(true);
|
|
});
|
|
|
|
test('runs and surfaces candidates when enabled', async () => {
|
|
await engine.setConfig('dream.drift.enabled', 'true');
|
|
await engine.setConfig('dream.drift.lookback_days', '30');
|
|
await engine.setConfig('dream.drift.budget', '1.0');
|
|
const r = await runPhaseDrift(engine, { dryRun: false, auditPath: join(tmpDir, 'd1.jsonl') });
|
|
expect(r.status).toBe('complete');
|
|
expect((r.totals as { candidates?: number }).candidates).toBeGreaterThanOrEqual(0);
|
|
await engine.setConfig('dream.drift.enabled', 'false');
|
|
});
|
|
|
|
test('dry-run returns skipped with candidate count', async () => {
|
|
await engine.setConfig('dream.drift.enabled', 'true');
|
|
const r = await runPhaseDrift(engine, { dryRun: true, auditPath: join(tmpDir, 'd2.jsonl') });
|
|
expect(r.status).toBe('skipped');
|
|
expect(r.detail).toContain('dry-run');
|
|
await engine.setConfig('dream.drift.enabled', 'false');
|
|
});
|
|
});
|