mirror of
https://github.com/garrytan/gbrain.git
synced 2026-07-30 11:22:34 +00:00
When the grade_takes phase auto-resolves a take as 'incorrect' or 'partial',
optionally write a learning entry to gstack's per-project learnings.jsonl
so other gstack skills (plan-ceo-review, ship, investigate, ...) can pull
it as context when relevant. The brain teaches every other tool about
the user's track record.
Config gate (D5 / CDX-17 mitigation):
`cycle.grade_takes.write_gstack_learnings` defaults FALSE. External
users may not have gstack installed; the gstack-learnings binary API
isn't stable yet. Garry's brain flips it true to opt in.
Quality gate:
Only 'incorrect' and 'partial' verdicts trigger the write. 'correct'
resolutions are noise (we expected the take to hold up — no learning).
'unresolvable' has no canonical column. Defense-in-depth runtime guard
in writeIncorrectResolution() rejects ineligible qualities with
reason='quality_not_eligible' so a caller misuse never surfaces a
malformed learning entry.
Auto-apply only:
Coupling fires only when grade_takes both auto-applies AND the verdict
is incorrect/partial AND the config flag is enabled. Manual resolutions
via `gbrain takes resolve` intentionally DO NOT propagate to gstack —
manual writes already carry operator intent; the calibration loop is
the noise-prone path that earns coupling.
Namespace:
Every entry's key starts with 'gbrain:calibration:v0.36.0.0:'. Lane D
`gbrain calibration --undo-wave v0.36.0.0` (T17) filters on this prefix
for the optional gstack-scrub step. First active bias tag suffixes the
key (e.g. 'take-42:over-confident-geography') so future analysis can
group learnings by bias pattern.
Architecture:
buildLearningEntry — pure. Truncates claim at 200 chars + ellipsis;
emits Pattern: line when activeBiasTags present; defaults confidence
to 0.8 when caller omits it.
writeIncorrectResolution — async wrapper. Honors config gate; honors
quality gate; calls the injected writer (or defaultGstackWriter in
production). Failures are non-fatal: returns
{ written: false, reason: 'write_failed' | 'binary_missing', error }.
The grade_takes phase logs to result.warnings and continues — gstack
coupling failure NEVER aborts a cycle.
defaultGstackWriter — shells out to gstack-learnings-log binary via
execFileSync. Throws GBrainError('GSTACK_BINARY_NOT_FOUND') when the
binary isn't on PATH; writeIncorrectResolution classifies that error
to reason='binary_missing' so the operator sees the install hint
instead of a generic write_failed.
Wired into grade-takes.ts after engine.resolveTake() inside the
auto-apply block. Only fires when shouldApply=true.
Tests: 14 cases.
buildLearningEntry (7): canonical shape, partial vs incorrect wording,
bias-tag suffix, no-tag fallback, claim truncation, default confidence,
no-reasoning omission.
writeIncorrectResolution (7): config gate, quality gate, happy path,
writer-throw graceful degrade, binary-missing classification, async
writer awaited, partial quality writes.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
210 lines
7.2 KiB
TypeScript
210 lines
7.2 KiB
TypeScript
/**
|
|
* v0.36.0.0 (T11 / E4) — gstack-learnings coupling tests.
|
|
*
|
|
* Hermetic. Pure-function tests + writer-injection tests. No real gstack
|
|
* binary, no shell-out.
|
|
*
|
|
* Tests cover:
|
|
* - config gate: enabled=false → skipped with reason='config_disabled'
|
|
* - quality gate: only 'incorrect' and 'partial' trigger
|
|
* - happy path: writer called with correct entry shape
|
|
* - entry shape: namespace prefix on key, files[] includes page slug,
|
|
* tag suffix when active bias tags present
|
|
* - graceful degrade: writer throw → reason='write_failed', no rethrow
|
|
* - binary-missing detection via error-message classification
|
|
* - long claim truncation
|
|
* - missing optional fields don't break entry construction
|
|
*/
|
|
|
|
import { describe, test, expect } from 'bun:test';
|
|
import {
|
|
writeIncorrectResolution,
|
|
buildLearningEntry,
|
|
GSTACK_LEARNING_NAMESPACE,
|
|
type IncorrectResolutionEvent,
|
|
type GstackLearningEntry,
|
|
} from '../src/core/calibration/gstack-coupling.ts';
|
|
import { GBrainError } from '../src/core/types.ts';
|
|
|
|
function buildEvent(overrides: Partial<IncorrectResolutionEvent> = {}): IncorrectResolutionEvent {
|
|
return {
|
|
takeId: 42,
|
|
pageSlug: 'wiki/companies/acme-example',
|
|
rowNum: 3,
|
|
holder: 'garry',
|
|
claim: 'Cold-start liquidity always wins in marketplaces.',
|
|
quality: 'incorrect',
|
|
weight: 0.85,
|
|
confidence: 0.95,
|
|
reasoning: 'Two competing marketplaces both failed to bootstrap demand-side liquidity.',
|
|
activeBiasTags: ['over-confident-market-timing'],
|
|
...overrides,
|
|
};
|
|
}
|
|
|
|
// ─── buildLearningEntry ─────────────────────────────────────────────
|
|
|
|
describe('buildLearningEntry', () => {
|
|
test('emits canonical entry shape', () => {
|
|
const entry = buildLearningEntry(buildEvent());
|
|
expect(entry.skill).toBe('gbrain-calibration');
|
|
expect(entry.type).toBe('observation');
|
|
expect(entry.source).toBe('observed');
|
|
expect(entry.key).toContain(GSTACK_LEARNING_NAMESPACE);
|
|
expect(entry.key).toContain('take-42');
|
|
expect(entry.files).toEqual(['wiki/companies/acme-example']);
|
|
expect(entry.insight).toContain('garry');
|
|
expect(entry.insight).toContain('was wrong');
|
|
expect(entry.insight).toContain('conviction 0.85');
|
|
});
|
|
|
|
test('uses "was partially wrong" wording on partial verdict', () => {
|
|
const entry = buildLearningEntry(buildEvent({ quality: 'partial' }));
|
|
expect(entry.insight).toContain('was partially wrong');
|
|
});
|
|
|
|
test('namespace tag suffix derived from first active bias tag', () => {
|
|
const entry = buildLearningEntry(
|
|
buildEvent({ activeBiasTags: ['over-confident-geography', 'late-on-macro'] }),
|
|
);
|
|
expect(entry.key).toContain('over-confident-geography');
|
|
expect(entry.insight).toContain('Pattern: over-confident-geography, late-on-macro');
|
|
});
|
|
|
|
test('omits Pattern: line when activeBiasTags empty', () => {
|
|
const entry = buildLearningEntry(buildEvent({ activeBiasTags: [] }));
|
|
expect(entry.insight).not.toContain('Pattern:');
|
|
});
|
|
|
|
test('truncates long claim text at 200 chars + ellipsis', () => {
|
|
const longClaim = 'x'.repeat(500);
|
|
const entry = buildLearningEntry(buildEvent({ claim: longClaim }));
|
|
// 200 chars + 1 ellipsis char = 201 visible chars in the quoted claim
|
|
expect(entry.insight).toContain('x'.repeat(200) + '…');
|
|
});
|
|
|
|
test('default confidence 0.8 when omitted', () => {
|
|
const ev = buildEvent();
|
|
delete (ev as IncorrectResolutionEvent & { confidence?: number }).confidence;
|
|
const entry = buildLearningEntry(ev);
|
|
expect(entry.confidence).toBe(0.8);
|
|
});
|
|
|
|
test('omits reasoning suffix when reasoning is undefined', () => {
|
|
const ev = buildEvent();
|
|
delete (ev as IncorrectResolutionEvent & { reasoning?: string }).reasoning;
|
|
const entry = buildLearningEntry(ev);
|
|
expect(entry.insight).not.toContain('Reasoning:');
|
|
});
|
|
});
|
|
|
|
// ─── writeIncorrectResolution ───────────────────────────────────────
|
|
|
|
describe('writeIncorrectResolution', () => {
|
|
test('config gate: enabled=false → skipped, no writer call', async () => {
|
|
let writerCalls = 0;
|
|
const result = await writeIncorrectResolution({
|
|
event: buildEvent(),
|
|
enabled: false,
|
|
writer: () => {
|
|
writerCalls++;
|
|
},
|
|
});
|
|
expect(result.written).toBe(false);
|
|
expect(result.reason).toBe('config_disabled');
|
|
expect(writerCalls).toBe(0);
|
|
});
|
|
|
|
test("quality gate: 'correct' or 'unresolvable' rejected (defensive)", async () => {
|
|
let writerCalls = 0;
|
|
const writer = () => {
|
|
writerCalls++;
|
|
};
|
|
// TypeScript will catch most misuses, but the runtime guard exists
|
|
// because the caller (grade-takes) determines quality from the verdict
|
|
// path — defense in depth.
|
|
const result = await writeIncorrectResolution({
|
|
event: buildEvent({ quality: 'correct' as IncorrectResolutionEvent['quality'] }),
|
|
enabled: true,
|
|
writer,
|
|
});
|
|
expect(result.written).toBe(false);
|
|
expect(result.reason).toBe('quality_not_eligible');
|
|
expect(writerCalls).toBe(0);
|
|
});
|
|
|
|
test('happy path: writer called with built entry, returns written=true', async () => {
|
|
let received: GstackLearningEntry | undefined;
|
|
const result = await writeIncorrectResolution({
|
|
event: buildEvent(),
|
|
enabled: true,
|
|
writer: (entry) => {
|
|
received = entry;
|
|
},
|
|
});
|
|
expect(result.written).toBe(true);
|
|
expect(received).toBeDefined();
|
|
expect(received!.skill).toBe('gbrain-calibration');
|
|
expect(received!.key).toContain('take-42');
|
|
});
|
|
|
|
test('writer throws → reason="write_failed", no rethrow', async () => {
|
|
const result = await writeIncorrectResolution({
|
|
event: buildEvent(),
|
|
enabled: true,
|
|
writer: () => {
|
|
throw new Error('connection refused');
|
|
},
|
|
});
|
|
expect(result.written).toBe(false);
|
|
expect(result.reason).toBe('write_failed');
|
|
expect(result.error).toContain('connection refused');
|
|
});
|
|
|
|
test('writer throws GBrainError(GSTACK_BINARY_NOT_FOUND) → reason="binary_missing"', async () => {
|
|
const result = await writeIncorrectResolution({
|
|
event: buildEvent(),
|
|
enabled: true,
|
|
writer: () => {
|
|
throw new GBrainError(
|
|
'GSTACK_BINARY_NOT_FOUND',
|
|
'gstack-learnings-log binary not on PATH',
|
|
'install gstack',
|
|
);
|
|
},
|
|
});
|
|
expect(result.written).toBe(false);
|
|
expect(result.reason).toBe('binary_missing');
|
|
});
|
|
|
|
test('writer that returns a Promise is awaited', async () => {
|
|
let resolved = false;
|
|
const writer = (_entry: GstackLearningEntry): Promise<void> =>
|
|
new Promise(r => {
|
|
setTimeout(() => {
|
|
resolved = true;
|
|
r();
|
|
}, 10);
|
|
});
|
|
const result = await writeIncorrectResolution({
|
|
event: buildEvent(),
|
|
enabled: true,
|
|
writer,
|
|
});
|
|
expect(result.written).toBe(true);
|
|
expect(resolved).toBe(true);
|
|
});
|
|
|
|
test('partial quality writes (not just incorrect)', async () => {
|
|
let writerCalls = 0;
|
|
await writeIncorrectResolution({
|
|
event: buildEvent({ quality: 'partial' }),
|
|
enabled: true,
|
|
writer: () => {
|
|
writerCalls++;
|
|
},
|
|
});
|
|
expect(writerCalls).toBe(1);
|
|
});
|
|
});
|