Files
gbrain/test/doctor-pool-reap-health.test.ts
T
3fe449361c v0.42.16.0 feat(doctor): brain health as a solved problem — cause-ranked doctor + OOM-loop line + auto-drain + pool-reap (#1685) (#1802)
* feat(minions): pool-recovery audit + reconnect reason-threading + shared drain helper (#1685 GAP B, 5A)

- pool-recovery-audit.ts: reap_detected (CONNECTION_ENDED) vs reconnect_other; recovered/failed split
- postgres-engine reconnect(ctx?) classifies the triggering error so only true pooler reaps are tagged (CODEX #8)
- retry.ts reconnect callback widened to thread the error; retry-matcher isConnectionEndedError
- runExtractAtomsDrainForSource shared helper (cycleLockIdFor + withRefreshingLock) — one drain path (5A)
- supervisor-audit readRecentSupervisorEvents (current+prev ISO week, CODEX #7)
- extract-atoms-drain PROTECTED; autopilot.auto_drain.* config keys

* feat(doctor): worker_oom_loop + pool_reap_health checks + cause-ranked top_issues (#1685 GAP A/B/C)

- computeWorkerOomLoopCheck: unions supervisor rss_watchdog + minion_jobs watchdog-abort (CODEX #5), cap fallback to resolveDefaultMaxRssMb (CODEX #6)
- computePoolReapHealthCheck: reaps-not-recovering fail, thrash warn
- doctor-cause-rank rankIssues: tier ordering + grounded downstream_of (CODEX #9) + drift guard (4A)
- supervisor causeStr + queue_health cross-reference worker_oom_loop (DRY 1C)
- register both checks in doctor-categories ops

* feat(autopilot): per-source extract_atoms auto-drain + handler + dream --drain refactor (#1685 GAP D)

- autopilot per-source gate: enabled + !packDeclares + backlog>threshold + daily cap; time-sloted idempotency key (CODEX #2)
- extract-atoms-drain Minion handler (thin wrapper, LockUnavailableError -> deferred)
- dream --drain routes through the shared helper (5A)

* chore: bump version and changelog (v0.42.12.0)

#1685 brain-health-as-solved-problem: cause-ranked doctor, worker_oom_loop
line, per-source auto-drain, pool-reap health. Layers on #1678/#1735.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* docs(todos): file #1685 GAP E + remote-path follow-ups (v0.42.12.0)

* fix(#1685): pre-landing review — multi-source auto-drain, honest pool-reap signal, lock-renewal reap labeling

- autopilot: drop maxWaiting (coalesces by name+queue not source → only one source drained + cap over-count); pre-check idempotency key so only genuinely-new sources submit+count
- pool_reap_health: fail on reconnect FAILURES (the real signal), not reaps>0&&failures>0 (false causality when a recovered reap + unrelated failure co-occur)
- lock-renewal-tick threads its triggering error to reconnect() so a CONNECTION_ENDED pooler reap is labeled reap_detected not reconnect_other (pool_reap_health now fires for the #1678 incident path)

* chore: re-version v0.42.12.0 → v0.42.16.0 (#1685)

Slot collision avoidance per queue.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* docs: restore slim CLAUDE.md + move #1685 entries to KEY_FILES.md (fix check:doc-history)

The master merge wrongly kept the pre-restructure 577KB CLAUDE.md; the
check:doc-history guard caps it at 60KB. Take master's slim CLAUDE.md and
record the #1685 files (doctor-cause-rank, pool-recovery-audit, worker_oom_loop
+ pool_reap_health checks, auto-drain, 5A helper) as current-state prose in
docs/architecture/KEY_FILES.md (no release markers). llms regenerated.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-03 07:27:34 -07:00

78 lines
3.1 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// #1685 GAP B — pool_reap_health doctor check.
//
// computePoolReapHealthCheck only touches engine.kind + the pool-recovery audit
// (filesystem), so a minimal `{ kind: 'postgres' }` stub drives it hermetically.
import { describe, expect, test, beforeEach, afterEach } from 'bun:test';
import * as fs from 'fs';
import * as os from 'os';
import * as path from 'path';
import { withEnv } from './helpers/with-env.ts';
import { logPoolRecovery } from '../src/core/audit/pool-recovery-audit.ts';
import { computePoolReapHealthCheck } from '../src/commands/doctor.ts';
// eslint-disable-next-line @typescript-eslint/no-explicit-any
const pg = { kind: 'postgres' } as any;
// eslint-disable-next-line @typescript-eslint/no-explicit-any
const pglite = { kind: 'pglite' } as any;
let tmpDir: string;
beforeEach(() => {
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'pool-reap-health-'));
});
afterEach(() => {
try { fs.rmSync(tmpDir, { recursive: true, force: true }); } catch { /* best-effort */ }
});
describe('computePoolReapHealthCheck', () => {
test('null on PGLite (no pool) and on null engine', async () => {
expect(await computePoolReapHealthCheck(pglite)).toBeNull();
expect(await computePoolReapHealthCheck(null)).toBeNull();
});
test('fail when reconnect failed (reconnect is throwing)', async () => {
await withEnv({ GBRAIN_AUDIT_DIR: tmpDir }, async () => {
logPoolRecovery('reap_detected');
logPoolRecovery('reconnect_failed', new Error('EHOSTUNREACH'));
const c = await computePoolReapHealthCheck(pg);
expect(c?.status).toBe('fail');
expect(c?.message).toContain('reconnect is throwing');
expect(c?.name).toBe('pool_reap_health');
});
});
// CODEX impl review #3: the fail trigger is the reconnect FAILURES themselves
// (reconnect throwing is the real, actionable problem), NOT a fabricated
// reap→failure causal link. A reconnect_failed with zero reaps still fails.
test('fail on reconnect failure even with zero reaps (no false causality)', async () => {
await withEnv({ GBRAIN_AUDIT_DIR: tmpDir }, async () => {
logPoolRecovery('reconnect_failed', new Error('password authentication failed'));
const c = await computePoolReapHealthCheck(pg);
expect(c?.status).toBe('fail');
expect(c?.message).toContain('0 pooler reap(s) detected');
expect(c?.message).not.toContain('not auto-recovering');
});
});
test('warn on pooler thrash (>=10 reaps all recovered)', async () => {
await withEnv({ GBRAIN_AUDIT_DIR: tmpDir }, async () => {
for (let i = 0; i < 12; i++) {
logPoolRecovery('reap_detected');
logPoolRecovery('reconnect_succeeded');
}
const c = await computePoolReapHealthCheck(pg);
expect(c?.status).toBe('warn');
expect(c?.message).toContain('12×');
});
});
test('null (quiet) when a few reaps all recovered', async () => {
await withEnv({ GBRAIN_AUDIT_DIR: tmpDir }, async () => {
logPoolRecovery('reap_detected');
logPoolRecovery('reconnect_succeeded');
const c = await computePoolReapHealthCheck(pg);
expect(c).toBeNull();
});
});
});