mirror of
https://github.com/garrytan/gbrain.git
synced 2026-07-27 22:15:33 +00:00
* fix(jobs): reap stale dead-holder cycle/sync locks (#1972) A crashed sync (OOM, recycle, SIGKILL) stranded its gbrain_cycle_locks row until something contended for it — reclaim was on-contention only. Add a host-scoped background reaper: reapDeadHolderLocks deletes locks whose holder PID is provably dead on this host, scoped to the gbrain-sync:*/gbrain-cycle* namespaces only (never elections/supervisor/reindex), with a snapshot-matched delete (date_trunc on acquired_at) that is TOCTOU-safe against PID reuse. Reuses isHolderDeadLocally (same-host + ESRCH + 60s grace). doctor --fix now auto-reaps for no-autopilot brains. DRY: selectLockRows + shared mapper now back inspectLock + listStaleLocks (killed the triplication). * fix(db): bound pool disconnect so teardown can't eat CLI output (#1972) pool.end() against PgBouncer transaction-mode never drained, so disconnect blocked until the CLI's 10s force-exit fired and process.exit()'d mid-write, truncating stdout (e.g. #1959's relational query returned empty). Add a gbrain-owned endPoolBounded(pool): Promise.race of pool.end({timeout}) against a hard timer, so teardown is bounded regardless of what postgres.js does and is testable. connection-manager ends its direct + read pools concurrently so the per-pool bounds don't stack. PGLite disconnect is unaffected. * fix(cycle): complete cooperative-abort coverage + wire lock reaper (#1972) v0.42.29 made only the embed phase honor the abort signal; a 24h pull still showed force-evicts from a long non-embed phase ignoring it. Thread the signal into every cycle-reachable long loop: extract (extractForSlugs + the full-walk extractLinksFromDir/extractTimelineFromDir), extract_facts (per-page loop + embed signal + the phantom-redirect 30s lock-retry), and consolidate's bucket loop. Add a terminal abort check so an aborted cycle never stamps last_full_cycle_at as a completed run (Codex #9). lint now yields + checks abort every 200 pages (it's synchronous; the yield is what lets the signal land). New phase-duration force-evict attribution log names any phase that crosses the 30s deadline. Wire reapDeadHolderLocks at cycle start. * chore: bump version and changelog (v0.42.38.0) #1972 — stale-lock reaper, bounded pool disconnect, and complete cooperative-abort coverage across cycle phases. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * docs(key-files): current-state for the #1972 reaper, bounded disconnect, abort coverage document-release: update db-lock.ts (reapDeadHolderLocks + selectLockRows DRY), db.ts (endPoolBounded), and abort-check.ts (coverage now spans extract/ extract_facts/consolidate/lint + terminal guard) entries to current truth. * test(isolation): fix shard-order flakes exposed by #1972's new test files Adding 3 new test files reshuffled the hash-based shards, exposing two pre-existing test-isolation bugs: - cycle-consolidate.test.ts assumed the global legacy-embedding preload's 1536-d gateway config still held at initSchema, but a co-sharded test that calls resetGateway() in teardown nulls it, so initSchema fell back to the 1280-d default and built a halfvec(1280) facts column its 1536-d fixtures can't fill. Re-pin the legacy OpenAI/1536 config in beforeAll (the pattern legacy-embedding-preload.ts documents for 1536-d fixture tests). - db-lock-heartbeat-takeover.test.ts (merged from master's #1794) mutated process.env.GBRAIN_LOCK_STEAL_GRACE_SECONDS raw, tripping check:test-isolation rule R1. Convert to withEnv(). --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
158 lines
6.0 KiB
TypeScript
158 lines
6.0 KiB
TypeScript
/**
|
|
* #1794 — heartbeat-aware lock takeover + direct-pool refresh.
|
|
*
|
|
* The lock-thrash fix: a holder that refreshed within the steal-grace window is
|
|
* NOT stolen even if its TTL lapsed (starved-but-alive), while a holder that
|
|
* stopped refreshing past the grace IS stolen. These tests isolate the ON
|
|
* CONFLICT grace logic by holding with `process.pid` (alive) so the dead-pid
|
|
* auto-takeover path can't fire — a successful steal therefore proves the
|
|
* grace/last_refreshed_at predicate did it.
|
|
*
|
|
* Plus the commit-8 causal-mechanism check: setImmediate yields let a
|
|
* setInterval tick fire during a busy loop (the reason the import loop's
|
|
* maybeYield keeps the refresh heartbeat alive).
|
|
*/
|
|
|
|
import { describe, test, expect, beforeAll, afterAll, beforeEach } from 'bun:test';
|
|
import { hostname } from 'os';
|
|
import { PGLiteEngine } from '../src/core/pglite-engine.ts';
|
|
import { withEnv } from './helpers/with-env.ts';
|
|
import {
|
|
tryAcquireDbLock,
|
|
resolveStealGraceSeconds,
|
|
DEFAULT_STEAL_GRACE_SECONDS,
|
|
} from '../src/core/db-lock.ts';
|
|
|
|
let engine: PGLiteEngine;
|
|
|
|
beforeAll(async () => {
|
|
engine = new PGLiteEngine();
|
|
await engine.connect({ database_url: '' });
|
|
await engine.initSchema();
|
|
});
|
|
|
|
afterAll(async () => {
|
|
await engine.disconnect();
|
|
});
|
|
|
|
beforeEach(async () => {
|
|
await engine.executeRaw(`DELETE FROM gbrain_cycle_locks WHERE id LIKE 'test-hb-%'`);
|
|
});
|
|
|
|
const LOCAL = hostname();
|
|
|
|
/**
|
|
* Seed a TTL-EXPIRED lock held by an ALIVE pid (process.pid), with a chosen
|
|
* last_refreshed_at age (or NULL). Alive holder → dead-pid auto-takeover can't
|
|
* fire, so the only reclaim path is the ON CONFLICT grace predicate.
|
|
*/
|
|
async function seedExpiredLock(id: string, refreshedSecondsAgo: number | null): Promise<void> {
|
|
if (refreshedSecondsAgo === null) {
|
|
await engine.executeRaw(
|
|
`INSERT INTO gbrain_cycle_locks (id, holder_pid, holder_host, acquired_at, ttl_expires_at, last_refreshed_at)
|
|
VALUES ($1, $2, $3, NOW() - INTERVAL '1 hour', NOW() - INTERVAL '5 minutes', NULL)`,
|
|
[id, process.pid, LOCAL],
|
|
);
|
|
} else {
|
|
await engine.executeRaw(
|
|
`INSERT INTO gbrain_cycle_locks (id, holder_pid, holder_host, acquired_at, ttl_expires_at, last_refreshed_at)
|
|
VALUES ($1, $2, $3, NOW() - INTERVAL '1 hour', NOW() - INTERVAL '5 minutes', NOW() - ($4 || ' seconds')::interval)`,
|
|
[id, process.pid, LOCAL, String(refreshedSecondsAgo)],
|
|
);
|
|
}
|
|
}
|
|
|
|
async function refreshedAge(id: string): Promise<Date | null> {
|
|
const rows = await engine.executeRaw<{ last_refreshed_at: string | null }>(
|
|
`SELECT last_refreshed_at FROM gbrain_cycle_locks WHERE id = $1`,
|
|
[id],
|
|
);
|
|
const v = rows[0]?.last_refreshed_at ?? null;
|
|
return v ? new Date(v) : null;
|
|
}
|
|
|
|
describe('resolveStealGraceSeconds', () => {
|
|
test('derives ~2 refresh ticks from the TTL (30min → 600s)', () => {
|
|
// refresh ~ttl/6 = 5min = 300s; *2 = 600s.
|
|
expect(resolveStealGraceSeconds(30)).toBe(DEFAULT_STEAL_GRACE_SECONDS);
|
|
expect(resolveStealGraceSeconds(30)).toBe(600);
|
|
});
|
|
|
|
test('floors at 60s for tiny TTLs', () => {
|
|
expect(resolveStealGraceSeconds(1)).toBe(60);
|
|
});
|
|
|
|
test('env override wins', async () => {
|
|
await withEnv({ GBRAIN_LOCK_STEAL_GRACE_SECONDS: '123' }, async () => {
|
|
expect(resolveStealGraceSeconds(30)).toBe(123);
|
|
});
|
|
});
|
|
|
|
test('bad env override falls back to derived', async () => {
|
|
await withEnv({ GBRAIN_LOCK_STEAL_GRACE_SECONDS: 'nope' }, async () => {
|
|
expect(resolveStealGraceSeconds(30)).toBe(600);
|
|
});
|
|
});
|
|
});
|
|
|
|
describe('heartbeat-aware takeover (ON CONFLICT grace predicate)', () => {
|
|
test('FRESH holder is NOT stolen even with an expired TTL', async () => {
|
|
// ttl expired 5min ago, but refreshed 1s ago → inside the 600s grace.
|
|
await seedExpiredLock('test-hb-fresh', 1);
|
|
const handle = await tryAcquireDbLock(engine, 'test-hb-fresh', 30);
|
|
expect(handle).toBeNull();
|
|
});
|
|
|
|
test('STALE-refresh holder IS stolen (refresh older than the grace)', async () => {
|
|
// ttl expired AND last refresh 20min ago (> 600s grace) → stealable, even
|
|
// though the holder pid is alive (proves the grace path, not auto-takeover).
|
|
await seedExpiredLock('test-hb-stale', 1200);
|
|
const handle = await tryAcquireDbLock(engine, 'test-hb-stale', 30);
|
|
expect(handle).not.toBeNull();
|
|
await handle!.release();
|
|
});
|
|
|
|
test('NULL last_refreshed_at (pre-v98 row) IS stolen on TTL expiry', async () => {
|
|
await seedExpiredLock('test-hb-null', null);
|
|
const handle = await tryAcquireDbLock(engine, 'test-hb-null', 30);
|
|
expect(handle).not.toBeNull();
|
|
await handle!.release();
|
|
});
|
|
|
|
test('a reclaimed handle refresh() bumps last_refreshed_at (direct pool path)', async () => {
|
|
await seedExpiredLock('test-hb-bump', 1200);
|
|
const handle = await tryAcquireDbLock(engine, 'test-hb-bump', 30);
|
|
expect(handle).not.toBeNull();
|
|
const before = await refreshedAge('test-hb-bump');
|
|
// Small real delay so NOW() advances measurably between acquire and refresh.
|
|
await new Promise((r) => setTimeout(r, 20));
|
|
await handle!.refresh();
|
|
const after = await refreshedAge('test-hb-bump');
|
|
expect(before).not.toBeNull();
|
|
expect(after).not.toBeNull();
|
|
expect(after!.getTime()).toBeGreaterThan(before!.getTime());
|
|
await handle!.release();
|
|
});
|
|
});
|
|
|
|
describe('event-loop yield keeps timers alive (commit 8 mechanism)', () => {
|
|
test('setTimeout(0) yields let a setInterval heartbeat fire during a busy loop', async () => {
|
|
let ticks = 0;
|
|
const iv = setInterval(() => { ticks++; }, 2);
|
|
try {
|
|
// Mirror the import loop's maybeYield: setTimeout(0) enters the timers
|
|
// phase, so the setInterval heartbeat can fire mid-loop. (A setImmediate
|
|
// loop starves the timers phase in Bun — the reason maybeYield uses
|
|
// setTimeout, not setImmediate.) Bound by wall-clock so the 2ms interval
|
|
// has real time to fire.
|
|
const start = Date.now();
|
|
while (Date.now() - start < 40) {
|
|
await new Promise<void>((r) => setTimeout(r, 0));
|
|
}
|
|
} finally {
|
|
clearInterval(iv);
|
|
}
|
|
expect(ticks).toBeGreaterThan(0);
|
|
});
|
|
});
|