From 00470e13a078490ff27bbee5f555b50aec5167dc Mon Sep 17 00:00:00 2001 From: jbarol Date: Wed, 10 Jun 2026 13:29:06 -0700 Subject: [PATCH] fix(cli): arm the disconnect hard-deadline at teardown entry, not before the op body MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 10s force-exit timer in the shared-op dispatch was armed BEFORE the try block, so any op whose handler ran past 10s wall-clock was killed mid-flight with process.exit(0) and zero stdout. On a slow Postgres pooler (6-10s per fresh connection) a healthy `gbrain search` was force-exited every time — an empty 'success' indistinguishable from no results. The v0.42.20.0 exitCode honor can't help: a mid-op kill fires before any error path sets exitCode. Move the arming into the finally (teardown entry), matching the fall-through owner-disconnect site later in main(): the timer still bounds a hung drain/disconnect (the C13 contract) but can no longer kill a slow-but-progressing op. Verified on a transaction-pooler Supabase brain: search went from 0 bytes/exit 0 at 10s to real results at ~21s. --- src/cli.ts | 50 ++++++++++++++++++++++++++++++-------------------- 1 file changed, 30 insertions(+), 20 deletions(-) diff --git a/src/cli.ts b/src/cli.ts index 09d187cea..4d3ce7183 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -360,26 +360,19 @@ async function main() { // // Defense-in-depth (adversarial-review C13): `engine.disconnect()` itself // can hang on PGLite (db.close() or releaseLock racing OS-level FS state). - // Install an unref'd setTimeout hard-exit fallback BEFORE entering the - // try/catch/finally so a hung disconnect cannot defeat the force-exit - // contract. Daemons (`serve`) are excluded so they stay alive. + // The unref'd hard-exit fallback is armed inside the `finally` below, so it + // bounds ONLY the teardown phase (drain + disconnect) — the same placement + // as the fall-through owner-disconnect site later in this file. It used to + // be armed HERE, before the try, which silently killed any op whose BODY ran + // past the deadline: on a slow Postgres pooler (6-10s per fresh connection) + // a healthy `gbrain search` was force-exited mid-handler with code 0 and + // ZERO stdout — an empty "success" indistinguishable from no results. The + // exitCode honor (v0.42.20.0) can't help there: a mid-op kill fires before + // any error path sets exitCode. Op-body wallclock bounds belong to the + // read-only dispatch's withTimeout wrap, not this teardown backstop. + // Daemons (`serve`) are excluded so they stay alive. const DISCONNECT_HARD_DEADLINE_MS = 10_000; let forceExitTimer: ReturnType | undefined; - if (shouldForceExitAfterMain()) { - forceExitTimer = setTimeout(() => { - console.warn( - `[cli] engine.disconnect() did not return within ${DISCONNECT_HARD_DEADLINE_MS}ms — force-exiting`, - ); - // v0.42.20.0 (codex): honor an exit code an errored op already set — - // a bare process.exit(0) here would mask a failed op as success if the - // drain/disconnect then hangs. - process.exit(process.exitCode ?? 0); - }, DISCONNECT_HARD_DEADLINE_MS); - // unref so the timer itself doesn't keep the event loop alive — only - // the actual pending work (PGLite WASM handle) does. Without unref, - // we'd block a clean exit by 10s on every successful CLI run. - forceExitTimer.unref?.(); - } try { const ctx = await makeContext(engine, params); @@ -412,8 +405,25 @@ async function main() { // DB logIngest gets the freshest live-engine window. 1s per-sink timeout: // read paths with no pending work pay the ~0ms fast path; capture/import // that DO enqueue pay up to 1s (+ facts shutdown grace) while in-flight - // Haiku finishes. The unref'd hard-deadline timer above is the backstop if - // disconnect or a lingering socket keeps Bun's loop alive. + // Haiku finishes. The unref'd hard-deadline timer armed here is the + // backstop if disconnect or a lingering socket keeps Bun's loop alive — + // armed at teardown entry (NOT before the op body; see the C13 comment + // above) so a slow-but-progressing op handler is never killed mid-flight. + if (shouldForceExitAfterMain()) { + forceExitTimer = setTimeout(() => { + console.warn( + `[cli] engine.disconnect() did not return within ${DISCONNECT_HARD_DEADLINE_MS}ms — force-exiting`, + ); + // v0.42.20.0 (codex): honor an exit code an errored op already set — + // a bare process.exit(0) here would mask a failed op as success if the + // drain/disconnect then hangs. + process.exit(process.exitCode ?? 0); + }, DISCONNECT_HARD_DEADLINE_MS); + // unref so the timer itself doesn't keep the event loop alive — only + // the actual pending work (PGLite WASM handle) does. Without unref, + // we'd block a clean exit by 10s on every successful CLI run. + forceExitTimer.unref?.(); + } await drainAllBackgroundWorkForCliExit({ timeoutMs: 1000 }); await engine.disconnect(); if (forceExitTimer) clearTimeout(forceExitTimer);