From df22a3db644fd6c41d8f8200c04a70c7e8f5da5b Mon Sep 17 00:00:00 2001 From: default Date: Thu, 9 Apr 2026 07:18:09 +0000 Subject: [PATCH] fix(kernel): preserve cron jobs across hand reactivation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bug: in activate_hand(), kill_agent() is called on the existing agent BEFORE the new agent is spawned. kill_agent() invokes cron_scheduler.remove_agent_jobs() which deletes all cron jobs from memory AND persists [] to cron_jobs.json. The reassign_agent_jobs() call further down was meant to migrate jobs from old to new (per #461), but it always runs as a no-op because the jobs are already gone — the order of operations defeats the fix. Symptom: every daemon restart silently destroys cron jobs for hand-style agents. cron_jobs.json is rewritten as []. /api/cron/jobs returns empty. No error message. Fix: snapshot the cron jobs into a local Vec BEFORE kill_agent (same pattern as saved_triggers above), then re-add them under the new agent_id AFTER spawn_agent_with_parent. Runtime state (next_run, last_run) is reset so jobs get a fresh start. The existing reassign_agent_jobs() block is kept as a defensive safety net but is now redundant in the common path. Verified with cargo check -p openfang-kernel --lib (clean compile, no warnings). Co-Authored-By: Claude Opus 4.6 (1M context) --- crates/openfang-kernel/src/kernel.rs | 44 ++++++++++++++++++++++++++-- 1 file changed, 41 insertions(+), 3 deletions(-) diff --git a/crates/openfang-kernel/src/kernel.rs b/crates/openfang-kernel/src/kernel.rs index cc070a15..d914370e 100644 --- a/crates/openfang-kernel/src/kernel.rs +++ b/crates/openfang-kernel/src/kernel.rs @@ -3490,6 +3490,15 @@ impl OpenFangKernel { let saved_triggers = old_agent_id .map(|id| self.triggers.take_agent_triggers(id)) .unwrap_or_default(); + // Snapshot cron jobs before kill_agent destroys them. kill_agent calls + // remove_agent_jobs() which deletes the jobs from memory and persists + // an empty cron_jobs.json to disk. The reassign_agent_jobs() call below + // would always be a no-op without this snapshot — same pattern as + // saved_triggers above. Fixes the silent loss of cron jobs across + // every daemon restart for hand-style agents. + let saved_crons: Vec = old_agent_id + .map(|id| self.cron_scheduler.list_jobs(id)) + .unwrap_or_default(); if let Some(old) = existing { info!(agent = %old.name, id = %old.id, "Removing existing hand agent for reactivation"); let _ = self.kill_agent(old.id); @@ -3520,9 +3529,38 @@ impl OpenFangKernel { } } - // Migrate cron jobs from old agent to new agent so they survive restarts. - // Without this, persisted cron jobs would reference the stale old UUID - // and fail silently (issue #461). + // Restore cron jobs that were snapshotted before kill_agent. They're + // re-added under the new agent_id (which equals old.id when fixed_id is + // derived from hand_id, but be explicit). Runtime state is reset so + // jobs get a fresh start. + if !saved_crons.is_empty() { + let mut restored = 0usize; + for mut job in saved_crons { + job.agent_id = agent_id; + job.next_run = None; + job.last_run = None; + if self.cron_scheduler.add_job(job, false).is_ok() { + restored += 1; + } + } + if restored > 0 { + info!( + agent = %agent_id, + restored, + "Restored cron jobs after hand reactivation" + ); + if let Err(e) = self.cron_scheduler.persist() { + warn!("Failed to persist cron jobs after restoration: {e}"); + } + } + } + + // Belt-and-braces: also reassign any jobs that somehow still reference + // the old UUID (shouldn't happen after the snapshot/restore above, but + // kept as a safety net for edge cases like out-of-band cron creation + // between kill and respawn). Removed reassign as primary path because + // kill_agent's remove_agent_jobs always wipes saved_crons before this + // could fire — see issue with #461's original fix. if let Some(old_id) = old_agent_id { let migrated = self.cron_scheduler.reassign_agent_jobs(old_id, agent_id); if migrated > 0 {