mirror of
https://github.com/tinyhumansai/openhuman.git
synced 2026-07-27 21:08:00 +00:00
342 lines
14 KiB
Rust
342 lines
14 KiB
Rust
//! Integration tests for the Ollama daemon lifecycle contract (issue #1622 / pr #1638).
|
|
//!
|
|
//! These tests exercise the ownership model through the public `LocalAiService`
|
|
//! API without launching a real Ollama binary. Three flows are covered:
|
|
//!
|
|
//! 1. **Owned-spawn → graceful exit**: `shutdown_owned_ollama` kills the child
|
|
//! process and clears the on-disk spawn marker.
|
|
//! 2. **External adoption → graceful exit**: when the daemon on `:11434` was not
|
|
//! spawned by openhuman (`owned_ollama == None`), `shutdown_owned_ollama` is
|
|
//! a no-op; a substitute long-running process stands in for the "external"
|
|
//! daemon and survives the call.
|
|
//! 3. **Crash recovery (stale marker + dead PID)**: `diagnostics` completes
|
|
//! successfully even when a leftover marker file references a PID that is no
|
|
//! longer alive, demonstrating that the reclaim guard in
|
|
//! `reclaim_orphan_if_ours` (called inside the production bootstrap) handles
|
|
//! the dead-marker case gracefully.
|
|
//!
|
|
//! # What requires a real Ollama binary
|
|
//!
|
|
//! Flows that exercise `start_and_wait_for_server` (i.e. the actual daemon
|
|
//! spawn loop with health polling) cannot be fully tested without a live
|
|
//! `ollama serve` process. The three scenarios above are covered at the
|
|
//! helper/shutdown level which is both necessary and sufficient to lock
|
|
//! the ownership contract. The spawn loop itself is tested indirectly via
|
|
//! `ensure_ollama_server_requires_external_runtime_when_unreachable` in
|
|
//! `ollama_admin_tests.rs`.
|
|
|
|
use std::sync::{Mutex, OnceLock};
|
|
|
|
use openhuman_core::openhuman::config::Config;
|
|
use openhuman_core::openhuman::inference::local::LocalAiService;
|
|
|
|
// ── Environment serialization lock ───────────────────────────────────────────
|
|
//
|
|
// Each test temporarily sets OPENHUMAN_WORKSPACE to redirect the marker path
|
|
// away from ~/.openhuman/. The mutex prevents parallel tests from stomping
|
|
// each other's env state.
|
|
|
|
static ENV_LOCK: OnceLock<Mutex<()>> = OnceLock::new();
|
|
|
|
fn env_lock() -> std::sync::MutexGuard<'static, ()> {
|
|
let m = ENV_LOCK.get_or_init(|| Mutex::new(()));
|
|
match m.lock() {
|
|
Ok(g) => g,
|
|
Err(p) => p.into_inner(),
|
|
}
|
|
}
|
|
|
|
// ── RAII env-var guard ────────────────────────────────────────────────────────
|
|
//
|
|
// Restores the previous env-var value (or removes it) when dropped.
|
|
// This ensures cleanup runs even if an assertion panics early, preventing
|
|
// env-var leakage that could destabilise subsequent tests.
|
|
|
|
struct EnvVarGuard {
|
|
key: &'static str,
|
|
prev: Option<std::ffi::OsString>,
|
|
}
|
|
|
|
impl EnvVarGuard {
|
|
fn set(key: &'static str, value: &std::ffi::OsStr) -> Self {
|
|
let prev = std::env::var_os(key);
|
|
std::env::set_var(key, value);
|
|
Self { key, prev }
|
|
}
|
|
}
|
|
|
|
impl Drop for EnvVarGuard {
|
|
fn drop(&mut self) {
|
|
match &self.prev {
|
|
Some(v) => std::env::set_var(self.key, v),
|
|
None => std::env::remove_var(self.key),
|
|
}
|
|
}
|
|
}
|
|
|
|
// ── Marker path helper ────────────────────────────────────────────────────────
|
|
//
|
|
// Mirrors the logic of `paths::ollama_spawn_marker_path`: when
|
|
// OPENHUMAN_WORKSPACE is set, the marker lives under config_path.parent()
|
|
// (i.e. the directory containing config.toml).
|
|
|
|
fn marker_path_for(config: &Config) -> std::path::PathBuf {
|
|
config
|
|
.config_path
|
|
.parent()
|
|
.expect("config_path must have a parent")
|
|
.join("local-ai")
|
|
.join("ollama.spawn")
|
|
}
|
|
|
|
/// Write a minimal spawn marker JSON directly (avoids needing pub(crate) helpers).
|
|
fn write_marker(path: &std::path::Path, pid: u32) {
|
|
if let Some(parent) = path.parent() {
|
|
std::fs::create_dir_all(parent).expect("create marker dir");
|
|
}
|
|
let json = format!(
|
|
r#"{{"pid":{pid},"started_at_unix":1700000000,"binary_path":"test-stub","openhuman_pid":{my_pid}}}"#,
|
|
pid = pid,
|
|
my_pid = std::process::id(),
|
|
);
|
|
let tmp = path.with_extension("spawn.tmp");
|
|
std::fs::write(&tmp, &json).expect("write marker tmp");
|
|
std::fs::rename(&tmp, path).expect("rename marker");
|
|
}
|
|
|
|
// ── Test 1: owned-spawn lifecycle — graceful exit ─────────────────────────────
|
|
|
|
/// When openhuman spawned Ollama itself (owned_ollama is Some), calling
|
|
/// `shutdown_owned_ollama` must:
|
|
/// - kill the owned child process,
|
|
/// - clear the on-disk spawn marker.
|
|
#[tokio::test]
|
|
async fn owned_spawn_shutdown_kills_child_and_clears_marker() {
|
|
let _guard = env_lock();
|
|
let tmp = tempfile::tempdir().unwrap();
|
|
|
|
// Set OPENHUMAN_WORKSPACE so the marker path resolves under our tempdir.
|
|
// EnvVarGuard restores the previous value on drop — even if an assertion panics.
|
|
let _ws_guard = EnvVarGuard::set("OPENHUMAN_WORKSPACE", tmp.path().as_os_str());
|
|
let mut config = Config::default();
|
|
config.workspace_dir = tmp.path().to_path_buf();
|
|
config.config_path = tmp.path().join("config.toml");
|
|
|
|
let service = LocalAiService::new(&config);
|
|
|
|
// Spawn a long-running stub process (acts as the "owned ollama" child).
|
|
let mut cmd = if cfg!(windows) {
|
|
let mut c = tokio::process::Command::new("powershell");
|
|
c.args(["-NoProfile", "-Command", "Start-Sleep -Seconds 30"]);
|
|
c
|
|
} else {
|
|
let mut c = tokio::process::Command::new("sleep");
|
|
c.arg("30");
|
|
c
|
|
};
|
|
cmd.stdout(std::process::Stdio::null())
|
|
.stderr(std::process::Stdio::null());
|
|
let child = cmd.spawn().expect("spawn stub child");
|
|
let child_pid = child.id().expect("child pid");
|
|
|
|
// Inject it as the owned child (mirrors what start_and_wait_for_server does).
|
|
service.inject_owned_ollama(child);
|
|
|
|
// Write the spawn marker (mirrors what start_and_wait_for_server does after
|
|
// the daemon health poll succeeds).
|
|
let marker_path = marker_path_for(&config);
|
|
write_marker(&marker_path, child_pid);
|
|
assert!(
|
|
marker_path.exists(),
|
|
"marker must be on disk before shutdown"
|
|
);
|
|
|
|
// Exercise the public shutdown hook.
|
|
service.shutdown_owned_ollama(&config).await;
|
|
|
|
// Marker must be gone.
|
|
assert!(
|
|
!marker_path.exists(),
|
|
"shutdown_owned_ollama must remove the spawn marker"
|
|
);
|
|
|
|
// Owned handle must be cleared.
|
|
assert!(
|
|
!service.has_owned_ollama(),
|
|
"owned_ollama must be None after shutdown"
|
|
);
|
|
|
|
// The child process must be dead within a brief settle window.
|
|
let mut still_alive = true;
|
|
for _ in 0..40 {
|
|
let mut sys = sysinfo::System::new();
|
|
let target = sysinfo::Pid::from_u32(child_pid);
|
|
sys.refresh_processes(sysinfo::ProcessesToUpdate::Some(&[target]), true);
|
|
if sys.process(target).is_none() {
|
|
still_alive = false;
|
|
break;
|
|
}
|
|
tokio::time::sleep(std::time::Duration::from_millis(50)).await;
|
|
}
|
|
assert!(
|
|
!still_alive,
|
|
"child pid {child_pid} should be dead after shutdown_owned_ollama"
|
|
);
|
|
// _ws_guard restores OPENHUMAN_WORKSPACE when it drops.
|
|
}
|
|
|
|
// ── Test 2: external adoption — shutdown leaves external daemon untouched ─────
|
|
|
|
/// When openhuman adopted an external Ollama (owned_ollama is None),
|
|
/// `shutdown_owned_ollama` must be a no-op: the external daemon must not be
|
|
/// killed. We simulate the external daemon with a second stub process whose
|
|
/// PID we track directly and assert is still alive after the call.
|
|
#[tokio::test]
|
|
async fn external_adoption_shutdown_leaves_external_process_running() {
|
|
let _guard = env_lock();
|
|
let tmp = tempfile::tempdir().unwrap();
|
|
|
|
let _ws_guard = EnvVarGuard::set("OPENHUMAN_WORKSPACE", tmp.path().as_os_str());
|
|
let mut config = Config::default();
|
|
config.workspace_dir = tmp.path().to_path_buf();
|
|
config.config_path = tmp.path().join("config.toml");
|
|
|
|
let service = LocalAiService::new(&config);
|
|
|
|
// `owned_ollama` starts as None — external daemon was adopted, not spawned.
|
|
assert!(
|
|
!service.has_owned_ollama(),
|
|
"fresh service must have no owned child"
|
|
);
|
|
|
|
// Spawn a separate stub to represent the "external" daemon so we can
|
|
// check it is NOT killed by shutdown. Keep it alive for >2 s.
|
|
let mut ext_cmd = if cfg!(windows) {
|
|
let mut c = tokio::process::Command::new("powershell");
|
|
c.args(["-NoProfile", "-Command", "Start-Sleep -Seconds 30"]);
|
|
c
|
|
} else {
|
|
let mut c = tokio::process::Command::new("sleep");
|
|
c.arg("30");
|
|
c
|
|
};
|
|
ext_cmd
|
|
.stdout(std::process::Stdio::null())
|
|
.stderr(std::process::Stdio::null());
|
|
let mut ext_child = ext_cmd.spawn().expect("spawn external stub");
|
|
let ext_pid = ext_child.id().expect("external stub pid");
|
|
|
|
// No marker file — we never wrote one because we adopted, not spawned.
|
|
let marker_path = marker_path_for(&config);
|
|
|
|
// Call shutdown with no owned child.
|
|
service.shutdown_owned_ollama(&config).await;
|
|
|
|
// Marker was never written, so it remains absent.
|
|
assert!(
|
|
!marker_path.exists(),
|
|
"no marker should appear when adopting an external daemon"
|
|
);
|
|
|
|
// The external stub must still be running.
|
|
let still_alive = {
|
|
let mut sys = sysinfo::System::new();
|
|
let target = sysinfo::Pid::from_u32(ext_pid);
|
|
sys.refresh_processes(sysinfo::ProcessesToUpdate::Some(&[target]), true);
|
|
sys.process(target).is_some()
|
|
};
|
|
assert!(
|
|
still_alive,
|
|
"external daemon pid {ext_pid} must still be running after no-op shutdown"
|
|
);
|
|
|
|
// Clean up the external stub ourselves.
|
|
let _ = ext_child.kill().await;
|
|
let _ = ext_child.wait().await;
|
|
// _ws_guard restores OPENHUMAN_WORKSPACE when it drops.
|
|
}
|
|
|
|
// ── Test 3: crash recovery — stale marker with dead PID ───────────────────────
|
|
|
|
/// Simulate a previous crash: a marker file exists on disk referencing a PID
|
|
/// that is no longer alive. On the next launch the service must handle this
|
|
/// gracefully. We test via `diagnostics` (a public, purely-read call that
|
|
/// triggers `reclaim_orphan_if_ours` indirectly through the bootstrap path
|
|
/// when a real server is present). Here we assert that `diagnostics` succeeds
|
|
/// even with a stale dead-PID marker present and without a live Ollama server —
|
|
/// the call must not panic or propagate the stale marker as an error.
|
|
///
|
|
/// NOTE: `reclaim_orphan_if_ours` is invoked inside `start_and_wait_for_server`
|
|
/// (the full bootstrap path), which requires a real Ollama binary to be
|
|
/// reachable. We test the marker-handling invariant through the path available
|
|
/// without a binary: `diagnostics` simply reports that the server is not
|
|
/// reachable, while the stale marker on disk is harmless. The dead-marker
|
|
/// clearing branch is already exercised by the inline `spawn_marker` unit
|
|
/// tests in `ollama_admin_tests.rs` (`pid_is_alive_rejects_dead_pid` +
|
|
/// `reclaim_orphan_if_ours` logic). What we add here is an integration-level
|
|
/// confirmation that the overall service stays functional when a stale marker
|
|
/// is present.
|
|
#[tokio::test]
|
|
async fn crash_recovery_stale_marker_does_not_break_service() {
|
|
let _guard = env_lock();
|
|
let tmp = tempfile::tempdir().unwrap();
|
|
|
|
let _ws_guard = EnvVarGuard::set("OPENHUMAN_WORKSPACE", tmp.path().as_os_str());
|
|
// Redirect Ollama health checks to a dead port so no real daemon is needed.
|
|
let _ollama_url_guard = EnvVarGuard::set(
|
|
"OPENHUMAN_OLLAMA_BASE_URL",
|
|
std::ffi::OsStr::new("http://127.0.0.1:1"),
|
|
);
|
|
|
|
let mut config = Config::default();
|
|
config.workspace_dir = tmp.path().to_path_buf();
|
|
config.config_path = tmp.path().join("config.toml");
|
|
|
|
// Write a stale marker with a PID that was recycled from a short-lived child.
|
|
let zombie = if cfg!(windows) {
|
|
std::process::Command::new("cmd")
|
|
.args(["/C", "exit 0"])
|
|
.spawn()
|
|
.expect("spawn cmd /C exit")
|
|
} else {
|
|
std::process::Command::new("true")
|
|
.spawn()
|
|
.expect("spawn /usr/bin/true")
|
|
};
|
|
let dead_pid = zombie.id();
|
|
let mut zombie = zombie;
|
|
let _ = zombie.wait();
|
|
// Brief settle so the OS fully reaps the zombie before we write the marker.
|
|
std::thread::sleep(std::time::Duration::from_millis(250));
|
|
|
|
let marker_path = marker_path_for(&config);
|
|
write_marker(&marker_path, dead_pid);
|
|
assert!(
|
|
marker_path.exists(),
|
|
"stale marker must be present to simulate a crash"
|
|
);
|
|
|
|
// A freshly constructed service must not panic and diagnostics must succeed.
|
|
let service = LocalAiService::new(&config);
|
|
let diag = service
|
|
.diagnostics(&config)
|
|
.await
|
|
.expect("diagnostics must succeed even with a stale spawn marker");
|
|
|
|
// Without a live Ollama server, diagnostics reports not running.
|
|
assert_eq!(
|
|
diag["ollama_running"], false,
|
|
"ollama_running must be false when port is unreachable"
|
|
);
|
|
let issues = diag["issues"].as_array().cloned().unwrap_or_default();
|
|
assert!(
|
|
!issues.is_empty(),
|
|
"diagnostics must surface issues when server is unreachable"
|
|
);
|
|
|
|
// The stale marker on disk is harmless at this level — it is consumed
|
|
// only during the bootstrap path (start_and_wait_for_server). The test
|
|
// confirms the service remains operational despite it.
|
|
// _ws_guard and _ollama_url_guard restore the env vars when they drop.
|
|
}
|