diff --git a/app/src/components/settings/panels/local-model/ModelStatusSection.test.tsx b/app/src/components/settings/panels/local-model/ModelStatusSection.test.tsx index 03bfcffbd..f9ba1039b 100644 --- a/app/src/components/settings/panels/local-model/ModelStatusSection.test.tsx +++ b/app/src/components/settings/panels/local-model/ModelStatusSection.test.tsx @@ -328,4 +328,72 @@ describe('ModelStatusSection diagnostics', () => { expect(screen.queryByRole('button', { name: 'Set Path' })).toBeNull(); expect(screen.getByRole('link', { name: 'Ollama docs' })).toBeTruthy(); }); + + it('accepts a model that meets the context minimum', () => { + render( + + ); + expect(screen.getByText('8,192 ctx ✓')).toBeTruthy(); + }); + + it('rejects and visually flags a model below the context minimum', () => { + render( + + ); + expect(screen.getByText('2,048 ctx — below 8,192 min')).toBeTruthy(); + // Model name is rendered in the rejection (red) treatment. + const name = screen.getByTitle('tiny-embed:latest'); + expect(name.className).toContain('text-red-700'); + }); + + it('marks an unknown context window without rejecting it', () => { + render( + + ); + expect(screen.getByText('ctx unknown')).toBeTruthy(); + }); + + it('renders models with no eligibility (older core) without a badge', () => { + render( + + ); + expect(screen.getByText('legacy:latest')).toBeTruthy(); + expect(screen.queryByText(/ctx/)).toBeNull(); + }); }); diff --git a/app/src/components/settings/panels/local-model/ModelStatusSection.tsx b/app/src/components/settings/panels/local-model/ModelStatusSection.tsx index a057daaf6..05edcaca2 100644 --- a/app/src/components/settings/panels/local-model/ModelStatusSection.tsx +++ b/app/src/components/settings/panels/local-model/ModelStatusSection.tsx @@ -4,9 +4,49 @@ import type { LocalAiDiagnostics, LocalAiDownloadsProgress, LocalAiStatus, + ModelContextEligibility, RepairAction, } from '../../../../utils/tauriCommands'; +/** + * Badge rendering a model's context-window verdict against the memory + * layer minimum. `below_minimum` is a hard rejection (the memory pipeline + * would silently truncate and corrupt recall); `unknown` is a soft warning. + */ +const ContextEligibilityBadge = ({ + eligibility, +}: { + eligibility: ModelContextEligibility | null | undefined; +}) => { + if (!eligibility) return null; + const fmt = (n: number) => n.toLocaleString(); + if (eligibility.status === 'ok') { + return ( + + {fmt(eligibility.context_length)} ctx ✓ + + ); + } + if (eligibility.status === 'below_minimum') { + return ( + + {fmt(eligibility.context_length)} ctx — below {fmt(eligibility.required)} min + + ); + } + return ( + + ctx unknown + + ); +}; + interface ModelStatusSectionProps { status: LocalAiStatus | null; downloads: LocalAiDownloadsProgress | null; @@ -394,18 +434,34 @@ const ModelStatusSection = ({ {diagnostics.installed_models.length})
- {diagnostics.installed_models.map(m => ( -
- - {m.name} - - - {typeof m.size === 'number' ? formatBytes(m.size) : ''} - -
- ))} + {diagnostics.installed_models.map(m => { + const rejected = m.eligibility?.status === 'below_minimum'; + return ( +
+ + {m.name} + + + + + {typeof m.size === 'number' ? formatBytes(m.size) : ''} + + +
+ ); + })}
)} diff --git a/app/src/utils/tauriCommands/localAi.ts b/app/src/utils/tauriCommands/localAi.ts index bcae39a19..11f824836 100644 --- a/app/src/utils/tauriCommands/localAi.ts +++ b/app/src/utils/tauriCommands/localAi.ts @@ -180,17 +180,42 @@ export type RepairAction = | { action: 'start_server'; binary_path: string | null } | { action: 'pull_model'; model: string }; +/** + * Verdict for a model's native context window against the memory-layer + * minimum. Mirrors the Rust `ContextEligibility` enum (serde tagged by + * `status`). `below_minimum` means the model is rejected for memory-layer + * use; `unknown` means the context window could not be determined (not a + * hard rejection). + */ +export type ModelContextEligibility = + | { status: 'ok'; context_length: number } + | { status: 'below_minimum'; context_length: number; required: number } + | { status: 'unknown'; required: number }; + +export interface InstalledModelInfo { + name: string; + size?: number | null; + modified_at?: string | null; + /** Native context window in tokens, or null when `/api/show` didn't report it. */ + context_length?: number | null; + eligibility?: ModelContextEligibility | null; +} + export interface LocalAiDiagnostics { ollama_running: boolean; ollama_base_url: string; ollama_binary_path: string | null; vision_mode?: string; - installed_models: Array<{ name: string; size?: number | null; modified_at?: string | null }>; + installed_models: InstalledModelInfo[]; + /** Memory-layer minimum a model's context window must meet to be accepted. */ + context_requirement?: { min_context_tokens: number }; expected: { chat_model: string; chat_found: boolean; + chat_eligibility?: ModelContextEligibility | null; embedding_model: string; embedding_found: boolean; + embedding_eligibility?: ModelContextEligibility | null; vision_model: string; vision_found: boolean; }; diff --git a/docs/TEST-COVERAGE-MATRIX.md b/docs/TEST-COVERAGE-MATRIX.md index 5b3e5b1bf..d7598021c 100644 --- a/docs/TEST-COVERAGE-MATRIX.md +++ b/docs/TEST-COVERAGE-MATRIX.md @@ -124,6 +124,7 @@ Canonical mapping of every product feature to its test source(s). Drives gap-fil | 3.1.2 | Model Download & Installation | WD | `local-model-runtime.spec.ts` | ✅ | | | 3.1.3 | Model Version Handling | RU | `src/openhuman/local_ai/model_ids.rs` | ✅ | | | 3.1.4 | LM Studio Model Discovery | RU+RI | `src/openhuman/local_ai/service/ollama_admin_tests.rs`, `tests/json_rpc_e2e.rs` | ✅ | Uses LM Studio's OpenAI-compatible `/v1/models` surface | +| 3.1.5 | Model Context-Window Requirement Gate | RU+VU | `src/openhuman/inference/local/model_requirements.rs`, `src/openhuman/inference/local/ollama.rs`, `src/openhuman/inference/local/service/ollama_admin_tests.rs`, `app/src/components/settings/panels/local-model/ModelStatusSection.test.tsx` | ✅ | Rejects Ollama models whose native context window is below the memory-layer minimum (`local_ai.model_context_check`) | ### 3.2 Runtime Execution diff --git a/src/openhuman/about_app/catalog.rs b/src/openhuman/about_app/catalog.rs index b05789d26..bcb09a626 100644 --- a/src/openhuman/about_app/catalog.rs +++ b/src/openhuman/about_app/catalog.rs @@ -565,6 +565,16 @@ const CAPABILITIES: &[Capability] = &[ status: CapabilityStatus::Beta, privacy: None, }, + Capability { + id: "local_ai.model_context_check", + name: "Model Context Requirement Check", + domain: "local_ai", + category: CapabilityCategory::LocalAI, + description: "Diagnostics report each installed Ollama model's native context window and reject any model below the minimum the memory layer requires (so short-context models can't silently truncate and corrupt recall).", + how_to: "Settings > Local AI Model > Run Diagnostics", + status: CapabilityStatus::Beta, + privacy: None, + }, Capability { id: "local_ai.embed_text", name: "Generate Text Embeddings", diff --git a/src/openhuman/inference/local/mod.rs b/src/openhuman/inference/local/mod.rs index e158f06ee..b86e7df29 100644 --- a/src/openhuman/inference/local/mod.rs +++ b/src/openhuman/inference/local/mod.rs @@ -33,9 +33,11 @@ pub mod install; pub(crate) mod install_piper; pub(crate) mod install_whisper; pub(crate) mod lm_studio; +pub(crate) mod model_requirements; mod ollama; mod process_util; pub(crate) mod provider; +pub(crate) use model_requirements::{evaluate_context, ContextEligibility, MIN_CONTEXT_TOKENS}; pub(crate) use ollama::{ollama_base_url, OLLAMA_BASE_URL}; pub mod service; pub(crate) mod voice_install_common; diff --git a/src/openhuman/inference/local/model_requirements.rs b/src/openhuman/inference/local/model_requirements.rs new file mode 100644 index 000000000..25e7c2418 --- /dev/null +++ b/src/openhuman/inference/local/model_requirements.rs @@ -0,0 +1,140 @@ +//! Minimum model requirements the memory layer imposes on local models. +//! +//! The memory tree's embedder (`bge-m3`) is requested with +//! `num_ctx = 8192` (see +//! [`crate::openhuman::memory::tree::score::embed::ollama::EMBED_NUM_CTX`]) +//! and the summariser hard-caps its output to fit that 8192-token embed +//! ceiling. A local model whose native context window is below this floor +//! silently truncates chunks/summaries and corrupts recall, so we refuse +//! to accept any local model reporting a context window smaller than +//! [`MIN_CONTEXT_TOKENS`]. There is **no upper requirement** — the +//! pipeline is deliberately capped at 8k and never needs more from a +//! local model. + +use serde::Serialize; + +/// Minimum native context window (tokens) a local model must advertise to +/// be accepted by the memory layer. +/// +/// Re-exported from the embedder's own `EMBED_NUM_CTX` so this gate can +/// never drift from what the memory pipeline actually requests at embed +/// time. Changing the embedder's context request automatically moves the +/// acceptance floor with it. +pub const MIN_CONTEXT_TOKENS: u64 = + crate::openhuman::memory::tree::score::embed::ollama::EMBED_NUM_CTX as u64; + +/// Verdict for a single model's context window against +/// [`MIN_CONTEXT_TOKENS`]. Serialized into the diagnostics payload so the +/// frontend can render / reject each installed model. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +#[serde(tag = "status", rename_all = "snake_case")] +pub enum ContextEligibility { + /// Context window ≥ the minimum. Accepted. + Ok { context_length: u64 }, + /// Context window below the minimum. Rejected for memory-layer use. + BelowMinimum { context_length: u64, required: u64 }, + /// Context window could not be determined (`/api/show` error or the + /// metadata key is absent). Not hard-rejected — surfaced as unknown so + /// the UI can warn without blocking a model that may still be fine. + Unknown { required: u64 }, +} + +impl ContextEligibility { + /// `true` only when the model is positively accepted. + pub fn is_accepted(&self) -> bool { + matches!(self, ContextEligibility::Ok { .. }) + } + + /// `true` when the model is conclusively rejected (reported a context + /// window below the floor). `Unknown` is **not** a rejection. + pub fn is_rejected(&self) -> bool { + matches!(self, ContextEligibility::BelowMinimum { .. }) + } +} + +/// Classify a model's (optional) reported context length against the +/// memory-layer minimum. +pub fn evaluate_context(context_length: Option) -> ContextEligibility { + match context_length { + Some(ctx) if ctx >= MIN_CONTEXT_TOKENS => ContextEligibility::Ok { + context_length: ctx, + }, + Some(ctx) => ContextEligibility::BelowMinimum { + context_length: ctx, + required: MIN_CONTEXT_TOKENS, + }, + None => ContextEligibility::Unknown { + required: MIN_CONTEXT_TOKENS, + }, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn min_context_tracks_embedder_request() { + // The acceptance floor must equal what the memory embedder actually + // requests; this guards against the two drifting apart. + assert_eq!( + MIN_CONTEXT_TOKENS, + crate::openhuman::memory::tree::score::embed::ollama::EMBED_NUM_CTX as u64 + ); + assert_eq!(MIN_CONTEXT_TOKENS, 8_192); + } + + #[test] + fn at_or_above_minimum_is_accepted() { + let exact = evaluate_context(Some(8_192)); + assert!(exact.is_accepted()); + assert_eq!( + exact, + ContextEligibility::Ok { + context_length: 8_192 + } + ); + + let above = evaluate_context(Some(32_768)); + assert!(above.is_accepted()); + assert!(!above.is_rejected()); + } + + #[test] + fn below_minimum_is_rejected_with_required_floor() { + let verdict = evaluate_context(Some(2_048)); + assert!(verdict.is_rejected()); + assert!(!verdict.is_accepted()); + assert_eq!( + verdict, + ContextEligibility::BelowMinimum { + context_length: 2_048, + required: 8_192, + } + ); + } + + #[test] + fn unknown_context_is_neither_accepted_nor_rejected() { + let verdict = evaluate_context(None); + assert!(!verdict.is_accepted()); + assert!(!verdict.is_rejected()); + assert_eq!(verdict, ContextEligibility::Unknown { required: 8_192 }); + } + + #[test] + fn eligibility_serializes_tagged() { + let json = serde_json::to_value(evaluate_context(Some(4_096))).unwrap(); + assert_eq!(json["status"], "below_minimum"); + assert_eq!(json["context_length"], 4_096); + assert_eq!(json["required"], 8_192); + + let ok = serde_json::to_value(evaluate_context(Some(8_192))).unwrap(); + assert_eq!(ok["status"], "ok"); + assert_eq!(ok["context_length"], 8_192); + + let unknown = serde_json::to_value(evaluate_context(None)).unwrap(); + assert_eq!(unknown["status"], "unknown"); + assert_eq!(unknown["required"], 8_192); + } +} diff --git a/src/openhuman/inference/local/ollama.rs b/src/openhuman/inference/local/ollama.rs index cb96fb275..e53cbb067 100644 --- a/src/openhuman/inference/local/ollama.rs +++ b/src/openhuman/inference/local/ollama.rs @@ -144,6 +144,64 @@ pub(crate) struct OllamaModelTag { pub modified_at: Option, } +/// Request body for Ollama `POST /api/show`. +#[derive(Debug, Serialize)] +pub(crate) struct OllamaShowRequest { + /// Model tag (e.g. `bge-m3:latest`). `model` is the current Ollama + /// field; older servers also accept the legacy `name`, which `model` + /// is forward-compatible with. + pub model: String, +} + +/// Subset of Ollama `POST /api/show` we consume. +/// +/// `model_info` is an open key/value map whose keys are architecture-scoped +/// (e.g. `llama.context_length`, `bert.embedding_length`). The prefix is +/// whatever `general.architecture` reports, so the context-length key is +/// resolved dynamically rather than hard-coded per model family. +#[derive(Debug, Default, Deserialize)] +pub(crate) struct OllamaShowResponse { + #[serde(default)] + pub model_info: serde_json::Map, + // Present in the Ollama API response; retained to document the wire shape, not yet consumed. + #[serde(default)] + #[allow(dead_code)] + pub capabilities: Vec, +} + +impl OllamaShowResponse { + /// Native context window (tokens) advertised by the model's GGUF + /// metadata, or `None` when the server did not report it. + pub(crate) fn context_length(&self) -> Option { + context_length_from_model_info(&self.model_info) + } +} + +/// Extract `.context_length` from an Ollama `model_info` map. +/// +/// Resolution order: +/// 1. `general.architecture` → `{arch}.context_length` (documented shape). +/// 2. Fallback: the first key ending in `.context_length` — covers servers +/// that omit `general.architecture` or use an unexpected prefix. +pub(crate) fn context_length_from_model_info( + info: &serde_json::Map, +) -> Option { + fn as_u64(v: &serde_json::Value) -> Option { + v.as_u64() + .or_else(|| v.as_i64().filter(|n| *n >= 0).map(|n| n as u64)) + .or_else(|| v.as_f64().filter(|n| *n >= 0.0).map(|n| n as u64)) + } + if let Some(arch) = info.get("general.architecture").and_then(|v| v.as_str()) { + if let Some(value) = info.get(&format!("{arch}.context_length")).and_then(as_u64) { + return Some(value); + } + } + info.iter() + .filter(|(key, _)| key.ends_with(".context_length")) + .filter_map(|(_, value)| as_u64(value)) + .max() +} + #[derive(Debug, Serialize)] pub(crate) struct OllamaGenerateRequest { pub model: String, @@ -282,6 +340,74 @@ mod tests { assert_eq!(progress.aggregate_total(), Some(120)); } + // ── /api/show context-length extraction ────────────────────────── + + fn show_response(json: serde_json::Value) -> OllamaShowResponse { + serde_json::from_value(json).expect("OllamaShowResponse") + } + + #[test] + fn context_length_uses_general_architecture_prefix() { + let resp = show_response(serde_json::json!({ + "model_info": { + "general.architecture": "bert", + "bert.context_length": 8192, + "bert.embedding_length": 1024 + } + })); + assert_eq!(resp.context_length(), Some(8192)); + } + + #[test] + fn context_length_falls_back_when_architecture_missing() { + let resp = show_response(serde_json::json!({ + "model_info": { "llama.context_length": 4096 } + })); + assert_eq!(resp.context_length(), Some(4096)); + } + + #[test] + fn context_length_handles_float_and_string_encodings() { + // Some servers serialize the metadata number as a float. + let float = show_response(serde_json::json!({ + "model_info": { "general.architecture": "qwen2", "qwen2.context_length": 32768.0 } + })); + assert_eq!(float.context_length(), Some(32768)); + + // Non-numeric / missing → None (caller treats as Unknown, not a hard fail). + let missing = show_response(serde_json::json!({ "model_info": {} })); + assert_eq!(missing.context_length(), None); + let absent_field = show_response(serde_json::json!({})); + assert_eq!(absent_field.context_length(), None); + } + + #[test] + fn context_length_prefers_architecture_key_over_unrelated_match() { + let resp = show_response(serde_json::json!({ + "model_info": { + "general.architecture": "llama", + "llama.context_length": 8192, + "clip.context_length": 77 + } + })); + assert_eq!(resp.context_length(), Some(8192)); + } + + #[test] + fn context_length_fallback_returns_max_not_first() { + // Without `general.architecture`, the fallback must pick the *largest* + // `.context_length` value, not the first one encountered. Multimodal + // models can carry a low secondary value (e.g. `clip.context_length:77`) + // which, if chosen first, would incorrectly mark the model below minimum. + let resp = show_response(serde_json::json!({ + "model_info": { + "clip.context_length": 77, + "llama.context_length": 32768 + } + })); + assert_eq!(resp.context_length(), Some(32768)); + } + // ── ollama_base_url env-override behaviour ─────────────────────── // // These tests mutate the process-global `OPENHUMAN_OLLAMA_BASE_URL` diff --git a/src/openhuman/inference/local/service/ollama_admin.rs b/src/openhuman/inference/local/service/ollama_admin.rs index 3741b5a39..c30d33169 100644 --- a/src/openhuman/inference/local/service/ollama_admin.rs +++ b/src/openhuman/inference/local/service/ollama_admin.rs @@ -7,9 +7,12 @@ use crate::openhuman::inference::local::install::{ find_system_ollama_binary, run_ollama_install_script, }; use crate::openhuman::inference::local::lm_studio::lm_studio_base_url; +use crate::openhuman::inference::local::model_requirements::{ + evaluate_context, ContextEligibility, MIN_CONTEXT_TOKENS, +}; use crate::openhuman::inference::local::ollama::{ ollama_base_url, OllamaModelTag, OllamaPullEvent, OllamaPullProgress, OllamaPullRequest, - OllamaTagsResponse, + OllamaShowRequest, OllamaShowResponse, OllamaTagsResponse, }; use crate::openhuman::inference::local::process_util::apply_no_window; use crate::openhuman::inference::local::provider::{provider_from_config, LocalAiProvider}; @@ -859,6 +862,56 @@ impl LocalAiService { let embedding_found = has(&expected_embedding); let vision_found = has(&expected_vision); + // Per-model native context window vs the memory-layer minimum. + // `/api/show` is one bounded round-trip per installed model, + // fetched concurrently and only on this diagnostics path. + let model_eligibilities: Vec = if healthy { + futures_util::future::join_all(models.iter().map(|m| self.fetch_model_context(&m.name))) + .await + .into_iter() + .map(evaluate_context) + .collect() + } else { + Vec::new() + }; + + let installed_models: Vec = models + .iter() + .enumerate() + .map(|(i, m)| { + let eligibility = model_eligibilities.get(i).cloned(); + let context_length = match eligibility.as_ref() { + Some(ContextEligibility::Ok { context_length }) + | Some(ContextEligibility::BelowMinimum { context_length, .. }) => { + Some(*context_length) + } + _ => None, + }; + serde_json::json!({ + "name": m.name, + "size": m.size, + "modified_at": m.modified_at, + "context_length": context_length, + "eligibility": eligibility, + }) + }) + .collect(); + + // Resolve the eligibility of an expected (active) model by tag prefix. + let eligibility_for = |target: &str| -> Option { + let t = target.to_ascii_lowercase(); + models + .iter() + .zip(model_eligibilities.iter()) + .find(|(m, _)| { + let n = m.name.to_ascii_lowercase(); + n == t || n.starts_with(&(t.clone() + ":")) + }) + .map(|(_, e)| e.clone()) + }; + let chat_eligibility = eligibility_for(&expected_chat); + let embedding_eligibility = eligibility_for(&expected_embedding); + let binary_path = self.resolve_binary_path(config); let mut issues: Vec = Vec::new(); @@ -894,6 +947,32 @@ impl LocalAiService { if let Some(ref e) = tags_error { issues.push(format!("Failed to list models: {e}")); } + // Reject installed-but-too-small active models: a context window + // below the memory-layer minimum silently truncates chunks / + // summaries and corrupts recall. + if let Some(ContextEligibility::BelowMinimum { + context_length, + required, + }) = embedding_eligibility.as_ref() + { + issues.push(format!( + "Embedding model `{}` has a {}-token context window; the memory layer \ + requires at least {}. Choose an embedding model with a larger context \ + (e.g. bge-m3).", + expected_embedding, context_length, required + )); + } + if let Some(ContextEligibility::BelowMinimum { + context_length, + required, + }) = chat_eligibility.as_ref() + { + issues.push(format!( + "Chat model `{}` has a {}-token context window; the memory layer \ + requires at least {}.", + expected_chat, context_length, required + )); + } log::debug!( "[local_ai] diagnostics: healthy={} models={} issues={} repair_actions={}", @@ -907,13 +986,18 @@ impl LocalAiService { "ollama_running": healthy, "ollama_base_url": base_url, "ollama_binary_path": binary_path, - "installed_models": models, + "installed_models": installed_models, + "context_requirement": { + "min_context_tokens": MIN_CONTEXT_TOKENS, + }, "vision_mode": presets::vision_mode_for_config(&config.local_ai), "expected": { "chat_model": expected_chat, "chat_found": chat_found, + "chat_eligibility": chat_eligibility, "embedding_model": expected_embedding, "embedding_found": embedding_found, + "embedding_eligibility": embedding_eligibility, "vision_model": expected_vision, "vision_found": vision_found, }, @@ -1005,6 +1089,60 @@ impl LocalAiService { Ok(payload.models) } + /// Fetch a model's native context window via Ollama `POST /api/show`. + /// + /// Returns `None` on any failure (unreachable, non-2xx, parse error, or + /// the metadata key is absent) — the caller maps that to an `Unknown` + /// eligibility verdict rather than a hard rejection. One bounded HTTP + /// round-trip per model; only ever invoked from the diagnostics path. + async fn fetch_model_context(&self, model: &str) -> Option { + let url = format!("{}/api/show", ollama_base_url()); + let resp = self + .http + .post(&url) + .json(&OllamaShowRequest { + model: model.to_string(), + }) + .timeout(std::time::Duration::from_secs(5)) + .send() + .await + .inspect_err(|e| { + tracing::debug!( + target: "local_ai::ollama_admin", + %url, model, error = %e, + "[local_ai:ollama_admin] fetch_model_context: request failed" + ); + }) + .ok()?; + let status = resp.status(); + if !status.is_success() { + tracing::debug!( + target: "local_ai::ollama_admin", + %url, model, %status, + "[local_ai:ollama_admin] fetch_model_context: non-success response" + ); + return None; + } + let parsed: OllamaShowResponse = resp + .json() + .await + .inspect_err(|e| { + tracing::debug!( + target: "local_ai::ollama_admin", + %url, model, error = %e, + "[local_ai:ollama_admin] fetch_model_context: JSON parse failed" + ); + }) + .ok()?; + let ctx = parsed.context_length(); + tracing::debug!( + target: "local_ai::ollama_admin", + model, context_length = ?ctx, + "[local_ai:ollama_admin] fetch_model_context: resolved" + ); + ctx + } + async fn lm_studio_diagnostics(&self, config: &Config) -> Result { let base_url = lm_studio_base_url(config); let models_result = self.list_lm_studio_models(config).await; diff --git a/src/openhuman/inference/local/service/ollama_admin_tests.rs b/src/openhuman/inference/local/service/ollama_admin_tests.rs index d85e9356f..26493ccbf 100644 --- a/src/openhuman/inference/local/service/ollama_admin_tests.rs +++ b/src/openhuman/inference/local/service/ollama_admin_tests.rs @@ -947,3 +947,72 @@ fn binary_present_uses_ollama_bin_env_var_when_set() { "OLLAMA_BIN pointing to a real file must make ollama_binary_present return true" ); } + +#[tokio::test] +async fn diagnostics_gates_models_by_context_window() { + let _guard = crate::openhuman::inference::inference_test_guard(); + + // /api/tags lists two models; /api/show reports their context windows: + // one at the 8192 floor (accepted) and one well below (rejected). + let app = Router::new() + .route( + "/api/tags", + get(|| async { + Json(json!({ + "models": [ + {"name": "bge-m3:latest", "modified_at": "", "size": 1u64, "digest": "d"}, + {"name": "tiny-embed:latest", "modified_at": "", "size": 2u64, "digest": "d"} + ] + })) + }), + ) + .route( + "/api/show", + axum::routing::post(|Json(body): Json| async move { + let model = body["model"].as_str().unwrap_or_default().to_string(); + let ctx = if model.starts_with("bge-m3") { 8192 } else { 2048 }; + Json(json!({ + "model_info": { + "general.architecture": "bert", + "bert.context_length": ctx, + }, + "capabilities": ["embedding"], + })) + }), + ); + let base = spawn_mock(app).await; + unsafe { + std::env::set_var("OPENHUMAN_OLLAMA_BASE_URL", &base); + } + + let config = Config::default(); + let service = LocalAiService::new(&config); + let diag = service.diagnostics(&config).await.expect("diagnostics"); + + assert_eq!(diag["ollama_running"], true); + assert_eq!(diag["context_requirement"]["min_context_tokens"], 8192); + + let models = diag["installed_models"] + .as_array() + .expect("installed_models"); + let by_name = |needle: &str| { + models + .iter() + .find(|m| m["name"].as_str().unwrap_or("").starts_with(needle)) + .unwrap_or_else(|| panic!("model {needle} missing")) + .clone() + }; + + let accepted = by_name("bge-m3"); + assert_eq!(accepted["context_length"], 8192); + assert_eq!(accepted["eligibility"]["status"], "ok"); + + let rejected = by_name("tiny-embed"); + assert_eq!(rejected["context_length"], 2048); + assert_eq!(rejected["eligibility"]["status"], "below_minimum"); + assert_eq!(rejected["eligibility"]["required"], 8192); + + unsafe { + std::env::remove_var("OPENHUMAN_OLLAMA_BASE_URL"); + } +} diff --git a/src/openhuman/inference/schemas.rs b/src/openhuman/inference/schemas.rs index c63568ef3..7253c4e86 100644 --- a/src/openhuman/inference/schemas.rs +++ b/src/openhuman/inference/schemas.rs @@ -303,7 +303,15 @@ pub fn schemas(function: &str) -> ControllerSchema { function: "diagnostics", description: "Run diagnostics for the configured local inference provider endpoint and expected models.", inputs: vec![], - outputs: vec![json_output("diagnostics", "Inference diagnostics payload.")], + outputs: vec![json_output( + "diagnostics", + "Inference diagnostics payload. `installed_models[]` carries \ + `context_length` and an `eligibility` verdict ({status: ok | \ + below_minimum | unknown}); `context_requirement.min_context_tokens` \ + is the memory-layer floor; `expected.{chat,embedding}_eligibility` \ + mirror it for the active models. Models below the floor are rejected \ + via `issues`.", + )], }, "summarize" => ControllerSchema { namespace: "inference", diff --git a/src/openhuman/memory/tree/score/embed/ollama.rs b/src/openhuman/memory/tree/score/embed/ollama.rs index 5884999ed..cd8db0f3b 100644 --- a/src/openhuman/memory/tree/score/embed/ollama.rs +++ b/src/openhuman/memory/tree/score/embed/ollama.rs @@ -115,7 +115,14 @@ impl OllamaEmbedder { /// silent prompt truncation; on models that natively support less, /// Ollama clamps `num_ctx` to the model's actual maximum, so this is /// safe to over-request. -const EMBED_NUM_CTX: u32 = 8192; +/// +/// This is also the **single source of truth for the memory layer's +/// minimum model context window**: a local model whose native context is +/// below this floor silently truncates chunks/summaries and corrupts +/// recall. `inference::local::model_requirements::MIN_CONTEXT_TOKENS` +/// re-exports this value so the model-acceptance gate can never drift +/// from what the embedder actually requests. +pub(crate) const EMBED_NUM_CTX: u32 = 8192; #[derive(Serialize)] struct EmbedRequest<'a> {