mirror of
https://github.com/tinyhumansai/openhuman.git
synced 2026-07-27 21:08:00 +00:00
feat(local_ai): unify memory embeddings — cloud Voyage default + live local toggle + dev infra (#1640)
Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.7
parent
41813dba14
commit
256f004c36
@@ -77,6 +77,15 @@ VITE_DEV_FORCE_ONBOARDING=false
|
||||
# core stops responding. Bounded [1000, 600000]; default 30000.
|
||||
# VITE_CORE_RPC_TIMEOUT_MS=30000
|
||||
|
||||
# [optional] Vite dev-server port for `pnpm dev:app:win` (read by both
|
||||
# vite.config.ts and scripts/run-dev-win.sh). Default 1420. Override per
|
||||
# worktree to avoid the hardcoded-1420 collision when running multiple
|
||||
# dev:app:win sessions in parallel. The HMR companion websocket binds
|
||||
# to (port + 1), so leave a 2-port gap between concurrent worktrees:
|
||||
# 1420, 1422, 1424, … When set to non-1420 the dev script also adds an
|
||||
# inline `tauri dev -c '{"build":{"devUrl":...}}'` override.
|
||||
# OPENHUMAN_DEV_PORT=1422
|
||||
|
||||
# [optional] Minimum desktop app semver to complete OAuth deep links (openhuman://oauth/success). Leave unset in dev.
|
||||
# VITE_MINIMUM_SUPPORTED_APP_VERSION=0.51.0
|
||||
# [optional] Download page when OAuth is blocked due to an outdated build (default: GitHub releases/latest).
|
||||
|
||||
@@ -1,133 +0,0 @@
|
||||
import { useState } from 'react';
|
||||
|
||||
import type { Backend } from '../../lib/intelligence/settingsApi';
|
||||
|
||||
interface BackendChooserProps {
|
||||
/** Currently selected backend. */
|
||||
value: Backend;
|
||||
/** Called when the user clicks a different card. */
|
||||
onChange: (next: Backend) => void;
|
||||
/** Optional cloud-cost estimate. Mock value until cost-tracker hook lands. */
|
||||
costEstimate?: string;
|
||||
/** Disabled while a backend switch is in flight. */
|
||||
busy?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Two large cards — Cloud (default, recommended) vs Local (advanced).
|
||||
*
|
||||
* Visual style intentionally matches the rest of the Intelligence page:
|
||||
* `bg-white` + `border-stone-200` + `rounded-2xl`, primary blue for the
|
||||
* selected accent. The inline tokens from the brief
|
||||
* (paper, hairline, ocean) map onto the existing stone/primary scale —
|
||||
* we keep the existing scale to avoid forking the design system.
|
||||
*/
|
||||
export default function BackendChooser({
|
||||
value,
|
||||
onChange,
|
||||
costEstimate = '$0.42 / mo est.',
|
||||
busy = false,
|
||||
}: BackendChooserProps) {
|
||||
const [hoveredCloud, setHoveredCloud] = useState(false);
|
||||
|
||||
const cardBase =
|
||||
'flex-1 min-h-[160px] px-6 py-5 rounded-2xl text-left transition-all duration-150 disabled:opacity-50 disabled:cursor-not-allowed';
|
||||
|
||||
return (
|
||||
<div className="flex gap-4 flex-col sm:flex-row" role="radiogroup" aria-label="AI backend">
|
||||
{/* Cloud */}
|
||||
<button
|
||||
type="button"
|
||||
role="radio"
|
||||
aria-checked={value === 'cloud'}
|
||||
disabled={busy}
|
||||
onClick={() => onChange('cloud')}
|
||||
onMouseEnter={() => setHoveredCloud(true)}
|
||||
onMouseLeave={() => setHoveredCloud(false)}
|
||||
onFocus={() => setHoveredCloud(true)}
|
||||
onBlur={() => setHoveredCloud(false)}
|
||||
className={`${cardBase} border-2 ${
|
||||
value === 'cloud'
|
||||
? 'border-primary-500 bg-white shadow-soft'
|
||||
: 'border-stone-200 bg-stone-50 hover:bg-white hover:border-stone-300'
|
||||
}`}>
|
||||
<div className="flex items-center justify-between mb-3">
|
||||
<div className="flex items-center gap-2">
|
||||
<RadioDot active={value === 'cloud'} />
|
||||
<span className="text-sm font-semibold text-stone-900">Cloud</span>
|
||||
<span className="text-[10px] uppercase tracking-wider px-1.5 py-0.5 rounded-full bg-primary-50 text-primary-700 border border-primary-100">
|
||||
Recommended
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
<p className="text-xs text-stone-600 leading-relaxed mb-3">
|
||||
Runs on OpenHuman servers. Costs credits. No local CPU.
|
||||
</p>
|
||||
<div className="font-mono text-[11px] text-stone-500">{costEstimate}</div>
|
||||
{/* Privacy reassurance — appears on hover/focus of the Cloud card. */}
|
||||
<div
|
||||
className={`mt-3 text-[11px] text-stone-500 leading-snug transition-opacity ${
|
||||
hoveredCloud ? 'opacity-100' : 'opacity-0'
|
||||
}`}
|
||||
aria-live="polite">
|
||||
Your data still stays local. bge-m3 embedder runs on your machine regardless.
|
||||
</div>
|
||||
</button>
|
||||
|
||||
{/* Local */}
|
||||
<button
|
||||
type="button"
|
||||
role="radio"
|
||||
aria-checked={value === 'local'}
|
||||
disabled={busy}
|
||||
onClick={() => onChange('local')}
|
||||
className={`${cardBase} border-2 ${
|
||||
value === 'local'
|
||||
? 'border-primary-500 bg-white shadow-soft'
|
||||
: 'border-stone-200 bg-stone-50 hover:bg-white hover:border-stone-300'
|
||||
}`}>
|
||||
<div className="flex items-center justify-between mb-3">
|
||||
<div className="flex items-center gap-2">
|
||||
<RadioDot active={value === 'local'} />
|
||||
<span className="text-sm font-semibold text-stone-900">Local</span>
|
||||
<span className="text-[10px] uppercase tracking-wider px-1.5 py-0.5 rounded-full bg-stone-100 text-stone-600 border border-stone-200">
|
||||
Advanced
|
||||
</span>
|
||||
</div>
|
||||
</div>
|
||||
<p className="text-xs text-stone-600 leading-relaxed mb-3">
|
||||
Runs on your machine. Free. Uses your CPU and battery.
|
||||
</p>
|
||||
<div className="flex items-center gap-1.5 text-[11px] text-amber-700">
|
||||
<svg
|
||||
className="w-3 h-3"
|
||||
viewBox="0 0 24 24"
|
||||
fill="none"
|
||||
stroke="currentColor"
|
||||
strokeWidth={2}>
|
||||
<path
|
||||
strokeLinecap="round"
|
||||
strokeLinejoin="round"
|
||||
d="M12 9v3.75m0 3.75h.008v.008H12v-.008zM21 12a9 9 0 11-18 0 9 9 0 0118 0z"
|
||||
/>
|
||||
</svg>
|
||||
<span>≥8 GB RAM recommended</span>
|
||||
</div>
|
||||
</button>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function RadioDot({ active }: { active: boolean }) {
|
||||
return (
|
||||
<span
|
||||
aria-hidden
|
||||
className={`w-3.5 h-3.5 rounded-full border-2 flex items-center justify-center ${
|
||||
active ? 'border-primary-500' : 'border-stone-300'
|
||||
}`}>
|
||||
<span
|
||||
className={`w-1.5 h-1.5 rounded-full ${active ? 'bg-primary-500' : 'bg-transparent'}`}
|
||||
/>
|
||||
</span>
|
||||
);
|
||||
}
|
||||
@@ -1,181 +0,0 @@
|
||||
import { useCallback, useEffect, useMemo, useState } from 'react';
|
||||
|
||||
import {
|
||||
type Backend,
|
||||
capabilityForModel,
|
||||
DEFAULT_EXTRACT_MODEL,
|
||||
downloadAsset,
|
||||
fetchInstalledModels,
|
||||
getMemoryTreeLlm,
|
||||
type ModelDescriptor,
|
||||
REQUIRED_EMBEDDER_MODEL,
|
||||
setMemoryTreeLlm,
|
||||
} from '../../lib/intelligence/settingsApi';
|
||||
import BackendChooser from './BackendChooser';
|
||||
import ModelAssignment from './ModelAssignment';
|
||||
import ModelCatalog from './ModelCatalog';
|
||||
|
||||
/**
|
||||
* Settings tab for the Intelligence page.
|
||||
*
|
||||
* Layout (top → bottom):
|
||||
* 1. AI Backend — Cloud / Local toggle
|
||||
* 2. Model Assignment — per-role dropdowns (visible only in Local mode)
|
||||
* 3. Model Catalog — full curated list with download / use / delete
|
||||
* 4. Currently Loaded — live `/api/ps`-style readout
|
||||
*
|
||||
* The orchestrator owns the cross-section state (backend, role assignments,
|
||||
* cached installed-models / status). Sections themselves stay presentational.
|
||||
*/
|
||||
export default function IntelligenceSettingsTab() {
|
||||
const [backend, setBackend] = useState<Backend>('cloud');
|
||||
const [backendBusy, setBackendBusy] = useState(false);
|
||||
// Single Memory LLM that drives both extractor and summariser. Most
|
||||
// users want one model for both; the rare case of mixing them is not
|
||||
// worth the second dropdown's cognitive cost.
|
||||
const [memoryModel, setMemoryModel] = useState<string>(DEFAULT_EXTRACT_MODEL);
|
||||
const [installedModels, setInstalledModels] = useState<string[]>([]);
|
||||
|
||||
// One-shot bootstrap — pull current backend and the installed-model list.
|
||||
useEffect(() => {
|
||||
let cancelled = false;
|
||||
void (async () => {
|
||||
try {
|
||||
console.debug('[intelligence-settings] bootstrap');
|
||||
const [bk, models] = await Promise.all([getMemoryTreeLlm(), fetchInstalledModels()]);
|
||||
if (cancelled) return;
|
||||
setBackend(bk);
|
||||
setInstalledModels(models.map(m => m.name));
|
||||
} catch (err) {
|
||||
if (!cancelled) {
|
||||
// Bootstrap failure leaves the tab on its useState defaults
|
||||
// (cloud backend, empty installed list) rather than throwing
|
||||
// an unhandled rejection. The user can still flip the backend
|
||||
// chooser; subsequent reads will retry the RPCs.
|
||||
console.error('[intelligence-settings] bootstrap failed', err);
|
||||
}
|
||||
}
|
||||
})();
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, []);
|
||||
|
||||
const handleBackendChange = useCallback(async (next: Backend) => {
|
||||
setBackendBusy(true);
|
||||
try {
|
||||
const { effective } = await setMemoryTreeLlm(next);
|
||||
setBackend(effective);
|
||||
} catch (err) {
|
||||
console.error('[intelligence-settings] backend switch failed', err);
|
||||
} finally {
|
||||
setBackendBusy(false);
|
||||
}
|
||||
}, []);
|
||||
|
||||
// Persist Memory LLM changes to config.toml. Fans out to both
|
||||
// extractor and summariser keys in a single atomic write — the unified
|
||||
// UI is one dropdown, but the underlying schema retains both keys so
|
||||
// power users can still split them via the RPC directly if needed.
|
||||
const handleMemoryModelChange = useCallback(
|
||||
async (id: string) => {
|
||||
console.debug('[intelligence-settings] memory model -> %s', id);
|
||||
const previous = memoryModel;
|
||||
setMemoryModel(id);
|
||||
try {
|
||||
await setMemoryTreeLlm('local', { extractModel: id, summariserModel: id });
|
||||
} catch (err) {
|
||||
// Persistence failed → roll back the optimistic UI update so the
|
||||
// dropdown reflects the value that's actually saved on disk
|
||||
// rather than the one the user just attempted.
|
||||
setMemoryModel(previous);
|
||||
console.error('[intelligence-settings] persist memory model failed', err);
|
||||
}
|
||||
},
|
||||
[memoryModel]
|
||||
);
|
||||
|
||||
const handleDownload = useCallback(async (model: ModelDescriptor) => {
|
||||
const cap = capabilityForModel(model);
|
||||
if (!cap) {
|
||||
console.debug('[intelligence-settings] no capability for model', { id: model.id });
|
||||
return;
|
||||
}
|
||||
try {
|
||||
await downloadAsset(cap);
|
||||
} catch (err) {
|
||||
console.error('[intelligence-settings] model download failed', err);
|
||||
} finally {
|
||||
// Refresh installed list after any download attempt — even on
|
||||
// failure, Ollama may have partially landed assets we should
|
||||
// surface; if it hasn't, the next bootstrap tick will catch up.
|
||||
const refreshed = await fetchInstalledModels();
|
||||
setInstalledModels(refreshed.map(m => m.name));
|
||||
}
|
||||
}, []);
|
||||
|
||||
const handleUse = useCallback(
|
||||
(model: ModelDescriptor) => {
|
||||
if (model.roles.includes('extract') || model.roles.includes('summariser')) {
|
||||
void handleMemoryModelChange(model.id);
|
||||
}
|
||||
},
|
||||
[handleMemoryModelChange]
|
||||
);
|
||||
|
||||
const activeModelIds = useMemo<string[]>(() => {
|
||||
const ids = new Set<string>();
|
||||
ids.add(memoryModel);
|
||||
ids.add(REQUIRED_EMBEDDER_MODEL);
|
||||
return [...ids];
|
||||
}, [memoryModel]);
|
||||
|
||||
return (
|
||||
<div className="space-y-10" data-testid="intelligence-settings-tab">
|
||||
<Section title="AI backend">
|
||||
<BackendChooser value={backend} onChange={handleBackendChange} busy={backendBusy} />
|
||||
</Section>
|
||||
|
||||
{/* All local-model sections (assignment, catalog, currently-loaded)
|
||||
are gated on local backend. Cloud users get just the backend
|
||||
chooser + the explanatory copy that lives inside it — they don't
|
||||
need to see Ollama-related UI at all. */}
|
||||
{backend === 'local' && (
|
||||
<>
|
||||
<Section title="Model assignment">
|
||||
<ModelAssignment
|
||||
installedModelIds={installedModels}
|
||||
memoryModel={memoryModel}
|
||||
onChangeMemory={handleMemoryModelChange}
|
||||
/>
|
||||
</Section>
|
||||
|
||||
<Section title="Model catalog">
|
||||
<ModelCatalog
|
||||
installedModelIds={installedModels}
|
||||
activeModelIds={activeModelIds}
|
||||
onDownload={handleDownload}
|
||||
onUse={handleUse}
|
||||
/>
|
||||
</Section>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
interface SectionProps {
|
||||
title: string;
|
||||
children: React.ReactNode;
|
||||
}
|
||||
|
||||
function Section({ title, children }: SectionProps) {
|
||||
return (
|
||||
<section>
|
||||
<h2 className="font-display text-[11px] uppercase tracking-[0.18em] text-stone-400 mb-3">
|
||||
{title}
|
||||
</h2>
|
||||
{children}
|
||||
</section>
|
||||
);
|
||||
}
|
||||
@@ -1,150 +0,0 @@
|
||||
import {
|
||||
DEFAULT_EXTRACT_MODEL,
|
||||
DEFAULT_SUMMARISER_MODEL,
|
||||
type ModelDescriptor,
|
||||
RECOMMENDED_MODEL_CATALOG,
|
||||
REQUIRED_EMBEDDER_MODEL,
|
||||
} from '../../lib/intelligence/settingsApi';
|
||||
|
||||
interface ModelAssignmentProps {
|
||||
/** Names of models that are already installed on the user's machine. */
|
||||
installedModelIds: ReadonlyArray<string>;
|
||||
/** Currently chosen memory LLM (used for both extract + summarise). */
|
||||
memoryModel: string;
|
||||
/** Called when the user picks a different memory LLM. The setting fans
|
||||
* out to both `llm_extractor_model` and `llm_summariser_model` in
|
||||
* config.toml — most users want one model for both roles, and the
|
||||
* cognitive load of two dropdowns isn't worth the rare power-user
|
||||
* case of mixing them. */
|
||||
onChangeMemory: (id: string) => void;
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-role assignment table — two rows: Memory LLM (covers both extract
|
||||
* and summarise), and Embedder.
|
||||
*
|
||||
* The embedder row is locked to `bge-m3` for v1 (the spec says we never
|
||||
* round-trip embeddings through the cloud). The Memory LLM dropdown is
|
||||
* populated from the recommended catalog filtered to models that can
|
||||
* serve both extract AND summarise roles, plus any locally-installed
|
||||
* models the user has pulled outside the curated catalog.
|
||||
*/
|
||||
export default function ModelAssignment({
|
||||
installedModelIds,
|
||||
memoryModel,
|
||||
onChangeMemory,
|
||||
}: ModelAssignmentProps) {
|
||||
// Ollama returns tags as `<name>:latest` for default-tag models. The
|
||||
// catalog stores bare names (e.g. `bge-m3`). Strip the `:latest` suffix
|
||||
// on the installed side so the bare-name comparison matches.
|
||||
const normalizedInstalled = installedModelIds.map(id =>
|
||||
id.endsWith(':latest') ? id.slice(0, -':latest'.length) : id
|
||||
);
|
||||
const memoryOptions = memoryLlmOptions(normalizedInstalled);
|
||||
const embedderDescriptor = RECOMMENDED_MODEL_CATALOG.find(m => m.id === REQUIRED_EMBEDDER_MODEL);
|
||||
const embedderInstalled = normalizedInstalled.includes(REQUIRED_EMBEDDER_MODEL);
|
||||
|
||||
return (
|
||||
<div className="border border-stone-200 rounded-2xl overflow-hidden">
|
||||
<Row
|
||||
label="Memory LLM"
|
||||
sublabel={describeMemory(memoryOptions.find(opt => opt.id === memoryModel))}>
|
||||
<select
|
||||
value={memoryModel}
|
||||
onChange={e => onChangeMemory(e.target.value)}
|
||||
className="w-full sm:w-64 px-3 py-1.5 text-sm bg-white border border-stone-200 rounded-lg text-stone-900 focus:outline-none focus:border-primary-500/50 transition-colors"
|
||||
aria-label="Memory LLM (extract + summarise)">
|
||||
{memoryOptions.map(opt => (
|
||||
<option key={opt.id} value={opt.id}>
|
||||
{opt.label ?? opt.id}
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
</Row>
|
||||
|
||||
<Row
|
||||
label="Embedder"
|
||||
sublabel={
|
||||
embedderDescriptor
|
||||
? `${embedderDescriptor.size} · required · 1024-dim`
|
||||
: 'required · 1024-dim'
|
||||
}
|
||||
last>
|
||||
<div className="flex items-center gap-2 text-sm font-mono text-stone-700">
|
||||
<span>{REQUIRED_EMBEDDER_MODEL}</span>
|
||||
{embedderInstalled ? (
|
||||
<span className="inline-flex items-center gap-1 text-sage-600 text-xs">
|
||||
<svg
|
||||
className="w-3 h-3"
|
||||
viewBox="0 0 24 24"
|
||||
fill="none"
|
||||
stroke="currentColor"
|
||||
strokeWidth={2.5}>
|
||||
<path strokeLinecap="round" strokeLinejoin="round" d="M5 13l4 4L19 7" />
|
||||
</svg>
|
||||
loaded
|
||||
</span>
|
||||
) : (
|
||||
<span className="text-amber-700 text-xs">not downloaded</span>
|
||||
)}
|
||||
</div>
|
||||
</Row>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
interface RowProps {
|
||||
label: string;
|
||||
sublabel: string;
|
||||
last?: boolean;
|
||||
children: React.ReactNode;
|
||||
}
|
||||
|
||||
function Row({ label, sublabel, last, children }: RowProps) {
|
||||
return (
|
||||
<div
|
||||
className={`grid grid-cols-1 sm:grid-cols-[1fr_auto] gap-2 sm:gap-6 px-5 py-4 ${
|
||||
last ? '' : 'border-b border-stone-100'
|
||||
}`}>
|
||||
<div>
|
||||
<div className="text-sm font-semibold text-stone-900">{label}</div>
|
||||
<div className="font-mono text-[11px] text-stone-500 mt-0.5">{sublabel}</div>
|
||||
</div>
|
||||
<div className="flex items-center sm:justify-end">{children}</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function describeMemory(model?: ModelDescriptor): string {
|
||||
if (!model) return 'used for extract + summarise';
|
||||
return `${model.size} · ${model.ramHint} · ${model.category}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the Memory LLM dropdown options. A model qualifies if it can serve
|
||||
* BOTH extract and summarise roles. Catalog entries come first; locally
|
||||
* installed extras (pulled outside the curated catalog) are appended so
|
||||
* they remain selectable.
|
||||
*/
|
||||
function memoryLlmOptions(installedModelIds: ReadonlyArray<string>): ModelDescriptor[] {
|
||||
const catalog = RECOMMENDED_MODEL_CATALOG.filter(
|
||||
m => m.roles.includes('extract') && m.roles.includes('summariser')
|
||||
);
|
||||
const known = new Set(catalog.map(m => m.id));
|
||||
const extras = installedModelIds
|
||||
.filter(id => !known.has(id) && id !== REQUIRED_EMBEDDER_MODEL)
|
||||
.map<ModelDescriptor>(id => ({
|
||||
id,
|
||||
size: '—',
|
||||
approxBytes: 0,
|
||||
ramHint: '—',
|
||||
category: 'balanced',
|
||||
note: 'locally installed',
|
||||
roles: ['extract', 'summariser'],
|
||||
}));
|
||||
return [...catalog, ...extras];
|
||||
}
|
||||
|
||||
// Re-export defaults so callers can still seed initial state via these
|
||||
// constants without chasing them through the API module.
|
||||
export { DEFAULT_EXTRACT_MODEL, DEFAULT_SUMMARISER_MODEL, REQUIRED_EMBEDDER_MODEL };
|
||||
@@ -1,252 +0,0 @@
|
||||
import { useState } from 'react';
|
||||
|
||||
import {
|
||||
capabilityForModel,
|
||||
type ModelDescriptor,
|
||||
RECOMMENDED_MODEL_CATALOG,
|
||||
} from '../../lib/intelligence/settingsApi';
|
||||
|
||||
interface ModelCatalogProps {
|
||||
/** Names of models that are already installed on the user's machine. */
|
||||
installedModelIds: ReadonlyArray<string>;
|
||||
/** Models in active use right now (assigned to a role). */
|
||||
activeModelIds: ReadonlyArray<string>;
|
||||
/** Called when the user kicks off a download for a catalog entry. */
|
||||
onDownload: (model: ModelDescriptor) => Promise<void>;
|
||||
/** Called when the user wants to assign an installed model to its role. */
|
||||
onUse: (model: ModelDescriptor) => void;
|
||||
/** Called when the user removes an installed model. */
|
||||
onDelete?: (model: ModelDescriptor) => Promise<void>;
|
||||
}
|
||||
|
||||
type RowState = 'idle' | 'downloading' | 'error';
|
||||
|
||||
/**
|
||||
* Single-column list of curated models. Each row is one card showing
|
||||
* <id> <size> <status> [action]
|
||||
* The action button changes by state:
|
||||
* - not installed → "Download" (clicks fire the per-capability RPC)
|
||||
* - installed but unused → "Use"
|
||||
* - installed and active → "Active"
|
||||
* - downloading → inline progress bar (mocked client-side animation
|
||||
* since the per-asset RPC is fire-and-forget; the real progress
|
||||
* stream is wired in `local_ai_downloads_progress` polling — out of
|
||||
* scope for v1)
|
||||
*/
|
||||
// Ollama reports tags as `<name>:<tag>` (e.g. `bge-m3:latest`,
|
||||
// `gemma3:1b-it-qat`). The recommended catalog uses bare names for the
|
||||
// default-`:latest` case (e.g. `bge-m3`) and full `<name>:<tag>` for
|
||||
// non-default tags. Normalize both sides by stripping the `:latest`
|
||||
// suffix before comparing — that way `bge-m3` matches `bge-m3:latest`,
|
||||
// while `gemma3:1b-it-qat` still requires the explicit tag.
|
||||
function normalizeModelId(id: string): string {
|
||||
return id.endsWith(':latest') ? id.slice(0, -':latest'.length) : id;
|
||||
}
|
||||
|
||||
export default function ModelCatalog({
|
||||
installedModelIds,
|
||||
activeModelIds,
|
||||
onDownload,
|
||||
onUse,
|
||||
onDelete,
|
||||
}: ModelCatalogProps) {
|
||||
const installedSet = new Set(installedModelIds.map(normalizeModelId));
|
||||
const activeSet = new Set(activeModelIds.map(normalizeModelId));
|
||||
|
||||
return (
|
||||
<div className="space-y-2">
|
||||
{RECOMMENDED_MODEL_CATALOG.map(model => (
|
||||
<CatalogRow
|
||||
key={model.id}
|
||||
model={model}
|
||||
installed={installedSet.has(normalizeModelId(model.id))}
|
||||
active={activeSet.has(normalizeModelId(model.id))}
|
||||
onDownload={onDownload}
|
||||
onUse={onUse}
|
||||
onDelete={onDelete}
|
||||
/>
|
||||
))}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
interface CatalogRowProps {
|
||||
model: ModelDescriptor;
|
||||
installed: boolean;
|
||||
active: boolean;
|
||||
onDownload: ModelCatalogProps['onDownload'];
|
||||
onUse: ModelCatalogProps['onUse'];
|
||||
onDelete: ModelCatalogProps['onDelete'];
|
||||
}
|
||||
|
||||
function CatalogRow({ model, installed, active, onDownload, onUse, onDelete }: CatalogRowProps) {
|
||||
const [state, setState] = useState<RowState>('idle');
|
||||
const [progress, setProgress] = useState(0);
|
||||
|
||||
const status: 'active' | 'installed' | 'available' = active
|
||||
? 'active'
|
||||
: installed
|
||||
? 'installed'
|
||||
: 'available';
|
||||
|
||||
const handleDownload = async () => {
|
||||
setState('downloading');
|
||||
setProgress(8);
|
||||
// Animated mock progress while the real per-capability RPC is in flight.
|
||||
// The real download progress stream comes from
|
||||
// `openhumanLocalAiDownloadsProgress` polling — wiring that in is
|
||||
// tracked separately and out of scope for v1.
|
||||
const tick = setInterval(() => {
|
||||
setProgress(prev => {
|
||||
if (prev >= 90) return prev;
|
||||
return prev + Math.max(2, Math.round((100 - prev) * 0.06));
|
||||
});
|
||||
}, 220);
|
||||
let didFail = false;
|
||||
try {
|
||||
await onDownload(model);
|
||||
setProgress(100);
|
||||
} catch (err) {
|
||||
console.debug('[intelligence-settings] catalog download failed', { id: model.id, err });
|
||||
setState('error');
|
||||
didFail = true;
|
||||
} finally {
|
||||
clearInterval(tick);
|
||||
// Hold the terminal state long enough for the user to actually read
|
||||
// it. Success collapses fast (~600 ms) so the row settles back to
|
||||
// its post-install state without a long pause; error lingers ~3s
|
||||
// so an unsuccessful pull doesn't snap back before the user has
|
||||
// a chance to notice. Tracked via a local flag because `state` is
|
||||
// React state and won't reflect the just-issued `setState('error')`
|
||||
// until the next render.
|
||||
const settleMs = didFail ? 3000 : 600;
|
||||
window.setTimeout(() => {
|
||||
setState('idle');
|
||||
setProgress(0);
|
||||
}, settleMs);
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div className="border border-stone-200 rounded-xl bg-white px-4 py-3">
|
||||
<div className="flex items-center justify-between gap-3 flex-wrap">
|
||||
<div className="flex items-center gap-3 min-w-0 flex-1">
|
||||
<div className="text-sm font-medium text-stone-900 truncate">{model.id}</div>
|
||||
<div className="font-mono text-[11px] text-stone-500 whitespace-nowrap">{model.size}</div>
|
||||
<StatusChip status={status} />
|
||||
</div>
|
||||
<div className="flex items-center gap-2">
|
||||
{state === 'downloading' ? (
|
||||
<ProgressBar progress={progress} />
|
||||
) : (
|
||||
<ActionButton
|
||||
status={status}
|
||||
hasDelete={!!onDelete}
|
||||
onDownload={handleDownload}
|
||||
onUse={() => onUse(model)}
|
||||
onDelete={onDelete ? () => onDelete(model) : undefined}
|
||||
/>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
<div className="mt-1 flex items-center gap-3 font-mono text-[11px] text-stone-500">
|
||||
<span>{model.ramHint}</span>
|
||||
<span>·</span>
|
||||
<span>{model.category}</span>
|
||||
<span>·</span>
|
||||
<span className="text-stone-400">{model.note}</span>
|
||||
{capabilityForModel(model) === null && (
|
||||
<span className="text-amber-600 ml-auto">no capability binding</span>
|
||||
)}
|
||||
</div>
|
||||
{state === 'error' && (
|
||||
<div className="mt-2 text-[11px] text-coral-700">
|
||||
Download failed — check Ollama is running and try again.
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
function StatusChip({ status }: { status: 'active' | 'installed' | 'available' }) {
|
||||
if (status === 'active') {
|
||||
return (
|
||||
<span className="inline-flex items-center gap-1 px-2 py-0.5 text-[10px] uppercase tracking-wider rounded-full bg-sage-50 text-sage-700 border border-sage-100">
|
||||
active
|
||||
</span>
|
||||
);
|
||||
}
|
||||
if (status === 'installed') {
|
||||
return (
|
||||
<span className="inline-flex items-center gap-1 px-2 py-0.5 text-[10px] uppercase tracking-wider rounded-full bg-stone-100 text-stone-600 border border-stone-200">
|
||||
installed
|
||||
</span>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<span className="inline-flex items-center gap-1 px-2 py-0.5 text-[10px] uppercase tracking-wider rounded-full bg-white text-stone-500 border border-stone-200">
|
||||
not downloaded
|
||||
</span>
|
||||
);
|
||||
}
|
||||
|
||||
interface ActionButtonProps {
|
||||
status: 'active' | 'installed' | 'available';
|
||||
hasDelete: boolean;
|
||||
onDownload: () => void;
|
||||
onUse: () => void;
|
||||
onDelete?: () => void;
|
||||
}
|
||||
|
||||
function ActionButton({ status, hasDelete, onDownload, onUse, onDelete }: ActionButtonProps) {
|
||||
if (status === 'active') {
|
||||
return (
|
||||
<span className="px-3 py-1.5 text-xs text-stone-500 border border-transparent">in use</span>
|
||||
);
|
||||
}
|
||||
if (status === 'installed') {
|
||||
return (
|
||||
<div className="flex items-center gap-1.5">
|
||||
<button
|
||||
type="button"
|
||||
onClick={onUse}
|
||||
className="px-3 py-1.5 text-xs font-medium bg-primary-50 hover:bg-primary-100 text-primary-700 border border-primary-100 rounded-lg transition-colors">
|
||||
Use
|
||||
</button>
|
||||
{hasDelete && onDelete && (
|
||||
<button
|
||||
type="button"
|
||||
onClick={onDelete}
|
||||
className="px-2 py-1.5 text-xs text-stone-500 hover:text-coral-700 border border-stone-200 rounded-lg transition-colors"
|
||||
aria-label="Delete model">
|
||||
Delete
|
||||
</button>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<button
|
||||
type="button"
|
||||
onClick={onDownload}
|
||||
className="px-3 py-1.5 text-xs font-medium bg-white hover:bg-stone-50 text-stone-700 border border-stone-200 rounded-lg transition-colors">
|
||||
Download
|
||||
</button>
|
||||
);
|
||||
}
|
||||
|
||||
function ProgressBar({ progress }: { progress: number }) {
|
||||
return (
|
||||
<div
|
||||
className="w-32 h-2 rounded-full bg-stone-100 overflow-hidden"
|
||||
role="progressbar"
|
||||
aria-valuemin={0}
|
||||
aria-valuemax={100}
|
||||
aria-valuenow={Math.round(progress)}>
|
||||
<div
|
||||
className="h-full bg-primary-500 transition-all duration-200"
|
||||
style={{ width: `${Math.min(100, Math.max(0, progress))}%` }}
|
||||
/>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
@@ -1,258 +0,0 @@
|
||||
import { fireEvent, screen, waitFor } from '@testing-library/react';
|
||||
import { beforeEach, describe, expect, it, type Mock, vi } from 'vitest';
|
||||
|
||||
import { renderWithProviders } from '../../../test/test-utils';
|
||||
import IntelligenceSettingsTab from '../IntelligenceSettingsTab';
|
||||
|
||||
// The orchestrator hits these RPCs on mount; the global tauriCommands mock
|
||||
// in setup.ts only stubs auth/service helpers, so we extend it here with
|
||||
// the local-AI surface the Settings tab uses, plus the new memory_tree
|
||||
// LLM-selector RPCs that replaced the dev-time mock backend.
|
||||
vi.mock('../../../utils/tauriCommands', () => ({
|
||||
isTauri: vi.fn(() => true),
|
||||
// memory_tree LLM selector — the BackendChooser polls these on mount and
|
||||
// again on every backend toggle. We track the value in a closure so the
|
||||
// set→get round-trip behaves like the real persistent core.
|
||||
memoryTreeGetLlm: vi.fn(),
|
||||
memoryTreeSetLlm: vi.fn(),
|
||||
openhumanLocalAiAssetsStatus: vi
|
||||
.fn()
|
||||
.mockResolvedValue({
|
||||
result: {
|
||||
chat: { state: 'NotInstalled', id: '', provider: 'ollama' },
|
||||
vision: { state: 'NotInstalled', id: '', provider: 'ollama' },
|
||||
embedding: { state: 'NotInstalled', id: '', provider: 'ollama' },
|
||||
stt: { state: 'NotInstalled', id: '', provider: 'ollama' },
|
||||
tts: { state: 'NotInstalled', id: '', provider: 'ollama' },
|
||||
quantization: 'q4_k_m',
|
||||
},
|
||||
}),
|
||||
openhumanLocalAiDiagnostics: vi.fn().mockResolvedValue({
|
||||
ollama_running: true,
|
||||
ollama_binary_path: '/usr/local/bin/ollama',
|
||||
installed_models: [
|
||||
{ name: 'gemma3:1b-it-qat', size: 1_700_000_000, modified_at: null },
|
||||
{ name: 'bge-m3', size: 1_300_000_000, modified_at: null },
|
||||
],
|
||||
expected: {
|
||||
chat_model: 'gemma3:1b-it-qat',
|
||||
chat_found: true,
|
||||
embedding_model: 'bge-m3',
|
||||
embedding_found: true,
|
||||
vision_model: '',
|
||||
vision_found: false,
|
||||
},
|
||||
issues: [],
|
||||
ok: true,
|
||||
}),
|
||||
openhumanLocalAiStatus: vi
|
||||
.fn()
|
||||
.mockResolvedValue({
|
||||
result: {
|
||||
state: 'Ready',
|
||||
model_id: 'gemma3:1b-it-qat',
|
||||
chat_model_id: 'gemma3:1b-it-qat',
|
||||
vision_model_id: '',
|
||||
embedding_model_id: 'bge-m3',
|
||||
stt_model_id: '',
|
||||
tts_voice_id: '',
|
||||
quantization: 'q4_k_m',
|
||||
vision_state: 'idle',
|
||||
vision_mode: 'off',
|
||||
embedding_state: 'Ready',
|
||||
stt_state: 'idle',
|
||||
tts_state: 'idle',
|
||||
provider: 'ollama',
|
||||
active_backend: 'cpu',
|
||||
last_latency_ms: 142,
|
||||
},
|
||||
}),
|
||||
openhumanLocalAiPresets: vi
|
||||
.fn()
|
||||
.mockResolvedValue({
|
||||
presets: [],
|
||||
recommended_tier: 'minimal',
|
||||
current_tier: 'minimal',
|
||||
device: {
|
||||
total_ram_bytes: 16_000_000_000,
|
||||
cpu_count: 8,
|
||||
cpu_brand: 'Test CPU',
|
||||
os_name: 'macos',
|
||||
os_version: '14',
|
||||
has_gpu: false,
|
||||
gpu_description: null,
|
||||
},
|
||||
local_ai_enabled: false,
|
||||
}),
|
||||
openhumanLocalAiDownloadAsset: vi
|
||||
.fn()
|
||||
.mockResolvedValue({
|
||||
result: {
|
||||
chat: { state: 'Ready', id: 'gemma3:1b-it-qat', provider: 'ollama' },
|
||||
vision: { state: 'NotInstalled', id: '', provider: 'ollama' },
|
||||
embedding: { state: 'Ready', id: 'bge-m3', provider: 'ollama' },
|
||||
stt: { state: 'NotInstalled', id: '', provider: 'ollama' },
|
||||
tts: { state: 'NotInstalled', id: '', provider: 'ollama' },
|
||||
quantization: 'q4_k_m',
|
||||
},
|
||||
}),
|
||||
}));
|
||||
|
||||
// Pull mocked references after vi.mock() has hoisted. Cast through unknown
|
||||
// because the import here is the typed wrapper module shape.
|
||||
const { memoryTreeGetLlm, memoryTreeSetLlm } =
|
||||
(await import('../../../utils/tauriCommands')) as unknown as {
|
||||
memoryTreeGetLlm: Mock;
|
||||
memoryTreeSetLlm: Mock;
|
||||
};
|
||||
|
||||
describe('IntelligenceSettingsTab', () => {
|
||||
beforeEach(() => {
|
||||
let backend: 'cloud' | 'local' = 'cloud';
|
||||
memoryTreeGetLlm.mockReset();
|
||||
memoryTreeSetLlm.mockReset();
|
||||
memoryTreeGetLlm.mockImplementation(async () => ({ current: backend }));
|
||||
// Accept both legacy (bare string) and the new request-object shape so
|
||||
// tests can assert on either call form.
|
||||
memoryTreeSetLlm.mockImplementation(
|
||||
async (req: 'cloud' | 'local' | { backend: 'cloud' | 'local' }) => {
|
||||
backend = typeof req === 'string' ? req : req.backend;
|
||||
return { current: backend };
|
||||
}
|
||||
);
|
||||
});
|
||||
|
||||
// Helper: bootstrap into Local mode so the model assignment + catalog
|
||||
// render. Cloud is the default; clicking the Advanced radio flips to
|
||||
// local and renders the Ollama-related sections.
|
||||
async function flipToLocal() {
|
||||
await waitFor(() => {
|
||||
expect(screen.getByText('AI backend')).toBeInTheDocument();
|
||||
});
|
||||
const radios = screen.getAllByRole('radio');
|
||||
const localCard = radios.find(el => /Advanced/.test(el.textContent ?? ''));
|
||||
expect(localCard).toBeDefined();
|
||||
fireEvent.click(localCard!);
|
||||
await waitFor(() => {
|
||||
expect(screen.getByText('Model assignment')).toBeInTheDocument();
|
||||
});
|
||||
}
|
||||
|
||||
it('renders the AI backend section in cloud mode (no local sections)', async () => {
|
||||
renderWithProviders(<IntelligenceSettingsTab />);
|
||||
|
||||
await waitFor(() => {
|
||||
expect(screen.getByText('AI backend')).toBeInTheDocument();
|
||||
});
|
||||
// Cloud is default — local-only sections are hidden so cloud users
|
||||
// never see Ollama-related UI.
|
||||
expect(screen.queryByText('Model assignment')).not.toBeInTheDocument();
|
||||
expect(screen.queryByText('Model catalog')).not.toBeInTheDocument();
|
||||
// Currently-loaded panel was removed entirely (was dev-debug noise).
|
||||
expect(screen.queryByText('Currently loaded')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('hides Model assignment in Cloud mode and reveals it in Local mode', async () => {
|
||||
renderWithProviders(<IntelligenceSettingsTab />);
|
||||
await flipToLocal();
|
||||
|
||||
// The new UI consolidates Extract + Summariser LLM into a single
|
||||
// Memory LLM picker (the underlying RPC still fans out to both
|
||||
// extract_model and summariser_model in config.toml).
|
||||
expect(screen.getByText('Memory LLM')).toBeInTheDocument();
|
||||
expect(screen.getByText('Embedder')).toBeInTheDocument();
|
||||
// Old separate dropdowns must be absent.
|
||||
expect(screen.queryByText('Extract LLM')).not.toBeInTheDocument();
|
||||
expect(screen.queryByText('Summariser LLM')).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('shows model catalog rows with sizes (in local mode)', async () => {
|
||||
renderWithProviders(<IntelligenceSettingsTab />);
|
||||
await flipToLocal();
|
||||
|
||||
await waitFor(() => {
|
||||
expect(screen.getAllByText('qwen2.5:0.5b').length).toBeGreaterThanOrEqual(1);
|
||||
});
|
||||
// Each model can appear in the Memory LLM dropdown AND the catalog,
|
||||
// so use getAllByText. Just confirm the catalog has at least one of
|
||||
// each curated entry rendered somewhere on the screen.
|
||||
expect(screen.getAllByText('gemma3:1b-it-qat').length).toBeGreaterThanOrEqual(1);
|
||||
expect(screen.getAllByText('gemma3:4b').length).toBeGreaterThanOrEqual(1);
|
||||
expect(screen.getAllByText('gemma3:12b-it-qat').length).toBeGreaterThanOrEqual(1);
|
||||
expect(screen.getAllByText('bge-m3').length).toBeGreaterThanOrEqual(1);
|
||||
|
||||
// 3.3 GB is unique to gemma3:4b in the catalog row meta.
|
||||
expect(screen.getByText('3.3 GB')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('renders a Download action for models that are not installed', async () => {
|
||||
renderWithProviders(<IntelligenceSettingsTab />);
|
||||
await flipToLocal();
|
||||
|
||||
// qwen2.5:0.5b is NOT in the diagnostics installed list, so it shows
|
||||
// a Download button.
|
||||
await waitFor(() => {
|
||||
expect(screen.getByText('qwen2.5:0.5b')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
const downloadButtons = screen.getAllByRole('button', { name: 'Download' });
|
||||
expect(downloadButtons.length).toBeGreaterThanOrEqual(1);
|
||||
});
|
||||
|
||||
it('reads the backend via memoryTreeGetLlm on mount and persists toggles via memoryTreeSetLlm', async () => {
|
||||
renderWithProviders(<IntelligenceSettingsTab />);
|
||||
|
||||
// Bootstrap: getMemoryTreeLlm must run once on mount.
|
||||
await waitFor(() => {
|
||||
expect(memoryTreeGetLlm).toHaveBeenCalled();
|
||||
});
|
||||
|
||||
// Click Local — setMemoryTreeLlm must be called with the request
|
||||
// object form `{ backend: 'local' }`. settingsApi.ts always normalizes
|
||||
// to the request-object shape because the wrapper now accepts both
|
||||
// forms but the API layer translates camelCase options through the
|
||||
// object shape. Model fields are absent so the corresponding
|
||||
// config keys stay untouched.
|
||||
const radios = screen.getAllByRole('radio');
|
||||
const localCard = radios.find(el => /Advanced/.test(el.textContent ?? ''));
|
||||
fireEvent.click(localCard!);
|
||||
|
||||
await waitFor(() => {
|
||||
expect(memoryTreeSetLlm).toHaveBeenCalledWith({ backend: 'local' });
|
||||
});
|
||||
|
||||
// The mocked setter persists state in the closure, so the bootstrap
|
||||
// value of any subsequent get_llm call would now be 'local' — sanity
|
||||
// check that the closure flipped.
|
||||
const after = await memoryTreeGetLlm();
|
||||
expect(after.current).toBe('local');
|
||||
});
|
||||
|
||||
it('persists Memory LLM dropdown changes via memoryTreeSetLlm with both extract_model and summariser_model', async () => {
|
||||
// The single Memory LLM picker fans out to BOTH extract_model and
|
||||
// summariser_model in one atomic write — the underlying schema keeps
|
||||
// the two keys separate so power users can split via the RPC, but the
|
||||
// UI consolidates them into one cognitive unit.
|
||||
renderWithProviders(<IntelligenceSettingsTab />);
|
||||
await flipToLocal();
|
||||
|
||||
// Reset call history so the assertion below is scoped to the
|
||||
// dropdown change, not the earlier backend toggle.
|
||||
memoryTreeSetLlm.mockClear();
|
||||
|
||||
// Pick a different memory LLM. `gemma3:12b-it-qat` is in the curated
|
||||
// catalog with both `extract` and `summariser` roles.
|
||||
const memorySelect = screen.getByLabelText(
|
||||
'Memory LLM (extract + summarise)'
|
||||
) as HTMLSelectElement;
|
||||
fireEvent.change(memorySelect, { target: { value: 'gemma3:12b-it-qat' } });
|
||||
|
||||
await waitFor(() => {
|
||||
expect(memoryTreeSetLlm).toHaveBeenCalledWith({
|
||||
backend: 'local',
|
||||
extract_model: 'gemma3:12b-it-qat',
|
||||
summariser_model: 'gemma3:12b-it-qat',
|
||||
});
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -1,142 +0,0 @@
|
||||
import { fireEvent, render, screen, waitFor } from '@testing-library/react';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import ModelCatalog from '../ModelCatalog';
|
||||
|
||||
describe('ModelCatalog', () => {
|
||||
it('renders one row per recommended model', () => {
|
||||
render(
|
||||
<ModelCatalog
|
||||
installedModelIds={[]}
|
||||
activeModelIds={[]}
|
||||
onDownload={vi.fn()}
|
||||
onUse={vi.fn()}
|
||||
/>
|
||||
);
|
||||
// Each id from RECOMMENDED_MODEL_CATALOG appears as a row title.
|
||||
expect(screen.getByText('qwen2.5:0.5b')).toBeInTheDocument();
|
||||
expect(screen.getByText('gemma3:1b-it-qat')).toBeInTheDocument();
|
||||
expect(screen.getByText('gemma3:4b')).toBeInTheDocument();
|
||||
expect(screen.getByText('gemma3:12b-it-qat')).toBeInTheDocument();
|
||||
expect(screen.getByText('bge-m3')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('shows "Download" for models that are not installed', () => {
|
||||
render(
|
||||
<ModelCatalog
|
||||
installedModelIds={[]}
|
||||
activeModelIds={[]}
|
||||
onDownload={vi.fn()}
|
||||
onUse={vi.fn()}
|
||||
/>
|
||||
);
|
||||
// Five models, all available → five Download buttons.
|
||||
expect(screen.getAllByRole('button', { name: /download/i })).toHaveLength(5);
|
||||
expect(screen.getAllByText('not downloaded').length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it('shows "Use" for installed-but-not-active models, "in use" for active', () => {
|
||||
render(
|
||||
<ModelCatalog
|
||||
installedModelIds={['gemma3:1b-it-qat', 'bge-m3:latest']}
|
||||
activeModelIds={['bge-m3']}
|
||||
onDownload={vi.fn()}
|
||||
onUse={vi.fn()}
|
||||
/>
|
||||
);
|
||||
// bge-m3 is installed AND active → "in use" pill, no Use button for it.
|
||||
expect(screen.getAllByText('in use').length).toBeGreaterThan(0);
|
||||
// gemma3 is installed but not active → Use button visible.
|
||||
expect(screen.getByRole('button', { name: 'Use' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('matches `bge-m3` against `bge-m3:latest` via the :latest normalization', () => {
|
||||
// Ollama tags everything as `:latest` by default; the catalog uses bare
|
||||
// names. The component must treat them as the same id.
|
||||
render(
|
||||
<ModelCatalog
|
||||
installedModelIds={['bge-m3:latest']}
|
||||
activeModelIds={[]}
|
||||
onDownload={vi.fn()}
|
||||
onUse={vi.fn()}
|
||||
/>
|
||||
);
|
||||
// bge-m3 row is now in the "installed" state — at least one Use button
|
||||
// appears (for bge-m3 specifically).
|
||||
expect(screen.getByRole('button', { name: 'Use' })).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('fires onUse with the matching model when the Use button is clicked', () => {
|
||||
const onUse = vi.fn();
|
||||
render(
|
||||
<ModelCatalog
|
||||
installedModelIds={['gemma3:4b']}
|
||||
activeModelIds={[]}
|
||||
onDownload={vi.fn()}
|
||||
onUse={onUse}
|
||||
/>
|
||||
);
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Use' }));
|
||||
expect(onUse).toHaveBeenCalledTimes(1);
|
||||
expect(onUse.mock.calls[0][0]).toMatchObject({ id: 'gemma3:4b' });
|
||||
});
|
||||
|
||||
it('renders Delete buttons only when onDelete is provided', () => {
|
||||
const { rerender } = render(
|
||||
<ModelCatalog
|
||||
installedModelIds={['gemma3:4b']}
|
||||
activeModelIds={[]}
|
||||
onDownload={vi.fn()}
|
||||
onUse={vi.fn()}
|
||||
/>
|
||||
);
|
||||
expect(screen.queryByLabelText('Delete model')).toBeNull();
|
||||
|
||||
rerender(
|
||||
<ModelCatalog
|
||||
installedModelIds={['gemma3:4b']}
|
||||
activeModelIds={[]}
|
||||
onDownload={vi.fn()}
|
||||
onUse={vi.fn()}
|
||||
onDelete={vi.fn()}
|
||||
/>
|
||||
);
|
||||
expect(screen.getByLabelText('Delete model')).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('shows a progress bar while a download is in flight, then clears it', async () => {
|
||||
let resolveDownload!: () => void;
|
||||
const onDownload = vi.fn(
|
||||
() =>
|
||||
new Promise<void>(resolve => {
|
||||
resolveDownload = resolve;
|
||||
})
|
||||
);
|
||||
render(
|
||||
<ModelCatalog
|
||||
installedModelIds={[]}
|
||||
activeModelIds={[]}
|
||||
onDownload={onDownload}
|
||||
onUse={vi.fn()}
|
||||
/>
|
||||
);
|
||||
|
||||
fireEvent.click(screen.getAllByRole('button', { name: /download/i })[0]);
|
||||
|
||||
// Mid-flight: a progressbar is rendered for that row.
|
||||
await waitFor(() => {
|
||||
expect(screen.queryAllByRole('progressbar').length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
resolveDownload();
|
||||
// After settle (~600 ms on success), the bar disappears and the row
|
||||
// returns to its post-install state. We just confirm the state
|
||||
// eventually clears — not the exact timing.
|
||||
await waitFor(
|
||||
() => {
|
||||
expect(screen.queryByRole('progressbar')).toBeNull();
|
||||
},
|
||||
{ timeout: 2000 }
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -9,8 +9,11 @@ import {
|
||||
} from '../../../utils/localAiHelpers';
|
||||
import {
|
||||
type ApplyPresetResult,
|
||||
type LlmBackend,
|
||||
type LocalAiDownloadsProgress,
|
||||
type LocalAiStatus,
|
||||
memoryTreeGetLlm,
|
||||
memoryTreeSetLlm,
|
||||
openhumanGetConfig,
|
||||
openhumanLocalAiApplyPreset,
|
||||
openhumanLocalAiDownload,
|
||||
@@ -59,6 +62,16 @@ const LocalModelPanel = () => {
|
||||
const [usageError, setUsageError] = useState('');
|
||||
const [usageSaving, setUsageSaving] = useState(false);
|
||||
|
||||
// Memory summarizer backend lives in `memory_tree.llm_backend`, which is
|
||||
// outside the `local_ai.usage.*` surface — so it has its own RPC pair
|
||||
// (`memoryTreeGetLlm` / `memoryTreeSetLlm`). This is now the only UI
|
||||
// surface for the cloud/local toggle (the Intelligence Memory tab's
|
||||
// BackendChooser was removed in this PR to eliminate the duplicate
|
||||
// control surface). The tab's local-only Ollama model picker reads
|
||||
// the same field at mount to decide visibility.
|
||||
const [summarizerBackend, setSummarizerBackend] = useState<LlmBackend>('cloud');
|
||||
const [summarizerSaving, setSummarizerSaving] = useState(false);
|
||||
|
||||
const progress = useMemo(() => {
|
||||
const downloadProgress = progressFromDownloads(downloads);
|
||||
if (downloadProgress != null) return downloadProgress;
|
||||
@@ -143,11 +156,48 @@ const LocalModelPanel = () => {
|
||||
}
|
||||
};
|
||||
|
||||
const loadSummarizerBackend = async () => {
|
||||
try {
|
||||
const resp = await memoryTreeGetLlm();
|
||||
if (resp?.current === 'local' || resp?.current === 'cloud') {
|
||||
setSummarizerBackend(resp.current);
|
||||
}
|
||||
} catch (err) {
|
||||
// Non-fatal — the row stays at its default (cloud) and the user
|
||||
// can still flip it; the next save attempt will surface the error.
|
||||
console.warn('[local-model-panel] memoryTreeGetLlm failed', err);
|
||||
}
|
||||
};
|
||||
|
||||
const updateSummarizerBackend = async (next: LlmBackend) => {
|
||||
// Belt-and-braces: the checkbox is already `disabled` when the
|
||||
// master runtime is off or a save is in flight, but the rendered
|
||||
// disabled attribute only stops real-browser click events — JSDOM
|
||||
// / fireEvent ignore it. Mirror the gate here so programmatic
|
||||
// dispatch can't sneak past the master switch either.
|
||||
if (!usageFlags.runtime_enabled || summarizerSaving) return;
|
||||
const prev = summarizerBackend;
|
||||
setSummarizerBackend(next);
|
||||
setSummarizerSaving(true);
|
||||
setUsageError('');
|
||||
try {
|
||||
await memoryTreeSetLlm({ backend: next });
|
||||
} catch (err) {
|
||||
// Roll back the optimistic toggle and surface the error.
|
||||
setSummarizerBackend(prev);
|
||||
const msg = err instanceof Error ? err.message : 'Failed to save memory summarizer backend';
|
||||
setUsageError(msg);
|
||||
} finally {
|
||||
setSummarizerSaving(false);
|
||||
}
|
||||
};
|
||||
|
||||
useEffect(() => {
|
||||
const initialLoad = window.setTimeout(() => {
|
||||
void loadStatus();
|
||||
void loadPresets();
|
||||
void loadUsage();
|
||||
void loadSummarizerBackend();
|
||||
}, 0);
|
||||
const timer = window.setInterval(() => {
|
||||
void loadStatus();
|
||||
@@ -366,6 +416,30 @@ const LocalModelPanel = () => {
|
||||
</label>
|
||||
|
||||
<div className={`space-y-2 pl-6 ${usageFlags.runtime_enabled ? '' : 'opacity-50'}`}>
|
||||
{/* Memory summarizer is special: it writes `memory_tree.llm_backend`,
|
||||
not `local_ai.usage.*`. This is now the sole UI surface for
|
||||
that field (the Intelligence Memory tab's BackendChooser was
|
||||
removed); the tab's Ollama model picker still reads it at
|
||||
mount to gate visibility. */}
|
||||
<label
|
||||
className="flex items-start gap-3 cursor-pointer"
|
||||
data-testid="local-ai-usage-memory-summarizer">
|
||||
<input
|
||||
type="checkbox"
|
||||
className="mt-0.5"
|
||||
checked={summarizerBackend === 'local'}
|
||||
disabled={!usageFlags.runtime_enabled || summarizerSaving}
|
||||
onChange={e => void updateSummarizerBackend(e.target.checked ? 'local' : 'cloud')}
|
||||
/>
|
||||
<div>
|
||||
<div className="text-sm text-stone-900">Memory summarizer</div>
|
||||
<div className="text-xs text-stone-500">
|
||||
Run memory-tree extract + summarise locally instead of in the cloud. The local
|
||||
model used comes from the Model Tier preset above.
|
||||
</div>
|
||||
</div>
|
||||
</label>
|
||||
|
||||
{(
|
||||
[
|
||||
{
|
||||
|
||||
@@ -8,6 +8,8 @@ import {
|
||||
isTauri,
|
||||
type LocalAiDownloadsProgress,
|
||||
type LocalAiStatus,
|
||||
memoryTreeGetLlm,
|
||||
memoryTreeSetLlm,
|
||||
openhumanGetConfig,
|
||||
openhumanLocalAiDownload,
|
||||
openhumanLocalAiDownloadAllAssets,
|
||||
@@ -21,6 +23,8 @@ import LocalModelPanel from '../LocalModelPanel';
|
||||
|
||||
vi.mock('../../../../utils/tauriCommands', () => ({
|
||||
isTauri: vi.fn(() => true),
|
||||
memoryTreeGetLlm: vi.fn(),
|
||||
memoryTreeSetLlm: vi.fn(),
|
||||
openhumanGetConfig: vi.fn(),
|
||||
openhumanLocalAiDownload: vi.fn(),
|
||||
openhumanLocalAiDownloadAllAssets: vi.fn(),
|
||||
@@ -124,6 +128,11 @@ describe('LocalModelPanel — usage flags', () => {
|
||||
};
|
||||
return makeSnapshot(runtime);
|
||||
});
|
||||
|
||||
// Memory summarizer backend defaults to cloud; tests that need a
|
||||
// specific seed value override this in the test body.
|
||||
vi.mocked(memoryTreeGetLlm).mockResolvedValue({ current: 'cloud' });
|
||||
vi.mocked(memoryTreeSetLlm).mockResolvedValue({ current: 'local' });
|
||||
});
|
||||
|
||||
it('renders all five usage toggles with sub-flags disabled when runtime is off', async () => {
|
||||
@@ -215,4 +224,86 @@ describe('LocalModelPanel — usage flags', () => {
|
||||
expect(openhumanUpdateLocalAiSettings).toHaveBeenCalledWith({ usage_embeddings: true });
|
||||
});
|
||||
});
|
||||
|
||||
// The Memory summarizer checkbox is special — it writes
|
||||
// `memory_tree.llm_backend` via memoryTreeSetLlm (the same field the
|
||||
// removed Intelligence → Memory BackendChooser used to edit), not
|
||||
// `local_ai.usage.*`. State seeds from memoryTreeGetLlm on mount.
|
||||
it('seeds the Memory summarizer checkbox state from memoryTreeGetLlm', async () => {
|
||||
vi.mocked(memoryTreeGetLlm).mockResolvedValueOnce({ current: 'local' });
|
||||
runtime.runtime_enabled = true;
|
||||
renderWithProviders(<LocalModelPanel />, { initialEntries: ['/settings/local-model'] });
|
||||
|
||||
const summarizerLabel = await screen.findByText('Memory summarizer');
|
||||
const checkbox = summarizerLabel
|
||||
.closest('label')
|
||||
?.querySelector('input[type="checkbox"]') as HTMLInputElement;
|
||||
expect(checkbox).toBeTruthy();
|
||||
await waitFor(() => {
|
||||
expect(memoryTreeGetLlm).toHaveBeenCalled();
|
||||
});
|
||||
await waitFor(() => {
|
||||
expect(checkbox.checked).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
it('flips the Memory summarizer checkbox and persists via memoryTreeSetLlm', async () => {
|
||||
runtime.runtime_enabled = true;
|
||||
renderWithProviders(<LocalModelPanel />, { initialEntries: ['/settings/local-model'] });
|
||||
|
||||
const summarizerLabel = await screen.findByText('Memory summarizer');
|
||||
const checkbox = summarizerLabel
|
||||
.closest('label')
|
||||
?.querySelector('input[type="checkbox"]') as HTMLInputElement;
|
||||
// The checkbox starts disabled until the async loadUsage() flips
|
||||
// `usageFlags.runtime_enabled` to true; wait for that before clicking.
|
||||
await waitFor(() => {
|
||||
expect(checkbox).not.toBeDisabled();
|
||||
});
|
||||
fireEvent.click(checkbox);
|
||||
|
||||
await waitFor(() => {
|
||||
expect(memoryTreeSetLlm).toHaveBeenCalledWith({ backend: 'local' });
|
||||
});
|
||||
});
|
||||
|
||||
it('rolls back the Memory summarizer optimistic toggle when memoryTreeSetLlm fails', async () => {
|
||||
runtime.runtime_enabled = true;
|
||||
vi.mocked(memoryTreeSetLlm).mockRejectedValueOnce(new Error('save: backend down'));
|
||||
renderWithProviders(<LocalModelPanel />, { initialEntries: ['/settings/local-model'] });
|
||||
|
||||
const summarizerLabel = await screen.findByText('Memory summarizer');
|
||||
const checkbox = summarizerLabel
|
||||
.closest('label')
|
||||
?.querySelector('input[type="checkbox"]') as HTMLInputElement;
|
||||
await waitFor(() => {
|
||||
expect(checkbox).not.toBeDisabled();
|
||||
});
|
||||
fireEvent.click(checkbox);
|
||||
|
||||
// The error message surfaces in the shared usageError block.
|
||||
await screen.findByText('save: backend down');
|
||||
// And the checkbox rolls back to its prior state (cloud → unchecked).
|
||||
await waitFor(() => {
|
||||
expect(checkbox.checked).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
it('does not call memoryTreeSetLlm when the Memory summarizer is disabled (runtime off)', async () => {
|
||||
// runtime is OFF by default — summarizer checkbox should be disabled
|
||||
// and clicks should not fire a setLlm call.
|
||||
renderWithProviders(<LocalModelPanel />, { initialEntries: ['/settings/local-model'] });
|
||||
|
||||
const summarizerLabel = await screen.findByText('Memory summarizer');
|
||||
const checkbox = summarizerLabel
|
||||
.closest('label')
|
||||
?.querySelector('input[type="checkbox"]') as HTMLInputElement;
|
||||
expect(checkbox).toBeDisabled();
|
||||
// Exercise the disabled-click path — fireEvent dispatches even on
|
||||
// disabled inputs (it bypasses React's synthetic event guard), so
|
||||
// this confirms the handler doesn't fire `setLlm` because of the
|
||||
// gating, not just because no click happened.
|
||||
fireEvent.click(checkbox);
|
||||
expect(memoryTreeSetLlm).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,174 +0,0 @@
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import {
|
||||
capabilityForModel,
|
||||
downloadAsset,
|
||||
fetchInstalledAssets,
|
||||
fetchInstalledModels,
|
||||
fetchLocalAiStatus,
|
||||
fetchPresets,
|
||||
formatBytes,
|
||||
getMemoryTreeLlm,
|
||||
type ModelDescriptor,
|
||||
setMemoryTreeLlm,
|
||||
} from '../settingsApi';
|
||||
|
||||
// Stub the underlying tauri-command wrappers; we're testing the
|
||||
// camelCase→snake_case translation + simple try/catch shells, not the
|
||||
// RPC plumbing.
|
||||
vi.mock('../../../utils/tauriCommands', () => ({
|
||||
isTauri: vi.fn(() => true),
|
||||
memoryTreeGetLlm: vi.fn(),
|
||||
memoryTreeSetLlm: vi.fn(),
|
||||
openhumanLocalAiAssetsStatus: vi.fn(),
|
||||
openhumanLocalAiStatus: vi.fn(),
|
||||
openhumanLocalAiDiagnostics: vi.fn(),
|
||||
openhumanLocalAiPresets: vi.fn(),
|
||||
openhumanLocalAiDownloadAsset: vi.fn(),
|
||||
}));
|
||||
|
||||
const tauri = (await import('../../../utils/tauriCommands')) as unknown as {
|
||||
memoryTreeGetLlm: ReturnType<typeof vi.fn>;
|
||||
memoryTreeSetLlm: ReturnType<typeof vi.fn>;
|
||||
openhumanLocalAiAssetsStatus: ReturnType<typeof vi.fn>;
|
||||
openhumanLocalAiStatus: ReturnType<typeof vi.fn>;
|
||||
openhumanLocalAiDiagnostics: ReturnType<typeof vi.fn>;
|
||||
openhumanLocalAiPresets: ReturnType<typeof vi.fn>;
|
||||
openhumanLocalAiDownloadAsset: ReturnType<typeof vi.fn>;
|
||||
};
|
||||
|
||||
beforeEach(() => {
|
||||
Object.values(tauri).forEach(fn => fn.mockReset());
|
||||
});
|
||||
|
||||
describe('getMemoryTreeLlm', () => {
|
||||
it('returns the current backend value from the RPC', async () => {
|
||||
tauri.memoryTreeGetLlm.mockResolvedValueOnce({ current: 'cloud' });
|
||||
await expect(getMemoryTreeLlm()).resolves.toBe('cloud');
|
||||
expect(tauri.memoryTreeGetLlm).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe('setMemoryTreeLlm', () => {
|
||||
it('passes only the backend when no options are supplied', async () => {
|
||||
tauri.memoryTreeSetLlm.mockResolvedValueOnce({ current: 'cloud' });
|
||||
await setMemoryTreeLlm('cloud');
|
||||
expect(tauri.memoryTreeSetLlm).toHaveBeenCalledWith({ backend: 'cloud' });
|
||||
});
|
||||
|
||||
it('translates camelCase options to snake_case wire fields and passes only those that are set', async () => {
|
||||
tauri.memoryTreeSetLlm.mockResolvedValueOnce({ current: 'local' });
|
||||
await setMemoryTreeLlm('local', { extractModel: 'a:b', summariserModel: 'c:d' });
|
||||
expect(tauri.memoryTreeSetLlm).toHaveBeenCalledWith({
|
||||
backend: 'local',
|
||||
extract_model: 'a:b',
|
||||
summariser_model: 'c:d',
|
||||
});
|
||||
// cloudModel was unset → cloud_model must NOT be on the wire payload.
|
||||
expect(tauri.memoryTreeSetLlm.mock.calls[0][0]).not.toHaveProperty('cloud_model');
|
||||
});
|
||||
|
||||
it('returns the effective backend value the core decided on', async () => {
|
||||
tauri.memoryTreeSetLlm.mockResolvedValueOnce({ current: 'cloud' });
|
||||
const out = await setMemoryTreeLlm('local');
|
||||
expect(out).toEqual({ effective: 'cloud' });
|
||||
});
|
||||
});
|
||||
|
||||
describe('fetchInstalledAssets', () => {
|
||||
it('unwraps the `result` field on success', async () => {
|
||||
tauri.openhumanLocalAiAssetsStatus.mockResolvedValueOnce({ result: { foo: 1 } });
|
||||
await expect(fetchInstalledAssets()).resolves.toEqual({ foo: 1 });
|
||||
});
|
||||
it('swallows RPC errors and returns null', async () => {
|
||||
tauri.openhumanLocalAiAssetsStatus.mockRejectedValueOnce(new Error('boom'));
|
||||
await expect(fetchInstalledAssets()).resolves.toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('fetchLocalAiStatus', () => {
|
||||
it('returns null on RPC failure', async () => {
|
||||
tauri.openhumanLocalAiStatus.mockRejectedValueOnce(new Error('nope'));
|
||||
await expect(fetchLocalAiStatus()).resolves.toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('fetchInstalledModels', () => {
|
||||
it('returns the installed_models array', async () => {
|
||||
tauri.openhumanLocalAiDiagnostics.mockResolvedValueOnce({
|
||||
installed_models: [{ name: 'bge-m3', size_bytes: 1 }],
|
||||
});
|
||||
const got = await fetchInstalledModels();
|
||||
expect(got).toHaveLength(1);
|
||||
expect(got[0]?.name).toBe('bge-m3');
|
||||
});
|
||||
it('returns [] when the RPC rejects', async () => {
|
||||
tauri.openhumanLocalAiDiagnostics.mockRejectedValueOnce(new Error('rpc down'));
|
||||
await expect(fetchInstalledModels()).resolves.toEqual([]);
|
||||
});
|
||||
it('returns [] when installed_models is missing', async () => {
|
||||
tauri.openhumanLocalAiDiagnostics.mockResolvedValueOnce({});
|
||||
await expect(fetchInstalledModels()).resolves.toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('fetchPresets', () => {
|
||||
it('forwards the response on success', async () => {
|
||||
tauri.openhumanLocalAiPresets.mockResolvedValueOnce({ presets: [] });
|
||||
await expect(fetchPresets()).resolves.toEqual({ presets: [] });
|
||||
});
|
||||
it('returns null on failure', async () => {
|
||||
tauri.openhumanLocalAiPresets.mockRejectedValueOnce(new Error('x'));
|
||||
await expect(fetchPresets()).resolves.toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('downloadAsset', () => {
|
||||
it('returns the result envelope on success', async () => {
|
||||
tauri.openhumanLocalAiDownloadAsset.mockResolvedValueOnce({ result: { ok: true } });
|
||||
await expect(downloadAsset('chat')).resolves.toEqual({ ok: true });
|
||||
expect(tauri.openhumanLocalAiDownloadAsset).toHaveBeenCalledWith('chat');
|
||||
});
|
||||
it('returns null on failure (and does not throw)', async () => {
|
||||
tauri.openhumanLocalAiDownloadAsset.mockRejectedValueOnce(new Error('disconnected'));
|
||||
await expect(downloadAsset('embedding')).resolves.toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('capabilityForModel', () => {
|
||||
const make = (roles: ModelDescriptor['roles']): ModelDescriptor => ({
|
||||
id: 'x',
|
||||
size: '0',
|
||||
approxBytes: 0,
|
||||
ramHint: '0',
|
||||
category: 'fast',
|
||||
note: '',
|
||||
roles,
|
||||
});
|
||||
it('maps embedder → embedding', () => {
|
||||
expect(capabilityForModel(make(['embedder']))).toBe('embedding');
|
||||
});
|
||||
it('maps extract / summariser → chat', () => {
|
||||
expect(capabilityForModel(make(['extract']))).toBe('chat');
|
||||
expect(capabilityForModel(make(['summariser']))).toBe('chat');
|
||||
expect(capabilityForModel(make(['extract', 'summariser']))).toBe('chat');
|
||||
});
|
||||
it('returns null when no role binds to a known capability', () => {
|
||||
expect(capabilityForModel(make([]))).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('formatBytes', () => {
|
||||
it('falls back to em-dash for non-finite or zero', () => {
|
||||
expect(formatBytes(Number.NaN)).toBe('—');
|
||||
expect(formatBytes(0)).toBe('—');
|
||||
expect(formatBytes(Number.POSITIVE_INFINITY)).toBe('—');
|
||||
});
|
||||
it('formats GB for >= 1 GiB', () => {
|
||||
expect(formatBytes(2.5 * 1024 ** 3)).toBe('2.5 GB');
|
||||
});
|
||||
it('formats MB (rounded) for sub-GB inputs', () => {
|
||||
expect(formatBytes(150 * 1024 ** 2)).toBe('150 MB');
|
||||
expect(formatBytes(1024 ** 2 + 1)).toBe('1 MB');
|
||||
});
|
||||
});
|
||||
@@ -1,269 +0,0 @@
|
||||
/**
|
||||
* Settings tab API layer for the Intelligence page.
|
||||
*
|
||||
* Wraps the existing `local_ai_*` core RPCs (re-exported with cleaner names)
|
||||
* and the canonical `openhuman.memory_tree_get_llm` / `set_llm` JSON-RPC
|
||||
* methods that drive the AI-backend selector. Both come from the shared
|
||||
* `utils/tauriCommands` barrel.
|
||||
*
|
||||
* Logging convention: `[intelligence-settings-api]` prefix for grep-friendly
|
||||
* tracing of the new flow per the project debug-logging rule.
|
||||
*/
|
||||
import {
|
||||
type LlmBackend,
|
||||
type LocalAiAssetsStatus,
|
||||
type LocalAiDiagnostics,
|
||||
type LocalAiStatus,
|
||||
memoryTreeGetLlm,
|
||||
memoryTreeSetLlm,
|
||||
openhumanLocalAiAssetsStatus,
|
||||
openhumanLocalAiDiagnostics,
|
||||
openhumanLocalAiDownloadAsset,
|
||||
openhumanLocalAiPresets,
|
||||
openhumanLocalAiStatus,
|
||||
type PresetsResponse,
|
||||
} from '../../utils/tauriCommands';
|
||||
|
||||
/**
|
||||
* AI backend the assistant is currently using for chat. Re-exports the
|
||||
* canonical `LlmBackend` from the wrapper so both names remain valid as
|
||||
* call-sites migrate.
|
||||
*/
|
||||
export type Backend = LlmBackend;
|
||||
|
||||
/** Static descriptor used by ModelAssignment + ModelCatalog. */
|
||||
export interface ModelDescriptor {
|
||||
/** Ollama-style identifier (e.g. `qwen2.5:0.5b`). */
|
||||
id: string;
|
||||
/** Pretty label shown in the UI (defaults to `id` when omitted). */
|
||||
label?: string;
|
||||
/** Human-readable disk size, e.g. `400 MB`. */
|
||||
size: string;
|
||||
/** Bytes — approximate; surfaced for sort / filter. */
|
||||
approxBytes: number;
|
||||
/** Approx RAM hint, e.g. `≤4 GB RAM`. */
|
||||
ramHint: string;
|
||||
/** Speed / quality tier — used for the inline annotation under each row. */
|
||||
category: 'fast' | 'balanced' | 'high quality' | 'embedder';
|
||||
/** One-sentence note about when to pick this model. */
|
||||
note: string;
|
||||
/** Role(s) this model is suitable for. */
|
||||
roles: ReadonlyArray<'extract' | 'summariser' | 'embedder'>;
|
||||
}
|
||||
|
||||
export type ModelRole = 'extract' | 'summariser' | 'embedder';
|
||||
|
||||
/**
|
||||
* Hard-coded recommended catalog. In a future wave this should come from
|
||||
* a `local_ai.recommended_catalog` RPC; for v1 we ship a curated list so
|
||||
* the UI is fully populated without a server roundtrip.
|
||||
*/
|
||||
export const RECOMMENDED_MODEL_CATALOG: ReadonlyArray<ModelDescriptor> = [
|
||||
{
|
||||
id: 'qwen2.5:0.5b',
|
||||
size: '400 MB',
|
||||
approxBytes: 400 * 1024 * 1024,
|
||||
ramHint: '≤4 GB RAM',
|
||||
category: 'fast',
|
||||
note: 'compact, lower quality',
|
||||
roles: ['extract'],
|
||||
},
|
||||
{
|
||||
id: 'gemma3:1b-it-qat',
|
||||
size: '1.0 GB',
|
||||
approxBytes: Math.round(1.0 * 1024 * 1024 * 1024),
|
||||
ramHint: '≤4 GB RAM',
|
||||
category: 'fast',
|
||||
note: 'compact Gemma; OK on laptops without a GPU',
|
||||
roles: ['extract', 'summariser'],
|
||||
},
|
||||
{
|
||||
id: 'gemma3:4b',
|
||||
size: '3.3 GB',
|
||||
approxBytes: Math.round(3.3 * 1024 * 1024 * 1024),
|
||||
ramHint: '≤8 GB RAM',
|
||||
category: 'balanced',
|
||||
note: 'default summariser — coherent abstractive output',
|
||||
roles: ['extract', 'summariser'],
|
||||
},
|
||||
{
|
||||
id: 'gemma3:12b-it-qat',
|
||||
size: '8.9 GB',
|
||||
approxBytes: Math.round(8.9 * 1024 * 1024 * 1024),
|
||||
ramHint: '≥16 GB RAM',
|
||||
category: 'high quality',
|
||||
note: 'larger Gemma; sharper summaries on capable hardware',
|
||||
roles: ['extract', 'summariser'],
|
||||
},
|
||||
{
|
||||
id: 'bge-m3',
|
||||
size: '1.3 GB',
|
||||
approxBytes: Math.round(1.3 * 1024 * 1024 * 1024),
|
||||
ramHint: '≥4 GB RAM',
|
||||
category: 'embedder',
|
||||
note: 'required for embeddings',
|
||||
roles: ['embedder'],
|
||||
},
|
||||
];
|
||||
|
||||
export const DEFAULT_EXTRACT_MODEL = 'gemma3:4b';
|
||||
export const DEFAULT_SUMMARISER_MODEL = 'gemma3:4b';
|
||||
export const REQUIRED_EMBEDDER_MODEL = 'bge-m3';
|
||||
|
||||
/**
|
||||
* Reads the currently configured chat backend from the core.
|
||||
*
|
||||
* Backed by `openhuman.memory_tree_get_llm` — the value persists across
|
||||
* sidecar restarts via `config.toml`.
|
||||
*/
|
||||
export async function getMemoryTreeLlm(): Promise<Backend> {
|
||||
console.debug('[intelligence-settings-api] getMemoryTreeLlm: entry');
|
||||
const resp = await memoryTreeGetLlm();
|
||||
console.debug('[intelligence-settings-api] getMemoryTreeLlm: exit current=%s', resp.current);
|
||||
return resp.current;
|
||||
}
|
||||
|
||||
/**
|
||||
* Optional per-role model picks for {@link setMemoryTreeLlm}. Field names
|
||||
* are camelCase here to match TS conventions; the wrapper translates them
|
||||
* to the snake_case wire shape the Rust `SetLlmRequest` expects:
|
||||
*
|
||||
* | TS option | Rust / wire field | Targets `memory_tree.*` |
|
||||
* | ----------------- | ------------------- | ----------------------- |
|
||||
* | `cloudModel` | `cloud_model` | `cloud_llm_model` |
|
||||
* | `extractModel` | `extract_model` | `llm_extractor_model` |
|
||||
* | `summariserModel` | `summariser_model` | `llm_summariser_model` |
|
||||
*
|
||||
* Each field follows "absent → unchanged, present → overwritten" so a
|
||||
* caller flipping just the backend doesn't have to re-supply every model
|
||||
* id, and a caller persisting just one role doesn't have to re-supply
|
||||
* the others.
|
||||
*/
|
||||
export interface SetMemoryTreeLlmOptions {
|
||||
cloudModel?: string;
|
||||
extractModel?: string;
|
||||
summariserModel?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Switches the chat backend and (optionally) persists per-role model
|
||||
* choices in the same atomic `config.toml` write. Returns the effective
|
||||
* value the core agreed on — today the handler accepts the input
|
||||
* verbatim, but a future revision may downgrade `local` → `cloud` when
|
||||
* the host can't satisfy the local minimums.
|
||||
*
|
||||
* Backed by `openhuman.memory_tree_set_llm`.
|
||||
*
|
||||
* Existing one-arg callers — `setMemoryTreeLlm('cloud')` — keep working
|
||||
* unchanged because `options` is optional.
|
||||
*/
|
||||
export async function setMemoryTreeLlm(
|
||||
next: Backend,
|
||||
options?: SetMemoryTreeLlmOptions
|
||||
): Promise<{ effective: Backend }> {
|
||||
console.debug(
|
||||
'[intelligence-settings-api] setMemoryTreeLlm: entry next=%s cloudModel=%s extractModel=%s summariserModel=%s',
|
||||
next,
|
||||
options?.cloudModel ?? '<none>',
|
||||
options?.extractModel ?? '<none>',
|
||||
options?.summariserModel ?? '<none>'
|
||||
);
|
||||
// camelCase → snake_case translation lives here, in one place. The
|
||||
// wrapper layer just forwards the snake_case shape to the wire.
|
||||
const resp = await memoryTreeSetLlm({
|
||||
backend: next,
|
||||
...(options?.cloudModel !== undefined && { cloud_model: options.cloudModel }),
|
||||
...(options?.extractModel !== undefined && { extract_model: options.extractModel }),
|
||||
...(options?.summariserModel !== undefined && { summariser_model: options.summariserModel }),
|
||||
});
|
||||
console.debug('[intelligence-settings-api] setMemoryTreeLlm: exit effective=%s', resp.current);
|
||||
return { effective: resp.current };
|
||||
}
|
||||
|
||||
/** Re-export the existing assets status fetch with a friendlier name. */
|
||||
export async function fetchInstalledAssets(): Promise<LocalAiAssetsStatus | null> {
|
||||
try {
|
||||
const response = await openhumanLocalAiAssetsStatus();
|
||||
return response.result;
|
||||
} catch (err) {
|
||||
console.debug('[intelligence-settings-api] fetchInstalledAssets failed', err);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch local AI status (includes per-capability state + last latency).
|
||||
* Used by `CurrentlyLoaded` to render Ollama-side telemetry.
|
||||
*/
|
||||
export async function fetchLocalAiStatus(): Promise<LocalAiStatus | null> {
|
||||
try {
|
||||
const response = await openhumanLocalAiStatus();
|
||||
return response.result;
|
||||
} catch (err) {
|
||||
console.debug('[intelligence-settings-api] fetchLocalAiStatus failed', err);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Reach into the existing diagnostics RPC for the list of installed Ollama
|
||||
* models. The diagnostics endpoint already enumerates them and is the
|
||||
* cleanest single source of truth — we do not duplicate the model table.
|
||||
*/
|
||||
export async function fetchInstalledModels(): Promise<LocalAiDiagnostics['installed_models']> {
|
||||
try {
|
||||
const response = await openhumanLocalAiDiagnostics();
|
||||
return response.installed_models ?? [];
|
||||
} catch (err) {
|
||||
console.debug('[intelligence-settings-api] fetchInstalledModels failed', err);
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
export async function fetchPresets(): Promise<PresetsResponse | null> {
|
||||
try {
|
||||
return await openhumanLocalAiPresets();
|
||||
} catch (err) {
|
||||
console.debug('[intelligence-settings-api] fetchPresets failed', err);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Trigger a download for a capability (chat / vision / embedding / stt / tts).
|
||||
* Used by ModelCatalog when the user clicks "Download".
|
||||
*
|
||||
* NOTE: the real RPC is per-capability, not per-model-id, so the catalog
|
||||
* picks the closest matching capability. This is acceptable for v1; future
|
||||
* iterations can swap in a per-model RPC.
|
||||
*/
|
||||
export async function downloadAsset(
|
||||
capability: 'chat' | 'vision' | 'embedding' | 'stt' | 'tts'
|
||||
): Promise<LocalAiAssetsStatus | null> {
|
||||
try {
|
||||
const response = await openhumanLocalAiDownloadAsset(capability);
|
||||
return response.result;
|
||||
} catch (err) {
|
||||
console.debug('[intelligence-settings-api] downloadAsset failed', { capability, err });
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Map a model descriptor to the closest capability bucket the core exposes. */
|
||||
export function capabilityForModel(model: ModelDescriptor): 'chat' | 'embedding' | null {
|
||||
if (model.roles.includes('embedder')) return 'embedding';
|
||||
if (model.roles.includes('extract') || model.roles.includes('summariser')) return 'chat';
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Cheap pretty-printer for a byte count. Mirrors the `JetBrains Mono`-style
|
||||
* compact format we want in the technical-readout sections.
|
||||
*/
|
||||
export function formatBytes(bytes: number): string {
|
||||
if (!Number.isFinite(bytes) || bytes <= 0) return '—';
|
||||
const gb = bytes / (1024 * 1024 * 1024);
|
||||
if (gb >= 1) return `${gb.toFixed(1)} GB`;
|
||||
const mb = bytes / (1024 * 1024);
|
||||
return `${Math.round(mb)} MB`;
|
||||
}
|
||||
@@ -3,7 +3,6 @@ import { useCallback, useEffect, useState } from 'react';
|
||||
import { ConfirmationModal } from '../components/intelligence/ConfirmationModal';
|
||||
import IntelligenceCallsTab from '../components/intelligence/IntelligenceCallsTab';
|
||||
import IntelligenceDreamsTab from '../components/intelligence/IntelligenceDreamsTab';
|
||||
import IntelligenceSettingsTab from '../components/intelligence/IntelligenceSettingsTab';
|
||||
import IntelligenceSubconsciousTab from '../components/intelligence/IntelligenceSubconsciousTab';
|
||||
import { MemoryWorkspace } from '../components/intelligence/MemoryWorkspace';
|
||||
import { ToastContainer } from '../components/intelligence/Toast';
|
||||
@@ -21,7 +20,7 @@ import type {
|
||||
ToastNotification,
|
||||
} from '../types/intelligence';
|
||||
|
||||
type IntelligenceTab = 'memory' | 'subconscious' | 'calls' | 'dreams' | 'settings';
|
||||
type IntelligenceTab = 'memory' | 'subconscious' | 'calls' | 'dreams';
|
||||
|
||||
export default function Intelligence() {
|
||||
const { aiStatus } = useIntelligenceStats();
|
||||
@@ -133,7 +132,6 @@ export default function Intelligence() {
|
||||
{ id: 'subconscious', label: 'Subconscious' },
|
||||
{ id: 'calls', label: 'Calls' },
|
||||
{ id: 'dreams', label: 'Dreams', comingSoon: true },
|
||||
{ id: 'settings', label: 'Settings' },
|
||||
];
|
||||
|
||||
return (
|
||||
@@ -244,8 +242,6 @@ export default function Intelligence() {
|
||||
{activeTab === 'calls' && <IntelligenceCallsTab onToast={addToast} />}
|
||||
|
||||
{activeTab === 'dreams' && <IntelligenceDreamsTab />}
|
||||
|
||||
{activeTab === 'settings' && <IntelligenceSettingsTab />}
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
+10
-4
@@ -10,6 +10,12 @@ import { nodePolyfills } from "vite-plugin-node-polyfills";
|
||||
|
||||
const host = process.env.TAURI_DEV_HOST;
|
||||
|
||||
// Optional override so parallel `dev:app:win` runs across worktrees can
|
||||
// avoid the hardcoded 1420 collision. Default 1420 preserves prior behavior;
|
||||
// HMR companion port is dev port + 1 (used only when TAURI_DEV_HOST is set).
|
||||
const devPort = Number(process.env.OPENHUMAN_DEV_PORT) || 1420;
|
||||
const hmrPort = devPort + 1;
|
||||
|
||||
const __dirname = dirname(fileURLToPath(import.meta.url));
|
||||
const pkg = JSON.parse(
|
||||
readFileSync(resolve(__dirname, "package.json"), "utf8"),
|
||||
@@ -138,7 +144,7 @@ export default defineConfig(async () => ({
|
||||
clearScreen: false,
|
||||
// 2. tauri expects a fixed port, fail if that port is not available
|
||||
server: {
|
||||
port: 1420,
|
||||
port: devPort,
|
||||
strictPort: true,
|
||||
// `false` lets Vite pick its own loopback default; on Windows that lands
|
||||
// on `::1` only, leaving 127.0.0.1 unbound. The Tauri dev-server proxy
|
||||
@@ -161,7 +167,7 @@ export default defineConfig(async () => ({
|
||||
? {
|
||||
protocol: "ws",
|
||||
host,
|
||||
port: 1421,
|
||||
port: hmrPort,
|
||||
}
|
||||
: {
|
||||
// Tauri CEF loads the app from tauri.localhost; without this the
|
||||
@@ -169,8 +175,8 @@ export default defineConfig(async () => ({
|
||||
// Force the client to connect to the Vite dev server directly.
|
||||
protocol: "ws",
|
||||
host: "localhost",
|
||||
port: 1420,
|
||||
clientPort: 1420,
|
||||
port: devPort,
|
||||
clientPort: devPort,
|
||||
},
|
||||
watch: {
|
||||
// 3. tell Vite to ignore watching `src-tauri` directory (includes src-tauri/ai)
|
||||
|
||||
+39
-4
@@ -199,7 +199,7 @@ if [[ -z "${WindowsSdkDir:-}" || "${WindowsSDKVersion:-}" == "\\" || -z "${Windo
|
||||
if [[ -d "$sdk_root_unix/Lib" ]]; then
|
||||
sdk_version="$(ls -d "$sdk_root_unix"/Lib/*/ 2>/dev/null \
|
||||
| sort -V | tail -n1 \
|
||||
| sed 's|.*/||; s|/||g')"
|
||||
| sed 's|/$||; s|.*/||')"
|
||||
if [[ -n "$sdk_version" && -f "$sdk_root_unix/Lib/$sdk_version/um/x64/kernel32.lib" ]]; then
|
||||
sdk_root_win="$(cygpath -w "$sdk_root_unix")"
|
||||
export WindowsSdkDir="${sdk_root_win}\\"
|
||||
@@ -527,8 +527,19 @@ export PATH="$PATH_PREFIX:$PATH"
|
||||
# destination, so subsequent dev runs are essentially free.
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
if [[ -n "${CEF_RUNTIME_PATH:-}" && -f "$CEF_RUNTIME_PATH/libcef.dll" ]]; then
|
||||
CARGO_TARGET_DIR_UNIX="$(to_unix_path "${CARGO_TARGET_DIR:-$REPO_ROOT/target}" 2>/dev/null || printf '%s' "${CARGO_TARGET_DIR:-$REPO_ROOT/target}")"
|
||||
CEF_STAGE_DIR="$CARGO_TARGET_DIR_UNIX/debug"
|
||||
# The dev OpenHuman.exe is produced by the *Tauri shell* crate
|
||||
# (app/src-tauri/Cargo.toml), not the root core crate. When
|
||||
# CARGO_TARGET_DIR is set both workspaces share it; when unset, the
|
||||
# Tauri shell builds into app/src-tauri/target while the root crate
|
||||
# builds into target/. Stage CEF next to where OpenHuman.exe will
|
||||
# actually live so Windows' DLL search order finds libcef.dll
|
||||
# regardless of how the exe is launched (terminal, OAuth deep-link,
|
||||
# double-click, etc).
|
||||
if [[ -n "${CARGO_TARGET_DIR:-}" ]]; then
|
||||
CEF_STAGE_DIR="$(to_unix_path "$CARGO_TARGET_DIR" 2>/dev/null || printf '%s' "$CARGO_TARGET_DIR")/debug"
|
||||
else
|
||||
CEF_STAGE_DIR="$REPO_ROOT/app/src-tauri/target/debug"
|
||||
fi
|
||||
mkdir -p "$CEF_STAGE_DIR"
|
||||
if [[ ! -f "$CEF_STAGE_DIR/libcef.dll" \
|
||||
|| "$CEF_RUNTIME_PATH/libcef.dll" -nt "$CEF_STAGE_DIR/libcef.dll" ]]; then
|
||||
@@ -546,4 +557,28 @@ fi
|
||||
# Use the vendored tauri-cef CLI (via the pnpm tauri script) so the
|
||||
# CEF runtime is correctly bundled. APPLE_SIGNING_IDENTITY is macOS-only
|
||||
# and is intentionally omitted here.
|
||||
"$PNPM_EXE" tauri dev
|
||||
#
|
||||
# OPENHUMAN_DEV_PORT lets parallel worktree dev sessions avoid the
|
||||
# hardcoded 1420 collision. Vite reads the same env var directly; the
|
||||
# tauri-cli inline override patches tauri.conf.json's `devUrl` so the
|
||||
# shell connects to the right Vite instance.
|
||||
# Validate OPENHUMAN_DEV_PORT before interpolating into JSON — a stray
|
||||
# space, alphabetic char, or out-of-range value would produce an invalid
|
||||
# devUrl and tauri would refuse to start (or worse, drift from Vite's
|
||||
# own numeric fallback). Trim whitespace, require pure digits in
|
||||
# [1, 65535], fall back to 1420 with a warning otherwise.
|
||||
raw_dev_port="${OPENHUMAN_DEV_PORT:-1420}"
|
||||
raw_dev_port="${raw_dev_port//[[:space:]]/}"
|
||||
if [[ "$raw_dev_port" =~ ^[0-9]+$ ]] && (( raw_dev_port >= 1 && raw_dev_port <= 65535 )); then
|
||||
DEV_PORT="$raw_dev_port"
|
||||
else
|
||||
echo "[run-dev-win] WARNING: invalid OPENHUMAN_DEV_PORT='$raw_dev_port'; falling back to 1420" >&2
|
||||
DEV_PORT=1420
|
||||
fi
|
||||
|
||||
if (( DEV_PORT != 1420 )); then
|
||||
echo "[run-dev-win] OPENHUMAN_DEV_PORT=$DEV_PORT — overriding tauri devUrl"
|
||||
"$PNPM_EXE" tauri dev -c "{\"build\":{\"devUrl\":\"http://localhost:$DEV_PORT\"}}"
|
||||
else
|
||||
"$PNPM_EXE" tauri dev
|
||||
fi
|
||||
|
||||
@@ -131,7 +131,11 @@ fn default_vision_model_id() -> String {
|
||||
}
|
||||
|
||||
fn default_embedding_model_id() -> String {
|
||||
"all-minilm:latest".to_string()
|
||||
// bge-m3 (1024 dims, 8192-token context). Required by the memory tree's
|
||||
// fixed on-disk embedding format (EMBEDDING_DIM=1024) — `all-minilm`
|
||||
// (384 dims) and `nomic-embed-text` (768 dims) would fail the
|
||||
// post-call dim validator at `memory::tree::score::embed::mod::embed`.
|
||||
"bge-m3".to_string()
|
||||
}
|
||||
|
||||
fn default_stt_model_id() -> String {
|
||||
|
||||
@@ -4,7 +4,10 @@
|
||||
//! This is the preferred local provider: Ollama handles model management,
|
||||
//! quantization, and GPU acceleration (Metal on macOS, CUDA on Linux/Windows).
|
||||
//!
|
||||
//! Default model: `nomic-embed-text:latest` (768 dimensions).
|
||||
//! Default model: `bge-m3` (1024 dimensions). Aligned with the memory
|
||||
//! tree's fixed on-disk format (`EMBEDDING_DIM=1024`) and the cloud
|
||||
//! Voyage default (`embedding-v1`, 1024 dims) so embeddings produced by
|
||||
//! either path are interchangeable.
|
||||
|
||||
use async_trait::async_trait;
|
||||
|
||||
@@ -13,11 +16,12 @@ use super::EmbeddingProvider;
|
||||
/// Default Ollama base URL.
|
||||
pub const DEFAULT_OLLAMA_URL: &str = "http://localhost:11434";
|
||||
|
||||
/// Default embedding model for Ollama.
|
||||
pub const DEFAULT_OLLAMA_MODEL: &str = "nomic-embed-text:latest";
|
||||
/// Default embedding model for Ollama. 1024-dim to match the memory
|
||||
/// tree's fixed on-disk format and the cloud Voyage default.
|
||||
pub const DEFAULT_OLLAMA_MODEL: &str = "bge-m3";
|
||||
|
||||
/// Default dimensions for nomic-embed-text.
|
||||
pub const DEFAULT_OLLAMA_DIMENSIONS: usize = 768;
|
||||
/// Default dimensions for `bge-m3`.
|
||||
pub const DEFAULT_OLLAMA_DIMENSIONS: usize = 1024;
|
||||
|
||||
/// Embedding provider backed by a local Ollama instance.
|
||||
///
|
||||
@@ -35,8 +39,8 @@ impl OllamaEmbedding {
|
||||
/// Creates a new Ollama embedding provider.
|
||||
///
|
||||
/// - `base_url`: Ollama server URL (default: `http://localhost:11434`)
|
||||
/// - `model`: Model name (default: `nomic-embed-text:latest`)
|
||||
/// - `dims`: Expected embedding dimensions (default: 768)
|
||||
/// - `model`: Model name (default: `bge-m3`)
|
||||
/// - `dims`: Expected embedding dimensions (default: 1024)
|
||||
pub fn try_new(base_url: &str, model: &str, dims: usize) -> anyhow::Result<Self> {
|
||||
let base_url = Self::normalize_base_url(base_url)?;
|
||||
let model = Self::normalize_model(model)?;
|
||||
|
||||
@@ -10,7 +10,7 @@ use crate::openhuman::config::Config;
|
||||
pub(crate) const DEFAULT_OLLAMA_MODEL: &str = "gemma3:1b-it-qat";
|
||||
pub(crate) const DEFAULT_OLLAMA_VISION_MODEL: &str = "";
|
||||
pub(crate) const DEFAULT_LOW_VISION_MODEL: &str = "moondream:1.8b-v2-q4_K_S";
|
||||
pub(crate) const DEFAULT_OLLAMA_EMBED_MODEL: &str = "all-minilm:latest";
|
||||
pub(crate) const DEFAULT_OLLAMA_EMBED_MODEL: &str = "bge-m3";
|
||||
|
||||
/// Chat models allowed in the current MVP build (2–4 GB tier only).
|
||||
/// Any resolved chat model ID not listed here is redirected to `MVP_DEFAULT_CHAT_MODEL`.
|
||||
@@ -22,7 +22,11 @@ const MVP_DEFAULT_CHAT_MODEL: &str = "gemma3:1b-it-qat";
|
||||
const MVP_ALLOWED_VISION_MODELS: &[&str] = &[""];
|
||||
|
||||
/// Embedding models allowed in MVP (2–4 GB tier uses all-minilm).
|
||||
const MVP_ALLOWED_EMBEDDING_MODELS: &[&str] = &["all-minilm:latest"];
|
||||
// bge-m3 (1024-dim, 8192-token context) is the canonical local embedder
|
||||
// for memory tree's fixed on-disk format. all-minilm (384-dim) is kept
|
||||
// for back-compat with users who pulled it under an older default, but
|
||||
// new selections should default to bge-m3.
|
||||
const MVP_ALLOWED_EMBEDDING_MODELS: &[&str] = &["bge-m3", "all-minilm:latest"];
|
||||
|
||||
fn enforce_mvp_chat_allowlist(resolved: &str) -> String {
|
||||
let lower = resolved.to_ascii_lowercase();
|
||||
@@ -195,6 +199,47 @@ mod tests {
|
||||
assert_eq!(effective_vision_model_id(&config), "");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedding_model_empty_falls_back_to_bge_m3() {
|
||||
// After the cloud-embeddings unification PR, the default embedder
|
||||
// for the local Ollama path is bge-m3 (1024 dim) to match memory
|
||||
// tree's fixed on-disk format. Empty / whitespace input must
|
||||
// resolve to that default, not the prior all-minilm:latest.
|
||||
let mut config = test_config();
|
||||
config.local_ai.embedding_model_id = String::new();
|
||||
assert_eq!(effective_embedding_model_id(&config), "bge-m3");
|
||||
|
||||
config.local_ai.embedding_model_id = " ".to_string();
|
||||
assert_eq!(effective_embedding_model_id(&config), "bge-m3");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedding_model_passes_through_allowlisted_legacy() {
|
||||
// all-minilm:latest is kept in MVP_ALLOWED_EMBEDDING_MODELS for
|
||||
// back-compat with users who already pulled it under the prior
|
||||
// default. It is NOT 1024-dim — memory tree's post-call validator
|
||||
// will surface that mismatch at embed time — but the allowlist
|
||||
// enforcer itself must let the value pass through unchanged.
|
||||
let mut config = test_config();
|
||||
config.local_ai.embedding_model_id = "all-minilm:latest".to_string();
|
||||
assert_eq!(effective_embedding_model_id(&config), "all-minilm:latest");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn embedding_model_rejects_non_allowlisted_and_redirects_to_default() {
|
||||
// Any non-allowlisted value (including legacy nomic-embed-text:latest
|
||||
// and arbitrary user input) is silently redirected to the canonical
|
||||
// default. This is the path that fired the "embedding model not in
|
||||
// MVP allowlist, redirecting to default" warning on every embed
|
||||
// resolution before bge-m3 was added to the allowlist.
|
||||
let mut config = test_config();
|
||||
config.local_ai.embedding_model_id = "nomic-embed-text:latest".to_string();
|
||||
assert_eq!(effective_embedding_model_id(&config), "bge-m3");
|
||||
|
||||
config.local_ai.embedding_model_id = "totally-made-up-model:v0".to_string();
|
||||
assert_eq!(effective_embedding_model_id(&config), "bge-m3");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stt_tts_and_quantization_defaults_are_applied() {
|
||||
let mut config = test_config();
|
||||
|
||||
@@ -128,13 +128,15 @@ pub fn all_presets() -> Vec<ModelPreset> {
|
||||
"Battery-friendly local summarization preset. Uses the 1B Gemma model with vision disabled.",
|
||||
chat_model_id: "gemma3:1b-it-qat",
|
||||
vision_model_id: "",
|
||||
embedding_model_id: "all-minilm:latest",
|
||||
// bge-m3 — 1024 dims required by memory tree's on-disk format
|
||||
// and 8192-token context for long-chunk embeds.
|
||||
embedding_model_id: "bge-m3",
|
||||
quantization: "qat",
|
||||
vision_mode: VisionMode::Disabled,
|
||||
supports_screen_summary: false,
|
||||
target_ram_gb: 2,
|
||||
min_ram_gb: 2,
|
||||
approx_download_gb: 1.1,
|
||||
approx_download_gb: 2.3,
|
||||
},
|
||||
ModelPreset {
|
||||
tier: ModelTier::Ram4To8Gb,
|
||||
|
||||
@@ -0,0 +1,101 @@
|
||||
//! Cloud (Voyage-backed) embedder for the memory tree.
|
||||
//!
|
||||
//! Adapts the OpenHuman backend's `POST /openai/v1/embeddings` surface
|
||||
//! (Voyage `voyage-3.5`, 1024 dims) to the memory_tree [`Embedder`] trait
|
||||
//! so Phase 4 ingest / bucket-seal can vectorize chunks without a local
|
||||
//! Ollama install.
|
||||
//!
|
||||
//! The 1024-dim output matches existing on-disk blobs (which were
|
||||
//! produced by `bge-m3`, also 1024-dim), so this is a drop-in replacement
|
||||
//! for the Ollama path — no migration of `mem_tree_chunks.embedding`
|
||||
//! required.
|
||||
//!
|
||||
//! Auth: the cloud embedder resolves the session JWT per call via
|
||||
//! [`OpenHumanCloudEmbedding`], so a session refresh between batches is
|
||||
//! picked up transparently. When the user is unauthenticated the first
|
||||
//! `embed()` returns an error; ingest treats that the same as any other
|
||||
//! embedder failure (don't persist the row, let job retry).
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use async_trait::async_trait;
|
||||
|
||||
use super::{Embedder, EMBEDDING_DIM};
|
||||
use crate::openhuman::config::Config;
|
||||
use crate::openhuman::embeddings::cloud::{
|
||||
OpenHumanCloudEmbedding, DEFAULT_CLOUD_EMBEDDING_DIMENSIONS, DEFAULT_CLOUD_EMBEDDING_MODEL,
|
||||
};
|
||||
use crate::openhuman::embeddings::EmbeddingProvider;
|
||||
|
||||
/// Cloud-backed memory_tree embedder.
|
||||
///
|
||||
/// Wraps [`OpenHumanCloudEmbedding`] (which speaks the OpenAI-compatible
|
||||
/// `/openai/v1/embeddings` shape backed by Voyage on the OpenHuman
|
||||
/// backend) and adapts it to the memory_tree [`Embedder`] trait.
|
||||
pub struct CloudEmbedder {
|
||||
inner: OpenHumanCloudEmbedding,
|
||||
}
|
||||
|
||||
impl CloudEmbedder {
|
||||
/// Build a cloud embedder using the same backend resolution as the
|
||||
/// main embeddings path: `api_url` falls back to
|
||||
/// [`effective_api_url`](crate::api::config::effective_api_url) and
|
||||
/// the workspace dir comes from `config.workspace_dir` so the auth
|
||||
/// service finds the user's session JWT.
|
||||
pub fn new(config: &Config) -> Self {
|
||||
let openhuman_dir = config.config_path.parent().map(std::path::PathBuf::from);
|
||||
Self {
|
||||
inner: OpenHumanCloudEmbedding::new(
|
||||
None,
|
||||
openhuman_dir,
|
||||
config.secrets.encrypt,
|
||||
DEFAULT_CLOUD_EMBEDDING_MODEL,
|
||||
DEFAULT_CLOUD_EMBEDDING_DIMENSIONS,
|
||||
),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
impl Embedder for CloudEmbedder {
|
||||
fn name(&self) -> &'static str {
|
||||
"cloud"
|
||||
}
|
||||
|
||||
async fn embed(&self, text: &str) -> Result<Vec<f32>> {
|
||||
let v = self
|
||||
.inner
|
||||
.embed_one(text)
|
||||
.await
|
||||
.context("cloud embeddings failed")?;
|
||||
if v.len() != EMBEDDING_DIM {
|
||||
anyhow::bail!(
|
||||
"cloud embedder returned {} dims, expected {}",
|
||||
v.len(),
|
||||
EMBEDDING_DIM
|
||||
);
|
||||
}
|
||||
Ok(v)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::openhuman::config::Config;
|
||||
use tempfile::TempDir;
|
||||
|
||||
fn test_config() -> (TempDir, Config) {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let mut cfg = Config::default();
|
||||
cfg.workspace_dir = tmp.path().to_path_buf();
|
||||
cfg.config_path = tmp.path().join("config.toml");
|
||||
(tmp, cfg)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn name_is_cloud() {
|
||||
let (_tmp, cfg) = test_config();
|
||||
let e = CloudEmbedder::new(&cfg);
|
||||
assert_eq!(e.name(), "cloud");
|
||||
}
|
||||
}
|
||||
@@ -1,12 +1,28 @@
|
||||
//! Build an [`Embedder`] from [`Config::memory_tree`] settings.
|
||||
//! Build an [`Embedder`] from [`Config`] settings.
|
||||
//!
|
||||
//! Resolution order:
|
||||
//! 1. `memory_tree.embedding_endpoint` + `memory_tree.embedding_model`
|
||||
//! both Some → [`OllamaEmbedder`]
|
||||
//! 2. Otherwise → depends on `memory_tree.embedding_strict`:
|
||||
//! - `true` → bail with a clear "configure Ollama for Phase 4" error
|
||||
//! - `false` → fall back to [`InertEmbedder`] (zero vectors) with a
|
||||
//! warn log so the operator notices embeddings are disabled
|
||||
//! 1. **Explicit override** — `memory_tree.embedding_endpoint` +
|
||||
//! `memory_tree.embedding_model` both Some → [`OllamaEmbedder`] with
|
||||
//! those exact values. For power users / E2E test rigs that want to
|
||||
//! point at a non-default Ollama endpoint.
|
||||
//! 2. **Local-AI usage flag** — `config.local_ai.use_local_for_embeddings()`
|
||||
//! (i.e. `runtime_enabled && usage.embeddings`) → [`OllamaEmbedder`]
|
||||
//! against [`ollama_base_url`] with the user's chosen
|
||||
//! `config.local_ai.embedding_model_id`. This is the path driven by
|
||||
//! the "Memory embeddings" checkbox in Local AI Settings.
|
||||
//! 3. **Default** — [`CloudEmbedder`] (OpenHuman backend / Voyage,
|
||||
//! 1024 dims). Auth failures surface at the first `embed()` call so
|
||||
//! ingest's existing retry-with-backoff logic handles them.
|
||||
//!
|
||||
//! NOTE on dimensions: the memory tree on-disk format is hard-coded at
|
||||
//! [`EMBEDDING_DIM`](super::EMBEDDING_DIM) (1024). If the user picks a
|
||||
//! local embedding model whose output is a different dimensionality,
|
||||
//! the trait's post-call validator rejects each embed with a clear
|
||||
//! `expected N dims, got M` error. Switching the local model picker in
|
||||
//! Local AI Settings is the fix.
|
||||
//!
|
||||
//! The historical `InertEmbedder` (zero vectors) path is retained for
|
||||
//! tests only — it is no longer the production lax-mode fallback.
|
||||
//!
|
||||
//! Env var overrides applied in [`crate::openhuman::config::load`]:
|
||||
//! - `OPENHUMAN_MEMORY_EMBED_ENDPOINT`
|
||||
@@ -15,8 +31,24 @@
|
||||
|
||||
use anyhow::Result;
|
||||
|
||||
use super::{Embedder, InertEmbedder, OllamaEmbedder};
|
||||
use super::{CloudEmbedder, Embedder, InertEmbedder, OllamaEmbedder};
|
||||
use crate::openhuman::config::Config;
|
||||
use crate::openhuman::local_ai::ollama_base_url;
|
||||
|
||||
/// Cheap heuristic for "is a backend session reachable?" — the cloud
|
||||
/// embedder needs one and bails on first embed call without it. We use
|
||||
/// the *presence* of `auth-profiles.json` next to the config file as a
|
||||
/// proxy: production after login has it, test harnesses and fresh
|
||||
/// pre-login installs don't. The CloudEmbedder still re-validates the
|
||||
/// JWT at every embed call, so a stale file just surfaces at embed
|
||||
/// time (not factory build), preserving the prior failure behavior.
|
||||
fn cloud_session_available(config: &Config) -> bool {
|
||||
config
|
||||
.config_path
|
||||
.parent()
|
||||
.map(|dir| dir.join("auth-profiles.json").exists())
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
/// Construct the active embedder for this process, honouring
|
||||
/// `config.memory_tree.*` and `embedding_strict`.
|
||||
@@ -48,23 +80,45 @@ pub fn build_embedder_from_config(config: &Config) -> Result<Box<dyn Embedder>>
|
||||
)))
|
||||
}
|
||||
_ => {
|
||||
if tree_cfg.embedding_strict {
|
||||
anyhow::bail!(
|
||||
"memory_tree embedding is required (embedding_strict=true) but \
|
||||
embedding_endpoint/embedding_model are unset. Set \
|
||||
`memory_tree.embedding_endpoint` + `.embedding_model` in \
|
||||
config.toml or export OPENHUMAN_MEMORY_EMBED_ENDPOINT / \
|
||||
OPENHUMAN_MEMORY_EMBED_MODEL — or set \
|
||||
`memory_tree.embedding_strict = false` to fall back to zero \
|
||||
vectors (embeddings will not contribute to retrieval rerank)."
|
||||
// Honor the Local AI Settings "Memory embeddings" checkbox.
|
||||
// `use_local_for_embeddings()` is `runtime_enabled && usage.embeddings`
|
||||
// so we never route to a disabled local runtime.
|
||||
if config.local_ai.use_local_for_embeddings() {
|
||||
let model = config.local_ai.embedding_model_id.clone();
|
||||
let endpoint = ollama_base_url();
|
||||
let timeout_ms = tree_cfg.embedding_timeout_ms.unwrap_or(0);
|
||||
log::debug!(
|
||||
"[memory_tree::embed::factory] usage.embeddings=true — using local Ollama endpoint={} model={} timeout_ms={}",
|
||||
endpoint, model, timeout_ms
|
||||
);
|
||||
Ok(Box::new(OllamaEmbedder::new(endpoint, model, timeout_ms)))
|
||||
} else if cloud_session_available(config) {
|
||||
// Default for logged-in users: cloud (OpenHuman backend /
|
||||
// Voyage `voyage-3.5`, 1024 dims). Matches the main
|
||||
// embeddings path so a fresh install needs zero local
|
||||
// Ollama setup. JWT failures (expired, invalid, etc.)
|
||||
// surface as embed-call errors so ingest's existing
|
||||
// retry-with-backoff logic handles them.
|
||||
log::debug!(
|
||||
"[memory_tree::embed::factory] using cloud (Voyage) — \
|
||||
flip 'Memory embeddings' in Local AI Settings to switch to local"
|
||||
);
|
||||
Ok(Box::new(CloudEmbedder::new(config)))
|
||||
} else {
|
||||
// Pre-login, test harness, or unauthenticated runtime
|
||||
// path — no auth-profiles.json on disk means the cloud
|
||||
// path has no chance of resolving a bearer. Drop to
|
||||
// InertEmbedder (zero vectors) so ingest/seal/retrieval
|
||||
// can run without panic; semantic rerank degrades to
|
||||
// recency only until the user logs in (or until they
|
||||
// flip "Memory embeddings" to local with Ollama running).
|
||||
log::warn!(
|
||||
"[memory_tree::embed::factory] no backend session found — \
|
||||
using InertEmbedder (zero vectors). Log in to OpenHuman, or \
|
||||
enable 'Memory embeddings' in Local AI Settings, to fix."
|
||||
);
|
||||
Ok(Box::new(InertEmbedder::new()))
|
||||
}
|
||||
log::warn!(
|
||||
"[memory_tree::embed::factory] no embedding endpoint/model — \
|
||||
falling back to InertEmbedder (zero vectors). Set \
|
||||
memory_tree.embedding_endpoint to enable semantic retrieval."
|
||||
);
|
||||
Ok(Box::new(InertEmbedder::new()))
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -78,9 +132,25 @@ mod tests {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let mut cfg = Config::default();
|
||||
cfg.workspace_dir = tmp.path().to_path_buf();
|
||||
// Plant config_path in the tempdir so cloud_session_available()
|
||||
// checks a writable directory; tests that need to simulate a
|
||||
// logged-in user just `touch` auth-profiles.json next to it.
|
||||
cfg.config_path = tmp.path().join("config.toml");
|
||||
(tmp, cfg)
|
||||
}
|
||||
|
||||
/// Drop a stub `auth-profiles.json` next to the test config so
|
||||
/// `cloud_session_available()` returns true. Contents don't matter
|
||||
/// — the factory only checks presence.
|
||||
fn touch_auth_profile(cfg: &Config) {
|
||||
let path = cfg
|
||||
.config_path
|
||||
.parent()
|
||||
.map(|p| p.join("auth-profiles.json"))
|
||||
.expect("config_path has a parent");
|
||||
std::fs::write(&path, "{}").expect("write stub auth-profiles.json");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ollama_chosen_when_endpoint_and_model_set() {
|
||||
let (_tmp, mut cfg) = test_config();
|
||||
@@ -92,36 +162,94 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strict_mode_bails_on_missing_endpoint() {
|
||||
let (_tmp, mut cfg) = test_config();
|
||||
cfg.memory_tree.embedding_endpoint = None;
|
||||
cfg.memory_tree.embedding_model = None;
|
||||
cfg.memory_tree.embedding_strict = true;
|
||||
// `Box<dyn Embedder>` isn't `Debug`, so go through `match` rather
|
||||
// than `unwrap_err` (which needs Debug on the Ok variant).
|
||||
match build_embedder_from_config(&cfg) {
|
||||
Ok(_) => panic!("expected strict-mode bail"),
|
||||
Err(e) => assert!(e.to_string().contains("embedding_strict"), "{e}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn lax_mode_falls_back_to_inert() {
|
||||
fn unset_endpoint_with_session_routes_to_cloud() {
|
||||
let (_tmp, mut cfg) = test_config();
|
||||
cfg.memory_tree.embedding_endpoint = None;
|
||||
cfg.memory_tree.embedding_model = None;
|
||||
cfg.memory_tree.embedding_strict = false;
|
||||
let e = build_embedder_from_config(&cfg).expect("lax path should build");
|
||||
touch_auth_profile(&cfg);
|
||||
let e = build_embedder_from_config(&cfg).expect("cloud default should build");
|
||||
assert_eq!(e.name(), "cloud");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unset_endpoint_without_session_falls_back_to_inert() {
|
||||
// Test harness / pre-login: no auth-profiles.json on disk,
|
||||
// factory degrades to InertEmbedder so callers don't crash on
|
||||
// first embed call.
|
||||
let (_tmp, mut cfg) = test_config();
|
||||
cfg.memory_tree.embedding_endpoint = None;
|
||||
cfg.memory_tree.embedding_model = None;
|
||||
cfg.memory_tree.embedding_strict = false;
|
||||
let e = build_embedder_from_config(&cfg).expect("inert fallback should build");
|
||||
assert_eq!(e.name(), "inert");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_strings_count_as_unset() {
|
||||
fn empty_strings_count_as_unset_with_session() {
|
||||
let (_tmp, mut cfg) = test_config();
|
||||
cfg.memory_tree.embedding_endpoint = Some("".into());
|
||||
cfg.memory_tree.embedding_model = Some("".into());
|
||||
cfg.memory_tree.embedding_strict = false;
|
||||
let e = build_embedder_from_config(&cfg).expect("lax path should build");
|
||||
assert_eq!(e.name(), "inert");
|
||||
touch_auth_profile(&cfg);
|
||||
let e = build_embedder_from_config(&cfg).expect("cloud default should build");
|
||||
assert_eq!(e.name(), "cloud");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strict_mode_no_longer_bails_with_cloud_default() {
|
||||
// Strict mode used to bail when endpoint/model were unset because
|
||||
// the only fallback was InertEmbedder. Now the lax-and-strict
|
||||
// paths share the cloud fallback; strict bail is a no-op here
|
||||
// and auth failures surface at first embed() call instead.
|
||||
let (_tmp, mut cfg) = test_config();
|
||||
cfg.memory_tree.embedding_endpoint = None;
|
||||
cfg.memory_tree.embedding_model = None;
|
||||
cfg.memory_tree.embedding_strict = true;
|
||||
touch_auth_profile(&cfg);
|
||||
let e = build_embedder_from_config(&cfg).expect("cloud default should build");
|
||||
assert_eq!(e.name(), "cloud");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn local_ai_usage_embeddings_routes_to_ollama() {
|
||||
// When the Local AI Settings "Memory embeddings" checkbox is on
|
||||
// (runtime_enabled && usage.embeddings), memory tree routes to
|
||||
// Ollama using the user's chosen `embedding_model_id`. The
|
||||
// explicit endpoint/model override is left unset so we exercise
|
||||
// the use_local_for_embeddings() branch.
|
||||
let (_tmp, mut cfg) = test_config();
|
||||
cfg.memory_tree.embedding_endpoint = None;
|
||||
cfg.memory_tree.embedding_model = None;
|
||||
cfg.local_ai.runtime_enabled = true;
|
||||
cfg.local_ai.usage.embeddings = true;
|
||||
cfg.local_ai.embedding_model_id = "all-minilm:latest".to_string();
|
||||
let e = build_embedder_from_config(&cfg).expect("ollama path should build");
|
||||
assert_eq!(e.name(), "ollama");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn local_ai_usage_off_with_session_falls_back_to_cloud() {
|
||||
// runtime_enabled=true but usage.embeddings=false → cloud (with session).
|
||||
let (_tmp, mut cfg) = test_config();
|
||||
cfg.memory_tree.embedding_endpoint = None;
|
||||
cfg.memory_tree.embedding_model = None;
|
||||
cfg.local_ai.runtime_enabled = true;
|
||||
cfg.local_ai.usage.embeddings = false;
|
||||
touch_auth_profile(&cfg);
|
||||
let e = build_embedder_from_config(&cfg).expect("cloud default should build");
|
||||
assert_eq!(e.name(), "cloud");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn explicit_endpoint_override_wins_over_local_ai_flag() {
|
||||
// Power-user override beats the checkbox.
|
||||
let (_tmp, mut cfg) = test_config();
|
||||
cfg.memory_tree.embedding_endpoint = Some("http://staging-embed:11434".into());
|
||||
cfg.memory_tree.embedding_model = Some("bge-m3".into());
|
||||
cfg.local_ai.runtime_enabled = true;
|
||||
cfg.local_ai.usage.embeddings = true;
|
||||
let e = build_embedder_from_config(&cfg).expect("override path should build");
|
||||
assert_eq!(e.name(), "ollama");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,10 +29,12 @@
|
||||
use anyhow::{Context, Result};
|
||||
use async_trait::async_trait;
|
||||
|
||||
pub mod cloud;
|
||||
pub mod factory;
|
||||
pub mod inert;
|
||||
pub mod ollama;
|
||||
|
||||
pub use cloud::CloudEmbedder;
|
||||
pub use factory::build_embedder_from_config;
|
||||
pub use inert::InertEmbedder;
|
||||
pub use ollama::OllamaEmbedder;
|
||||
|
||||
Reference in New Issue
Block a user