diff --git a/README.md b/README.md index 1f87d5883..2cd26578d 100644 --- a/README.md +++ b/README.md @@ -71,8 +71,8 @@ GBrain is designed to be installed and operated by an AI agent. The fastest path If you don't already have an AI agent platform running, start with one of these. Both are designed to read GBrain's install protocol and execute it: -- **[OpenClaw](https://github.com/openclawagents/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM) -- **[Hermes](https://github.com/openclawagents/hermes)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click) +- **[OpenClaw](https://github.com/openclaw/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM) +- **[Hermes](https://github.com/NousResearch/hermes-agent)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click) Then paste this into your agent: diff --git a/docs/guides/skillpacks-as-scaffolding.md b/docs/guides/skillpacks-as-scaffolding.md index b58d8e511..db33a8a1d 100644 --- a/docs/guides/skillpacks-as-scaffolding.md +++ b/docs/guides/skillpacks-as-scaffolding.md @@ -131,7 +131,9 @@ into gbrain so other clients can scaffold it. Default behavior: `~/.gbrain/harvest-private-patterns.txt` plus built-in defaults (canonical private fork name, common email regex, Slack channel pattern). Any match → rollback (delete the harvested files) and exit non-zero. -- `openclaw.plugin.json` updated with the new slug, sorted. +- `openclaw.plugin.json` updated with the new slug, sorted. Harvest must preserve + the top-level OpenClaw-native plugin fields (`id`, `configSchema`, `contracts`) + because OpenClaw validates those before it can install the package. - `--no-lint` bypasses the linter (after a manual editorial scrub). Use the `skillpack-harvest` skill (its companion editorial workflow) diff --git a/docs/tutorials/improving-skills-with-skillopt.md b/docs/tutorials/improving-skills-with-skillopt.md index 6c2a7fd53..009cac18d 100644 --- a/docs/tutorials/improving-skills-with-skillopt.md +++ b/docs/tutorials/improving-skills-with-skillopt.md @@ -233,13 +233,14 @@ keep it or `git checkout` to throw it away. Nothing is committed for you. **For a skill that ships with gbrain** (anything under the gbrain repo's own `skills/`): SkillOpt refuses to overwrite it by default and writes the winner to -`skills//skillopt/best.md` instead, so an optimization pass can never -silently mutate a skill other people depend on. Two ways to handle that: +`skills//skillopt/proposed.md` instead (while keeping `best.md` as the +optimizer's current-best pointer), so an optimization pass can never silently +mutate a skill other people depend on. Two ways to handle that: ```bash # See the proposed improvement without touching SKILL.md (works for ANY skill): gbrain skillopt meeting-prep --split 1:1:1 --no-mutate -# → writes skills/meeting-prep/skillopt/best.md (the proposed rewrite), prints its path. Copy what you want. +# → writes skills/meeting-prep/skillopt/proposed.md, updates best.md, and prints the proposal path. # Actually rewrite a bundled skill (explicit opt-in + an independent held-out set): gbrain skillopt brain-ops --split 1:1:1 --allow-mutate-bundled \ diff --git a/llms-full.txt b/llms-full.txt index 4de2b11e4..0ef8ca577 100644 --- a/llms-full.txt +++ b/llms-full.txt @@ -1565,8 +1565,8 @@ GBrain is designed to be installed and operated by an AI agent. The fastest path If you don't already have an AI agent platform running, start with one of these. Both are designed to read GBrain's install protocol and execute it: -- **[OpenClaw](https://github.com/openclawagents/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM) -- **[Hermes](https://github.com/openclawagents/hermes)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click) +- **[OpenClaw](https://github.com/openclaw/openclaw)** — deploy [AlphaClaw on Render](https://render.com/deploy?repo=https://github.com/chrysb/alphaclaw) (one click, 8GB+ RAM) +- **[Hermes](https://github.com/NousResearch/hermes-agent)** — deploy on [Railway](https://github.com/praveen-ks-2001/hermes-agent-template) (one click) Then paste this into your agent: diff --git a/openclaw.plugin.json b/openclaw.plugin.json index e6c1b0533..f12bb6445 100644 --- a/openclaw.plugin.json +++ b/openclaw.plugin.json @@ -1,4 +1,5 @@ { + "id": "gbrain-context-engine", "name": "gbrain", "version": "0.32.3.0", "description": "Personal knowledge brain with Postgres + pgvector hybrid search", diff --git a/skills/skillpack-harvest/SKILL.md b/skills/skillpack-harvest/SKILL.md index a19e217b5..bd6f2dfbb 100644 --- a/skills/skillpack-harvest/SKILL.md +++ b/skills/skillpack-harvest/SKILL.md @@ -266,4 +266,5 @@ editorial pass. (e.g. `src/commands/.ts` if the host SKILL.md declares it in frontmatter) - gbrain's `openclaw.plugin.json` — adds the slug to `skills:` - array, sorted alphabetically + array, sorted alphabetically, without removing OpenClaw-native plugin fields + like `id`, `configSchema`, or `contracts` diff --git a/skills/testing/SKILL.md b/skills/testing/SKILL.md index 790820c56..9d73437d1 100644 --- a/skills/testing/SKILL.md +++ b/skills/testing/SKILL.md @@ -57,6 +57,8 @@ This mode guarantees: - `skills/manifest.json` lists every skill directory - `skills/RESOLVER.md` references every skill in the manifest - `openclaw.plugin.json` `skills[]` round-trips with both +- `openclaw.plugin.json` keeps OpenClaw install-required native plugin fields + (`id`, object `configSchema`, and `contracts.contextEngines` when applicable) - No MECE violations (duplicate triggers across skills) ### Phases @@ -72,7 +74,7 @@ This mode guarantees: ### Automation ```bash -bun test test/skills-conformance.test.ts test/resolver.test.ts +bun test test/skills-conformance.test.ts test/resolver.test.ts test/openclaw-plugin-manifest.test.ts ``` The CI-gated check is the package.json `test` script. diff --git a/src/cli.ts b/src/cli.ts index 804d7dbf3..3ca8a66fd 100755 --- a/src/cli.ts +++ b/src/cli.ts @@ -54,7 +54,7 @@ export function bigintToStringReplacer(_key: string, value: unknown): unknown { } // CLI-only commands that bypass the operation layer -export const CLI_ONLY = new Set(['init', 'reinit-pglite', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'sync', 'extract', 'extract-conversation-facts', 'enrich', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'sources', 'mounts', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test', 'providers', 'storage', 'repos', 'code-def', 'code-refs', 'reindex', 'reindex-code', 'reindex-frontmatter', 'code-callers', 'code-callees', 'reconcile-links', 'frontmatter', 'auth', 'friction', 'claw-test', 'book-mirror', 'takes', 'think', 'salience', 'anomalies', 'calibration', 'transcripts', 'models', 'remote', 'recall', 'forget', 'edges-backfill', 'cache', 'ze-switch', 'founder', 'brainstorm', 'lsd', 'schema', 'capture', 'onboard', 'conversation-parser', 'status', 'connect', 'skillopt', 'quarantine', 'self-upgrade', 'advisor', 'watch', 'reindex-search-vector']); +export const CLI_ONLY = new Set(['init', 'reinit-pglite', 'upgrade', 'post-upgrade', 'check-update', 'integrations', 'publish', 'check-backlinks', 'lint', 'report', 'import', 'export', 'files', 'embed', 'serve', 'call', 'config', 'doctor', 'migrate', 'eval', 'sync', 'extract', 'extract-conversation-facts', 'enrich', 'features', 'autopilot', 'graph-query', 'jobs', 'agent', 'apply-migrations', 'skillpack-check', 'skillpack', 'resolvers', 'integrity', 'repair-jsonb', 'orphans', 'maintain', 'sources', 'mounts', 'dream', 'check-resolvable', 'routing-eval', 'skillify', 'smoke-test', 'providers', 'storage', 'repos', 'code-def', 'code-refs', 'reindex', 'reindex-code', 'reindex-frontmatter', 'code-callers', 'code-callees', 'reconcile-links', 'frontmatter', 'auth', 'friction', 'claw-test', 'book-mirror', 'takes', 'think', 'salience', 'anomalies', 'calibration', 'transcripts', 'models', 'remote', 'recall', 'forget', 'edges-backfill', 'cache', 'ze-switch', 'founder', 'brainstorm', 'lsd', 'schema', 'capture', 'onboard', 'conversation-parser', 'status', 'connect', 'skillopt', 'quarantine', 'self-upgrade', 'advisor', 'watch', 'reindex-search-vector']); // CLI-only commands whose handlers print their own --help text. These are // excluded from the generic short-circuit so detailed per-command and // per-subcommand usage stays reachable. @@ -78,6 +78,8 @@ const CLI_ONLY_SELF_HELP = new Set([ 'capture', // v0.42 self-upgrade ships its own usage (flags + the agent-skill story). 'self-upgrade', + // maintain (#3015) prints its own usage block (modes + not-auto-applied list). + 'maintain', // v0.43 (#2095): watch ships WATCH_HELP (flags + the stdin-turn protocol). 'watch', // v0.37 fix wave (Lane D.4 + CDX2-12): sync's --no-embed flag was @@ -1757,6 +1759,11 @@ async function handleCliOnly(command: string, args: string[]) { await runOrphans(engine, args); break; } + case 'maintain': { + const { runMaintain } = await import('./commands/maintain.ts'); + await runMaintain(engine, args); + break; + } // v0.32.7 CJK wave — post-upgrade markdown re-chunk sweep. // v0.36 Phase 3 wave — `gbrain reindex --multimodal` re-embeds content_chunks // into the unified Voyage multimodal-3 column. diff --git a/src/commands/embed.ts b/src/commands/embed.ts index 6d8816a6e..348d76812 100644 --- a/src/commands/embed.ts +++ b/src/commands/embed.ts @@ -581,7 +581,7 @@ async function embedPage( for (let j = 0; j < toEmbed.length; j++) { embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]); } - const updated: ChunkInput[] = chunks.map(c => ({ + const updated: ChunkInput[] = chunks.map(c => preserveCodeMetadata(c, { chunk_index: c.chunk_index, chunk_text: c.chunk_text, chunk_source: c.chunk_source, @@ -605,6 +605,31 @@ async function embedPage( slog(`${slug}: embedded ${toEmbed.length} chunks`); } +/** + * Carry code-chunk metadata (language, symbol_name, symbol_type, line range, + * parent scope, doc comment, qualified name) from a loaded Chunk back into a + * ChunkInput destined for upsertChunks. + * + * Issue #769: every re-embed used to strip these fields, and upsertChunks + * overwrites (does not COALESCE) the metadata columns from EXCLUDED, so + * each pass clobbered code-def's primary index to NULL. Pulling the + * preservation into one helper keeps the three re-embed call sites + * (embedPage, embedAll non-stale, embedAllStale) in lock-step. + */ +function preserveCodeMetadata(loaded: any, base: ChunkInput): ChunkInput { + return { + ...base, + language: loaded.language ?? undefined, + symbol_name: loaded.symbol_name ?? undefined, + symbol_type: loaded.symbol_type ?? undefined, + start_line: loaded.start_line ?? undefined, + end_line: loaded.end_line ?? undefined, + parent_symbol_path: loaded.parent_symbol_path ?? undefined, + doc_comment: loaded.doc_comment ?? undefined, + symbol_name_qualified: loaded.symbol_name_qualified ?? undefined, + }; +} + async function embedAll( engine: BrainEngine, staleOnly: boolean, @@ -717,8 +742,10 @@ async function embedAll( for (let j = 0; j < toEmbed.length; j++) { embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]); } - // Preserve ALL chunks, only update embeddings for stale ones - const updated: ChunkInput[] = chunks.map(c => ({ + // Preserve ALL chunks, only update embeddings for stale ones. + // preserveCodeMetadata threads code-chunk metadata (#769) so re-embed + // doesn't clobber language/symbol_name/symbol_type to NULL. + const updated: ChunkInput[] = chunks.map(c => preserveCodeMetadata(c, { chunk_index: c.chunk_index, chunk_text: c.chunk_text, chunk_source: c.chunk_source, @@ -1012,7 +1039,10 @@ async function embedAllStale( for (let j = 0; j < stale.length; j++) { staleIdxToEmbedding.set(stale[j].chunk_index, embeddings[j]); } - const merged: ChunkInput[] = existing.map(c => ({ + // preserveCodeMetadata threads code-chunk metadata (#769) so the + // autopilot --stale path doesn't clobber language/symbol_name/etc + // to NULL on every cycle. + const merged: ChunkInput[] = existing.map(c => preserveCodeMetadata(c, { chunk_index: c.chunk_index, chunk_text: c.chunk_text, chunk_source: c.chunk_source, diff --git a/src/commands/extract.ts b/src/commands/extract.ts index 21eeaaef5..db232ec51 100644 --- a/src/commands/extract.ts +++ b/src/commands/extract.ts @@ -1651,7 +1651,7 @@ async function extractTimelineFromDB( * make re-extraction idempotent). EVERY processed page is stamped, including * zero-link pages — they WERE processed. */ -async function extractStaleFromDB( +export async function extractStaleFromDB( engine: BrainEngine, opts: { dryRun: boolean; diff --git a/src/commands/maintain.ts b/src/commands/maintain.ts new file mode 100644 index 000000000..2442b4844 --- /dev/null +++ b/src/commands/maintain.ts @@ -0,0 +1,224 @@ +/** + * gbrain maintain — conservative self-healing maintenance. + * + * This command automates the safe parts of the operator runbook: + * - stale link/timeline extraction + * - stale per-source dream cycles when doctor reports cycle_freshness + * + * It deliberately does NOT mutate source files, apply schema-pack upgrades, or + * invent semantic hub links. Those need review or a separate command with an + * auditable proposal surface. + */ + +import { existsSync } from 'fs'; +import type { BrainEngine } from '../core/engine.ts'; +import type { BrainHealth } from '../core/types.ts'; +import { buildChecks, computeDoctorReport, type DoctorReport, type Check } from './doctor.ts'; +import { extractStaleFromDB } from './extract.ts'; +import { runCycle, type CycleReport } from '../core/cycle.ts'; + +type ActionStatus = 'ok' | 'would_apply' | 'applied' | 'blocked' | 'skipped'; + +export interface MaintenanceAction { + name: string; + status: ActionStatus; + message: string; + details?: Record; +} + +export interface MaintainOptions { + json: boolean; + safe: boolean; + dryRun: boolean; + help: boolean; +} + +export interface MaintainReport { + mode: 'dry-run' | 'safe'; + before: { + health: BrainHealth; + doctor: DoctorReport; + }; + actions: MaintenanceAction[]; + after: { + health: BrainHealth; + doctor: DoctorReport; + }; +} + +export function parseMaintainArgs(args: string[]): MaintainOptions { + const safe = args.includes('--safe'); + return { + json: args.includes('--json'), + safe, + dryRun: args.includes('--dry-run') || !safe, + help: args.includes('--help') || args.includes('-h'), + }; +} + +export function extractCycleFreshnessSourceIds(checks: Check[]): string[] { + const ids = new Set(); + for (const check of checks) { + if (check.name !== 'cycle_freshness' || check.status === 'ok') continue; + const re = /Source '([^']+)' last cycled/g; + for (const match of check.message.matchAll(re)) { + const id = match[1]?.trim(); + if (id) ids.add(id); + } + } + return [...ids].sort(); +} + +async function buildDoctorReport(engine: BrainEngine): Promise { + const checks = await buildChecks(engine, ['--json', '--scope=brain']); + return computeDoctorReport(checks); +} + +async function runStaleExtraction( + engine: BrainEngine, + beforeHealth: BrainHealth, + dryRun: boolean, +): Promise { + if (beforeHealth.stale_pages <= 0) { + return { name: 'extract_stale', status: 'ok', message: 'No stale pages.' }; + } + + if (dryRun) { + return { + name: 'extract_stale', + status: 'would_apply', + message: `Would run DB-backed stale extraction for ${beforeHealth.stale_pages} page(s).`, + details: { stale_pages: beforeHealth.stale_pages }, + }; + } + + const result = await extractStaleFromDB(engine, { + dryRun: false, + jsonMode: false, + includeFrontmatter: false, + catchUp: false, + }); + + return { + name: 'extract_stale', + status: 'applied', + message: `Processed ${result.pagesProcessed} stale page(s); ${result.staleRemaining} remain.`, + details: { + links_created: result.linksCreated, + timeline_created: result.timelineCreated, + pages_processed: result.pagesProcessed, + stale_remaining: result.staleRemaining, + }, + }; +} + +async function runCycleFreshnessMaintenance( + engine: BrainEngine, + beforeDoctor: DoctorReport, + dryRun: boolean, +): Promise { + const sourceIds = extractCycleFreshnessSourceIds(beforeDoctor.checks); + if (sourceIds.length === 0) { + return [{ name: 'cycle_freshness', status: 'ok', message: 'All sources cycled recently.' }]; + } + + if (dryRun) { + return sourceIds.map((sourceId) => ({ + name: 'cycle_freshness', + status: 'would_apply', + message: `Would run source-scoped dream cycle for ${sourceId}.`, + details: { source_id: sourceId }, + })); + } + + const sources = await engine.listAllSources(); + const actions: MaintenanceAction[] = []; + + for (const sourceId of sourceIds) { + const source = sources.find((s) => s.id === sourceId); + const localPath = source?.local_path ?? null; + const brainDir = localPath && existsSync(localPath) ? localPath : null; + const report: CycleReport = await runCycle(engine, { + brainDir, + dryRun: false, + pull: false, + sourceId, + }); + actions.push({ + name: 'cycle_freshness', + status: report.status === 'failed' ? 'blocked' : 'applied', + message: `Ran source-scoped dream cycle for ${sourceId}: ${report.status}.`, + details: { + source_id: sourceId, + brain_dir: brainDir, + cycle_status: report.status, + phases: report.phases.map((p) => ({ phase: p.phase, status: p.status })), + }, + }); + } + + return actions; +} + +export async function runMaintain(engine: BrainEngine, args: string[]): Promise { + const opts = parseMaintainArgs(args); + if (opts.help) { + console.log(`Usage: gbrain maintain [--safe] [--dry-run] [--json] + +Conservative self-healing maintenance. + +Modes: + --dry-run Preview safe actions without writes. Default when --safe is absent. + --safe Apply safe actions: stale extraction and source cycle freshness. + --json Emit a structured before/action/after report. + +Not auto-applied: + source-file frontmatter fixes, schema-pack upgrades, atom-pack changes, + semantic hub-link guesses, and destructive cleanup. +`); + return; + } + + const beforeHealth = await engine.getHealth(); + const beforeDoctor = await buildDoctorReport(engine); + const actions: MaintenanceAction[] = []; + + actions.push(await runStaleExtraction(engine, beforeHealth, opts.dryRun)); + actions.push(...await runCycleFreshnessMaintenance(engine, beforeDoctor, opts.dryRun)); + + const afterHealth = await engine.getHealth(); + const afterDoctor = await buildDoctorReport(engine); + const report: MaintainReport = { + mode: opts.dryRun ? 'dry-run' : 'safe', + before: { health: beforeHealth, doctor: beforeDoctor }, + actions, + after: { health: afterHealth, doctor: afterDoctor }, + }; + + if (opts.json) { + console.log(JSON.stringify(report, null, 2)); + } else { + printMaintainReport(report); + } + return report; +} + +function printMaintainReport(report: MaintainReport): void { + console.log(`GBrain maintain (${report.mode})`); + console.log( + `Before: brain_score=${Math.round(report.before.health.brain_score)}/100 ` + + `stale=${report.before.health.stale_pages} islands=${report.before.health.orphan_pages} ` + + `doctor=${report.before.doctor.status}`, + ); + for (const action of report.actions) { + console.log(` ${action.status}: ${action.name} — ${action.message}`); + } + console.log( + `After: brain_score=${Math.round(report.after.health.brain_score)}/100 ` + + `stale=${report.after.health.stale_pages} islands=${report.after.health.orphan_pages} ` + + `doctor=${report.after.doctor.status}`, + ); + if (report.mode === 'dry-run') { + console.log('Run `gbrain maintain --safe` to apply safe actions.'); + } +} diff --git a/src/commands/orphans.ts b/src/commands/orphans.ts index a440c1017..645dd51e2 100644 --- a/src/commands/orphans.ts +++ b/src/commands/orphans.ts @@ -15,6 +15,11 @@ import type { BrainEngine } from '../core/engine.ts'; import { createProgress, startHeartbeat } from '../core/progress.ts'; import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts'; +import { + shouldExcludeFromOrphanReporting, + loadOrphanPolicyOverrides, + type OrphanPolicyOverrides, +} from '../core/orphan-policy.ts'; // --- Types --- @@ -32,65 +37,14 @@ export interface OrphanResult { excluded: number; } -// --- Filter constants --- - -/** Slug suffixes that are always auto-generated root files */ -const AUTO_SUFFIX_PATTERNS = ['/_index', '/log']; - -/** Page slugs that are pseudo-pages by convention */ -const PSEUDO_SLUGS = new Set(['_atlas', '_index', '_stats', '_orphans', '_scratch', 'claude']); - -/** Slug segment that marks raw sources */ -const RAW_SEGMENT = '/raw/'; - -/** Slug prefixes where no inbound links is expected */ -const DENY_PREFIXES = [ - 'output/', - 'dashboards/', - 'scripts/', - 'templates/', - 'openclaw/config/', -]; - -/** First slug segments where no inbound links is expected */ -const FIRST_SEGMENT_EXCLUSIONS = new Set([ - 'scratch', - 'thoughts', - 'catalog', - 'entities', - 'raw', - 'atoms', - 'skills', -]); - // --- Filter logic --- /** * Returns true if a slug should be excluded from orphan reporting by default. * These are pages where having no inbound links is expected / not a content problem. */ -export function shouldExclude(slug: string): boolean { - // Pseudo-pages (exact match) - if (PSEUDO_SLUGS.has(slug)) return true; - - // Auto-generated suffix patterns - for (const suffix of AUTO_SUFFIX_PATTERNS) { - if (slug.endsWith(suffix)) return true; - } - - // Raw source slugs - if (slug.includes(RAW_SEGMENT)) return true; - - // Deny-prefix slugs - for (const prefix of DENY_PREFIXES) { - if (slug.startsWith(prefix)) return true; - } - - // First-segment exclusions - const firstSegment = slug.split('/')[0]; - if (FIRST_SEGMENT_EXCLUSIONS.has(firstSegment)) return true; - - return false; +export function shouldExclude(slug: string, overrides?: OrphanPolicyOverrides): boolean { + return shouldExcludeFromOrphanReporting(slug, overrides); } /** @@ -156,6 +110,7 @@ export async function findOrphans( let allOrphans: { slug: string; title: string; domain: string | null }[]; let total: number; let excludedAll: number; + const overrides = includePseudo ? undefined : await loadOrphanPolicyOverrides(engine); try { allOrphans = await engine.findOrphanPages( sourceIds ? { sourceIds } : sourceId ? { sourceId } : undefined, @@ -184,7 +139,7 @@ export async function findOrphans( total = liveRows.length; excludedAll = includePseudo ? 0 - : liveRows.reduce((n, r) => n + (shouldExclude(r.slug) ? 1 : 0), 0); + : liveRows.reduce((n, r) => n + (shouldExclude(r.slug, overrides) ? 1 : 0), 0); } finally { stopHb(); progress.finish(); @@ -192,7 +147,7 @@ export async function findOrphans( const filtered = includePseudo ? allOrphans - : allOrphans.filter(row => !shouldExclude(row.slug)); + : allOrphans.filter(row => !shouldExclude(row.slug, overrides)); const orphans: OrphanPage[] = filtered.map(row => ({ slug: row.slug, diff --git a/src/core/contextual-retrieval-service.ts b/src/core/contextual-retrieval-service.ts index 0e22b7746..a65c8e074 100644 --- a/src/core/contextual-retrieval-service.ts +++ b/src/core/contextual-retrieval-service.ts @@ -54,6 +54,7 @@ import { import { generatePerChunkSynopsis, SYNOPSIS_PROMPT_VERSION, + SYNOPSIS_DOC_MAX_CHARS, type GeneratePerChunkSynopsisResult, } from './page-summary.ts'; import { @@ -103,8 +104,17 @@ function getEmbeddingModelTag(): string { export function computeCorpusGeneration(args: { crMode: CRMode; haikuModel: string; + /** + * Resolved `SYNOPSIS_DOC_MAX_CHARS` for per_chunk_synopsis runs. When + * present, folded into the hash so changes to + * `GBRAIN_SYNOPSIS_DOC_MAX_CHARS` invalidate the prior cache cleanly. + * Omit for `crMode !== 'per_chunk_synopsis'` — title / none modes + * don't consult the cap and the field stays out of the hash for + * back-compat with pre-cap embeddings. + */ + synopsisDocMaxChars?: number; }): string { - return createHash('sha256') + const h = createHash('sha256') .update(args.crMode) .update('|') .update(String(SYNOPSIS_PROMPT_VERSION)) @@ -113,9 +123,11 @@ export function computeCorpusGeneration(args: { .update('|') .update(String(TITLE_WRAPPER_VERSION)) .update('|') - .update(getEmbeddingModelTag()) - .digest('hex') - .slice(0, 16); + .update(getEmbeddingModelTag()); + if (args.synopsisDocMaxChars !== undefined) { + h.update('|doc_cap=').update(String(args.synopsisDocMaxChars)); + } + return h.digest('hex').slice(0, 16); } /** @@ -253,7 +265,11 @@ export async function reembedPageWithContextualRetrieval( args.pageSlug, args.sourceId, resolution.mode, - computeCorpusGeneration({ crMode: resolution.mode, haikuModel: args.haikuModel ?? DEFAULT_HAIKU_MODEL }), + computeCorpusGeneration({ + crMode: resolution.mode, + haikuModel: args.haikuModel ?? DEFAULT_HAIKU_MODEL, + synopsisDocMaxChars: resolution.mode === 'per_chunk_synopsis' ? SYNOPSIS_DOC_MAX_CHARS : undefined, + }), ); return { kind: 'skipped', reason: 'no_chunks' }; } @@ -282,6 +298,7 @@ export async function reembedPageWithContextualRetrieval( const corpus_generation = computeCorpusGeneration({ crMode: attemptMode, haikuModel, + synopsisDocMaxChars: attemptMode === 'per_chunk_synopsis' ? SYNOPSIS_DOC_MAX_CHARS : undefined, }); // ── PHASE 2: single DB transaction ─────────────────────────── diff --git a/src/core/import-file.ts b/src/core/import-file.ts index 4e7722abb..f988ee1cf 100644 --- a/src/core/import-file.ts +++ b/src/core/import-file.ts @@ -733,6 +733,11 @@ export async function importFromContent( : computeCorpusGeneration({ crMode: effectiveCRMode, haikuModel: 'anthropic:claude-haiku-4-5-20251001', + // Inline import-file path never uses per_chunk_synopsis (refuses + // upstream); pass undefined so the doc-cap field stays out of + // the hash here. Per_chunk_synopsis runs through the Minion + // backfill handler which threads SYNOPSIS_DOC_MAX_CHARS through + // the service layer. }); // Transaction wraps all DB writes. Every per-page tx call carries the diff --git a/src/core/link-extraction.ts b/src/core/link-extraction.ts index c0fc2644a..8f27903e6 100644 --- a/src/core/link-extraction.ts +++ b/src/core/link-extraction.ts @@ -489,7 +489,22 @@ export async function extractPageLinks( // text inside `[[...]]` before any `|`), NOT the display alias // (ref.name = match[2]). `[[struktura|the project]]` must resolve // `struktura`, not "the project". The display text is for context only. - const matches = await resolver.resolveBasenameMatches(ref.slug); + // + // The literal may be path-qualified (`[[notes/struktura]]`). The FS + // path (resolveSlugAll) strips the dirname before its basename lookup, + // but this path passed the raw literal to an index keyed by final + // segments only — so every slash-containing wikilink outside + // DIR_PATTERN silently resolved to nothing. Query by the final + // segment, then use the written path as a disambiguation filter + // (the analogue of the FS ancestor walk honoring the written path): + // a match must end with the literal, so `[[notes/struktura]]` can + // resolve to `vault/notes/struktura` but never to `wiki/struktura`. + const slashIdx = ref.slug.lastIndexOf('/'); + const basename = slashIdx === -1 ? ref.slug : ref.slug.slice(slashIdx + 1); + let matches = await resolver.resolveBasenameMatches(basename); + if (slashIdx !== -1) { + matches = matches.filter(m => m === ref.slug || m.endsWith(`/${ref.slug}`)); + } if (matches.length === 0) continue; const idx = content.indexOf(ref.slug); const context = idx >= 0 ? excerpt(content, idx, 240) : ref.name; diff --git a/src/core/orphan-policy.ts b/src/core/orphan-policy.ts new file mode 100644 index 000000000..321777ee8 --- /dev/null +++ b/src/core/orphan-policy.ts @@ -0,0 +1,116 @@ +/** + * Shared orphan-reporting exclusion policy. + * + * These are pages where "no inbound links" is expected and should not count + * against health. Keep this in core so the CLI orphan report and engine health + * dashboard cannot drift. + * + * Defaults are GBrain-wide conventions only. Brain-specific exclusions + * (private folder names, one-off fixture slugs) belong in the brain's own + * config, not here: + * + * gbrain config set orphans.exclude_prefixes "my-private-folder/,archive/" + * gbrain config set orphans.exclude_slugs "some-one-off-page" + */ + +const AUTO_SUFFIX_PATTERNS = ['/_index', '/log']; + +const PSEUDO_SLUGS = new Set(['_atlas', '_index', '_stats', '_orphans', '_scratch', 'claude']); + +const RAW_SEGMENT = '/raw/'; + +const DENY_PREFIXES = [ + 'output/', + 'dashboards/', + 'scripts/', + 'templates/', + '_templates/', + 'openclaw/config/', + 'extracts/', +]; + +const FIRST_SEGMENT_EXCLUSIONS = new Set([ + 'scratch', + 'thoughts', + 'catalog', + 'entities', + 'raw', + 'atoms', + 'skills', + 'dreaming', + 'daily', +]); + +const ROOT_DATE_SLUG = /^\d{4}-\d{2}-\d{2}(?:-.+)?$/; + +function isAgentWorkspaceConvention(slug: string): boolean { + if (!slug.startsWith('agents/')) return false; + if (slug.includes('/memory/dreaming/')) return true; + return /^agents\/[^/]+\/(?:agents|identity|soul|tools|user|heartbeat|dreams|dormant)$/.test(slug); +} + +/** Per-brain additions to the convention defaults (from config). */ +export interface OrphanPolicyOverrides { + excludePrefixes?: string[]; + excludeSlugs?: string[]; +} + +/** Config keys for per-brain orphan exclusions (comma-separated values). */ +export const ORPHAN_EXCLUDE_PREFIXES_KEY = 'orphans.exclude_prefixes'; +export const ORPHAN_EXCLUDE_SLUGS_KEY = 'orphans.exclude_slugs'; + +function parseList(value: string | null): string[] { + if (!value) return []; + return value.split(',').map(s => s.trim()).filter(Boolean); +} + +/** + * Load per-brain orphan exclusions from the brain config table. Callers with + * an engine in hand (getHealth, `gbrain orphans`) pass the result as the + * second argument to shouldExcludeFromOrphanReporting. + */ +export async function loadOrphanPolicyOverrides( + engine: { getConfig(key: string): Promise }, +): Promise { + const [prefixes, slugs] = await Promise.all([ + engine.getConfig(ORPHAN_EXCLUDE_PREFIXES_KEY), + engine.getConfig(ORPHAN_EXCLUDE_SLUGS_KEY), + ]); + return { excludePrefixes: parseList(prefixes), excludeSlugs: parseList(slugs) }; +} + +export function shouldExcludeFromOrphanReporting( + slug: string, + overrides?: OrphanPolicyOverrides, +): boolean { + if (PSEUDO_SLUGS.has(slug)) return true; + + for (const suffix of AUTO_SUFFIX_PATTERNS) { + if (slug.endsWith(suffix)) return true; + } + + if (slug.includes(RAW_SEGMENT)) return true; + if (slug.includes('/daily/')) return true; + + for (const prefix of DENY_PREFIXES) { + if (slug.startsWith(prefix)) return true; + } + + const firstSegment = slug.split('/')[0]; + if (FIRST_SEGMENT_EXCLUSIONS.has(firstSegment)) return true; + + if (ROOT_DATE_SLUG.test(slug)) return true; + + if (slug.startsWith('_brain-')) return true; + + if (isAgentWorkspaceConvention(slug)) return true; + + if (overrides) { + if (overrides.excludeSlugs?.includes(slug)) return true; + for (const prefix of overrides.excludePrefixes ?? []) { + if (slug.startsWith(prefix)) return true; + } + } + + return false; +} diff --git a/src/core/page-summary.ts b/src/core/page-summary.ts index c4d7deedc..78e04d41c 100644 --- a/src/core/page-summary.ts +++ b/src/core/page-summary.ts @@ -44,6 +44,33 @@ const HAIKU_MAX_TOKENS = 200; /** Default model when caller doesn't override. Resolves through the gateway. */ const DEFAULT_SYNOPSIS_MODEL = 'anthropic:claude-haiku-4-5-20251001'; +/** + * Hard cap on `documentText` length (chars) before send. + * + * 2026-05-25 fix wave: small local chat models (Gemma 4 E2B, Qwen3 4B) get + * dramatically slower on long contexts even with 131K-token windows declared. + * A 73K-char page synopsis on Gemma 4 E2B takes 60-120s, exceeding the + * worker's default 30s `lockDuration` and tripping `lock-lost` errors. + * + * Truncate to a budget that fits a small model's effective throughput while + * preserving enough document context for the synopsis to be useful. Truncates + * the TAIL because the head (title, frontmatter, intro) carries the + * document-level anchor the synopsis needs. + * + * Override per workload via `GBRAIN_SYNOPSIS_DOC_MAX_CHARS`. Default 32768 + * (~8K tokens at 4 chars/tok) keeps small-model synopsis under ~30s. + * Anthropic Haiku is unaffected at this cap; bump higher when running + * frontier models if you want richer document anchoring. + */ +export const SYNOPSIS_DOC_MAX_CHARS = (() => { + const env = process.env.GBRAIN_SYNOPSIS_DOC_MAX_CHARS; + if (env && /^\d+$/.test(env)) { + const n = parseInt(env, 10); + if (n >= 512 && n <= 1_048_576) return n; + } + return 32768; +})(); + /** * Synopsis prompt version. Folded into corpus_generation so prompt edits * invalidate prior embeddings via the v0.40.3.0 query_cache.page_generations @@ -188,11 +215,19 @@ function buildUserPrompt( documentText: string, chunkText: string, ): string { + // Tail-truncate `documentText` to `SYNOPSIS_DOC_MAX_CHARS` so small local + // chat models don't stall on >100KB pages. Head preserved (title block, + // frontmatter, intro paragraphs carry the document-level anchor). + let trimmedDoc = documentText; + if (documentText.length > SYNOPSIS_DOC_MAX_CHARS) { + trimmedDoc = documentText.slice(0, SYNOPSIS_DOC_MAX_CHARS) + + `\n\n[... ${documentText.length - SYNOPSIS_DOC_MAX_CHARS} chars truncated for synopsis budget ...]`; + } return [ `${pageTitle}`, '', '', - documentText, + trimmedDoc, '', '', '', diff --git a/src/core/pglite-engine.ts b/src/core/pglite-engine.ts index 54efd0873..257090a23 100644 --- a/src/core/pglite-engine.ts +++ b/src/core/pglite-engine.ts @@ -57,6 +57,8 @@ import { finalizeLastSeen } from './chronicle/last-seen.ts'; import { computeAnomaliesFromBuckets } from './cycle/anomaly.ts'; import { resolveBoostMap, resolveHardExcludes } from './search/source-boost.ts'; import { buildSourceFactorCase, buildHardExcludeClause, buildVisibilityClause, buildRecencyComponentSql, buildBestPerPagePoolCte, buildOrFallbackWebsearchQuery } from './search/sql-ranking.ts'; +import { shouldExcludeFromOrphanReporting, loadOrphanPolicyOverrides } from './orphan-policy.ts'; +import { LINK_EXTRACTOR_VERSION_TS } from './link-extraction.ts'; import { normalizeEngineColumn, buildVectorCastFragment, @@ -2322,6 +2324,10 @@ export class PGLiteEngine implements BrainEngine { // v0.40.3.0 D24 NULL→non-NULL race fix mirrors postgres-engine.ts. Two writers // racing on the same chunk previously raced last-write-wins; the fix lets the // fresher `embedded_at` win in the text-unchanged branch. + // + // Code-chunk metadata columns follow the same chunk_text-gated CASE pattern as `embedding` + // (#769). Re-chunk trusts EXCLUDED outright; pure re-embed COALESCEs so a caller carrying + // only embedding-shaped fields doesn't clobber metadata to NULL. await this.db.query( `INSERT INTO content_chunks ${cols} VALUES ${rowParts.join(', ')} ON CONFLICT (page_id, chunk_index) DO UPDATE SET @@ -2345,14 +2351,14 @@ export class PGLiteEngine implements BrainEngine { THEN EXCLUDED.embedded_at ELSE content_chunks.embedded_at END, - language = EXCLUDED.language, - symbol_name = EXCLUDED.symbol_name, - symbol_type = EXCLUDED.symbol_type, - start_line = EXCLUDED.start_line, - end_line = EXCLUDED.end_line, - parent_symbol_path = EXCLUDED.parent_symbol_path, - doc_comment = EXCLUDED.doc_comment, - symbol_name_qualified = EXCLUDED.symbol_name_qualified, + language = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.language ELSE COALESCE(EXCLUDED.language, content_chunks.language) END, + symbol_name = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name ELSE COALESCE(EXCLUDED.symbol_name, content_chunks.symbol_name) END, + symbol_type = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_type ELSE COALESCE(EXCLUDED.symbol_type, content_chunks.symbol_type) END, + start_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.start_line ELSE COALESCE(EXCLUDED.start_line, content_chunks.start_line) END, + end_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.end_line ELSE COALESCE(EXCLUDED.end_line, content_chunks.end_line) END, + parent_symbol_path = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.parent_symbol_path ELSE COALESCE(EXCLUDED.parent_symbol_path, content_chunks.parent_symbol_path) END, + doc_comment = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.doc_comment ELSE COALESCE(EXCLUDED.doc_comment, content_chunks.doc_comment) END, + symbol_name_qualified = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name_qualified ELSE COALESCE(EXCLUDED.symbol_name_qualified, content_chunks.symbol_name_qualified) END, modality = EXCLUDED.modality, embedding_image = COALESCE(EXCLUDED.embedding_image, content_chunks.embedding_image)`, params @@ -5207,15 +5213,10 @@ export class PGLiteEngine implements BrainEngine { (SELECT count(*) FROM pages) as page_count, (SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float / GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage, - (SELECT count(*) FROM pages p - WHERE p.updated_at < (SELECT MAX(te.created_at) FROM timeline_entries te WHERE te.page_id = p.id) - ) as stale_pages, - -- Bug 11 — orphan = islanded (no inbound AND no outbound). - -- See BrainHealth.orphan_pages docstring; docs updated to match this. - (SELECT count(*) FROM pages p - WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id) - AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id) - ) as orphan_pages, + 0 as stale_pages, + -- Bug 11 — orphan = islanded (no inbound AND no outbound). The raw + -- list is filtered in TS using the shared orphan-reporting policy. + 0 as orphan_pages, (SELECT count(*) FROM links l WHERE NOT EXISTS (SELECT 1 FROM pages p WHERE p.id = l.to_page_id) ) as dead_links, @@ -5240,10 +5241,20 @@ export class PGLiteEngine implements BrainEngine { LIMIT 5 `); + const { rows: islandedRows } = await this.db.query(` + SELECT p.slug + FROM pages p + WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id) + AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id) + `); + const r = h as Record; const pageCount = Number(r.page_count); const embedCoverage = Number(r.embed_coverage); - const orphanPages = Number(r.orphan_pages); + const stalePages = await this.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS }); + const orphanOverrides = await loadOrphanPolicyOverrides(this); + const orphanPages = (islandedRows as { slug: string }[]) + .filter(row => !shouldExcludeFromOrphanReporting(row.slug, orphanOverrides)).length; const deadLinks = Number(r.dead_links); const linkCount = Number(r.link_count); const pagesWithTimeline = Number(r.pages_with_timeline); @@ -5271,7 +5282,7 @@ export class PGLiteEngine implements BrainEngine { return { page_count: pageCount, embed_coverage: embedCoverage, - stale_pages: Number(r.stale_pages), + stale_pages: stalePages, orphan_pages: orphanPages, missing_embeddings: Number(r.missing_embeddings), brain_score: brainScore, diff --git a/src/core/postgres-engine.ts b/src/core/postgres-engine.ts index b42f2a909..0e8f658f8 100644 --- a/src/core/postgres-engine.ts +++ b/src/core/postgres-engine.ts @@ -67,6 +67,8 @@ import { resolveBoostMap, resolveHardExcludes } from './search/source-boost.ts'; import { buildSourceFactorCase, buildHardExcludeClause, buildVisibilityClause, buildRecencyComponentSql, buildBestPerPagePoolCte, buildOrFallbackWebsearchQuery } from './search/sql-ranking.ts'; import { DEFAULT_EMBEDDING_MODEL, DEFAULT_EMBEDDING_DIMENSIONS } from './ai/defaults.ts'; import { DELETE_BATCH_SIZE } from './engine-constants.ts'; +import { shouldExcludeFromOrphanReporting, loadOrphanPolicyOverrides } from './orphan-policy.ts'; +import { LINK_EXTRACTOR_VERSION_TS } from './link-extraction.ts'; function escapeSqlStringLiteral(value: string): string { return value.replace(/'/g, "''"); @@ -2473,6 +2475,13 @@ export class PostgresEngine implements BrainEngine { // - new is fresher (embedded_at > existing.embedded_at) → take new // - otherwise → keep existing (slower writer with stale embedding loses) // Mirrored in pglite-engine.ts; pinned by test/e2e/concurrent-embed-race.test.ts. + // + // Code-chunk metadata columns (language / symbol_name / symbol_type / line range / + // parent_symbol_path / doc_comment / symbol_name_qualified) follow the SAME chunk_text-gated + // CASE pattern as `embedding` (#769). Re-chunk (chunk_text changed) trusts EXCLUDED outright; + // pure re-embed (chunk_text unchanged) COALESCEs so a caller that only carries embedding + // doesn't clobber metadata to NULL. Without this, every embed --stale pass nuked code-def's + // primary index for thousands of chunks at once. await sql.unsafe( `INSERT INTO content_chunks ${cols} VALUES ${rows.join(', ')} ON CONFLICT (page_id, chunk_index) DO UPDATE SET @@ -2496,14 +2505,14 @@ export class PostgresEngine implements BrainEngine { THEN EXCLUDED.embedded_at ELSE content_chunks.embedded_at END, - language = EXCLUDED.language, - symbol_name = EXCLUDED.symbol_name, - symbol_type = EXCLUDED.symbol_type, - start_line = EXCLUDED.start_line, - end_line = EXCLUDED.end_line, - parent_symbol_path = EXCLUDED.parent_symbol_path, - doc_comment = EXCLUDED.doc_comment, - symbol_name_qualified = EXCLUDED.symbol_name_qualified, + language = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.language ELSE COALESCE(EXCLUDED.language, content_chunks.language) END, + symbol_name = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name ELSE COALESCE(EXCLUDED.symbol_name, content_chunks.symbol_name) END, + symbol_type = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_type ELSE COALESCE(EXCLUDED.symbol_type, content_chunks.symbol_type) END, + start_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.start_line ELSE COALESCE(EXCLUDED.start_line, content_chunks.start_line) END, + end_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.end_line ELSE COALESCE(EXCLUDED.end_line, content_chunks.end_line) END, + parent_symbol_path = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.parent_symbol_path ELSE COALESCE(EXCLUDED.parent_symbol_path, content_chunks.parent_symbol_path) END, + doc_comment = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.doc_comment ELSE COALESCE(EXCLUDED.doc_comment, content_chunks.doc_comment) END, + symbol_name_qualified = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name_qualified ELSE COALESCE(EXCLUDED.symbol_name_qualified, content_chunks.symbol_name_qualified) END, modality = EXCLUDED.modality, embedding_image = COALESCE(EXCLUDED.embedding_image, content_chunks.embedding_image)`, params as Parameters[1], @@ -5313,11 +5322,9 @@ export class PostgresEngine implements BrainEngine { async getHealth(): Promise { const sql = this.sql; // Bug 11 doc-drift fix — orphan_pages means "islanded" (no inbound AND - // no outbound links), aligning both engines with the user-facing - // definition. The type comment previously said "no inbound" but the - // SQL required both — docs now match code so users can trust the - // number. A hub page that links out to many but has no back-references - // is working as intended, not an orphan. + // no outbound links). The raw islanded list is filtered through the same + // policy as `gbrain orphans` so convention pages do not count against + // dashboard health. const [h] = await sql` WITH entity_pages AS ( SELECT id, slug FROM pages WHERE type IN ('person', 'company') @@ -5326,13 +5333,8 @@ export class PostgresEngine implements BrainEngine { (SELECT count(*) FROM pages) as page_count, (SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float / GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage, - (SELECT count(*) FROM pages p - WHERE p.updated_at < (SELECT MAX(te.created_at) FROM timeline_entries te WHERE te.page_id = p.id) - ) as stale_pages, - (SELECT count(*) FROM pages p - WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id) - AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id) - ) as orphan_pages, + 0 as stale_pages, + 0 as orphan_pages, (SELECT count(*) FROM links l WHERE NOT EXISTS (SELECT 1 FROM pages p WHERE p.id = l.to_page_id) ) as dead_links, @@ -5356,9 +5358,18 @@ export class PostgresEngine implements BrainEngine { LIMIT 5 `; + const islandedRows = await sql<{ slug: string }[]>` + SELECT p.slug + FROM pages p + WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id) + AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id) + `; + const pageCount = Number(h.page_count); const embedCoverage = Number(h.embed_coverage); - const orphanPages = Number(h.orphan_pages); + const stalePages = await this.countStalePagesForExtraction({ versionTs: LINK_EXTRACTOR_VERSION_TS }); + const orphanOverrides = await loadOrphanPolicyOverrides(this); + const orphanPages = islandedRows.filter(row => !shouldExcludeFromOrphanReporting(row.slug, orphanOverrides)).length; const deadLinks = Number(h.dead_links); const linkCount = Number(h.link_count); const pagesWithTimeline = Number(h.pages_with_timeline); @@ -5386,7 +5397,7 @@ export class PostgresEngine implements BrainEngine { return { page_count: pageCount, embed_coverage: embedCoverage, - stale_pages: Number(h.stale_pages), + stale_pages: stalePages, orphan_pages: orphanPages, missing_embeddings: Number(h.missing_embeddings), brain_score: brainScore, diff --git a/src/core/skillopt/orchestrator.ts b/src/core/skillopt/orchestrator.ts index ffb858a83..99df6e1c4 100644 --- a/src/core/skillopt/orchestrator.ts +++ b/src/core/skillopt/orchestrator.ts @@ -93,7 +93,13 @@ import { resolveLrSchedule } from './lr-schedule.ts'; import { preflight, formatPreflightReport } from './preflight.ts'; import { isRejected, loadRejectedBuffer, makeRejectedEntry, saveRejectedBuffer } from './rejected-buffer.ts'; import { runReflect, runOneShotRewrite, describeJudges } from './reflect.ts'; -import { acceptCandidate, bestPath, revertAllPending, skillPath, writeProposed } from './version-store.ts'; +import { + acceptCandidate, + proposedPath as proposedFilePath, + revertAllPending, + skillPath, + writeProposed, +} from './version-store.ts'; import { runValidationGate, scoreSkillOnTasks } from './validate-gate.ts'; import { ROLLOUT_SUCCESS_THRESHOLD } from './types.ts'; import type { SkillOptOpts, EditOp, RunReceipt, BenchmarkTask } from './types.ts'; @@ -702,9 +708,9 @@ async function runOptimizationLoop( // to the catch's assignment values only (it can't prove the async callback ran). const finalOutcome = outcome as 'accepted' | 'no_improvement' | 'aborted' | 'errored'; if (!mutateDecision.mutate && finalOutcome === 'accepted') { - // best.md was written by writeProposed() in the accept branch (no-mutate - // path); it doubles as proposed.md for human review. SKILL.md untouched. - proposedPath = bestPath(skillsDir, skillName); + // writeProposed() emitted both the best pointer and the stable review + // artifact in the accept branch. SKILL.md remains untouched. + proposedPath = proposedFilePath(skillsDir, skillName); } else if (mutateDecision.mutate) { mutatedSkillFile = finalOutcome === 'accepted'; } diff --git a/src/core/skillopt/version-store.ts b/src/core/skillopt/version-store.ts index c723ccd10..ce2eef72a 100644 --- a/src/core/skillopt/version-store.ts +++ b/src/core/skillopt/version-store.ts @@ -23,6 +23,7 @@ * * history.json * best.md + * proposed.md * versions/ * v0001_e1_s1.md * v0002_e1_s2.md @@ -52,6 +53,10 @@ export function bestPath(skillsDir: string, skillName: string): string { return path.join(skilloptDir(skillsDir, skillName), 'best.md'); } +export function proposedPath(skillsDir: string, skillName: string): string { + return path.join(skilloptDir(skillsDir, skillName), 'proposed.md'); +} + export function skillPath(skillsDir: string, skillName: string): string { return path.join(skillsDir, skillName, 'SKILL.md'); } @@ -171,17 +176,18 @@ export function acceptCandidate(input: AcceptInput): AcceptResult { } /** - * Write the candidate to `best.md` (which doubles as `proposed.md`) WITHOUT - * touching SKILL.md or the history ledger. Used by the `--no-mutate` / - * bundled-without-allow paths: the optimizer found a better candidate but the - * caller opted out of in-place mutation, so we surface it for human review. - * Returns the path written. Atomic (.tmp + rename). + * Write the candidate to both `best.md` and `proposed.md` WITHOUT touching + * SKILL.md or the history ledger. `best.md` remains the optimizer's current + * best pointer; `proposed.md` is the stable human-review artifact promised by + * `--no-mutate`. Returns the proposal path. Each write is atomic (.tmp + rename). */ export function writeProposed(skillsDir: string, skillName: string, candidateText: string): string { - const p = bestPath(skillsDir, skillName); - fs.mkdirSync(path.dirname(p), { recursive: true }); - atomicWrite(p, candidateText); - return p; + const best = bestPath(skillsDir, skillName); + const proposed = proposedPath(skillsDir, skillName); + fs.mkdirSync(path.dirname(best), { recursive: true }); + atomicWrite(best, candidateText); + atomicWrite(proposed, candidateText); + return proposed; } /** diff --git a/src/openclaw-context-engine.ts b/src/openclaw-context-engine.ts index b1809523f..a1055cf29 100644 --- a/src/openclaw-context-engine.ts +++ b/src/openclaw-context-engine.ts @@ -63,25 +63,26 @@ interface PluginCtx { [key: string]: unknown; } +export function register(api: PluginApi) { + api.registerContextEngine(ENGINE_ID, (ctx: PluginCtx) => { + const hostResolver = + typeof ctx.resolveEntities === 'function' + ? ctx.resolveEntities + : typeof ctx.brainQuery === 'function' + ? ctx.brainQuery + : undefined; + return createGBrainContextEngine({ + workspaceDir: ctx.workspaceDir, + resolveEntities: hostResolver, + }); + }); +} + const entry: PluginEntry = { id: 'gbrain-context-engine', name: 'GBrain Context Engine', description: 'Deterministic temporal/spatial context injection on every turn', - - register(api: PluginApi) { - api.registerContextEngine(ENGINE_ID, (ctx: PluginCtx) => { - const hostResolver = - typeof ctx.resolveEntities === 'function' - ? ctx.resolveEntities - : typeof ctx.brainQuery === 'function' - ? ctx.brainQuery - : undefined; - return createGBrainContextEngine({ - workspaceDir: ctx.workspaceDir, - resolveEntities: hostResolver, - }); - }); - }, + register, }; export default entry; diff --git a/test/e2e/global-basename-pglite.test.ts b/test/e2e/global-basename-pglite.test.ts index d3c767b46..3c5af045b 100644 --- a/test/e2e/global-basename-pglite.test.ts +++ b/test/e2e/global-basename-pglite.test.ts @@ -172,6 +172,59 @@ describe('issue #972 — DB-source (gbrain extract links --source db)', () => { expect(strk!.link_type).toBe('wikilink_basename'); }); + test('flag ON → path-qualified wikilink outside DIR_PATTERN resolves via DB path', async () => { + // `[[notes/struktura]]` — `notes` is not in DIR_PATTERN, so the ref + // reaches the generic pass with its dirname intact. Regression: the DB + // path queried the basename index with the raw literal (which is keyed + // by final segments only), so path-qualified wikilinks outside + // DIR_PATTERN silently produced zero edges while the FS path resolved + // the identical content. + await engine.putPage('notes/struktura', { + type: 'concept' as any, title: 'Struktura Notes', + compiled_truth: '', timeline: '', + }); + await engine.putPage('concepts/knowledge-graph', { + type: 'concept', title: 'Knowledge Graph', + compiled_truth: 'Background in [[notes/struktura]].', timeline: '', + }); + await engine.setConfig('link_resolution.global_basename', 'true'); + + await runExtract(engine, ['links', '--source', 'db']); + + const outLinks = await engine.getLinks('concepts/knowledge-graph'); + const strk = outLinks.find(l => l.to_slug === 'notes/struktura'); + expect(strk).toBeDefined(); + expect(strk!.link_type).toBe('wikilink_basename'); + expect(strk!.link_source).toBe('wikilink-resolved'); + }); + + test('path-qualified wikilink never attaches to a basename-only sibling', async () => { + // Both notes/struktura and wiki/struktura exist. The author wrote + // `[[notes/struktura]]` — the written path must exclude wiki/struktura + // (a bare `[[struktura]]` would legitimately match both). + await engine.putPage('notes/struktura', { + type: 'concept' as any, title: 'Struktura Notes', + compiled_truth: '', timeline: '', + }); + await engine.putPage('wiki/struktura', { + type: 'concept' as any, title: 'Struktura Wiki', + compiled_truth: '', timeline: '', + }); + await engine.putPage('concepts/x', { + type: 'concept', title: 'X', + compiled_truth: 'See [[notes/struktura]].', timeline: '', + }); + await engine.setConfig('link_resolution.global_basename', 'true'); + + await runExtract(engine, ['links', '--source', 'db']); + + const outLinks = await engine.getLinks('concepts/x'); + const basenameLinks = outLinks + .filter(l => l.link_type === 'wikilink_basename') + .map(l => l.to_slug); + expect(basenameLinks).toEqual(['notes/struktura']); + }); + test('flag OFF → no basename edges via DB path (back-compat)', async () => { await engine.putPage('projects/struktura', { type: 'project', title: 'Struktura', diff --git a/test/e2e/skillopt-loop.serial.test.ts b/test/e2e/skillopt-loop.serial.test.ts index 5b5038b3d..72cf6ae42 100644 --- a/test/e2e/skillopt-loop.serial.test.ts +++ b/test/e2e/skillopt-loop.serial.test.ts @@ -39,6 +39,7 @@ import { runSkillOpt } from '../../src/core/skillopt/orchestrator.ts'; import { bestPath, loadHistory, + proposedPath, skillPath, } from '../../src/core/skillopt/version-store.ts'; import { loadRejectedBuffer } from '../../src/core/skillopt/rejected-buffer.ts'; @@ -741,7 +742,7 @@ describe('skillopt T3 — F11 held-out gate, ablation opts, no-DB-pollution', () } finally { fixture.cleanup(); } }); - test('--no-mutate writes proposed.md (best.md), leaves SKILL.md untouched', async () => { + test('--no-mutate writes proposed.md and best.md, leaves SKILL.md untouched', async () => { const fixture = setupFixture(SKILL_PEOPLE_ONLY, CITATIONS_BENCHMARK); try { installStub({ @@ -753,10 +754,9 @@ describe('skillopt T3 — F11 held-out gate, ablation opts, no-DB-pollution', () const result = await runOnce(fixture, { noMutate: true }); expect(result.outcome).toBe('accepted'); expect(result.mutatedSkillFile).toBe(false); - expect(result.proposedPath).toBeDefined(); - // proposed.md (best.md) exists and carries the improvement. - expect(fs.existsSync(result.proposedPath!)).toBe(true); + expect(result.proposedPath).toBe(proposedPath(fixture.skillsDir, SKILL)); expect(fs.readFileSync(result.proposedPath!, 'utf8')).toContain('## Citations'); + expect(fs.readFileSync(bestPath(fixture.skillsDir, SKILL), 'utf8')).toContain('## Citations'); // SKILL.md on disk is UNCHANGED (still People-only). const skill = fs.readFileSync(skillPath(fixture.skillsDir, SKILL), 'utf8'); expect(skill).not.toContain('## Citations'); diff --git a/test/embed.serial.test.ts b/test/embed.serial.test.ts index b4db83beb..a0bf89bea 100644 --- a/test/embed.serial.test.ts +++ b/test/embed.serial.test.ts @@ -803,3 +803,107 @@ describe('embedAllStale --source threading (D7)', () => { expect((firstCallOpts as { sourceId?: string }).sourceId).toBe('media-corpus'); }); }); + +// ──────────────────────────────────────────────────────────────── +// Code metadata preservation across re-embed (regression for #769) +// ──────────────────────────────────────────────────────────────── +// +// gbrain v0.30.1 and earlier silently clobbered code-chunk metadata +// (language, symbol_name, symbol_type, start_line, end_line, +// parent_symbol_path, doc_comment, symbol_name_qualified) on every +// re-embed pass. The chunker populated those columns at import time, +// but embed.ts loaded chunks via getChunks then mapped them to a +// stripped ChunkInput carrying only 5 fields. upsertChunks then +// OVERWROTE (not COALESCEd) the metadata columns from EXCLUDED, so +// re-embed wiped them to NULL. End result on a real brain: 4875 code +// pages, 47866 chunks, all with NULL language/symbol_name/symbol_type; +// code-def returned 0 hits across every indexed repo. +// +// All three runEmbed paths (--stale autopilot, --all, --slugs) must +// thread metadata through the re-upsert. Tests below assert that the +// engine.upsertChunks call carries the same metadata it loaded. + +describe('runEmbed preserves code-chunk metadata across re-embed (regression for #769)', () => { + const fullCodeChunk = { + chunk_index: 0, + chunk_text: '[Java] foo/Bar.java:10-20 method baz', + chunk_source: 'compiled_truth' as const, + embedded_at: null, + token_count: 12, + language: 'java', + symbol_name: 'baz', + symbol_type: 'function', + start_line: 10, + end_line: 20, + parent_symbol_path: ['Bar'], + doc_comment: 'does the thing', + symbol_name_qualified: 'Bar.baz', + }; + + function metadataOf(chunk: any) { + return { + language: chunk.language, + symbol_name: chunk.symbol_name, + symbol_type: chunk.symbol_type, + start_line: chunk.start_line, + end_line: chunk.end_line, + parent_symbol_path: chunk.parent_symbol_path, + doc_comment: chunk.doc_comment, + symbol_name_qualified: chunk.symbol_name_qualified, + }; + } + + test('--stale (autopilot path) carries code metadata into upsertChunks', async () => { + const stale = [{ + slug: 'code-page', + chunk_index: 0, + chunk_text: fullCodeChunk.chunk_text, + chunk_source: 'compiled_truth', + model: null, + token_count: 12, + }]; + let upsertChunkArgs: any[] | null = null; + const engine = mockEngine({ + countStaleChunks: async () => 1, + listStaleChunks: async () => stale, + getChunks: async () => [fullCodeChunk], + upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; }, + }); + + await runEmbed(engine, ['--stale']); + + expect(upsertChunkArgs).not.toBeNull(); + expect(upsertChunkArgs!).toHaveLength(1); + expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk)); + }); + + test('--all (full re-embed) carries code metadata into upsertChunks', async () => { + let upsertChunkArgs: any[] | null = null; + const engine = mockEngine({ + listPages: async () => [{ slug: 'code-page' }], + getChunks: async () => [fullCodeChunk], + upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; }, + }); + + await runEmbed(engine, ['--all']); + + expect(upsertChunkArgs).not.toBeNull(); + expect(upsertChunkArgs!).toHaveLength(1); + expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk)); + }); + + test('--slugs (per-page embed) carries code metadata into upsertChunks', async () => { + let upsertChunkArgs: any[] | null = null; + const engine = mockEngine({ + getPage: async () => ({ slug: 'code-page', compiled_truth: 'x', timeline: '' }), + getChunks: async () => [fullCodeChunk], + upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; }, + }); + + await runEmbed(engine, ['--slugs', 'code-page']); + + expect(upsertChunkArgs).not.toBeNull(); + expect(upsertChunkArgs!).toHaveLength(1); + expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk)); + }); +}); diff --git a/test/link-extraction.test.ts b/test/link-extraction.test.ts index e0a741220..f7c003ab5 100644 --- a/test/link-extraction.test.ts +++ b/test/link-extraction.test.ts @@ -403,6 +403,77 @@ describe('extractPageLinks', () => { expect(candidates).toEqual([]); }); + test('path-qualified wikilink outside DIR_PATTERN queries by final segment', async () => { + // `[[notes/struktura]]` (dir not in DIR_PATTERN) falls to the generic + // pass. The resolver's basename index is keyed by final path segments, + // so the lookup must strip the dirname — mirroring the FS path + // (resolveSlugAll). Regression: the raw literal was passed through, + // which never matched, so these links silently dropped. + const seen: string[] = []; + const resolver: SlugResolver = { + resolve: async () => null, + resolveBasenameMatches: async (name) => { + seen.push(name); + return name === 'struktura' ? ['notes/struktura'] : []; + }, + }; + const { candidates } = await extractPageLinks( + 'concepts/x', 'See [[notes/struktura]].', + {}, 'concept', resolver, { globalBasename: true }, + ); + expect(seen).toContain('struktura'); + expect(seen).not.toContain('notes/struktura'); + expect(candidates.map(c => c.targetSlug)).toEqual(['notes/struktura']); + expect(candidates[0].linkType).toBe('wikilink_basename'); + expect(candidates[0].linkSource).toBe('wikilink-resolved'); + }); + + test('path-qualified wikilink keeps only matches ending with the written path', async () => { + // The written path disambiguates: `[[notes/struktura]]` must never + // attach to `wiki/struktura` even though both share the basename. + const resolver: SlugResolver = { + resolve: async () => null, + resolveBasenameMatches: async (name) => + name === 'struktura' ? ['notes/struktura', 'wiki/struktura'] : [], + }; + const { candidates } = await extractPageLinks( + 'concepts/x', 'See [[notes/struktura]].', + {}, 'concept', resolver, { globalBasename: true }, + ); + expect(candidates.map(c => c.targetSlug)).toEqual(['notes/struktura']); + }); + + test('path-qualified wikilink matches a deeper real slug by path suffix', async () => { + // The page lives at vault/notes/struktura; the author wrote the shorter + // tail `[[notes/struktura]]`. Suffix matching connects them, while the + // basename-only sibling `wiki/struktura` stays excluded. + const resolver: SlugResolver = { + resolve: async () => null, + resolveBasenameMatches: async (name) => + name === 'struktura' ? ['vault/notes/struktura', 'wiki/struktura'] : [], + }; + const { candidates } = await extractPageLinks( + 'concepts/x', 'See [[notes/struktura]].', + {}, 'concept', resolver, { globalBasename: true }, + ); + expect(candidates.map(c => c.targetSlug)).toEqual(['vault/notes/struktura']); + }); + + test('path-qualified self-link is dropped like the bare form', async () => { + // `[[notes/struktura]]` written on notes/struktura itself must not + // produce a self-loop (same guard as the bare `[[own-tail]]` case). + const resolver: SlugResolver = { + resolve: async () => null, + resolveBasenameMatches: async (name) => + name === 'struktura' ? ['notes/struktura'] : [], + }; + const { candidates } = await extractPageLinks( + 'notes/struktura', 'See [[notes/struktura]].', + {}, 'concept', resolver, { globalBasename: true }, + ); + expect(candidates).toEqual([]); + }); + test('bare wikilink resolution does not interfere with DIR_PATTERN wikilinks', async () => { // 2b refs (people/alice) take the verb-inferred type; // 2c refs (struktura) take wikilink_basename. Same call. diff --git a/test/maintain.test.ts b/test/maintain.test.ts new file mode 100644 index 000000000..0db805a87 --- /dev/null +++ b/test/maintain.test.ts @@ -0,0 +1,62 @@ +import { describe, expect, test } from 'bun:test'; +import { + extractCycleFreshnessSourceIds, + parseMaintainArgs, +} from '../src/commands/maintain.ts'; +import type { Check } from '../src/commands/doctor.ts'; + +describe('maintain args', () => { + test('defaults to dry-run unless --safe is explicit', () => { + expect(parseMaintainArgs([])).toMatchObject({ + safe: false, + dryRun: true, + json: false, + }); + }); + + test('--safe enables mutating safe mode', () => { + expect(parseMaintainArgs(['--safe', '--json'])).toMatchObject({ + safe: true, + dryRun: false, + json: true, + }); + }); + + test('--dry-run wins over --safe', () => { + expect(parseMaintainArgs(['--safe', '--dry-run'])).toMatchObject({ + safe: true, + dryRun: true, + }); + }); +}); + +describe('cycle freshness source extraction', () => { + test('extracts stale source ids from doctor messages', () => { + const checks: Check[] = [ + { + name: 'cycle_freshness', + status: 'fail', + message: "Source 'brain-sync-remote-teffur' last cycled 40h ago. Run `gbrain dream --source `.", + }, + { + name: 'cycle_freshness', + status: 'fail', + message: "Source 'wiki' last cycled 25h ago. Source 'wiki' last cycled 25h ago.", + }, + ]; + + expect(extractCycleFreshnessSourceIds(checks)).toEqual([ + 'brain-sync-remote-teffur', + 'wiki', + ]); + }); + + test('ignores ok and unrelated checks', () => { + const checks: Check[] = [ + { name: 'cycle_freshness', status: 'ok', message: "Source 'fresh' last cycled recently." }, + { name: 'frontmatter_integrity', status: 'warn', message: "Source 'wiki' has frontmatter issues." }, + ]; + + expect(extractCycleFreshnessSourceIds(checks)).toEqual([]); + }); +}); diff --git a/test/openclaw-plugin-manifest.test.ts b/test/openclaw-plugin-manifest.test.ts new file mode 100644 index 000000000..f06882f8d --- /dev/null +++ b/test/openclaw-plugin-manifest.test.ts @@ -0,0 +1,17 @@ +import { describe, expect, it } from 'bun:test'; +import { readFileSync } from 'fs'; +import { join } from 'path'; + +describe('root OpenClaw plugin manifest', () => { + it('declares the id required by OpenClaw plugin installs', () => { + const manifest = JSON.parse(readFileSync(join(import.meta.dir, '..', 'openclaw.plugin.json'), 'utf8')); + const entrySource = readFileSync(join(import.meta.dir, '..', 'src', 'openclaw-context-engine.ts'), 'utf8'); + const entryId = entrySource.match(/id:\s*'([^']+)'/)?.[1]; + + expect(manifest.id).toBe(entryId); + expect(manifest.configSchema).toBeDefined(); + expect(typeof manifest.configSchema).toBe('object'); + expect(manifest.contracts?.contextEngines).toContain('gbrain-context'); + expect(entrySource).toContain('export function register'); + }); +}); diff --git a/test/orphans-pure-fn.test.ts b/test/orphans-pure-fn.test.ts index ada6a7d09..79df84b79 100644 --- a/test/orphans-pure-fn.test.ts +++ b/test/orphans-pure-fn.test.ts @@ -186,11 +186,67 @@ describe('shouldExclude — orphan filter regression (preserve curation)', () => expect(shouldExclude('entities/anonymous')).toBe(true); expect(shouldExclude('atoms/fact-123')).toBe(true); expect(shouldExclude('skills/gbrain-operations')).toBe(true); + expect(shouldExclude('dreaming/light/2026-07-20')).toBe(true); + expect(shouldExclude('daily/2026-07-20')).toBe(true); + expect(shouldExclude('agent-openclaw/daily/2026-07-20')).toBe(true); + }); + + test('workspace convention slugs are excluded', () => { + expect(shouldExclude('_brain-conventions')).toBe(true); + expect(shouldExclude('_templates/decision')).toBe(true); + expect(shouldExclude('extracts/2026-06-30/takes.proposed/round-single')).toBe(true); + expect(shouldExclude('2026-07-20')).toBe(true); + expect(shouldExclude('2026-07-20-qa-sweep')).toBe(true); + expect(shouldExclude('agents/arya/identity')).toBe(true); + expect(shouldExclude('agents/arya/memory/dreaming/deep/2026-07-20')).toBe(true); }); test('regular slugs are NOT excluded', () => { expect(shouldExclude('people/alice')).toBe(false); expect(shouldExclude('companies/acme')).toBe(false); expect(shouldExclude('writing/post-1')).toBe(false); + expect(shouldExclude('agents/arya/qa-reports/launch-review')).toBe(false); + }); +}); + +describe('getHealth orphan_pages uses shared exclusion policy', () => { + test('excluded convention islands do not count against health', async () => { + await engine.putPage('_templates/decision', { + type: 'template', title: 'Decision', compiled_truth: 'template', timeline: '', frontmatter: {}, + }); + await engine.putPage('skills/arya/source-check', { + type: 'concept', title: 'Skill', compiled_truth: 'skill', timeline: '', frontmatter: {}, + }); + await engine.putPage('agents/arya/identity', { + type: 'note', title: 'Identity', compiled_truth: 'identity', timeline: '', frontmatter: {}, + }); + await engine.putPage('people/alice', { + type: 'person', title: 'Alice', compiled_truth: 'real island', timeline: '', frontmatter: {}, + }); + + const health = await engine.getHealth(); + + expect(health.orphan_pages).toBe(1); + }); + + test('per-brain config overrides (orphans.exclude_*) also apply to health', async () => { + await engine.putPage('my-private-folder/secret-ref', { + type: 'note', title: 'Ref', compiled_truth: 'ref', timeline: '', frontmatter: {}, + }); + await engine.putPage('one-off-fixture-page', { + type: 'note', title: 'Fixture', compiled_truth: 'fixture', timeline: '', frontmatter: {}, + }); + await engine.putPage('people/alice', { + type: 'person', title: 'Alice', compiled_truth: 'real island', timeline: '', frontmatter: {}, + }); + + expect((await engine.getHealth()).orphan_pages).toBe(3); + + await engine.setConfig('orphans.exclude_prefixes', 'my-private-folder/'); + await engine.setConfig('orphans.exclude_slugs', 'one-off-fixture-page'); + expect((await engine.getHealth()).orphan_pages).toBe(1); + + await engine.unsetConfig('orphans.exclude_prefixes'); + await engine.unsetConfig('orphans.exclude_slugs'); }); }); diff --git a/test/orphans.test.ts b/test/orphans.test.ts index 7d56bce01..ecbfb74f0 100644 --- a/test/orphans.test.ts +++ b/test/orphans.test.ts @@ -66,6 +66,10 @@ describe('shouldExclude', () => { expect(shouldExclude('templates/meeting-note')).toBe(true); }); + test('excludes deny-prefix: _templates/', () => { + expect(shouldExclude('_templates/meeting-note')).toBe(true); + }); + test('excludes deny-prefix: openclaw/config/', () => { expect(shouldExclude('openclaw/config/agent')).toBe(true); }); @@ -86,10 +90,44 @@ describe('shouldExclude', () => { expect(shouldExclude('entities/product-hunt')).toBe(true); }); + test('excludes first-segment: skills, dreaming, and daily', () => { + expect(shouldExclude('skills/arya/source-check')).toBe(true); + expect(shouldExclude('dreaming/light/2026-07-20')).toBe(true); + expect(shouldExclude('daily/2026-07-20')).toBe(true); + expect(shouldExclude('agent-openclaw/daily/2026-07-20')).toBe(true); + }); + + test('excludes root date logs and agent workspace conventions', () => { + expect(shouldExclude('_brain-conventions')).toBe(true); + expect(shouldExclude('2026-07-20')).toBe(true); + expect(shouldExclude('2026-07-20-qa-sweep')).toBe(true); + expect(shouldExclude('agents/arya/identity')).toBe(true); + expect(shouldExclude('agents/arya/memory/dreaming/deep/2026-07-20')).toBe(true); + }); + + test('excludes generated extracts', () => { + expect(shouldExclude('extracts/2026-06-30/takes.proposed/round-single')).toBe(true); + }); + + test('brain-specific exclusions come from config overrides, not global defaults', () => { + // No baked-in defaults for these: + expect(shouldExclude('my-private-folder/some-secret-ref.md')).toBe(false); + expect(shouldExclude('one-off-fixture-page')).toBe(false); + // The per-brain config plane (orphans.exclude_prefixes / exclude_slugs): + const overrides = { + excludePrefixes: ['my-private-folder/'], + excludeSlugs: ['one-off-fixture-page'], + }; + expect(shouldExclude('my-private-folder/some-secret-ref.md', overrides)).toBe(true); + expect(shouldExclude('one-off-fixture-page', overrides)).toBe(true); + expect(shouldExclude('people/jane-doe', overrides)).toBe(false); + }); + test('does NOT exclude a normal content page', () => { expect(shouldExclude('companies/acme')).toBe(false); expect(shouldExclude('people/jane-doe')).toBe(false); expect(shouldExclude('projects/gbrain')).toBe(false); + expect(shouldExclude('agents/arya/qa-reports/launch-review')).toBe(false); }); test('does NOT exclude a page ending with log-like text that is not /log', () => { diff --git a/test/skillopt/version-store.test.ts b/test/skillopt/version-store.test.ts index 736d52611..129aec849 100644 --- a/test/skillopt/version-store.test.ts +++ b/test/skillopt/version-store.test.ts @@ -12,9 +12,11 @@ import { bestPath, historyPath, loadHistory, + proposedPath, revertAllPending, skillPath, versionsDir, + writeProposed, } from '../../src/core/skillopt/version-store.ts'; let tmpDir: string; @@ -79,6 +81,19 @@ describe('acceptCandidate (D8 two-phase commit)', () => { }); }); +describe('writeProposed', () => { + test('writes distinct best and proposed artifacts without mutating SKILL.md (#2635)', () => { + const candidate = '---\nname: test\n---\nproposed body\n'; + + const written = writeProposed(tmpDir, SKILL, candidate); + + expect(written).toBe(proposedPath(tmpDir, SKILL)); + expect(fs.readFileSync(bestPath(tmpDir, SKILL), 'utf8')).toBe(candidate); + expect(fs.readFileSync(proposedPath(tmpDir, SKILL), 'utf8')).toBe(candidate); + expect(fs.readFileSync(skillPath(tmpDir, SKILL), 'utf8')).toContain('baseline body'); + }); +}); + describe('revertAllPending (D8 crash recovery)', () => { test('no-op when no pending rows', () => { const reverted = revertAllPending(tmpDir, SKILL);