From 4cf1aa90391d84101471083a74e3b3ec55f574ab Mon Sep 17 00:00:00 2001 From: Garry Tan Date: Tue, 21 Jul 2026 14:50:41 -0700 Subject: [PATCH] fix(doctor): brain_score orphan/timeline components use the orphans-audit linkable scope Takeover of #2525, rebased onto current master. getHealth() in both engines now computes orphan_pages and the timeline component over a linkable_pages CTE driven by the same constants the orphans audit uses (src/core/linkable-scope.ts), so one doctor report can no longer show a 19% orphan_ratio next to a no-orphans score implying ~70%. Master's newer first-segment exclusions (raw, atoms, skills) are folded into the shared scope so the orphans audit loses nothing in the move. Co-authored-by: pabloglzg Co-Authored-By: Claude Fable 5 --- src/commands/orphans.ts | 67 ++--------- src/core/linkable-scope.ts | 138 +++++++++++++++++++++++ src/core/pglite-engine.ts | 29 ++++- src/core/postgres-engine.ts | 28 ++++- src/core/types.ts | 12 +- test/advisor-ranking-eval.test.ts | 4 +- test/brain-score-breakdown.test.ts | 46 ++++++++ test/brain-score-recommendations.test.ts | 1 + test/orphans-pure-fn.test.ts | 39 +++++++ 9 files changed, 298 insertions(+), 66 deletions(-) create mode 100644 src/core/linkable-scope.ts diff --git a/src/commands/orphans.ts b/src/commands/orphans.ts index a440c1017..49a415790 100644 --- a/src/commands/orphans.ts +++ b/src/commands/orphans.ts @@ -15,6 +15,7 @@ import type { BrainEngine } from '../core/engine.ts'; import { createProgress, startHeartbeat } from '../core/progress.ts'; import { getCliOptions, cliOptsToProgressOptions } from '../core/cli-options.ts'; +import { shouldExcludeFromLinkableScope } from '../core/linkable-scope.ts'; // --- Types --- @@ -32,65 +33,19 @@ export interface OrphanResult { excluded: number; } -// --- Filter constants --- - -/** Slug suffixes that are always auto-generated root files */ -const AUTO_SUFFIX_PATTERNS = ['/_index', '/log']; - -/** Page slugs that are pseudo-pages by convention */ -const PSEUDO_SLUGS = new Set(['_atlas', '_index', '_stats', '_orphans', '_scratch', 'claude']); - -/** Slug segment that marks raw sources */ -const RAW_SEGMENT = '/raw/'; - -/** Slug prefixes where no inbound links is expected */ -const DENY_PREFIXES = [ - 'output/', - 'dashboards/', - 'scripts/', - 'templates/', - 'openclaw/config/', -]; - -/** First slug segments where no inbound links is expected */ -const FIRST_SEGMENT_EXCLUSIONS = new Set([ - 'scratch', - 'thoughts', - 'catalog', - 'entities', - 'raw', - 'atoms', - 'skills', -]); - // --- Filter logic --- +// +// The exclusion constants and predicate moved to src/core/linkable-scope.ts +// so the orphans audit and brain_score's no-orphans / timeline components +// evaluate the SAME page scope (they previously disagreed inside one doctor +// report). Re-exported here for the existing public surface. /** * Returns true if a slug should be excluded from orphan reporting by default. * These are pages where having no inbound links is expected / not a content problem. */ export function shouldExclude(slug: string): boolean { - // Pseudo-pages (exact match) - if (PSEUDO_SLUGS.has(slug)) return true; - - // Auto-generated suffix patterns - for (const suffix of AUTO_SUFFIX_PATTERNS) { - if (slug.endsWith(suffix)) return true; - } - - // Raw source slugs - if (slug.includes(RAW_SEGMENT)) return true; - - // Deny-prefix slugs - for (const prefix of DENY_PREFIXES) { - if (slug.startsWith(prefix)) return true; - } - - // First-segment exclusions - const firstSegment = slug.split('/')[0]; - if (FIRST_SEGMENT_EXCLUSIONS.has(firstSegment)) return true; - - return false; + return shouldExcludeFromLinkableScope(slug); } /** @@ -128,10 +83,10 @@ export async function queryOrphanPages( * v0.42.0.0 (D1 from /plan-eng-review): this is the canonical pure data * fn for "what counts as an orphan in this brain." Re-exported as * `getOrphansData` for the doctor `orphan_ratio` check and any other - * consumer that needs the same exclusion logic (AUTO_SUFFIX_PATTERNS, - * PSEUDO_SLUGS, RAW_SEGMENT, DENY_PREFIXES, FIRST_SEGMENT_EXCLUSIONS). - * Two consumers sharing one definition = doctor and `gbrain orphans` - * cannot disagree on the orphan count. + * consumer that needs the same exclusion logic (now centralized in + * src/core/linkable-scope.ts, which getHealth's brain_score also uses). + * All consumers sharing one definition = doctor, `gbrain orphans`, and + * brain_score cannot disagree on the orphan count. */ export async function findOrphans( engine: BrainEngine, diff --git a/src/core/linkable-scope.ts b/src/core/linkable-scope.ts new file mode 100644 index 000000000..2a6205b04 --- /dev/null +++ b/src/core/linkable-scope.ts @@ -0,0 +1,138 @@ +/** + * Linkable-page scope — the single definition of which pages are expected + * to participate in the link graph. + * + * Two consumers MUST agree on this definition or `gbrain doctor` contradicts + * itself inside one report: + * + * - the orphans audit (`src/commands/orphans.ts`) excludes archive and + * generated pages from orphan accounting ("no inbound links is + * expected / not a content problem"), and + * - brain_score's no-orphans and timeline-coverage components + * (`getHealth()` in both engines) previously counted EVERY page. + * + * On a brain with a large raw archive the same report read "orphan ratio + * 19% (linkable pages)" in one check and a no-orphans score implying ~70% + * orphans two lines later. Same word, two denominators. + * + * The JS predicate (`shouldExcludeFromLinkableScope`) and the SQL fragments + * (`LINKABLE_EXCLUDE_*`) are generated from the same constants so the two + * evaluation paths cannot drift. + */ + +/** + * Slug suffixes that are always auto-generated root files. + * `/readme` — a README is a folder descriptor, not a knowledge node; + * nothing is expected to wikilink to it. + */ +export const AUTO_SUFFIX_PATTERNS = ['/_index', '/log', '/readme']; + +/** + * Page slugs that are pseudo-pages by convention. + * `readme` / `index` — root-level folder descriptors, same rationale as + * the `/readme` suffix and the existing `_index` pseudo-page. + * `schema` — written by the schema pack on init; `log` — the root brain log. + */ +export const PSEUDO_SLUGS = new Set(['_atlas', '_index', '_stats', '_orphans', '_scratch', 'claude', 'readme', 'index', 'schema', 'log']); + +/** Slug segment that marks raw sources */ +export const RAW_SEGMENT = '/raw/'; + +/** + * `raw/` as the FIRST slug segment — the same archive convention as + * RAW_SEGMENT, previously missed because `includes('/raw/')` requires a + * leading segment. Brains that keep their archive at the repo root + * (`raw/whatsapp/...`) had every archive page counted as an orphan. + */ +export const RAW_PREFIX = 'raw/'; + +/** Slug prefixes where no inbound links is expected */ +export const DENY_PREFIXES = [ + 'output/', + 'outputs/', + 'dashboards/', + 'scripts/', + 'templates/', + 'openclaw/config/', +]; + +/** + * First slug segments where no inbound links is expected. + * + * `daily` — daily records (calendar sync, email digests, activity logs — + * the pages gbrain's own calendar/email integrations write under + * `daily/...`) link OUT to the graph; nothing is expected to link back to + * a dated log page. + * + * `extracts` — machine-generated takes/extraction pages gbrain's own + * pipelines write (e.g. `extracts//takes.proposed/...`); they exist + * only in the DB and are not curated content. + */ +export const FIRST_SEGMENT_EXCLUSIONS = new Set([ + 'scratch', + 'thoughts', + 'catalog', + 'entities', + 'raw', + 'atoms', + 'skills', + 'daily', + 'extracts', + // 'inbox' — GTD-style intake tray: dated collector records in transit + // (email digests, alerts) awaiting triage; nothing links INTO an inbox + // item, same rationale as 'daily'. + 'inbox', +]); + +/** + * Returns true if a slug is excluded from the linkable scope: pages where + * having no inbound links is expected / not a content problem. + */ +export function shouldExcludeFromLinkableScope(slug: string): boolean { + // Pseudo-pages (exact match) + if (PSEUDO_SLUGS.has(slug)) return true; + + // Auto-generated suffix patterns + for (const suffix of AUTO_SUFFIX_PATTERNS) { + if (slug.endsWith(suffix)) return true; + } + + // Raw source slugs (any segment, including leading) + if (slug.includes(RAW_SEGMENT)) return true; + if (slug.startsWith(RAW_PREFIX)) return true; + + // Deny-prefix slugs + for (const prefix of DENY_PREFIXES) { + if (slug.startsWith(prefix)) return true; + } + + // First-segment exclusions + const firstSegment = slug.split('/')[0]; + if (FIRST_SEGMENT_EXCLUSIONS.has(firstSegment)) return true; + + return false; +} + +// --- SQL projection of the same predicate --- +// +// getHealth() evaluates the scope inside one SQL statement. These arrays are +// derived from the constants above; a slug is excluded when it matches ANY +// LIKE pattern, equals ANY exact slug, or its first segment equals ANY +// excluded segment: +// +// slug NOT LIKE ALL($like) AND slug != ALL($exact) +// AND split_part(slug, '/', 1) != ALL($segments) + +/** LIKE patterns — a slug matching any of these is excluded. */ +export const LINKABLE_EXCLUDE_LIKE: string[] = [ + `%${RAW_SEGMENT}%`, + `${RAW_PREFIX}%`, + ...DENY_PREFIXES.map((p) => `${p}%`), + ...AUTO_SUFFIX_PATTERNS.map((s) => `%${s}`), +]; + +/** Exact slugs excluded (pseudo-pages). */ +export const LINKABLE_EXCLUDE_EXACT: string[] = [...PSEUDO_SLUGS]; + +/** First slug segments excluded. */ +export const LINKABLE_EXCLUDE_FIRST_SEGMENTS: string[] = [...FIRST_SEGMENT_EXCLUSIONS]; diff --git a/src/core/pglite-engine.ts b/src/core/pglite-engine.ts index 54efd0873..06e3b6f00 100644 --- a/src/core/pglite-engine.ts +++ b/src/core/pglite-engine.ts @@ -24,6 +24,7 @@ import { PGLITE_SCHEMA_SQL, getPGLiteSchema } from './pglite-schema.ts'; import { DEFAULT_EMBEDDING_MODEL, DEFAULT_EMBEDDING_DIMENSIONS } from './ai/defaults.ts'; import { DELETE_BATCH_SIZE } from './engine-constants.ts'; import { MARKDOWN_CHUNKER_VERSION } from './chunkers/recursive.ts'; +import { LINKABLE_EXCLUDE_LIKE, LINKABLE_EXCLUDE_EXACT, LINKABLE_EXCLUDE_FIRST_SEGMENTS } from './linkable-scope.ts'; import { acquireLock, releaseLock, type LockHandle } from './pglite-lock.ts'; import { getFtsLanguage } from './fts-language.ts'; import type { @@ -5199,12 +5200,23 @@ export class PGLiteEngine implements BrainEngine { // pages_with_timeline) and v0.10.3 graph layer (link_coverage, timeline_coverage, // most_connected). Both coexist: master's brain_score is the composite // dashboard, v0.10.3 metrics give entity-page-level granularity. + // orphan_pages and timeline coverage are computed over LINKABLE pages + // (src/core/linkable-scope.ts) — the same scope the orphans audit uses — + // so brain_score and `gbrain orphans` / doctor's orphan_ratio cannot + // disagree on what counts. Archive (raw/), generated, and daily-log + // pages are not expected to participate in the curated graph. const { rows: [h] } = await this.db.query(` WITH entity_pages AS ( SELECT id, slug FROM pages WHERE type IN ('person', 'company') + ), linkable_pages AS ( + SELECT id FROM pages + WHERE slug NOT LIKE ALL($1) + AND NOT (slug = ANY($2)) + AND NOT (split_part(slug, '/', 1) = ANY($3)) ) SELECT (SELECT count(*) FROM pages) as page_count, + (SELECT count(*) FROM linkable_pages) as linkable_page_count, (SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float / GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage, (SELECT count(*) FROM pages p @@ -5212,7 +5224,7 @@ export class PGLiteEngine implements BrainEngine { ) as stale_pages, -- Bug 11 — orphan = islanded (no inbound AND no outbound). -- See BrainHealth.orphan_pages docstring; docs updated to match this. - (SELECT count(*) FROM pages p + (SELECT count(*) FROM linkable_pages p WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id) AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id) ) as orphan_pages, @@ -5222,13 +5234,15 @@ export class PGLiteEngine implements BrainEngine { (SELECT count(*) FROM content_chunks WHERE embedded_at IS NULL) as missing_embeddings, (SELECT count(*) FROM links) as link_count, (SELECT count(DISTINCT page_id) FROM timeline_entries) as pages_with_timeline, + (SELECT count(DISTINCT te.page_id) FROM timeline_entries te + WHERE te.page_id IN (SELECT id FROM linkable_pages)) as linkable_pages_with_timeline, (SELECT count(*) FROM entity_pages e WHERE EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = e.id))::float / GREATEST((SELECT count(*) FROM entity_pages), 1)::float as link_coverage, (SELECT count(*) FROM entity_pages e WHERE EXISTS (SELECT 1 FROM timeline_entries te WHERE te.page_id = e.id))::float / GREATEST((SELECT count(*) FROM entity_pages), 1)::float as timeline_coverage - `); + `, [LINKABLE_EXCLUDE_LIKE, LINKABLE_EXCLUDE_EXACT, LINKABLE_EXCLUDE_FIRST_SEGMENTS]); // Top 5 most connected entities by total link count (in + out). const { rows: connected } = await this.db.query(` @@ -5242,15 +5256,18 @@ export class PGLiteEngine implements BrainEngine { const r = h as Record; const pageCount = Number(r.page_count); + const linkablePageCount = Number(r.linkable_page_count); const embedCoverage = Number(r.embed_coverage); const orphanPages = Number(r.orphan_pages); const deadLinks = Number(r.dead_links); const linkCount = Number(r.link_count); const pagesWithTimeline = Number(r.pages_with_timeline); + const linkableTimelinePages = Number(r.linkable_pages_with_timeline); const linkDensity = pageCount > 0 ? Math.min(linkCount / pageCount, 1) : 0; - const timelineCoverageDensity = pageCount > 0 ? Math.min(pagesWithTimeline / pageCount, 1) : 0; - const noOrphans = pageCount > 0 ? 1 - (orphanPages / pageCount) : 1; + const timelineCoverageDensity = + linkablePageCount > 0 ? Math.min(linkableTimelinePages / linkablePageCount, 1) : 1; + const noOrphans = linkablePageCount > 0 ? 1 - (orphanPages / linkablePageCount) : 1; const noDeadLinks = pageCount > 0 ? 1 - Math.min(deadLinks / pageCount, 1) : 1; // Bug 11 — per-component points. Sum equals brainScore by construction // so `doctor` can render a breakdown that adds up to the total. @@ -5261,6 +5278,9 @@ export class PGLiteEngine implements BrainEngine { // pre-fix "empty = 0" caused fresh-init brains to score as critically // unhealthy on `gbrain doctor`, which was a structural surprise to users // who'd just successfully run init. + // The same rule extends to linkablePageCount === 0 for the orphan / + // timeline components: an all-archive brain has no curated graph to + // penalize. const embedCoverageScore = pageCount === 0 ? 35 : Math.round(embedCoverage * 35); const linkDensityScore = pageCount === 0 ? 25 : Math.round(linkDensity * 25); const timelineCoverageScore = pageCount === 0 ? 15 : Math.round(timelineCoverageDensity * 15); @@ -5270,6 +5290,7 @@ export class PGLiteEngine implements BrainEngine { return { page_count: pageCount, + linkable_page_count: linkablePageCount, embed_coverage: embedCoverage, stale_pages: Number(r.stale_pages), orphan_pages: orphanPages, diff --git a/src/core/postgres-engine.ts b/src/core/postgres-engine.ts index b42f2a909..75fbc9e28 100644 --- a/src/core/postgres-engine.ts +++ b/src/core/postgres-engine.ts @@ -67,6 +67,7 @@ import { resolveBoostMap, resolveHardExcludes } from './search/source-boost.ts'; import { buildSourceFactorCase, buildHardExcludeClause, buildVisibilityClause, buildRecencyComponentSql, buildBestPerPagePoolCte, buildOrFallbackWebsearchQuery } from './search/sql-ranking.ts'; import { DEFAULT_EMBEDDING_MODEL, DEFAULT_EMBEDDING_DIMENSIONS } from './ai/defaults.ts'; import { DELETE_BATCH_SIZE } from './engine-constants.ts'; +import { LINKABLE_EXCLUDE_LIKE, LINKABLE_EXCLUDE_EXACT, LINKABLE_EXCLUDE_FIRST_SEGMENTS } from './linkable-scope.ts'; function escapeSqlStringLiteral(value: string): string { return value.replace(/'/g, "''"); @@ -5318,18 +5319,30 @@ export class PostgresEngine implements BrainEngine { // SQL required both — docs now match code so users can trust the // number. A hub page that links out to many but has no back-references // is working as intended, not an orphan. + // + // orphan_pages and timeline coverage are computed over LINKABLE pages + // (src/core/linkable-scope.ts) — the same scope the orphans audit uses — + // so brain_score and `gbrain orphans` / doctor's orphan_ratio cannot + // disagree on what counts. Archive (raw/), generated, and daily-log + // pages are not expected to participate in the curated graph. const [h] = await sql` WITH entity_pages AS ( SELECT id, slug FROM pages WHERE type IN ('person', 'company') + ), linkable_pages AS ( + SELECT id FROM pages + WHERE slug NOT LIKE ALL(${LINKABLE_EXCLUDE_LIKE}) + AND NOT (slug = ANY(${LINKABLE_EXCLUDE_EXACT})) + AND NOT (split_part(slug, '/', 1) = ANY(${LINKABLE_EXCLUDE_FIRST_SEGMENTS})) ) SELECT (SELECT count(*) FROM pages) as page_count, + (SELECT count(*) FROM linkable_pages) as linkable_page_count, (SELECT count(*) FROM content_chunks WHERE embedded_at IS NOT NULL)::float / GREATEST((SELECT count(*) FROM content_chunks), 1)::float as embed_coverage, (SELECT count(*) FROM pages p WHERE p.updated_at < (SELECT MAX(te.created_at) FROM timeline_entries te WHERE te.page_id = p.id) ) as stale_pages, - (SELECT count(*) FROM pages p + (SELECT count(*) FROM linkable_pages p WHERE NOT EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = p.id) AND NOT EXISTS (SELECT 1 FROM links l WHERE l.from_page_id = p.id) ) as orphan_pages, @@ -5339,6 +5352,8 @@ export class PostgresEngine implements BrainEngine { (SELECT count(*) FROM content_chunks WHERE embedded_at IS NULL) as missing_embeddings, (SELECT count(*) FROM links) as link_count, (SELECT count(DISTINCT page_id) FROM timeline_entries) as pages_with_timeline, + (SELECT count(DISTINCT te.page_id) FROM timeline_entries te + WHERE te.page_id IN (SELECT id FROM linkable_pages)) as linkable_pages_with_timeline, (SELECT count(*) FROM entity_pages e WHERE EXISTS (SELECT 1 FROM links l WHERE l.to_page_id = e.id))::float / GREATEST((SELECT count(*) FROM entity_pages), 1)::float as link_coverage, @@ -5357,16 +5372,19 @@ export class PostgresEngine implements BrainEngine { `; const pageCount = Number(h.page_count); + const linkablePageCount = Number(h.linkable_page_count); const embedCoverage = Number(h.embed_coverage); const orphanPages = Number(h.orphan_pages); const deadLinks = Number(h.dead_links); const linkCount = Number(h.link_count); const pagesWithTimeline = Number(h.pages_with_timeline); + const linkableTimelinePages = Number(h.linkable_pages_with_timeline); // brain_score: 0-100 weighted average const linkDensity = pageCount > 0 ? Math.min(linkCount / pageCount, 1) : 0; - const timelineCoverageWhole = pageCount > 0 ? Math.min(pagesWithTimeline / pageCount, 1) : 0; - const noOrphans = pageCount > 0 ? 1 - (orphanPages / pageCount) : 1; + const timelineCoverageWhole = + linkablePageCount > 0 ? Math.min(linkableTimelinePages / linkablePageCount, 1) : 1; + const noOrphans = linkablePageCount > 0 ? 1 - (orphanPages / linkablePageCount) : 1; const noDeadLinks = pageCount > 0 ? 1 - Math.min(deadLinks / pageCount, 1) : 1; // Per-component points. Sum equals brainScore by construction. // @@ -5376,6 +5394,9 @@ export class PostgresEngine implements BrainEngine { // pre-fix "empty = 0" caused fresh-init brains to score as critically // unhealthy on `gbrain doctor`, which was a structural surprise to users // who'd just successfully run init. PGLite path has the same fix. + // The same rule extends to linkablePageCount === 0 for the orphan / + // timeline components: an all-archive brain has no curated graph to + // penalize. const embedCoverageScore = pageCount === 0 ? 35 : Math.round(embedCoverage * 35); const linkDensityScore = pageCount === 0 ? 25 : Math.round(linkDensity * 25); const timelineCoverageScore = pageCount === 0 ? 15 : Math.round(timelineCoverageWhole * 15); @@ -5385,6 +5406,7 @@ export class PostgresEngine implements BrainEngine { return { page_count: pageCount, + linkable_page_count: linkablePageCount, embed_coverage: embedCoverage, stale_pages: Number(h.stale_pages), orphan_pages: orphanPages, diff --git a/src/core/types.ts b/src/core/types.ts index 9464d16de..2c29c291d 100644 --- a/src/core/types.ts +++ b/src/core/types.ts @@ -1422,10 +1422,20 @@ export interface BrainStats { export interface BrainHealth { page_count: number; + /** + * Pages inside the linkable scope (src/core/linkable-scope.ts) — the + * pages expected to participate in the curated link graph. Excludes + * archive (raw/), generated, and daily-log pages; the same scope the + * orphans audit uses. Denominator for the no-orphans and + * timeline-coverage score components. + */ + linkable_page_count: number; embed_coverage: number; stale_pages: number; /** - * Islanded pages — zero inbound AND zero outbound links. A hub page + * Islanded pages — zero inbound AND zero outbound links, counted over + * LINKABLE pages only (the same scope as the `gbrain orphans` audit, so + * doctor cannot report two contradictory orphan numbers). A hub page * that has references out but no back-references is NOT an orphan under * this definition (it's working as intended as an index). The metric * aims at "pages I forgot to connect to anything", not the stricter diff --git a/test/advisor-ranking-eval.test.ts b/test/advisor-ranking-eval.test.ts index 3d828c08c..219f33a7f 100644 --- a/test/advisor-ranking-eval.test.ts +++ b/test/advisor-ranking-eval.test.ts @@ -35,7 +35,7 @@ const HEALTHY_STATS = { page_count: 500, chunk_count: 0, embedded_count: 0, link_count: 0, tag_count: 0, timeline_entry_count: 0, pages_by_type: {}, }), getHealth: async () => ({ - page_count: 500, embed_coverage: 0.99, stale_pages: 0, orphan_pages: 0, missing_embeddings: 0, + page_count: 500, linkable_page_count: 500, embed_coverage: 0.99, stale_pages: 0, orphan_pages: 0, missing_embeddings: 0, brain_score: 95, dead_links: 0, link_coverage: 1, timeline_coverage: 1, most_connected: [], embed_coverage_score: 35, link_density_score: 25, timeline_coverage_score: 15, no_orphans_score: 15, no_dead_links_score: 10, }), @@ -55,7 +55,7 @@ const FIXTURES: Fixture[] = [ engine: { ...HEALTHY_STATS, getHealth: async () => ({ - page_count: 500, embed_coverage: 0.3, stale_pages: 0, orphan_pages: 12, missing_embeddings: 350, + page_count: 500, linkable_page_count: 500, embed_coverage: 0.3, stale_pages: 0, orphan_pages: 12, missing_embeddings: 350, brain_score: 40, dead_links: 2, link_coverage: 0.2, timeline_coverage: 0.2, most_connected: [], embed_coverage_score: 10, link_density_score: 5, timeline_coverage_score: 3, no_orphans_score: 2, no_dead_links_score: 8, }), diff --git a/test/brain-score-breakdown.test.ts b/test/brain-score-breakdown.test.ts index 55b9a6e96..5eb609c23 100644 --- a/test/brain-score-breakdown.test.ts +++ b/test/brain-score-breakdown.test.ts @@ -143,3 +143,49 @@ describe('Bug 11 — BrainHealth type shape', () => { expect(typesSource).toContain('0-100'); }); }); + +describe('linkable scope — archive pages do not drag the score', () => { + test('islanded raw/ and daily/ pages are excluded from the orphan component', async () => { + // Curated, connected pages. + await engine.putPage('people/alice-example', { type: 'person', title: 'Alice', compiled_truth: 'x', frontmatter: {} }); + await engine.putPage('companies/acme-example', { type: 'company', title: 'Acme', compiled_truth: 'x', frontmatter: {} }); + const { rows: ids } = await (engine as any).db.query( + `SELECT id, slug FROM pages ORDER BY slug`, + ); + const bySlug = Object.fromEntries(ids.map((r: any) => [r.slug, r.id])); + await (engine as any).db.query( + `INSERT INTO links (from_page_id, to_page_id, link_type) VALUES ($1, $2, 'works_at')`, + [bySlug['people/alice-example'], bySlug['companies/acme-example']], + ); + // Archive + daily-log pages: no links, no timeline — by design. + await engine.putPage('raw/whatsapp/2025-01/log-page', { type: 'note', title: 'raw log', compiled_truth: 'x', frontmatter: {} }); + await engine.putPage('deals/acme-seed/raw/transcript', { type: 'note', title: 'raw t', compiled_truth: 'x', frontmatter: {} }); + await engine.putPage('daily/calendar/2025/2025-01-01', { type: 'note', title: 'day', compiled_truth: 'x', frontmatter: {} }); + + const h = await engine.getHealth(); + // The three archive/log pages are outside the linkable scope... + expect(h.linkable_page_count).toBe(2); + // ...so none of them is an orphan, and the connected pair keeps 15/15. + expect(h.orphan_pages).toBe(0); + expect(h.no_orphans_score).toBe(15); + }); + + test('timeline coverage is measured over linkable pages only', async () => { + await engine.putPage('people/alice-example', { type: 'person', title: 'Alice', compiled_truth: 'x', frontmatter: {} }); + await engine.addTimelineEntry('people/alice-example', { date: '2025-01-01', source: 'note', summary: 'joined' }); + // A raw archive page without timeline must not dilute coverage. + await engine.putPage('raw/whatsapp/2025-01/log-page', { type: 'note', title: 'raw log', compiled_truth: 'x', frontmatter: {} }); + + const h = await engine.getHealth(); + expect(h.linkable_page_count).toBe(1); + expect(h.timeline_coverage_score).toBe(15); // 1/1 linkable pages covered + }); + + test('an islanded curated page still counts as an orphan', async () => { + await engine.putPage('people/forgotten-example', { type: 'person', title: 'F', compiled_truth: 'x', frontmatter: {} }); + const h = await engine.getHealth(); + expect(h.orphan_pages).toBe(1); + expect(h.linkable_page_count).toBe(1); + expect(h.no_orphans_score).toBe(0); + }); +}); diff --git a/test/brain-score-recommendations.test.ts b/test/brain-score-recommendations.test.ts index 998d5f1b4..d3935de9c 100644 --- a/test/brain-score-recommendations.test.ts +++ b/test/brain-score-recommendations.test.ts @@ -77,6 +77,7 @@ describe('embeddingProviderConfigured (recipe-aware helper)', () => { function makeHealth(overrides: Partial = {}): BrainHealth { return { page_count: 100, + linkable_page_count: 100, embed_coverage: 1.0, stale_pages: 0, orphan_pages: 0, diff --git a/test/orphans-pure-fn.test.ts b/test/orphans-pure-fn.test.ts index ada6a7d09..f2ead624e 100644 --- a/test/orphans-pure-fn.test.ts +++ b/test/orphans-pure-fn.test.ts @@ -172,6 +172,45 @@ describe('shouldExclude — orphan filter regression (preserve curation)', () => expect(shouldExclude('raw/chats/claude-code/session')).toBe(true); }); + test('leading raw/ segment is excluded (same archive convention)', () => { + expect(shouldExclude('raw/whatsapp/2025-01/chat-log')).toBe(true); + expect(shouldExclude('raw/transcripts/meeting')).toBe(true); + // 'rawhide/...' must NOT match — prefix is 'raw/', not 'raw'. + expect(shouldExclude('rawhide/notes')).toBe(false); + }); + + test('daily-log pages are excluded (calendar/email integrations write these)', () => { + expect(shouldExclude('daily/calendar/2025/2025-01-01')).toBe(true); + expect(shouldExclude('daily/x/2025-06-13')).toBe(true); + }); + + test('outputs/ plural prefix is excluded like output/', () => { + expect(shouldExclude('outputs/render-batch-3')).toBe(true); + }); + + test('readme folder descriptors are excluded at any depth', () => { + expect(shouldExclude('readme')).toBe(true); + expect(shouldExclude('index')).toBe(true); + expect(shouldExclude('projects/readme')).toBe(true); + expect(shouldExclude('media/readme')).toBe(true); + // A page merely mentioning readme in its name is NOT excluded. + expect(shouldExclude('concepts/readme-driven-development')).toBe(false); + }); + + test('machine-generated extracts pages are excluded', () => { + expect(shouldExclude('extracts/2026-06-12/takes.proposed/host/propose-x/round-single')).toBe(true); + }); + + test('inbox intake-tray pages are excluded (same rationale as daily)', () => { + expect(shouldExclude('inbox/some-renewal-notice-2026-06-22')).toBe(true); + }); + + test('root schema and log pages are excluded', () => { + expect(shouldExclude('schema')).toBe(true); + expect(shouldExclude('log')).toBe(true); + expect(shouldExclude('concepts/schema-design')).toBe(false); + }); + test('deny-prefixes are excluded', () => { expect(shouldExclude('templates/meeting')).toBe(true); expect(shouldExclude('dashboards/_index')).toBe(true);