diff --git a/src/commands/embed.ts b/src/commands/embed.ts
index 6d8816a6e..348d76812 100644
--- a/src/commands/embed.ts
+++ b/src/commands/embed.ts
@@ -581,7 +581,7 @@ async function embedPage(
for (let j = 0; j < toEmbed.length; j++) {
embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]);
}
- const updated: ChunkInput[] = chunks.map(c => ({
+ const updated: ChunkInput[] = chunks.map(c => preserveCodeMetadata(c, {
chunk_index: c.chunk_index,
chunk_text: c.chunk_text,
chunk_source: c.chunk_source,
@@ -605,6 +605,31 @@ async function embedPage(
slog(`${slug}: embedded ${toEmbed.length} chunks`);
}
+/**
+ * Carry code-chunk metadata (language, symbol_name, symbol_type, line range,
+ * parent scope, doc comment, qualified name) from a loaded Chunk back into a
+ * ChunkInput destined for upsertChunks.
+ *
+ * Issue #769: every re-embed used to strip these fields, and upsertChunks
+ * overwrites (does not COALESCE) the metadata columns from EXCLUDED, so
+ * each pass clobbered code-def's primary index to NULL. Pulling the
+ * preservation into one helper keeps the three re-embed call sites
+ * (embedPage, embedAll non-stale, embedAllStale) in lock-step.
+ */
+function preserveCodeMetadata(loaded: any, base: ChunkInput): ChunkInput {
+ return {
+ ...base,
+ language: loaded.language ?? undefined,
+ symbol_name: loaded.symbol_name ?? undefined,
+ symbol_type: loaded.symbol_type ?? undefined,
+ start_line: loaded.start_line ?? undefined,
+ end_line: loaded.end_line ?? undefined,
+ parent_symbol_path: loaded.parent_symbol_path ?? undefined,
+ doc_comment: loaded.doc_comment ?? undefined,
+ symbol_name_qualified: loaded.symbol_name_qualified ?? undefined,
+ };
+}
+
async function embedAll(
engine: BrainEngine,
staleOnly: boolean,
@@ -717,8 +742,10 @@ async function embedAll(
for (let j = 0; j < toEmbed.length; j++) {
embeddingMap.set(toEmbed[j].chunk_index, embeddings[j]);
}
- // Preserve ALL chunks, only update embeddings for stale ones
- const updated: ChunkInput[] = chunks.map(c => ({
+ // Preserve ALL chunks, only update embeddings for stale ones.
+ // preserveCodeMetadata threads code-chunk metadata (#769) so re-embed
+ // doesn't clobber language/symbol_name/symbol_type to NULL.
+ const updated: ChunkInput[] = chunks.map(c => preserveCodeMetadata(c, {
chunk_index: c.chunk_index,
chunk_text: c.chunk_text,
chunk_source: c.chunk_source,
@@ -1012,7 +1039,10 @@ async function embedAllStale(
for (let j = 0; j < stale.length; j++) {
staleIdxToEmbedding.set(stale[j].chunk_index, embeddings[j]);
}
- const merged: ChunkInput[] = existing.map(c => ({
+ // preserveCodeMetadata threads code-chunk metadata (#769) so the
+ // autopilot --stale path doesn't clobber language/symbol_name/etc
+ // to NULL on every cycle.
+ const merged: ChunkInput[] = existing.map(c => preserveCodeMetadata(c, {
chunk_index: c.chunk_index,
chunk_text: c.chunk_text,
chunk_source: c.chunk_source,
diff --git a/src/core/contextual-retrieval-service.ts b/src/core/contextual-retrieval-service.ts
index 0e22b7746..a65c8e074 100644
--- a/src/core/contextual-retrieval-service.ts
+++ b/src/core/contextual-retrieval-service.ts
@@ -54,6 +54,7 @@ import {
import {
generatePerChunkSynopsis,
SYNOPSIS_PROMPT_VERSION,
+ SYNOPSIS_DOC_MAX_CHARS,
type GeneratePerChunkSynopsisResult,
} from './page-summary.ts';
import {
@@ -103,8 +104,17 @@ function getEmbeddingModelTag(): string {
export function computeCorpusGeneration(args: {
crMode: CRMode;
haikuModel: string;
+ /**
+ * Resolved `SYNOPSIS_DOC_MAX_CHARS` for per_chunk_synopsis runs. When
+ * present, folded into the hash so changes to
+ * `GBRAIN_SYNOPSIS_DOC_MAX_CHARS` invalidate the prior cache cleanly.
+ * Omit for `crMode !== 'per_chunk_synopsis'` — title / none modes
+ * don't consult the cap and the field stays out of the hash for
+ * back-compat with pre-cap embeddings.
+ */
+ synopsisDocMaxChars?: number;
}): string {
- return createHash('sha256')
+ const h = createHash('sha256')
.update(args.crMode)
.update('|')
.update(String(SYNOPSIS_PROMPT_VERSION))
@@ -113,9 +123,11 @@ export function computeCorpusGeneration(args: {
.update('|')
.update(String(TITLE_WRAPPER_VERSION))
.update('|')
- .update(getEmbeddingModelTag())
- .digest('hex')
- .slice(0, 16);
+ .update(getEmbeddingModelTag());
+ if (args.synopsisDocMaxChars !== undefined) {
+ h.update('|doc_cap=').update(String(args.synopsisDocMaxChars));
+ }
+ return h.digest('hex').slice(0, 16);
}
/**
@@ -253,7 +265,11 @@ export async function reembedPageWithContextualRetrieval(
args.pageSlug,
args.sourceId,
resolution.mode,
- computeCorpusGeneration({ crMode: resolution.mode, haikuModel: args.haikuModel ?? DEFAULT_HAIKU_MODEL }),
+ computeCorpusGeneration({
+ crMode: resolution.mode,
+ haikuModel: args.haikuModel ?? DEFAULT_HAIKU_MODEL,
+ synopsisDocMaxChars: resolution.mode === 'per_chunk_synopsis' ? SYNOPSIS_DOC_MAX_CHARS : undefined,
+ }),
);
return { kind: 'skipped', reason: 'no_chunks' };
}
@@ -282,6 +298,7 @@ export async function reembedPageWithContextualRetrieval(
const corpus_generation = computeCorpusGeneration({
crMode: attemptMode,
haikuModel,
+ synopsisDocMaxChars: attemptMode === 'per_chunk_synopsis' ? SYNOPSIS_DOC_MAX_CHARS : undefined,
});
// ── PHASE 2: single DB transaction ───────────────────────────
diff --git a/src/core/import-file.ts b/src/core/import-file.ts
index 4e7722abb..f988ee1cf 100644
--- a/src/core/import-file.ts
+++ b/src/core/import-file.ts
@@ -733,6 +733,11 @@ export async function importFromContent(
: computeCorpusGeneration({
crMode: effectiveCRMode,
haikuModel: 'anthropic:claude-haiku-4-5-20251001',
+ // Inline import-file path never uses per_chunk_synopsis (refuses
+ // upstream); pass undefined so the doc-cap field stays out of
+ // the hash here. Per_chunk_synopsis runs through the Minion
+ // backfill handler which threads SYNOPSIS_DOC_MAX_CHARS through
+ // the service layer.
});
// Transaction wraps all DB writes. Every per-page tx call carries the
diff --git a/src/core/link-extraction.ts b/src/core/link-extraction.ts
index c0fc2644a..8f27903e6 100644
--- a/src/core/link-extraction.ts
+++ b/src/core/link-extraction.ts
@@ -489,7 +489,22 @@ export async function extractPageLinks(
// text inside `[[...]]` before any `|`), NOT the display alias
// (ref.name = match[2]). `[[struktura|the project]]` must resolve
// `struktura`, not "the project". The display text is for context only.
- const matches = await resolver.resolveBasenameMatches(ref.slug);
+ //
+ // The literal may be path-qualified (`[[notes/struktura]]`). The FS
+ // path (resolveSlugAll) strips the dirname before its basename lookup,
+ // but this path passed the raw literal to an index keyed by final
+ // segments only — so every slash-containing wikilink outside
+ // DIR_PATTERN silently resolved to nothing. Query by the final
+ // segment, then use the written path as a disambiguation filter
+ // (the analogue of the FS ancestor walk honoring the written path):
+ // a match must end with the literal, so `[[notes/struktura]]` can
+ // resolve to `vault/notes/struktura` but never to `wiki/struktura`.
+ const slashIdx = ref.slug.lastIndexOf('/');
+ const basename = slashIdx === -1 ? ref.slug : ref.slug.slice(slashIdx + 1);
+ let matches = await resolver.resolveBasenameMatches(basename);
+ if (slashIdx !== -1) {
+ matches = matches.filter(m => m === ref.slug || m.endsWith(`/${ref.slug}`));
+ }
if (matches.length === 0) continue;
const idx = content.indexOf(ref.slug);
const context = idx >= 0 ? excerpt(content, idx, 240) : ref.name;
diff --git a/src/core/page-summary.ts b/src/core/page-summary.ts
index c4d7deedc..78e04d41c 100644
--- a/src/core/page-summary.ts
+++ b/src/core/page-summary.ts
@@ -44,6 +44,33 @@ const HAIKU_MAX_TOKENS = 200;
/** Default model when caller doesn't override. Resolves through the gateway. */
const DEFAULT_SYNOPSIS_MODEL = 'anthropic:claude-haiku-4-5-20251001';
+/**
+ * Hard cap on `documentText` length (chars) before send.
+ *
+ * 2026-05-25 fix wave: small local chat models (Gemma 4 E2B, Qwen3 4B) get
+ * dramatically slower on long contexts even with 131K-token windows declared.
+ * A 73K-char page synopsis on Gemma 4 E2B takes 60-120s, exceeding the
+ * worker's default 30s `lockDuration` and tripping `lock-lost` errors.
+ *
+ * Truncate to a budget that fits a small model's effective throughput while
+ * preserving enough document context for the synopsis to be useful. Truncates
+ * the TAIL because the head (title, frontmatter, intro) carries the
+ * document-level anchor the synopsis needs.
+ *
+ * Override per workload via `GBRAIN_SYNOPSIS_DOC_MAX_CHARS`. Default 32768
+ * (~8K tokens at 4 chars/tok) keeps small-model synopsis under ~30s.
+ * Anthropic Haiku is unaffected at this cap; bump higher when running
+ * frontier models if you want richer document anchoring.
+ */
+export const SYNOPSIS_DOC_MAX_CHARS = (() => {
+ const env = process.env.GBRAIN_SYNOPSIS_DOC_MAX_CHARS;
+ if (env && /^\d+$/.test(env)) {
+ const n = parseInt(env, 10);
+ if (n >= 512 && n <= 1_048_576) return n;
+ }
+ return 32768;
+})();
+
/**
* Synopsis prompt version. Folded into corpus_generation so prompt edits
* invalidate prior embeddings via the v0.40.3.0 query_cache.page_generations
@@ -188,11 +215,19 @@ function buildUserPrompt(
documentText: string,
chunkText: string,
): string {
+ // Tail-truncate `documentText` to `SYNOPSIS_DOC_MAX_CHARS` so small local
+ // chat models don't stall on >100KB pages. Head preserved (title block,
+ // frontmatter, intro paragraphs carry the document-level anchor).
+ let trimmedDoc = documentText;
+ if (documentText.length > SYNOPSIS_DOC_MAX_CHARS) {
+ trimmedDoc = documentText.slice(0, SYNOPSIS_DOC_MAX_CHARS) +
+ `\n\n[... ${documentText.length - SYNOPSIS_DOC_MAX_CHARS} chars truncated for synopsis budget ...]`;
+ }
return [
`${pageTitle}`,
'',
'',
- documentText,
+ trimmedDoc,
'',
'',
'',
diff --git a/src/core/pglite-engine.ts b/src/core/pglite-engine.ts
index e3dfc98d7..257090a23 100644
--- a/src/core/pglite-engine.ts
+++ b/src/core/pglite-engine.ts
@@ -2324,6 +2324,10 @@ export class PGLiteEngine implements BrainEngine {
// v0.40.3.0 D24 NULL→non-NULL race fix mirrors postgres-engine.ts. Two writers
// racing on the same chunk previously raced last-write-wins; the fix lets the
// fresher `embedded_at` win in the text-unchanged branch.
+ //
+ // Code-chunk metadata columns follow the same chunk_text-gated CASE pattern as `embedding`
+ // (#769). Re-chunk trusts EXCLUDED outright; pure re-embed COALESCEs so a caller carrying
+ // only embedding-shaped fields doesn't clobber metadata to NULL.
await this.db.query(
`INSERT INTO content_chunks ${cols} VALUES ${rowParts.join(', ')}
ON CONFLICT (page_id, chunk_index) DO UPDATE SET
@@ -2347,14 +2351,14 @@ export class PGLiteEngine implements BrainEngine {
THEN EXCLUDED.embedded_at
ELSE content_chunks.embedded_at
END,
- language = EXCLUDED.language,
- symbol_name = EXCLUDED.symbol_name,
- symbol_type = EXCLUDED.symbol_type,
- start_line = EXCLUDED.start_line,
- end_line = EXCLUDED.end_line,
- parent_symbol_path = EXCLUDED.parent_symbol_path,
- doc_comment = EXCLUDED.doc_comment,
- symbol_name_qualified = EXCLUDED.symbol_name_qualified,
+ language = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.language ELSE COALESCE(EXCLUDED.language, content_chunks.language) END,
+ symbol_name = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name ELSE COALESCE(EXCLUDED.symbol_name, content_chunks.symbol_name) END,
+ symbol_type = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_type ELSE COALESCE(EXCLUDED.symbol_type, content_chunks.symbol_type) END,
+ start_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.start_line ELSE COALESCE(EXCLUDED.start_line, content_chunks.start_line) END,
+ end_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.end_line ELSE COALESCE(EXCLUDED.end_line, content_chunks.end_line) END,
+ parent_symbol_path = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.parent_symbol_path ELSE COALESCE(EXCLUDED.parent_symbol_path, content_chunks.parent_symbol_path) END,
+ doc_comment = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.doc_comment ELSE COALESCE(EXCLUDED.doc_comment, content_chunks.doc_comment) END,
+ symbol_name_qualified = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name_qualified ELSE COALESCE(EXCLUDED.symbol_name_qualified, content_chunks.symbol_name_qualified) END,
modality = EXCLUDED.modality,
embedding_image = COALESCE(EXCLUDED.embedding_image, content_chunks.embedding_image)`,
params
diff --git a/src/core/postgres-engine.ts b/src/core/postgres-engine.ts
index dc6c9608c..0e8f658f8 100644
--- a/src/core/postgres-engine.ts
+++ b/src/core/postgres-engine.ts
@@ -2475,6 +2475,13 @@ export class PostgresEngine implements BrainEngine {
// - new is fresher (embedded_at > existing.embedded_at) → take new
// - otherwise → keep existing (slower writer with stale embedding loses)
// Mirrored in pglite-engine.ts; pinned by test/e2e/concurrent-embed-race.test.ts.
+ //
+ // Code-chunk metadata columns (language / symbol_name / symbol_type / line range /
+ // parent_symbol_path / doc_comment / symbol_name_qualified) follow the SAME chunk_text-gated
+ // CASE pattern as `embedding` (#769). Re-chunk (chunk_text changed) trusts EXCLUDED outright;
+ // pure re-embed (chunk_text unchanged) COALESCEs so a caller that only carries embedding
+ // doesn't clobber metadata to NULL. Without this, every embed --stale pass nuked code-def's
+ // primary index for thousands of chunks at once.
await sql.unsafe(
`INSERT INTO content_chunks ${cols} VALUES ${rows.join(', ')}
ON CONFLICT (page_id, chunk_index) DO UPDATE SET
@@ -2498,14 +2505,14 @@ export class PostgresEngine implements BrainEngine {
THEN EXCLUDED.embedded_at
ELSE content_chunks.embedded_at
END,
- language = EXCLUDED.language,
- symbol_name = EXCLUDED.symbol_name,
- symbol_type = EXCLUDED.symbol_type,
- start_line = EXCLUDED.start_line,
- end_line = EXCLUDED.end_line,
- parent_symbol_path = EXCLUDED.parent_symbol_path,
- doc_comment = EXCLUDED.doc_comment,
- symbol_name_qualified = EXCLUDED.symbol_name_qualified,
+ language = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.language ELSE COALESCE(EXCLUDED.language, content_chunks.language) END,
+ symbol_name = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name ELSE COALESCE(EXCLUDED.symbol_name, content_chunks.symbol_name) END,
+ symbol_type = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_type ELSE COALESCE(EXCLUDED.symbol_type, content_chunks.symbol_type) END,
+ start_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.start_line ELSE COALESCE(EXCLUDED.start_line, content_chunks.start_line) END,
+ end_line = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.end_line ELSE COALESCE(EXCLUDED.end_line, content_chunks.end_line) END,
+ parent_symbol_path = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.parent_symbol_path ELSE COALESCE(EXCLUDED.parent_symbol_path, content_chunks.parent_symbol_path) END,
+ doc_comment = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.doc_comment ELSE COALESCE(EXCLUDED.doc_comment, content_chunks.doc_comment) END,
+ symbol_name_qualified = CASE WHEN EXCLUDED.chunk_text != content_chunks.chunk_text THEN EXCLUDED.symbol_name_qualified ELSE COALESCE(EXCLUDED.symbol_name_qualified, content_chunks.symbol_name_qualified) END,
modality = EXCLUDED.modality,
embedding_image = COALESCE(EXCLUDED.embedding_image, content_chunks.embedding_image)`,
params as Parameters[1],
diff --git a/test/e2e/global-basename-pglite.test.ts b/test/e2e/global-basename-pglite.test.ts
index d3c767b46..3c5af045b 100644
--- a/test/e2e/global-basename-pglite.test.ts
+++ b/test/e2e/global-basename-pglite.test.ts
@@ -172,6 +172,59 @@ describe('issue #972 — DB-source (gbrain extract links --source db)', () => {
expect(strk!.link_type).toBe('wikilink_basename');
});
+ test('flag ON → path-qualified wikilink outside DIR_PATTERN resolves via DB path', async () => {
+ // `[[notes/struktura]]` — `notes` is not in DIR_PATTERN, so the ref
+ // reaches the generic pass with its dirname intact. Regression: the DB
+ // path queried the basename index with the raw literal (which is keyed
+ // by final segments only), so path-qualified wikilinks outside
+ // DIR_PATTERN silently produced zero edges while the FS path resolved
+ // the identical content.
+ await engine.putPage('notes/struktura', {
+ type: 'concept' as any, title: 'Struktura Notes',
+ compiled_truth: '', timeline: '',
+ });
+ await engine.putPage('concepts/knowledge-graph', {
+ type: 'concept', title: 'Knowledge Graph',
+ compiled_truth: 'Background in [[notes/struktura]].', timeline: '',
+ });
+ await engine.setConfig('link_resolution.global_basename', 'true');
+
+ await runExtract(engine, ['links', '--source', 'db']);
+
+ const outLinks = await engine.getLinks('concepts/knowledge-graph');
+ const strk = outLinks.find(l => l.to_slug === 'notes/struktura');
+ expect(strk).toBeDefined();
+ expect(strk!.link_type).toBe('wikilink_basename');
+ expect(strk!.link_source).toBe('wikilink-resolved');
+ });
+
+ test('path-qualified wikilink never attaches to a basename-only sibling', async () => {
+ // Both notes/struktura and wiki/struktura exist. The author wrote
+ // `[[notes/struktura]]` — the written path must exclude wiki/struktura
+ // (a bare `[[struktura]]` would legitimately match both).
+ await engine.putPage('notes/struktura', {
+ type: 'concept' as any, title: 'Struktura Notes',
+ compiled_truth: '', timeline: '',
+ });
+ await engine.putPage('wiki/struktura', {
+ type: 'concept' as any, title: 'Struktura Wiki',
+ compiled_truth: '', timeline: '',
+ });
+ await engine.putPage('concepts/x', {
+ type: 'concept', title: 'X',
+ compiled_truth: 'See [[notes/struktura]].', timeline: '',
+ });
+ await engine.setConfig('link_resolution.global_basename', 'true');
+
+ await runExtract(engine, ['links', '--source', 'db']);
+
+ const outLinks = await engine.getLinks('concepts/x');
+ const basenameLinks = outLinks
+ .filter(l => l.link_type === 'wikilink_basename')
+ .map(l => l.to_slug);
+ expect(basenameLinks).toEqual(['notes/struktura']);
+ });
+
test('flag OFF → no basename edges via DB path (back-compat)', async () => {
await engine.putPage('projects/struktura', {
type: 'project', title: 'Struktura',
diff --git a/test/embed.serial.test.ts b/test/embed.serial.test.ts
index b4db83beb..a0bf89bea 100644
--- a/test/embed.serial.test.ts
+++ b/test/embed.serial.test.ts
@@ -803,3 +803,107 @@ describe('embedAllStale --source threading (D7)', () => {
expect((firstCallOpts as { sourceId?: string }).sourceId).toBe('media-corpus');
});
});
+
+// ────────────────────────────────────────────────────────────────
+// Code metadata preservation across re-embed (regression for #769)
+// ────────────────────────────────────────────────────────────────
+//
+// gbrain v0.30.1 and earlier silently clobbered code-chunk metadata
+// (language, symbol_name, symbol_type, start_line, end_line,
+// parent_symbol_path, doc_comment, symbol_name_qualified) on every
+// re-embed pass. The chunker populated those columns at import time,
+// but embed.ts loaded chunks via getChunks then mapped them to a
+// stripped ChunkInput carrying only 5 fields. upsertChunks then
+// OVERWROTE (not COALESCEd) the metadata columns from EXCLUDED, so
+// re-embed wiped them to NULL. End result on a real brain: 4875 code
+// pages, 47866 chunks, all with NULL language/symbol_name/symbol_type;
+// code-def returned 0 hits across every indexed repo.
+//
+// All three runEmbed paths (--stale autopilot, --all, --slugs) must
+// thread metadata through the re-upsert. Tests below assert that the
+// engine.upsertChunks call carries the same metadata it loaded.
+
+describe('runEmbed preserves code-chunk metadata across re-embed (regression for #769)', () => {
+ const fullCodeChunk = {
+ chunk_index: 0,
+ chunk_text: '[Java] foo/Bar.java:10-20 method baz',
+ chunk_source: 'compiled_truth' as const,
+ embedded_at: null,
+ token_count: 12,
+ language: 'java',
+ symbol_name: 'baz',
+ symbol_type: 'function',
+ start_line: 10,
+ end_line: 20,
+ parent_symbol_path: ['Bar'],
+ doc_comment: 'does the thing',
+ symbol_name_qualified: 'Bar.baz',
+ };
+
+ function metadataOf(chunk: any) {
+ return {
+ language: chunk.language,
+ symbol_name: chunk.symbol_name,
+ symbol_type: chunk.symbol_type,
+ start_line: chunk.start_line,
+ end_line: chunk.end_line,
+ parent_symbol_path: chunk.parent_symbol_path,
+ doc_comment: chunk.doc_comment,
+ symbol_name_qualified: chunk.symbol_name_qualified,
+ };
+ }
+
+ test('--stale (autopilot path) carries code metadata into upsertChunks', async () => {
+ const stale = [{
+ slug: 'code-page',
+ chunk_index: 0,
+ chunk_text: fullCodeChunk.chunk_text,
+ chunk_source: 'compiled_truth',
+ model: null,
+ token_count: 12,
+ }];
+ let upsertChunkArgs: any[] | null = null;
+ const engine = mockEngine({
+ countStaleChunks: async () => 1,
+ listStaleChunks: async () => stale,
+ getChunks: async () => [fullCodeChunk],
+ upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; },
+ });
+
+ await runEmbed(engine, ['--stale']);
+
+ expect(upsertChunkArgs).not.toBeNull();
+ expect(upsertChunkArgs!).toHaveLength(1);
+ expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk));
+ });
+
+ test('--all (full re-embed) carries code metadata into upsertChunks', async () => {
+ let upsertChunkArgs: any[] | null = null;
+ const engine = mockEngine({
+ listPages: async () => [{ slug: 'code-page' }],
+ getChunks: async () => [fullCodeChunk],
+ upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; },
+ });
+
+ await runEmbed(engine, ['--all']);
+
+ expect(upsertChunkArgs).not.toBeNull();
+ expect(upsertChunkArgs!).toHaveLength(1);
+ expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk));
+ });
+
+ test('--slugs (per-page embed) carries code metadata into upsertChunks', async () => {
+ let upsertChunkArgs: any[] | null = null;
+ const engine = mockEngine({
+ getPage: async () => ({ slug: 'code-page', compiled_truth: 'x', timeline: '' }),
+ getChunks: async () => [fullCodeChunk],
+ upsertChunks: async (_slug: string, chunks: any[]) => { upsertChunkArgs = chunks; },
+ });
+
+ await runEmbed(engine, ['--slugs', 'code-page']);
+
+ expect(upsertChunkArgs).not.toBeNull();
+ expect(upsertChunkArgs!).toHaveLength(1);
+ expect(metadataOf(upsertChunkArgs![0])).toEqual(metadataOf(fullCodeChunk));
+ });
+});
diff --git a/test/link-extraction.test.ts b/test/link-extraction.test.ts
index e0a741220..f7c003ab5 100644
--- a/test/link-extraction.test.ts
+++ b/test/link-extraction.test.ts
@@ -403,6 +403,77 @@ describe('extractPageLinks', () => {
expect(candidates).toEqual([]);
});
+ test('path-qualified wikilink outside DIR_PATTERN queries by final segment', async () => {
+ // `[[notes/struktura]]` (dir not in DIR_PATTERN) falls to the generic
+ // pass. The resolver's basename index is keyed by final path segments,
+ // so the lookup must strip the dirname — mirroring the FS path
+ // (resolveSlugAll). Regression: the raw literal was passed through,
+ // which never matched, so these links silently dropped.
+ const seen: string[] = [];
+ const resolver: SlugResolver = {
+ resolve: async () => null,
+ resolveBasenameMatches: async (name) => {
+ seen.push(name);
+ return name === 'struktura' ? ['notes/struktura'] : [];
+ },
+ };
+ const { candidates } = await extractPageLinks(
+ 'concepts/x', 'See [[notes/struktura]].',
+ {}, 'concept', resolver, { globalBasename: true },
+ );
+ expect(seen).toContain('struktura');
+ expect(seen).not.toContain('notes/struktura');
+ expect(candidates.map(c => c.targetSlug)).toEqual(['notes/struktura']);
+ expect(candidates[0].linkType).toBe('wikilink_basename');
+ expect(candidates[0].linkSource).toBe('wikilink-resolved');
+ });
+
+ test('path-qualified wikilink keeps only matches ending with the written path', async () => {
+ // The written path disambiguates: `[[notes/struktura]]` must never
+ // attach to `wiki/struktura` even though both share the basename.
+ const resolver: SlugResolver = {
+ resolve: async () => null,
+ resolveBasenameMatches: async (name) =>
+ name === 'struktura' ? ['notes/struktura', 'wiki/struktura'] : [],
+ };
+ const { candidates } = await extractPageLinks(
+ 'concepts/x', 'See [[notes/struktura]].',
+ {}, 'concept', resolver, { globalBasename: true },
+ );
+ expect(candidates.map(c => c.targetSlug)).toEqual(['notes/struktura']);
+ });
+
+ test('path-qualified wikilink matches a deeper real slug by path suffix', async () => {
+ // The page lives at vault/notes/struktura; the author wrote the shorter
+ // tail `[[notes/struktura]]`. Suffix matching connects them, while the
+ // basename-only sibling `wiki/struktura` stays excluded.
+ const resolver: SlugResolver = {
+ resolve: async () => null,
+ resolveBasenameMatches: async (name) =>
+ name === 'struktura' ? ['vault/notes/struktura', 'wiki/struktura'] : [],
+ };
+ const { candidates } = await extractPageLinks(
+ 'concepts/x', 'See [[notes/struktura]].',
+ {}, 'concept', resolver, { globalBasename: true },
+ );
+ expect(candidates.map(c => c.targetSlug)).toEqual(['vault/notes/struktura']);
+ });
+
+ test('path-qualified self-link is dropped like the bare form', async () => {
+ // `[[notes/struktura]]` written on notes/struktura itself must not
+ // produce a self-loop (same guard as the bare `[[own-tail]]` case).
+ const resolver: SlugResolver = {
+ resolve: async () => null,
+ resolveBasenameMatches: async (name) =>
+ name === 'struktura' ? ['notes/struktura'] : [],
+ };
+ const { candidates } = await extractPageLinks(
+ 'notes/struktura', 'See [[notes/struktura]].',
+ {}, 'concept', resolver, { globalBasename: true },
+ );
+ expect(candidates).toEqual([]);
+ });
+
test('bare wikilink resolution does not interfere with DIR_PATTERN wikilinks', async () => {
// 2b refs (people/alice) take the verb-inferred type;
// 2c refs (struktura) take wikilink_basename. Same call.