diff --git a/src/commands/conversation-parser.ts b/src/commands/conversation-parser.ts index 7ae262bbe..ea0186ef0 100644 --- a/src/commands/conversation-parser.ts +++ b/src/commands/conversation-parser.ts @@ -16,6 +16,7 @@ import { readFileSync, existsSync } from 'node:fs'; import { BUILTIN_PATTERNS } from '../core/conversation-parser/builtins.ts'; +import { readConversationBodyForParsing } from '../core/conversation-parser/body.ts'; import { parseConversation } from '../core/conversation-parser/parse.ts'; import type { BrainEngine } from '../core/engine.ts'; @@ -170,8 +171,7 @@ async function runScan( process.exit(2); } - // Concatenate compiled_truth + timeline (matches the real parser's body shape). - const body = `${page.compiled_truth ?? ''}\n${page.timeline ?? ''}`.trim(); + const body = await readConversationBodyForParsing(engine, page); const result = parseConversation(body, { page, diagnostic: true }); diff --git a/src/commands/doctor.ts b/src/commands/doctor.ts index 9b0dc3a4b..134646dc5 100644 --- a/src/commands/doctor.ts +++ b/src/commands/doctor.ts @@ -4861,6 +4861,7 @@ export async function buildChecks( // triage the misses interactively. if (engine) { try { + const { readConversationBodyForParsing } = await import('../core/conversation-parser/body.ts'); const { parseConversation } = await import('../core/conversation-parser/parse.ts'); const allowedTypes = ['conversation', 'meeting', 'slack', 'email'] as const; // PageFilters supports singular `type` only; iterate the 4 types @@ -4880,7 +4881,7 @@ export async function buildChecks( const hitsByPattern: Record = {}; let unmatched = 0; for (const page of sample) { - const body = `${page.compiled_truth ?? ''}\n${page.timeline ?? ''}`.trim(); + const body = await readConversationBodyForParsing(engine, page); const result = parseConversation(body, { page, noPolish: true, noFallback: true }); const id = result.matched_pattern_id ?? '_no_match'; hitsByPattern[id] = (hitsByPattern[id] ?? 0) + 1; diff --git a/src/commands/extract-conversation-facts.ts b/src/commands/extract-conversation-facts.ts index ea9f2c373..f1d04e4e3 100644 --- a/src/commands/extract-conversation-facts.ts +++ b/src/commands/extract-conversation-facts.ts @@ -35,9 +35,10 @@ * `listPages({type, sourceId, limit: PAGE_LIST_BATCH})` so worst * case is BATCH × 25MB per batch (currently 10 × 25MB = 250MB * bounded). Per-page body cap drops oversize before parsing. - * - Body read covers compiled_truth + timeline. parseMarkdown splits - * conversation imports across both columns; reading only - * compiled_truth silently drops half on iMessage/Slack imports. + * - Body read prefers frontmatter.raw_transcript when present, then + * falls back to compiled_truth + timeline. Meeting pages often + * store the real turn-by-turn transcript in a sidecar file while + * compiled_truth is just the human summary. * - Page-global row_num accumulator. facts table unique index is * (source_id, source_markdown_slug, row_num); per-segment row_num * would collide on segment 2. Per-page counter increments across @@ -286,6 +287,7 @@ import { parseConversation, type ParseConversationOpts as OrchestratorParseOpts, } from '../core/conversation-parser/parse.ts'; +import { readConversationBodyForParsing } from '../core/conversation-parser/body.ts'; /** * v0.41.13.0 — back-compat shape for direct callers + the existing @@ -474,16 +476,6 @@ function pageBodyBytes(page: Page): number { return Buffer.byteLength(compiled, 'utf8') + Buffer.byteLength(timeline, 'utf8'); } -function readPageBody(page: Page): string { - // F1: read BOTH compiled_truth AND timeline; iMessage importers - // place chronological message stream in timeline. - const compiled = page.compiled_truth ?? ''; - const timeline = page.timeline ?? ''; - if (!compiled) return timeline; - if (!timeline) return compiled; - return `${compiled}\n\n${timeline}`; -} - // --------------------------------------------------------------------------- // Types config resolver (Eng-v2 A2 — unified single source of truth). // --------------------------------------------------------------------------- @@ -682,7 +674,7 @@ async function processPage( return { newEndIso: null }; } - const body = readPageBody(page); + const body = await readConversationBodyForParsing(state.engine, page); // v0.41.13.0: thread the full Page through the orchestrator so D8 // date-derivation chain (frontmatter.date > effective_date > // '1970-01-01') AND timezone_policy warnings apply. The historical diff --git a/src/core/conversation-parser/body.ts b/src/core/conversation-parser/body.ts new file mode 100644 index 000000000..fd24345d8 --- /dev/null +++ b/src/core/conversation-parser/body.ts @@ -0,0 +1,39 @@ +import { existsSync, readFileSync } from 'node:fs'; +import { isAbsolute, join } from 'node:path'; +import type { BrainEngine } from '../engine.ts'; +import type { Page } from '../types.ts'; + +export function readSummaryBody(page: Page): string { + const compiled = page.compiled_truth ?? ''; + const timeline = page.timeline ?? ''; + if (!compiled) return timeline; + if (!timeline) return compiled; + return `${compiled}\n\n${timeline}`; +} + +function extractRawTranscriptPath(page: Page): string | null { + const raw = page.frontmatter?.raw_transcript; + if (typeof raw !== 'string') return null; + const trimmed = raw.trim(); + return trimmed.length > 0 ? trimmed : null; +} + +export async function readConversationBodyForParsing( + engine: BrainEngine, + page: Page, +): Promise { + const rawTranscript = extractRawTranscriptPath(page); + if (rawTranscript) { + const repoPath = await engine.getConfig('sync.repo_path'); + const resolved = isAbsolute(rawTranscript) + ? rawTranscript + : repoPath + ? join(repoPath, rawTranscript) + : null; + if (resolved && existsSync(resolved)) { + const rawBody = readFileSync(resolved, 'utf8').trim(); + if (rawBody.length > 0) return rawBody; + } + } + return readSummaryBody(page); +} diff --git a/src/core/conversation-parser/builtins.ts b/src/core/conversation-parser/builtins.ts index b28ffba27..07f1cd99f 100644 --- a/src/core/conversation-parser/builtins.ts +++ b/src/core/conversation-parser/builtins.ts @@ -178,6 +178,39 @@ export const BUILTIN_PATTERNS: readonly PatternEntry[] = [ 'OpenClaw meeting-ingestion pipeline reformat of Circleback transcripts (see your OpenClaw skills/meeting-ingestion/SKILL.md)', }, + { + // Fathom/phone-call raw transcripts in this workspace use a plain + // `Speaker A: ...` / `Speaker B: ...` shape with no per-line time. + // Narrow on the literal `Speaker ` prefix so we don't accidentally + // parse ordinary prose labels (`Owner:`, `Decision:`) as chat. + id: 'speaker-letter-no-time', + origin: 'builtin', + regex: /^(Speaker [A-Z0-9]+):\s*(.*)$/, + captures: { + speaker_group: 1, + text_group: 2, + }, + date_source: 'frontmatter', + time_format: '24h', + timezone_policy: 'utc_assumed_with_warn', + multi_line: false, + quick_reject: /^Speaker /, + score_full_body: true, + test_positive: [ + 'Speaker A: That is exactly the issue.', + 'Speaker B: Yeah, I know.', + 'Speaker Z9: Let me ask him.', + ], + test_negative: [ + '**Speaker A:** bold no-time shape', + 'Speaker: missing participant suffix', + 'Owner: this is a prose label, not a transcript line', + 'Participant 2: different raw format', + ], + source_doc: + 'Workspace raw transcript sidecar shape from capture-cli / phone-call transcripts: `Speaker A: ...`', + }, + { // Modern meeting-transcription tools (Circleback, Granola, Zoom) // emit `**Speaker Name:** message text` with NO per-line diff --git a/test/conversation-parser/parse.test.ts b/test/conversation-parser/parse.test.ts index 33cbf36fe..34bb7d222 100644 --- a/test/conversation-parser/parse.test.ts +++ b/test/conversation-parser/parse.test.ts @@ -585,6 +585,38 @@ describe('bold-name-no-time pattern (Circleback/Granola/Zoom, no timestamp)', () }); }); +describe('speaker-letter-no-time pattern (raw transcript sidecars)', () => { + test('parses plain Speaker A / Speaker B transcripts', () => { + const body = [ + 'Speaker A: That is exactly the issue.', + 'Speaker B: Yeah, I know.', + 'Speaker A: Let me ask him.', + 'Speaker B: Sounds good.', + ].join('\n'); + const r = parseConversation(body, { fallbackDate: '2026-06-01' }); + expect(r.phase).toBe('regex_match'); + expect(r.matched_pattern_id).toBe('speaker-letter-no-time'); + expect(r.messages).toHaveLength(4); + expect(r.messages[0]).toEqual({ + speaker: 'Speaker A', + timestamp: '2026-06-01T00:00:00Z', + text: 'That is exactly the issue.', + }); + expect(r.messages[1].speaker).toBe('Speaker B'); + expect(r.messages[3].text).toBe('Sounds good.'); + }); + + test('does not parse ordinary prose labels as transcript lines', () => { + const body = [ + 'Owner: Elliot', + 'Decision: Ship the parser fix', + 'Next step: rerun extraction', + ].join('\n'); + const r = parseConversation(body, { fallbackDate: '2026-06-01' }); + expect(r.phase).toBe('no_match'); + }); +}); + // --------------------------------------------------------------------------- // parseConversation — full-body fallback (v0.41.18+ #1533 + Codex P1 #1, #2, #8) // --------------------------------------------------------------------------- diff --git a/test/extract-conversation-facts.test.ts b/test/extract-conversation-facts.test.ts index 387502b67..bcc0e4d1f 100644 --- a/test/extract-conversation-facts.test.ts +++ b/test/extract-conversation-facts.test.ts @@ -12,6 +12,9 @@ */ import { describe, expect, test, beforeAll, afterAll, beforeEach } from 'bun:test'; +import { mkdtempSync, mkdirSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; import { PGLiteEngine } from '../src/core/pglite-engine.ts'; import { __setChatTransportForTests, @@ -246,11 +249,13 @@ const SAMPLE_BODY = [ describe('runExtractConversationFactsCore', () => { let engine: PGLiteEngine; + let repoDir: string; beforeAll(async () => { engine = new PGLiteEngine(); await engine.connect({}); await engine.initSchema(); + repoDir = mkdtempSync(join(tmpdir(), 'gbrain-convo-facts-')); // Deterministic chat-transport stub. Records calls + returns one // fact per turn. Real-LLM extraction quality is the eval suite's job. @@ -293,6 +298,7 @@ describe('runExtractConversationFactsCore', () => { __setEmbedTransportForTests(null); resetGateway(); await engine.disconnect(); + rmSync(repoDir, { recursive: true, force: true }); }); beforeEach(async () => { @@ -303,6 +309,7 @@ describe('runExtractConversationFactsCore', () => { await engine.executeRaw(`DELETE FROM pages WHERE slug LIKE 'conversations/%' OR slug LIKE 'people/alice%'`); // Set facts.extraction_enabled=true so kill-switch doesn't refuse. await engine.setConfig('facts.extraction_enabled', 'true'); + await engine.setConfig('sync.repo_path', repoDir); // Seed test pages. await engine.putPage('conversations/imessage/alice-example', { type: 'conversation', @@ -318,6 +325,31 @@ describe('runExtractConversationFactsCore', () => { timeline: '', frontmatter: {}, }); + const rawDir = join(repoDir, 'meetings/raw-speaker-example.raw'); + mkdirSync(rawDir, { recursive: true }); + writeFileSync( + join(rawDir, 'transcript.txt'), + [ + 'Speaker A: We finally shipped the parser fix.', + 'Speaker B: Good. Now rerun extraction.', + 'Speaker A: I also turned the fallback flag on.', + 'Speaker B: Perfect.', + ].join('\n'), + 'utf8', + ); + await engine.putPage('meetings/raw-speaker-example', { + type: 'meeting', + title: 'Raw speaker transcript example', + compiled_truth: [ + '## Executive Summary', + '- This is a polished meeting note, not the transcript.', + ].join('\n'), + timeline: '', + frontmatter: { + date: '2026-06-01', + raw_transcript: 'meetings/raw-speaker-example.raw/transcript.txt', + }, + }); }); test('dry-run reports segmentation without writing facts', async () => { @@ -356,6 +388,18 @@ describe('runExtractConversationFactsCore', () => { expect(result.pages_skipped).toBe(1); }); + test('meeting page reads raw_transcript sidecar instead of polished summary body', async () => { + const result = await runExtractConversationFactsCore(engine, { + sourceId: 'default', + slug: 'meetings/raw-speaker-example', + dryRun: true, + sleepMs: 0, + }); + expect(result.pages_processed).toBe(1); + expect(result.segments_processed).toBeGreaterThanOrEqual(1); + expect(result.pages_skipped).toBe(0); + }); + test('writes facts with per-segment source_session AND terminal audit row (E16)', async () => { const result = await runExtractConversationFactsCore(engine, { sourceId: 'default',