/** * Markdown text extraction → canonical text + gap-free offset map. * * Implements ADR-0003 for the `markdown-rendered` representation. Markup is * stripped to plain text before `normalize()` is applied. */ import { normalize, type OffsetMap } from "@citation-evidence/engine/shared"; import { buildWholeTextOffsetMap } from "../shared/offset-map"; const FENCED_CODE = /```[\s\S]*?```/g; const INLINE_CODE = /`([^`]+)`/g; const IMAGE = /!\[([^\]]*)\]\([^)]+\)/g; const LINK = /\[([^\]]+)\]\([^)]+\)/g; const HEADING = /^#{1,6}\s+/gm; const BLOCKQUOTE = /^>\s?/gm; const LIST_MARKER = /^[\s]*[-*+]\s+/gm; const ORDERED_LIST = /^[\s]*\d+\.\s+/gm; const EMPHASIS = /(\*\*|__)(.*?)\1/g; const ITALIC = /(\*|_)([^*_]+)\1/g; const STRIKETHROUGH = /~~(.*?)~~/g; const HORIZONTAL_RULE = /^[-*_]{3,}\s*$/gm; export interface MarkdownExtractionResult { readonly canonicalText: string; readonly offsetMap: OffsetMap; } export function extractMarkdown(source: string): MarkdownExtractionResult { const rawText = markdownToPlainText(source); const canonicalText = normalize(rawText).text; return { canonicalText, offsetMap: buildWholeTextOffsetMap(canonicalText), }; } function markdownToPlainText(source: string): string { let text = source; text = text.replace(FENCED_CODE, "\n\n"); text = text.replace(INLINE_CODE, "$1"); text = text.replace(IMAGE, "$1"); text = text.replace(LINK, "$1"); text = text.replace(HEADING, ""); text = text.replace(BLOCKQUOTE, ""); text = text.replace(LIST_MARKER, ""); text = text.replace(ORDERED_LIST, ""); text = text.replace(EMPHASIS, "$2"); text = text.replace(ITALIC, "$2"); text = text.replace(STRIKETHROUGH, "$1"); text = text.replace(HORIZONTAL_RULE, "\n\n"); return text; }