"""Markdown content parser and excerpt extraction utilities.""" from __future__ import annotations import re def strip_markdown(markdown_text: str) -> str: """Remove markdown syntax markers (headers, bold, italics, links, code blocks) to obtain plain text.""" if not markdown_text: return "" text = markdown_text # Remove code blocks text = re.sub(r"```[\s\S]*?```", " ", text) text = re.sub(r"`[^`]*`", " ", text) # Remove headers (# Header) text = re.sub(r"^#+\s+", " ", text, flags=re.MULTILINE) # Replace markdown links [anchor](url) with just anchor text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text) # Remove image links ![alt](url) text = re.sub(r"!\[[^\]]*\]\([^)]+\)", " ", text) # Remove bold/italics (*, _, **, __) text = re.sub(r"(\*\*|__)(.*?)\1", r"\2", text) text = re.sub(r"(\*|_)(.*?)\1", r"\2", text) # Remove blockquotes and list markers text = re.sub(r"^\s*[-*+]\s+", " ", text, flags=re.MULTILINE) text = re.sub(r"^\s*\d+\.\s+", " ", text, flags=re.MULTILINE) text = re.sub(r"^\s*>\s*", " ", text, flags=re.MULTILINE) # Normalize whitespace text = re.sub(r"\s+", " ", text).strip() return text def extract_sentences(text: str) -> list[str]: """Split text into individual sentences.""" # Split by period, exclamation, question mark followed by space or newline raw_sentences = re.split(r"(?<=[.!?])\s+", text.strip()) sentences = [s.strip() for s in raw_sentences if len(s.strip()) > 3] return sentences def extract_evidence_snippets( markdown_text: str, match_terms: list[str], max_snippets: int = 3 ) -> list[str]: """ Extract relevant sentence excerpts from Markdown text that contain any of the given match terms. Preserves original phrasing and formats as clean evidence. """ if not markdown_text or not match_terms: return [] plain_text = strip_markdown(markdown_text) sentences = extract_sentences(plain_text) if not sentences: sentences = [plain_text] lower_terms = [t.lower() for t in match_terms if t] evidence: list[str] = [] for sentence in sentences: lower_sent = sentence.lower() for term in lower_terms: if re.search(r"\b" + re.escape(term) + r"\b", lower_sent) or term in lower_sent: clean_snippet = sentence.strip() if clean_snippet and clean_snippet not in evidence: evidence.append(clean_snippet) if len(evidence) >= max_snippets: return evidence break return evidence