- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
80 lines
2.6 KiB
Python
80 lines
2.6 KiB
Python
"""Markdown content parser and excerpt extraction utilities."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
|
|
|
|
def strip_markdown(markdown_text: str) -> str:
|
|
"""Remove markdown syntax markers (headers, bold, italics, links, code blocks) to obtain plain text."""
|
|
if not markdown_text:
|
|
return ""
|
|
|
|
text = markdown_text
|
|
|
|
# Remove code blocks
|
|
text = re.sub(r"```[\s\S]*?```", " ", text)
|
|
text = re.sub(r"`[^`]*`", " ", text)
|
|
|
|
# Remove headers (# Header)
|
|
text = re.sub(r"^#+\s+", " ", text, flags=re.MULTILINE)
|
|
|
|
# Replace markdown links [anchor](url) with just anchor
|
|
text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text)
|
|
|
|
# Remove image links 
|
|
text = re.sub(r"!\[[^\]]*\]\([^)]+\)", " ", text)
|
|
|
|
# Remove bold/italics (*, _, **, __)
|
|
text = re.sub(r"(\*\*|__)(.*?)\1", r"\2", text)
|
|
text = re.sub(r"(\*|_)(.*?)\1", r"\2", text)
|
|
|
|
# Remove blockquotes and list markers
|
|
text = re.sub(r"^\s*[-*+]\s+", " ", text, flags=re.MULTILINE)
|
|
text = re.sub(r"^\s*\d+\.\s+", " ", text, flags=re.MULTILINE)
|
|
text = re.sub(r"^\s*>\s*", " ", text, flags=re.MULTILINE)
|
|
|
|
# Normalize whitespace
|
|
text = re.sub(r"\s+", " ", text).strip()
|
|
return text
|
|
|
|
|
|
def extract_sentences(text: str) -> list[str]:
|
|
"""Split text into individual sentences."""
|
|
# Split by period, exclamation, question mark followed by space or newline
|
|
raw_sentences = re.split(r"(?<=[.!?])\s+", text.strip())
|
|
sentences = [s.strip() for s in raw_sentences if len(s.strip()) > 3]
|
|
return sentences
|
|
|
|
|
|
def extract_evidence_snippets(
|
|
markdown_text: str, match_terms: list[str], max_snippets: int = 3
|
|
) -> list[str]:
|
|
"""
|
|
Extract relevant sentence excerpts from Markdown text that contain any of the given match terms.
|
|
Preserves original phrasing and formats as clean evidence.
|
|
"""
|
|
if not markdown_text or not match_terms:
|
|
return []
|
|
|
|
plain_text = strip_markdown(markdown_text)
|
|
sentences = extract_sentences(plain_text)
|
|
if not sentences:
|
|
sentences = [plain_text]
|
|
|
|
lower_terms = [t.lower() for t in match_terms if t]
|
|
evidence: list[str] = []
|
|
|
|
for sentence in sentences:
|
|
lower_sent = sentence.lower()
|
|
for term in lower_terms:
|
|
if re.search(r"\b" + re.escape(term) + r"\b", lower_sent) or term in lower_sent:
|
|
clean_snippet = sentence.strip()
|
|
if clean_snippet and clean_snippet not in evidence:
|
|
evidence.append(clean_snippet)
|
|
if len(evidence) >= max_snippets:
|
|
return evidence
|
|
break
|
|
|
|
return evidence
|