Files
TextNLPClassifierApp/src/parser.py
T
andreferraro 6a45368cb0 feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability)
- Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing)
- Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts
- Passed ruff linting/formatting and mypy type checking cleanly
2026-08-20 19:22:20 -03:00

80 lines
2.6 KiB
Python

"""Markdown content parser and excerpt extraction utilities."""
from __future__ import annotations
import re
def strip_markdown(markdown_text: str) -> str:
"""Remove markdown syntax markers (headers, bold, italics, links, code blocks) to obtain plain text."""
if not markdown_text:
return ""
text = markdown_text
# Remove code blocks
text = re.sub(r"```[\s\S]*?```", " ", text)
text = re.sub(r"`[^`]*`", " ", text)
# Remove headers (# Header)
text = re.sub(r"^#+\s+", " ", text, flags=re.MULTILINE)
# Replace markdown links [anchor](url) with just anchor
text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text)
# Remove image links ![alt](url)
text = re.sub(r"!\[[^\]]*\]\([^)]+\)", " ", text)
# Remove bold/italics (*, _, **, __)
text = re.sub(r"(\*\*|__)(.*?)\1", r"\2", text)
text = re.sub(r"(\*|_)(.*?)\1", r"\2", text)
# Remove blockquotes and list markers
text = re.sub(r"^\s*[-*+]\s+", " ", text, flags=re.MULTILINE)
text = re.sub(r"^\s*\d+\.\s+", " ", text, flags=re.MULTILINE)
text = re.sub(r"^\s*>\s*", " ", text, flags=re.MULTILINE)
# Normalize whitespace
text = re.sub(r"\s+", " ", text).strip()
return text
def extract_sentences(text: str) -> list[str]:
"""Split text into individual sentences."""
# Split by period, exclamation, question mark followed by space or newline
raw_sentences = re.split(r"(?<=[.!?])\s+", text.strip())
sentences = [s.strip() for s in raw_sentences if len(s.strip()) > 3]
return sentences
def extract_evidence_snippets(
markdown_text: str, match_terms: list[str], max_snippets: int = 3
) -> list[str]:
"""
Extract relevant sentence excerpts from Markdown text that contain any of the given match terms.
Preserves original phrasing and formats as clean evidence.
"""
if not markdown_text or not match_terms:
return []
plain_text = strip_markdown(markdown_text)
sentences = extract_sentences(plain_text)
if not sentences:
sentences = [plain_text]
lower_terms = [t.lower() for t in match_terms if t]
evidence: list[str] = []
for sentence in sentences:
lower_sent = sentence.lower()
for term in lower_terms:
if re.search(r"\b" + re.escape(term) + r"\b", lower_sent) or term in lower_sent:
clean_snippet = sentence.strip()
if clean_snippet and clean_snippet not in evidence:
evidence.append(clean_snippet)
if len(evidence) >= max_snippets:
return evidence
break
return evidence