feat(extractor): implement multi-engine article content extractor

- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability)
- Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing)
- Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts
- Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
2026-08-20 19:22:20 -03:00
parent 6e3d57619b
commit 6a45368cb0
85 changed files with 18345 additions and 3897 deletions
+5 -4
View File
@@ -3,7 +3,6 @@
from __future__ import annotations
import re
from typing import List, Tuple
def strip_markdown(markdown_text: str) -> str:
@@ -40,7 +39,7 @@ def strip_markdown(markdown_text: str) -> str:
return text
def extract_sentences(text: str) -> List[str]:
def extract_sentences(text: str) -> list[str]:
"""Split text into individual sentences."""
# Split by period, exclamation, question mark followed by space or newline
raw_sentences = re.split(r"(?<=[.!?])\s+", text.strip())
@@ -48,7 +47,9 @@ def extract_sentences(text: str) -> List[str]:
return sentences
def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_snippets: int = 3) -> List[str]:
def extract_evidence_snippets(
markdown_text: str, match_terms: list[str], max_snippets: int = 3
) -> list[str]:
"""
Extract relevant sentence excerpts from Markdown text that contain any of the given match terms.
Preserves original phrasing and formats as clean evidence.
@@ -62,7 +63,7 @@ def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_sn
sentences = [plain_text]
lower_terms = [t.lower() for t in match_terms if t]
evidence: List[str] = []
evidence: list[str] = []
for sentence in sentences:
lower_sent = sentence.lower()