feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
+5
-4
@@ -3,7 +3,6 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import List, Tuple
|
||||
|
||||
|
||||
def strip_markdown(markdown_text: str) -> str:
|
||||
@@ -40,7 +39,7 @@ def strip_markdown(markdown_text: str) -> str:
|
||||
return text
|
||||
|
||||
|
||||
def extract_sentences(text: str) -> List[str]:
|
||||
def extract_sentences(text: str) -> list[str]:
|
||||
"""Split text into individual sentences."""
|
||||
# Split by period, exclamation, question mark followed by space or newline
|
||||
raw_sentences = re.split(r"(?<=[.!?])\s+", text.strip())
|
||||
@@ -48,7 +47,9 @@ def extract_sentences(text: str) -> List[str]:
|
||||
return sentences
|
||||
|
||||
|
||||
def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_snippets: int = 3) -> List[str]:
|
||||
def extract_evidence_snippets(
|
||||
markdown_text: str, match_terms: list[str], max_snippets: int = 3
|
||||
) -> list[str]:
|
||||
"""
|
||||
Extract relevant sentence excerpts from Markdown text that contain any of the given match terms.
|
||||
Preserves original phrasing and formats as clean evidence.
|
||||
@@ -62,7 +63,7 @@ def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_sn
|
||||
sentences = [plain_text]
|
||||
|
||||
lower_terms = [t.lower() for t in match_terms if t]
|
||||
evidence: List[str] = []
|
||||
evidence: list[str] = []
|
||||
|
||||
for sentence in sentences:
|
||||
lower_sent = sentence.lower()
|
||||
|
||||
Reference in New Issue
Block a user