feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
+1
-2
@@ -4,13 +4,12 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from src import __version__
|
||||
from src.models import ECPSnapshot, ClassificationError, ErrorCode
|
||||
from src.classifier import InherenceClassifier
|
||||
from src.models import ClassificationError, ECPSnapshot, ErrorCode
|
||||
|
||||
# Ensure UTF-8 output streams across all platforms
|
||||
if hasattr(sys.stdout, "reconfigure"):
|
||||
|
||||
Reference in New Issue
Block a user