feat(extractor): implement multi-engine article content extractor

- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability)
- Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing)
- Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts
- Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
2026-08-20 19:22:20 -03:00
parent 6e3d57619b
commit 6a45368cb0
85 changed files with 18345 additions and 3897 deletions
+2 -6
View File
@@ -42,9 +42,7 @@ class SearchQuery:
raise ValueError("A palavra-chave não pode ser vazia.")
if not self.language or len(self.language.strip()) < 2:
raise ValueError(
"O idioma deve conter pelo menos 2 caracteres (ex: 'pt', 'en', 'es')."
)
raise ValueError("O idioma deve conter pelo menos 2 caracteres (ex: 'pt', 'en', 'es').")
if self.max_pages < 1 or self.max_pages > 10:
raise ValueError("O número máximo de páginas deve estar entre 1 e 10.")
@@ -434,9 +432,7 @@ def main(argv: list[str] | None = None) -> int:
verbose = not getattr(args, "silent", False)
try:
result = extract_google_news(
query, resolve_urls=args.resolve_urls, verbose=verbose
)
result = extract_google_news(query, resolve_urls=args.resolve_urls, verbose=verbose)
json_output = json.dumps(
result.to_dict(),
ensure_ascii=False,