feat(extractor): implement multi-engine article content extractor

- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability)
- Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing)
- Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts
- Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
2026-08-20 19:22:20 -03:00
parent 6e3d57619b
commit 6a45368cb0
85 changed files with 18345 additions and 3897 deletions
+770
View File
@@ -0,0 +1,770 @@
#!/usr/bin/env python3
"""
Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability).
Lê listagens JSON de notícias (ex: out/river_plate.json), acessa e renderiza as páginas
em modo stealth headless utilizando Foxcape reutilizando a mesma sessão de navegador,
executa a extração em paralelo/sequência com 3 motores de conteúdo (Trafilatura,
Newspaper4k e Readability) e salva o resultado enriquecido e higienizado em JSON.
"""
from __future__ import annotations
import argparse
import json
import sys
import time
from dataclasses import dataclass, field
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Literal
import trafilatura
from bs4 import BeautifulSoup
from foxcape import Foxcape, FoxcapeConfig
from newspaper import Article
from readability import Document
# ==============================================================================
# Modelos de Dados e Dataclasses
# ==============================================================================
@dataclass(frozen=True)
class InputArticle:
"""Metadados originais da notícia contida no JSON de entrada."""
titulo: str
url: str
subtitulo: str | None = None
quando_publicado: str | None = None
pagina: int = 1
def to_dict(self) -> dict[str, Any]:
return {
"titulo": self.titulo,
"subtitulo": self.subtitulo,
"quando_publicado": self.quando_publicado,
"url": self.url,
"pagina": self.pagina,
}
@dataclass(frozen=True)
class TrafilaturaData:
"""Dados completos extraídos pelo motor Trafilatura."""
title: str | None = None
author: str | None = None
date: str | None = None
description: str | None = None
sitename: str | None = None
hostname: str | None = None
language: str | None = None
categories: list[str] = field(default_factory=list)
tags: list[str] = field(default_factory=list)
canonical_url: str | None = None
image: str | None = None
pagetype: str | None = None
fingerprint: str | None = None
license: str | None = None
comments: str | None = None
text: str = ""
markdown: str | None = None
raw_json: dict[str, Any] | None = None
error: str | None = None
def to_dict(self) -> dict[str, Any]:
return {
"title": self.title,
"author": self.author,
"date": self.date,
"description": self.description,
"sitename": self.sitename,
"hostname": self.hostname,
"language": self.language,
"categories": self.categories,
"tags": self.tags,
"canonical_url": self.canonical_url,
"image": self.image,
"pagetype": self.pagetype,
"fingerprint": self.fingerprint,
"license": self.license,
"comments": self.comments,
"text": self.text,
"markdown": self.markdown,
"raw_json": self.raw_json,
"error": self.error,
}
@dataclass(frozen=True)
class NewspaperData:
"""Dados completos extraídos e enriquecidos com NLP pelo motor Newspaper4k."""
title: str | None = None
authors: list[str] = field(default_factory=list)
publish_date: str | None = None
text: str = ""
summary: str | None = None
keywords: list[str] = field(default_factory=list)
keyword_scores: dict[str, float] = field(default_factory=dict)
top_image: str | None = None
images: list[str] = field(default_factory=list)
movies: list[str] = field(default_factory=list)
tags: list[str] = field(default_factory=list)
canonical_link: str | None = None
article_html: str | None = None
meta_description: str | None = None
meta_keywords: list[str] = field(default_factory=list)
meta_favicon: str | None = None
meta_site_name: str | None = None
meta_lang: str | None = None
meta_data: dict[str, Any] = field(default_factory=dict)
error: str | None = None
def to_dict(self) -> dict[str, Any]:
return {
"title": self.title,
"authors": self.authors,
"publish_date": self.publish_date,
"text": self.text,
"summary": self.summary,
"keywords": self.keywords,
"keyword_scores": self.keyword_scores,
"top_image": self.top_image,
"images": self.images,
"movies": self.movies,
"tags": self.tags,
"canonical_link": self.canonical_link,
"article_html": self.article_html,
"meta_description": self.meta_description,
"meta_keywords": self.meta_keywords,
"meta_favicon": self.meta_favicon,
"meta_site_name": self.meta_site_name,
"meta_lang": self.meta_lang,
"meta_data": self.meta_data,
"error": self.error,
}
@dataclass(frozen=True)
class ReadabilityData:
"""Dados completos higienizados pelo algoritmo Readability."""
title: str | None = None
short_title: str | None = None
author: str | None = None
cleaned_html: str | None = None
cleaned_text: str | None = None
error: str | None = None
def to_dict(self) -> dict[str, Any]:
return {
"title": self.title,
"short_title": self.short_title,
"author": self.author,
"cleaned_html": self.cleaned_html,
"cleaned_text": self.cleaned_text,
"error": self.error,
}
@dataclass(frozen=True)
class ExtractedArticle:
"""Resultado consolidado da extração de um artigo."""
input_meta: InputArticle
extraction_status: Literal["success", "failed"]
error_message: str | None
crawled_url: str
page_title: str | None
http_status: int | None
trafilatura: TrafilaturaData | None = None
newspaper4k: NewspaperData | None = None
readability: ReadabilityData | None = None
def to_dict(self) -> dict[str, Any]:
return {
"input_meta": self.input_meta.to_dict(),
"extraction_status": self.extraction_status,
"error_message": self.error_message,
"crawled_url": self.crawled_url,
"page_title": self.page_title,
"http_status": self.http_status,
"trafilatura": self.trafilatura.to_dict() if self.trafilatura else None,
"newspaper4k": self.newspaper4k.to_dict() if self.newspaper4k else None,
"readability": self.readability.to_dict() if self.readability else None,
}
@dataclass(frozen=True)
class ExtractionBatchReport:
"""Relatório consolidado de saída do processamento de um lote."""
source_file: str
processed_at: str
total_articles: int
successful_articles: int
failed_articles: int
articles: list[ExtractedArticle] = field(default_factory=list)
def to_dict(self) -> dict[str, Any]:
return {
"source_file": self.source_file,
"processed_at": self.processed_at,
"total_articles": self.total_articles,
"successful_articles": self.successful_articles,
"failed_articles": self.failed_articles,
"articles": [a.to_dict() for a in self.articles],
}
# ==============================================================================
# Parsers / Extratores Especializados
# ==============================================================================
class TrafilaturaExtractor:
"""Motor de extração baseado na biblioteca Trafilatura."""
@staticmethod
def extract(html: str, url: str | None = None) -> TrafilaturaData:
try:
# Extração bare document completa
doc = trafilatura.bare_extraction(
html,
url=url,
include_comments=True,
include_tables=True,
include_images=True,
include_links=True,
include_formatting=True,
with_metadata=True,
)
# Extração em markdown
markdown_text = trafilatura.extract(
html,
output_format="markdown",
include_comments=True,
include_tables=True,
include_images=True,
include_links=True,
include_formatting=True,
url=url,
)
# Extração em JSON nativo
json_output_str = trafilatura.extract(
html,
output_format="json",
include_comments=True,
include_tables=True,
include_images=True,
include_links=True,
url=url,
)
raw_json = json.loads(json_output_str) if json_output_str else None
if doc:
if isinstance(doc, dict):
categories = list(doc.get("categories", [])) if doc.get("categories") else []
tags = list(doc.get("tags", [])) if doc.get("tags") else []
return TrafilaturaData(
title=doc.get("title"),
author=doc.get("author"),
date=doc.get("date"),
description=doc.get("description"),
sitename=doc.get("sitename"),
hostname=doc.get("hostname"),
language=doc.get("language"),
categories=categories,
tags=tags,
canonical_url=doc.get("url") or url,
image=doc.get("image"),
pagetype=doc.get("pagetype"),
fingerprint=doc.get("fingerprint"),
license=doc.get("license"),
comments=doc.get("comments"),
text=(doc.get("text") or "").strip(),
markdown=(markdown_text or "").strip() if markdown_text else None,
raw_json=raw_json,
error=None,
)
else:
categories = list(doc.categories) if doc.categories else []
tags = list(doc.tags) if doc.tags else []
return TrafilaturaData(
title=doc.title,
author=doc.author,
date=doc.date,
description=doc.description,
sitename=doc.sitename,
hostname=doc.hostname,
language=doc.language,
categories=categories,
tags=tags,
canonical_url=doc.url or url,
image=doc.image,
pagetype=doc.pagetype,
fingerprint=doc.fingerprint,
license=doc.license,
comments=doc.comments,
text=(doc.text or "").strip(),
markdown=(markdown_text or "").strip() if markdown_text else None,
raw_json=raw_json,
error=None,
)
else:
raw_text = trafilatura.extract(html, output_format="txt", url=url) or ""
return TrafilaturaData(
title=raw_json.get("title") if raw_json else None,
text=raw_text.strip(),
markdown=markdown_text.strip() if markdown_text else None,
raw_json=raw_json,
error=None,
)
except Exception as e:
return TrafilaturaData(error=str(e))
class NewspaperExtractor:
"""Motor de extração baseado no Newspaper4k com NLP."""
@staticmethod
def extract(html: str, url: str = "", language: str = "en") -> NewspaperData:
try:
lang_code = language.split("-")[0].lower() if language else "en"
article = Article(url=url, language=lang_code)
article.download(input_html=html)
article.parse()
# Executar NLP para summary e keywords com fallback gracioso
try:
article.nlp()
summary = article.summary
keywords = list(article.keywords) if article.keywords else []
keyword_scores = getattr(article, "keyword_scores", {}) or {}
except Exception:
summary = None
keywords = []
keyword_scores = {}
publish_date_str = (
article.publish_date.isoformat()
if article.publish_date and hasattr(article.publish_date, "isoformat")
else str(article.publish_date)
if article.publish_date
else None
)
# Metadados e tags adicionais
meta_keywords = list(article.meta_keywords) if article.meta_keywords else []
tags = list(article.tags) if getattr(article, "tags", None) else []
movies = list(article.movies) if getattr(article, "movies", None) else []
images = list(article.images) if article.images else []
return NewspaperData(
title=article.title or None,
authors=list(article.authors) if article.authors else [],
publish_date=publish_date_str,
text=article.text or "",
summary=summary,
keywords=keywords,
keyword_scores=dict(keyword_scores),
top_image=article.top_image or getattr(article, "meta_img", None) or None,
images=images,
movies=movies,
tags=tags,
canonical_link=getattr(article, "canonical_link", None) or None,
article_html=getattr(article, "article_html", None) or None,
meta_description=getattr(article, "meta_description", None) or None,
meta_keywords=meta_keywords,
meta_favicon=getattr(article, "meta_favicon", None) or None,
meta_site_name=getattr(article, "meta_site_name", None) or None,
meta_lang=getattr(article, "meta_lang", None) or None,
meta_data=dict(article.meta_data) if article.meta_data else {},
error=None,
)
except Exception as e:
return NewspaperData(error=str(e))
class ReadabilityExtractor:
"""Motor de extração baseado no algoritmo Readability (readability-lxml)."""
@staticmethod
def extract(html: str) -> ReadabilityData:
try:
doc = Document(html)
title = doc.title()
short_title = doc.short_title()
cleaned_html = doc.summary()
author = None
try:
author = doc.author()
except Exception:
pass
# Extração de texto limpo a partir do HTML higienizado
soup = BeautifulSoup(cleaned_html, "html.parser")
cleaned_text = soup.get_text(separator="\n\n", strip=True)
return ReadabilityData(
title=title or None,
short_title=short_title or None,
author=author or None,
cleaned_html=cleaned_html or None,
cleaned_text=cleaned_text or None,
error=None,
)
except Exception as e:
return ReadabilityData(error=str(e))
def extract_all_engines(
html: str, url: str = "", language: str = "en"
) -> tuple[TrafilaturaData, NewspaperData, ReadabilityData]:
"""Executa a extração simultânea pelos três motores de conteúdo com isolamento defensivo."""
try:
traf_data = TrafilaturaExtractor.extract(html, url=url)
except Exception as exc:
traf_data = TrafilaturaData(
title=None,
author=None,
date=None,
description=None,
categories=[],
tags=[],
canonical_url=None,
text="",
raw_json=None,
error=str(exc),
)
try:
newspaper_data = NewspaperExtractor.extract(html, url=url, language=language)
except Exception as exc:
newspaper_data = NewspaperData(
title=None,
authors=[],
publish_date=None,
text="",
summary=None,
keywords=[],
top_image=None,
images=[],
meta_data={},
error=str(exc),
)
try:
readability_data = ReadabilityExtractor.extract(html)
except Exception as exc:
readability_data = ReadabilityData(
title=None,
short_title=None,
cleaned_html=None,
cleaned_text=None,
error=str(exc),
)
return traf_data, newspaper_data, readability_data
# ==============================================================================
# Motor de Navegação Stealth Headless com Foxcape
# ==============================================================================
class ArticleCrawler:
"""Gerenciador de ciclo de vida e requisições via Foxcape Headless."""
def __init__(self, timeout_sec: int = 30) -> None:
self.timeout_sec = timeout_sec
self.timeout_ms = timeout_sec * 1000
self._scraper: Foxcape | None = None
def start(self) -> None:
if self._scraper is None:
config = FoxcapeConfig(headless=True, humanize=False)
self._scraper = Foxcape(config=config)
self._scraper.start()
def close(self) -> None:
if self._scraper is not None:
try:
self._scraper.close()
except Exception:
pass
self._scraper = None
def __enter__(self) -> ArticleCrawler:
self.start()
return self
def __exit__(self, exc_type: Any, exc_val: Any, exc_tb: Any) -> None:
self.close()
def crawl(self, url: str) -> tuple[str, str | None, int | None]:
"""
Navega até a URL, aguarda o carregamento do DOM e retorna (html, page_title, http_status).
"""
if self._scraper is None:
self.start()
assert self._scraper is not None
result = self._scraper.get(
url,
wait_until="domcontentloaded",
timeout_ms=self.timeout_ms,
human_delay=False,
)
return result.html, result.title, result.status_code
# ==============================================================================
# Helpers de I/O e Orquestrador de Lote
# ==============================================================================
def log_info(message: str, silent: bool = False) -> None:
"""Escreve mensagem informativa no stderr."""
if not silent:
sys.stderr.write(f"[INFO] {message}\n")
sys.stderr.flush()
def load_search_json(file_path: Path) -> tuple[str | None, str, list[InputArticle]]:
"""Carrega o arquivo JSON gerado pelo extrator de notícias."""
if not file_path.exists():
raise FileNotFoundError(f"Arquivo de entrada não encontrado: {file_path}")
with file_path.open("r", encoding="utf-8") as f:
data = json.load(f)
query = data.get("query")
language = data.get("language", "en")
raw_items = data.get("items", [])
articles = []
for item in raw_items:
if isinstance(item, dict) and "url" in item and "titulo" in item:
articles.append(
InputArticle(
titulo=item["titulo"],
url=item["url"],
subtitulo=item.get("subtitulo"),
quando_publicado=item.get("quando_publicado"),
pagina=item.get("pagina", 1),
)
)
return query, language, articles
def save_extracted_json(report: ExtractionBatchReport, output_path: Path) -> None:
"""Salva o relatório consolidado em formato JSON com UTF-8."""
output_path.parent.mkdir(parents=True, exist_ok=True)
with output_path.open("w", encoding="utf-8") as f:
json.dump(report.to_dict(), f, ensure_ascii=False, indent=2)
def process_batch(
input_path: Path,
output_path: Path | None = None,
limit: int | None = None,
language_override: str | None = None,
timeout: int = 30,
silent: bool = False,
) -> ExtractionBatchReport:
"""
Executa o pipeline completo de extração em lote para o arquivo de entrada.
"""
query, search_lang, input_articles = load_search_json(input_path)
effective_lang = language_override or search_lang or "en"
if limit is not None and limit > 0:
input_articles = input_articles[:limit]
total = len(input_articles)
log_info(
f"🚀 Iniciando extração de {total} artigo(s) a partir de '{input_path}' (Idioma NLP: '{effective_lang}')...",
silent=silent,
)
# Determinar caminho de saída padrão se não especificado
if output_path is None:
output_path = input_path.parent / f"{input_path.stem}_extracted.json"
extracted_list: list[ExtractedArticle] = []
successful_count = 0
failed_count = 0
start_time = time.time()
with ArticleCrawler(timeout_sec=timeout) as crawler:
for idx, article in enumerate(input_articles, start=1):
url = article.url
log_info(f"🌐 [{idx}/{total}] Navegando com Foxcape: {url}", silent=silent)
try:
html, page_title, http_status = crawler.crawl(url)
log_info(
f"⚙️ [{idx}/{total}] Processando extratores (Trafilatura, Newspaper4k, Readability)...",
silent=silent,
)
traf_data, newspaper_data, readability_data = extract_all_engines(
html=html, url=url, language=effective_lang
)
extracted_article = ExtractedArticle(
input_meta=article,
extraction_status="success",
error_message=None,
crawled_url=url,
page_title=page_title,
http_status=http_status,
trafilatura=traf_data,
newspaper4k=newspaper_data,
readability=readability_data,
)
successful_count += 1
log_info(
f'✅ [{idx}/{total}] Sucesso (Título: "{article.titulo[:50]}...")',
silent=silent,
)
except Exception as exc:
failed_count += 1
error_msg = str(exc)
log_info(
f"⚠️ [{idx}/{total}] Falha ao processar URL '{url}': {error_msg}",
silent=silent,
)
extracted_article = ExtractedArticle(
input_meta=article,
extraction_status="failed",
error_message=error_msg,
crawled_url=url,
page_title=None,
http_status=None,
trafilatura=None,
newspaper4k=None,
readability=None,
)
extracted_list.append(extracted_article)
elapsed = time.time() - start_time
now_iso = datetime.now(timezone.utc).isoformat()
report = ExtractionBatchReport(
source_file=str(input_path),
processed_at=now_iso,
total_articles=total,
successful_articles=successful_count,
failed_articles=failed_count,
articles=extracted_list,
)
save_extracted_json(report, output_path)
log_info(f"💾 Relatório final gravado com sucesso em: '{output_path}'", silent=silent)
log_info(
f"📊 Resumo: {total} total | {successful_count} sucessos | {failed_count} falhas | Tempo: {elapsed:.2f}s",
silent=silent,
)
return report
# ==============================================================================
# Interface CLI
# ==============================================================================
def parse_arguments(args: list[str] | None = None) -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability)",
formatter_class=argparse.RawTextHelpFormatter,
)
parser.add_argument(
"-i",
"--input",
required=True,
type=str,
help="Caminho para o arquivo JSON de busca de notícias (ex: out/river_plate.json)",
)
parser.add_argument(
"-o",
"--output",
required=False,
type=str,
default=None,
help="Caminho do arquivo JSON de destino (padrão: <input_stem>_extracted.json)",
)
parser.add_argument(
"-l",
"--limit",
required=False,
type=int,
default=None,
help="Limita a quantidade máxima de artigos a serem processados",
)
parser.add_argument(
"--lang",
"--language",
dest="language",
required=False,
type=str,
default=None,
help="Sobrescreve o código de idioma para o NLP do Newspaper4k (ex: pt, es, en)",
)
parser.add_argument(
"-t",
"--timeout",
required=False,
type=int,
default=30,
help="Timeout em segundos para carregamento do DOM de cada página no Foxcape (padrão: 30)",
)
parser.add_argument(
"-s",
"--silent",
action="store_true",
help="Suprime mensagens de log e progresso no stderr",
)
return parser.parse_args(args)
def main(args: list[str] | None = None) -> int:
try:
parsed = parse_arguments(args)
input_path = Path(parsed.input)
output_path = Path(parsed.output) if parsed.output else None
if not input_path.exists():
sys.stderr.write(f"Erro: Arquivo de entrada '{input_path}' não existe.\n")
return 1
process_batch(
input_path=input_path,
output_path=output_path,
limit=parsed.limit,
language_override=parsed.language,
timeout=parsed.timeout,
silent=parsed.silent,
)
return 0
except KeyboardInterrupt:
sys.stderr.write("\nExecução cancelada pelo usuário.\n")
return 130
except Exception as exc:
sys.stderr.write(f"Erro fatal durante a execução: {exc}\n")
return 2
if __name__ == "__main__":
sys.exit(main())