#!/usr/bin/env python3 """ Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability). Lê listagens JSON de notícias (ex: out/river_plate.json), acessa e renderiza as páginas em modo stealth headless utilizando Foxcape reutilizando a mesma sessão de navegador, executa a extração em paralelo/sequência com 3 motores de conteúdo (Trafilatura, Newspaper4k e Readability) e salva o resultado enriquecido e higienizado em JSON. """ from __future__ import annotations import argparse import json import sys import time from dataclasses import dataclass, field from datetime import datetime, timezone from pathlib import Path from typing import Any, Literal import trafilatura from bs4 import BeautifulSoup from foxcape import Foxcape, FoxcapeConfig from newspaper import Article from readability import Document # ============================================================================== # Modelos de Dados e Dataclasses # ============================================================================== @dataclass(frozen=True) class InputArticle: """Metadados originais da notícia contida no JSON de entrada.""" titulo: str url: str subtitulo: str | None = None quando_publicado: str | None = None pagina: int = 1 def to_dict(self) -> dict[str, Any]: return { "titulo": self.titulo, "subtitulo": self.subtitulo, "quando_publicado": self.quando_publicado, "url": self.url, "pagina": self.pagina, } @dataclass(frozen=True) class TrafilaturaData: """Dados completos extraídos pelo motor Trafilatura.""" title: str | None = None author: str | None = None date: str | None = None description: str | None = None sitename: str | None = None hostname: str | None = None language: str | None = None categories: list[str] = field(default_factory=list) tags: list[str] = field(default_factory=list) canonical_url: str | None = None image: str | None = None pagetype: str | None = None fingerprint: str | None = None license: str | None = None comments: str | None = None text: str = "" markdown: str | None = None raw_json: dict[str, Any] | None = None error: str | None = None def to_dict(self) -> dict[str, Any]: return { "title": self.title, "author": self.author, "date": self.date, "description": self.description, "sitename": self.sitename, "hostname": self.hostname, "language": self.language, "categories": self.categories, "tags": self.tags, "canonical_url": self.canonical_url, "image": self.image, "pagetype": self.pagetype, "fingerprint": self.fingerprint, "license": self.license, "comments": self.comments, "text": self.text, "markdown": self.markdown, "raw_json": self.raw_json, "error": self.error, } @dataclass(frozen=True) class NewspaperData: """Dados completos extraídos e enriquecidos com NLP pelo motor Newspaper4k.""" title: str | None = None authors: list[str] = field(default_factory=list) publish_date: str | None = None text: str = "" summary: str | None = None keywords: list[str] = field(default_factory=list) keyword_scores: dict[str, float] = field(default_factory=dict) top_image: str | None = None images: list[str] = field(default_factory=list) movies: list[str] = field(default_factory=list) tags: list[str] = field(default_factory=list) canonical_link: str | None = None article_html: str | None = None meta_description: str | None = None meta_keywords: list[str] = field(default_factory=list) meta_favicon: str | None = None meta_site_name: str | None = None meta_lang: str | None = None meta_data: dict[str, Any] = field(default_factory=dict) error: str | None = None def to_dict(self) -> dict[str, Any]: return { "title": self.title, "authors": self.authors, "publish_date": self.publish_date, "text": self.text, "summary": self.summary, "keywords": self.keywords, "keyword_scores": self.keyword_scores, "top_image": self.top_image, "images": self.images, "movies": self.movies, "tags": self.tags, "canonical_link": self.canonical_link, "article_html": self.article_html, "meta_description": self.meta_description, "meta_keywords": self.meta_keywords, "meta_favicon": self.meta_favicon, "meta_site_name": self.meta_site_name, "meta_lang": self.meta_lang, "meta_data": self.meta_data, "error": self.error, } @dataclass(frozen=True) class ReadabilityData: """Dados completos higienizados pelo algoritmo Readability.""" title: str | None = None short_title: str | None = None author: str | None = None cleaned_html: str | None = None cleaned_text: str | None = None error: str | None = None def to_dict(self) -> dict[str, Any]: return { "title": self.title, "short_title": self.short_title, "author": self.author, "cleaned_html": self.cleaned_html, "cleaned_text": self.cleaned_text, "error": self.error, } @dataclass(frozen=True) class ExtractedArticle: """Resultado consolidado da extração de um artigo.""" input_meta: InputArticle extraction_status: Literal["success", "failed"] error_message: str | None crawled_url: str page_title: str | None http_status: int | None trafilatura: TrafilaturaData | None = None newspaper4k: NewspaperData | None = None readability: ReadabilityData | None = None def to_dict(self) -> dict[str, Any]: return { "input_meta": self.input_meta.to_dict(), "extraction_status": self.extraction_status, "error_message": self.error_message, "crawled_url": self.crawled_url, "page_title": self.page_title, "http_status": self.http_status, "trafilatura": self.trafilatura.to_dict() if self.trafilatura else None, "newspaper4k": self.newspaper4k.to_dict() if self.newspaper4k else None, "readability": self.readability.to_dict() if self.readability else None, } @dataclass(frozen=True) class ExtractionBatchReport: """Relatório consolidado de saída do processamento de um lote.""" source_file: str processed_at: str total_articles: int successful_articles: int failed_articles: int articles: list[ExtractedArticle] = field(default_factory=list) def to_dict(self) -> dict[str, Any]: return { "source_file": self.source_file, "processed_at": self.processed_at, "total_articles": self.total_articles, "successful_articles": self.successful_articles, "failed_articles": self.failed_articles, "articles": [a.to_dict() for a in self.articles], } # ============================================================================== # Parsers / Extratores Especializados # ============================================================================== class TrafilaturaExtractor: """Motor de extração baseado na biblioteca Trafilatura.""" @staticmethod def extract(html: str, url: str | None = None) -> TrafilaturaData: try: # Extração bare document completa doc = trafilatura.bare_extraction( html, url=url, include_comments=True, include_tables=True, include_images=True, include_links=True, include_formatting=True, with_metadata=True, ) # Extração em markdown markdown_text = trafilatura.extract( html, output_format="markdown", include_comments=True, include_tables=True, include_images=True, include_links=True, include_formatting=True, url=url, ) # Extração em JSON nativo json_output_str = trafilatura.extract( html, output_format="json", include_comments=True, include_tables=True, include_images=True, include_links=True, url=url, ) raw_json = json.loads(json_output_str) if json_output_str else None if doc: if isinstance(doc, dict): categories = list(doc.get("categories", [])) if doc.get("categories") else [] tags = list(doc.get("tags", [])) if doc.get("tags") else [] return TrafilaturaData( title=doc.get("title"), author=doc.get("author"), date=doc.get("date"), description=doc.get("description"), sitename=doc.get("sitename"), hostname=doc.get("hostname"), language=doc.get("language"), categories=categories, tags=tags, canonical_url=doc.get("url") or url, image=doc.get("image"), pagetype=doc.get("pagetype"), fingerprint=doc.get("fingerprint"), license=doc.get("license"), comments=doc.get("comments"), text=(doc.get("text") or "").strip(), markdown=(markdown_text or "").strip() if markdown_text else None, raw_json=raw_json, error=None, ) else: categories = list(doc.categories) if doc.categories else [] tags = list(doc.tags) if doc.tags else [] return TrafilaturaData( title=doc.title, author=doc.author, date=doc.date, description=doc.description, sitename=doc.sitename, hostname=doc.hostname, language=doc.language, categories=categories, tags=tags, canonical_url=doc.url or url, image=doc.image, pagetype=doc.pagetype, fingerprint=doc.fingerprint, license=doc.license, comments=doc.comments, text=(doc.text or "").strip(), markdown=(markdown_text or "").strip() if markdown_text else None, raw_json=raw_json, error=None, ) else: raw_text = trafilatura.extract(html, output_format="txt", url=url) or "" return TrafilaturaData( title=raw_json.get("title") if raw_json else None, text=raw_text.strip(), markdown=markdown_text.strip() if markdown_text else None, raw_json=raw_json, error=None, ) except Exception as e: return TrafilaturaData(error=str(e)) class NewspaperExtractor: """Motor de extração baseado no Newspaper4k com NLP.""" @staticmethod def extract(html: str, url: str = "", language: str = "en") -> NewspaperData: try: lang_code = language.split("-")[0].lower() if language else "en" article = Article(url=url, language=lang_code) article.download(input_html=html) article.parse() # Executar NLP para summary e keywords com fallback gracioso try: article.nlp() summary = article.summary keywords = list(article.keywords) if article.keywords else [] keyword_scores = getattr(article, "keyword_scores", {}) or {} except Exception: summary = None keywords = [] keyword_scores = {} publish_date_str = ( article.publish_date.isoformat() if article.publish_date and hasattr(article.publish_date, "isoformat") else str(article.publish_date) if article.publish_date else None ) # Metadados e tags adicionais meta_keywords = list(article.meta_keywords) if article.meta_keywords else [] tags = list(article.tags) if getattr(article, "tags", None) else [] movies = list(article.movies) if getattr(article, "movies", None) else [] images = list(article.images) if article.images else [] return NewspaperData( title=article.title or None, authors=list(article.authors) if article.authors else [], publish_date=publish_date_str, text=article.text or "", summary=summary, keywords=keywords, keyword_scores=dict(keyword_scores), top_image=article.top_image or getattr(article, "meta_img", None) or None, images=images, movies=movies, tags=tags, canonical_link=getattr(article, "canonical_link", None) or None, article_html=getattr(article, "article_html", None) or None, meta_description=getattr(article, "meta_description", None) or None, meta_keywords=meta_keywords, meta_favicon=getattr(article, "meta_favicon", None) or None, meta_site_name=getattr(article, "meta_site_name", None) or None, meta_lang=getattr(article, "meta_lang", None) or None, meta_data=dict(article.meta_data) if article.meta_data else {}, error=None, ) except Exception as e: return NewspaperData(error=str(e)) class ReadabilityExtractor: """Motor de extração baseado no algoritmo Readability (readability-lxml).""" @staticmethod def extract(html: str) -> ReadabilityData: try: doc = Document(html) title = doc.title() short_title = doc.short_title() cleaned_html = doc.summary() author = None try: author = doc.author() except Exception: pass # Extração de texto limpo a partir do HTML higienizado soup = BeautifulSoup(cleaned_html, "html.parser") cleaned_text = soup.get_text(separator="\n\n", strip=True) return ReadabilityData( title=title or None, short_title=short_title or None, author=author or None, cleaned_html=cleaned_html or None, cleaned_text=cleaned_text or None, error=None, ) except Exception as e: return ReadabilityData(error=str(e)) def extract_all_engines( html: str, url: str = "", language: str = "en" ) -> tuple[TrafilaturaData, NewspaperData, ReadabilityData]: """Executa a extração simultânea pelos três motores de conteúdo com isolamento defensivo.""" try: traf_data = TrafilaturaExtractor.extract(html, url=url) except Exception as exc: traf_data = TrafilaturaData( title=None, author=None, date=None, description=None, categories=[], tags=[], canonical_url=None, text="", raw_json=None, error=str(exc), ) try: newspaper_data = NewspaperExtractor.extract(html, url=url, language=language) except Exception as exc: newspaper_data = NewspaperData( title=None, authors=[], publish_date=None, text="", summary=None, keywords=[], top_image=None, images=[], meta_data={}, error=str(exc), ) try: readability_data = ReadabilityExtractor.extract(html) except Exception as exc: readability_data = ReadabilityData( title=None, short_title=None, cleaned_html=None, cleaned_text=None, error=str(exc), ) return traf_data, newspaper_data, readability_data # ============================================================================== # Motor de Navegação Stealth Headless com Foxcape # ============================================================================== class ArticleCrawler: """Gerenciador de ciclo de vida e requisições via Foxcape Headless.""" def __init__(self, timeout_sec: int = 30) -> None: self.timeout_sec = timeout_sec self.timeout_ms = timeout_sec * 1000 self._scraper: Foxcape | None = None def start(self) -> None: if self._scraper is None: config = FoxcapeConfig(headless=True, humanize=False) self._scraper = Foxcape(config=config) self._scraper.start() def close(self) -> None: if self._scraper is not None: try: self._scraper.close() except Exception: pass self._scraper = None def __enter__(self) -> ArticleCrawler: self.start() return self def __exit__(self, exc_type: Any, exc_val: Any, exc_tb: Any) -> None: self.close() def crawl(self, url: str) -> tuple[str, str | None, int | None]: """ Navega até a URL, aguarda o carregamento do DOM e retorna (html, page_title, http_status). """ if self._scraper is None: self.start() assert self._scraper is not None result = self._scraper.get( url, wait_until="domcontentloaded", timeout_ms=self.timeout_ms, human_delay=False, ) return result.html, result.title, result.status_code # ============================================================================== # Helpers de I/O e Orquestrador de Lote # ============================================================================== def log_info(message: str, silent: bool = False) -> None: """Escreve mensagem informativa no stderr.""" if not silent: sys.stderr.write(f"[INFO] {message}\n") sys.stderr.flush() def load_search_json(file_path: Path) -> tuple[str | None, str, list[InputArticle]]: """Carrega o arquivo JSON gerado pelo extrator de notícias.""" if not file_path.exists(): raise FileNotFoundError(f"Arquivo de entrada não encontrado: {file_path}") with file_path.open("r", encoding="utf-8") as f: data = json.load(f) query = data.get("query") language = data.get("language", "en") raw_items = data.get("items", []) articles = [] for item in raw_items: if isinstance(item, dict) and "url" in item and "titulo" in item: articles.append( InputArticle( titulo=item["titulo"], url=item["url"], subtitulo=item.get("subtitulo"), quando_publicado=item.get("quando_publicado"), pagina=item.get("pagina", 1), ) ) return query, language, articles def save_extracted_json(report: ExtractionBatchReport, output_path: Path) -> None: """Salva o relatório consolidado em formato JSON com UTF-8.""" output_path.parent.mkdir(parents=True, exist_ok=True) with output_path.open("w", encoding="utf-8") as f: json.dump(report.to_dict(), f, ensure_ascii=False, indent=2) def process_batch( input_path: Path, output_path: Path | None = None, limit: int | None = None, language_override: str | None = None, timeout: int = 30, silent: bool = False, ) -> ExtractionBatchReport: """ Executa o pipeline completo de extração em lote para o arquivo de entrada. """ query, search_lang, input_articles = load_search_json(input_path) effective_lang = language_override or search_lang or "en" if limit is not None and limit > 0: input_articles = input_articles[:limit] total = len(input_articles) log_info( f"🚀 Iniciando extração de {total} artigo(s) a partir de '{input_path}' (Idioma NLP: '{effective_lang}')...", silent=silent, ) # Determinar caminho de saída padrão se não especificado if output_path is None: output_path = input_path.parent / f"{input_path.stem}_extracted.json" extracted_list: list[ExtractedArticle] = [] successful_count = 0 failed_count = 0 start_time = time.time() with ArticleCrawler(timeout_sec=timeout) as crawler: for idx, article in enumerate(input_articles, start=1): url = article.url log_info(f"🌐 [{idx}/{total}] Navegando com Foxcape: {url}", silent=silent) try: html, page_title, http_status = crawler.crawl(url) log_info( f"⚙️ [{idx}/{total}] Processando extratores (Trafilatura, Newspaper4k, Readability)...", silent=silent, ) traf_data, newspaper_data, readability_data = extract_all_engines( html=html, url=url, language=effective_lang ) extracted_article = ExtractedArticle( input_meta=article, extraction_status="success", error_message=None, crawled_url=url, page_title=page_title, http_status=http_status, trafilatura=traf_data, newspaper4k=newspaper_data, readability=readability_data, ) successful_count += 1 log_info( f'✅ [{idx}/{total}] Sucesso (Título: "{article.titulo[:50]}...")', silent=silent, ) except Exception as exc: failed_count += 1 error_msg = str(exc) log_info( f"⚠️ [{idx}/{total}] Falha ao processar URL '{url}': {error_msg}", silent=silent, ) extracted_article = ExtractedArticle( input_meta=article, extraction_status="failed", error_message=error_msg, crawled_url=url, page_title=None, http_status=None, trafilatura=None, newspaper4k=None, readability=None, ) extracted_list.append(extracted_article) elapsed = time.time() - start_time now_iso = datetime.now(timezone.utc).isoformat() report = ExtractionBatchReport( source_file=str(input_path), processed_at=now_iso, total_articles=total, successful_articles=successful_count, failed_articles=failed_count, articles=extracted_list, ) save_extracted_json(report, output_path) log_info(f"💾 Relatório final gravado com sucesso em: '{output_path}'", silent=silent) log_info( f"📊 Resumo: {total} total | {successful_count} sucessos | {failed_count} falhas | Tempo: {elapsed:.2f}s", silent=silent, ) return report # ============================================================================== # Interface CLI # ============================================================================== def parse_arguments(args: list[str] | None = None) -> argparse.Namespace: parser = argparse.ArgumentParser( description="Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability)", formatter_class=argparse.RawTextHelpFormatter, ) parser.add_argument( "-i", "--input", required=True, type=str, help="Caminho para o arquivo JSON de busca de notícias (ex: out/river_plate.json)", ) parser.add_argument( "-o", "--output", required=False, type=str, default=None, help="Caminho do arquivo JSON de destino (padrão: _extracted.json)", ) parser.add_argument( "-l", "--limit", required=False, type=int, default=None, help="Limita a quantidade máxima de artigos a serem processados", ) parser.add_argument( "--lang", "--language", dest="language", required=False, type=str, default=None, help="Sobrescreve o código de idioma para o NLP do Newspaper4k (ex: pt, es, en)", ) parser.add_argument( "-t", "--timeout", required=False, type=int, default=30, help="Timeout em segundos para carregamento do DOM de cada página no Foxcape (padrão: 30)", ) parser.add_argument( "-s", "--silent", action="store_true", help="Suprime mensagens de log e progresso no stderr", ) return parser.parse_args(args) def main(args: list[str] | None = None) -> int: try: parsed = parse_arguments(args) input_path = Path(parsed.input) output_path = Path(parsed.output) if parsed.output else None if not input_path.exists(): sys.stderr.write(f"Erro: Arquivo de entrada '{input_path}' não existe.\n") return 1 process_batch( input_path=input_path, output_path=output_path, limit=parsed.limit, language_override=parsed.language, timeout=parsed.timeout, silent=parsed.silent, ) return 0 except KeyboardInterrupt: sys.stderr.write("\nExecução cancelada pelo usuário.\n") return 130 except Exception as exc: sys.stderr.write(f"Erro fatal durante a execução: {exc}\n") return 2 if __name__ == "__main__": sys.exit(main())