- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
464 lines
14 KiB
Python
464 lines
14 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Extrator de Manchetes do Google News via RSS.
|
|
|
|
Script CLI autônomo para busca, extração e estruturação de notícias
|
|
por termo/assunto, idioma e localização geográfica (locale), utilizando
|
|
a biblioteca Foxcape para raspagem indetectável e decodificação automática
|
|
das URLs intermediárias do Google News para as URLs finais dos veículos.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import concurrent.futures
|
|
import json
|
|
import re
|
|
import sys
|
|
import urllib.error
|
|
import urllib.parse
|
|
import urllib.request
|
|
from dataclasses import dataclass, field
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from bs4 import BeautifulSoup, FeatureNotFound
|
|
from foxcape import Foxcape, FoxcapeConfig
|
|
from googlenewsdecoder import gnewsdecoder # type: ignore[import-untyped]
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class SearchQuery:
|
|
"""Value Object com parâmetros de busca validados."""
|
|
|
|
keyword: str
|
|
language: str = "pt"
|
|
locale: str | None = None
|
|
max_pages: int = 1
|
|
|
|
def __post_init__(self) -> None:
|
|
if not self.keyword or not self.keyword.strip():
|
|
raise ValueError("A palavra-chave não pode ser vazia.")
|
|
|
|
if not self.language or len(self.language.strip()) < 2:
|
|
raise ValueError("O idioma deve conter pelo menos 2 caracteres (ex: 'pt', 'en', 'es').")
|
|
|
|
if self.max_pages < 1 or self.max_pages > 10:
|
|
raise ValueError("O número máximo de páginas deve estar entre 1 e 10.")
|
|
|
|
@property
|
|
def clean_keyword(self) -> str:
|
|
return self.keyword.strip()
|
|
|
|
@property
|
|
def clean_language(self) -> str:
|
|
return self.language.strip().lower()
|
|
|
|
@property
|
|
def clean_locale(self) -> str | None:
|
|
return self.locale.strip().upper() if self.locale else None
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class NewsArticle:
|
|
"""Entidade que representa uma notícia extraída."""
|
|
|
|
titulo: str
|
|
url: str
|
|
pagina: int
|
|
subtitulo: str | None = None
|
|
quando_publicado: str | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"titulo": self.titulo,
|
|
"subtitulo": self.subtitulo,
|
|
"quando_publicado": self.quando_publicado,
|
|
"url": self.url,
|
|
"pagina": self.pagina,
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ExtractionResult:
|
|
"""Resultado consolidado da extração."""
|
|
|
|
query: str
|
|
language: str
|
|
locale: str
|
|
total_paginas: int
|
|
total_itens: int
|
|
scraped_at: str
|
|
items: list[NewsArticle] = field(default_factory=list)
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"query": self.query,
|
|
"language": self.language,
|
|
"locale": self.locale,
|
|
"total_paginas": self.total_paginas,
|
|
"total_itens": self.total_itens,
|
|
"scraped_at": self.scraped_at,
|
|
"items": [item.to_dict() for item in self.items],
|
|
}
|
|
|
|
|
|
def get_hl_gl_ceid(lang_raw: str, locale: str | None = None) -> tuple[str, str, str]:
|
|
"""
|
|
Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News.
|
|
"""
|
|
lang_clean = lang_raw.lower().replace("-", "_")
|
|
|
|
locale_map: dict[str, tuple[str, str]] = {
|
|
"pt": ("pt-BR", "BR"),
|
|
"pt_br": ("pt-BR", "BR"),
|
|
"es": ("es-419", "AR"),
|
|
"es_mx": ("es-419", "MX"),
|
|
"es_es": ("es", "ES"),
|
|
"en": ("en-US", "US"),
|
|
"en_gb": ("en-GB", "GB"),
|
|
"en_uk": ("en-GB", "GB"),
|
|
"en_us": ("en-US", "US"),
|
|
"de": ("de", "DE"),
|
|
"de_de": ("de", "DE"),
|
|
"it": ("it", "IT"),
|
|
"it_it": ("it", "IT"),
|
|
"fr": ("fr", "FR"),
|
|
"fr_fr": ("fr", "FR"),
|
|
}
|
|
|
|
if lang_clean in locale_map:
|
|
hl, gl = locale_map[lang_clean]
|
|
else:
|
|
parts = lang_clean.split("_")
|
|
if len(parts) == 2:
|
|
hl, gl = f"{parts[0]}-{parts[1].upper()}", parts[1].upper()
|
|
else:
|
|
hl, gl = lang_clean, lang_clean.upper()
|
|
|
|
if locale:
|
|
gl = locale.strip().upper()
|
|
if lang_clean == "es" and gl == "ES":
|
|
hl = "es"
|
|
elif lang_clean == "en" and gl == "GB":
|
|
hl = "en-GB"
|
|
|
|
ceid = f"{gl}:{hl}"
|
|
return hl, gl, ceid
|
|
|
|
|
|
def _normalize_text_for_comparison(text: str) -> str:
|
|
"""Remove pontuação e espaços extras para comparação de redundância."""
|
|
return re.sub(r"[^\w\s]", "", text).lower().strip()
|
|
|
|
|
|
def parse_google_news_rss(xml_content: str, max_pages: int = 1) -> list[NewsArticle]:
|
|
"""
|
|
Parseia o XML do RSS do Google News e extrai os itens estruturados.
|
|
"""
|
|
try:
|
|
soup = BeautifulSoup(xml_content, "xml")
|
|
except FeatureNotFound:
|
|
soup = BeautifulSoup(xml_content, "html.parser")
|
|
|
|
rss_items = soup.find_all("item")
|
|
articles: list[NewsArticle] = []
|
|
max_allowed = max_pages * 10
|
|
|
|
for idx, item in enumerate(rss_items[:max_allowed]):
|
|
page_number = (idx // 10) + 1
|
|
|
|
title_tag = item.find("title")
|
|
link_tag = item.find("link")
|
|
pubdate_tag = item.find("pubDate")
|
|
desc_tag = item.find("description")
|
|
|
|
title = title_tag.get_text(strip=True) if title_tag else ""
|
|
link = link_tag.get_text(strip=True) if link_tag else ""
|
|
pub_date = pubdate_tag.get_text(strip=True) if pubdate_tag else None
|
|
desc_raw = desc_tag.get_text() if desc_tag else ""
|
|
|
|
# Limpeza de HTML do resumo / snippet
|
|
snippet = None
|
|
if desc_raw:
|
|
desc_soup = BeautifulSoup(desc_raw, "html.parser")
|
|
clean_snippet = desc_soup.get_text(separator=" ", strip=True)
|
|
if clean_snippet:
|
|
norm_snippet = _normalize_text_for_comparison(clean_snippet)
|
|
norm_title = _normalize_text_for_comparison(title)
|
|
if norm_snippet != norm_title:
|
|
snippet = clean_snippet
|
|
|
|
if title and link:
|
|
articles.append(
|
|
NewsArticle(
|
|
titulo=title,
|
|
subtitulo=snippet,
|
|
quando_publicado=pub_date,
|
|
url=link,
|
|
pagina=page_number,
|
|
)
|
|
)
|
|
|
|
return articles
|
|
|
|
|
|
def resolve_article_url(url: str) -> str:
|
|
"""Resolve a URL intermediária do Google News para a URL real do veículo."""
|
|
if not url or "news.google.com" not in url:
|
|
return url
|
|
try:
|
|
res = gnewsdecoder(url)
|
|
if isinstance(res, dict) and res.get("status") and res.get("decoded_url"):
|
|
return str(res["decoded_url"])
|
|
except (ValueError, OSError, RuntimeError, TypeError):
|
|
pass
|
|
return url
|
|
|
|
|
|
def resolve_articles_urls(
|
|
articles: list[NewsArticle], max_workers: int = 5, verbose: bool = True
|
|
) -> list[NewsArticle]:
|
|
"""Resolve em paralelo as URLs intermediárias do Google News para os links finais dos veículos."""
|
|
if not articles:
|
|
return articles
|
|
|
|
if verbose:
|
|
sys.stderr.write(
|
|
f"[INFO] 🔗 Decodificando {len(articles)} URLs do Google News para os portais reais...\n"
|
|
)
|
|
|
|
def _resolve_single(art: NewsArticle) -> NewsArticle:
|
|
return NewsArticle(
|
|
titulo=art.titulo,
|
|
subtitulo=art.subtitulo,
|
|
quando_publicado=art.quando_publicado,
|
|
url=resolve_article_url(art.url),
|
|
pagina=art.pagina,
|
|
)
|
|
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
|
|
resolved = list(executor.map(_resolve_single, articles))
|
|
|
|
if verbose:
|
|
resolved_count = sum(1 for a in resolved if "news.google.com" not in a.url)
|
|
sys.stderr.write(
|
|
f"[INFO] ✅ {resolved_count}/{len(resolved)} URLs resolvidas com sucesso para os domínios de origem.\n"
|
|
)
|
|
|
|
return resolved
|
|
|
|
|
|
def _fetch_rss_content(rss_url: str) -> str:
|
|
"""
|
|
Realiza a requisição ao feed RSS utilizando a biblioteca Foxcape como
|
|
motor primário de extração e evasão anti-bot em modo headless.
|
|
"""
|
|
config = FoxcapeConfig(
|
|
headless=True,
|
|
humanize=False,
|
|
)
|
|
try:
|
|
result = Foxcape.fetch(
|
|
rss_url,
|
|
config=config,
|
|
timeout_ms=15000,
|
|
human_delay=False,
|
|
)
|
|
if result and result.html:
|
|
return str(result.html)
|
|
raise RuntimeError("Foxcape não retornou conteúdo para a URL informada.")
|
|
except (RuntimeError, OSError, TimeoutError, ValueError) as exc:
|
|
# Fallback de contingência caso o ambiente não possua suporte a subprocessos de navegador
|
|
try:
|
|
headers = {
|
|
"User-Agent": (
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
|
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/132.0.0.0 Safari/537.36"
|
|
),
|
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|
"Accept-Language": "pt-BR,pt;q=0.9,en-US;q=0.8,en;q=0.7,es;q=0.6",
|
|
}
|
|
req = urllib.request.Request(rss_url, headers=headers)
|
|
with urllib.request.urlopen(req, timeout=10) as response:
|
|
return response.read().decode("utf-8", errors="replace")
|
|
except (
|
|
urllib.error.URLError,
|
|
urllib.error.HTTPError,
|
|
OSError,
|
|
TimeoutError,
|
|
) as fallback_exc:
|
|
raise RuntimeError(
|
|
f"Erro na extração Foxcape: {exc} | Contingência: {fallback_exc}"
|
|
) from exc
|
|
|
|
|
|
def extract_google_news(
|
|
query: SearchQuery, resolve_urls: bool = True, verbose: bool = True
|
|
) -> ExtractionResult:
|
|
"""
|
|
Orquestra a consulta ao Google News e retorna o resultado consolidado
|
|
com URLs resolvidas para os portais de notícias.
|
|
"""
|
|
encoded_query = urllib.parse.quote_plus(query.clean_keyword)
|
|
hl, gl, ceid = get_hl_gl_ceid(query.clean_language, query.clean_locale)
|
|
rss_url = f"https://news.google.com/rss/search?q={encoded_query}&hl={hl}&gl={gl}&ceid={ceid}"
|
|
|
|
if verbose:
|
|
sys.stderr.write(
|
|
f"[INFO] 🔍 Consultando Google News: '{query.clean_keyword}' "
|
|
f"(idioma: {query.clean_language}, locale: {gl}, max_pages: {query.max_pages})...\n"
|
|
)
|
|
|
|
xml_content = _fetch_rss_content(rss_url)
|
|
if verbose:
|
|
sys.stderr.write(f"[INFO] 📥 Feed RSS recebido ({len(xml_content)} bytes).\n")
|
|
|
|
articles = parse_google_news_rss(xml_content, max_pages=query.max_pages)
|
|
if verbose:
|
|
sys.stderr.write(f"[INFO] 📰 {len(articles)} artigos extraídos do feed XML.\n")
|
|
|
|
if resolve_urls and articles:
|
|
articles = resolve_articles_urls(articles, verbose=verbose)
|
|
|
|
return ExtractionResult(
|
|
query=query.clean_keyword,
|
|
language=query.clean_language,
|
|
locale=gl,
|
|
total_paginas=query.max_pages,
|
|
total_itens=len(articles),
|
|
scraped_at=datetime.now(timezone.utc).isoformat(),
|
|
items=articles,
|
|
)
|
|
|
|
|
|
def build_parser() -> argparse.ArgumentParser:
|
|
"""Cria e configura o parser de argumentos CLI."""
|
|
parser = argparse.ArgumentParser(
|
|
prog="extract_google_news",
|
|
description="Extrator de manchetes do Google News RSS por assunto, idioma e região.",
|
|
)
|
|
parser.add_argument(
|
|
"-q",
|
|
"--query",
|
|
"--keyword",
|
|
dest="query",
|
|
required=True,
|
|
help="Termo ou expressão de busca (obrigatório).",
|
|
)
|
|
parser.add_argument(
|
|
"-l",
|
|
"--lang",
|
|
"--language",
|
|
dest="lang",
|
|
default="pt",
|
|
help="Código do idioma (padrão: pt).",
|
|
)
|
|
parser.add_argument(
|
|
"--locale",
|
|
"--country",
|
|
dest="locale",
|
|
default=None,
|
|
help="Código do país/região (ex: BR, US, MX, ES).",
|
|
)
|
|
parser.add_argument(
|
|
"-p",
|
|
"--max-pages",
|
|
dest="max_pages",
|
|
type=int,
|
|
default=1,
|
|
help="Número de páginas (1 a 10, com 10 itens por página; padrão: 1).",
|
|
)
|
|
parser.add_argument(
|
|
"-o",
|
|
"--output",
|
|
dest="output",
|
|
default=None,
|
|
help="Caminho do arquivo para salvar a saída JSON.",
|
|
)
|
|
parser.add_argument(
|
|
"--pretty",
|
|
action="store_true",
|
|
help="Formata a saída JSON com indentação legível.",
|
|
)
|
|
parser.add_argument(
|
|
"--no-resolve-urls",
|
|
dest="resolve_urls",
|
|
action="store_false",
|
|
default=True,
|
|
help="Desativa a decodificação para as URLs diretas dos portais de notícias.",
|
|
)
|
|
parser.add_argument(
|
|
"-s",
|
|
"--silent",
|
|
"--quiet",
|
|
dest="silent",
|
|
action="store_true",
|
|
help="Suprime as mensagens informativas de progresso.",
|
|
)
|
|
return parser
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
"""Ponto de entrada do script CLI."""
|
|
if hasattr(sys.stdout, "reconfigure"):
|
|
try:
|
|
sys.stdout.reconfigure(encoding="utf-8")
|
|
except OSError:
|
|
pass
|
|
if hasattr(sys.stderr, "reconfigure"):
|
|
try:
|
|
sys.stderr.reconfigure(encoding="utf-8")
|
|
except OSError:
|
|
pass
|
|
|
|
parser = build_parser()
|
|
|
|
try:
|
|
args = parser.parse_args(argv)
|
|
query = SearchQuery(
|
|
keyword=args.query,
|
|
language=args.lang,
|
|
locale=args.locale,
|
|
max_pages=args.max_pages,
|
|
)
|
|
except (ValueError, argparse.ArgumentError) as err:
|
|
sys.stderr.write(f"Erro de validação: {err}\n")
|
|
return 1
|
|
except SystemExit as err:
|
|
return err.code if isinstance(err.code, int) else 1
|
|
|
|
verbose = not getattr(args, "silent", False)
|
|
|
|
try:
|
|
result = extract_google_news(query, resolve_urls=args.resolve_urls, verbose=verbose)
|
|
json_output = json.dumps(
|
|
result.to_dict(),
|
|
ensure_ascii=False,
|
|
indent=2 if args.pretty else None,
|
|
)
|
|
|
|
if args.output:
|
|
out_path = Path(args.output)
|
|
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
out_path.write_text(json_output + "\n", encoding="utf-8")
|
|
if verbose:
|
|
sys.stderr.write(
|
|
f"[INFO] 💾 Arquivo salvo com sucesso: '{args.output}' ({result.total_itens} notícias).\n"
|
|
)
|
|
else:
|
|
sys.stdout.write(json_output + "\n")
|
|
|
|
return 0
|
|
except RuntimeError as err:
|
|
sys.stderr.write(f"Erro na extração: {err}\n")
|
|
return 2
|
|
except OSError as err:
|
|
sys.stderr.write(f"Erro de arquivo/sistema: {err}\n")
|
|
return 2
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|