feat(extractor): add Google News headlines extractor with Foxcape headless and URL resolution

- Add standalone CLI script scripts/extract_google_news.py for Google News RSS scraping
- Integrate foxcape in headless mode as primary stealth anti-bot engine
- Implement parallel article URL resolution using googlenewsdecoder and ThreadPoolExecutor
- Support language and regional locale mapping (-l, --lang, --locale)
- Implement real-time progress logging in stderr and --silent flag
- Add unit, integration, and live E2E tests in tests/test_extract_google_news.py
- Add full SpecKit documentation (specs/002-google-news-extractor/)
- Create comprehensive README.md covering both NLP Classifier and Google News Extractor
This commit is contained in:
2026-08-20 11:50:16 -03:00
parent 67cc40f91a
commit 6e3d57619b
59 changed files with 16118 additions and 2160 deletions
+467
View File
@@ -0,0 +1,467 @@
#!/usr/bin/env python3
"""
Extrator de Manchetes do Google News via RSS.
Script CLI autônomo para busca, extração e estruturação de notícias
por termo/assunto, idioma e localização geográfica (locale), utilizando
a biblioteca Foxcape para raspagem indetectável e decodificação automática
das URLs intermediárias do Google News para as URLs finais dos veículos.
"""
from __future__ import annotations
import argparse
import concurrent.futures
import json
import re
import sys
import urllib.error
import urllib.parse
import urllib.request
from dataclasses import dataclass, field
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
from bs4 import BeautifulSoup, FeatureNotFound
from foxcape import Foxcape, FoxcapeConfig
from googlenewsdecoder import gnewsdecoder # type: ignore[import-untyped]
@dataclass(frozen=True)
class SearchQuery:
"""Value Object com parâmetros de busca validados."""
keyword: str
language: str = "pt"
locale: str | None = None
max_pages: int = 1
def __post_init__(self) -> None:
if not self.keyword or not self.keyword.strip():
raise ValueError("A palavra-chave não pode ser vazia.")
if not self.language or len(self.language.strip()) < 2:
raise ValueError(
"O idioma deve conter pelo menos 2 caracteres (ex: 'pt', 'en', 'es')."
)
if self.max_pages < 1 or self.max_pages > 10:
raise ValueError("O número máximo de páginas deve estar entre 1 e 10.")
@property
def clean_keyword(self) -> str:
return self.keyword.strip()
@property
def clean_language(self) -> str:
return self.language.strip().lower()
@property
def clean_locale(self) -> str | None:
return self.locale.strip().upper() if self.locale else None
@dataclass(frozen=True)
class NewsArticle:
"""Entidade que representa uma notícia extraída."""
titulo: str
url: str
pagina: int
subtitulo: str | None = None
quando_publicado: str | None = None
def to_dict(self) -> dict[str, Any]:
return {
"titulo": self.titulo,
"subtitulo": self.subtitulo,
"quando_publicado": self.quando_publicado,
"url": self.url,
"pagina": self.pagina,
}
@dataclass(frozen=True)
class ExtractionResult:
"""Resultado consolidado da extração."""
query: str
language: str
locale: str
total_paginas: int
total_itens: int
scraped_at: str
items: list[NewsArticle] = field(default_factory=list)
def to_dict(self) -> dict[str, Any]:
return {
"query": self.query,
"language": self.language,
"locale": self.locale,
"total_paginas": self.total_paginas,
"total_itens": self.total_itens,
"scraped_at": self.scraped_at,
"items": [item.to_dict() for item in self.items],
}
def get_hl_gl_ceid(lang_raw: str, locale: str | None = None) -> tuple[str, str, str]:
"""
Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News.
"""
lang_clean = lang_raw.lower().replace("-", "_")
locale_map: dict[str, tuple[str, str]] = {
"pt": ("pt-BR", "BR"),
"pt_br": ("pt-BR", "BR"),
"es": ("es-419", "AR"),
"es_mx": ("es-419", "MX"),
"es_es": ("es", "ES"),
"en": ("en-US", "US"),
"en_gb": ("en-GB", "GB"),
"en_uk": ("en-GB", "GB"),
"en_us": ("en-US", "US"),
"de": ("de", "DE"),
"de_de": ("de", "DE"),
"it": ("it", "IT"),
"it_it": ("it", "IT"),
"fr": ("fr", "FR"),
"fr_fr": ("fr", "FR"),
}
if lang_clean in locale_map:
hl, gl = locale_map[lang_clean]
else:
parts = lang_clean.split("_")
if len(parts) == 2:
hl, gl = f"{parts[0]}-{parts[1].upper()}", parts[1].upper()
else:
hl, gl = lang_clean, lang_clean.upper()
if locale:
gl = locale.strip().upper()
if lang_clean == "es" and gl == "ES":
hl = "es"
elif lang_clean == "en" and gl == "GB":
hl = "en-GB"
ceid = f"{gl}:{hl}"
return hl, gl, ceid
def _normalize_text_for_comparison(text: str) -> str:
"""Remove pontuação e espaços extras para comparação de redundância."""
return re.sub(r"[^\w\s]", "", text).lower().strip()
def parse_google_news_rss(xml_content: str, max_pages: int = 1) -> list[NewsArticle]:
"""
Parseia o XML do RSS do Google News e extrai os itens estruturados.
"""
try:
soup = BeautifulSoup(xml_content, "xml")
except FeatureNotFound:
soup = BeautifulSoup(xml_content, "html.parser")
rss_items = soup.find_all("item")
articles: list[NewsArticle] = []
max_allowed = max_pages * 10
for idx, item in enumerate(rss_items[:max_allowed]):
page_number = (idx // 10) + 1
title_tag = item.find("title")
link_tag = item.find("link")
pubdate_tag = item.find("pubDate")
desc_tag = item.find("description")
title = title_tag.get_text(strip=True) if title_tag else ""
link = link_tag.get_text(strip=True) if link_tag else ""
pub_date = pubdate_tag.get_text(strip=True) if pubdate_tag else None
desc_raw = desc_tag.get_text() if desc_tag else ""
# Limpeza de HTML do resumo / snippet
snippet = None
if desc_raw:
desc_soup = BeautifulSoup(desc_raw, "html.parser")
clean_snippet = desc_soup.get_text(separator=" ", strip=True)
if clean_snippet:
norm_snippet = _normalize_text_for_comparison(clean_snippet)
norm_title = _normalize_text_for_comparison(title)
if norm_snippet != norm_title:
snippet = clean_snippet
if title and link:
articles.append(
NewsArticle(
titulo=title,
subtitulo=snippet,
quando_publicado=pub_date,
url=link,
pagina=page_number,
)
)
return articles
def resolve_article_url(url: str) -> str:
"""Resolve a URL intermediária do Google News para a URL real do veículo."""
if not url or "news.google.com" not in url:
return url
try:
res = gnewsdecoder(url)
if isinstance(res, dict) and res.get("status") and res.get("decoded_url"):
return str(res["decoded_url"])
except (ValueError, OSError, RuntimeError, TypeError):
pass
return url
def resolve_articles_urls(
articles: list[NewsArticle], max_workers: int = 5, verbose: bool = True
) -> list[NewsArticle]:
"""Resolve em paralelo as URLs intermediárias do Google News para os links finais dos veículos."""
if not articles:
return articles
if verbose:
sys.stderr.write(
f"[INFO] 🔗 Decodificando {len(articles)} URLs do Google News para os portais reais...\n"
)
def _resolve_single(art: NewsArticle) -> NewsArticle:
return NewsArticle(
titulo=art.titulo,
subtitulo=art.subtitulo,
quando_publicado=art.quando_publicado,
url=resolve_article_url(art.url),
pagina=art.pagina,
)
with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
resolved = list(executor.map(_resolve_single, articles))
if verbose:
resolved_count = sum(1 for a in resolved if "news.google.com" not in a.url)
sys.stderr.write(
f"[INFO] ✅ {resolved_count}/{len(resolved)} URLs resolvidas com sucesso para os domínios de origem.\n"
)
return resolved
def _fetch_rss_content(rss_url: str) -> str:
"""
Realiza a requisição ao feed RSS utilizando a biblioteca Foxcape como
motor primário de extração e evasão anti-bot em modo headless.
"""
config = FoxcapeConfig(
headless=True,
humanize=False,
)
try:
result = Foxcape.fetch(
rss_url,
config=config,
timeout_ms=15000,
human_delay=False,
)
if result and result.html:
return str(result.html)
raise RuntimeError("Foxcape não retornou conteúdo para a URL informada.")
except (RuntimeError, OSError, TimeoutError, ValueError) as exc:
# Fallback de contingência caso o ambiente não possua suporte a subprocessos de navegador
try:
headers = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/132.0.0.0 Safari/537.36"
),
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "pt-BR,pt;q=0.9,en-US;q=0.8,en;q=0.7,es;q=0.6",
}
req = urllib.request.Request(rss_url, headers=headers)
with urllib.request.urlopen(req, timeout=10) as response:
return response.read().decode("utf-8", errors="replace")
except (
urllib.error.URLError,
urllib.error.HTTPError,
OSError,
TimeoutError,
) as fallback_exc:
raise RuntimeError(
f"Erro na extração Foxcape: {exc} | Contingência: {fallback_exc}"
) from exc
def extract_google_news(
query: SearchQuery, resolve_urls: bool = True, verbose: bool = True
) -> ExtractionResult:
"""
Orquestra a consulta ao Google News e retorna o resultado consolidado
com URLs resolvidas para os portais de notícias.
"""
encoded_query = urllib.parse.quote_plus(query.clean_keyword)
hl, gl, ceid = get_hl_gl_ceid(query.clean_language, query.clean_locale)
rss_url = f"https://news.google.com/rss/search?q={encoded_query}&hl={hl}&gl={gl}&ceid={ceid}"
if verbose:
sys.stderr.write(
f"[INFO] 🔍 Consultando Google News: '{query.clean_keyword}' "
f"(idioma: {query.clean_language}, locale: {gl}, max_pages: {query.max_pages})...\n"
)
xml_content = _fetch_rss_content(rss_url)
if verbose:
sys.stderr.write(f"[INFO] 📥 Feed RSS recebido ({len(xml_content)} bytes).\n")
articles = parse_google_news_rss(xml_content, max_pages=query.max_pages)
if verbose:
sys.stderr.write(f"[INFO] 📰 {len(articles)} artigos extraídos do feed XML.\n")
if resolve_urls and articles:
articles = resolve_articles_urls(articles, verbose=verbose)
return ExtractionResult(
query=query.clean_keyword,
language=query.clean_language,
locale=gl,
total_paginas=query.max_pages,
total_itens=len(articles),
scraped_at=datetime.now(timezone.utc).isoformat(),
items=articles,
)
def build_parser() -> argparse.ArgumentParser:
"""Cria e configura o parser de argumentos CLI."""
parser = argparse.ArgumentParser(
prog="extract_google_news",
description="Extrator de manchetes do Google News RSS por assunto, idioma e região.",
)
parser.add_argument(
"-q",
"--query",
"--keyword",
dest="query",
required=True,
help="Termo ou expressão de busca (obrigatório).",
)
parser.add_argument(
"-l",
"--lang",
"--language",
dest="lang",
default="pt",
help="Código do idioma (padrão: pt).",
)
parser.add_argument(
"--locale",
"--country",
dest="locale",
default=None,
help="Código do país/região (ex: BR, US, MX, ES).",
)
parser.add_argument(
"-p",
"--max-pages",
dest="max_pages",
type=int,
default=1,
help="Número de páginas (1 a 10, com 10 itens por página; padrão: 1).",
)
parser.add_argument(
"-o",
"--output",
dest="output",
default=None,
help="Caminho do arquivo para salvar a saída JSON.",
)
parser.add_argument(
"--pretty",
action="store_true",
help="Formata a saída JSON com indentação legível.",
)
parser.add_argument(
"--no-resolve-urls",
dest="resolve_urls",
action="store_false",
default=True,
help="Desativa a decodificação para as URLs diretas dos portais de notícias.",
)
parser.add_argument(
"-s",
"--silent",
"--quiet",
dest="silent",
action="store_true",
help="Suprime as mensagens informativas de progresso.",
)
return parser
def main(argv: list[str] | None = None) -> int:
"""Ponto de entrada do script CLI."""
if hasattr(sys.stdout, "reconfigure"):
try:
sys.stdout.reconfigure(encoding="utf-8")
except OSError:
pass
if hasattr(sys.stderr, "reconfigure"):
try:
sys.stderr.reconfigure(encoding="utf-8")
except OSError:
pass
parser = build_parser()
try:
args = parser.parse_args(argv)
query = SearchQuery(
keyword=args.query,
language=args.lang,
locale=args.locale,
max_pages=args.max_pages,
)
except (ValueError, argparse.ArgumentError) as err:
sys.stderr.write(f"Erro de validação: {err}\n")
return 1
except SystemExit as err:
return err.code if isinstance(err.code, int) else 1
verbose = not getattr(args, "silent", False)
try:
result = extract_google_news(
query, resolve_urls=args.resolve_urls, verbose=verbose
)
json_output = json.dumps(
result.to_dict(),
ensure_ascii=False,
indent=2 if args.pretty else None,
)
if args.output:
out_path = Path(args.output)
out_path.parent.mkdir(parents=True, exist_ok=True)
out_path.write_text(json_output + "\n", encoding="utf-8")
if verbose:
sys.stderr.write(
f"[INFO] 💾 Arquivo salvo com sucesso: '{args.output}' ({result.total_itens} notícias).\n"
)
else:
sys.stdout.write(json_output + "\n")
return 0
except RuntimeError as err:
sys.stderr.write(f"Erro na extração: {err}\n")
return 2
except OSError as err:
sys.stderr.write(f"Erro de arquivo/sistema: {err}\n")
return 2
if __name__ == "__main__":
sys.exit(main())