feat(extractor): add Google News headlines extractor with Foxcape headless and URL resolution
- Add standalone CLI script scripts/extract_google_news.py for Google News RSS scraping - Integrate foxcape in headless mode as primary stealth anti-bot engine - Implement parallel article URL resolution using googlenewsdecoder and ThreadPoolExecutor - Support language and regional locale mapping (-l, --lang, --locale) - Implement real-time progress logging in stderr and --silent flag - Add unit, integration, and live E2E tests in tests/test_extract_google_news.py - Add full SpecKit documentation (specs/002-google-news-extractor/) - Create comprehensive README.md covering both NLP Classifier and Google News Extractor
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Módulo de scripts utilitários do projeto."""
|
||||
@@ -0,0 +1,467 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Extrator de Manchetes do Google News via RSS.
|
||||
|
||||
Script CLI autônomo para busca, extração e estruturação de notícias
|
||||
por termo/assunto, idioma e localização geográfica (locale), utilizando
|
||||
a biblioteca Foxcape para raspagem indetectável e decodificação automática
|
||||
das URLs intermediárias do Google News para as URLs finais dos veículos.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import concurrent.futures
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from bs4 import BeautifulSoup, FeatureNotFound
|
||||
from foxcape import Foxcape, FoxcapeConfig
|
||||
from googlenewsdecoder import gnewsdecoder # type: ignore[import-untyped]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SearchQuery:
|
||||
"""Value Object com parâmetros de busca validados."""
|
||||
|
||||
keyword: str
|
||||
language: str = "pt"
|
||||
locale: str | None = None
|
||||
max_pages: int = 1
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not self.keyword or not self.keyword.strip():
|
||||
raise ValueError("A palavra-chave não pode ser vazia.")
|
||||
|
||||
if not self.language or len(self.language.strip()) < 2:
|
||||
raise ValueError(
|
||||
"O idioma deve conter pelo menos 2 caracteres (ex: 'pt', 'en', 'es')."
|
||||
)
|
||||
|
||||
if self.max_pages < 1 or self.max_pages > 10:
|
||||
raise ValueError("O número máximo de páginas deve estar entre 1 e 10.")
|
||||
|
||||
@property
|
||||
def clean_keyword(self) -> str:
|
||||
return self.keyword.strip()
|
||||
|
||||
@property
|
||||
def clean_language(self) -> str:
|
||||
return self.language.strip().lower()
|
||||
|
||||
@property
|
||||
def clean_locale(self) -> str | None:
|
||||
return self.locale.strip().upper() if self.locale else None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NewsArticle:
|
||||
"""Entidade que representa uma notícia extraída."""
|
||||
|
||||
titulo: str
|
||||
url: str
|
||||
pagina: int
|
||||
subtitulo: str | None = None
|
||||
quando_publicado: str | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"titulo": self.titulo,
|
||||
"subtitulo": self.subtitulo,
|
||||
"quando_publicado": self.quando_publicado,
|
||||
"url": self.url,
|
||||
"pagina": self.pagina,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ExtractionResult:
|
||||
"""Resultado consolidado da extração."""
|
||||
|
||||
query: str
|
||||
language: str
|
||||
locale: str
|
||||
total_paginas: int
|
||||
total_itens: int
|
||||
scraped_at: str
|
||||
items: list[NewsArticle] = field(default_factory=list)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"query": self.query,
|
||||
"language": self.language,
|
||||
"locale": self.locale,
|
||||
"total_paginas": self.total_paginas,
|
||||
"total_itens": self.total_itens,
|
||||
"scraped_at": self.scraped_at,
|
||||
"items": [item.to_dict() for item in self.items],
|
||||
}
|
||||
|
||||
|
||||
def get_hl_gl_ceid(lang_raw: str, locale: str | None = None) -> tuple[str, str, str]:
|
||||
"""
|
||||
Mapeia idioma e locale para os parâmetros hl, gl e ceid do Google News.
|
||||
"""
|
||||
lang_clean = lang_raw.lower().replace("-", "_")
|
||||
|
||||
locale_map: dict[str, tuple[str, str]] = {
|
||||
"pt": ("pt-BR", "BR"),
|
||||
"pt_br": ("pt-BR", "BR"),
|
||||
"es": ("es-419", "AR"),
|
||||
"es_mx": ("es-419", "MX"),
|
||||
"es_es": ("es", "ES"),
|
||||
"en": ("en-US", "US"),
|
||||
"en_gb": ("en-GB", "GB"),
|
||||
"en_uk": ("en-GB", "GB"),
|
||||
"en_us": ("en-US", "US"),
|
||||
"de": ("de", "DE"),
|
||||
"de_de": ("de", "DE"),
|
||||
"it": ("it", "IT"),
|
||||
"it_it": ("it", "IT"),
|
||||
"fr": ("fr", "FR"),
|
||||
"fr_fr": ("fr", "FR"),
|
||||
}
|
||||
|
||||
if lang_clean in locale_map:
|
||||
hl, gl = locale_map[lang_clean]
|
||||
else:
|
||||
parts = lang_clean.split("_")
|
||||
if len(parts) == 2:
|
||||
hl, gl = f"{parts[0]}-{parts[1].upper()}", parts[1].upper()
|
||||
else:
|
||||
hl, gl = lang_clean, lang_clean.upper()
|
||||
|
||||
if locale:
|
||||
gl = locale.strip().upper()
|
||||
if lang_clean == "es" and gl == "ES":
|
||||
hl = "es"
|
||||
elif lang_clean == "en" and gl == "GB":
|
||||
hl = "en-GB"
|
||||
|
||||
ceid = f"{gl}:{hl}"
|
||||
return hl, gl, ceid
|
||||
|
||||
|
||||
def _normalize_text_for_comparison(text: str) -> str:
|
||||
"""Remove pontuação e espaços extras para comparação de redundância."""
|
||||
return re.sub(r"[^\w\s]", "", text).lower().strip()
|
||||
|
||||
|
||||
def parse_google_news_rss(xml_content: str, max_pages: int = 1) -> list[NewsArticle]:
|
||||
"""
|
||||
Parseia o XML do RSS do Google News e extrai os itens estruturados.
|
||||
"""
|
||||
try:
|
||||
soup = BeautifulSoup(xml_content, "xml")
|
||||
except FeatureNotFound:
|
||||
soup = BeautifulSoup(xml_content, "html.parser")
|
||||
|
||||
rss_items = soup.find_all("item")
|
||||
articles: list[NewsArticle] = []
|
||||
max_allowed = max_pages * 10
|
||||
|
||||
for idx, item in enumerate(rss_items[:max_allowed]):
|
||||
page_number = (idx // 10) + 1
|
||||
|
||||
title_tag = item.find("title")
|
||||
link_tag = item.find("link")
|
||||
pubdate_tag = item.find("pubDate")
|
||||
desc_tag = item.find("description")
|
||||
|
||||
title = title_tag.get_text(strip=True) if title_tag else ""
|
||||
link = link_tag.get_text(strip=True) if link_tag else ""
|
||||
pub_date = pubdate_tag.get_text(strip=True) if pubdate_tag else None
|
||||
desc_raw = desc_tag.get_text() if desc_tag else ""
|
||||
|
||||
# Limpeza de HTML do resumo / snippet
|
||||
snippet = None
|
||||
if desc_raw:
|
||||
desc_soup = BeautifulSoup(desc_raw, "html.parser")
|
||||
clean_snippet = desc_soup.get_text(separator=" ", strip=True)
|
||||
if clean_snippet:
|
||||
norm_snippet = _normalize_text_for_comparison(clean_snippet)
|
||||
norm_title = _normalize_text_for_comparison(title)
|
||||
if norm_snippet != norm_title:
|
||||
snippet = clean_snippet
|
||||
|
||||
if title and link:
|
||||
articles.append(
|
||||
NewsArticle(
|
||||
titulo=title,
|
||||
subtitulo=snippet,
|
||||
quando_publicado=pub_date,
|
||||
url=link,
|
||||
pagina=page_number,
|
||||
)
|
||||
)
|
||||
|
||||
return articles
|
||||
|
||||
|
||||
def resolve_article_url(url: str) -> str:
|
||||
"""Resolve a URL intermediária do Google News para a URL real do veículo."""
|
||||
if not url or "news.google.com" not in url:
|
||||
return url
|
||||
try:
|
||||
res = gnewsdecoder(url)
|
||||
if isinstance(res, dict) and res.get("status") and res.get("decoded_url"):
|
||||
return str(res["decoded_url"])
|
||||
except (ValueError, OSError, RuntimeError, TypeError):
|
||||
pass
|
||||
return url
|
||||
|
||||
|
||||
def resolve_articles_urls(
|
||||
articles: list[NewsArticle], max_workers: int = 5, verbose: bool = True
|
||||
) -> list[NewsArticle]:
|
||||
"""Resolve em paralelo as URLs intermediárias do Google News para os links finais dos veículos."""
|
||||
if not articles:
|
||||
return articles
|
||||
|
||||
if verbose:
|
||||
sys.stderr.write(
|
||||
f"[INFO] 🔗 Decodificando {len(articles)} URLs do Google News para os portais reais...\n"
|
||||
)
|
||||
|
||||
def _resolve_single(art: NewsArticle) -> NewsArticle:
|
||||
return NewsArticle(
|
||||
titulo=art.titulo,
|
||||
subtitulo=art.subtitulo,
|
||||
quando_publicado=art.quando_publicado,
|
||||
url=resolve_article_url(art.url),
|
||||
pagina=art.pagina,
|
||||
)
|
||||
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
|
||||
resolved = list(executor.map(_resolve_single, articles))
|
||||
|
||||
if verbose:
|
||||
resolved_count = sum(1 for a in resolved if "news.google.com" not in a.url)
|
||||
sys.stderr.write(
|
||||
f"[INFO] ✅ {resolved_count}/{len(resolved)} URLs resolvidas com sucesso para os domínios de origem.\n"
|
||||
)
|
||||
|
||||
return resolved
|
||||
|
||||
|
||||
def _fetch_rss_content(rss_url: str) -> str:
|
||||
"""
|
||||
Realiza a requisição ao feed RSS utilizando a biblioteca Foxcape como
|
||||
motor primário de extração e evasão anti-bot em modo headless.
|
||||
"""
|
||||
config = FoxcapeConfig(
|
||||
headless=True,
|
||||
humanize=False,
|
||||
)
|
||||
try:
|
||||
result = Foxcape.fetch(
|
||||
rss_url,
|
||||
config=config,
|
||||
timeout_ms=15000,
|
||||
human_delay=False,
|
||||
)
|
||||
if result and result.html:
|
||||
return str(result.html)
|
||||
raise RuntimeError("Foxcape não retornou conteúdo para a URL informada.")
|
||||
except (RuntimeError, OSError, TimeoutError, ValueError) as exc:
|
||||
# Fallback de contingência caso o ambiente não possua suporte a subprocessos de navegador
|
||||
try:
|
||||
headers = {
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/132.0.0.0 Safari/537.36"
|
||||
),
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "pt-BR,pt;q=0.9,en-US;q=0.8,en;q=0.7,es;q=0.6",
|
||||
}
|
||||
req = urllib.request.Request(rss_url, headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=10) as response:
|
||||
return response.read().decode("utf-8", errors="replace")
|
||||
except (
|
||||
urllib.error.URLError,
|
||||
urllib.error.HTTPError,
|
||||
OSError,
|
||||
TimeoutError,
|
||||
) as fallback_exc:
|
||||
raise RuntimeError(
|
||||
f"Erro na extração Foxcape: {exc} | Contingência: {fallback_exc}"
|
||||
) from exc
|
||||
|
||||
|
||||
def extract_google_news(
|
||||
query: SearchQuery, resolve_urls: bool = True, verbose: bool = True
|
||||
) -> ExtractionResult:
|
||||
"""
|
||||
Orquestra a consulta ao Google News e retorna o resultado consolidado
|
||||
com URLs resolvidas para os portais de notícias.
|
||||
"""
|
||||
encoded_query = urllib.parse.quote_plus(query.clean_keyword)
|
||||
hl, gl, ceid = get_hl_gl_ceid(query.clean_language, query.clean_locale)
|
||||
rss_url = f"https://news.google.com/rss/search?q={encoded_query}&hl={hl}&gl={gl}&ceid={ceid}"
|
||||
|
||||
if verbose:
|
||||
sys.stderr.write(
|
||||
f"[INFO] 🔍 Consultando Google News: '{query.clean_keyword}' "
|
||||
f"(idioma: {query.clean_language}, locale: {gl}, max_pages: {query.max_pages})...\n"
|
||||
)
|
||||
|
||||
xml_content = _fetch_rss_content(rss_url)
|
||||
if verbose:
|
||||
sys.stderr.write(f"[INFO] 📥 Feed RSS recebido ({len(xml_content)} bytes).\n")
|
||||
|
||||
articles = parse_google_news_rss(xml_content, max_pages=query.max_pages)
|
||||
if verbose:
|
||||
sys.stderr.write(f"[INFO] 📰 {len(articles)} artigos extraídos do feed XML.\n")
|
||||
|
||||
if resolve_urls and articles:
|
||||
articles = resolve_articles_urls(articles, verbose=verbose)
|
||||
|
||||
return ExtractionResult(
|
||||
query=query.clean_keyword,
|
||||
language=query.clean_language,
|
||||
locale=gl,
|
||||
total_paginas=query.max_pages,
|
||||
total_itens=len(articles),
|
||||
scraped_at=datetime.now(timezone.utc).isoformat(),
|
||||
items=articles,
|
||||
)
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
"""Cria e configura o parser de argumentos CLI."""
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="extract_google_news",
|
||||
description="Extrator de manchetes do Google News RSS por assunto, idioma e região.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-q",
|
||||
"--query",
|
||||
"--keyword",
|
||||
dest="query",
|
||||
required=True,
|
||||
help="Termo ou expressão de busca (obrigatório).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-l",
|
||||
"--lang",
|
||||
"--language",
|
||||
dest="lang",
|
||||
default="pt",
|
||||
help="Código do idioma (padrão: pt).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--locale",
|
||||
"--country",
|
||||
dest="locale",
|
||||
default=None,
|
||||
help="Código do país/região (ex: BR, US, MX, ES).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-p",
|
||||
"--max-pages",
|
||||
dest="max_pages",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Número de páginas (1 a 10, com 10 itens por página; padrão: 1).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output",
|
||||
dest="output",
|
||||
default=None,
|
||||
help="Caminho do arquivo para salvar a saída JSON.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--pretty",
|
||||
action="store_true",
|
||||
help="Formata a saída JSON com indentação legível.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--no-resolve-urls",
|
||||
dest="resolve_urls",
|
||||
action="store_false",
|
||||
default=True,
|
||||
help="Desativa a decodificação para as URLs diretas dos portais de notícias.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-s",
|
||||
"--silent",
|
||||
"--quiet",
|
||||
dest="silent",
|
||||
action="store_true",
|
||||
help="Suprime as mensagens informativas de progresso.",
|
||||
)
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
"""Ponto de entrada do script CLI."""
|
||||
if hasattr(sys.stdout, "reconfigure"):
|
||||
try:
|
||||
sys.stdout.reconfigure(encoding="utf-8")
|
||||
except OSError:
|
||||
pass
|
||||
if hasattr(sys.stderr, "reconfigure"):
|
||||
try:
|
||||
sys.stderr.reconfigure(encoding="utf-8")
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
parser = build_parser()
|
||||
|
||||
try:
|
||||
args = parser.parse_args(argv)
|
||||
query = SearchQuery(
|
||||
keyword=args.query,
|
||||
language=args.lang,
|
||||
locale=args.locale,
|
||||
max_pages=args.max_pages,
|
||||
)
|
||||
except (ValueError, argparse.ArgumentError) as err:
|
||||
sys.stderr.write(f"Erro de validação: {err}\n")
|
||||
return 1
|
||||
except SystemExit as err:
|
||||
return err.code if isinstance(err.code, int) else 1
|
||||
|
||||
verbose = not getattr(args, "silent", False)
|
||||
|
||||
try:
|
||||
result = extract_google_news(
|
||||
query, resolve_urls=args.resolve_urls, verbose=verbose
|
||||
)
|
||||
json_output = json.dumps(
|
||||
result.to_dict(),
|
||||
ensure_ascii=False,
|
||||
indent=2 if args.pretty else None,
|
||||
)
|
||||
|
||||
if args.output:
|
||||
out_path = Path(args.output)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
out_path.write_text(json_output + "\n", encoding="utf-8")
|
||||
if verbose:
|
||||
sys.stderr.write(
|
||||
f"[INFO] 💾 Arquivo salvo com sucesso: '{args.output}' ({result.total_itens} notícias).\n"
|
||||
)
|
||||
else:
|
||||
sys.stdout.write(json_output + "\n")
|
||||
|
||||
return 0
|
||||
except RuntimeError as err:
|
||||
sys.stderr.write(f"Erro na extração: {err}\n")
|
||||
return 2
|
||||
except OSError as err:
|
||||
sys.stderr.write(f"Erro de arquivo/sistema: {err}\n")
|
||||
return 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user