feat(extractor): add Google News headlines extractor with Foxcape headless and URL resolution
- Add standalone CLI script scripts/extract_google_news.py for Google News RSS scraping - Integrate foxcape in headless mode as primary stealth anti-bot engine - Implement parallel article URL resolution using googlenewsdecoder and ThreadPoolExecutor - Support language and regional locale mapping (-l, --lang, --locale) - Implement real-time progress logging in stderr and --silent flag - Add unit, integration, and live E2E tests in tests/test_extract_google_news.py - Add full SpecKit documentation (specs/002-google-news-extractor/) - Create comprehensive README.md covering both NLP Classifier and Google News Extractor
This commit is contained in:
@@ -0,0 +1,355 @@
|
||||
"""
|
||||
Testes unitários e de integração para o Extrator de Manchetes do Google News.
|
||||
|
||||
Cobre validação de entrada, mapeamento de idiomas/locales, parsing e
|
||||
higienização de XML/HTML, orquestração, decodificação de URLs e contrato de execução CLI.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from scripts.extract_google_news import (
|
||||
ExtractionResult,
|
||||
NewsArticle,
|
||||
SearchQuery,
|
||||
extract_google_news,
|
||||
get_hl_gl_ceid,
|
||||
main,
|
||||
parse_google_news_rss,
|
||||
resolve_article_url,
|
||||
resolve_articles_urls,
|
||||
)
|
||||
|
||||
FIXTURE_PATH = Path(__file__).parent / "fixtures" / "google_news_sample.xml"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_rss_xml() -> str:
|
||||
"""Fixture que fornece o conteúdo do XML de exemplo para testes offline."""
|
||||
return FIXTURE_PATH.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def test_get_hl_gl_ceid_default_mappings():
|
||||
"""Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)."""
|
||||
hl, gl, ceid = get_hl_gl_ceid("pt")
|
||||
assert hl == "pt-BR"
|
||||
assert gl == "BR"
|
||||
assert ceid == "BR:pt-BR"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("en")
|
||||
assert hl == "en-US"
|
||||
assert gl == "US"
|
||||
assert ceid == "US:en-US"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("es")
|
||||
assert hl == "es-419"
|
||||
assert gl == "AR"
|
||||
assert ceid == "AR:es-419"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("de")
|
||||
assert hl == "de"
|
||||
assert gl == "DE"
|
||||
assert ceid == "DE:de"
|
||||
|
||||
|
||||
def test_get_hl_gl_ceid_with_custom_locale():
|
||||
"""Valida a sobrescrita geográfica quando o argumento locale é especificado."""
|
||||
hl, gl, ceid = get_hl_gl_ceid("es", locale="MX")
|
||||
assert hl == "es-419"
|
||||
assert gl == "MX"
|
||||
assert ceid == "MX:es-419"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("en", locale="GB")
|
||||
assert hl == "en-GB"
|
||||
assert gl == "GB"
|
||||
assert ceid == "GB:en-GB"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("es", locale="ES")
|
||||
assert hl == "es"
|
||||
assert gl == "ES"
|
||||
assert ceid == "ES:es"
|
||||
|
||||
|
||||
def test_get_hl_gl_ceid_dynamic_fallback():
|
||||
"""Valida fallback dinâmico para idiomas regionais não listados explicitamente."""
|
||||
hl, gl, ceid = get_hl_gl_ceid("ja_jp")
|
||||
assert hl == "ja-JP"
|
||||
assert gl == "JP"
|
||||
assert ceid == "JP:ja-JP"
|
||||
|
||||
|
||||
def test_search_query_validation():
|
||||
"""Valida as regras de negócio e limites de SearchQuery."""
|
||||
# Instanciação válida
|
||||
q = SearchQuery(keyword="inteligencia artificial", language="pt", max_pages=1)
|
||||
assert q.clean_keyword == "inteligencia artificial"
|
||||
assert q.clean_language == "pt"
|
||||
assert q.clean_locale is None
|
||||
assert q.max_pages == 1
|
||||
|
||||
# Palavra-chave vazia ou apenas espaços deve lançar ValueError
|
||||
with pytest.raises(ValueError, match="palavra-chave"):
|
||||
SearchQuery(keyword=" ", language="pt")
|
||||
|
||||
# Idioma com menos de 2 caracteres deve lançar ValueError
|
||||
with pytest.raises(ValueError, match="idioma"):
|
||||
SearchQuery(keyword="test", language="p")
|
||||
|
||||
# Intervalo de páginas fora de 1..10 deve lançar ValueError
|
||||
with pytest.raises(ValueError, match="páginas"):
|
||||
SearchQuery(keyword="test", language="pt", max_pages=0)
|
||||
|
||||
with pytest.raises(ValueError, match="páginas"):
|
||||
SearchQuery(keyword="test", language="pt", max_pages=11)
|
||||
|
||||
|
||||
def test_parse_google_news_rss_with_fixture(sample_rss_xml: str):
|
||||
"""Valida o parsing do feed RSS, higienização de tags HTML e deduplicação."""
|
||||
articles = parse_google_news_rss(sample_rss_xml, max_pages=1)
|
||||
|
||||
# 4 itens no fixture, mas 1 não possui link -> exatamente 3 válidos
|
||||
assert len(articles) == 3
|
||||
|
||||
# Artigo 1: InfoMoney com HTML no description que deve ser limpo
|
||||
art1 = articles[0]
|
||||
assert "InfoMoney" in art1.titulo
|
||||
assert art1.url.startswith("https://news.google.com/rss/articles/")
|
||||
assert art1.quando_publicado == "Thu, 20 Aug 2026 10:30:00 GMT"
|
||||
assert art1.pagina == 1
|
||||
assert "<" not in (art1.subtitulo or "")
|
||||
assert ">" not in (art1.subtitulo or "")
|
||||
assert "crescimento expressivo" in (art1.subtitulo or "")
|
||||
|
||||
# Artigo 2: G1 com parágrafos limpos
|
||||
art2 = articles[1]
|
||||
assert "G1" in art2.titulo
|
||||
assert "<p>" not in (art2.subtitulo or "")
|
||||
|
||||
# Artigo 3: Folha com descrição redundante/igual ao título -> subtitulo deve ser None
|
||||
art3 = articles[2]
|
||||
assert art3.subtitulo is None
|
||||
|
||||
|
||||
def test_resolve_article_url_fallback():
|
||||
"""Valida fallback gracioso de URL quando não é link do Google News ou em erro."""
|
||||
direct_url = "https://www.globo.com/noticia/123"
|
||||
assert resolve_article_url(direct_url) == direct_url
|
||||
|
||||
with patch(
|
||||
"scripts.extract_google_news.gnewsdecoder", return_value={"status": False}
|
||||
):
|
||||
gn_url = "https://news.google.com/rss/articles/fake_token"
|
||||
assert resolve_article_url(gn_url) == gn_url
|
||||
|
||||
|
||||
def test_resolve_article_url_success():
|
||||
"""Valida resolução bem-sucedida de URL do Google News para o portal destino."""
|
||||
gn_url = "https://news.google.com/rss/articles/valid_token"
|
||||
dest_url = "https://infomoney.com.br/mercados/artigo-ia"
|
||||
|
||||
with patch(
|
||||
"scripts.extract_google_news.gnewsdecoder",
|
||||
return_value={"status": True, "decoded_url": dest_url},
|
||||
):
|
||||
resolved = resolve_article_url(gn_url)
|
||||
assert resolved == dest_url
|
||||
|
||||
|
||||
def test_resolve_articles_urls_batch():
|
||||
"""Valida a resolução concorrente em lote de uma lista de NewsArticle."""
|
||||
articles = [
|
||||
NewsArticle(
|
||||
titulo="Notícia 1",
|
||||
url="https://news.google.com/rss/articles/1",
|
||||
pagina=1,
|
||||
),
|
||||
NewsArticle(
|
||||
titulo="Notícia 2",
|
||||
url="https://news.google.com/rss/articles/2",
|
||||
pagina=1,
|
||||
),
|
||||
]
|
||||
|
||||
with patch(
|
||||
"scripts.extract_google_news.resolve_article_url",
|
||||
side_effect=lambda u: f"https://destinofinal.com/{u.split('/')[-1]}",
|
||||
):
|
||||
resolved = resolve_articles_urls(articles)
|
||||
assert len(resolved) == 2
|
||||
assert resolved[0].url == "https://destinofinal.com/1"
|
||||
assert resolved[1].url == "https://destinofinal.com/2"
|
||||
|
||||
|
||||
def test_extract_google_news_orchestration_mocked(sample_rss_xml: str):
|
||||
"""Valida a consolidação do ExtractionResult a partir da busca mockada com URLs resolvidas."""
|
||||
query = SearchQuery(
|
||||
keyword="inteligência artificial", language="pt", locale="BR", max_pages=1
|
||||
)
|
||||
|
||||
with (
|
||||
patch(
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
return_value=sample_rss_xml,
|
||||
),
|
||||
patch(
|
||||
"scripts.extract_google_news.resolve_article_url",
|
||||
side_effect=lambda u: f"https://resolved.com/{u[-5:]}",
|
||||
),
|
||||
):
|
||||
result = extract_google_news(query, resolve_urls=True)
|
||||
|
||||
assert isinstance(result, ExtractionResult)
|
||||
assert result.query == "inteligência artificial"
|
||||
assert result.language == "pt"
|
||||
assert result.locale == "BR"
|
||||
assert result.total_paginas == 1
|
||||
assert result.total_itens == 3
|
||||
assert len(result.items) == 3
|
||||
assert result.scraped_at is not None
|
||||
assert result.items[0].url.startswith("https://resolved.com/")
|
||||
|
||||
|
||||
def test_cli_execution_stdout(sample_rss_xml: str, capsys: pytest.CaptureFixture[str]):
|
||||
"""Valida execução padrão do CLI com saída JSON no stdout."""
|
||||
with (
|
||||
patch(
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
return_value=sample_rss_xml,
|
||||
),
|
||||
patch(
|
||||
"scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u
|
||||
),
|
||||
):
|
||||
exit_code = main(
|
||||
["--query", "inteligencia artificial", "--lang", "pt", "--pretty"]
|
||||
)
|
||||
assert exit_code == 0
|
||||
|
||||
captured = capsys.readouterr()
|
||||
data = json.loads(captured.out)
|
||||
assert data["query"] == "inteligencia artificial"
|
||||
assert data["total_itens"] == 3
|
||||
assert len(data["items"]) == 3
|
||||
# Validar indentação presente por causa de --pretty
|
||||
assert "\n " in captured.out
|
||||
|
||||
|
||||
def test_cli_execution_file_output(sample_rss_xml: str, tmp_path: Path):
|
||||
"""Valida gravação em arquivo com criação automática de diretórios pais."""
|
||||
out_file = tmp_path / "sub_dir" / "news_out.json"
|
||||
|
||||
with (
|
||||
patch(
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
return_value=sample_rss_xml,
|
||||
),
|
||||
patch(
|
||||
"scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u
|
||||
),
|
||||
):
|
||||
exit_code = main(["-q", "IA", "-p", "1", "-o", str(out_file)])
|
||||
assert exit_code == 0
|
||||
|
||||
assert out_file.exists()
|
||||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert data["total_itens"] == 3
|
||||
|
||||
|
||||
def test_cli_empty_query_error(capsys: pytest.CaptureFixture[str]):
|
||||
"""Valida tratamento de erro e código de saída 1 para parâmetro vazio."""
|
||||
exit_code = main(["--query", " "])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
assert "Erro de validação" in captured.err
|
||||
|
||||
|
||||
def test_cli_network_error_handling(capsys: pytest.CaptureFixture[str]):
|
||||
"""Valida tratamento de erro e código de saída 2 para falhas de rede."""
|
||||
with patch(
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
side_effect=RuntimeError("Connection refused"),
|
||||
):
|
||||
exit_code = main(["--query", "IA"])
|
||||
assert exit_code == 2
|
||||
|
||||
captured = capsys.readouterr()
|
||||
assert "Erro na extração" in captured.err
|
||||
assert "Connection refused" in captured.err
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Testes E2E (End-to-End) com resolução real de rede e validação de URLs finais
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_e2e_resolve_real_google_news_url():
|
||||
"""Valida E2E que o decodificador resolve uma URL real do Google News para o veículo de imprensa."""
|
||||
# URL real de artigo extraída do Google News RSS
|
||||
sample_gn_url = (
|
||||
"https://news.google.com/rss/articles/"
|
||||
"CBMi6wFBVV95cUxQUS0tMGxacUJycDlCWXpWSGR3T0hfR1E0Q0txWjdqek9LWjNZTUo1WEx3"
|
||||
"UEw3Skt1eW1Wa2VYUUE2ajNYMmpEckhsdGFXUGtmQktYX2JWa1hmclhEZEVIa3hNODhpVXNO"
|
||||
"UGN4cDhmWmJMczBEUVFDNS1aX3EzRHh6VmQ3cVY0ZnZmaW9YWDZJSE9UNFJ2dXpyNFlHaVVs"
|
||||
"VlY0V1FiZ0tzZ3FpRVhUYnhPbmdnOFRveW5oOVB3WDAzS3c0eWFjMDBZSERwNmRkRk1MRHZF"
|
||||
"UE92UE9GMmpRcFZ5cUU2Ym1NeDdQU2tn?oc=5"
|
||||
)
|
||||
|
||||
resolved_url = resolve_article_url(sample_gn_url)
|
||||
|
||||
# Não deve mais ser URL do Google News
|
||||
assert "news.google.com" not in resolved_url
|
||||
# Deve ser uma URL absoluta http/https apontando para o portal real (TyC Sports)
|
||||
assert resolved_url.startswith("http")
|
||||
assert "tycsports.com" in resolved_url
|
||||
|
||||
|
||||
def test_e2e_extract_google_news_live_pipeline():
|
||||
"""Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo."""
|
||||
query = SearchQuery(keyword="tecnologia", language="pt", locale="BR", max_pages=1)
|
||||
result = extract_google_news(query, resolve_urls=True)
|
||||
|
||||
assert isinstance(result, ExtractionResult)
|
||||
assert result.total_itens > 0
|
||||
assert len(result.items) == result.total_itens
|
||||
|
||||
for article in result.items:
|
||||
assert article.titulo
|
||||
assert article.url.startswith("http")
|
||||
# Garante que as URLs foram decodificadas e não permanecem no formato intermediário
|
||||
assert "news.google.com/rss/articles/" not in article.url
|
||||
|
||||
|
||||
def test_e2e_cli_live_file_output(tmp_path: Path):
|
||||
"""Valida E2E a execução do CLI com saída real em arquivo e URLs decodificadas."""
|
||||
out_file = tmp_path / "e2e_result.json"
|
||||
exit_code = main(
|
||||
[
|
||||
"--query",
|
||||
"economia",
|
||||
"--lang",
|
||||
"pt",
|
||||
"--max-pages",
|
||||
"1",
|
||||
"--output",
|
||||
str(out_file),
|
||||
]
|
||||
)
|
||||
|
||||
assert exit_code == 0
|
||||
assert out_file.exists()
|
||||
|
||||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert data["query"] == "economia"
|
||||
assert data["total_itens"] > 0
|
||||
assert len(data["items"]) > 0
|
||||
|
||||
first_item = data["items"][0]
|
||||
assert first_item["titulo"]
|
||||
assert first_item["url"].startswith("http")
|
||||
assert "news.google.com/rss/articles/" not in first_item["url"]
|
||||
Reference in New Issue
Block a user