Files
TextNLPClassifierApp/tests/tools/test_extract_article_contents.py
T

373 lines
13 KiB
Python

#!/usr/bin/env python3
"""
Testes automatizados para o Extrator e Parser Multimotor de Artigos.
Cobre modelos de dados, parsers (Trafilatura, Newspaper4k, Readability),
isolamento de falhas, orquestração de lote e interface CLI.
"""
from __future__ import annotations
import json
from pathlib import Path
from unittest.mock import patch
import pytest
from scripts.extract_article_contents import (
ArticleCrawler,
ExtractedArticle,
ExtractionBatchReport,
InputArticle,
NewspaperData,
NewspaperExtractor,
ReadabilityData,
ReadabilityExtractor,
TrafilaturaData,
TrafilaturaExtractor,
extract_all_engines,
load_search_json,
main,
process_batch,
save_extracted_json,
)
SAMPLE_HTML = """
<!DOCTYPE html>
<html lang="es">
<head>
<meta charset="utf-8">
<title>River Plate igualó sin goles ante Independiente Santa Fe - Olé</title>
<meta name="description" content="El equipo de Núñez empató 0-0 en Bogotá por los octavos de final.">
<meta name="author" content="Juan Pérez">
<meta property="og:title" content="River Plate igualó sin goles ante Independiente Santa Fe">
<meta property="og:image" content="https://media.ole.com.ar/river.jpg">
</head>
<body>
<header><nav><a href="/">Inicio</a></nav></header>
<article>
<h1>River Plate igualó sin goles ante Independiente Santa Fe</h1>
<p class="byline">Por Juan Pérez - 20 de Agosto de 2026</p>
<p class="lead">El equipo de Núñez empató 0-0 en Bogotá por la Copa Sudamericana.</p>
<p>Franco Armani fue la gran figura del encuentro con tres atajadas espectaculares en el primer tiempo.</p>
<p>El partido de vuelta se disputará en el estadio Monumental la próxima semana ante una multitud.</p>
</article>
<footer><p>Copyright 2026 Olé</p></footer>
</body>
</html>
"""
@pytest.fixture
def sample_input_json(tmp_path: Path) -> Path:
data = {
"query": "River Plate",
"language": "es",
"locale": "AR",
"total_itens": 2,
"items": [
{
"titulo": "River Plate igualó sin goles ante Santa Fe",
"subtitulo": "Empate en Bogotá",
"quando_publicado": "Thu, 20 Aug 2026 03:27:26 GMT",
"url": "https://www.ole.com.ar/river-0-0-santa-fe.html",
"pagina": 1,
},
{
"titulo": "Armani fue la figura de River",
"subtitulo": "Gran actuación del arquero",
"quando_publicado": "Thu, 20 Aug 2026 04:00:00 GMT",
"url": "https://www.tycsports.com/armani-figura.html",
"pagina": 1,
},
],
}
input_file = tmp_path / "river_plate.json"
input_file.write_text(json.dumps(data, ensure_ascii=False), encoding="utf-8")
return input_file
# ==============================================================================
# 1. Testes de Modelos e I/O de JSON
# ==============================================================================
def test_input_article_creation():
article = InputArticle(
titulo="Notícia Teste",
url="https://example.com/noticia",
subtitulo="Subtítulo",
quando_publicado="Thu, 20 Aug 2026",
pagina=1,
)
assert article.titulo == "Notícia Teste"
assert article.url == "https://example.com/noticia"
assert article.pagina == 1
d = article.to_dict()
assert d["titulo"] == "Notícia Teste"
assert d["url"] == "https://example.com/noticia"
def test_load_search_json_valid(sample_input_json: Path):
query, lang, items = load_search_json(sample_input_json)
assert query == "River Plate"
assert lang == "es"
assert len(items) == 2
assert items[0].url == "https://www.ole.com.ar/river-0-0-santa-fe.html"
def test_load_search_json_invalid_file(tmp_path: Path):
non_existent = tmp_path / "missing.json"
with pytest.raises(FileNotFoundError):
load_search_json(non_existent)
def test_save_extracted_json(tmp_path: Path):
report = ExtractionBatchReport(
source_file="test.json",
processed_at="2026-08-20T12:00:00Z",
total_articles=1,
successful_articles=1,
failed_articles=0,
articles=[
ExtractedArticle(
input_meta=InputArticle(titulo="Teste", url="https://example.com/noticia"),
extraction_status="success",
error_message=None,
crawled_url="https://example.com/noticia",
page_title="Página Teste",
http_status=200,
trafilatura=TrafilaturaData(
title="Teste",
author="Autor",
date="2026-08-20",
description="Desc",
categories=[],
tags=[],
canonical_url=None,
text="Texto do teste.",
raw_json=None,
error=None,
),
newspaper4k=NewspaperData(
title="Teste",
authors=["Autor"],
publish_date="2026-08-20",
text="Texto do teste.",
summary="Resumo",
keywords=["teste"],
top_image=None,
images=[],
meta_data={},
error=None,
),
readability=ReadabilityData(
title="Teste",
short_title="Teste",
cleaned_html="<p>Texto do teste.</p>",
cleaned_text="Texto do teste.",
error=None,
),
)
],
)
out_file = tmp_path / "out" / "result.json"
save_extracted_json(report, out_file)
assert out_file.exists()
content = json.loads(out_file.read_text(encoding="utf-8"))
assert content["total_articles"] == 1
assert content["articles"][0]["trafilatura"]["title"] == "Teste"
# ==============================================================================
# 2. Testes Unitários dos Parsers (Trafilatura, Newspaper4k, Readability)
# ==============================================================================
def test_trafilatura_extractor():
extractor = TrafilaturaExtractor()
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia")
assert isinstance(res, TrafilaturaData)
assert res.error is None
assert "Armani" in res.text or "River Plate" in res.text
assert res.title is not None
def test_newspaper_extractor():
extractor = NewspaperExtractor()
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es")
assert isinstance(res, NewspaperData)
assert res.error is None
assert "Armani" in res.text or "River" in res.text
assert isinstance(res.keywords, list)
assert len(res.keywords) > 0
def test_readability_extractor():
extractor = ReadabilityExtractor()
res = extractor.extract(SAMPLE_HTML)
assert isinstance(res, ReadabilityData)
assert res.error is None
assert res.title is not None
assert res.cleaned_html is not None
assert "Armani" in (res.cleaned_text or "") or "River" in (res.cleaned_text or "")
def test_extract_all_engines():
traf, news, read = extract_all_engines(
SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es"
)
assert traf.error is None
assert news.error is None
assert read.error is None
# ==============================================================================
# 3. Testes de Isolamento de Falhas (Resiliência)
# ==============================================================================
def test_extractor_error_isolation_on_faulty_engine():
with patch(
"scripts.extract_article_contents.TrafilaturaExtractor.extract",
side_effect=RuntimeError("Trafilatura crash"),
):
traf, news, read = extract_all_engines(
SAMPLE_HTML, url="https://example.com", language="es"
)
assert traf.error == "Trafilatura crash"
assert news.error is None
assert read.error is None
def test_crawler_error_isolation(sample_input_json: Path, tmp_path: Path):
out_file = tmp_path / "river_plate_extracted.json"
def mock_crawl(url, timeout_sec=30):
if "ole.com.ar" in url:
return SAMPLE_HTML, "River Plate Olé", 200
raise ConnectionError("Connection refused by tycsports.com")
with patch.object(ArticleCrawler, "crawl", side_effect=mock_crawl):
with patch.object(ArticleCrawler, "start"), patch.object(ArticleCrawler, "close"):
report = process_batch(
input_path=sample_input_json,
output_path=out_file,
silent=True,
)
assert report.total_articles == 2
assert report.successful_articles == 1
assert report.failed_articles == 1
assert report.articles[0].extraction_status == "success"
assert report.articles[1].extraction_status == "failed"
assert "Connection refused" in (report.articles[1].error_message or "")
# ==============================================================================
# 4. Testes de CLI e Limitação (--limit, --silent, --language)
# ==============================================================================
def test_process_batch_with_limit(sample_input_json: Path, tmp_path: Path):
out_file = tmp_path / "limit_extracted.json"
with (
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
patch.object(ArticleCrawler, "start"),
patch.object(ArticleCrawler, "close"),
):
report = process_batch(
input_path=sample_input_json,
output_path=out_file,
limit=1,
silent=True,
)
assert report.total_articles == 1
assert len(report.articles) == 1
assert out_file.exists()
def test_cli_main_success(sample_input_json: Path, tmp_path: Path, capsys):
out_file = tmp_path / "cli_out.json"
with (
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
patch.object(ArticleCrawler, "start"),
patch.object(ArticleCrawler, "close"),
):
exit_code = main(["-i", str(sample_input_json), "-o", str(out_file), "--limit", "1", "-s"])
assert exit_code == 0
assert out_file.exists()
def test_cli_main_missing_input_file(tmp_path: Path, capsys):
missing_file = tmp_path / "does_not_exist.json"
exit_code = main(["-i", str(missing_file), "-s"])
assert exit_code == 1
# ==============================================================================
# 5. Testes End-to-End (E2E) ao Vivo (Live Network)
# ==============================================================================
def test_e2e_live_article_extraction(tmp_path: Path):
"""Valida E2E a extração real ao vivo com Foxcape e os 3 motores em lote."""
live_input_file = Path("out/river_plate.json")
if not live_input_file.exists():
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E ao vivo.")
out_file = tmp_path / "e2e_live_extracted.json"
# Executa o batch real com limite de 1 notícia
report = process_batch(
input_path=live_input_file,
output_path=out_file,
limit=1,
silent=True,
)
assert report.total_articles == 1
assert report.successful_articles == 1
assert report.failed_articles == 0
assert len(report.articles) == 1
art = report.articles[0]
assert art.extraction_status == "success"
assert art.crawled_url.startswith("http")
# Valida que todos os 3 motores extraíram dados reais
assert art.trafilatura is not None and art.trafilatura.error is None
assert len(art.trafilatura.text) > 50
assert art.newspaper4k is not None and art.newspaper4k.error is None
assert len(art.newspaper4k.text) > 50
assert isinstance(art.newspaper4k.keywords, list)
assert art.readability is not None and art.readability.error is None
assert art.readability.cleaned_html is not None
assert len(art.readability.cleaned_text or "") > 50
# Valida arquivo JSON gravado
assert out_file.exists()
saved = json.loads(out_file.read_text(encoding="utf-8"))
assert saved["total_articles"] == 1
assert saved["articles"][0]["extraction_status"] == "success"
def test_e2e_cli_live_execution(tmp_path: Path):
"""Valida E2E a execução do CLI real de ponta a ponta."""
live_input_file = Path("out/river_plate.json")
if not live_input_file.exists():
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E.")
out_file = tmp_path / "e2e_cli_live.json"
exit_code = main(["-i", str(live_input_file), "-o", str(out_file), "--limit", "1", "-s"])
assert exit_code == 0
assert out_file.exists()
data = json.loads(out_file.read_text(encoding="utf-8"))
assert data["successful_articles"] == 1