feat(runtime): implement single-article consolidation runtime and modularize codebase
This commit is contained in:
@@ -0,0 +1,372 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Testes automatizados para o Extrator e Parser Multimotor de Artigos.
|
||||
Cobre modelos de dados, parsers (Trafilatura, Newspaper4k, Readability),
|
||||
isolamento de falhas, orquestração de lote e interface CLI.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from scripts.extract_article_contents import (
|
||||
ArticleCrawler,
|
||||
ExtractedArticle,
|
||||
ExtractionBatchReport,
|
||||
InputArticle,
|
||||
NewspaperData,
|
||||
NewspaperExtractor,
|
||||
ReadabilityData,
|
||||
ReadabilityExtractor,
|
||||
TrafilaturaData,
|
||||
TrafilaturaExtractor,
|
||||
extract_all_engines,
|
||||
load_search_json,
|
||||
main,
|
||||
process_batch,
|
||||
save_extracted_json,
|
||||
)
|
||||
|
||||
SAMPLE_HTML = """
|
||||
<!DOCTYPE html>
|
||||
<html lang="es">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>River Plate igualó sin goles ante Independiente Santa Fe - Olé</title>
|
||||
<meta name="description" content="El equipo de Núñez empató 0-0 en Bogotá por los octavos de final.">
|
||||
<meta name="author" content="Juan Pérez">
|
||||
<meta property="og:title" content="River Plate igualó sin goles ante Independiente Santa Fe">
|
||||
<meta property="og:image" content="https://media.ole.com.ar/river.jpg">
|
||||
</head>
|
||||
<body>
|
||||
<header><nav><a href="/">Inicio</a></nav></header>
|
||||
<article>
|
||||
<h1>River Plate igualó sin goles ante Independiente Santa Fe</h1>
|
||||
<p class="byline">Por Juan Pérez - 20 de Agosto de 2026</p>
|
||||
<p class="lead">El equipo de Núñez empató 0-0 en Bogotá por la Copa Sudamericana.</p>
|
||||
<p>Franco Armani fue la gran figura del encuentro con tres atajadas espectaculares en el primer tiempo.</p>
|
||||
<p>El partido de vuelta se disputará en el estadio Monumental la próxima semana ante una multitud.</p>
|
||||
</article>
|
||||
<footer><p>Copyright 2026 Olé</p></footer>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_input_json(tmp_path: Path) -> Path:
|
||||
data = {
|
||||
"query": "River Plate",
|
||||
"language": "es",
|
||||
"locale": "AR",
|
||||
"total_itens": 2,
|
||||
"items": [
|
||||
{
|
||||
"titulo": "River Plate igualó sin goles ante Santa Fe",
|
||||
"subtitulo": "Empate en Bogotá",
|
||||
"quando_publicado": "Thu, 20 Aug 2026 03:27:26 GMT",
|
||||
"url": "https://www.ole.com.ar/river-0-0-santa-fe.html",
|
||||
"pagina": 1,
|
||||
},
|
||||
{
|
||||
"titulo": "Armani fue la figura de River",
|
||||
"subtitulo": "Gran actuación del arquero",
|
||||
"quando_publicado": "Thu, 20 Aug 2026 04:00:00 GMT",
|
||||
"url": "https://www.tycsports.com/armani-figura.html",
|
||||
"pagina": 1,
|
||||
},
|
||||
],
|
||||
}
|
||||
input_file = tmp_path / "river_plate.json"
|
||||
input_file.write_text(json.dumps(data, ensure_ascii=False), encoding="utf-8")
|
||||
return input_file
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 1. Testes de Modelos e I/O de JSON
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_input_article_creation():
|
||||
article = InputArticle(
|
||||
titulo="Notícia Teste",
|
||||
url="https://example.com/noticia",
|
||||
subtitulo="Subtítulo",
|
||||
quando_publicado="Thu, 20 Aug 2026",
|
||||
pagina=1,
|
||||
)
|
||||
assert article.titulo == "Notícia Teste"
|
||||
assert article.url == "https://example.com/noticia"
|
||||
assert article.pagina == 1
|
||||
d = article.to_dict()
|
||||
assert d["titulo"] == "Notícia Teste"
|
||||
assert d["url"] == "https://example.com/noticia"
|
||||
|
||||
|
||||
def test_load_search_json_valid(sample_input_json: Path):
|
||||
query, lang, items = load_search_json(sample_input_json)
|
||||
assert query == "River Plate"
|
||||
assert lang == "es"
|
||||
assert len(items) == 2
|
||||
assert items[0].url == "https://www.ole.com.ar/river-0-0-santa-fe.html"
|
||||
|
||||
|
||||
def test_load_search_json_invalid_file(tmp_path: Path):
|
||||
non_existent = tmp_path / "missing.json"
|
||||
with pytest.raises(FileNotFoundError):
|
||||
load_search_json(non_existent)
|
||||
|
||||
|
||||
def test_save_extracted_json(tmp_path: Path):
|
||||
report = ExtractionBatchReport(
|
||||
source_file="test.json",
|
||||
processed_at="2026-08-20T12:00:00Z",
|
||||
total_articles=1,
|
||||
successful_articles=1,
|
||||
failed_articles=0,
|
||||
articles=[
|
||||
ExtractedArticle(
|
||||
input_meta=InputArticle(titulo="Teste", url="https://example.com/noticia"),
|
||||
extraction_status="success",
|
||||
error_message=None,
|
||||
crawled_url="https://example.com/noticia",
|
||||
page_title="Página Teste",
|
||||
http_status=200,
|
||||
trafilatura=TrafilaturaData(
|
||||
title="Teste",
|
||||
author="Autor",
|
||||
date="2026-08-20",
|
||||
description="Desc",
|
||||
categories=[],
|
||||
tags=[],
|
||||
canonical_url=None,
|
||||
text="Texto do teste.",
|
||||
raw_json=None,
|
||||
error=None,
|
||||
),
|
||||
newspaper4k=NewspaperData(
|
||||
title="Teste",
|
||||
authors=["Autor"],
|
||||
publish_date="2026-08-20",
|
||||
text="Texto do teste.",
|
||||
summary="Resumo",
|
||||
keywords=["teste"],
|
||||
top_image=None,
|
||||
images=[],
|
||||
meta_data={},
|
||||
error=None,
|
||||
),
|
||||
readability=ReadabilityData(
|
||||
title="Teste",
|
||||
short_title="Teste",
|
||||
cleaned_html="<p>Texto do teste.</p>",
|
||||
cleaned_text="Texto do teste.",
|
||||
error=None,
|
||||
),
|
||||
)
|
||||
],
|
||||
)
|
||||
out_file = tmp_path / "out" / "result.json"
|
||||
save_extracted_json(report, out_file)
|
||||
assert out_file.exists()
|
||||
content = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert content["total_articles"] == 1
|
||||
assert content["articles"][0]["trafilatura"]["title"] == "Teste"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 2. Testes Unitários dos Parsers (Trafilatura, Newspaper4k, Readability)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_trafilatura_extractor():
|
||||
extractor = TrafilaturaExtractor()
|
||||
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia")
|
||||
assert isinstance(res, TrafilaturaData)
|
||||
assert res.error is None
|
||||
assert "Armani" in res.text or "River Plate" in res.text
|
||||
assert res.title is not None
|
||||
|
||||
|
||||
def test_newspaper_extractor():
|
||||
extractor = NewspaperExtractor()
|
||||
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es")
|
||||
assert isinstance(res, NewspaperData)
|
||||
assert res.error is None
|
||||
assert "Armani" in res.text or "River" in res.text
|
||||
assert isinstance(res.keywords, list)
|
||||
assert len(res.keywords) > 0
|
||||
|
||||
|
||||
def test_readability_extractor():
|
||||
extractor = ReadabilityExtractor()
|
||||
res = extractor.extract(SAMPLE_HTML)
|
||||
assert isinstance(res, ReadabilityData)
|
||||
assert res.error is None
|
||||
assert res.title is not None
|
||||
assert res.cleaned_html is not None
|
||||
assert "Armani" in (res.cleaned_text or "") or "River" in (res.cleaned_text or "")
|
||||
|
||||
|
||||
def test_extract_all_engines():
|
||||
traf, news, read = extract_all_engines(
|
||||
SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es"
|
||||
)
|
||||
assert traf.error is None
|
||||
assert news.error is None
|
||||
assert read.error is None
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 3. Testes de Isolamento de Falhas (Resiliência)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_extractor_error_isolation_on_faulty_engine():
|
||||
with patch(
|
||||
"scripts.extract_article_contents.TrafilaturaExtractor.extract",
|
||||
side_effect=RuntimeError("Trafilatura crash"),
|
||||
):
|
||||
traf, news, read = extract_all_engines(
|
||||
SAMPLE_HTML, url="https://example.com", language="es"
|
||||
)
|
||||
assert traf.error == "Trafilatura crash"
|
||||
assert news.error is None
|
||||
assert read.error is None
|
||||
|
||||
|
||||
def test_crawler_error_isolation(sample_input_json: Path, tmp_path: Path):
|
||||
out_file = tmp_path / "river_plate_extracted.json"
|
||||
|
||||
def mock_crawl(url, timeout_sec=30):
|
||||
if "ole.com.ar" in url:
|
||||
return SAMPLE_HTML, "River Plate Olé", 200
|
||||
raise ConnectionError("Connection refused by tycsports.com")
|
||||
|
||||
with patch.object(ArticleCrawler, "crawl", side_effect=mock_crawl):
|
||||
with patch.object(ArticleCrawler, "start"), patch.object(ArticleCrawler, "close"):
|
||||
report = process_batch(
|
||||
input_path=sample_input_json,
|
||||
output_path=out_file,
|
||||
silent=True,
|
||||
)
|
||||
|
||||
assert report.total_articles == 2
|
||||
assert report.successful_articles == 1
|
||||
assert report.failed_articles == 1
|
||||
assert report.articles[0].extraction_status == "success"
|
||||
assert report.articles[1].extraction_status == "failed"
|
||||
assert "Connection refused" in (report.articles[1].error_message or "")
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 4. Testes de CLI e Limitação (--limit, --silent, --language)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_process_batch_with_limit(sample_input_json: Path, tmp_path: Path):
|
||||
out_file = tmp_path / "limit_extracted.json"
|
||||
|
||||
with (
|
||||
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
|
||||
patch.object(ArticleCrawler, "start"),
|
||||
patch.object(ArticleCrawler, "close"),
|
||||
):
|
||||
report = process_batch(
|
||||
input_path=sample_input_json,
|
||||
output_path=out_file,
|
||||
limit=1,
|
||||
silent=True,
|
||||
)
|
||||
|
||||
assert report.total_articles == 1
|
||||
assert len(report.articles) == 1
|
||||
assert out_file.exists()
|
||||
|
||||
|
||||
def test_cli_main_success(sample_input_json: Path, tmp_path: Path, capsys):
|
||||
out_file = tmp_path / "cli_out.json"
|
||||
|
||||
with (
|
||||
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
|
||||
patch.object(ArticleCrawler, "start"),
|
||||
patch.object(ArticleCrawler, "close"),
|
||||
):
|
||||
exit_code = main(["-i", str(sample_input_json), "-o", str(out_file), "--limit", "1", "-s"])
|
||||
|
||||
assert exit_code == 0
|
||||
assert out_file.exists()
|
||||
|
||||
|
||||
def test_cli_main_missing_input_file(tmp_path: Path, capsys):
|
||||
missing_file = tmp_path / "does_not_exist.json"
|
||||
exit_code = main(["-i", str(missing_file), "-s"])
|
||||
assert exit_code == 1
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 5. Testes End-to-End (E2E) ao Vivo (Live Network)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_e2e_live_article_extraction(tmp_path: Path):
|
||||
"""Valida E2E a extração real ao vivo com Foxcape e os 3 motores em lote."""
|
||||
live_input_file = Path("out/river_plate.json")
|
||||
if not live_input_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E ao vivo.")
|
||||
|
||||
out_file = tmp_path / "e2e_live_extracted.json"
|
||||
|
||||
# Executa o batch real com limite de 1 notícia
|
||||
report = process_batch(
|
||||
input_path=live_input_file,
|
||||
output_path=out_file,
|
||||
limit=1,
|
||||
silent=True,
|
||||
)
|
||||
|
||||
assert report.total_articles == 1
|
||||
assert report.successful_articles == 1
|
||||
assert report.failed_articles == 0
|
||||
assert len(report.articles) == 1
|
||||
|
||||
art = report.articles[0]
|
||||
assert art.extraction_status == "success"
|
||||
assert art.crawled_url.startswith("http")
|
||||
|
||||
# Valida que todos os 3 motores extraíram dados reais
|
||||
assert art.trafilatura is not None and art.trafilatura.error is None
|
||||
assert len(art.trafilatura.text) > 50
|
||||
|
||||
assert art.newspaper4k is not None and art.newspaper4k.error is None
|
||||
assert len(art.newspaper4k.text) > 50
|
||||
assert isinstance(art.newspaper4k.keywords, list)
|
||||
|
||||
assert art.readability is not None and art.readability.error is None
|
||||
assert art.readability.cleaned_html is not None
|
||||
assert len(art.readability.cleaned_text or "") > 50
|
||||
|
||||
# Valida arquivo JSON gravado
|
||||
assert out_file.exists()
|
||||
saved = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert saved["total_articles"] == 1
|
||||
assert saved["articles"][0]["extraction_status"] == "success"
|
||||
|
||||
|
||||
def test_e2e_cli_live_execution(tmp_path: Path):
|
||||
"""Valida E2E a execução do CLI real de ponta a ponta."""
|
||||
live_input_file = Path("out/river_plate.json")
|
||||
if not live_input_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E.")
|
||||
|
||||
out_file = tmp_path / "e2e_cli_live.json"
|
||||
exit_code = main(["-i", str(live_input_file), "-o", str(out_file), "--limit", "1", "-s"])
|
||||
|
||||
assert exit_code == 0
|
||||
assert out_file.exists()
|
||||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert data["successful_articles"] == 1
|
||||
Reference in New Issue
Block a user