- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
373 lines
13 KiB
Python
373 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Testes automatizados para o Extrator e Parser Multimotor de Artigos.
|
|
Cobre modelos de dados, parsers (Trafilatura, Newspaper4k, Readability),
|
|
isolamento de falhas, orquestração de lote e interface CLI.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from pathlib import Path
|
|
from unittest.mock import patch
|
|
|
|
import pytest
|
|
|
|
from scripts.extract_article_contents import (
|
|
ArticleCrawler,
|
|
ExtractedArticle,
|
|
ExtractionBatchReport,
|
|
InputArticle,
|
|
NewspaperData,
|
|
NewspaperExtractor,
|
|
ReadabilityData,
|
|
ReadabilityExtractor,
|
|
TrafilaturaData,
|
|
TrafilaturaExtractor,
|
|
extract_all_engines,
|
|
load_search_json,
|
|
main,
|
|
process_batch,
|
|
save_extracted_json,
|
|
)
|
|
|
|
SAMPLE_HTML = """
|
|
<!DOCTYPE html>
|
|
<html lang="es">
|
|
<head>
|
|
<meta charset="utf-8">
|
|
<title>River Plate igualó sin goles ante Independiente Santa Fe - Olé</title>
|
|
<meta name="description" content="El equipo de Núñez empató 0-0 en Bogotá por los octavos de final.">
|
|
<meta name="author" content="Juan Pérez">
|
|
<meta property="og:title" content="River Plate igualó sin goles ante Independiente Santa Fe">
|
|
<meta property="og:image" content="https://media.ole.com.ar/river.jpg">
|
|
</head>
|
|
<body>
|
|
<header><nav><a href="/">Inicio</a></nav></header>
|
|
<article>
|
|
<h1>River Plate igualó sin goles ante Independiente Santa Fe</h1>
|
|
<p class="byline">Por Juan Pérez - 20 de Agosto de 2026</p>
|
|
<p class="lead">El equipo de Núñez empató 0-0 en Bogotá por la Copa Sudamericana.</p>
|
|
<p>Franco Armani fue la gran figura del encuentro con tres atajadas espectaculares en el primer tiempo.</p>
|
|
<p>El partido de vuelta se disputará en el estadio Monumental la próxima semana ante una multitud.</p>
|
|
</article>
|
|
<footer><p>Copyright 2026 Olé</p></footer>
|
|
</body>
|
|
</html>
|
|
"""
|
|
|
|
|
|
@pytest.fixture
|
|
def sample_input_json(tmp_path: Path) -> Path:
|
|
data = {
|
|
"query": "River Plate",
|
|
"language": "es",
|
|
"locale": "AR",
|
|
"total_itens": 2,
|
|
"items": [
|
|
{
|
|
"titulo": "River Plate igualó sin goles ante Santa Fe",
|
|
"subtitulo": "Empate en Bogotá",
|
|
"quando_publicado": "Thu, 20 Aug 2026 03:27:26 GMT",
|
|
"url": "https://www.ole.com.ar/river-0-0-santa-fe.html",
|
|
"pagina": 1,
|
|
},
|
|
{
|
|
"titulo": "Armani fue la figura de River",
|
|
"subtitulo": "Gran actuación del arquero",
|
|
"quando_publicado": "Thu, 20 Aug 2026 04:00:00 GMT",
|
|
"url": "https://www.tycsports.com/armani-figura.html",
|
|
"pagina": 1,
|
|
},
|
|
],
|
|
}
|
|
input_file = tmp_path / "river_plate.json"
|
|
input_file.write_text(json.dumps(data, ensure_ascii=False), encoding="utf-8")
|
|
return input_file
|
|
|
|
|
|
# ==============================================================================
|
|
# 1. Testes de Modelos e I/O de JSON
|
|
# ==============================================================================
|
|
|
|
|
|
def test_input_article_creation():
|
|
article = InputArticle(
|
|
titulo="Notícia Teste",
|
|
url="https://example.com/noticia",
|
|
subtitulo="Subtítulo",
|
|
quando_publicado="Thu, 20 Aug 2026",
|
|
pagina=1,
|
|
)
|
|
assert article.titulo == "Notícia Teste"
|
|
assert article.url == "https://example.com/noticia"
|
|
assert article.pagina == 1
|
|
d = article.to_dict()
|
|
assert d["titulo"] == "Notícia Teste"
|
|
assert d["url"] == "https://example.com/noticia"
|
|
|
|
|
|
def test_load_search_json_valid(sample_input_json: Path):
|
|
query, lang, items = load_search_json(sample_input_json)
|
|
assert query == "River Plate"
|
|
assert lang == "es"
|
|
assert len(items) == 2
|
|
assert items[0].url == "https://www.ole.com.ar/river-0-0-santa-fe.html"
|
|
|
|
|
|
def test_load_search_json_invalid_file(tmp_path: Path):
|
|
non_existent = tmp_path / "missing.json"
|
|
with pytest.raises(FileNotFoundError):
|
|
load_search_json(non_existent)
|
|
|
|
|
|
def test_save_extracted_json(tmp_path: Path):
|
|
report = ExtractionBatchReport(
|
|
source_file="test.json",
|
|
processed_at="2026-08-20T12:00:00Z",
|
|
total_articles=1,
|
|
successful_articles=1,
|
|
failed_articles=0,
|
|
articles=[
|
|
ExtractedArticle(
|
|
input_meta=InputArticle(titulo="Teste", url="https://example.com/noticia"),
|
|
extraction_status="success",
|
|
error_message=None,
|
|
crawled_url="https://example.com/noticia",
|
|
page_title="Página Teste",
|
|
http_status=200,
|
|
trafilatura=TrafilaturaData(
|
|
title="Teste",
|
|
author="Autor",
|
|
date="2026-08-20",
|
|
description="Desc",
|
|
categories=[],
|
|
tags=[],
|
|
canonical_url=None,
|
|
text="Texto do teste.",
|
|
raw_json=None,
|
|
error=None,
|
|
),
|
|
newspaper4k=NewspaperData(
|
|
title="Teste",
|
|
authors=["Autor"],
|
|
publish_date="2026-08-20",
|
|
text="Texto do teste.",
|
|
summary="Resumo",
|
|
keywords=["teste"],
|
|
top_image=None,
|
|
images=[],
|
|
meta_data={},
|
|
error=None,
|
|
),
|
|
readability=ReadabilityData(
|
|
title="Teste",
|
|
short_title="Teste",
|
|
cleaned_html="<p>Texto do teste.</p>",
|
|
cleaned_text="Texto do teste.",
|
|
error=None,
|
|
),
|
|
)
|
|
],
|
|
)
|
|
out_file = tmp_path / "out" / "result.json"
|
|
save_extracted_json(report, out_file)
|
|
assert out_file.exists()
|
|
content = json.loads(out_file.read_text(encoding="utf-8"))
|
|
assert content["total_articles"] == 1
|
|
assert content["articles"][0]["trafilatura"]["title"] == "Teste"
|
|
|
|
|
|
# ==============================================================================
|
|
# 2. Testes Unitários dos Parsers (Trafilatura, Newspaper4k, Readability)
|
|
# ==============================================================================
|
|
|
|
|
|
def test_trafilatura_extractor():
|
|
extractor = TrafilaturaExtractor()
|
|
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia")
|
|
assert isinstance(res, TrafilaturaData)
|
|
assert res.error is None
|
|
assert "Armani" in res.text or "River Plate" in res.text
|
|
assert res.title is not None
|
|
|
|
|
|
def test_newspaper_extractor():
|
|
extractor = NewspaperExtractor()
|
|
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es")
|
|
assert isinstance(res, NewspaperData)
|
|
assert res.error is None
|
|
assert "Armani" in res.text or "River" in res.text
|
|
assert isinstance(res.keywords, list)
|
|
assert len(res.keywords) > 0
|
|
|
|
|
|
def test_readability_extractor():
|
|
extractor = ReadabilityExtractor()
|
|
res = extractor.extract(SAMPLE_HTML)
|
|
assert isinstance(res, ReadabilityData)
|
|
assert res.error is None
|
|
assert res.title is not None
|
|
assert res.cleaned_html is not None
|
|
assert "Armani" in (res.cleaned_text or "") or "River" in (res.cleaned_text or "")
|
|
|
|
|
|
def test_extract_all_engines():
|
|
traf, news, read = extract_all_engines(
|
|
SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es"
|
|
)
|
|
assert traf.error is None
|
|
assert news.error is None
|
|
assert read.error is None
|
|
|
|
|
|
# ==============================================================================
|
|
# 3. Testes de Isolamento de Falhas (Resiliência)
|
|
# ==============================================================================
|
|
|
|
|
|
def test_extractor_error_isolation_on_faulty_engine():
|
|
with patch(
|
|
"scripts.extract_article_contents.TrafilaturaExtractor.extract",
|
|
side_effect=RuntimeError("Trafilatura crash"),
|
|
):
|
|
traf, news, read = extract_all_engines(
|
|
SAMPLE_HTML, url="https://example.com", language="es"
|
|
)
|
|
assert traf.error == "Trafilatura crash"
|
|
assert news.error is None
|
|
assert read.error is None
|
|
|
|
|
|
def test_crawler_error_isolation(sample_input_json: Path, tmp_path: Path):
|
|
out_file = tmp_path / "river_plate_extracted.json"
|
|
|
|
def mock_crawl(url, timeout_sec=30):
|
|
if "ole.com.ar" in url:
|
|
return SAMPLE_HTML, "River Plate Olé", 200
|
|
raise ConnectionError("Connection refused by tycsports.com")
|
|
|
|
with patch.object(ArticleCrawler, "crawl", side_effect=mock_crawl):
|
|
with patch.object(ArticleCrawler, "start"), patch.object(ArticleCrawler, "close"):
|
|
report = process_batch(
|
|
input_path=sample_input_json,
|
|
output_path=out_file,
|
|
silent=True,
|
|
)
|
|
|
|
assert report.total_articles == 2
|
|
assert report.successful_articles == 1
|
|
assert report.failed_articles == 1
|
|
assert report.articles[0].extraction_status == "success"
|
|
assert report.articles[1].extraction_status == "failed"
|
|
assert "Connection refused" in (report.articles[1].error_message or "")
|
|
|
|
|
|
# ==============================================================================
|
|
# 4. Testes de CLI e Limitação (--limit, --silent, --language)
|
|
# ==============================================================================
|
|
|
|
|
|
def test_process_batch_with_limit(sample_input_json: Path, tmp_path: Path):
|
|
out_file = tmp_path / "limit_extracted.json"
|
|
|
|
with (
|
|
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
|
|
patch.object(ArticleCrawler, "start"),
|
|
patch.object(ArticleCrawler, "close"),
|
|
):
|
|
report = process_batch(
|
|
input_path=sample_input_json,
|
|
output_path=out_file,
|
|
limit=1,
|
|
silent=True,
|
|
)
|
|
|
|
assert report.total_articles == 1
|
|
assert len(report.articles) == 1
|
|
assert out_file.exists()
|
|
|
|
|
|
def test_cli_main_success(sample_input_json: Path, tmp_path: Path, capsys):
|
|
out_file = tmp_path / "cli_out.json"
|
|
|
|
with (
|
|
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
|
|
patch.object(ArticleCrawler, "start"),
|
|
patch.object(ArticleCrawler, "close"),
|
|
):
|
|
exit_code = main(["-i", str(sample_input_json), "-o", str(out_file), "--limit", "1", "-s"])
|
|
|
|
assert exit_code == 0
|
|
assert out_file.exists()
|
|
|
|
|
|
def test_cli_main_missing_input_file(tmp_path: Path, capsys):
|
|
missing_file = tmp_path / "does_not_exist.json"
|
|
exit_code = main(["-i", str(missing_file), "-s"])
|
|
assert exit_code == 1
|
|
|
|
|
|
# ==============================================================================
|
|
# 5. Testes End-to-End (E2E) ao Vivo (Live Network)
|
|
# ==============================================================================
|
|
|
|
|
|
def test_e2e_live_article_extraction(tmp_path: Path):
|
|
"""Valida E2E a extração real ao vivo com Foxcape e os 3 motores em lote."""
|
|
live_input_file = Path("out/river_plate.json")
|
|
if not live_input_file.exists():
|
|
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E ao vivo.")
|
|
|
|
out_file = tmp_path / "e2e_live_extracted.json"
|
|
|
|
# Executa o batch real com limite de 1 notícia
|
|
report = process_batch(
|
|
input_path=live_input_file,
|
|
output_path=out_file,
|
|
limit=1,
|
|
silent=True,
|
|
)
|
|
|
|
assert report.total_articles == 1
|
|
assert report.successful_articles == 1
|
|
assert report.failed_articles == 0
|
|
assert len(report.articles) == 1
|
|
|
|
art = report.articles[0]
|
|
assert art.extraction_status == "success"
|
|
assert art.crawled_url.startswith("http")
|
|
|
|
# Valida que todos os 3 motores extraíram dados reais
|
|
assert art.trafilatura is not None and art.trafilatura.error is None
|
|
assert len(art.trafilatura.text) > 50
|
|
|
|
assert art.newspaper4k is not None and art.newspaper4k.error is None
|
|
assert len(art.newspaper4k.text) > 50
|
|
assert isinstance(art.newspaper4k.keywords, list)
|
|
|
|
assert art.readability is not None and art.readability.error is None
|
|
assert art.readability.cleaned_html is not None
|
|
assert len(art.readability.cleaned_text or "") > 50
|
|
|
|
# Valida arquivo JSON gravado
|
|
assert out_file.exists()
|
|
saved = json.loads(out_file.read_text(encoding="utf-8"))
|
|
assert saved["total_articles"] == 1
|
|
assert saved["articles"][0]["extraction_status"] == "success"
|
|
|
|
|
|
def test_e2e_cli_live_execution(tmp_path: Path):
|
|
"""Valida E2E a execução do CLI real de ponta a ponta."""
|
|
live_input_file = Path("out/river_plate.json")
|
|
if not live_input_file.exists():
|
|
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E.")
|
|
|
|
out_file = tmp_path / "e2e_cli_live.json"
|
|
exit_code = main(["-i", str(live_input_file), "-o", str(out_file), "--limit", "1", "-s"])
|
|
|
|
assert exit_code == 0
|
|
assert out_file.exists()
|
|
data = json.loads(out_file.read_text(encoding="utf-8"))
|
|
assert data["successful_articles"] == 1
|