Files
TextNLPClassifierApp/tests/tools/test_select_article_extractor.py

616 lines
24 KiB
Python

"""
Suíte de Testes Automatizados para o Seletor Determinístico de Extrator.
Cobre 100% dos Casos de Teste Obrigatórios do PRD (CT-001 a CT-014), testes unitários
de normalização e shingles, testes de integração de lote e testes E2E via subprocess.
"""
from __future__ import annotations
import json
import subprocess
import sys
from pathlib import Path
import pytest
from scripts.select_article_extractor import (
ExtractorName,
generate_shingles,
normalize_text,
process_batch,
select_article_extractor,
)
# ==============================================================================
# Testes Unitários de Normalização e Tokenização
# ==============================================================================
def test_normalize_text_empty_and_invalid():
assert normalize_text(None) == []
assert normalize_text("") == []
assert normalize_text(" \n\t ") == []
assert normalize_text(12345) == []
def test_normalize_text_html_entities_and_tags():
raw = "<p>El &amp; <b>futebol</b> mundial &quot;está&quot; mudando.</p>"
tokens = normalize_text(raw)
assert tokens == ["el", "futebol", "mundial", "está", "mudando"]
def test_normalize_text_markdown_links():
raw = "Veja mais no [Portal de Notícias](https://example.com/noticias) hoje."
tokens = normalize_text(raw)
assert tokens == ["veja", "mais", "no", "portal", "de", "notícias", "hoje"]
def test_normalize_text_markdown_images_stripped_while_links_preserved():
"""Garante que marcação de imagem Markdown ![alt](url) seja descartada e link [texto](url) seja preservado."""
raw = (
"Texto inicial do artigo. "
"![Legenda da foto e imagem](https://example.com/imagem.webp) "
"Mais texto com [link importante](https://example.com/pagina) e outra "
"![Outra foto](https://example.com/foto2.jpg) informação."
)
tokens = normalize_text(raw)
assert "legenda" not in tokens
assert "foto" not in tokens
assert "imagem" not in tokens
assert "link" in tokens
assert "importante" in tokens
assert tokens == [
"texto",
"inicial",
"do",
"artigo",
"mais",
"texto",
"com",
"link",
"importante",
"e",
"outra",
"informação",
]
def test_normalize_text_nfkc_unicode_and_punctuation():
# Caracteres combinados e pontuação
raw = "River Plate venceu por 3-0! (Com gol de pênalti & golaço de falta)."
tokens = normalize_text(raw)
assert tokens == [
"river",
"plate",
"venceu",
"por",
"3",
"0",
"com",
"gol",
"de",
"pênalti",
"golaço",
"de",
"falta",
]
# ==============================================================================
# Testes Unitários de Shingles
# ==============================================================================
def test_generate_shingles_sliding_window():
tokens = ["um", "dois", "três", "quatro", "cinco", "seis"]
shingles = generate_shingles(tokens, window_size=5)
assert len(shingles) == 2
assert ("um", "dois", "três", "quatro", "cinco") in shingles
assert ("dois", "três", "quatro", "cinco", "seis") in shingles
def test_generate_shingles_short_text():
# Entre 1 e 4 tokens deve gerar 1 único shingle com a tupla completa
tokens = ["river", "plate", "campeão"]
shingles = generate_shingles(tokens, window_size=5)
assert len(shingles) == 1
assert ("river", "plate", "campeão") in shingles
def test_generate_shingles_empty():
assert generate_shingles([]) == set()
# ==============================================================================
# Casos de Teste Obrigatórios do PRD (§12: CT-001 a CT-014)
# ==============================================================================
def test_ct_001_three_candidates_clear_winner():
"""CT-001: Três candidatos com consenso e um vencedor claro -> Selecionar o maior score."""
base = "river plate venceu o clássico ontem a noite no estádio monumental"
article = {
"trafilatura": {"text": base, "error": None},
"newspaper4k": {"text": base, "error": None},
"readability": {
"cleaned_text": "texto completamente diferente sem nenhuma relação",
"error": None,
},
}
result = select_article_extractor(article)
assert result.selected_extractor in (ExtractorName.NEWSPAPER4K, ExtractorName.TRAFILATURA)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
def test_ct_002_technical_tie_smallest_shingles():
"""CT-002: Dois ou mais candidatos dentro de 0,03 do maior score -> Selecionar o de menor quantidade de shingles."""
tokens_comuns = (
"o rio de janeiro continua lindo e sempre maravilhoso em todas as estações do ano"
)
article = {
"trafilatura": {"text": tokens_comuns, "error": None},
"readability": {
"cleaned_text": tokens_comuns + " propaganda extra adicionada no fim",
"error": None,
},
"newspaper4k": {"text": tokens_comuns, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
def test_ct_003_technical_tie_priority_fallback():
"""CT-003: Empate técnico e mesma quantidade de shingles -> Aplicar prioridade final (newspaper4k > readability > trafilatura)."""
texto = "o time jogou muito bem durante toda a partida de futebol"
article = {
"trafilatura": {"text": texto, "error": None},
"readability": {"cleaned_text": texto, "error": None},
"newspaper4k": {"text": texto, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
# Agora sem newspaper4k ativo (somente readability e trafilatura idênticos)
article_two = {
"trafilatura": {"text": texto, "error": None},
"readability": {"cleaned_text": texto, "error": None},
"newspaper4k": {"text": None, "error": "Crash"},
}
result_two = select_article_extractor(article_two)
assert result_two.selected_extractor == ExtractorName.READABILITY
def test_ct_004_three_candidates_no_consensus():
"""CT-004: Três candidatos sem consenso -> Selecionar a quantidade mediana de shingles."""
t1 = "alfa bravo charlie delta echo foxtrot golf hotel india juliet" # 10 tokens -> 6 shingles
t2 = "kilo lima mike november oscar papa quebec romeo sierra tango uniform victor" # 12 tokens -> 8 shingles (MEDIANA)
t3 = "whiskey xray yankee zulu zero one two three four five six seven eight nine" # 14 tokens -> 10 shingles
article = {
"trafilatura": {"text": t1, "error": None},
"readability": {"cleaned_text": t2, "error": None},
"newspaper4k": {"text": t3, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.READABILITY
assert result.selection_reason == "no_consensus_median_shingles"
def test_ct_005_two_candidates_no_consensus():
"""CT-005: Dois candidatos sem consenso -> Selecionar a maior quantidade de shingles."""
t_short = "alfa bravo charlie delta echo foxtrot" # 6 tokens -> 2 shingles
t_long = (
"kilo lima mike november oscar papa quebec romeo sierra tango" # 10 tokens -> 6 shingles
)
article = {
"trafilatura": {"text": t_short, "error": None},
"newspaper4k": {"text": t_long, "error": None},
"readability": {"cleaned_text": None, "error": "Not extracted"},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
assert result.selection_reason == "no_consensus_max_shingles"
def test_ct_006_single_usable_candidate():
"""CT-006: Somente um candidato utilizável -> Selecionar esse candidato."""
article = {
"trafilatura": {
"text": "conteúdo válido e utilizável extraído com sucesso aqui",
"error": None,
},
"newspaper4k": {"text": "", "error": None},
"readability": {"cleaned_text": None, "error": "Timeout error"},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.TRAFILATURA
assert result.selection_reason == "single_usable_candidate"
def test_ct_007_degraded_candidates_only():
"""CT-007: Nenhum utilizável, mas existe candidato degradado -> Executar o algoritmo somente com os degradados."""
texto_comum = "artigo relevante sobre economia global e finanças internacionais com detalhes"
article = {
"trafilatura": {"text": texto_comum, "error": "Warning: partial parse"},
"newspaper4k": {"text": texto_comum, "error": "HTTP 403 partial"},
"readability": {"cleaned_text": None, "error": "Fatal exception"},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
def test_ct_008_all_candidates_unavailable():
"""CT-008: Todos os candidatos indisponíveis -> Selecionar newspaper4k."""
article = {
"trafilatura": {"text": None, "error": "Error"},
"newspaper4k": {"text": "", "error": "Empty"},
"readability": {"cleaned_text": " ", "error": "Blank"},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
assert result.selection_reason == "fallback_all_unavailable"
def test_ct_009_small_fragment_loses_due_to_low_coverage():
"""CT-009: Readability retorna apenas um fragmento pequeno enquanto os outros concordam -> Perde por baixa cobertura."""
full_text = (
"o presidente da república anunciou novas medidas econômicas para conter a inflação "
"e estimular o crescimento industrial em todo o território nacional durante o pronunciamento oficial"
)
small_fragment = "o presidente da república anunciou"
article = {
"trafilatura": {"text": full_text, "error": None},
"newspaper4k": {"text": full_text, "error": None},
"readability": {"cleaned_text": small_fragment, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor in (ExtractorName.TRAFILATURA, ExtractorName.NEWSPAPER4K)
assert result.selected_extractor != ExtractorName.READABILITY
def test_ct_010_excessive_boilerplate_loses_due_to_low_support():
"""CT-010: Um candidato contém o conteúdo comum e muito conteúdo excedente -> Perde suporte e reduz pontuação."""
common_content = (
"notícia oficial com dados apurados sobre a operação policial realizada nesta manhã"
)
massive_boilerplate = common_content + (
" compartilhe no facebook twitter whatsapp veja também esportes receitas horóscopo política "
" e assine nossa newsletter diária para receber mais novidades sobre culinária e fofocas"
)
article = {
"trafilatura": {"text": common_content, "error": None},
"readability": {"cleaned_text": common_content, "error": None},
"newspaper4k": {"text": massive_boilerplate, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.READABILITY
def test_ct_011_partial_content_loses_due_to_low_coverage():
"""CT-011: Um candidato contém somente parte do conteúdo comum -> Perde cobertura e reduz pontuação."""
full_content = "primeiro parágrafo do artigo completo segundo parágrafo com explicações terceiro parágrafo final"
half_content = "primeiro parágrafo do artigo completo"
article = {
"newspaper4k": {"text": full_content, "error": None},
"readability": {"cleaned_text": full_content, "error": None},
"trafilatura": {"text": half_content, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
def test_ct_012_recalculate_existing_selected_extractor(tmp_path: Path):
"""CT-012: A entrada já contém selected_extractor -> Recalcular e substituir somente essa chave."""
input_data = {
"articles": [
{
"titulo": "Teste",
"selected_extractor": "trafilatura",
"trafilatura": {"text": "lixo sem sentido", "error": None},
"newspaper4k": {
"text": "conteúdo correto compartilhado por dois motores",
"error": None,
},
"readability": {
"cleaned_text": "conteúdo correto compartilhado por dois motores",
"error": None,
},
}
]
}
in_file = tmp_path / "artigos.json"
in_file.write_text(json.dumps(input_data, ensure_ascii=False), encoding="utf-8")
res = process_batch(in_file)
out_file = Path(res.output_file)
assert out_file.exists()
with open(out_file, "r", encoding="utf-8") as f:
out_data = json.load(f)
assert out_data["articles"][0]["selected_extractor"] == "newspaper4k"
def test_ct_013_empty_articles_list(tmp_path: Path):
"""CT-013: articles está vazio -> Gerar saída válida com articles vazio."""
input_data = {"metadata": "info", "articles": []}
in_file = tmp_path / "empty.json"
in_file.write_text(json.dumps(input_data), encoding="utf-8")
res = process_batch(in_file)
out_file = Path(res.output_file)
assert out_file.exists()
with open(out_file, "r", encoding="utf-8") as f:
out_data = json.load(f)
assert out_data["metadata"] == "info"
assert out_data["articles"] == []
assert res.total_articles == 0
assert res.processed_count == 0
def test_ct_014_invalid_json_fails_atomically(tmp_path: Path):
"""CT-014: JSON inválido -> Não gerar saída."""
in_file = tmp_path / "invalid.json"
in_file.write_text("{articles: [ malformed json", encoding="utf-8")
expected_out = tmp_path / "invalid_selected.json"
with pytest.raises(ValueError, match="JSON inválido"):
process_batch(in_file)
assert not expected_out.exists()
# ==============================================================================
# Testes de Integração
# ==============================================================================
def test_integration_reference_file(tmp_path: Path):
"""Testa o processamento em lote completo sobre o arquivo real out/river_plate_extracted.json."""
ref_file = Path("out/river_plate_extracted.json")
if not ref_file.exists():
pytest.skip(
"Arquivo out/river_plate_extracted.json não encontrado para teste de integração."
)
out_file = tmp_path / "river_plate_extracted_selected.json"
result = process_batch(ref_file, output_path=out_file, verbose=True)
assert result.total_articles == 20
assert result.processed_count == 20
assert out_file.exists()
with open(out_file, "r", encoding="utf-8") as f:
data = json.load(f)
assert len(data["articles"]) == 20
for art in data["articles"]:
assert "selected_extractor" in art
assert art["selected_extractor"] in ["trafilatura", "newspaper4k", "readability"]
# Validar a distribuição exata conforme o algoritmo do PRD
assert result.selection_distribution == {
"newspaper4k": 9,
"readability": 9,
"trafilatura": 2,
}
def test_article_1_regression_technical_tie_markdown_images():
"""
Teste de regressão para o Artigo 1 (Los puntajes de River vs. Independiente Santa Fe):
Valida que com o descarte de imagens Markdown ![alt](url), Readability e Newspaper4k
entram em empate técnico (diff <= 0.03) e Newspaper4k vence por possuir menor quantidade
de shingles (1009 vs 1057).
"""
ref_file = Path("out/river_plate_extracted.json")
if not ref_file.exists():
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
with open(ref_file, "r", encoding="utf-8") as f:
data = json.load(f)
art1 = data["articles"][0]
result = select_article_extractor(art1, article_index=0)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
assert result.selection_reason == "technical_tie_smallest_shingles"
cand_news = result.candidates[ExtractorName.NEWSPAPER4K]
cand_read = result.candidates[ExtractorName.READABILITY]
cand_traf = result.candidates[ExtractorName.TRAFILATURA]
assert cand_news.shingle_count == 1009
assert cand_read.shingle_count == 1057
assert cand_traf.shingle_count == 1091
# Diferença para o maior score <= 0.03 (empate técnico)
max_score = max(cand_news.score, cand_read.score, cand_traf.score)
assert (max_score - cand_news.score) <= 0.03
assert (max_score - cand_read.score) <= 0.03
def test_integration_large_batch_determinism(tmp_path: Path):
"""Gera um lote de 100 artigos sintéticos e verifica 100% de repetibilidade determinística entre 2 execuções."""
articles = []
for i in range(100):
if i % 4 == 0:
art = {
"trafilatura": {
"text": f"artigo numero {i} sobre futebol internacional no estadio",
"error": None,
},
"newspaper4k": {
"text": f"artigo numero {i} sobre futebol internacional no estadio",
"error": None,
},
"readability": {"cleaned_text": "sem relacao", "error": None},
}
elif i % 4 == 1:
art = {
"trafilatura": {"text": None, "error": "timeout"},
"newspaper4k": {"text": f"noticia exclusiva {i} com detalhes", "error": None},
"readability": {
"cleaned_text": f"noticia exclusiva {i} com detalhes",
"error": None,
},
}
elif i % 4 == 2:
art = {
"trafilatura": {"text": f"texto a {i}", "error": None},
"newspaper4k": {"text": f"texto b diferente {i}", "error": None},
"readability": {"cleaned_text": f"texto c terceiro {i}", "error": None},
}
else:
art = {
"trafilatura": {"text": None, "error": "err"},
"newspaper4k": {"text": "", "error": "err"},
"readability": {"cleaned_text": None, "error": "err"},
}
articles.append(art)
batch_payload = {"articles": articles}
in_file = tmp_path / "large_batch.json"
in_file.write_text(json.dumps(batch_payload, ensure_ascii=False), encoding="utf-8")
out1 = tmp_path / "large_batch_run1.json"
out2 = tmp_path / "large_batch_run2.json"
res1 = process_batch(in_file, output_path=out1)
res2 = process_batch(in_file, output_path=out2)
assert res1.processed_count == 100
assert res2.processed_count == 100
# Os resultados devem ser 100% idênticos
selections1 = [s.selected_extractor for s in res1.selections]
selections2 = [s.selected_extractor for s in res2.selections]
assert selections1 == selections2
def test_integration_pipeline_downstream_consumer(tmp_path: Path):
"""
Testa a integração end-to-end do pipeline downstream:
Lê o JSON enriquecido com selected_extractor, recupera o conteúdo do extrator vencedor e
garante que o texto está higienizado e pronto para os classificadores NLP.
"""
ref_file = Path("out/river_plate_extracted.json")
if not ref_file.exists():
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
out_file = tmp_path / "downstream_test.json"
process_batch(ref_file, output_path=out_file)
with open(out_file, "r", encoding="utf-8") as f:
data = json.load(f)
for idx, art in enumerate(data["articles"]):
winner = art["selected_extractor"]
assert winner in ["trafilatura", "newspaper4k", "readability"]
# Recuperar texto do extrator vencedor conforme mapeamento do PRD
if winner == "trafilatura":
chosen_text = art["trafilatura"]["text"]
elif winner == "newspaper4k":
chosen_text = art["newspaper4k"]["text"]
elif winner == "readability":
chosen_text = art["readability"]["cleaned_text"]
assert isinstance(chosen_text, str)
assert len(chosen_text.strip()) > 0
# ==============================================================================
# Testes End-to-End (E2E) via Subprocess CLI
# ==============================================================================
def test_e2e_cli_subprocess_real_execution(tmp_path: Path):
"""E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha de comando."""
ref_file = Path("out/river_plate_extracted.json")
if not ref_file.exists():
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado para teste E2E.")
out_file = tmp_path / "e2e_river_plate_selected.json"
script_path = Path("scripts/select_article_extractor.py").resolve()
cmd = [
sys.executable,
str(script_path),
str(ref_file.resolve()),
"-o",
str(out_file),
"--verbose",
"--indent",
"2",
]
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
assert proc.returncode == 0
assert out_file.exists()
# Validar saída estruturada JSON do stdout
summary = json.loads(proc.stdout)
assert summary["status"] == "success"
assert summary["total_articles"] == 20
assert summary["processed_count"] == 20
assert "distribution" in summary
# Validar que logs de verbose foram emitidos no stderr
assert "[Artigo #001]" in proc.stderr
assert "[Artigo #020]" in proc.stderr
def test_e2e_cli_subprocess_default_naming(tmp_path: Path):
"""E2E: Executa CLI sem a flag -o e valida criação automática de <nome>_selected.json."""
sample_data = {
"articles": [
{
"titulo": "Artigo Automático",
"trafilatura": {"text": "conteúdo padrão", "error": None},
"newspaper4k": {"text": "conteúdo padrão", "error": None},
"readability": {"cleaned_text": "conteúdo padrão", "error": None},
}
]
}
in_file = tmp_path / "my_news.json"
in_file.write_text(json.dumps(sample_data), encoding="utf-8")
expected_out = tmp_path / "my_news_selected.json"
script_path = Path("scripts/select_article_extractor.py").resolve()
cmd = [sys.executable, str(script_path), str(in_file)]
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
assert proc.returncode == 0
assert expected_out.exists()
with open(expected_out, "r", encoding="utf-8") as f:
data = json.load(f)
assert data["articles"][0]["selected_extractor"] == "newspaper4k"
def test_e2e_cli_subprocess_invalid_input(tmp_path: Path):
"""E2E: Executa CLI com JSON inválido e valida código de saída e erro no stderr."""
invalid_file = tmp_path / "broken.json"
invalid_file.write_text("not a valid json {", encoding="utf-8")
script_path = Path("scripts/select_article_extractor.py").resolve()
cmd = [sys.executable, str(script_path), str(invalid_file)]
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
assert proc.returncode == 2
assert "ERRO DE VALIDAÇÃO" in proc.stderr
def test_e2e_cli_subprocess_missing_file():
"""E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr."""
script_path = Path("scripts/select_article_extractor.py").resolve()
cmd = [sys.executable, str(script_path), "non_existent_file_12345.json"]
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
assert proc.returncode == 1
assert "ERRO DE ARQUIVO" in proc.stderr