feat: add deterministic content extractor selector engine with F1 consensus
This commit is contained in:
@@ -0,0 +1,615 @@
|
||||
"""
|
||||
Suíte de Testes Automatizados para o Seletor Determinístico de Extrator.
|
||||
|
||||
Cobre 100% dos Casos de Teste Obrigatórios do PRD (CT-001 a CT-014), testes unitários
|
||||
de normalização e shingles, testes de integração de lote e testes E2E via subprocess.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from scripts.select_article_extractor import (
|
||||
ExtractorName,
|
||||
generate_shingles,
|
||||
normalize_text,
|
||||
process_batch,
|
||||
select_article_extractor,
|
||||
)
|
||||
|
||||
# ==============================================================================
|
||||
# Testes Unitários de Normalização e Tokenização
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_normalize_text_empty_and_invalid():
|
||||
assert normalize_text(None) == []
|
||||
assert normalize_text("") == []
|
||||
assert normalize_text(" \n\t ") == []
|
||||
assert normalize_text(12345) == []
|
||||
|
||||
|
||||
def test_normalize_text_html_entities_and_tags():
|
||||
raw = "<p>El & <b>futebol</b> mundial "está" mudando.</p>"
|
||||
tokens = normalize_text(raw)
|
||||
assert tokens == ["el", "futebol", "mundial", "está", "mudando"]
|
||||
|
||||
|
||||
def test_normalize_text_markdown_links():
|
||||
raw = "Veja mais no [Portal de Notícias](https://example.com/noticias) hoje."
|
||||
tokens = normalize_text(raw)
|
||||
assert tokens == ["veja", "mais", "no", "portal", "de", "notícias", "hoje"]
|
||||
|
||||
|
||||
def test_normalize_text_markdown_images_stripped_while_links_preserved():
|
||||
"""Garante que marcação de imagem Markdown  seja descartada e link [texto](url) seja preservado."""
|
||||
raw = (
|
||||
"Texto inicial do artigo. "
|
||||
" "
|
||||
"Mais texto com [link importante](https://example.com/pagina) e outra "
|
||||
" informação."
|
||||
)
|
||||
tokens = normalize_text(raw)
|
||||
assert "legenda" not in tokens
|
||||
assert "foto" not in tokens
|
||||
assert "imagem" not in tokens
|
||||
assert "link" in tokens
|
||||
assert "importante" in tokens
|
||||
assert tokens == [
|
||||
"texto",
|
||||
"inicial",
|
||||
"do",
|
||||
"artigo",
|
||||
"mais",
|
||||
"texto",
|
||||
"com",
|
||||
"link",
|
||||
"importante",
|
||||
"e",
|
||||
"outra",
|
||||
"informação",
|
||||
]
|
||||
|
||||
|
||||
def test_normalize_text_nfkc_unicode_and_punctuation():
|
||||
# Caracteres combinados e pontuação
|
||||
raw = "River Plate venceu por 3-0! (Com gol de pênalti & golaço de falta)."
|
||||
tokens = normalize_text(raw)
|
||||
assert tokens == [
|
||||
"river",
|
||||
"plate",
|
||||
"venceu",
|
||||
"por",
|
||||
"3",
|
||||
"0",
|
||||
"com",
|
||||
"gol",
|
||||
"de",
|
||||
"pênalti",
|
||||
"golaço",
|
||||
"de",
|
||||
"falta",
|
||||
]
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Testes Unitários de Shingles
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_generate_shingles_sliding_window():
|
||||
tokens = ["um", "dois", "três", "quatro", "cinco", "seis"]
|
||||
shingles = generate_shingles(tokens, window_size=5)
|
||||
assert len(shingles) == 2
|
||||
assert ("um", "dois", "três", "quatro", "cinco") in shingles
|
||||
assert ("dois", "três", "quatro", "cinco", "seis") in shingles
|
||||
|
||||
|
||||
def test_generate_shingles_short_text():
|
||||
# Entre 1 e 4 tokens deve gerar 1 único shingle com a tupla completa
|
||||
tokens = ["river", "plate", "campeão"]
|
||||
shingles = generate_shingles(tokens, window_size=5)
|
||||
assert len(shingles) == 1
|
||||
assert ("river", "plate", "campeão") in shingles
|
||||
|
||||
|
||||
def test_generate_shingles_empty():
|
||||
assert generate_shingles([]) == set()
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Casos de Teste Obrigatórios do PRD (§12: CT-001 a CT-014)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_ct_001_three_candidates_clear_winner():
|
||||
"""CT-001: Três candidatos com consenso e um vencedor claro -> Selecionar o maior score."""
|
||||
base = "river plate venceu o clássico ontem a noite no estádio monumental"
|
||||
article = {
|
||||
"trafilatura": {"text": base, "error": None},
|
||||
"newspaper4k": {"text": base, "error": None},
|
||||
"readability": {
|
||||
"cleaned_text": "texto completamente diferente sem nenhuma relação",
|
||||
"error": None,
|
||||
},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor in (ExtractorName.NEWSPAPER4K, ExtractorName.TRAFILATURA)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
|
||||
def test_ct_002_technical_tie_smallest_shingles():
|
||||
"""CT-002: Dois ou mais candidatos dentro de 0,03 do maior score -> Selecionar o de menor quantidade de shingles."""
|
||||
tokens_comuns = (
|
||||
"o rio de janeiro continua lindo e sempre maravilhoso em todas as estações do ano"
|
||||
)
|
||||
article = {
|
||||
"trafilatura": {"text": tokens_comuns, "error": None},
|
||||
"readability": {
|
||||
"cleaned_text": tokens_comuns + " propaganda extra adicionada no fim",
|
||||
"error": None,
|
||||
},
|
||||
"newspaper4k": {"text": tokens_comuns, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
|
||||
def test_ct_003_technical_tie_priority_fallback():
|
||||
"""CT-003: Empate técnico e mesma quantidade de shingles -> Aplicar prioridade final (newspaper4k > readability > trafilatura)."""
|
||||
texto = "o time jogou muito bem durante toda a partida de futebol"
|
||||
article = {
|
||||
"trafilatura": {"text": texto, "error": None},
|
||||
"readability": {"cleaned_text": texto, "error": None},
|
||||
"newspaper4k": {"text": texto, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
# Agora sem newspaper4k ativo (somente readability e trafilatura idênticos)
|
||||
article_two = {
|
||||
"trafilatura": {"text": texto, "error": None},
|
||||
"readability": {"cleaned_text": texto, "error": None},
|
||||
"newspaper4k": {"text": None, "error": "Crash"},
|
||||
}
|
||||
result_two = select_article_extractor(article_two)
|
||||
assert result_two.selected_extractor == ExtractorName.READABILITY
|
||||
|
||||
|
||||
def test_ct_004_three_candidates_no_consensus():
|
||||
"""CT-004: Três candidatos sem consenso -> Selecionar a quantidade mediana de shingles."""
|
||||
t1 = "alfa bravo charlie delta echo foxtrot golf hotel india juliet" # 10 tokens -> 6 shingles
|
||||
t2 = "kilo lima mike november oscar papa quebec romeo sierra tango uniform victor" # 12 tokens -> 8 shingles (MEDIANA)
|
||||
t3 = "whiskey xray yankee zulu zero one two three four five six seven eight nine" # 14 tokens -> 10 shingles
|
||||
article = {
|
||||
"trafilatura": {"text": t1, "error": None},
|
||||
"readability": {"cleaned_text": t2, "error": None},
|
||||
"newspaper4k": {"text": t3, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.READABILITY
|
||||
assert result.selection_reason == "no_consensus_median_shingles"
|
||||
|
||||
|
||||
def test_ct_005_two_candidates_no_consensus():
|
||||
"""CT-005: Dois candidatos sem consenso -> Selecionar a maior quantidade de shingles."""
|
||||
t_short = "alfa bravo charlie delta echo foxtrot" # 6 tokens -> 2 shingles
|
||||
t_long = (
|
||||
"kilo lima mike november oscar papa quebec romeo sierra tango" # 10 tokens -> 6 shingles
|
||||
)
|
||||
article = {
|
||||
"trafilatura": {"text": t_short, "error": None},
|
||||
"newspaper4k": {"text": t_long, "error": None},
|
||||
"readability": {"cleaned_text": None, "error": "Not extracted"},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
assert result.selection_reason == "no_consensus_max_shingles"
|
||||
|
||||
|
||||
def test_ct_006_single_usable_candidate():
|
||||
"""CT-006: Somente um candidato utilizável -> Selecionar esse candidato."""
|
||||
article = {
|
||||
"trafilatura": {
|
||||
"text": "conteúdo válido e utilizável extraído com sucesso aqui",
|
||||
"error": None,
|
||||
},
|
||||
"newspaper4k": {"text": "", "error": None},
|
||||
"readability": {"cleaned_text": None, "error": "Timeout error"},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.TRAFILATURA
|
||||
assert result.selection_reason == "single_usable_candidate"
|
||||
|
||||
|
||||
def test_ct_007_degraded_candidates_only():
|
||||
"""CT-007: Nenhum utilizável, mas existe candidato degradado -> Executar o algoritmo somente com os degradados."""
|
||||
texto_comum = "artigo relevante sobre economia global e finanças internacionais com detalhes"
|
||||
article = {
|
||||
"trafilatura": {"text": texto_comum, "error": "Warning: partial parse"},
|
||||
"newspaper4k": {"text": texto_comum, "error": "HTTP 403 partial"},
|
||||
"readability": {"cleaned_text": None, "error": "Fatal exception"},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
|
||||
def test_ct_008_all_candidates_unavailable():
|
||||
"""CT-008: Todos os candidatos indisponíveis -> Selecionar newspaper4k."""
|
||||
article = {
|
||||
"trafilatura": {"text": None, "error": "Error"},
|
||||
"newspaper4k": {"text": "", "error": "Empty"},
|
||||
"readability": {"cleaned_text": " ", "error": "Blank"},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
assert result.selection_reason == "fallback_all_unavailable"
|
||||
|
||||
|
||||
def test_ct_009_small_fragment_loses_due_to_low_coverage():
|
||||
"""CT-009: Readability retorna apenas um fragmento pequeno enquanto os outros concordam -> Perde por baixa cobertura."""
|
||||
full_text = (
|
||||
"o presidente da república anunciou novas medidas econômicas para conter a inflação "
|
||||
"e estimular o crescimento industrial em todo o território nacional durante o pronunciamento oficial"
|
||||
)
|
||||
small_fragment = "o presidente da república anunciou"
|
||||
article = {
|
||||
"trafilatura": {"text": full_text, "error": None},
|
||||
"newspaper4k": {"text": full_text, "error": None},
|
||||
"readability": {"cleaned_text": small_fragment, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor in (ExtractorName.TRAFILATURA, ExtractorName.NEWSPAPER4K)
|
||||
assert result.selected_extractor != ExtractorName.READABILITY
|
||||
|
||||
|
||||
def test_ct_010_excessive_boilerplate_loses_due_to_low_support():
|
||||
"""CT-010: Um candidato contém o conteúdo comum e muito conteúdo excedente -> Perde suporte e reduz pontuação."""
|
||||
common_content = (
|
||||
"notícia oficial com dados apurados sobre a operação policial realizada nesta manhã"
|
||||
)
|
||||
massive_boilerplate = common_content + (
|
||||
" compartilhe no facebook twitter whatsapp veja também esportes receitas horóscopo política "
|
||||
" e assine nossa newsletter diária para receber mais novidades sobre culinária e fofocas"
|
||||
)
|
||||
article = {
|
||||
"trafilatura": {"text": common_content, "error": None},
|
||||
"readability": {"cleaned_text": common_content, "error": None},
|
||||
"newspaper4k": {"text": massive_boilerplate, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.READABILITY
|
||||
|
||||
|
||||
def test_ct_011_partial_content_loses_due_to_low_coverage():
|
||||
"""CT-011: Um candidato contém somente parte do conteúdo comum -> Perde cobertura e reduz pontuação."""
|
||||
full_content = "primeiro parágrafo do artigo completo segundo parágrafo com explicações terceiro parágrafo final"
|
||||
half_content = "primeiro parágrafo do artigo completo"
|
||||
article = {
|
||||
"newspaper4k": {"text": full_content, "error": None},
|
||||
"readability": {"cleaned_text": full_content, "error": None},
|
||||
"trafilatura": {"text": half_content, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
|
||||
def test_ct_012_recalculate_existing_selected_extractor(tmp_path: Path):
|
||||
"""CT-012: A entrada já contém selected_extractor -> Recalcular e substituir somente essa chave."""
|
||||
input_data = {
|
||||
"articles": [
|
||||
{
|
||||
"titulo": "Teste",
|
||||
"selected_extractor": "trafilatura",
|
||||
"trafilatura": {"text": "lixo sem sentido", "error": None},
|
||||
"newspaper4k": {
|
||||
"text": "conteúdo correto compartilhado por dois motores",
|
||||
"error": None,
|
||||
},
|
||||
"readability": {
|
||||
"cleaned_text": "conteúdo correto compartilhado por dois motores",
|
||||
"error": None,
|
||||
},
|
||||
}
|
||||
]
|
||||
}
|
||||
in_file = tmp_path / "artigos.json"
|
||||
in_file.write_text(json.dumps(input_data, ensure_ascii=False), encoding="utf-8")
|
||||
|
||||
res = process_batch(in_file)
|
||||
out_file = Path(res.output_file)
|
||||
assert out_file.exists()
|
||||
|
||||
with open(out_file, "r", encoding="utf-8") as f:
|
||||
out_data = json.load(f)
|
||||
|
||||
assert out_data["articles"][0]["selected_extractor"] == "newspaper4k"
|
||||
|
||||
|
||||
def test_ct_013_empty_articles_list(tmp_path: Path):
|
||||
"""CT-013: articles está vazio -> Gerar saída válida com articles vazio."""
|
||||
input_data = {"metadata": "info", "articles": []}
|
||||
in_file = tmp_path / "empty.json"
|
||||
in_file.write_text(json.dumps(input_data), encoding="utf-8")
|
||||
|
||||
res = process_batch(in_file)
|
||||
out_file = Path(res.output_file)
|
||||
assert out_file.exists()
|
||||
|
||||
with open(out_file, "r", encoding="utf-8") as f:
|
||||
out_data = json.load(f)
|
||||
|
||||
assert out_data["metadata"] == "info"
|
||||
assert out_data["articles"] == []
|
||||
assert res.total_articles == 0
|
||||
assert res.processed_count == 0
|
||||
|
||||
|
||||
def test_ct_014_invalid_json_fails_atomically(tmp_path: Path):
|
||||
"""CT-014: JSON inválido -> Não gerar saída."""
|
||||
in_file = tmp_path / "invalid.json"
|
||||
in_file.write_text("{articles: [ malformed json", encoding="utf-8")
|
||||
expected_out = tmp_path / "invalid_selected.json"
|
||||
|
||||
with pytest.raises(ValueError, match="JSON inválido"):
|
||||
process_batch(in_file)
|
||||
|
||||
assert not expected_out.exists()
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Testes de Integração
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_integration_reference_file(tmp_path: Path):
|
||||
"""Testa o processamento em lote completo sobre o arquivo real out/river_plate_extracted.json."""
|
||||
ref_file = Path("out/river_plate_extracted.json")
|
||||
if not ref_file.exists():
|
||||
pytest.skip(
|
||||
"Arquivo out/river_plate_extracted.json não encontrado para teste de integração."
|
||||
)
|
||||
|
||||
out_file = tmp_path / "river_plate_extracted_selected.json"
|
||||
result = process_batch(ref_file, output_path=out_file, verbose=True)
|
||||
|
||||
assert result.total_articles == 20
|
||||
assert result.processed_count == 20
|
||||
assert out_file.exists()
|
||||
|
||||
with open(out_file, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
assert len(data["articles"]) == 20
|
||||
for art in data["articles"]:
|
||||
assert "selected_extractor" in art
|
||||
assert art["selected_extractor"] in ["trafilatura", "newspaper4k", "readability"]
|
||||
|
||||
# Validar a distribuição exata conforme o algoritmo do PRD
|
||||
assert result.selection_distribution == {
|
||||
"newspaper4k": 9,
|
||||
"readability": 9,
|
||||
"trafilatura": 2,
|
||||
}
|
||||
|
||||
|
||||
def test_article_1_regression_technical_tie_markdown_images():
|
||||
"""
|
||||
Teste de regressão para o Artigo 1 (Los puntajes de River vs. Independiente Santa Fe):
|
||||
Valida que com o descarte de imagens Markdown , Readability e Newspaper4k
|
||||
entram em empate técnico (diff <= 0.03) e Newspaper4k vence por possuir menor quantidade
|
||||
de shingles (1009 vs 1057).
|
||||
"""
|
||||
ref_file = Path("out/river_plate_extracted.json")
|
||||
if not ref_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
|
||||
|
||||
with open(ref_file, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
art1 = data["articles"][0]
|
||||
result = select_article_extractor(art1, article_index=0)
|
||||
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
assert result.selection_reason == "technical_tie_smallest_shingles"
|
||||
|
||||
cand_news = result.candidates[ExtractorName.NEWSPAPER4K]
|
||||
cand_read = result.candidates[ExtractorName.READABILITY]
|
||||
cand_traf = result.candidates[ExtractorName.TRAFILATURA]
|
||||
|
||||
assert cand_news.shingle_count == 1009
|
||||
assert cand_read.shingle_count == 1057
|
||||
assert cand_traf.shingle_count == 1091
|
||||
|
||||
# Diferença para o maior score <= 0.03 (empate técnico)
|
||||
max_score = max(cand_news.score, cand_read.score, cand_traf.score)
|
||||
assert (max_score - cand_news.score) <= 0.03
|
||||
assert (max_score - cand_read.score) <= 0.03
|
||||
|
||||
|
||||
def test_integration_large_batch_determinism(tmp_path: Path):
|
||||
"""Gera um lote de 100 artigos sintéticos e verifica 100% de repetibilidade determinística entre 2 execuções."""
|
||||
articles = []
|
||||
for i in range(100):
|
||||
if i % 4 == 0:
|
||||
art = {
|
||||
"trafilatura": {
|
||||
"text": f"artigo numero {i} sobre futebol internacional no estadio",
|
||||
"error": None,
|
||||
},
|
||||
"newspaper4k": {
|
||||
"text": f"artigo numero {i} sobre futebol internacional no estadio",
|
||||
"error": None,
|
||||
},
|
||||
"readability": {"cleaned_text": "sem relacao", "error": None},
|
||||
}
|
||||
elif i % 4 == 1:
|
||||
art = {
|
||||
"trafilatura": {"text": None, "error": "timeout"},
|
||||
"newspaper4k": {"text": f"noticia exclusiva {i} com detalhes", "error": None},
|
||||
"readability": {
|
||||
"cleaned_text": f"noticia exclusiva {i} com detalhes",
|
||||
"error": None,
|
||||
},
|
||||
}
|
||||
elif i % 4 == 2:
|
||||
art = {
|
||||
"trafilatura": {"text": f"texto a {i}", "error": None},
|
||||
"newspaper4k": {"text": f"texto b diferente {i}", "error": None},
|
||||
"readability": {"cleaned_text": f"texto c terceiro {i}", "error": None},
|
||||
}
|
||||
else:
|
||||
art = {
|
||||
"trafilatura": {"text": None, "error": "err"},
|
||||
"newspaper4k": {"text": "", "error": "err"},
|
||||
"readability": {"cleaned_text": None, "error": "err"},
|
||||
}
|
||||
articles.append(art)
|
||||
|
||||
batch_payload = {"articles": articles}
|
||||
in_file = tmp_path / "large_batch.json"
|
||||
in_file.write_text(json.dumps(batch_payload, ensure_ascii=False), encoding="utf-8")
|
||||
|
||||
out1 = tmp_path / "large_batch_run1.json"
|
||||
out2 = tmp_path / "large_batch_run2.json"
|
||||
|
||||
res1 = process_batch(in_file, output_path=out1)
|
||||
res2 = process_batch(in_file, output_path=out2)
|
||||
|
||||
assert res1.processed_count == 100
|
||||
assert res2.processed_count == 100
|
||||
|
||||
# Os resultados devem ser 100% idênticos
|
||||
selections1 = [s.selected_extractor for s in res1.selections]
|
||||
selections2 = [s.selected_extractor for s in res2.selections]
|
||||
assert selections1 == selections2
|
||||
|
||||
|
||||
def test_integration_pipeline_downstream_consumer(tmp_path: Path):
|
||||
"""
|
||||
Testa a integração end-to-end do pipeline downstream:
|
||||
Lê o JSON enriquecido com selected_extractor, recupera o conteúdo do extrator vencedor e
|
||||
garante que o texto está higienizado e pronto para os classificadores NLP.
|
||||
"""
|
||||
ref_file = Path("out/river_plate_extracted.json")
|
||||
if not ref_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
|
||||
|
||||
out_file = tmp_path / "downstream_test.json"
|
||||
process_batch(ref_file, output_path=out_file)
|
||||
|
||||
with open(out_file, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
for idx, art in enumerate(data["articles"]):
|
||||
winner = art["selected_extractor"]
|
||||
assert winner in ["trafilatura", "newspaper4k", "readability"]
|
||||
|
||||
# Recuperar texto do extrator vencedor conforme mapeamento do PRD
|
||||
if winner == "trafilatura":
|
||||
chosen_text = art["trafilatura"]["text"]
|
||||
elif winner == "newspaper4k":
|
||||
chosen_text = art["newspaper4k"]["text"]
|
||||
elif winner == "readability":
|
||||
chosen_text = art["readability"]["cleaned_text"]
|
||||
|
||||
assert isinstance(chosen_text, str)
|
||||
assert len(chosen_text.strip()) > 0
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Testes End-to-End (E2E) via Subprocess CLI
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_e2e_cli_subprocess_real_execution(tmp_path: Path):
|
||||
"""E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha de comando."""
|
||||
ref_file = Path("out/river_plate_extracted.json")
|
||||
if not ref_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado para teste E2E.")
|
||||
|
||||
out_file = tmp_path / "e2e_river_plate_selected.json"
|
||||
script_path = Path("scripts/select_article_extractor.py").resolve()
|
||||
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(script_path),
|
||||
str(ref_file.resolve()),
|
||||
"-o",
|
||||
str(out_file),
|
||||
"--verbose",
|
||||
"--indent",
|
||||
"2",
|
||||
]
|
||||
|
||||
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||
|
||||
assert proc.returncode == 0
|
||||
assert out_file.exists()
|
||||
|
||||
# Validar saída estruturada JSON do stdout
|
||||
summary = json.loads(proc.stdout)
|
||||
assert summary["status"] == "success"
|
||||
assert summary["total_articles"] == 20
|
||||
assert summary["processed_count"] == 20
|
||||
assert "distribution" in summary
|
||||
|
||||
# Validar que logs de verbose foram emitidos no stderr
|
||||
assert "[Artigo #001]" in proc.stderr
|
||||
assert "[Artigo #020]" in proc.stderr
|
||||
|
||||
|
||||
def test_e2e_cli_subprocess_default_naming(tmp_path: Path):
|
||||
"""E2E: Executa CLI sem a flag -o e valida criação automática de <nome>_selected.json."""
|
||||
sample_data = {
|
||||
"articles": [
|
||||
{
|
||||
"titulo": "Artigo Automático",
|
||||
"trafilatura": {"text": "conteúdo padrão", "error": None},
|
||||
"newspaper4k": {"text": "conteúdo padrão", "error": None},
|
||||
"readability": {"cleaned_text": "conteúdo padrão", "error": None},
|
||||
}
|
||||
]
|
||||
}
|
||||
in_file = tmp_path / "my_news.json"
|
||||
in_file.write_text(json.dumps(sample_data), encoding="utf-8")
|
||||
expected_out = tmp_path / "my_news_selected.json"
|
||||
|
||||
script_path = Path("scripts/select_article_extractor.py").resolve()
|
||||
cmd = [sys.executable, str(script_path), str(in_file)]
|
||||
|
||||
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||
assert proc.returncode == 0
|
||||
assert expected_out.exists()
|
||||
|
||||
with open(expected_out, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
assert data["articles"][0]["selected_extractor"] == "newspaper4k"
|
||||
|
||||
|
||||
def test_e2e_cli_subprocess_invalid_input(tmp_path: Path):
|
||||
"""E2E: Executa CLI com JSON inválido e valida código de saída e erro no stderr."""
|
||||
invalid_file = tmp_path / "broken.json"
|
||||
invalid_file.write_text("not a valid json {", encoding="utf-8")
|
||||
|
||||
script_path = Path("scripts/select_article_extractor.py").resolve()
|
||||
cmd = [sys.executable, str(script_path), str(invalid_file)]
|
||||
|
||||
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||
assert proc.returncode == 2
|
||||
assert "ERRO DE VALIDAÇÃO" in proc.stderr
|
||||
|
||||
|
||||
def test_e2e_cli_subprocess_missing_file():
|
||||
"""E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr."""
|
||||
script_path = Path("scripts/select_article_extractor.py").resolve()
|
||||
cmd = [sys.executable, str(script_path), "non_existent_file_12345.json"]
|
||||
|
||||
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||
assert proc.returncode == 1
|
||||
assert "ERRO DE ARQUIVO" in proc.stderr
|
||||
Reference in New Issue
Block a user