616 lines
24 KiB
Python
616 lines
24 KiB
Python
"""
|
|
Suíte de Testes Automatizados para o Seletor Determinístico de Extrator.
|
|
|
|
Cobre 100% dos Casos de Teste Obrigatórios do PRD (CT-001 a CT-014), testes unitários
|
|
de normalização e shingles, testes de integração de lote e testes E2E via subprocess.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import subprocess
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from scripts.select_article_extractor import (
|
|
ExtractorName,
|
|
generate_shingles,
|
|
normalize_text,
|
|
process_batch,
|
|
select_article_extractor,
|
|
)
|
|
|
|
# ==============================================================================
|
|
# Testes Unitários de Normalização e Tokenização
|
|
# ==============================================================================
|
|
|
|
|
|
def test_normalize_text_empty_and_invalid():
|
|
assert normalize_text(None) == []
|
|
assert normalize_text("") == []
|
|
assert normalize_text(" \n\t ") == []
|
|
assert normalize_text(12345) == []
|
|
|
|
|
|
def test_normalize_text_html_entities_and_tags():
|
|
raw = "<p>El & <b>futebol</b> mundial "está" mudando.</p>"
|
|
tokens = normalize_text(raw)
|
|
assert tokens == ["el", "futebol", "mundial", "está", "mudando"]
|
|
|
|
|
|
def test_normalize_text_markdown_links():
|
|
raw = "Veja mais no [Portal de Notícias](https://example.com/noticias) hoje."
|
|
tokens = normalize_text(raw)
|
|
assert tokens == ["veja", "mais", "no", "portal", "de", "notícias", "hoje"]
|
|
|
|
|
|
def test_normalize_text_markdown_images_stripped_while_links_preserved():
|
|
"""Garante que marcação de imagem Markdown  seja descartada e link [texto](url) seja preservado."""
|
|
raw = (
|
|
"Texto inicial do artigo. "
|
|
" "
|
|
"Mais texto com [link importante](https://example.com/pagina) e outra "
|
|
" informação."
|
|
)
|
|
tokens = normalize_text(raw)
|
|
assert "legenda" not in tokens
|
|
assert "foto" not in tokens
|
|
assert "imagem" not in tokens
|
|
assert "link" in tokens
|
|
assert "importante" in tokens
|
|
assert tokens == [
|
|
"texto",
|
|
"inicial",
|
|
"do",
|
|
"artigo",
|
|
"mais",
|
|
"texto",
|
|
"com",
|
|
"link",
|
|
"importante",
|
|
"e",
|
|
"outra",
|
|
"informação",
|
|
]
|
|
|
|
|
|
def test_normalize_text_nfkc_unicode_and_punctuation():
|
|
# Caracteres combinados e pontuação
|
|
raw = "River Plate venceu por 3-0! (Com gol de pênalti & golaço de falta)."
|
|
tokens = normalize_text(raw)
|
|
assert tokens == [
|
|
"river",
|
|
"plate",
|
|
"venceu",
|
|
"por",
|
|
"3",
|
|
"0",
|
|
"com",
|
|
"gol",
|
|
"de",
|
|
"pênalti",
|
|
"golaço",
|
|
"de",
|
|
"falta",
|
|
]
|
|
|
|
|
|
# ==============================================================================
|
|
# Testes Unitários de Shingles
|
|
# ==============================================================================
|
|
|
|
|
|
def test_generate_shingles_sliding_window():
|
|
tokens = ["um", "dois", "três", "quatro", "cinco", "seis"]
|
|
shingles = generate_shingles(tokens, window_size=5)
|
|
assert len(shingles) == 2
|
|
assert ("um", "dois", "três", "quatro", "cinco") in shingles
|
|
assert ("dois", "três", "quatro", "cinco", "seis") in shingles
|
|
|
|
|
|
def test_generate_shingles_short_text():
|
|
# Entre 1 e 4 tokens deve gerar 1 único shingle com a tupla completa
|
|
tokens = ["river", "plate", "campeão"]
|
|
shingles = generate_shingles(tokens, window_size=5)
|
|
assert len(shingles) == 1
|
|
assert ("river", "plate", "campeão") in shingles
|
|
|
|
|
|
def test_generate_shingles_empty():
|
|
assert generate_shingles([]) == set()
|
|
|
|
|
|
# ==============================================================================
|
|
# Casos de Teste Obrigatórios do PRD (§12: CT-001 a CT-014)
|
|
# ==============================================================================
|
|
|
|
|
|
def test_ct_001_three_candidates_clear_winner():
|
|
"""CT-001: Três candidatos com consenso e um vencedor claro -> Selecionar o maior score."""
|
|
base = "river plate venceu o clássico ontem a noite no estádio monumental"
|
|
article = {
|
|
"trafilatura": {"text": base, "error": None},
|
|
"newspaper4k": {"text": base, "error": None},
|
|
"readability": {
|
|
"cleaned_text": "texto completamente diferente sem nenhuma relação",
|
|
"error": None,
|
|
},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor in (ExtractorName.NEWSPAPER4K, ExtractorName.TRAFILATURA)
|
|
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
|
|
|
|
|
def test_ct_002_technical_tie_smallest_shingles():
|
|
"""CT-002: Dois ou mais candidatos dentro de 0,03 do maior score -> Selecionar o de menor quantidade de shingles."""
|
|
tokens_comuns = (
|
|
"o rio de janeiro continua lindo e sempre maravilhoso em todas as estações do ano"
|
|
)
|
|
article = {
|
|
"trafilatura": {"text": tokens_comuns, "error": None},
|
|
"readability": {
|
|
"cleaned_text": tokens_comuns + " propaganda extra adicionada no fim",
|
|
"error": None,
|
|
},
|
|
"newspaper4k": {"text": tokens_comuns, "error": None},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
|
|
|
|
|
def test_ct_003_technical_tie_priority_fallback():
|
|
"""CT-003: Empate técnico e mesma quantidade de shingles -> Aplicar prioridade final (newspaper4k > readability > trafilatura)."""
|
|
texto = "o time jogou muito bem durante toda a partida de futebol"
|
|
article = {
|
|
"trafilatura": {"text": texto, "error": None},
|
|
"readability": {"cleaned_text": texto, "error": None},
|
|
"newspaper4k": {"text": texto, "error": None},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
|
|
|
# Agora sem newspaper4k ativo (somente readability e trafilatura idênticos)
|
|
article_two = {
|
|
"trafilatura": {"text": texto, "error": None},
|
|
"readability": {"cleaned_text": texto, "error": None},
|
|
"newspaper4k": {"text": None, "error": "Crash"},
|
|
}
|
|
result_two = select_article_extractor(article_two)
|
|
assert result_two.selected_extractor == ExtractorName.READABILITY
|
|
|
|
|
|
def test_ct_004_three_candidates_no_consensus():
|
|
"""CT-004: Três candidatos sem consenso -> Selecionar a quantidade mediana de shingles."""
|
|
t1 = "alfa bravo charlie delta echo foxtrot golf hotel india juliet" # 10 tokens -> 6 shingles
|
|
t2 = "kilo lima mike november oscar papa quebec romeo sierra tango uniform victor" # 12 tokens -> 8 shingles (MEDIANA)
|
|
t3 = "whiskey xray yankee zulu zero one two three four five six seven eight nine" # 14 tokens -> 10 shingles
|
|
article = {
|
|
"trafilatura": {"text": t1, "error": None},
|
|
"readability": {"cleaned_text": t2, "error": None},
|
|
"newspaper4k": {"text": t3, "error": None},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor == ExtractorName.READABILITY
|
|
assert result.selection_reason == "no_consensus_median_shingles"
|
|
|
|
|
|
def test_ct_005_two_candidates_no_consensus():
|
|
"""CT-005: Dois candidatos sem consenso -> Selecionar a maior quantidade de shingles."""
|
|
t_short = "alfa bravo charlie delta echo foxtrot" # 6 tokens -> 2 shingles
|
|
t_long = (
|
|
"kilo lima mike november oscar papa quebec romeo sierra tango" # 10 tokens -> 6 shingles
|
|
)
|
|
article = {
|
|
"trafilatura": {"text": t_short, "error": None},
|
|
"newspaper4k": {"text": t_long, "error": None},
|
|
"readability": {"cleaned_text": None, "error": "Not extracted"},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
|
assert result.selection_reason == "no_consensus_max_shingles"
|
|
|
|
|
|
def test_ct_006_single_usable_candidate():
|
|
"""CT-006: Somente um candidato utilizável -> Selecionar esse candidato."""
|
|
article = {
|
|
"trafilatura": {
|
|
"text": "conteúdo válido e utilizável extraído com sucesso aqui",
|
|
"error": None,
|
|
},
|
|
"newspaper4k": {"text": "", "error": None},
|
|
"readability": {"cleaned_text": None, "error": "Timeout error"},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor == ExtractorName.TRAFILATURA
|
|
assert result.selection_reason == "single_usable_candidate"
|
|
|
|
|
|
def test_ct_007_degraded_candidates_only():
|
|
"""CT-007: Nenhum utilizável, mas existe candidato degradado -> Executar o algoritmo somente com os degradados."""
|
|
texto_comum = "artigo relevante sobre economia global e finanças internacionais com detalhes"
|
|
article = {
|
|
"trafilatura": {"text": texto_comum, "error": "Warning: partial parse"},
|
|
"newspaper4k": {"text": texto_comum, "error": "HTTP 403 partial"},
|
|
"readability": {"cleaned_text": None, "error": "Fatal exception"},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
|
|
|
|
|
def test_ct_008_all_candidates_unavailable():
|
|
"""CT-008: Todos os candidatos indisponíveis -> Selecionar newspaper4k."""
|
|
article = {
|
|
"trafilatura": {"text": None, "error": "Error"},
|
|
"newspaper4k": {"text": "", "error": "Empty"},
|
|
"readability": {"cleaned_text": " ", "error": "Blank"},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
|
assert result.selection_reason == "fallback_all_unavailable"
|
|
|
|
|
|
def test_ct_009_small_fragment_loses_due_to_low_coverage():
|
|
"""CT-009: Readability retorna apenas um fragmento pequeno enquanto os outros concordam -> Perde por baixa cobertura."""
|
|
full_text = (
|
|
"o presidente da república anunciou novas medidas econômicas para conter a inflação "
|
|
"e estimular o crescimento industrial em todo o território nacional durante o pronunciamento oficial"
|
|
)
|
|
small_fragment = "o presidente da república anunciou"
|
|
article = {
|
|
"trafilatura": {"text": full_text, "error": None},
|
|
"newspaper4k": {"text": full_text, "error": None},
|
|
"readability": {"cleaned_text": small_fragment, "error": None},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor in (ExtractorName.TRAFILATURA, ExtractorName.NEWSPAPER4K)
|
|
assert result.selected_extractor != ExtractorName.READABILITY
|
|
|
|
|
|
def test_ct_010_excessive_boilerplate_loses_due_to_low_support():
|
|
"""CT-010: Um candidato contém o conteúdo comum e muito conteúdo excedente -> Perde suporte e reduz pontuação."""
|
|
common_content = (
|
|
"notícia oficial com dados apurados sobre a operação policial realizada nesta manhã"
|
|
)
|
|
massive_boilerplate = common_content + (
|
|
" compartilhe no facebook twitter whatsapp veja também esportes receitas horóscopo política "
|
|
" e assine nossa newsletter diária para receber mais novidades sobre culinária e fofocas"
|
|
)
|
|
article = {
|
|
"trafilatura": {"text": common_content, "error": None},
|
|
"readability": {"cleaned_text": common_content, "error": None},
|
|
"newspaper4k": {"text": massive_boilerplate, "error": None},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor == ExtractorName.READABILITY
|
|
|
|
|
|
def test_ct_011_partial_content_loses_due_to_low_coverage():
|
|
"""CT-011: Um candidato contém somente parte do conteúdo comum -> Perde cobertura e reduz pontuação."""
|
|
full_content = "primeiro parágrafo do artigo completo segundo parágrafo com explicações terceiro parágrafo final"
|
|
half_content = "primeiro parágrafo do artigo completo"
|
|
article = {
|
|
"newspaper4k": {"text": full_content, "error": None},
|
|
"readability": {"cleaned_text": full_content, "error": None},
|
|
"trafilatura": {"text": half_content, "error": None},
|
|
}
|
|
result = select_article_extractor(article)
|
|
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
|
|
|
|
|
def test_ct_012_recalculate_existing_selected_extractor(tmp_path: Path):
|
|
"""CT-012: A entrada já contém selected_extractor -> Recalcular e substituir somente essa chave."""
|
|
input_data = {
|
|
"articles": [
|
|
{
|
|
"titulo": "Teste",
|
|
"selected_extractor": "trafilatura",
|
|
"trafilatura": {"text": "lixo sem sentido", "error": None},
|
|
"newspaper4k": {
|
|
"text": "conteúdo correto compartilhado por dois motores",
|
|
"error": None,
|
|
},
|
|
"readability": {
|
|
"cleaned_text": "conteúdo correto compartilhado por dois motores",
|
|
"error": None,
|
|
},
|
|
}
|
|
]
|
|
}
|
|
in_file = tmp_path / "artigos.json"
|
|
in_file.write_text(json.dumps(input_data, ensure_ascii=False), encoding="utf-8")
|
|
|
|
res = process_batch(in_file)
|
|
out_file = Path(res.output_file)
|
|
assert out_file.exists()
|
|
|
|
with open(out_file, "r", encoding="utf-8") as f:
|
|
out_data = json.load(f)
|
|
|
|
assert out_data["articles"][0]["selected_extractor"] == "newspaper4k"
|
|
|
|
|
|
def test_ct_013_empty_articles_list(tmp_path: Path):
|
|
"""CT-013: articles está vazio -> Gerar saída válida com articles vazio."""
|
|
input_data = {"metadata": "info", "articles": []}
|
|
in_file = tmp_path / "empty.json"
|
|
in_file.write_text(json.dumps(input_data), encoding="utf-8")
|
|
|
|
res = process_batch(in_file)
|
|
out_file = Path(res.output_file)
|
|
assert out_file.exists()
|
|
|
|
with open(out_file, "r", encoding="utf-8") as f:
|
|
out_data = json.load(f)
|
|
|
|
assert out_data["metadata"] == "info"
|
|
assert out_data["articles"] == []
|
|
assert res.total_articles == 0
|
|
assert res.processed_count == 0
|
|
|
|
|
|
def test_ct_014_invalid_json_fails_atomically(tmp_path: Path):
|
|
"""CT-014: JSON inválido -> Não gerar saída."""
|
|
in_file = tmp_path / "invalid.json"
|
|
in_file.write_text("{articles: [ malformed json", encoding="utf-8")
|
|
expected_out = tmp_path / "invalid_selected.json"
|
|
|
|
with pytest.raises(ValueError, match="JSON inválido"):
|
|
process_batch(in_file)
|
|
|
|
assert not expected_out.exists()
|
|
|
|
|
|
# ==============================================================================
|
|
# Testes de Integração
|
|
# ==============================================================================
|
|
|
|
|
|
def test_integration_reference_file(tmp_path: Path):
|
|
"""Testa o processamento em lote completo sobre o arquivo real out/river_plate_extracted.json."""
|
|
ref_file = Path("out/river_plate_extracted.json")
|
|
if not ref_file.exists():
|
|
pytest.skip(
|
|
"Arquivo out/river_plate_extracted.json não encontrado para teste de integração."
|
|
)
|
|
|
|
out_file = tmp_path / "river_plate_extracted_selected.json"
|
|
result = process_batch(ref_file, output_path=out_file, verbose=True)
|
|
|
|
assert result.total_articles == 20
|
|
assert result.processed_count == 20
|
|
assert out_file.exists()
|
|
|
|
with open(out_file, "r", encoding="utf-8") as f:
|
|
data = json.load(f)
|
|
|
|
assert len(data["articles"]) == 20
|
|
for art in data["articles"]:
|
|
assert "selected_extractor" in art
|
|
assert art["selected_extractor"] in ["trafilatura", "newspaper4k", "readability"]
|
|
|
|
# Validar a distribuição exata conforme o algoritmo do PRD
|
|
assert result.selection_distribution == {
|
|
"newspaper4k": 9,
|
|
"readability": 9,
|
|
"trafilatura": 2,
|
|
}
|
|
|
|
|
|
def test_article_1_regression_technical_tie_markdown_images():
|
|
"""
|
|
Teste de regressão para o Artigo 1 (Los puntajes de River vs. Independiente Santa Fe):
|
|
Valida que com o descarte de imagens Markdown , Readability e Newspaper4k
|
|
entram em empate técnico (diff <= 0.03) e Newspaper4k vence por possuir menor quantidade
|
|
de shingles (1009 vs 1057).
|
|
"""
|
|
ref_file = Path("out/river_plate_extracted.json")
|
|
if not ref_file.exists():
|
|
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
|
|
|
|
with open(ref_file, "r", encoding="utf-8") as f:
|
|
data = json.load(f)
|
|
|
|
art1 = data["articles"][0]
|
|
result = select_article_extractor(art1, article_index=0)
|
|
|
|
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
|
assert result.selection_reason == "technical_tie_smallest_shingles"
|
|
|
|
cand_news = result.candidates[ExtractorName.NEWSPAPER4K]
|
|
cand_read = result.candidates[ExtractorName.READABILITY]
|
|
cand_traf = result.candidates[ExtractorName.TRAFILATURA]
|
|
|
|
assert cand_news.shingle_count == 1009
|
|
assert cand_read.shingle_count == 1057
|
|
assert cand_traf.shingle_count == 1091
|
|
|
|
# Diferença para o maior score <= 0.03 (empate técnico)
|
|
max_score = max(cand_news.score, cand_read.score, cand_traf.score)
|
|
assert (max_score - cand_news.score) <= 0.03
|
|
assert (max_score - cand_read.score) <= 0.03
|
|
|
|
|
|
def test_integration_large_batch_determinism(tmp_path: Path):
|
|
"""Gera um lote de 100 artigos sintéticos e verifica 100% de repetibilidade determinística entre 2 execuções."""
|
|
articles = []
|
|
for i in range(100):
|
|
if i % 4 == 0:
|
|
art = {
|
|
"trafilatura": {
|
|
"text": f"artigo numero {i} sobre futebol internacional no estadio",
|
|
"error": None,
|
|
},
|
|
"newspaper4k": {
|
|
"text": f"artigo numero {i} sobre futebol internacional no estadio",
|
|
"error": None,
|
|
},
|
|
"readability": {"cleaned_text": "sem relacao", "error": None},
|
|
}
|
|
elif i % 4 == 1:
|
|
art = {
|
|
"trafilatura": {"text": None, "error": "timeout"},
|
|
"newspaper4k": {"text": f"noticia exclusiva {i} com detalhes", "error": None},
|
|
"readability": {
|
|
"cleaned_text": f"noticia exclusiva {i} com detalhes",
|
|
"error": None,
|
|
},
|
|
}
|
|
elif i % 4 == 2:
|
|
art = {
|
|
"trafilatura": {"text": f"texto a {i}", "error": None},
|
|
"newspaper4k": {"text": f"texto b diferente {i}", "error": None},
|
|
"readability": {"cleaned_text": f"texto c terceiro {i}", "error": None},
|
|
}
|
|
else:
|
|
art = {
|
|
"trafilatura": {"text": None, "error": "err"},
|
|
"newspaper4k": {"text": "", "error": "err"},
|
|
"readability": {"cleaned_text": None, "error": "err"},
|
|
}
|
|
articles.append(art)
|
|
|
|
batch_payload = {"articles": articles}
|
|
in_file = tmp_path / "large_batch.json"
|
|
in_file.write_text(json.dumps(batch_payload, ensure_ascii=False), encoding="utf-8")
|
|
|
|
out1 = tmp_path / "large_batch_run1.json"
|
|
out2 = tmp_path / "large_batch_run2.json"
|
|
|
|
res1 = process_batch(in_file, output_path=out1)
|
|
res2 = process_batch(in_file, output_path=out2)
|
|
|
|
assert res1.processed_count == 100
|
|
assert res2.processed_count == 100
|
|
|
|
# Os resultados devem ser 100% idênticos
|
|
selections1 = [s.selected_extractor for s in res1.selections]
|
|
selections2 = [s.selected_extractor for s in res2.selections]
|
|
assert selections1 == selections2
|
|
|
|
|
|
def test_integration_pipeline_downstream_consumer(tmp_path: Path):
|
|
"""
|
|
Testa a integração end-to-end do pipeline downstream:
|
|
Lê o JSON enriquecido com selected_extractor, recupera o conteúdo do extrator vencedor e
|
|
garante que o texto está higienizado e pronto para os classificadores NLP.
|
|
"""
|
|
ref_file = Path("out/river_plate_extracted.json")
|
|
if not ref_file.exists():
|
|
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
|
|
|
|
out_file = tmp_path / "downstream_test.json"
|
|
process_batch(ref_file, output_path=out_file)
|
|
|
|
with open(out_file, "r", encoding="utf-8") as f:
|
|
data = json.load(f)
|
|
|
|
for idx, art in enumerate(data["articles"]):
|
|
winner = art["selected_extractor"]
|
|
assert winner in ["trafilatura", "newspaper4k", "readability"]
|
|
|
|
# Recuperar texto do extrator vencedor conforme mapeamento do PRD
|
|
if winner == "trafilatura":
|
|
chosen_text = art["trafilatura"]["text"]
|
|
elif winner == "newspaper4k":
|
|
chosen_text = art["newspaper4k"]["text"]
|
|
elif winner == "readability":
|
|
chosen_text = art["readability"]["cleaned_text"]
|
|
|
|
assert isinstance(chosen_text, str)
|
|
assert len(chosen_text.strip()) > 0
|
|
|
|
|
|
# ==============================================================================
|
|
# Testes End-to-End (E2E) via Subprocess CLI
|
|
# ==============================================================================
|
|
|
|
|
|
def test_e2e_cli_subprocess_real_execution(tmp_path: Path):
|
|
"""E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha de comando."""
|
|
ref_file = Path("out/river_plate_extracted.json")
|
|
if not ref_file.exists():
|
|
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado para teste E2E.")
|
|
|
|
out_file = tmp_path / "e2e_river_plate_selected.json"
|
|
script_path = Path("scripts/select_article_extractor.py").resolve()
|
|
|
|
cmd = [
|
|
sys.executable,
|
|
str(script_path),
|
|
str(ref_file.resolve()),
|
|
"-o",
|
|
str(out_file),
|
|
"--verbose",
|
|
"--indent",
|
|
"2",
|
|
]
|
|
|
|
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
|
|
|
assert proc.returncode == 0
|
|
assert out_file.exists()
|
|
|
|
# Validar saída estruturada JSON do stdout
|
|
summary = json.loads(proc.stdout)
|
|
assert summary["status"] == "success"
|
|
assert summary["total_articles"] == 20
|
|
assert summary["processed_count"] == 20
|
|
assert "distribution" in summary
|
|
|
|
# Validar que logs de verbose foram emitidos no stderr
|
|
assert "[Artigo #001]" in proc.stderr
|
|
assert "[Artigo #020]" in proc.stderr
|
|
|
|
|
|
def test_e2e_cli_subprocess_default_naming(tmp_path: Path):
|
|
"""E2E: Executa CLI sem a flag -o e valida criação automática de <nome>_selected.json."""
|
|
sample_data = {
|
|
"articles": [
|
|
{
|
|
"titulo": "Artigo Automático",
|
|
"trafilatura": {"text": "conteúdo padrão", "error": None},
|
|
"newspaper4k": {"text": "conteúdo padrão", "error": None},
|
|
"readability": {"cleaned_text": "conteúdo padrão", "error": None},
|
|
}
|
|
]
|
|
}
|
|
in_file = tmp_path / "my_news.json"
|
|
in_file.write_text(json.dumps(sample_data), encoding="utf-8")
|
|
expected_out = tmp_path / "my_news_selected.json"
|
|
|
|
script_path = Path("scripts/select_article_extractor.py").resolve()
|
|
cmd = [sys.executable, str(script_path), str(in_file)]
|
|
|
|
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
|
assert proc.returncode == 0
|
|
assert expected_out.exists()
|
|
|
|
with open(expected_out, "r", encoding="utf-8") as f:
|
|
data = json.load(f)
|
|
assert data["articles"][0]["selected_extractor"] == "newspaper4k"
|
|
|
|
|
|
def test_e2e_cli_subprocess_invalid_input(tmp_path: Path):
|
|
"""E2E: Executa CLI com JSON inválido e valida código de saída e erro no stderr."""
|
|
invalid_file = tmp_path / "broken.json"
|
|
invalid_file.write_text("not a valid json {", encoding="utf-8")
|
|
|
|
script_path = Path("scripts/select_article_extractor.py").resolve()
|
|
cmd = [sys.executable, str(script_path), str(invalid_file)]
|
|
|
|
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
|
assert proc.returncode == 2
|
|
assert "ERRO DE VALIDAÇÃO" in proc.stderr
|
|
|
|
|
|
def test_e2e_cli_subprocess_missing_file():
|
|
"""E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr."""
|
|
script_path = Path("scripts/select_article_extractor.py").resolve()
|
|
cmd = [sys.executable, str(script_path), "non_existent_file_12345.json"]
|
|
|
|
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
|
assert proc.returncode == 1
|
|
assert "ERRO DE ARQUIVO" in proc.stderr
|