""" Suíte de Testes Automatizados para o Seletor Determinístico de Extrator. Cobre 100% dos Casos de Teste Obrigatórios do PRD (CT-001 a CT-014), testes unitários de normalização e shingles, testes de integração de lote e testes E2E via subprocess. """ from __future__ import annotations import json import subprocess import sys from pathlib import Path import pytest from scripts.select_article_extractor import ( ExtractorName, generate_shingles, normalize_text, process_batch, select_article_extractor, ) # ============================================================================== # Testes Unitários de Normalização e Tokenização # ============================================================================== def test_normalize_text_empty_and_invalid(): assert normalize_text(None) == [] assert normalize_text("") == [] assert normalize_text(" \n\t ") == [] assert normalize_text(12345) == [] def test_normalize_text_html_entities_and_tags(): raw = "

El & futebol mundial "está" mudando.

" tokens = normalize_text(raw) assert tokens == ["el", "futebol", "mundial", "está", "mudando"] def test_normalize_text_markdown_links(): raw = "Veja mais no [Portal de Notícias](https://example.com/noticias) hoje." tokens = normalize_text(raw) assert tokens == ["veja", "mais", "no", "portal", "de", "notícias", "hoje"] def test_normalize_text_markdown_images_stripped_while_links_preserved(): """Garante que marcação de imagem Markdown ![alt](url) seja descartada e link [texto](url) seja preservado.""" raw = ( "Texto inicial do artigo. " "![Legenda da foto e imagem](https://example.com/imagem.webp) " "Mais texto com [link importante](https://example.com/pagina) e outra " "![Outra foto](https://example.com/foto2.jpg) informação." ) tokens = normalize_text(raw) assert "legenda" not in tokens assert "foto" not in tokens assert "imagem" not in tokens assert "link" in tokens assert "importante" in tokens assert tokens == [ "texto", "inicial", "do", "artigo", "mais", "texto", "com", "link", "importante", "e", "outra", "informação", ] def test_normalize_text_nfkc_unicode_and_punctuation(): # Caracteres combinados e pontuação raw = "River Plate venceu por 3-0! (Com gol de pênalti & golaço de falta)." tokens = normalize_text(raw) assert tokens == [ "river", "plate", "venceu", "por", "3", "0", "com", "gol", "de", "pênalti", "golaço", "de", "falta", ] # ============================================================================== # Testes Unitários de Shingles # ============================================================================== def test_generate_shingles_sliding_window(): tokens = ["um", "dois", "três", "quatro", "cinco", "seis"] shingles = generate_shingles(tokens, window_size=5) assert len(shingles) == 2 assert ("um", "dois", "três", "quatro", "cinco") in shingles assert ("dois", "três", "quatro", "cinco", "seis") in shingles def test_generate_shingles_short_text(): # Entre 1 e 4 tokens deve gerar 1 único shingle com a tupla completa tokens = ["river", "plate", "campeão"] shingles = generate_shingles(tokens, window_size=5) assert len(shingles) == 1 assert ("river", "plate", "campeão") in shingles def test_generate_shingles_empty(): assert generate_shingles([]) == set() # ============================================================================== # Casos de Teste Obrigatórios do PRD (§12: CT-001 a CT-014) # ============================================================================== def test_ct_001_three_candidates_clear_winner(): """CT-001: Três candidatos com consenso e um vencedor claro -> Selecionar o maior score.""" base = "river plate venceu o clássico ontem a noite no estádio monumental" article = { "trafilatura": {"text": base, "error": None}, "newspaper4k": {"text": base, "error": None}, "readability": { "cleaned_text": "texto completamente diferente sem nenhuma relação", "error": None, }, } result = select_article_extractor(article) assert result.selected_extractor in (ExtractorName.NEWSPAPER4K, ExtractorName.TRAFILATURA) assert result.selected_extractor == ExtractorName.NEWSPAPER4K def test_ct_002_technical_tie_smallest_shingles(): """CT-002: Dois ou mais candidatos dentro de 0,03 do maior score -> Selecionar o de menor quantidade de shingles.""" tokens_comuns = ( "o rio de janeiro continua lindo e sempre maravilhoso em todas as estações do ano" ) article = { "trafilatura": {"text": tokens_comuns, "error": None}, "readability": { "cleaned_text": tokens_comuns + " propaganda extra adicionada no fim", "error": None, }, "newspaper4k": {"text": tokens_comuns, "error": None}, } result = select_article_extractor(article) assert result.selected_extractor == ExtractorName.NEWSPAPER4K def test_ct_003_technical_tie_priority_fallback(): """CT-003: Empate técnico e mesma quantidade de shingles -> Aplicar prioridade final (newspaper4k > readability > trafilatura).""" texto = "o time jogou muito bem durante toda a partida de futebol" article = { "trafilatura": {"text": texto, "error": None}, "readability": {"cleaned_text": texto, "error": None}, "newspaper4k": {"text": texto, "error": None}, } result = select_article_extractor(article) assert result.selected_extractor == ExtractorName.NEWSPAPER4K # Agora sem newspaper4k ativo (somente readability e trafilatura idênticos) article_two = { "trafilatura": {"text": texto, "error": None}, "readability": {"cleaned_text": texto, "error": None}, "newspaper4k": {"text": None, "error": "Crash"}, } result_two = select_article_extractor(article_two) assert result_two.selected_extractor == ExtractorName.READABILITY def test_ct_004_three_candidates_no_consensus(): """CT-004: Três candidatos sem consenso -> Selecionar a quantidade mediana de shingles.""" t1 = "alfa bravo charlie delta echo foxtrot golf hotel india juliet" # 10 tokens -> 6 shingles t2 = "kilo lima mike november oscar papa quebec romeo sierra tango uniform victor" # 12 tokens -> 8 shingles (MEDIANA) t3 = "whiskey xray yankee zulu zero one two three four five six seven eight nine" # 14 tokens -> 10 shingles article = { "trafilatura": {"text": t1, "error": None}, "readability": {"cleaned_text": t2, "error": None}, "newspaper4k": {"text": t3, "error": None}, } result = select_article_extractor(article) assert result.selected_extractor == ExtractorName.READABILITY assert result.selection_reason == "no_consensus_median_shingles" def test_ct_005_two_candidates_no_consensus(): """CT-005: Dois candidatos sem consenso -> Selecionar a maior quantidade de shingles.""" t_short = "alfa bravo charlie delta echo foxtrot" # 6 tokens -> 2 shingles t_long = ( "kilo lima mike november oscar papa quebec romeo sierra tango" # 10 tokens -> 6 shingles ) article = { "trafilatura": {"text": t_short, "error": None}, "newspaper4k": {"text": t_long, "error": None}, "readability": {"cleaned_text": None, "error": "Not extracted"}, } result = select_article_extractor(article) assert result.selected_extractor == ExtractorName.NEWSPAPER4K assert result.selection_reason == "no_consensus_max_shingles" def test_ct_006_single_usable_candidate(): """CT-006: Somente um candidato utilizável -> Selecionar esse candidato.""" article = { "trafilatura": { "text": "conteúdo válido e utilizável extraído com sucesso aqui", "error": None, }, "newspaper4k": {"text": "", "error": None}, "readability": {"cleaned_text": None, "error": "Timeout error"}, } result = select_article_extractor(article) assert result.selected_extractor == ExtractorName.TRAFILATURA assert result.selection_reason == "single_usable_candidate" def test_ct_007_degraded_candidates_only(): """CT-007: Nenhum utilizável, mas existe candidato degradado -> Executar o algoritmo somente com os degradados.""" texto_comum = "artigo relevante sobre economia global e finanças internacionais com detalhes" article = { "trafilatura": {"text": texto_comum, "error": "Warning: partial parse"}, "newspaper4k": {"text": texto_comum, "error": "HTTP 403 partial"}, "readability": {"cleaned_text": None, "error": "Fatal exception"}, } result = select_article_extractor(article) assert result.selected_extractor == ExtractorName.NEWSPAPER4K def test_ct_008_all_candidates_unavailable(): """CT-008: Todos os candidatos indisponíveis -> Selecionar newspaper4k.""" article = { "trafilatura": {"text": None, "error": "Error"}, "newspaper4k": {"text": "", "error": "Empty"}, "readability": {"cleaned_text": " ", "error": "Blank"}, } result = select_article_extractor(article) assert result.selected_extractor == ExtractorName.NEWSPAPER4K assert result.selection_reason == "fallback_all_unavailable" def test_ct_009_small_fragment_loses_due_to_low_coverage(): """CT-009: Readability retorna apenas um fragmento pequeno enquanto os outros concordam -> Perde por baixa cobertura.""" full_text = ( "o presidente da república anunciou novas medidas econômicas para conter a inflação " "e estimular o crescimento industrial em todo o território nacional durante o pronunciamento oficial" ) small_fragment = "o presidente da república anunciou" article = { "trafilatura": {"text": full_text, "error": None}, "newspaper4k": {"text": full_text, "error": None}, "readability": {"cleaned_text": small_fragment, "error": None}, } result = select_article_extractor(article) assert result.selected_extractor in (ExtractorName.TRAFILATURA, ExtractorName.NEWSPAPER4K) assert result.selected_extractor != ExtractorName.READABILITY def test_ct_010_excessive_boilerplate_loses_due_to_low_support(): """CT-010: Um candidato contém o conteúdo comum e muito conteúdo excedente -> Perde suporte e reduz pontuação.""" common_content = ( "notícia oficial com dados apurados sobre a operação policial realizada nesta manhã" ) massive_boilerplate = common_content + ( " compartilhe no facebook twitter whatsapp veja também esportes receitas horóscopo política " " e assine nossa newsletter diária para receber mais novidades sobre culinária e fofocas" ) article = { "trafilatura": {"text": common_content, "error": None}, "readability": {"cleaned_text": common_content, "error": None}, "newspaper4k": {"text": massive_boilerplate, "error": None}, } result = select_article_extractor(article) assert result.selected_extractor == ExtractorName.READABILITY def test_ct_011_partial_content_loses_due_to_low_coverage(): """CT-011: Um candidato contém somente parte do conteúdo comum -> Perde cobertura e reduz pontuação.""" full_content = "primeiro parágrafo do artigo completo segundo parágrafo com explicações terceiro parágrafo final" half_content = "primeiro parágrafo do artigo completo" article = { "newspaper4k": {"text": full_content, "error": None}, "readability": {"cleaned_text": full_content, "error": None}, "trafilatura": {"text": half_content, "error": None}, } result = select_article_extractor(article) assert result.selected_extractor == ExtractorName.NEWSPAPER4K def test_ct_012_recalculate_existing_selected_extractor(tmp_path: Path): """CT-012: A entrada já contém selected_extractor -> Recalcular e substituir somente essa chave.""" input_data = { "articles": [ { "titulo": "Teste", "selected_extractor": "trafilatura", "trafilatura": {"text": "lixo sem sentido", "error": None}, "newspaper4k": { "text": "conteúdo correto compartilhado por dois motores", "error": None, }, "readability": { "cleaned_text": "conteúdo correto compartilhado por dois motores", "error": None, }, } ] } in_file = tmp_path / "artigos.json" in_file.write_text(json.dumps(input_data, ensure_ascii=False), encoding="utf-8") res = process_batch(in_file) out_file = Path(res.output_file) assert out_file.exists() with open(out_file, "r", encoding="utf-8") as f: out_data = json.load(f) assert out_data["articles"][0]["selected_extractor"] == "newspaper4k" def test_ct_013_empty_articles_list(tmp_path: Path): """CT-013: articles está vazio -> Gerar saída válida com articles vazio.""" input_data = {"metadata": "info", "articles": []} in_file = tmp_path / "empty.json" in_file.write_text(json.dumps(input_data), encoding="utf-8") res = process_batch(in_file) out_file = Path(res.output_file) assert out_file.exists() with open(out_file, "r", encoding="utf-8") as f: out_data = json.load(f) assert out_data["metadata"] == "info" assert out_data["articles"] == [] assert res.total_articles == 0 assert res.processed_count == 0 def test_ct_014_invalid_json_fails_atomically(tmp_path: Path): """CT-014: JSON inválido -> Não gerar saída.""" in_file = tmp_path / "invalid.json" in_file.write_text("{articles: [ malformed json", encoding="utf-8") expected_out = tmp_path / "invalid_selected.json" with pytest.raises(ValueError, match="JSON inválido"): process_batch(in_file) assert not expected_out.exists() # ============================================================================== # Testes de Integração # ============================================================================== def test_integration_reference_file(tmp_path: Path): """Testa o processamento em lote completo sobre o arquivo real out/river_plate_extracted.json.""" ref_file = Path("out/river_plate_extracted.json") if not ref_file.exists(): pytest.skip( "Arquivo out/river_plate_extracted.json não encontrado para teste de integração." ) out_file = tmp_path / "river_plate_extracted_selected.json" result = process_batch(ref_file, output_path=out_file, verbose=True) assert result.total_articles == 20 assert result.processed_count == 20 assert out_file.exists() with open(out_file, "r", encoding="utf-8") as f: data = json.load(f) assert len(data["articles"]) == 20 for art in data["articles"]: assert "selected_extractor" in art assert art["selected_extractor"] in ["trafilatura", "newspaper4k", "readability"] # Validar a distribuição exata conforme o algoritmo do PRD assert result.selection_distribution == { "newspaper4k": 9, "readability": 9, "trafilatura": 2, } def test_article_1_regression_technical_tie_markdown_images(): """ Teste de regressão para o Artigo 1 (Los puntajes de River vs. Independiente Santa Fe): Valida que com o descarte de imagens Markdown ![alt](url), Readability e Newspaper4k entram em empate técnico (diff <= 0.03) e Newspaper4k vence por possuir menor quantidade de shingles (1009 vs 1057). """ ref_file = Path("out/river_plate_extracted.json") if not ref_file.exists(): pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.") with open(ref_file, "r", encoding="utf-8") as f: data = json.load(f) art1 = data["articles"][0] result = select_article_extractor(art1, article_index=0) assert result.selected_extractor == ExtractorName.NEWSPAPER4K assert result.selection_reason == "technical_tie_smallest_shingles" cand_news = result.candidates[ExtractorName.NEWSPAPER4K] cand_read = result.candidates[ExtractorName.READABILITY] cand_traf = result.candidates[ExtractorName.TRAFILATURA] assert cand_news.shingle_count == 1009 assert cand_read.shingle_count == 1057 assert cand_traf.shingle_count == 1091 # Diferença para o maior score <= 0.03 (empate técnico) max_score = max(cand_news.score, cand_read.score, cand_traf.score) assert (max_score - cand_news.score) <= 0.03 assert (max_score - cand_read.score) <= 0.03 def test_integration_large_batch_determinism(tmp_path: Path): """Gera um lote de 100 artigos sintéticos e verifica 100% de repetibilidade determinística entre 2 execuções.""" articles = [] for i in range(100): if i % 4 == 0: art = { "trafilatura": { "text": f"artigo numero {i} sobre futebol internacional no estadio", "error": None, }, "newspaper4k": { "text": f"artigo numero {i} sobre futebol internacional no estadio", "error": None, }, "readability": {"cleaned_text": "sem relacao", "error": None}, } elif i % 4 == 1: art = { "trafilatura": {"text": None, "error": "timeout"}, "newspaper4k": {"text": f"noticia exclusiva {i} com detalhes", "error": None}, "readability": { "cleaned_text": f"noticia exclusiva {i} com detalhes", "error": None, }, } elif i % 4 == 2: art = { "trafilatura": {"text": f"texto a {i}", "error": None}, "newspaper4k": {"text": f"texto b diferente {i}", "error": None}, "readability": {"cleaned_text": f"texto c terceiro {i}", "error": None}, } else: art = { "trafilatura": {"text": None, "error": "err"}, "newspaper4k": {"text": "", "error": "err"}, "readability": {"cleaned_text": None, "error": "err"}, } articles.append(art) batch_payload = {"articles": articles} in_file = tmp_path / "large_batch.json" in_file.write_text(json.dumps(batch_payload, ensure_ascii=False), encoding="utf-8") out1 = tmp_path / "large_batch_run1.json" out2 = tmp_path / "large_batch_run2.json" res1 = process_batch(in_file, output_path=out1) res2 = process_batch(in_file, output_path=out2) assert res1.processed_count == 100 assert res2.processed_count == 100 # Os resultados devem ser 100% idênticos selections1 = [s.selected_extractor for s in res1.selections] selections2 = [s.selected_extractor for s in res2.selections] assert selections1 == selections2 def test_integration_pipeline_downstream_consumer(tmp_path: Path): """ Testa a integração end-to-end do pipeline downstream: Lê o JSON enriquecido com selected_extractor, recupera o conteúdo do extrator vencedor e garante que o texto está higienizado e pronto para os classificadores NLP. """ ref_file = Path("out/river_plate_extracted.json") if not ref_file.exists(): pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.") out_file = tmp_path / "downstream_test.json" process_batch(ref_file, output_path=out_file) with open(out_file, "r", encoding="utf-8") as f: data = json.load(f) for idx, art in enumerate(data["articles"]): winner = art["selected_extractor"] assert winner in ["trafilatura", "newspaper4k", "readability"] # Recuperar texto do extrator vencedor conforme mapeamento do PRD if winner == "trafilatura": chosen_text = art["trafilatura"]["text"] elif winner == "newspaper4k": chosen_text = art["newspaper4k"]["text"] elif winner == "readability": chosen_text = art["readability"]["cleaned_text"] assert isinstance(chosen_text, str) assert len(chosen_text.strip()) > 0 # ============================================================================== # Testes End-to-End (E2E) via Subprocess CLI # ============================================================================== def test_e2e_cli_subprocess_real_execution(tmp_path: Path): """E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha de comando.""" ref_file = Path("out/river_plate_extracted.json") if not ref_file.exists(): pytest.skip("Arquivo out/river_plate_extracted.json não encontrado para teste E2E.") out_file = tmp_path / "e2e_river_plate_selected.json" script_path = Path("scripts/select_article_extractor.py").resolve() cmd = [ sys.executable, str(script_path), str(ref_file.resolve()), "-o", str(out_file), "--verbose", "--indent", "2", ] proc = subprocess.run(cmd, capture_output=True, text=True, check=False) assert proc.returncode == 0 assert out_file.exists() # Validar saída estruturada JSON do stdout summary = json.loads(proc.stdout) assert summary["status"] == "success" assert summary["total_articles"] == 20 assert summary["processed_count"] == 20 assert "distribution" in summary # Validar que logs de verbose foram emitidos no stderr assert "[Artigo #001]" in proc.stderr assert "[Artigo #020]" in proc.stderr def test_e2e_cli_subprocess_default_naming(tmp_path: Path): """E2E: Executa CLI sem a flag -o e valida criação automática de _selected.json.""" sample_data = { "articles": [ { "titulo": "Artigo Automático", "trafilatura": {"text": "conteúdo padrão", "error": None}, "newspaper4k": {"text": "conteúdo padrão", "error": None}, "readability": {"cleaned_text": "conteúdo padrão", "error": None}, } ] } in_file = tmp_path / "my_news.json" in_file.write_text(json.dumps(sample_data), encoding="utf-8") expected_out = tmp_path / "my_news_selected.json" script_path = Path("scripts/select_article_extractor.py").resolve() cmd = [sys.executable, str(script_path), str(in_file)] proc = subprocess.run(cmd, capture_output=True, text=True, check=False) assert proc.returncode == 0 assert expected_out.exists() with open(expected_out, "r", encoding="utf-8") as f: data = json.load(f) assert data["articles"][0]["selected_extractor"] == "newspaper4k" def test_e2e_cli_subprocess_invalid_input(tmp_path: Path): """E2E: Executa CLI com JSON inválido e valida código de saída e erro no stderr.""" invalid_file = tmp_path / "broken.json" invalid_file.write_text("not a valid json {", encoding="utf-8") script_path = Path("scripts/select_article_extractor.py").resolve() cmd = [sys.executable, str(script_path), str(invalid_file)] proc = subprocess.run(cmd, capture_output=True, text=True, check=False) assert proc.returncode == 2 assert "ERRO DE VALIDAÇÃO" in proc.stderr def test_e2e_cli_subprocess_missing_file(): """E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr.""" script_path = Path("scripts/select_article_extractor.py").resolve() cmd = [sys.executable, str(script_path), "non_existent_file_12345.json"] proc = subprocess.run(cmd, capture_output=True, text=True, check=False) assert proc.returncode == 1 assert "ERRO DE ARQUIVO" in proc.stderr