feat(runtime): implement single-article consolidation runtime and modularize codebase

This commit is contained in:
2026-08-24 00:14:07 -03:00
parent e1e0be1353
commit 23de7d8fe7
176 changed files with 266754 additions and 10179 deletions
+34
View File
@@ -0,0 +1,34 @@
"""Unit tests for optional adapter interfaces (Tier 2 / Tier 3)."""
from src.tools.adapters.embeddings import LocalEmbeddingsAdapter
from src.tools.adapters.llm import LLMFallbackAdapter
from src.tools.classifier import InherenceClassifier
from src.tools.models import ECPSnapshot
def test_embeddings_adapter_interface():
adapter = LocalEmbeddingsAdapter()
assert isinstance(adapter.is_available(), bool)
assert adapter.evaluate_similarity("test text", ["term1", "term2"]) == 0.0
def test_llm_adapter_interface():
adapter = LLMFallbackAdapter()
assert isinstance(adapter.is_available(), bool)
def test_classifier_with_adapter_flags():
classifier = InherenceClassifier(enable_embeddings=True, enable_llm=True)
assert classifier._embeddings_adapter is not None
assert classifier._llm_adapter is not None
ecp = ECPSnapshot(
target_entity_id="ent_test",
target_name="TestCorp",
aliases=["TestCorp"],
domain="Tech",
anchors=["software"],
)
res = classifier.classify(ecp, "TestCorp builds enterprise cloud software.")
assert res.decision.value == "DIRECT_INHERENT"
assert res.is_inherent is True
+242
View File
@@ -0,0 +1,242 @@
"""Adversarial and robustness test suite for Multilingual NLP Entity Inherence Classifier.
Validates homonym disambiguation, isolated related entities, edge cases,
and CLI execution behavior via subprocess (exit codes, stream purity, JSON parsing).
"""
import json
import subprocess
import sys
from src.tools.models import DecisionCategory, ECPSnapshot, RelatedEntity
def test_adversarial_sao_paulo_city_vs_fc():
"""Content about city/state governance of São Paulo against ECP for São Paulo FC."""
ecp = ECPSnapshot(
target_entity_id="ent_spfc",
target_name="São Paulo Futebol Clube",
aliases=["São Paulo", "SPFC", "Tricolor Paulista"],
domain="Futebol e Esportes",
anchors=["Morumbi", "futebol", "campeonato", "Copa Libertadores", "elenco", "estádio"],
negative_anchors=[
"prefeitura de são paulo",
"governo do estado de são paulo",
"trânsito na capital paulista",
],
graph_version="1.0.0",
related_entities=[],
)
content = (
"# Obras Viárias na Capital\n\n"
"A prefeitura de São Paulo anunciou novas intervenções no trânsito na capital paulista "
"para desafogar o fluxo de veículos na região central durante os horários de pico."
)
from src.tools.classifier import InherenceClassifier
classifier = InherenceClassifier()
result = classifier.classify(ecp, content)
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
assert result.is_inherent is False
assert result.decision != DecisionCategory.DIRECT_INHERENT
def test_adversarial_apple_fruit_recipe():
"""Content about apple fruit/culinary recipe against Apple Inc. tech entity."""
ecp = ECPSnapshot(
target_entity_id="ent_apple",
target_name="Apple",
aliases=["Apple Inc.", "Apple"],
domain="Technology",
anchors=["iPhone", "MacBook", "iOS", "silicon", "hardware"],
negative_anchors=["apple pie", "orchard harvest", "doce de maçã"],
graph_version="1.0.0",
related_entities=[],
)
content = (
"# Receita Caseira\n\n"
"Comprei maçãs frescas no mercado para preparar um doce de maçã com canela e açúcar mascavo."
)
from src.tools.classifier import InherenceClassifier
classifier = InherenceClassifier()
result = classifier.classify(ecp, content)
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
assert result.is_inherent is False
def test_adversarial_related_entity_without_scope_context():
"""High-weight related entity mentioned in passing without required domain anchors."""
ecp = ECPSnapshot(
target_entity_id="ent_volkswagen",
target_name="Volkswagen",
aliases=["Volkswagen AG", "VW"],
domain="Automotive & Electric Vehicles",
anchors=["Elektrofahrzeuge", "Batteriezellen", "Fahrzeugproduktion"],
graph_version="1.0.0",
related_entities=[
RelatedEntity(
entity_id="ent_northvolt",
name="Northvolt",
relation_type="SUPPLIER_OF",
weight=0.95,
scope="battery_technology",
confidence=0.99,
)
],
)
# Content mentions Northvolt in an unrelated/passing architectural context without domain anchors
content = (
"# Architekturbericht aus Stockholm\n\n"
"Während unseres Stadtrundgangs besuchten wir das neue Bürogebäude von Northvolt "
"mit moderner Holzfassade und Blick auf den See."
)
from src.tools.classifier import InherenceClassifier
classifier = InherenceClassifier()
result = classifier.classify(ecp, content)
# Must be TANGENTIAL or NOT_RELATED, NEVER CONTEXTUAL_INHERENT
assert result.decision in (DecisionCategory.TANGENTIAL, DecisionCategory.NOT_RELATED)
assert result.is_inherent is False
assert result.decision != DecisionCategory.CONTEXTUAL_INHERENT
def test_adversarial_subprocess_cli_success_stdout(tmp_path):
"""Run CLI via subprocess without --output and verify stdout is pure parseable JSON."""
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_petrobras",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text(
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
encoding="utf-8",
)
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
res = subprocess.run(
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
capture_output=True,
text=True,
encoding="utf-8",
env=env,
)
assert res.returncode == 0
# Stdout must be directly parseable as JSON without extraneous log text
assert res.stdout is not None and len(res.stdout.strip()) > 0
parsed = json.loads(res.stdout)
assert parsed["decision"] == "DIRECT_INHERENT"
assert parsed["is_inherent"] is True
assert parsed["confidence"] >= 0.85
assert len(parsed["evidence"]) > 0
def test_adversarial_subprocess_cli_empty_content(tmp_path):
"""Run CLI via subprocess with empty content and verify error code and exit code."""
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Test",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "empty.md"
content_file.write_text(" \n\n ", encoding="utf-8")
res = subprocess.run(
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
capture_output=True,
text=True,
encoding="utf-8",
env=env,
)
assert res.returncode != 0
# Stderr must contain pure parseable error JSON
parsed_err = json.loads(res.stderr)
assert parsed_err["error_code"] == "empty_content"
def test_adversarial_subprocess_cli_missing_required_field(tmp_path):
"""Run CLI via subprocess with missing target_name and verify error payload."""
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
ecp_file = tmp_path / "ecp_bad.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("Conteúdo de teste válido.", encoding="utf-8")
res = subprocess.run(
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
capture_output=True,
text=True,
encoding="utf-8",
env=env,
)
assert res.returncode != 0
parsed_err = json.loads(res.stderr)
assert parsed_err["error_code"] == "missing_required_field"
def test_adversarial_subprocess_cli_corrupted_json(tmp_path):
"""Run CLI via subprocess with corrupted JSON and verify error payload."""
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
ecp_file = tmp_path / "ecp_corrupted.json"
ecp_file.write_text("{ target_entity_id: not_valid_json }", encoding="utf-8")
content_file = tmp_path / "content.md"
content_file.write_text("Conteúdo de teste válido.", encoding="utf-8")
res = subprocess.run(
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
capture_output=True,
text=True,
encoding="utf-8",
env=env,
)
assert res.returncode != 0
parsed_err = json.loads(res.stderr)
assert parsed_err["error_code"] == "invalid_ecp_json"
+69
View File
@@ -0,0 +1,69 @@
"""Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence Classifier.
Matrix: 6 Languages (PT, EN, ES, DE, IT, FR) x 4 Decisions (DIRECT, CONTEXTUAL, TANGENTIAL, NOT_RELATED).
Target Success Criterion: Precision >= 90% over the 24 cases.
"""
import json
from pathlib import Path
import pytest
from src.tools.classifier import InherenceClassifier
from src.tools.models import ECPSnapshot
FIXTURES_DIR = Path(__file__).parent.parent / "fixtures" / "benchmark_24"
LANGUAGES = ["pt", "en", "es", "de", "it", "fr"]
DECISION_TYPES = ["direct", "contextual", "tangential", "not_related"]
BENCHMARK_CASES = [(lang, dec_type) for lang in LANGUAGES for dec_type in DECISION_TYPES]
@pytest.fixture(scope="module")
def classifier():
return InherenceClassifier()
@pytest.mark.parametrize("lang,dec_type", BENCHMARK_CASES)
def test_benchmark_case(classifier, lang: str, dec_type: str):
case_dir = FIXTURES_DIR / lang
ecp_file = case_dir / "ecp.json"
content_file = case_dir / f"{dec_type}.md"
expected_file = case_dir / f"{dec_type}_expected.json"
assert ecp_file.is_file(), f"Missing ECP fixture: {ecp_file}"
assert content_file.is_file(), f"Missing Content fixture: {content_file}"
assert expected_file.is_file(), f"Missing Expected fixture: {expected_file}"
ecp = ECPSnapshot.from_json_str(ecp_file.read_text(encoding="utf-8"))
content = content_file.read_text(encoding="utf-8")
expected = json.loads(expected_file.read_text(encoding="utf-8"))
result = classifier.classify(ecp, content)
# 1. Decision category validation
assert result.decision.value == expected["expected_decision"], (
f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
)
# 2. Derived is_inherent boolean validation
assert result.is_inherent == expected["expected_is_inherent"], (
f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
)
# 3. Language detection validation
assert result.detected_language == expected["expected_language"], (
f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
)
# 4. Confidence threshold validation
min_conf = expected.get("min_confidence", 0.0)
assert result.confidence >= min_conf, (
f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
)
# 5. Evidence presence for inherent content
if result.is_inherent:
assert len(result.evidence) > 0, (
f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
)
+100
View File
@@ -0,0 +1,100 @@
"""Unit tests for deterministic classification decision logic."""
import pytest
from src.tools.classifier import InherenceClassifier
from src.tools.models import DecisionCategory, ECPSnapshot, RelatedEntity
@pytest.fixture
def petrobras_ecp():
return ECPSnapshot(
target_entity_id="ent_petrobras",
target_name="Petrobras",
aliases=["Petróleo Brasileiro S.A.", "Petrobras", "Petrobrás"],
domain="Oil & Gas",
anchors=["pré-sal", "refinaria", "combustíveis", "petróleo", "exploração"],
negative_anchors=["posto de combustíveis pirata"],
graph_version="1.0.0",
related_entities=[
RelatedEntity(
entity_id="ent_transpetro",
name="Transpetro",
relation_type="SUBSIDIARY_OF",
weight=0.85,
aliases=[],
scope="logistics",
confidence=1.0,
)
],
)
def test_direct_inherent(petrobras_ecp):
classifier = InherenceClassifier()
content = (
"# Expansão da Produção Nacional\n\n"
"A Petrobras anunciou um aumento expressivo na produção de petróleo na camada pré-sal. "
"Os investimentos em novas plataformas devem acelerar a exploração offshore."
)
result = classifier.classify(petrobras_ecp, content)
assert result.decision == DecisionCategory.DIRECT_INHERENT
assert result.is_inherent is True
assert result.confidence >= 0.85
assert result.detected_language == "pt"
assert "Petrobras" in result.matched_anchors
assert len(result.evidence) > 0
def test_contextual_inherent(petrobras_ecp):
classifier = InherenceClassifier()
content = (
"# Logística de Combustíveis no Brasil\n\n"
"A Transpetro ampliou a sua frota de navios para o transporte de combustíveis e derivados "
"pelo litoral brasileiro, reforçando a infraestrutura energética."
)
result = classifier.classify(petrobras_ecp, content)
assert result.decision == DecisionCategory.CONTEXTUAL_INHERENT
assert result.is_inherent is True
assert result.confidence >= 0.70
assert len(result.graph_matches) == 1
assert result.graph_matches[0]["name"] == "Transpetro"
def test_tangential_inherent(petrobras_ecp):
classifier = InherenceClassifier()
content = (
"# Crônica de Viagem pelo Interior\n\n"
"Passamos perto de um prédio da Petrobras enquanto procurávamos um café na praça central. "
"A tarde estava quente e os pássaros cantavam nas árvores antigas."
)
result = classifier.classify(petrobras_ecp, content)
assert result.decision == DecisionCategory.TANGENTIAL
assert result.is_inherent is False
assert result.confidence < 0.60
assert len(result.warnings) > 0
def test_not_related(petrobras_ecp):
classifier = InherenceClassifier()
content = (
"# Como Fazer Bolo de Cenoura com Cobertura de Chocolate\n\n"
"Bata as cenouras raladas no liquidificador com os ovos e o óleo. "
"Acrescente a farinha de trigo e o açúcar aos poucos até obter uma massa homogênea."
)
result = classifier.classify(petrobras_ecp, content)
assert result.decision == DecisionCategory.NOT_RELATED
assert result.is_inherent is False
assert result.confidence >= 0.85
def test_negative_anchor_suppression(petrobras_ecp):
classifier = InherenceClassifier()
content = (
"# Operação Policial Fecha Estabelecimento\n\n"
"A polícia interditou um posto de combustíveis pirata na rodovia estadual por adulteração."
)
result = classifier.classify(petrobras_ecp, content)
assert result.decision == DecisionCategory.NOT_RELATED
assert result.is_inherent is False
assert len(result.negative_matches) > 0
@@ -0,0 +1,806 @@
"""
Suíte de Testes Exaustiva para o Classificador de Inerência (classify.py e src/).
Cobre 100% dos caminhos felizes, infelizes, limiares, de ambiguidade,
fallback de LLM (OpenAI e Gemini), resiliência de API, erros de contrato CLI
e suporte aos 6 idiomas conforme a metodologia da skill-suite-tests.
"""
from __future__ import annotations
import json
from pathlib import Path
from unittest.mock import MagicMock, patch
import pytest
from classify import main
from src.tools.adapters.llm import LLMFallbackAdapter
from src.tools.classifier import InherenceClassifier
from src.tools.models import (
ClassificationResult,
DecisionCategory,
ECPSnapshot,
RelatedEntity,
)
CLASSIFY_CLI = Path(__file__).parent.parent.parent / "classify.py"
# ==============================================================================
# Fixtures Universais
# ==============================================================================
@pytest.fixture
def ecp_tech_corp() -> ECPSnapshot:
return ECPSnapshot(
target_entity_id="ent_tech_corp",
target_name="TechCorp Global",
aliases=["TechCorp", "TechCorp Global", "TCG"],
domain="Tecnologia e Cloud",
anchors=[
"cloud",
"computação em nuvem",
"software",
"inteligência artificial",
"datacenter",
],
negative_anchors=["TechCorp Calçados", "TechCorp Imóveis", "homônimo"],
related_entities=[
RelatedEntity(
entity_id="ent_cloud_subsidiary",
name="CloudPlatform Solutions",
relation_type="SUBSIDIARY_OF",
weight=0.90,
aliases=["CloudPlatform"],
scope="cloud_services",
),
RelatedEntity(
entity_id="ent_ceo_tech",
name="Alan Turing Silva",
relation_type="CEO_OF",
weight=0.80,
aliases=["Alan Turing"],
scope="executive",
),
],
)
# ==============================================================================
# 1. Casos Felizes (Happy Paths) - NLP Determinístico (Tier 1)
# ==============================================================================
def test_happy_path_direct_inherent_with_canonical_and_anchors(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.1: Nome canônico + múltiplas âncoras temáticas -> DIRECT_INHERENT com alta confiança."""
classifier = InherenceClassifier()
content = (
"# TechCorp Global anuncia novo datacenter de computação em nuvem\n\n"
"A TechCorp Global investiu 500 milhões para expandir sua infraestrutura de software "
"e inteligência artificial na América Latina."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.is_inherent is True
assert res.confidence >= 0.90
assert "TechCorp Global" in res.matched_anchors or "TechCorp" in res.matched_anchors
assert len(res.evidence) >= 1
def test_happy_path_direct_inherent_via_alias_and_acronym(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.2: Apenas o alias / sigla 'TCG' é mencionado, com âncoras do domínio."""
classifier = InherenceClassifier()
content = (
"# Inovação em Cloud\n\n"
"A TCG lançou hoje uma nova plataforma de software baseada em computação em nuvem."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.is_inherent is True
assert res.confidence >= 0.85
def test_happy_path_direct_inherent_by_repetition_without_heavy_anchors(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.3: O nome 'TechCorp' aparece 3 vezes no texto, satisfazendo a regra de menção múltipla."""
classifier = InherenceClassifier()
content = (
"# Relatório Corporativo Trimestral\n\n"
"A TechCorp divulgou seus resultados. A TechCorp superou as estimativas de analistas. "
"O conselho da TechCorp aprovou dividendos extraordinários."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.is_inherent is True
assert res.confidence >= 0.85
def test_happy_path_contextual_inherent_via_subsidiary_graph_entity(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.4: Menção da subsidiária 'CloudPlatform Solutions' com âncoras de cloud."""
classifier = InherenceClassifier()
content = (
"# Expansão de Infraestrutura de Nuvem\n\n"
"A CloudPlatform Solutions ativou novos servidores em seu datacenter de computação em nuvem."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
assert res.is_inherent is True
assert res.confidence >= 0.75
assert len(res.graph_matches) >= 1
assert res.graph_matches[0]["name"] == "CloudPlatform Solutions"
def test_happy_path_contextual_inherent_via_executive_graph_entity(ecp_tech_corp: ECPSnapshot):
"""Cenário 1.5: Menção ao CEO no grafo + âncoras de tecnologia."""
classifier = InherenceClassifier()
content = (
"# Discurso na Conferência de Tecnologia\n\n"
"O executivo Alan Turing Silva discursou sobre o futuro da inteligência artificial e software."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
assert res.is_inherent is True
assert any(g["name"] == "Alan Turing Silva" for g in res.graph_matches)
# ==============================================================================
# 2. Casos Infelizes e Rejeições (Sad Paths) - NLP Determinístico (Tier 1)
# ==============================================================================
def test_sad_path_not_related_completely_off_topic(ecp_tech_corp: ECPSnapshot):
"""Cenário 2.1: Conteúdo totalmente desvinculado (culinária/jardinagem)."""
classifier = InherenceClassifier()
content = (
"# Receita de Pão Caseiro Fácil\n\n"
"Misture a farinha, o fermento biológico seco e a água morna. "
"Deixe a massa descansar por 40 minutos em local aquecido."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
assert res.confidence >= 0.90
assert len(res.matched_anchors) == 0
def test_sad_path_not_related_generic_domain_without_target_or_graph(ecp_tech_corp: ECPSnapshot):
"""Cenário 2.2: Artigo cita muitas âncoras ('cloud', 'software'), mas NÃO cita a TechCorp nem o grafo."""
classifier = InherenceClassifier()
content = (
"# O Mercado Global de Computação em Nuvem\n\n"
"O setor de computação em nuvem, datacenter e inteligência artificial cresceu 25% este ano."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
assert (
"General domain topics mentioned, but target entity or related entities are absent."
in res.rationale
)
def test_sad_path_not_related_negative_anchor_dominance(ecp_tech_corp: ECPSnapshot):
"""Cenário 2.3: Homônimo 'TechCorp Calçados' dispara âncora negativa dominante."""
classifier = InherenceClassifier()
content = (
"# Feira de Moda e Varejo\n\n"
"A TechCorp Calçados apresentou sua nova linha de sandálias de couro para o verão."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
assert "TechCorp Calçados" in res.negative_matches
def test_sad_path_not_related_negative_anchor_ties_with_positive_anchor(ecp_tech_corp: ECPSnapshot):
"""Cenário 2.4: 1 âncora negativa e 1 positiva -> prioridade de segurança rejeita para NOT_RELATED."""
classifier = InherenceClassifier()
content = "A TechCorp Calçados adotou um novo software interno de gestão."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
# ==============================================================================
# 3. Casos Limiares e Ambiguidades (Borderline / Tangential)
# ==============================================================================
def test_borderline_tangential_single_passing_mention(ecp_tech_corp: ECPSnapshot):
"""Cenário 3.1: Menção única isolada sem âncoras temáticas -> TANGENTIAL com baixa confiança."""
classifier = InherenceClassifier()
content = "Estávamos caminhando pela avenida e vimos a placa da TechCorp ao longe na esquina."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert res.is_inherent is False
assert res.confidence == 0.40
assert any("Low contextual density" in w for w in res.warnings)
def test_borderline_tangential_graph_entity_in_isolation(ecp_tech_corp: ECPSnapshot):
"""Cenário 3.2: Entidade do grafo mencionada sem contexto de domínio -> TANGENTIAL."""
classifier = InherenceClassifier()
content = "Alan Turing Silva participou de uma corrida beneficente no parque no domingo."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert res.is_inherent is False
assert res.confidence == 0.45
# ==============================================================================
# 4. Suíte Abrangente de Fallback para LLM (Tier 3)
# ==============================================================================
def test_llm_happy_path_upgrade_tangential_to_direct_inherent(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.1: Caso ambíguo elevado para DIRECT_INHERENT pelo LLM."""
mock_resp = json.dumps(
{
"analysis_summary": "Artigo detalha o projeto estratégico secreto da TechCorp.",
"decision": "DIRECT_INHERENT",
"confidence": 0.95,
"rationale": "Embora a redação use linguagem coloquial, o artigo foca inteiramente na estratégia da TechCorp.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "A diretoria da TechCorp finalizou as negociações confidenciais da rodada."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.is_inherent is True
assert res.confidence == 0.95
assert "[Tier 3 LLM]" in res.rationale
assert "[Tier 3 LLM Override applied]" in res.warnings
def test_llm_happy_path_upgrade_tangential_to_contextual_inherent(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.2: Caso ambíguo elevado para CONTEXTUAL_INHERENT pelo LLM."""
mock_resp = json.dumps(
{
"analysis_summary": "Matéria sobre fusão de fornecedores onde a TechCorp é impactada diretamente.",
"decision": "CONTEXTUAL_INHERENT",
"confidence": 0.88,
"rationale": "A TechCorp é parte material do ecossistema afetado pela fusão anunciada.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "O consórcio fornecedor foi reestruturado e envolverá contratos com a TechCorp."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
assert res.is_inherent is True
assert res.confidence == 0.88
def test_llm_happy_path_confirmation_of_tangential(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.3: LLM confirma categoricamente que a menção é periférica / irrelevante."""
mock_resp = json.dumps(
{
"analysis_summary": "Crônica sobre trânsito urbano com citação lateral a um outdoor da TechCorp.",
"decision": "TANGENTIAL",
"confidence": 0.97,
"rationale": "A empresa é apenas uma referência visual casual sem relação com a narrativa de trânsito.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "O tráfego estava parado bem em frente ao painel da TechCorp na autoestrada."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert res.is_inherent is False
assert res.confidence == 0.97
def test_llm_happy_path_rejection_to_not_related(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.4: LLM identifica homônimo não mapeado nas regras determinísticas e rebaixa para NOT_RELATED."""
mock_resp = json.dumps(
{
"analysis_summary": "Artigo sobre uma banda de rock indie com nome idêntico.",
"decision": "NOT_RELATED",
"confidence": 0.99,
"rationale": "O texto refere-se a um grupo musical e não à empresa de tecnologia.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "A banda TechCorp tocou seus novos acordes no festival de música independente."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.NOT_RELATED
assert res.is_inherent is False
assert res.confidence == 0.99
def test_llm_sad_path_llm_disabled_by_default_never_invokes_adapter(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.5: Quando enable_llm=False (padrão), o LLM NUNCA é chamado mesmo em caso limiar."""
called = {"status": False}
def tracking_fn(p: str) -> str:
called["status"] = True
return "{}"
adapter = LLMFallbackAdapter(provider_fn=tracking_fn)
classifier = InherenceClassifier(enable_llm=False, llm_adapter=adapter)
content = "Menção isolada da TechCorp sem contexto algum."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert called["status"] is False
def test_llm_sad_path_flag_enabled_without_api_key_or_provider(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.6: enable_llm=True mas sem chaves no ambiente -> degrada sem quebrar, retém Tier 1."""
with patch.dict("os.environ", {}, clear=True):
adapter = LLMFallbackAdapter(api_key="", provider_fn=None)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção isolada da TechCorp em relatório breve."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert res.is_inherent is False
def test_llm_sad_path_network_timeout_graceful_degradation(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.7: API do LLM sofre TimeoutError -> retém Tier 1 e registra aviso em warnings."""
def timeout_fn(p: str) -> str:
raise TimeoutError("Conexão com gateway do LLM excedeu tempo limite de 30s.")
adapter = LLMFallbackAdapter(provider_fn=timeout_fn)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "A TechCorp esteve presente no evento de premiação."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert any("LLM fallback failed" in w for w in res.warnings)
def test_llm_sad_path_http_500_server_error_graceful_degradation(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.8: API do LLM retorna erro 500 / ConnectionError -> retém Tier 1 com aviso."""
def error_500_fn(p: str) -> str:
raise ConnectionError("HTTP 500: Internal Server Error do provedor de IA.")
adapter = LLMFallbackAdapter(provider_fn=error_500_fn)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção da TechCorp em comunicado à imprensa."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
assert any("LLM fallback failed" in w for w in res.warnings)
def test_llm_sad_path_malformed_json_and_non_json_strings(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.9: LLM retorna texto livre ou JSON quebrado -> parser ignora com segurança."""
def make_bad_provider(resp_text: str):
def _prov(prompt: str) -> str:
return resp_text
return _prov
for bad_resp in [
"Não tenho certeza sobre este documento.",
"{json_quebrado_sem_aspas: true",
"```json\n{invalido: 123}\n```",
]:
adapter = LLMFallbackAdapter(provider_fn=make_bad_provider(bad_resp))
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção isolada da TechCorp."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
def test_llm_sad_path_missing_decision_key_in_json(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.10: LLM retorna JSON válido mas sem o campo obrigatório 'decision'."""
adapter = LLMFallbackAdapter(
provider_fn=lambda p: json.dumps({"confidence": 0.90, "rationale": "Faltou a decisao"})
)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção isolada da TechCorp."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
def test_llm_sad_path_unknown_hallucinated_decision_enum(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.11: LLM alucina uma categoria inexistente (ex: 'SUPER_INHERENT')."""
adapter = LLMFallbackAdapter(
provider_fn=lambda p: json.dumps({"decision": "SUPER_INHERENT", "confidence": 0.99})
)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "Menção isolada da TechCorp."
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.TANGENTIAL
def test_llm_resilience_confidence_clipping(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.12: LLM retorna confidence fora do intervalo [0.0, 1.0] -> clippa com segurança."""
def make_clipping_provider(c_val: float):
def _prov(prompt: str) -> str:
return json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": c_val,
"rationale": "Teste de clipping.",
}
)
return _prov
for raw_conf, expected_conf in [(1.5, 1.0), (-0.5, 0.0), (0.85432, 0.8543)]:
adapter = LLMFallbackAdapter(provider_fn=make_clipping_provider(raw_conf))
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
res = classifier.classify(ecp_tech_corp, "Menção da TechCorp.")
assert res.confidence == expected_conf
def test_llm_optimization_clear_case_bypasses_llm(ecp_tech_corp: ECPSnapshot):
"""Cenário 4.13: Caso claro de alta densidade NÃO chama LLM mesmo com enable_llm=True."""
called = {"status": False}
def tracking_fn(p: str) -> str:
called["status"] = True
return json.dumps({"decision": "DIRECT_INHERENT"})
adapter = LLMFallbackAdapter(provider_fn=tracking_fn)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = (
"# TechCorp Global anuncia nova inteligência artificial para computação em nuvem\n\n"
"A TechCorp Global ativou hoje novos clusters de datacenter com software avançado."
)
res = classifier.classify(ecp_tech_corp, content)
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert called["status"] is False # LLM NÃO foi acionado
# ==============================================================================
# 5. Provedores Reais de LLM (OpenAI Mock e Gemini REST Mock)
# ==============================================================================
def test_llm_provider_openai_client_execution(ecp_tech_corp: ECPSnapshot):
"""Cenário 5.1: Simula execução bem-sucedida via cliente OpenAI SDK."""
mock_chat_completion = MagicMock()
mock_choice = MagicMock()
mock_choice.message.content = json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": 0.96,
"rationale": "OpenAI validou o contexto corporativo com precisão.",
}
)
mock_chat_completion.choices = [mock_choice]
mock_openai_instance = MagicMock()
mock_openai_instance.chat.completions.create.return_value = mock_chat_completion
with patch("openai.OpenAI", return_value=mock_openai_instance):
adapter = LLMFallbackAdapter(api_key="sk-mock-openai-key")
initial_res = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=[],
negative_matches=[],
graph_matches=[],
evidence=[],
rationale="Passing.",
warnings=[],
)
res = adapter.disambiguate(ecp_tech_corp, "Artigo sobre TechCorp.", initial_res)
assert res is not None
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.confidence == 0.96
def test_llm_provider_gemini_rest_execution(ecp_tech_corp: ECPSnapshot):
"""Cenário 5.2: Simula execução bem-sucedida via API REST do Google Gemini."""
gemini_payload = {
"candidates": [
{
"content": {
"parts": [
{
"text": json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": 0.98,
"rationale": "Gemini 2.5 Flash confirmou aderência direta ao tópico.",
}
)
}
]
}
}
]
}
mock_response = MagicMock()
mock_response.read.return_value = json.dumps(gemini_payload).encode("utf-8")
mock_response.__enter__.return_value = mock_response
with patch("urllib.request.urlopen", return_value=mock_response):
with patch.dict("os.environ", {"GEMINI_API_KEY": "mock-gemini-key"}):
adapter = LLMFallbackAdapter(api_key="")
initial_res = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=[],
negative_matches=[],
graph_matches=[],
evidence=[],
rationale="Passing.",
warnings=[],
)
res = adapter.disambiguate(ecp_tech_corp, "Artigo sobre TechCorp.", initial_res)
assert res is not None
assert res.decision == DecisionCategory.DIRECT_INHERENT
assert res.confidence == 0.98
# ==============================================================================
# 6. Suíte de Contrato e Erros da CLI classify.py
# ==============================================================================
def test_cli_error_ecp_file_does_not_exist(tmp_path: Path, capsys):
"""Cenário 6.1: Caminho de ECP inexistente -> Exit Code 1, error_code: invalid_ecp_json."""
content_file = tmp_path / "valid.md"
content_file.write_text("# Conteúdo válido", encoding="utf-8")
exit_code = main(["--ecp", str(tmp_path / "nao_existe.json"), "--content", str(content_file)])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "invalid_ecp_json"
def test_cli_error_ecp_corrupted_json_syntax(tmp_path: Path, capsys):
"""Cenário 6.2: Arquivo ECP com sintaxe JSON corrompida."""
bad_ecp = tmp_path / "corrupt.json"
bad_ecp.write_text("{ target_name: 'sem_aspas' ", encoding="utf-8")
content_file = tmp_path / "valid.md"
content_file.write_text("# Conteúdo válido", encoding="utf-8")
exit_code = main(["--ecp", str(bad_ecp), "--content", str(content_file)])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "invalid_ecp_json"
def test_cli_error_ecp_missing_each_required_field(tmp_path: Path, capsys):
"""Cenário 6.3: Valida erro para falta de cada um dos campos obrigatórios do ECP."""
required_fields = ["target_entity_id", "target_name", "aliases", "domain", "anchors"]
base_ecp = {
"target_entity_id": "ent_1",
"target_name": "Nome",
"aliases": ["Alias"],
"domain": "Domínio",
"anchors": ["Âncora"],
}
content_file = tmp_path / "valid.md"
content_file.write_text("# Conteúdo válido", encoding="utf-8")
for field in required_fields:
bad_data = base_ecp.copy()
del bad_data[field]
bad_file = tmp_path / f"missing_{field}.json"
bad_file.write_text(json.dumps(bad_data), encoding="utf-8")
exit_code = main(["--ecp", str(bad_file), "--content", str(content_file)])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "missing_required_field"
assert field in err_json["message"]
def test_cli_error_content_file_does_not_exist(tmp_path: Path, capsys):
"""Cenário 6.4: Caminho de arquivo Markdown inexistente."""
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "TechCorp",
"aliases": ["TechCorp"],
"domain": "Tech",
"anchors": ["cloud"],
}
),
encoding="utf-8",
)
exit_code = main(["--ecp", str(ecp_file), "--content", str(tmp_path / "doc_fantasma.md")])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "invalid_markdown"
def test_cli_error_empty_and_whitespace_content(tmp_path: Path, capsys):
"""Cenário 6.5: Arquivo Markdown vazio ou contendo apenas espaços em branco."""
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "TechCorp",
"aliases": ["TechCorp"],
"domain": "Tech",
"anchors": ["cloud"],
}
),
encoding="utf-8",
)
for empty_text in ["", " \n\n\t \n "]:
empty_file = tmp_path / "empty.md"
empty_file.write_text(empty_text, encoding="utf-8")
exit_code = main(["--ecp", str(ecp_file), "--content", str(empty_file)])
assert exit_code == 1
captured = capsys.readouterr()
err_json = json.loads(captured.err)
assert err_json["error_code"] == "empty_content"
def test_cli_output_file_creates_nested_directories(tmp_path: Path):
"""Cenário 6.6: A flag -o / --output cria diretórios aninhados automaticamente."""
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "TechCorp",
"aliases": ["TechCorp"],
"domain": "Tech",
"anchors": ["cloud"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("# TechCorp\n\nTechCorp cloud computing.", encoding="utf-8")
nested_out = tmp_path / "deep" / "nested" / "folder" / "resultado.json"
exit_code = main(
["--ecp", str(ecp_file), "--content", str(content_file), "-o", str(nested_out)]
)
assert exit_code == 0
assert nested_out.exists()
payload = json.loads(nested_out.read_text(encoding="utf-8"))
assert payload["decision"] == "DIRECT_INHERENT"
# ==============================================================================
# 7. Matriz Multilíngue Completa (6 Idiomas)
# ==============================================================================
@pytest.mark.parametrize(
"lang,target,aliases,domain,anchors,content,expected_decision,expected_lang",
[
# Português
(
"pt",
"Petrobras",
["Petrobras"],
"Energia",
["pré-sal", "petróleo", "refinaria"],
"# Petrobras bate recorde de produção no pré-sal com novas plataformas.",
DecisionCategory.DIRECT_INHERENT,
"pt",
),
# Inglês
(
"en",
"Apple Inc.",
["Apple", "Apple Inc."],
"Technology",
["iPhone", "MacBook", "iOS", "silicon"],
"# Apple unveils new MacBook Pro with M4 silicon and advanced iOS features.",
DecisionCategory.DIRECT_INHERENT,
"en",
),
# Espanhol
(
"es",
"River Plate",
["River Plate", "River"],
"Fútbol",
["Monumental", "Libertadores", "Sudamericana"],
"# River Plate se prepara para disputar el torneo continental en el Estadio Monumental.",
DecisionCategory.DIRECT_INHERENT,
"es",
),
# Alemão (Compostos e Diacríticos)
(
"de",
"Volkswagen AG",
["Volkswagen", "VW"],
"Automobilindustrie",
["Elektroauto", "Batteriefabrik", "Produktion"],
"# Volkswagen investiert Milliarden in eine neue Batteriefabrik für Elektroautos in Deutschland.",
DecisionCategory.DIRECT_INHERENT,
"de",
),
# Italiano
(
"it",
"Scuderia Ferrari",
["Ferrari", "Scuderia Ferrari"],
"Automobilismo",
["Monza", "Gran Premio", "motore", "pole position"],
"# La Ferrari conquista una straordinaria pole position nel Gran Premio di Monza.",
DecisionCategory.DIRECT_INHERENT,
"it",
),
# Francês (Elisão e Apóstrofos)
(
"fr",
"TotalEnergies",
["TotalEnergies", "Total"],
"Énergie",
["énergie solaire", "pétrole", "renouvelable", "électricité"],
"# L'entreprise TotalEnergies accélère ses investissements dans l'énergie solaire et l'électricité en France.",
DecisionCategory.DIRECT_INHERENT,
"fr",
),
],
)
def test_multilingual_matrix_6_languages(
lang: str,
target: str,
aliases: list[str],
domain: str,
anchors: list[str],
content: str,
expected_decision: DecisionCategory,
expected_lang: str,
):
"""Garante a precisão e robustez do classificador nos 6 idiomas suportados pela POC."""
ecp = ECPSnapshot(
target_entity_id=f"ent_{lang}",
target_name=target,
aliases=aliases,
domain=domain,
anchors=anchors,
)
classifier = InherenceClassifier()
res = classifier.classify(ecp, content)
assert res.decision == expected_decision
assert res.is_inherent is True
assert res.detected_language == expected_lang
assert res.confidence >= 0.85
+126
View File
@@ -0,0 +1,126 @@
"""CLI execution tests covering flags, arguments, stdout, and error handling."""
import json
from classify import main
def test_cli_success_stdout(tmp_path, capsys):
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text(
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
encoding="utf-8",
)
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
assert exit_code == 0
captured = capsys.readouterr()
result = json.loads(captured.out)
assert result["decision"] == "DIRECT_INHERENT"
assert result["is_inherent"] is True
assert result["detected_language"] == "pt"
def test_cli_output_file(tmp_path):
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text(
"Petrobras anunciou investimentos bilionários em novas refinarias de petróleo.",
encoding="utf-8",
)
output_file = tmp_path / "out" / "result.json"
exit_code = main(
["--ecp", str(ecp_file), "--content", str(content_file), "--output", str(output_file)]
)
assert exit_code == 0
assert output_file.exists()
result = json.loads(output_file.read_text(encoding="utf-8"))
assert result["decision"] == "DIRECT_INHERENT"
assert result["is_inherent"] is True
def test_cli_missing_ecp_file(tmp_path, capsys):
content_file = tmp_path / "content.md"
content_file.write_text("Algum conteúdo válido aqui.", encoding="utf-8")
exit_code = main(["--ecp", str(tmp_path / "non_existent.json"), "--content", str(content_file)])
assert exit_code == 1
captured = capsys.readouterr()
err = json.loads(captured.err)
assert err["error_code"] == "invalid_ecp_json"
def test_cli_empty_content_file(tmp_path, capsys):
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "empty.md"
content_file.write_text(" \n\n ", encoding="utf-8")
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
assert exit_code == 1
captured = capsys.readouterr()
err = json.loads(captured.err)
assert err["error_code"] == "empty_content"
def test_cli_missing_required_ecp_field(tmp_path, capsys):
ecp_file = tmp_path / "bad_ecp.json"
ecp_file.write_text(
json.dumps({"target_entity_id": "ent_1", "domain": "Oil & Gas", "anchors": ["petróleo"]}),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("Algum conteúdo válido para testar o erro.", encoding="utf-8")
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
assert exit_code == 1
captured = capsys.readouterr()
err = json.loads(captured.err)
assert err["error_code"] == "missing_required_field"
@@ -0,0 +1,759 @@
"""
Suíte de Testes Automatizados para Conversão de Artigo JSON para Markdown.
Cobre 100% dos Requisitos Funcionais (FR-001 a FR-018), Requisitos Não Funcionais (RNF-001 a RNF-006),
Critérios de Aceite (CA-001 a CA-013), Casos de Teste do PRD (CT-001 a CT-012),
Matriz de Erros (12 condições), Testes Unitários de Prioridade/Normalização,
Testes de Integração de Arquivo e Testes E2E de Pipeline via subprocess.
"""
from __future__ import annotations
import hashlib
import json
import subprocess
import sys
from pathlib import Path
from typing import Any
import pytest
from scripts.convert_article_to_markdown import (
assemble_markdown_document,
clean_body_images,
convert_article,
convert_html_to_markdown,
normalize_date,
normalize_list,
normalize_scalar,
parse_arguments,
remove_duplicate_initial_h1,
resolve_article_body,
resolve_article_metadata,
validate_url,
)
FIXTURES_DIR = Path(__file__).parent.parent / "fixtures" / "markdown_conversion"
SCRIPT_PATH = Path(__file__).parent.parent.parent / "scripts" / "convert_article_to_markdown.py"
# ==============================================================================
# 1. Testes Unitários de Normalização e Sanitização Escalar
# ==============================================================================
def test_normalize_scalar_complex_html_entities():
"""Valida decodificação de entidades HTML nomeadas e numéricas."""
assert (
normalize_scalar("River &amp; Boca &quot;Supercl&aacute;sico&quot;")
== 'River & Boca "Superclásico"'
)
assert normalize_scalar("Pre&ccedil;o: R&#36; 50&#44;00 &euro;") == "Preço: R$ 50,00 €"
def test_normalize_scalar_whitespace_collapsing():
"""Valida colapso de tabs, quebras de linha e espaços múltiplos em um único espaço."""
assert (
normalize_scalar(" Texto com \t\t múltiplos \n\n espaços ")
== "Texto com múltiplos espaços"
)
@pytest.mark.parametrize(
"placeholder",
[
"null",
"Null",
"NULL",
"none",
"None",
"NONE",
"n/a",
"N/A",
"N/a",
"unknown",
"Unknown",
"UNKNOWN",
"[no-author]",
"[No-Author]",
"[NO-AUTHOR]",
"no-author",
"No-Author",
"NO-AUTHOR",
],
)
def test_normalize_scalar_placeholders_discarded(placeholder: str):
"""Garante que todos os placeholders documentados no PRD sejam descartados (retornando None)."""
assert normalize_scalar(placeholder) is None
assert normalize_scalar(f" {placeholder} ") is None
def test_normalize_scalar_non_string_types():
"""Valida que entradas não string retornem None de forma segura."""
assert normalize_scalar(None) is None
assert normalize_scalar(12345) is None
assert normalize_scalar(["lista"]) is None
assert normalize_scalar({"chave": "valor"}) is None
# ==============================================================================
# 2. Testes Unitários de Normalização de Listas
# ==============================================================================
def test_normalize_list_semicolon_and_comma_split():
"""Testa divisão por ponto e vírgula na string e vírgulas em elementos de lista para tags/categorias."""
raw_str = "Futebol; Copa Libertadores; Conmebol; Notícias de Hoje"
expected_str = ["Futebol", "Copa Libertadores", "Conmebol", "Notícias de Hoje"]
assert normalize_list(raw_str) == expected_str
raw_list = ["Futebol", "Copa Libertadores, Conmebol", "Notícias de Hoje"]
expected_list = ["Futebol", "Copa Libertadores", "Conmebol", "Notícias de Hoje"]
assert normalize_list(raw_list) == expected_list
def test_normalize_list_author_url_filtering():
"""Garante que URLs em campos de autor sejam estritamente descartadas."""
raw_authors = [
"Ernesto Provitilo",
"https://twitter.com/eprovitilo",
"http://www.instagram.com/reporter",
"www.tycsports.com/autor",
"Juan Pablo Varsky",
]
result = normalize_list(raw_authors, is_author=True)
assert result == ["Ernesto Provitilo", "Juan Pablo Varsky"]
def test_normalize_list_deduplication_preserves_case_and_order():
"""Testa deduplicação case-insensitive preservando a grafia e ordem da primeira ocorrência."""
items = ["River Plate", "Boca Juniors", "river plate", "RIVER PLATE", "boca juniors", "Racing"]
assert normalize_list(items) == ["River Plate", "Boca Juniors", "Racing"]
def test_normalize_list_empty_and_invalid():
"""Testa comportamento com listas vazias, nulas ou contendo apenas placeholders."""
assert normalize_list([]) == []
assert normalize_list(None) == []
assert normalize_list(["n/a", "unknown", "[no-author]", " "]) == []
# ==============================================================================
# 3. Testes Unitários de Parsing de Datas
# ==============================================================================
def test_normalize_date_iso_8601_variants():
"""Valida parsing de datas ISO 8601 em múltiplos formatos e fusos."""
assert normalize_date("2026-08-20T00:36:33-03:00") == "2026-08-20T00:36:33-03:00"
assert normalize_date("2026-08-20T03:36:33+00:00") == "2026-08-20T03:36:33+00:00"
# Data pura sem hora
assert normalize_date("2026-08-20") == "2026-08-20"
def test_normalize_date_rfc_2822_variants():
"""Valida parsing de datas no formato RFC 2822 (usado em feeds RSS e cabeçalhos HTTP)."""
d1 = normalize_date("Thu, 20 Aug 2026 03:27:26 GMT")
assert d1 is not None and "2026-08-20" in d1
d2 = normalize_date("Wed, 19 Aug 2026 21:00:00 -0300")
assert d2 is not None and "2026-08-19" in d2
def test_normalize_date_invalid_and_placeholders():
"""Garante que datas inválidas ou placeholders retornem None sem lançar exceção não tratada."""
assert normalize_date("data-invalida") is None
assert normalize_date("2026/99/99") is None
assert normalize_date("n/a") is None
assert normalize_date(None) is None
assert normalize_date(123456789) is None
# ==============================================================================
# 4. Testes Unitários de Validação de URLs
# ==============================================================================
@pytest.mark.parametrize(
"valid_url",
[
"https://www.tycsports.com/river-plate/los-puntajes.html",
"http://globoesporte.globo.com/futebol/times/flamengo",
"https://sub.dominio.co.uk:8080/path/to/resource?param=1&query=test#hash",
"https://example.com/noticia-com-acentos-%C3%A1%C3%A9%C3%AD",
],
)
def test_validate_url_valid_schemes(valid_url: str):
"""Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos."""
assert validate_url(valid_url) == valid_url
@pytest.mark.parametrize(
"invalid_url",
[
"ftp://ftp.is.co.za/rfc/rfc1808.txt",
"file:///C:/Users/test/file.txt",
"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAUA",
"javascript:alert('xss')",
"/caminho/relativo/artigo.html",
"http://",
"https://",
"",
" ",
None,
],
)
def test_validate_url_invalid_schemes(invalid_url: str | None):
"""Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias."""
assert validate_url(invalid_url) is None
# ==============================================================================
# 5. Testes Unitários de Conversão HTML para Markdown
# ==============================================================================
def test_convert_html_to_markdown_rich_formatting():
"""Valida conversão de elementos HTML estruturados para Markdown com títulos ATX."""
html_raw = (
"<div>"
"<h1>Título H1</h1>"
"<h2>Subtítulo H2</h2>"
"<h3>Seção H3</h3>"
"<p>Parágrafo com <b>negrito</b>, <strong>forte</strong>, <i>itálico</i> e <em>ênfase</em>.</p>"
"<blockquote>Uma citação memorável.</blockquote>"
"<ul><li>Item 1</li><li>Item 2</li></ul>"
"<p>Link para o <a href='https://example.com/fonte'>portal oficial</a>.</p>"
"<code>codigo_inline()</code>"
"<script>alert('remover');</script>"
"<style>.esconder { display: none; }</style>"
"</div>"
)
md = convert_html_to_markdown(html_raw)
assert "# Título H1" in md
assert "## Subtítulo H2" in md
assert "### Seção H3" in md
assert "**negrito**" in md or "__negrito__" in md
assert "> Uma citação memorável." in md
assert "* Item 1" in md or "- Item 1" in md
assert "[portal oficial](https://example.com/fonte)" in md
assert "`codigo_inline()`" in md
assert "alert('remover')" not in md
assert "display: none" not in md
def test_convert_html_to_markdown_empty_or_whitespace():
"""Testa conversão de HTML vazio retornando string vazia."""
assert convert_html_to_markdown("") == ""
assert convert_html_to_markdown(" \n\t ") == ""
assert convert_html_to_markdown(None) == ""
# ==============================================================================
# 6. Testes de Isolamento Estrito do Extrator e Fallback Interno
# ==============================================================================
def test_resolve_article_body_trafilatura_primary_and_fallback():
"""Testa prioridade trafilatura.markdown sobre trafilatura.text."""
# Primário
art1 = {
"selected_extractor": "trafilatura",
"trafilatura": {"markdown": "Corpo primário Trafilatura", "text": "Texto secundário"},
}
assert resolve_article_body(art1) == "Corpo primário Trafilatura"
# Fallback
art2 = {
"selected_extractor": "trafilatura",
"trafilatura": {"markdown": None, "text": "Texto secundário Trafilatura"},
}
assert resolve_article_body(art2) == "Texto secundário Trafilatura"
def test_resolve_article_body_newspaper4k_primary_and_fallback():
"""Testa prioridade newspaper4k.article_html sobre newspaper4k.text."""
# Primário HTML -> MD
art1 = {
"selected_extractor": "newspaper4k",
"newspaper4k": {
"article_html": "<p>Artigo em <b>HTML</b></p>",
"text": "Artigo em texto puro",
},
}
assert "**HTML**" in resolve_article_body(art1)
# Fallback
art2 = {
"selected_extractor": "newspaper4k",
"newspaper4k": {"article_html": None, "text": "Artigo em texto puro Newspaper"},
}
assert resolve_article_body(art2) == "Artigo em texto puro Newspaper"
def test_resolve_article_body_readability_primary_and_fallback():
"""Testa prioridade readability.cleaned_html sobre readability.cleaned_text."""
# Primário HTML -> MD
art1 = {
"selected_extractor": "readability",
"readability": {
"cleaned_html": "<p>Conteúdo <i>Readability</i></p>",
"cleaned_text": "Texto puro Readability",
},
}
body = resolve_article_body(art1)
assert "*Readability*" in body or "_Readability_" in body
# Fallback
art2 = {
"selected_extractor": "readability",
"readability": {"cleaned_html": "", "cleaned_text": "Texto puro Readability Fallback"},
}
assert resolve_article_body(art2) == "Texto puro Readability Fallback"
@pytest.mark.parametrize("extractor", ["trafilatura", "newspaper4k", "readability"])
def test_resolve_article_body_strict_isolation_all_extractors(extractor: str):
"""Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback para outro extrator."""
article = {
"selected_extractor": extractor,
"trafilatura": {"markdown": "Texto Trafilatura", "text": "Texto Trafilatura"},
"newspaper4k": {"article_html": "<p>Texto Newspaper</p>", "text": "Texto Newspaper"},
"readability": {
"cleaned_html": "<p>Texto Readability</p>",
"cleaned_text": "Texto Readability",
},
}
# Esvazia o corpo do extrator selecionado
if extractor == "trafilatura":
article["trafilatura"] = {"markdown": None, "text": ""}
elif extractor == "newspaper4k":
article["newspaper4k"] = {"article_html": "", "text": None}
elif extractor == "readability":
article["readability"] = {"cleaned_html": None, "cleaned_text": ""}
with pytest.raises(ValueError, match="Corpo do extrator selecionado.*vazio|indisponível"):
resolve_article_body(article)
def test_resolve_article_body_invalid_selected_extractor():
"""Garante erro ao receber selected_extractor ausente ou não reconhecido."""
with pytest.raises(ValueError, match="selected_extractor inválido ou ausente"):
resolve_article_body({"selected_extractor": "extrator_desconhecido"})
with pytest.raises(ValueError, match="selected_extractor inválido ou ausente"):
resolve_article_body({"selected_extractor": None})
# ==============================================================================
# 7. Testes da Matriz Determinística de Resolução de Metadados
# ==============================================================================
def test_metadata_priority_title_all_fallbacks():
"""Valida a cadeia de fallback completa para o campo TÍTULO (6 níveis)."""
# 1. Do selecionado
art1 = {"selected_extractor": "trafilatura", "trafilatura": {"title": "Título Selecionado"}}
assert (
resolve_article_metadata({**art1, "crawled_url": "https://e.com"})["title"]
== "Título Selecionado"
)
# 2. input_meta.titulo
art2 = {
"selected_extractor": "trafilatura",
"input_meta": {"titulo": "Título Input Meta"},
"crawled_url": "https://e.com",
}
assert resolve_article_metadata(art2)["title"] == "Título Input Meta"
# 3. page_title
art3 = {
"selected_extractor": "trafilatura",
"page_title": "Título Page Title",
"crawled_url": "https://e.com",
}
assert resolve_article_metadata(art3)["title"] == "Título Page Title"
# 4. newspaper4k.title
art4 = {
"selected_extractor": "trafilatura",
"newspaper4k": {"title": "Título Newspaper"},
"crawled_url": "https://e.com",
}
assert resolve_article_metadata(art4)["title"] == "Título Newspaper"
# 5. readability.title
art5 = {
"selected_extractor": "trafilatura",
"readability": {"title": "Título Readability"},
"crawled_url": "https://e.com",
}
assert resolve_article_metadata(art5)["title"] == "Título Readability"
def test_metadata_priority_original_url_all_fallbacks():
"""Valida a cadeia de fallback completa para a URL ORIGINAL (5 níveis)."""
# 1. input_meta.url
art1 = {
"selected_extractor": "trafilatura",
"trafilatura": {"title": "T"},
"input_meta": {"url": "https://example.com/input-meta"},
"crawled_url": "https://example.com/crawled",
}
assert resolve_article_metadata(art1)["original_url"] == "https://example.com/input-meta"
# 2. crawled_url
art2 = {
"selected_extractor": "trafilatura",
"trafilatura": {"title": "T"},
"crawled_url": "https://example.com/crawled",
}
assert resolve_article_metadata(art2)["original_url"] == "https://example.com/crawled"
# 3. Canonical do selecionado
art3 = {
"selected_extractor": "trafilatura",
"trafilatura": {"title": "T", "canonical_url": "https://example.com/canonical-trafilatura"},
}
assert (
resolve_article_metadata(art3)["original_url"]
== "https://example.com/canonical-trafilatura"
)
# 4. Canonical do newspaper4k
art4 = {
"selected_extractor": "readability",
"readability": {"title": "T"},
"newspaper4k": {"canonical_link": "https://example.com/canonical-newspaper"},
}
assert (
resolve_article_metadata(art4)["original_url"] == "https://example.com/canonical-newspaper"
)
def test_metadata_priority_subtitle_omitted_when_equal_to_title():
"""Garante que subtítulo idêntico ao título seja automaticamente omitido (None)."""
art = {
"selected_extractor": "trafilatura",
"trafilatura": {
"title": "Grande Vitória no Clássico",
"description": " grande vitória no clássico ",
},
"crawled_url": "https://example.com/noticia",
}
meta = resolve_article_metadata(art)
assert meta["title"] == "Grande Vitória no Clássico"
assert meta["subtitle"] is None
def test_metadata_priority_first_valid_source_no_cross_merging():
"""Garante que listas de autores/tags usem apenas a primeira fonte válida, sem merge cruzado."""
art = {
"selected_extractor": "readability",
"crawled_url": "https://example.com/noticia",
"readability": {"title": "Título", "author": "Carlos Bilardo"},
"newspaper4k": {"authors": ["Juan Pérez", "María Gómez"]},
}
meta = resolve_article_metadata(art)
# Deve pegar o autor de Readability (selecionado), sem misturar com Newspaper4k
assert meta["authors"] == ["Carlos Bilardo"]
# ==============================================================================
# 8. Testes de Higienização de Cabeçalhos e Imagens no Corpo
# ==============================================================================
def test_remove_duplicate_initial_h1_exact_and_variations():
"""Testa remoção de H1 inicial coincidente com título com variações de espaços e caixa."""
title = "River Plate Conquista a Copa"
# H1 inicial igual
body1 = "# river plate conquista a copa\n\nPrimeiro parágrafo do artigo."
assert remove_duplicate_initial_h1(body1, title).strip() == "Primeiro parágrafo do artigo."
# H1 inicial diferente (deve ser preservado)
body2 = "# Outro Título Diferente\n\nPrimeiro parágrafo."
assert remove_duplicate_initial_h1(body2, title) == body2
# Sem H1 inicial
body3 = "Parágrafo sem nenhum título H1 inicial."
assert remove_duplicate_initial_h1(body3, title) == body3
def test_clean_body_images_removes_invalid_and_deduplicates():
"""Valida descarte de data:, relativos e deduplicação mantendo a primeira ocorrência."""
body = (
"![Legenda 1](https://example.com/img1.jpg)\n\n"
"![Legenda Relativa](/imagem.png)\n\n"
"![Legenda Data](data:image/png;base64,AAAA)\n\n"
"![Legenda 1 Repetida](https://example.com/img1.jpg)\n\n"
"![Legenda 2](https://example.com/img2.webp)"
)
cleaned = clean_body_images(body)
assert cleaned.count("https://example.com/img1.jpg") == 1
assert "https://example.com/img2.webp" in cleaned
assert "/imagem.png" not in cleaned
assert "data:image" not in cleaned
# ==============================================================================
# 9. Testes de Montagem e Formatação do Documento Markdown
# ==============================================================================
def test_assemble_markdown_document_full_and_minimal():
"""Testa montagem com todos os campos e apenas com campos obrigatórios."""
# Artigo Mínimo (apenas Título e URL Original)
min_meta: dict[str, Any] = {
"title": "Título Mínimo",
"original_url": "https://example.com/minimo",
"subtitle": None,
"authors": [],
"publish_date": None,
"site_name": None,
"categories": [],
"tags": [],
"keywords": [],
"language": None,
"top_image": None,
}
doc_min = assemble_markdown_document(min_meta, "Corpo do texto simples.")
assert doc_min.startswith("# Título Mínimo\n\n")
assert "**Fonte original:** [https://example.com/minimo](https://example.com/minimo)" in doc_min
assert "**Autor:**" not in doc_min
assert "**Site:**" not in doc_min
assert "![Imagem principal]" not in doc_min
assert "\n\n---\n\nCorpo do texto simples.\n" in doc_min
assert doc_min.endswith("\n")
assert not doc_min.endswith("\n\n")
# ==============================================================================
# 10. Golden Test Fixtures (Conformidade 100% Byte a Byte)
# ==============================================================================
def test_golden_fixtures_byte_level_precision(tmp_path):
"""Garante correspondência exata byte a byte para Trafilatura, Newspaper4k e Readability."""
for extractor in ["trafilatura", "newspaper4k", "readability"]:
input_json = FIXTURES_DIR / f"valid_{extractor}.json"
expected_md = (FIXTURES_DIR / f"valid_{extractor}.md").read_text(encoding="utf-8")
output_md = tmp_path / f"valid_{extractor}.md"
res = subprocess.run(
[sys.executable, str(SCRIPT_PATH), "-i", str(input_json), "-o", str(output_md)],
capture_output=True,
text=True,
)
assert res.returncode == 0, f"Erro no extrator {extractor}: {res.stderr}"
generated_md = output_md.read_text(encoding="utf-8")
assert generated_md == expected_md, f"Divergência byte a byte na fixture {extractor}"
def test_conversion_determinism_sha256_repeatability(tmp_path):
"""Garante que múltiplas execuções no mesmo arquivo produzam hashes SHA-256 idênticos."""
input_json = FIXTURES_DIR / "valid_trafilatura.json"
hashes = set()
for i in range(5):
out_md = tmp_path / f"deterministic_{i}.md"
res = subprocess.run(
[sys.executable, str(SCRIPT_PATH), "-i", str(input_json), "-o", str(out_md)],
capture_output=True,
text=True,
)
assert res.returncode == 0
content_bytes = out_md.read_bytes()
hashes.add(hashlib.sha256(content_bytes).hexdigest())
assert len(hashes) == 1, "A conversão não foi 100% determinística entre execuções repetidas."
# ==============================================================================
# 11. Testes de Integração CLI, Validação e Códigos de Saída
# ==============================================================================
def test_cli_exit_codes_and_error_handling(tmp_path):
"""Testa toda a matriz de códigos de saída da CLI (0, 1, 2)."""
# Código 2: Sintaxe / Argumentos Faltantes
res_no_args = subprocess.run([sys.executable, str(SCRIPT_PATH)], capture_output=True, text=True)
assert res_no_args.returncode == 2
# Código 1: Arquivo Inexistente
res_missing_file = subprocess.run(
[sys.executable, str(SCRIPT_PATH), "-i", "arquivo_que_nao_existe_xyz.json"],
capture_output=True,
text=True,
)
assert res_missing_file.returncode == 1
assert "não encontrado" in res_missing_file.stderr.lower()
# Código 1: Rejeição de Batch com chave 'articles'
res_batch = subprocess.run(
[sys.executable, str(SCRIPT_PATH), "-i", str(FIXTURES_DIR / "batch_articles_invalid.json")],
capture_output=True,
text=True,
)
assert res_batch.returncode == 1
assert "articles" in res_batch.stderr.lower()
# Código 1: JSON Corrompido
res_corrupt = subprocess.run(
[sys.executable, str(SCRIPT_PATH), "-i", str(FIXTURES_DIR / "corrupt_json_invalid.json")],
capture_output=True,
text=True,
)
assert res_corrupt.returncode == 1
assert "json" in res_corrupt.stderr.lower()
def test_cli_json_root_must_be_object(tmp_path):
"""Garante encerramento com código 1 caso a raiz do JSON seja lista, número ou string."""
for invalid_root in [["item1", "item2"], 12345, "string simples"]:
bad_json = tmp_path / "bad_root.json"
bad_json.write_text(json.dumps(invalid_root), encoding="utf-8")
res = subprocess.run(
[sys.executable, str(SCRIPT_PATH), "-i", str(bad_json)],
capture_output=True,
text=True,
)
assert res.returncode == 1
assert "objeto" in res.stderr.lower() or "dict" in res.stderr.lower()
def test_cli_no_stdout_pollution_and_atomic_preservation(tmp_path):
"""Garante que stdout permaneça limpo e gravação atômica preserve arquivos preexistentes em falha."""
target_md = tmp_path / "target_document.md"
target_md.write_text("Versão Original Preservada", encoding="utf-8")
# Executa conversão bem-sucedida
res_ok = subprocess.run(
[
sys.executable,
str(SCRIPT_PATH),
"-i",
str(FIXTURES_DIR / "valid_trafilatura.json"),
"-o",
str(target_md),
],
capture_output=True,
text=True,
)
assert res_ok.returncode == 0
assert res_ok.stdout == "" # Não polui stdout
assert "[INFO]" in res_ok.stderr
# Executa falha direcionada ao mesmo target
res_fail = subprocess.run(
[
sys.executable,
str(SCRIPT_PATH),
"-i",
str(FIXTURES_DIR / "missing_body_invalid.json"),
"-o",
str(target_md),
],
capture_output=True,
text=True,
)
assert res_fail.returncode == 1
# O arquivo target deve manter o conteúdo do sucesso anterior, não foi apagado/corrompido
assert "# Los puntajes de River" in target_md.read_text(encoding="utf-8")
# Verifica que não há arquivos temporários .tmp no diretório
tmp_files = list(tmp_path.glob("*.tmp"))
assert len(tmp_files) == 0
def test_cli_default_output_naming(tmp_path):
"""Garante geração automática de <input_stem>.md quando -o não é informado."""
sample_file = tmp_path / "meu_artigo_editorial.json"
sample_file.write_text(
json.dumps(
{
"selected_extractor": "trafilatura",
"crawled_url": "https://example.com/editorial",
"trafilatura": {"title": "Editorial do Dia", "markdown": "Texto do editorial."},
}
),
encoding="utf-8",
)
res = subprocess.run(
[sys.executable, str(SCRIPT_PATH), "-i", str(sample_file)],
capture_output=True,
text=True,
)
assert res.returncode == 0
expected_md = tmp_path / "meu_artigo_editorial.md"
assert expected_md.exists()
assert "# Editorial do Dia" in expected_md.read_text(encoding="utf-8")
# ==============================================================================
# 12. Teste E2E de Pipeline Real (Artigo Autêntico)
# ==============================================================================
def test_e2e_pipeline_with_real_extracted_selected_json(tmp_path):
"""Valida a conversão E2E de um artigo real extraído do arquivo out/river_plate_extracted_selected.json."""
sample_source = Path(__file__).parent.parent.parent / "out" / "river_plate_extracted_selected.json"
if not sample_source.exists():
pytest.skip(
"Arquivo out/river_plate_extracted_selected.json não encontrado para teste de integração real."
)
data = json.loads(sample_source.read_text(encoding="utf-8"))
assert "articles" in data and len(data["articles"]) > 0
first_article = data["articles"][0]
input_json = tmp_path / "river_first_article.json"
input_json.write_text(json.dumps(first_article, indent=2, ensure_ascii=False), encoding="utf-8")
output_md = tmp_path / "river_first_article.md"
res = subprocess.run(
[sys.executable, str(SCRIPT_PATH), "-i", str(input_json), "-o", str(output_md)],
capture_output=True,
text=True,
)
assert res.returncode == 0, f"Erro na conversão E2E real: {res.stderr}"
assert output_md.exists()
md_content = output_md.read_text(encoding="utf-8")
assert md_content.startswith("# ")
assert "**Fonte original:** [" in md_content
assert "\n\n---\n\n" in md_content
assert len(md_content.splitlines()) > 10
def test_convert_article_api_direct(tmp_path):
"""Testa a chamada direta da função convert_article em código Python."""
in_file = tmp_path / "direct.json"
in_file.write_text(
json.dumps(
{
"selected_extractor": "trafilatura",
"crawled_url": "https://example.com/direct",
"trafilatura": {"title": "Título Direto", "markdown": "Conteúdo direto."},
}
),
encoding="utf-8",
)
out_file = tmp_path / "direct.md"
result = convert_article(in_file, out_file)
assert result == out_file
assert out_file.exists()
assert "# Título Direto" in out_file.read_text(encoding="utf-8")
def test_parse_arguments_api_direct():
"""Testa a chamada direta do parse_arguments."""
args = parse_arguments(["-i", "input_test.json", "-o", "output_test.md"])
assert args.input == Path("input_test.json")
assert args.output == Path("output_test.md")
@@ -0,0 +1,388 @@
"""
Suíte de Testes E2E e de Integração Completa para Análise de Texto e Classificação de Inerência (QA Sênior).
Valida todo o funil de análise de texto:
1. Sucesso no NLP Determinístico (Tier 1 DIRECT_INHERENT com bypass de LLM).
2. Insucesso / Rejeição no NLP (Tier 1 NOT_RELATED por âncoras negativas e homônimos).
3. Ambiguidade / Limiar detectada no NLP e resolvida com sucesso no LLM (Tier 3 Upgrade).
4. Ambiguidade confirmada pelo LLM como TANGENTIAL (Tier 3 Confirmation).
5. Insucesso / Falha de API de LLM com degradação graciosa para Tier 1.
6. Cobertura Multilíngue E2E nos 6 idiomas (PT, EN, ES, DE, IT, FR).
7. Execução E2E via CLI subprocess com contratos de entrada, saída e flags.
8. Chamada real ao vivo a provedores de LLM (OpenAI/Gemini) quando credenciais estiverem disponíveis.
"""
from __future__ import annotations
import json
import subprocess
import sys
from pathlib import Path
import pytest
from src.tools.adapters.llm import LLMFallbackAdapter
from src.tools.classifier import InherenceClassifier
from src.tools.models import (
ClassificationResult,
DecisionCategory,
ECPSnapshot,
RelatedEntity,
)
CLASSIFY_CLI = Path(__file__).parent.parent.parent / "classify.py"
# ==============================================================================
# Fixtures e Helpers para o Funil de Teste E2E
# ==============================================================================
@pytest.fixture
def ecp_river_plate() -> ECPSnapshot:
"""Fixture ECP oficial para o Club Atlético River Plate."""
return ECPSnapshot(
target_entity_id="ecp_river_plate",
target_name="Club Atlético River Plate",
aliases=[
"Club Atlético River Plate",
"River Plate",
"River",
"El Millonario",
"La Banda",
"CARP",
],
domain="Futebol / Esportes",
anchors=[
"fútbol",
"Copa Libertadores",
"Libertadores",
"Copa Sudamericana",
"Sudamericana",
"Monumental",
"Estadio Monumental",
"Eduardo Coudet",
"Coudet",
"Nicolás Otamendi",
"Otamendi",
"Rafael Santos Borré",
],
negative_anchors=[
"River Plate de Montevideo",
"River Plate de Asunción",
"River de Piauí",
"Rio da Prata",
"Bacia do Rio da Prata",
],
related_entities=[
RelatedEntity(
entity_id="estadio_monumental",
name="Estadio Mâs Monumental",
relation_type="HOME_VENUE_OF",
weight=0.95,
aliases=["Monumental", "El Monumental"],
),
RelatedEntity(
entity_id="copa_sudamericana",
name="Copa Sudamericana",
relation_type="COMPETES_IN",
weight=0.85,
aliases=["Sudamericana"],
),
],
)
# ==============================================================================
# 1. Funil de Sucesso NLP Determinístico (Tier 1)
# ==============================================================================
def test_funnel_nlp_deterministic_success(ecp_river_plate: ECPSnapshot):
"""
Cenário 1: Artigo com alta densidade de âncoras do River Plate.
Oráculo: Decisão DIRECT_INHERENT, confiança alta (>=0.95), LLM não é chamado.
"""
content = """
# River Plate vence com autoridade na Copa Sudamericana
Em noite histórica no Estadio Monumental, o River dominou a partida sob o comando de Eduardo Coudet.
Otamendi e Borré marcaram os gols que garantiram a classificação na Sudamericana.
"""
llm_called = {"status": False}
def mock_llm(prompt: str) -> str:
llm_called["status"] = True
return json.dumps({"decision": "DIRECT_INHERENT", "confidence": 0.99})
adapter = LLMFallbackAdapter(provider_fn=mock_llm)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
result = classifier.classify(ecp_river_plate, content)
assert result.decision == DecisionCategory.DIRECT_INHERENT
assert result.is_inherent is True
assert result.confidence >= 0.95
assert "River Plate" in result.matched_anchors or "River" in result.matched_anchors
assert len(result.graph_matches) >= 1
# Otimização de custo: LLM NÃO deve ser chamado em casos determinísticos claros
assert llm_called["status"] is False
# ==============================================================================
# 2. Funil de Insucesso / Rejeição NLP (Tier 1 Negativas e Homônimos)
# ==============================================================================
def test_funnel_nlp_deterministic_rejection_homonym(ecp_river_plate: ECPSnapshot):
"""
Cenário 2: Artigo sobre a Bacia do Rio da Prata ou clube homônimo do Uruguai.
Oráculo: Decisão NOT_RELATED, is_inherent=False, âncoras negativas detectadas.
"""
content = """
# Expedição ambiental navega pela Bacia do Rio da Prata
Pesquisadores mapearam a biodiversidade fluvial e os sedimentos do Rio da Prata durante o verão.
"""
classifier = InherenceClassifier(enable_llm=False)
result = classifier.classify(ecp_river_plate, content)
assert result.decision == DecisionCategory.NOT_RELATED
assert result.is_inherent is False
assert any("Rio da Prata" in neg for neg in result.negative_matches)
assert "Negative anchor" in result.rationale
# ==============================================================================
# 3. Funil de Ambiguidade NLP -> Resolução com Sucesso no LLM (Tier 3)
# ==============================================================================
def test_funnel_nlp_ambiguity_resolved_by_llm_upgrade(ecp_river_plate: ECPSnapshot):
"""
Cenário 3: Menção isolada do clube ('River') em contexto com poucas âncoras explícitas.
Tier 1 preliminar: TANGENTIAL (baixa densidade).
LLM Fallback: Analisa o contexto profundo e eleva para DIRECT_INHERENT.
"""
ambiguous_content = """
# Bastidores do mercado sul-americano
A diretoria do River finalizou os últimos detalhes contratuais para a renovação de jovens promessas.
"""
mock_llm_response = json.dumps(
{
"analysis_summary": "O artigo trata da gestão de elenco e renovações contratuais do clube River Plate.",
"decision": "DIRECT_INHERENT",
"confidence": 0.94,
"rationale": "A análise contextual profunda comprova que a matéria é focada na administração do River Plate.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda prompt: mock_llm_response)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
result = classifier.classify(ecp_river_plate, ambiguous_content)
assert result.decision == DecisionCategory.DIRECT_INHERENT
assert result.is_inherent is True
assert result.confidence == 0.94
assert "[Tier 3 LLM]" in result.rationale
assert "[Tier 3 LLM Override applied]" in result.warnings
# ==============================================================================
# 4. Funil de Ambiguidade NLP -> Confirmação de Tangencial no LLM
# ==============================================================================
def test_funnel_nlp_ambiguity_confirmed_tangential_by_llm(ecp_river_plate: ECPSnapshot):
"""
Cenário 4: Menção metafórica ou turística a um local próximo.
Tier 1 preliminar: TANGENTIAL.
LLM Fallback: Confirma que é meramente periférico/ilustrativo.
"""
tangential_content = """
# Melhores restaurantes do bairro de Núñez em Buenos Aires
Ao visitar a capital portenha, próximo de onde fica o River, você encontra excelentes opções gastronômicas.
"""
mock_llm_response = json.dumps(
{
"analysis_summary": "Guia gastronômico sobre o bairro de Núñez com citação geográfica casual ao clube.",
"decision": "TANGENTIAL",
"confidence": 0.96,
"rationale": "A entidade é usada apenas como ponto de referência geográfica em um artigo sobre restaurantes.",
}
)
adapter = LLMFallbackAdapter(provider_fn=lambda prompt: mock_llm_response)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
result = classifier.classify(ecp_river_plate, tangential_content)
assert result.decision == DecisionCategory.TANGENTIAL
assert result.is_inherent is False
assert result.confidence == 0.96
assert "[Tier 3 LLM]" in result.rationale
# ==============================================================================
# 5. Funil de Insucesso / Degradação Graciosa em Falha do LLM
# ==============================================================================
def test_funnel_llm_failure_graceful_degradation(ecp_river_plate: ECPSnapshot):
"""
Cenário 5: LLM configurado, caso ambíguo, mas a API externa sofre timeout/500.
Oráculo: Mantém o resultado do Tier 1 determinístico com warning detalhado e sem quebrar.
"""
def broken_llm(prompt: str) -> str:
raise TimeoutError("Conexão com serviço de LLM excedeu 30 segundos.")
adapter = LLMFallbackAdapter(provider_fn=broken_llm)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
content = "O River esteve presente no evento de inauguração da praça."
result = classifier.classify(ecp_river_plate, content)
# Mantém o Tier 1 determinístico
assert result.decision == DecisionCategory.TANGENTIAL
assert result.is_inherent is False
assert any("LLM fallback failed" in w for w in result.warnings)
# ==============================================================================
# 6. Cobertura Multilíngue nos 6 Idiomas (PT, EN, ES, DE, IT, FR)
# ==============================================================================
@pytest.mark.parametrize(
"lang_code,content,expected_lang",
[
("pt", "# Petrobras anuncia perfuração no pré-sal com tecnologia nacional.", "pt"),
("en", "# Apple unveils new generative AI features for upcoming devices.", "en"),
(
"es",
"# River Plate prepara su viaje a Bogotá para disputar el torneo continental.",
"es",
),
(
"de",
"# Volkswagen investiert Milliarden in neue Batterie-Fabriken in Deutschland.",
"de",
),
("it", "# Ferrari conquista la pole position nel Gran Premio di Monza.", "it"),
(
"fr",
"# L'entreprise TotalEnergies accélère ses investissements solaires en France.",
"fr",
),
],
)
def test_funnel_multilingual_language_detection(
lang_code: str, content: str, expected_lang: str, ecp_river_plate: ECPSnapshot
):
"""Garante a identificação precisa de idioma e integridade nos 6 idiomas suportados."""
classifier = InherenceClassifier(enable_llm=False)
result = classifier.classify(ecp_river_plate, content)
assert result.detected_language == expected_lang
# ==============================================================================
# 7. Execução E2E via CLI Subprocess
# ==============================================================================
def test_funnel_cli_subprocess_end_to_end(tmp_path: Path):
"""Valida o contrato CLI completo classify.py com saída em arquivo JSON e flags ativas."""
ecp_path = tmp_path / "ecp.json"
ecp_path.write_text(
json.dumps(
{
"target_entity_id": "ecp_test_e2e",
"target_name": "Clube Teste",
"aliases": ["Clube Teste", "Clube"],
"domain": "Esportes",
"anchors": ["campeonato", "vitória", "torneio"],
}
),
encoding="utf-8",
)
doc_path = tmp_path / "artigo.md"
doc_path.write_text(
"# Clube Teste comemora vitória histórica no campeonato\n\nEquipe foi campeã do torneio.",
encoding="utf-8",
)
out_path = tmp_path / "resultado.json"
res = subprocess.run(
[
sys.executable,
str(CLASSIFY_CLI),
"--ecp",
str(ecp_path),
"--content",
str(doc_path),
"--output",
str(out_path),
"--enable-llm",
],
capture_output=True,
text=True,
)
assert res.returncode == 0, f"Erro na CLI: {res.stderr}"
assert out_path.exists()
payload = json.loads(out_path.read_text(encoding="utf-8"))
assert payload["decision"] == "DIRECT_INHERENT"
assert payload["is_inherent"] is True
assert payload["confidence"] >= 0.85
assert isinstance(payload["matched_anchors"], list)
assert isinstance(payload["evidence"], list)
# ==============================================================================
# 8. Teste Live Opt-In com API Real (OpenAI / Gemini) se .env Estiver Presente
# ==============================================================================
def test_funnel_live_api_execution_if_configured():
"""
Executa chamada ao vivo contra OpenAI ou Gemini caso OPENAI_API_KEY ou GEMINI_API_KEY
esteja configurada no ambiente ou no arquivo .env.
"""
adapter = LLMFallbackAdapter()
if not (adapter.openai_api_key or adapter.gemini_api_key):
pytest.skip(
"Chaves de API reais (OPENAI_API_KEY ou GEMINI_API_KEY) não configuradas no .env"
)
ecp = ECPSnapshot(
target_entity_id="ecp_live_test",
target_name="Club Atlético River Plate",
aliases=["River Plate", "River"],
domain="Futebol",
anchors=["Monumental", "Libertadores"],
)
initial_res = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="es",
matched_anchors=["River"],
negative_matches=[],
graph_matches=[],
evidence=["River"],
rationale="Passing mention detected by Tier 1.",
warnings=[],
)
# Texto de teste para a API ao vivo
content = "O River Plate empatou em 1 a 1 em Bogotá com gols de Otamendi na Copa Sul-Americana."
refined = adapter.disambiguate(ecp, content, initial_res)
assert refined is not None
assert refined.decision in [
DecisionCategory.DIRECT_INHERENT,
DecisionCategory.CONTEXTUAL_INHERENT,
]
assert refined.is_inherent is True
assert "[Tier 3 LLM]" in refined.rationale
@@ -0,0 +1,372 @@
#!/usr/bin/env python3
"""
Testes automatizados para o Extrator e Parser Multimotor de Artigos.
Cobre modelos de dados, parsers (Trafilatura, Newspaper4k, Readability),
isolamento de falhas, orquestração de lote e interface CLI.
"""
from __future__ import annotations
import json
from pathlib import Path
from unittest.mock import patch
import pytest
from scripts.extract_article_contents import (
ArticleCrawler,
ExtractedArticle,
ExtractionBatchReport,
InputArticle,
NewspaperData,
NewspaperExtractor,
ReadabilityData,
ReadabilityExtractor,
TrafilaturaData,
TrafilaturaExtractor,
extract_all_engines,
load_search_json,
main,
process_batch,
save_extracted_json,
)
SAMPLE_HTML = """
<!DOCTYPE html>
<html lang="es">
<head>
<meta charset="utf-8">
<title>River Plate igualó sin goles ante Independiente Santa Fe - Olé</title>
<meta name="description" content="El equipo de Núñez empató 0-0 en Bogotá por los octavos de final.">
<meta name="author" content="Juan Pérez">
<meta property="og:title" content="River Plate igualó sin goles ante Independiente Santa Fe">
<meta property="og:image" content="https://media.ole.com.ar/river.jpg">
</head>
<body>
<header><nav><a href="/">Inicio</a></nav></header>
<article>
<h1>River Plate igualó sin goles ante Independiente Santa Fe</h1>
<p class="byline">Por Juan Pérez - 20 de Agosto de 2026</p>
<p class="lead">El equipo de Núñez empató 0-0 en Bogotá por la Copa Sudamericana.</p>
<p>Franco Armani fue la gran figura del encuentro con tres atajadas espectaculares en el primer tiempo.</p>
<p>El partido de vuelta se disputará en el estadio Monumental la próxima semana ante una multitud.</p>
</article>
<footer><p>Copyright 2026 Olé</p></footer>
</body>
</html>
"""
@pytest.fixture
def sample_input_json(tmp_path: Path) -> Path:
data = {
"query": "River Plate",
"language": "es",
"locale": "AR",
"total_itens": 2,
"items": [
{
"titulo": "River Plate igualó sin goles ante Santa Fe",
"subtitulo": "Empate en Bogotá",
"quando_publicado": "Thu, 20 Aug 2026 03:27:26 GMT",
"url": "https://www.ole.com.ar/river-0-0-santa-fe.html",
"pagina": 1,
},
{
"titulo": "Armani fue la figura de River",
"subtitulo": "Gran actuación del arquero",
"quando_publicado": "Thu, 20 Aug 2026 04:00:00 GMT",
"url": "https://www.tycsports.com/armani-figura.html",
"pagina": 1,
},
],
}
input_file = tmp_path / "river_plate.json"
input_file.write_text(json.dumps(data, ensure_ascii=False), encoding="utf-8")
return input_file
# ==============================================================================
# 1. Testes de Modelos e I/O de JSON
# ==============================================================================
def test_input_article_creation():
article = InputArticle(
titulo="Notícia Teste",
url="https://example.com/noticia",
subtitulo="Subtítulo",
quando_publicado="Thu, 20 Aug 2026",
pagina=1,
)
assert article.titulo == "Notícia Teste"
assert article.url == "https://example.com/noticia"
assert article.pagina == 1
d = article.to_dict()
assert d["titulo"] == "Notícia Teste"
assert d["url"] == "https://example.com/noticia"
def test_load_search_json_valid(sample_input_json: Path):
query, lang, items = load_search_json(sample_input_json)
assert query == "River Plate"
assert lang == "es"
assert len(items) == 2
assert items[0].url == "https://www.ole.com.ar/river-0-0-santa-fe.html"
def test_load_search_json_invalid_file(tmp_path: Path):
non_existent = tmp_path / "missing.json"
with pytest.raises(FileNotFoundError):
load_search_json(non_existent)
def test_save_extracted_json(tmp_path: Path):
report = ExtractionBatchReport(
source_file="test.json",
processed_at="2026-08-20T12:00:00Z",
total_articles=1,
successful_articles=1,
failed_articles=0,
articles=[
ExtractedArticle(
input_meta=InputArticle(titulo="Teste", url="https://example.com/noticia"),
extraction_status="success",
error_message=None,
crawled_url="https://example.com/noticia",
page_title="Página Teste",
http_status=200,
trafilatura=TrafilaturaData(
title="Teste",
author="Autor",
date="2026-08-20",
description="Desc",
categories=[],
tags=[],
canonical_url=None,
text="Texto do teste.",
raw_json=None,
error=None,
),
newspaper4k=NewspaperData(
title="Teste",
authors=["Autor"],
publish_date="2026-08-20",
text="Texto do teste.",
summary="Resumo",
keywords=["teste"],
top_image=None,
images=[],
meta_data={},
error=None,
),
readability=ReadabilityData(
title="Teste",
short_title="Teste",
cleaned_html="<p>Texto do teste.</p>",
cleaned_text="Texto do teste.",
error=None,
),
)
],
)
out_file = tmp_path / "out" / "result.json"
save_extracted_json(report, out_file)
assert out_file.exists()
content = json.loads(out_file.read_text(encoding="utf-8"))
assert content["total_articles"] == 1
assert content["articles"][0]["trafilatura"]["title"] == "Teste"
# ==============================================================================
# 2. Testes Unitários dos Parsers (Trafilatura, Newspaper4k, Readability)
# ==============================================================================
def test_trafilatura_extractor():
extractor = TrafilaturaExtractor()
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia")
assert isinstance(res, TrafilaturaData)
assert res.error is None
assert "Armani" in res.text or "River Plate" in res.text
assert res.title is not None
def test_newspaper_extractor():
extractor = NewspaperExtractor()
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es")
assert isinstance(res, NewspaperData)
assert res.error is None
assert "Armani" in res.text or "River" in res.text
assert isinstance(res.keywords, list)
assert len(res.keywords) > 0
def test_readability_extractor():
extractor = ReadabilityExtractor()
res = extractor.extract(SAMPLE_HTML)
assert isinstance(res, ReadabilityData)
assert res.error is None
assert res.title is not None
assert res.cleaned_html is not None
assert "Armani" in (res.cleaned_text or "") or "River" in (res.cleaned_text or "")
def test_extract_all_engines():
traf, news, read = extract_all_engines(
SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es"
)
assert traf.error is None
assert news.error is None
assert read.error is None
# ==============================================================================
# 3. Testes de Isolamento de Falhas (Resiliência)
# ==============================================================================
def test_extractor_error_isolation_on_faulty_engine():
with patch(
"scripts.extract_article_contents.TrafilaturaExtractor.extract",
side_effect=RuntimeError("Trafilatura crash"),
):
traf, news, read = extract_all_engines(
SAMPLE_HTML, url="https://example.com", language="es"
)
assert traf.error == "Trafilatura crash"
assert news.error is None
assert read.error is None
def test_crawler_error_isolation(sample_input_json: Path, tmp_path: Path):
out_file = tmp_path / "river_plate_extracted.json"
def mock_crawl(url, timeout_sec=30):
if "ole.com.ar" in url:
return SAMPLE_HTML, "River Plate Olé", 200
raise ConnectionError("Connection refused by tycsports.com")
with patch.object(ArticleCrawler, "crawl", side_effect=mock_crawl):
with patch.object(ArticleCrawler, "start"), patch.object(ArticleCrawler, "close"):
report = process_batch(
input_path=sample_input_json,
output_path=out_file,
silent=True,
)
assert report.total_articles == 2
assert report.successful_articles == 1
assert report.failed_articles == 1
assert report.articles[0].extraction_status == "success"
assert report.articles[1].extraction_status == "failed"
assert "Connection refused" in (report.articles[1].error_message or "")
# ==============================================================================
# 4. Testes de CLI e Limitação (--limit, --silent, --language)
# ==============================================================================
def test_process_batch_with_limit(sample_input_json: Path, tmp_path: Path):
out_file = tmp_path / "limit_extracted.json"
with (
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
patch.object(ArticleCrawler, "start"),
patch.object(ArticleCrawler, "close"),
):
report = process_batch(
input_path=sample_input_json,
output_path=out_file,
limit=1,
silent=True,
)
assert report.total_articles == 1
assert len(report.articles) == 1
assert out_file.exists()
def test_cli_main_success(sample_input_json: Path, tmp_path: Path, capsys):
out_file = tmp_path / "cli_out.json"
with (
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
patch.object(ArticleCrawler, "start"),
patch.object(ArticleCrawler, "close"),
):
exit_code = main(["-i", str(sample_input_json), "-o", str(out_file), "--limit", "1", "-s"])
assert exit_code == 0
assert out_file.exists()
def test_cli_main_missing_input_file(tmp_path: Path, capsys):
missing_file = tmp_path / "does_not_exist.json"
exit_code = main(["-i", str(missing_file), "-s"])
assert exit_code == 1
# ==============================================================================
# 5. Testes End-to-End (E2E) ao Vivo (Live Network)
# ==============================================================================
def test_e2e_live_article_extraction(tmp_path: Path):
"""Valida E2E a extração real ao vivo com Foxcape e os 3 motores em lote."""
live_input_file = Path("out/river_plate.json")
if not live_input_file.exists():
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E ao vivo.")
out_file = tmp_path / "e2e_live_extracted.json"
# Executa o batch real com limite de 1 notícia
report = process_batch(
input_path=live_input_file,
output_path=out_file,
limit=1,
silent=True,
)
assert report.total_articles == 1
assert report.successful_articles == 1
assert report.failed_articles == 0
assert len(report.articles) == 1
art = report.articles[0]
assert art.extraction_status == "success"
assert art.crawled_url.startswith("http")
# Valida que todos os 3 motores extraíram dados reais
assert art.trafilatura is not None and art.trafilatura.error is None
assert len(art.trafilatura.text) > 50
assert art.newspaper4k is not None and art.newspaper4k.error is None
assert len(art.newspaper4k.text) > 50
assert isinstance(art.newspaper4k.keywords, list)
assert art.readability is not None and art.readability.error is None
assert art.readability.cleaned_html is not None
assert len(art.readability.cleaned_text or "") > 50
# Valida arquivo JSON gravado
assert out_file.exists()
saved = json.loads(out_file.read_text(encoding="utf-8"))
assert saved["total_articles"] == 1
assert saved["articles"][0]["extraction_status"] == "success"
def test_e2e_cli_live_execution(tmp_path: Path):
"""Valida E2E a execução do CLI real de ponta a ponta."""
live_input_file = Path("out/river_plate.json")
if not live_input_file.exists():
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E.")
out_file = tmp_path / "e2e_cli_live.json"
exit_code = main(["-i", str(live_input_file), "-o", str(out_file), "--limit", "1", "-s"])
assert exit_code == 0
assert out_file.exists()
data = json.loads(out_file.read_text(encoding="utf-8"))
assert data["successful_articles"] == 1
+345
View File
@@ -0,0 +1,345 @@
"""
Testes unitários e de integração para o Extrator de Manchetes do Google News.
Cobre validação de entrada, mapeamento de idiomas/locales, parsing e
higienização de XML/HTML, orquestração, decodificação de URLs e contrato de execução CLI.
"""
from __future__ import annotations
import json
from pathlib import Path
from unittest.mock import patch
import pytest
from scripts.extract_google_news import (
ExtractionResult,
NewsArticle,
SearchQuery,
extract_google_news,
get_hl_gl_ceid,
main,
parse_google_news_rss,
resolve_article_url,
resolve_articles_urls,
)
FIXTURE_PATH = Path(__file__).parent.parent / "fixtures" / "google_news_sample.xml"
@pytest.fixture
def sample_rss_xml() -> str:
"""Fixture que fornece o conteúdo do XML de exemplo para testes offline."""
return FIXTURE_PATH.read_text(encoding="utf-8")
def test_get_hl_gl_ceid_default_mappings():
"""Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)."""
hl, gl, ceid = get_hl_gl_ceid("pt")
assert hl == "pt-BR"
assert gl == "BR"
assert ceid == "BR:pt-BR"
hl, gl, ceid = get_hl_gl_ceid("en")
assert hl == "en-US"
assert gl == "US"
assert ceid == "US:en-US"
hl, gl, ceid = get_hl_gl_ceid("es")
assert hl == "es-419"
assert gl == "AR"
assert ceid == "AR:es-419"
hl, gl, ceid = get_hl_gl_ceid("de")
assert hl == "de"
assert gl == "DE"
assert ceid == "DE:de"
def test_get_hl_gl_ceid_with_custom_locale():
"""Valida a sobrescrita geográfica quando o argumento locale é especificado."""
hl, gl, ceid = get_hl_gl_ceid("es", locale="MX")
assert hl == "es-419"
assert gl == "MX"
assert ceid == "MX:es-419"
hl, gl, ceid = get_hl_gl_ceid("en", locale="GB")
assert hl == "en-GB"
assert gl == "GB"
assert ceid == "GB:en-GB"
hl, gl, ceid = get_hl_gl_ceid("es", locale="ES")
assert hl == "es"
assert gl == "ES"
assert ceid == "ES:es"
def test_get_hl_gl_ceid_dynamic_fallback():
"""Valida fallback dinâmico para idiomas regionais não listados explicitamente."""
hl, gl, ceid = get_hl_gl_ceid("ja_jp")
assert hl == "ja-JP"
assert gl == "JP"
assert ceid == "JP:ja-JP"
def test_search_query_validation():
"""Valida as regras de negócio e limites de SearchQuery."""
# Instanciação válida
q = SearchQuery(keyword="inteligencia artificial", language="pt", max_pages=1)
assert q.clean_keyword == "inteligencia artificial"
assert q.clean_language == "pt"
assert q.clean_locale is None
assert q.max_pages == 1
# Palavra-chave vazia ou apenas espaços deve lançar ValueError
with pytest.raises(ValueError, match="palavra-chave"):
SearchQuery(keyword=" ", language="pt")
# Idioma com menos de 2 caracteres deve lançar ValueError
with pytest.raises(ValueError, match="idioma"):
SearchQuery(keyword="test", language="p")
# Intervalo de páginas fora de 1..10 deve lançar ValueError
with pytest.raises(ValueError, match="páginas"):
SearchQuery(keyword="test", language="pt", max_pages=0)
with pytest.raises(ValueError, match="páginas"):
SearchQuery(keyword="test", language="pt", max_pages=11)
def test_parse_google_news_rss_with_fixture(sample_rss_xml: str):
"""Valida o parsing do feed RSS, higienização de tags HTML e deduplicação."""
articles = parse_google_news_rss(sample_rss_xml, max_pages=1)
# 4 itens no fixture, mas 1 não possui link -> exatamente 3 válidos
assert len(articles) == 3
# Artigo 1: InfoMoney com HTML no description que deve ser limpo
art1 = articles[0]
assert "InfoMoney" in art1.titulo
assert art1.url.startswith("https://news.google.com/rss/articles/")
assert art1.quando_publicado == "Thu, 20 Aug 2026 10:30:00 GMT"
assert art1.pagina == 1
assert "<" not in (art1.subtitulo or "")
assert ">" not in (art1.subtitulo or "")
assert "crescimento expressivo" in (art1.subtitulo or "")
# Artigo 2: G1 com parágrafos limpos
art2 = articles[1]
assert "G1" in art2.titulo
assert "<p>" not in (art2.subtitulo or "")
# Artigo 3: Folha com descrição redundante/igual ao título -> subtitulo deve ser None
art3 = articles[2]
assert art3.subtitulo is None
def test_resolve_article_url_fallback():
"""Valida fallback gracioso de URL quando não é link do Google News ou em erro."""
direct_url = "https://www.globo.com/noticia/123"
assert resolve_article_url(direct_url) == direct_url
with patch("scripts.extract_google_news.gnewsdecoder", return_value={"status": False}):
gn_url = "https://news.google.com/rss/articles/fake_token"
assert resolve_article_url(gn_url) == gn_url
def test_resolve_article_url_success():
"""Valida resolução bem-sucedida de URL do Google News para o portal destino."""
gn_url = "https://news.google.com/rss/articles/valid_token"
dest_url = "https://infomoney.com.br/mercados/artigo-ia"
with patch(
"scripts.extract_google_news.gnewsdecoder",
return_value={"status": True, "decoded_url": dest_url},
):
resolved = resolve_article_url(gn_url)
assert resolved == dest_url
def test_resolve_articles_urls_batch():
"""Valida a resolução concorrente em lote de uma lista de NewsArticle."""
articles = [
NewsArticle(
titulo="Notícia 1",
url="https://news.google.com/rss/articles/1",
pagina=1,
),
NewsArticle(
titulo="Notícia 2",
url="https://news.google.com/rss/articles/2",
pagina=1,
),
]
with patch(
"scripts.extract_google_news.resolve_article_url",
side_effect=lambda u: f"https://destinofinal.com/{u.split('/')[-1]}",
):
resolved = resolve_articles_urls(articles)
assert len(resolved) == 2
assert resolved[0].url == "https://destinofinal.com/1"
assert resolved[1].url == "https://destinofinal.com/2"
def test_extract_google_news_orchestration_mocked(sample_rss_xml: str):
"""Valida a consolidação do ExtractionResult a partir da busca mockada com URLs resolvidas."""
query = SearchQuery(keyword="inteligência artificial", language="pt", locale="BR", max_pages=1)
with (
patch(
"scripts.extract_google_news._fetch_rss_content",
return_value=sample_rss_xml,
),
patch(
"scripts.extract_google_news.resolve_article_url",
side_effect=lambda u: f"https://resolved.com/{u[-5:]}",
),
):
result = extract_google_news(query, resolve_urls=True)
assert isinstance(result, ExtractionResult)
assert result.query == "inteligência artificial"
assert result.language == "pt"
assert result.locale == "BR"
assert result.total_paginas == 1
assert result.total_itens == 3
assert len(result.items) == 3
assert result.scraped_at is not None
assert result.items[0].url.startswith("https://resolved.com/")
def test_cli_execution_stdout(sample_rss_xml: str, capsys: pytest.CaptureFixture[str]):
"""Valida execução padrão do CLI com saída JSON no stdout."""
with (
patch(
"scripts.extract_google_news._fetch_rss_content",
return_value=sample_rss_xml,
),
patch("scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u),
):
exit_code = main(["--query", "inteligencia artificial", "--lang", "pt", "--pretty"])
assert exit_code == 0
captured = capsys.readouterr()
data = json.loads(captured.out)
assert data["query"] == "inteligencia artificial"
assert data["total_itens"] == 3
assert len(data["items"]) == 3
# Validar indentação presente por causa de --pretty
assert "\n " in captured.out
def test_cli_execution_file_output(sample_rss_xml: str, tmp_path: Path):
"""Valida gravação em arquivo com criação automática de diretórios pais."""
out_file = tmp_path / "sub_dir" / "news_out.json"
with (
patch(
"scripts.extract_google_news._fetch_rss_content",
return_value=sample_rss_xml,
),
patch("scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u),
):
exit_code = main(["-q", "IA", "-p", "1", "-o", str(out_file)])
assert exit_code == 0
assert out_file.exists()
data = json.loads(out_file.read_text(encoding="utf-8"))
assert data["total_itens"] == 3
def test_cli_empty_query_error(capsys: pytest.CaptureFixture[str]):
"""Valida tratamento de erro e código de saída 1 para parâmetro vazio."""
exit_code = main(["--query", " "])
assert exit_code == 1
captured = capsys.readouterr()
assert "Erro de validação" in captured.err
def test_cli_network_error_handling(capsys: pytest.CaptureFixture[str]):
"""Valida tratamento de erro e código de saída 2 para falhas de rede."""
with patch(
"scripts.extract_google_news._fetch_rss_content",
side_effect=RuntimeError("Connection refused"),
):
exit_code = main(["--query", "IA"])
assert exit_code == 2
captured = capsys.readouterr()
assert "Erro na extração" in captured.err
assert "Connection refused" in captured.err
# ---------------------------------------------------------------------------
# Testes E2E (End-to-End) com resolução real de rede e validação de URLs finais
# ---------------------------------------------------------------------------
def test_e2e_resolve_real_google_news_url():
"""Valida E2E que o decodificador resolve uma URL real do Google News para o veículo de imprensa."""
# URL real de artigo extraída do Google News RSS
sample_gn_url = (
"https://news.google.com/rss/articles/"
"CBMi6wFBVV95cUxQUS0tMGxacUJycDlCWXpWSGR3T0hfR1E0Q0txWjdqek9LWjNZTUo1WEx3"
"UEw3Skt1eW1Wa2VYUUE2ajNYMmpEckhsdGFXUGtmQktYX2JWa1hmclhEZEVIa3hNODhpVXNO"
"UGN4cDhmWmJMczBEUVFDNS1aX3EzRHh6VmQ3cVY0ZnZmaW9YWDZJSE9UNFJ2dXpyNFlHaVVs"
"VlY0V1FiZ0tzZ3FpRVhUYnhPbmdnOFRveW5oOVB3WDAzS3c0eWFjMDBZSERwNmRkRk1MRHZF"
"UE92UE9GMmpRcFZ5cUU2Ym1NeDdQU2tn?oc=5"
)
resolved_url = resolve_article_url(sample_gn_url)
# Não deve mais ser URL do Google News
assert "news.google.com" not in resolved_url
# Deve ser uma URL absoluta http/https apontando para o portal real (TyC Sports)
assert resolved_url.startswith("http")
assert "tycsports.com" in resolved_url
def test_e2e_extract_google_news_live_pipeline():
"""Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo."""
query = SearchQuery(keyword="tecnologia", language="pt", locale="BR", max_pages=1)
result = extract_google_news(query, resolve_urls=True)
assert isinstance(result, ExtractionResult)
assert result.total_itens > 0
assert len(result.items) == result.total_itens
for article in result.items:
assert article.titulo
assert article.url.startswith("http")
# Garante que as URLs foram decodificadas e não permanecem no formato intermediário
assert "news.google.com/rss/articles/" not in article.url
def test_e2e_cli_live_file_output(tmp_path: Path):
"""Valida E2E a execução do CLI com saída real em arquivo e URLs decodificadas."""
out_file = tmp_path / "e2e_result.json"
exit_code = main(
[
"--query",
"economia",
"--lang",
"pt",
"--max-pages",
"1",
"--output",
str(out_file),
]
)
assert exit_code == 0
assert out_file.exists()
data = json.loads(out_file.read_text(encoding="utf-8"))
assert data["query"] == "economia"
assert data["total_itens"] > 0
assert len(data["items"]) > 0
first_item = data["items"][0]
assert first_item["titulo"]
assert first_item["url"].startswith("http")
assert "news.google.com/rss/articles/" not in first_item["url"]
+61
View File
@@ -0,0 +1,61 @@
"""Unit tests for language detection and text normalization."""
from src.tools.language import detect_language, normalize_text
def test_normalize_text():
assert normalize_text("São Paulo & Petróleo") == "sao paulo & petroleo"
assert normalize_text(
"Über große Veränderungen"
) == "uber grosse veranderungen" or "uber" in normalize_text("Über")
assert normalize_text("Crème brûlée") == "creme brulee"
def test_detect_portuguese():
text = "A Petrobras anunciou um novo plano de investimentos para a exploração de petróleo na camada pré-sal."
lang, conf = detect_language(text)
assert lang == "pt"
assert conf > 0.5
def test_detect_english():
text = "Apple announced its new silicon chip with improved machine learning performance and battery life."
lang, conf = detect_language(text)
assert lang == "en"
assert conf > 0.5
def test_detect_spanish():
text = (
"La empresa petrolera anunció una nueva inversión en el sector energético durante este año."
)
lang, conf = detect_language(text)
assert lang == "es"
assert conf > 0.5
def test_detect_german():
text = "Volkswagen plant eine umfassende Transformation zur Elektromobilität in den kommenden Jahren."
lang, conf = detect_language(text)
assert lang == "de"
assert conf > 0.5
def test_detect_italian():
text = "La Ferrari ha presentato la nuova vettura da competizione per il campionato mondiale di Formula 1."
lang, conf = detect_language(text)
assert lang == "it"
assert conf > 0.5
def test_detect_french():
text = "Le groupe TotalEnergies a confirmé ses nouveaux projets de développement dans les énergies renouvelables."
lang, conf = detect_language(text)
assert lang == "fr"
assert conf > 0.5
def test_empty_language():
lang, conf = detect_language("")
assert lang == "unknown"
assert conf == 0.0
+329
View File
@@ -0,0 +1,329 @@
"""
Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador de Inerência.
Cobre cenários unitários, de integração de pipeline, de parsing estruturado, de desambiguação
de casos limiares e de degradação graciosa em falhas de API conforme o requisito FR-004.
"""
from __future__ import annotations
import json
import subprocess
import sys
from pathlib import Path
from src.tools.adapters.llm import LLMFallbackAdapter
from src.tools.classifier import InherenceClassifier
from src.tools.models import (
ClassificationResult,
DecisionCategory,
ECPSnapshot,
)
SCRIPT_PATH = Path(__file__).parent.parent.parent / "classify.py"
# ==============================================================================
# 1. Testes Unitários do LLMFallbackAdapter
# ==============================================================================
def test_llm_adapter_availability_detection():
"""Valida detecção de disponibilidade por chave de API ou provider customizado."""
# Sem chave e sem provider
adapter_empty = LLMFallbackAdapter(api_key="")
assert adapter_empty.is_available() is False
# Com chave de API
adapter_with_key = LLMFallbackAdapter(api_key="sk-test-key-12345")
assert adapter_with_key.is_available() is True
# Com provider function
adapter_with_fn = LLMFallbackAdapter(
api_key="", provider_fn=lambda p: '{"decision": "DIRECT_INHERENT"}'
)
assert adapter_with_fn.is_available() is True
def test_llm_adapter_build_prompt_structure():
"""Valida a montagem do prompt de desambiguação com metadados do ECP e documento."""
adapter = LLMFallbackAdapter(api_key="test")
ecp = ECPSnapshot(
target_entity_id="ecp_river",
target_name="River Plate",
aliases=["Club Atlético River Plate", "CARP"],
domain="Futebol",
anchors=["Monumental", "Libertadores"],
)
initial_res = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="es",
matched_anchors=["River"],
negative_matches=[],
graph_matches=[],
evidence=["River"],
rationale="Passing mention.",
warnings=[],
)
prompt = adapter.build_prompt(ecp, "# Título do Artigo\n\nConteúdo sobre o jogo.", initial_res)
assert "River Plate" in prompt
assert "Futebol" in prompt
assert "TANGENTIAL" in prompt
assert "Título do Artigo" in prompt
def test_llm_adapter_parsing_valid_json_response():
"""Valida o parsing e instanciação correta do ClassificationResult a partir da resposta do LLM."""
adapter = LLMFallbackAdapter(
provider_fn=lambda p: json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": 0.95,
"rationale": "Artigo detalha o desempenho da equipe no torneio.",
}
)
)
ecp = ECPSnapshot(
target_entity_id="ecp_test",
target_name="Test Entity",
aliases=["Test"],
domain="Tech",
anchors=["cloud"],
)
initial = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=["Test"],
negative_matches=[],
graph_matches=[],
evidence=["Test"],
rationale="Weak match.",
warnings=["Low contextual density."],
)
refined = adapter.disambiguate(ecp, "Document content...", initial)
assert refined is not None
assert refined.decision == DecisionCategory.DIRECT_INHERENT
assert refined.is_inherent is True
assert refined.confidence == 0.95
assert "[Tier 3 LLM]" in refined.rationale
assert "[Tier 3 LLM Override applied]" in refined.warnings
def test_llm_adapter_parsing_json_wrapped_in_markdown_codeblock():
"""Valida extração de JSON quando a resposta do LLM vem formatada em bloco markdown ```json ... ```."""
raw_md_json = '```json\n{\n "decision": "CONTEXTUAL_INHERENT",\n "confidence": 0.88,\n "rationale": "Conexão contextual forte através da subsidiária."\n}\n```'
adapter = LLMFallbackAdapter(provider_fn=lambda p: raw_md_json)
ecp = ECPSnapshot(
target_entity_id="ecp_test",
target_name="Test Entity",
aliases=["Test"],
domain="Tech",
anchors=["cloud"],
)
initial = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=[],
negative_matches=[],
graph_matches=[],
evidence=[],
rationale="Weak match.",
warnings=[],
)
refined = adapter.disambiguate(ecp, "Content...", initial)
assert refined is not None
assert refined.decision == DecisionCategory.CONTEXTUAL_INHERENT
assert refined.is_inherent is True
assert refined.confidence == 0.88
def test_llm_adapter_handling_invalid_and_corrupt_responses():
"""Valida que respostas corrompidas ou JSONs sem campos obrigatórios retornem None com segurança."""
def make_bad_provider(resp_str: str):
def _prov(prompt: str) -> str:
return resp_str
return _prov
for bad_response in [
"Desculpe, não consegui avaliar o texto.",
"{json_invalido_sem_fechamento",
json.dumps({"campo_desconhecido": "valor"}),
json.dumps({"decision": "DECISAO_INEXISTENTE"}),
]:
adapter = LLMFallbackAdapter(provider_fn=make_bad_provider(bad_response))
ecp = ECPSnapshot(
target_entity_id="ecp_test",
target_name="Test Entity",
aliases=["Test"],
domain="Tech",
anchors=["cloud"],
)
initial = ClassificationResult(
decision=DecisionCategory.TANGENTIAL,
is_inherent=False,
confidence=0.40,
detected_language="pt",
matched_anchors=[],
negative_matches=[],
graph_matches=[],
evidence=[],
rationale="Initial.",
warnings=[],
)
assert adapter.disambiguate(ecp, "Content...", initial) is None
# ==============================================================================
# 2. Testes de Integração de Pipeline (InherenceClassifier com Tier 3)
# ==============================================================================
def test_classifier_triggers_tier3_on_ambiguous_tangential_case():
"""
Garante que o classificador dispare o Tier 3 LLM para casos ambíguos (TANGENTIAL)
e adote o refinamento retornado.
"""
mock_adapter = LLMFallbackAdapter(
provider_fn=lambda prompt: json.dumps(
{
"decision": "DIRECT_INHERENT",
"confidence": 0.92,
"rationale": "Análise profunda revelou que o texto é focado na entidade alvo.",
}
)
)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=mock_adapter)
ecp = ECPSnapshot(
target_entity_id="ecp_empresa",
target_name="EmpresaAlfa",
aliases=["EmpresaAlfa"],
domain="Tecnologia",
anchors=["software", "computação em nuvem"],
)
# Texto com menção única sem âncoras temáticas (Tier 1 produziria TANGENTIAL)
ambiguous_content = "A EmpresaAlfa esteve presente no evento de encerramento anual da cidade."
result = classifier.classify(ecp, ambiguous_content)
# Como enable_llm=True e o caso era TANGENTIAL, o Tier 3 substitui a decisão
assert result.decision == DecisionCategory.DIRECT_INHERENT
assert result.is_inherent is True
assert result.confidence == 0.92
assert "[Tier 3 LLM]" in result.rationale
def test_classifier_skips_tier3_on_clear_direct_inherent_case():
"""
Garante que casos claros (alta confiança e alta densidade de âncoras) NÃO chamem
o LLM, economizando chamadas desnecessárias conforme FR-004.
"""
call_tracker = {"called": False}
def tracking_provider(prompt: str) -> str:
call_tracker["called"] = True
return json.dumps({"decision": "DIRECT_INHERENT", "confidence": 0.99})
mock_adapter = LLMFallbackAdapter(provider_fn=tracking_provider)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=mock_adapter)
ecp = ECPSnapshot(
target_entity_id="ecp_empresa",
target_name="EmpresaAlfa",
aliases=["EmpresaAlfa"],
domain="Tecnologia",
anchors=["software", "computação em nuvem", "inteligência artificial"],
)
# Caso claro com alta densidade de âncoras
clear_content = "A EmpresaAlfa desenvolveu uma nova plataforma de software baseada em computação em nuvem e inteligência artificial."
result = classifier.classify(ecp, clear_content)
assert result.decision == DecisionCategory.DIRECT_INHERENT
assert result.confidence >= 0.85
# O LLM NÃO deve ter sido chamado
assert call_tracker["called"] is False
def test_classifier_graceful_degradation_when_llm_raises_exception():
"""
Garante que se o LLM falhar por erro de rede ou timeout, o classificador mantenha
o resultado do Tier 1 com degradação graciosa e registre o aviso em warnings.
"""
def failing_provider(prompt: str) -> str:
raise ConnectionError("Timeout ao conectar com a API do modelo de linguagem.")
mock_adapter = LLMFallbackAdapter(provider_fn=failing_provider)
classifier = InherenceClassifier(enable_llm=True, llm_adapter=mock_adapter)
ecp = ECPSnapshot(
target_entity_id="ecp_empresa",
target_name="EmpresaAlfa",
aliases=["EmpresaAlfa"],
domain="Tecnologia",
anchors=["software"],
)
ambiguous_content = "A EmpresaAlfa participou da conferência."
result = classifier.classify(ecp, ambiguous_content)
# Retém a decisão original do Tier 1
assert result.decision == DecisionCategory.TANGENTIAL
assert result.is_inherent is False
# Contém aviso sobre a falha do LLM sem quebrar a execução
assert any("LLM fallback failed" in w for w in result.warnings)
# ==============================================================================
# 3. Teste de Integração CLI com a Flag --enable-llm
# ==============================================================================
def test_cli_execution_with_enable_llm_flag(tmp_path):
"""Garante que a CLI classify.py aceite e processe a flag --enable-llm sem erros."""
ecp_file = tmp_path / "test_ecp.json"
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ecp_test",
"target_name": "TestCorp",
"aliases": ["TestCorp"],
"domain": "Tech",
"anchors": ["software", "cloud"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "test_doc.md"
content_file.write_text("# TestCorp\n\nTestCorp builds cloud software.", encoding="utf-8")
output_file = tmp_path / "out.json"
res = subprocess.run(
[
sys.executable,
str(SCRIPT_PATH),
"--ecp",
str(ecp_file),
"--content",
str(content_file),
"--output",
str(output_file),
"--enable-llm",
],
capture_output=True,
text=True,
)
assert res.returncode == 0, f"Erro na CLI: {res.stderr}"
assert output_file.exists()
data = json.loads(output_file.read_text(encoding="utf-8"))
assert data["decision"] == "DIRECT_INHERENT"
assert data["is_inherent"] is True
+114
View File
@@ -0,0 +1,114 @@
"""Unit tests for ECP models, schema validation, and structured error handling."""
import pytest
from src.tools.models import (
ClassificationError,
ClassificationResult,
DecisionCategory,
ECPSnapshot,
ErrorCode,
)
from src.tools.parser import extract_evidence_snippets, strip_markdown
def test_ecp_snapshot_valid():
data = {
"target_entity_id": "ent_123",
"target_name": "Petrobras",
"aliases": ["Petróleo Brasileiro S.A.", "Petrobras"],
"domain": "Oil & Gas",
"anchors": ["pré-sal", "refinaria", "combustíveis"],
"negative_anchors": ["petrobras posto pirata"],
"graph_version": "1.0.0",
"related_entities": [
{
"entity_id": "ent_456",
"name": "Transpetro",
"relation_type": "SUBSIDIARY_OF",
"weight": 0.9,
"aliases": ["Transpetro Logística"],
"scope": "logistics",
"confidence": 0.95,
}
],
}
snapshot = ECPSnapshot.from_dict(data)
assert snapshot.target_entity_id == "ent_123"
assert snapshot.target_name == "Petrobras"
assert len(snapshot.aliases) == 2
assert len(snapshot.related_entities) == 1
assert snapshot.related_entities[0].name == "Transpetro"
assert snapshot.related_entities[0].weight == 0.9
def test_ecp_snapshot_defaults():
data = {
"target_entity_id": "ent_123",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["energia"],
}
snapshot = ECPSnapshot.from_dict(data)
assert snapshot.negative_anchors == []
assert snapshot.graph_version == "1.0.0"
assert snapshot.related_entities == []
def test_ecp_snapshot_missing_required():
data = {
"target_entity_id": "ent_123",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["energia"],
}
with pytest.raises(ValueError, match="Missing required field"):
ECPSnapshot.from_dict(data)
def test_classification_result_serialization():
res = ClassificationResult(
decision=DecisionCategory.DIRECT_INHERENT,
is_inherent=True,
confidence=0.95,
detected_language="pt",
matched_anchors=["Petrobras"],
evidence=["Petrobras anunciou investimentos no pré-sal."],
rationale="Match forte da entidade alvo.",
)
d = res.to_dict()
assert d["decision"] == "DIRECT_INHERENT"
assert d["is_inherent"] is True
assert d["confidence"] == 0.95
assert d["detected_language"] == "pt"
assert "Petrobras" in d["matched_anchors"]
def test_classification_error_serialization():
err = ClassificationError(
error_code=ErrorCode.INVALID_ECP_JSON,
message="Malformed JSON syntax",
details={"path": "snapshot.json"},
)
d = err.to_dict()
assert d["error_code"] == "invalid_ecp_json"
assert d["message"] == "Malformed JSON syntax"
assert d["details"]["path"] == "snapshot.json"
def test_parser_strip_markdown():
md = "# Title\n\nThis is **bold** text and [link](https://example.com).\n- item 1\n- item 2"
plain = strip_markdown(md)
assert "Title" in plain
assert "bold text" in plain
assert "link" in plain
assert "[" not in plain
assert "*" not in plain
def test_extract_evidence_snippets():
md = "O pré-sal brasileiro é uma das maiores reservas de petróleo. A Petrobras lidera a exploração técnica."
snippets = extract_evidence_snippets(md, ["Petrobras"])
assert len(snippets) > 0
assert "Petrobras lidera" in snippets[0]
@@ -0,0 +1,615 @@
"""
Suíte de Testes Automatizados para o Seletor Determinístico de Extrator.
Cobre 100% dos Casos de Teste Obrigatórios do PRD (CT-001 a CT-014), testes unitários
de normalização e shingles, testes de integração de lote e testes E2E via subprocess.
"""
from __future__ import annotations
import json
import subprocess
import sys
from pathlib import Path
import pytest
from scripts.select_article_extractor import (
ExtractorName,
generate_shingles,
normalize_text,
process_batch,
select_article_extractor,
)
# ==============================================================================
# Testes Unitários de Normalização e Tokenização
# ==============================================================================
def test_normalize_text_empty_and_invalid():
assert normalize_text(None) == []
assert normalize_text("") == []
assert normalize_text(" \n\t ") == []
assert normalize_text(12345) == []
def test_normalize_text_html_entities_and_tags():
raw = "<p>El &amp; <b>futebol</b> mundial &quot;está&quot; mudando.</p>"
tokens = normalize_text(raw)
assert tokens == ["el", "futebol", "mundial", "está", "mudando"]
def test_normalize_text_markdown_links():
raw = "Veja mais no [Portal de Notícias](https://example.com/noticias) hoje."
tokens = normalize_text(raw)
assert tokens == ["veja", "mais", "no", "portal", "de", "notícias", "hoje"]
def test_normalize_text_markdown_images_stripped_while_links_preserved():
"""Garante que marcação de imagem Markdown ![alt](url) seja descartada e link [texto](url) seja preservado."""
raw = (
"Texto inicial do artigo. "
"![Legenda da foto e imagem](https://example.com/imagem.webp) "
"Mais texto com [link importante](https://example.com/pagina) e outra "
"![Outra foto](https://example.com/foto2.jpg) informação."
)
tokens = normalize_text(raw)
assert "legenda" not in tokens
assert "foto" not in tokens
assert "imagem" not in tokens
assert "link" in tokens
assert "importante" in tokens
assert tokens == [
"texto",
"inicial",
"do",
"artigo",
"mais",
"texto",
"com",
"link",
"importante",
"e",
"outra",
"informação",
]
def test_normalize_text_nfkc_unicode_and_punctuation():
# Caracteres combinados e pontuação
raw = "River Plate venceu por 3-0! (Com gol de pênalti & golaço de falta)."
tokens = normalize_text(raw)
assert tokens == [
"river",
"plate",
"venceu",
"por",
"3",
"0",
"com",
"gol",
"de",
"pênalti",
"golaço",
"de",
"falta",
]
# ==============================================================================
# Testes Unitários de Shingles
# ==============================================================================
def test_generate_shingles_sliding_window():
tokens = ["um", "dois", "três", "quatro", "cinco", "seis"]
shingles = generate_shingles(tokens, window_size=5)
assert len(shingles) == 2
assert ("um", "dois", "três", "quatro", "cinco") in shingles
assert ("dois", "três", "quatro", "cinco", "seis") in shingles
def test_generate_shingles_short_text():
# Entre 1 e 4 tokens deve gerar 1 único shingle com a tupla completa
tokens = ["river", "plate", "campeão"]
shingles = generate_shingles(tokens, window_size=5)
assert len(shingles) == 1
assert ("river", "plate", "campeão") in shingles
def test_generate_shingles_empty():
assert generate_shingles([]) == set()
# ==============================================================================
# Casos de Teste Obrigatórios do PRD (§12: CT-001 a CT-014)
# ==============================================================================
def test_ct_001_three_candidates_clear_winner():
"""CT-001: Três candidatos com consenso e um vencedor claro -> Selecionar o maior score."""
base = "river plate venceu o clássico ontem a noite no estádio monumental"
article = {
"trafilatura": {"text": base, "error": None},
"newspaper4k": {"text": base, "error": None},
"readability": {
"cleaned_text": "texto completamente diferente sem nenhuma relação",
"error": None,
},
}
result = select_article_extractor(article)
assert result.selected_extractor in (ExtractorName.NEWSPAPER4K, ExtractorName.TRAFILATURA)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
def test_ct_002_technical_tie_smallest_shingles():
"""CT-002: Dois ou mais candidatos dentro de 0,03 do maior score -> Selecionar o de menor quantidade de shingles."""
tokens_comuns = (
"o rio de janeiro continua lindo e sempre maravilhoso em todas as estações do ano"
)
article = {
"trafilatura": {"text": tokens_comuns, "error": None},
"readability": {
"cleaned_text": tokens_comuns + " propaganda extra adicionada no fim",
"error": None,
},
"newspaper4k": {"text": tokens_comuns, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
def test_ct_003_technical_tie_priority_fallback():
"""CT-003: Empate técnico e mesma quantidade de shingles -> Aplicar prioridade final (newspaper4k > readability > trafilatura)."""
texto = "o time jogou muito bem durante toda a partida de futebol"
article = {
"trafilatura": {"text": texto, "error": None},
"readability": {"cleaned_text": texto, "error": None},
"newspaper4k": {"text": texto, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
# Agora sem newspaper4k ativo (somente readability e trafilatura idênticos)
article_two = {
"trafilatura": {"text": texto, "error": None},
"readability": {"cleaned_text": texto, "error": None},
"newspaper4k": {"text": None, "error": "Crash"},
}
result_two = select_article_extractor(article_two)
assert result_two.selected_extractor == ExtractorName.READABILITY
def test_ct_004_three_candidates_no_consensus():
"""CT-004: Três candidatos sem consenso -> Selecionar a quantidade mediana de shingles."""
t1 = "alfa bravo charlie delta echo foxtrot golf hotel india juliet" # 10 tokens -> 6 shingles
t2 = "kilo lima mike november oscar papa quebec romeo sierra tango uniform victor" # 12 tokens -> 8 shingles (MEDIANA)
t3 = "whiskey xray yankee zulu zero one two three four five six seven eight nine" # 14 tokens -> 10 shingles
article = {
"trafilatura": {"text": t1, "error": None},
"readability": {"cleaned_text": t2, "error": None},
"newspaper4k": {"text": t3, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.READABILITY
assert result.selection_reason == "no_consensus_median_shingles"
def test_ct_005_two_candidates_no_consensus():
"""CT-005: Dois candidatos sem consenso -> Selecionar a maior quantidade de shingles."""
t_short = "alfa bravo charlie delta echo foxtrot" # 6 tokens -> 2 shingles
t_long = (
"kilo lima mike november oscar papa quebec romeo sierra tango" # 10 tokens -> 6 shingles
)
article = {
"trafilatura": {"text": t_short, "error": None},
"newspaper4k": {"text": t_long, "error": None},
"readability": {"cleaned_text": None, "error": "Not extracted"},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
assert result.selection_reason == "no_consensus_max_shingles"
def test_ct_006_single_usable_candidate():
"""CT-006: Somente um candidato utilizável -> Selecionar esse candidato."""
article = {
"trafilatura": {
"text": "conteúdo válido e utilizável extraído com sucesso aqui",
"error": None,
},
"newspaper4k": {"text": "", "error": None},
"readability": {"cleaned_text": None, "error": "Timeout error"},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.TRAFILATURA
assert result.selection_reason == "single_usable_candidate"
def test_ct_007_degraded_candidates_only():
"""CT-007: Nenhum utilizável, mas existe candidato degradado -> Executar o algoritmo somente com os degradados."""
texto_comum = "artigo relevante sobre economia global e finanças internacionais com detalhes"
article = {
"trafilatura": {"text": texto_comum, "error": "Warning: partial parse"},
"newspaper4k": {"text": texto_comum, "error": "HTTP 403 partial"},
"readability": {"cleaned_text": None, "error": "Fatal exception"},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
def test_ct_008_all_candidates_unavailable():
"""CT-008: Todos os candidatos indisponíveis -> Selecionar newspaper4k."""
article = {
"trafilatura": {"text": None, "error": "Error"},
"newspaper4k": {"text": "", "error": "Empty"},
"readability": {"cleaned_text": " ", "error": "Blank"},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
assert result.selection_reason == "fallback_all_unavailable"
def test_ct_009_small_fragment_loses_due_to_low_coverage():
"""CT-009: Readability retorna apenas um fragmento pequeno enquanto os outros concordam -> Perde por baixa cobertura."""
full_text = (
"o presidente da república anunciou novas medidas econômicas para conter a inflação "
"e estimular o crescimento industrial em todo o território nacional durante o pronunciamento oficial"
)
small_fragment = "o presidente da república anunciou"
article = {
"trafilatura": {"text": full_text, "error": None},
"newspaper4k": {"text": full_text, "error": None},
"readability": {"cleaned_text": small_fragment, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor in (ExtractorName.TRAFILATURA, ExtractorName.NEWSPAPER4K)
assert result.selected_extractor != ExtractorName.READABILITY
def test_ct_010_excessive_boilerplate_loses_due_to_low_support():
"""CT-010: Um candidato contém o conteúdo comum e muito conteúdo excedente -> Perde suporte e reduz pontuação."""
common_content = (
"notícia oficial com dados apurados sobre a operação policial realizada nesta manhã"
)
massive_boilerplate = common_content + (
" compartilhe no facebook twitter whatsapp veja também esportes receitas horóscopo política "
" e assine nossa newsletter diária para receber mais novidades sobre culinária e fofocas"
)
article = {
"trafilatura": {"text": common_content, "error": None},
"readability": {"cleaned_text": common_content, "error": None},
"newspaper4k": {"text": massive_boilerplate, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.READABILITY
def test_ct_011_partial_content_loses_due_to_low_coverage():
"""CT-011: Um candidato contém somente parte do conteúdo comum -> Perde cobertura e reduz pontuação."""
full_content = "primeiro parágrafo do artigo completo segundo parágrafo com explicações terceiro parágrafo final"
half_content = "primeiro parágrafo do artigo completo"
article = {
"newspaper4k": {"text": full_content, "error": None},
"readability": {"cleaned_text": full_content, "error": None},
"trafilatura": {"text": half_content, "error": None},
}
result = select_article_extractor(article)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
def test_ct_012_recalculate_existing_selected_extractor(tmp_path: Path):
"""CT-012: A entrada já contém selected_extractor -> Recalcular e substituir somente essa chave."""
input_data = {
"articles": [
{
"titulo": "Teste",
"selected_extractor": "trafilatura",
"trafilatura": {"text": "lixo sem sentido", "error": None},
"newspaper4k": {
"text": "conteúdo correto compartilhado por dois motores",
"error": None,
},
"readability": {
"cleaned_text": "conteúdo correto compartilhado por dois motores",
"error": None,
},
}
]
}
in_file = tmp_path / "artigos.json"
in_file.write_text(json.dumps(input_data, ensure_ascii=False), encoding="utf-8")
res = process_batch(in_file)
out_file = Path(res.output_file)
assert out_file.exists()
with open(out_file, "r", encoding="utf-8") as f:
out_data = json.load(f)
assert out_data["articles"][0]["selected_extractor"] == "newspaper4k"
def test_ct_013_empty_articles_list(tmp_path: Path):
"""CT-013: articles está vazio -> Gerar saída válida com articles vazio."""
input_data = {"metadata": "info", "articles": []}
in_file = tmp_path / "empty.json"
in_file.write_text(json.dumps(input_data), encoding="utf-8")
res = process_batch(in_file)
out_file = Path(res.output_file)
assert out_file.exists()
with open(out_file, "r", encoding="utf-8") as f:
out_data = json.load(f)
assert out_data["metadata"] == "info"
assert out_data["articles"] == []
assert res.total_articles == 0
assert res.processed_count == 0
def test_ct_014_invalid_json_fails_atomically(tmp_path: Path):
"""CT-014: JSON inválido -> Não gerar saída."""
in_file = tmp_path / "invalid.json"
in_file.write_text("{articles: [ malformed json", encoding="utf-8")
expected_out = tmp_path / "invalid_selected.json"
with pytest.raises(ValueError, match="JSON inválido"):
process_batch(in_file)
assert not expected_out.exists()
# ==============================================================================
# Testes de Integração
# ==============================================================================
def test_integration_reference_file(tmp_path: Path):
"""Testa o processamento em lote completo sobre o arquivo real out/river_plate_extracted.json."""
ref_file = Path("out/river_plate_extracted.json")
if not ref_file.exists():
pytest.skip(
"Arquivo out/river_plate_extracted.json não encontrado para teste de integração."
)
out_file = tmp_path / "river_plate_extracted_selected.json"
result = process_batch(ref_file, output_path=out_file, verbose=True)
assert result.total_articles == 20
assert result.processed_count == 20
assert out_file.exists()
with open(out_file, "r", encoding="utf-8") as f:
data = json.load(f)
assert len(data["articles"]) == 20
for art in data["articles"]:
assert "selected_extractor" in art
assert art["selected_extractor"] in ["trafilatura", "newspaper4k", "readability"]
# Validar a distribuição exata conforme o algoritmo do PRD
assert result.selection_distribution == {
"newspaper4k": 9,
"readability": 9,
"trafilatura": 2,
}
def test_article_1_regression_technical_tie_markdown_images():
"""
Teste de regressão para o Artigo 1 (Los puntajes de River vs. Independiente Santa Fe):
Valida que com o descarte de imagens Markdown ![alt](url), Readability e Newspaper4k
entram em empate técnico (diff <= 0.03) e Newspaper4k vence por possuir menor quantidade
de shingles (1009 vs 1057).
"""
ref_file = Path("out/river_plate_extracted.json")
if not ref_file.exists():
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
with open(ref_file, "r", encoding="utf-8") as f:
data = json.load(f)
art1 = data["articles"][0]
result = select_article_extractor(art1, article_index=0)
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
assert result.selection_reason == "technical_tie_smallest_shingles"
cand_news = result.candidates[ExtractorName.NEWSPAPER4K]
cand_read = result.candidates[ExtractorName.READABILITY]
cand_traf = result.candidates[ExtractorName.TRAFILATURA]
assert cand_news.shingle_count == 1009
assert cand_read.shingle_count == 1057
assert cand_traf.shingle_count == 1091
# Diferença para o maior score <= 0.03 (empate técnico)
max_score = max(cand_news.score, cand_read.score, cand_traf.score)
assert (max_score - cand_news.score) <= 0.03
assert (max_score - cand_read.score) <= 0.03
def test_integration_large_batch_determinism(tmp_path: Path):
"""Gera um lote de 100 artigos sintéticos e verifica 100% de repetibilidade determinística entre 2 execuções."""
articles = []
for i in range(100):
if i % 4 == 0:
art = {
"trafilatura": {
"text": f"artigo numero {i} sobre futebol internacional no estadio",
"error": None,
},
"newspaper4k": {
"text": f"artigo numero {i} sobre futebol internacional no estadio",
"error": None,
},
"readability": {"cleaned_text": "sem relacao", "error": None},
}
elif i % 4 == 1:
art = {
"trafilatura": {"text": None, "error": "timeout"},
"newspaper4k": {"text": f"noticia exclusiva {i} com detalhes", "error": None},
"readability": {
"cleaned_text": f"noticia exclusiva {i} com detalhes",
"error": None,
},
}
elif i % 4 == 2:
art = {
"trafilatura": {"text": f"texto a {i}", "error": None},
"newspaper4k": {"text": f"texto b diferente {i}", "error": None},
"readability": {"cleaned_text": f"texto c terceiro {i}", "error": None},
}
else:
art = {
"trafilatura": {"text": None, "error": "err"},
"newspaper4k": {"text": "", "error": "err"},
"readability": {"cleaned_text": None, "error": "err"},
}
articles.append(art)
batch_payload = {"articles": articles}
in_file = tmp_path / "large_batch.json"
in_file.write_text(json.dumps(batch_payload, ensure_ascii=False), encoding="utf-8")
out1 = tmp_path / "large_batch_run1.json"
out2 = tmp_path / "large_batch_run2.json"
res1 = process_batch(in_file, output_path=out1)
res2 = process_batch(in_file, output_path=out2)
assert res1.processed_count == 100
assert res2.processed_count == 100
# Os resultados devem ser 100% idênticos
selections1 = [s.selected_extractor for s in res1.selections]
selections2 = [s.selected_extractor for s in res2.selections]
assert selections1 == selections2
def test_integration_pipeline_downstream_consumer(tmp_path: Path):
"""
Testa a integração end-to-end do pipeline downstream:
Lê o JSON enriquecido com selected_extractor, recupera o conteúdo do extrator vencedor e
garante que o texto está higienizado e pronto para os classificadores NLP.
"""
ref_file = Path("out/river_plate_extracted.json")
if not ref_file.exists():
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
out_file = tmp_path / "downstream_test.json"
process_batch(ref_file, output_path=out_file)
with open(out_file, "r", encoding="utf-8") as f:
data = json.load(f)
for idx, art in enumerate(data["articles"]):
winner = art["selected_extractor"]
assert winner in ["trafilatura", "newspaper4k", "readability"]
# Recuperar texto do extrator vencedor conforme mapeamento do PRD
if winner == "trafilatura":
chosen_text = art["trafilatura"]["text"]
elif winner == "newspaper4k":
chosen_text = art["newspaper4k"]["text"]
elif winner == "readability":
chosen_text = art["readability"]["cleaned_text"]
assert isinstance(chosen_text, str)
assert len(chosen_text.strip()) > 0
# ==============================================================================
# Testes End-to-End (E2E) via Subprocess CLI
# ==============================================================================
def test_e2e_cli_subprocess_real_execution(tmp_path: Path):
"""E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha de comando."""
ref_file = Path("out/river_plate_extracted.json")
if not ref_file.exists():
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado para teste E2E.")
out_file = tmp_path / "e2e_river_plate_selected.json"
script_path = Path("scripts/select_article_extractor.py").resolve()
cmd = [
sys.executable,
str(script_path),
str(ref_file.resolve()),
"-o",
str(out_file),
"--verbose",
"--indent",
"2",
]
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
assert proc.returncode == 0
assert out_file.exists()
# Validar saída estruturada JSON do stdout
summary = json.loads(proc.stdout)
assert summary["status"] == "success"
assert summary["total_articles"] == 20
assert summary["processed_count"] == 20
assert "distribution" in summary
# Validar que logs de verbose foram emitidos no stderr
assert "[Artigo #001]" in proc.stderr
assert "[Artigo #020]" in proc.stderr
def test_e2e_cli_subprocess_default_naming(tmp_path: Path):
"""E2E: Executa CLI sem a flag -o e valida criação automática de <nome>_selected.json."""
sample_data = {
"articles": [
{
"titulo": "Artigo Automático",
"trafilatura": {"text": "conteúdo padrão", "error": None},
"newspaper4k": {"text": "conteúdo padrão", "error": None},
"readability": {"cleaned_text": "conteúdo padrão", "error": None},
}
]
}
in_file = tmp_path / "my_news.json"
in_file.write_text(json.dumps(sample_data), encoding="utf-8")
expected_out = tmp_path / "my_news_selected.json"
script_path = Path("scripts/select_article_extractor.py").resolve()
cmd = [sys.executable, str(script_path), str(in_file)]
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
assert proc.returncode == 0
assert expected_out.exists()
with open(expected_out, "r", encoding="utf-8") as f:
data = json.load(f)
assert data["articles"][0]["selected_extractor"] == "newspaper4k"
def test_e2e_cli_subprocess_invalid_input(tmp_path: Path):
"""E2E: Executa CLI com JSON inválido e valida código de saída e erro no stderr."""
invalid_file = tmp_path / "broken.json"
invalid_file.write_text("not a valid json {", encoding="utf-8")
script_path = Path("scripts/select_article_extractor.py").resolve()
cmd = [sys.executable, str(script_path), str(invalid_file)]
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
assert proc.returncode == 2
assert "ERRO DE VALIDAÇÃO" in proc.stderr
def test_e2e_cli_subprocess_missing_file():
"""E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr."""
script_path = Path("scripts/select_article_extractor.py").resolve()
cmd = [sys.executable, str(script_path), "non_existent_file_12345.json"]
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
assert proc.returncode == 1
assert "ERRO DE ARQUIVO" in proc.stderr