feat(runtime): implement single-article consolidation runtime and modularize codebase
This commit is contained in:
@@ -0,0 +1,34 @@
|
||||
"""Unit tests for optional adapter interfaces (Tier 2 / Tier 3)."""
|
||||
|
||||
from src.tools.adapters.embeddings import LocalEmbeddingsAdapter
|
||||
from src.tools.adapters.llm import LLMFallbackAdapter
|
||||
from src.tools.classifier import InherenceClassifier
|
||||
from src.tools.models import ECPSnapshot
|
||||
|
||||
|
||||
def test_embeddings_adapter_interface():
|
||||
adapter = LocalEmbeddingsAdapter()
|
||||
assert isinstance(adapter.is_available(), bool)
|
||||
assert adapter.evaluate_similarity("test text", ["term1", "term2"]) == 0.0
|
||||
|
||||
|
||||
def test_llm_adapter_interface():
|
||||
adapter = LLMFallbackAdapter()
|
||||
assert isinstance(adapter.is_available(), bool)
|
||||
|
||||
|
||||
def test_classifier_with_adapter_flags():
|
||||
classifier = InherenceClassifier(enable_embeddings=True, enable_llm=True)
|
||||
assert classifier._embeddings_adapter is not None
|
||||
assert classifier._llm_adapter is not None
|
||||
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ent_test",
|
||||
target_name="TestCorp",
|
||||
aliases=["TestCorp"],
|
||||
domain="Tech",
|
||||
anchors=["software"],
|
||||
)
|
||||
res = classifier.classify(ecp, "TestCorp builds enterprise cloud software.")
|
||||
assert res.decision.value == "DIRECT_INHERENT"
|
||||
assert res.is_inherent is True
|
||||
@@ -0,0 +1,242 @@
|
||||
"""Adversarial and robustness test suite for Multilingual NLP Entity Inherence Classifier.
|
||||
|
||||
Validates homonym disambiguation, isolated related entities, edge cases,
|
||||
and CLI execution behavior via subprocess (exit codes, stream purity, JSON parsing).
|
||||
"""
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
from src.tools.models import DecisionCategory, ECPSnapshot, RelatedEntity
|
||||
|
||||
|
||||
def test_adversarial_sao_paulo_city_vs_fc():
|
||||
"""Content about city/state governance of São Paulo against ECP for São Paulo FC."""
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ent_spfc",
|
||||
target_name="São Paulo Futebol Clube",
|
||||
aliases=["São Paulo", "SPFC", "Tricolor Paulista"],
|
||||
domain="Futebol e Esportes",
|
||||
anchors=["Morumbi", "futebol", "campeonato", "Copa Libertadores", "elenco", "estádio"],
|
||||
negative_anchors=[
|
||||
"prefeitura de são paulo",
|
||||
"governo do estado de são paulo",
|
||||
"trânsito na capital paulista",
|
||||
],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[],
|
||||
)
|
||||
content = (
|
||||
"# Obras Viárias na Capital\n\n"
|
||||
"A prefeitura de São Paulo anunciou novas intervenções no trânsito na capital paulista "
|
||||
"para desafogar o fluxo de veículos na região central durante os horários de pico."
|
||||
)
|
||||
from src.tools.classifier import InherenceClassifier
|
||||
|
||||
classifier = InherenceClassifier()
|
||||
result = classifier.classify(ecp, content)
|
||||
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
|
||||
assert result.is_inherent is False
|
||||
assert result.decision != DecisionCategory.DIRECT_INHERENT
|
||||
|
||||
|
||||
def test_adversarial_apple_fruit_recipe():
|
||||
"""Content about apple fruit/culinary recipe against Apple Inc. tech entity."""
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ent_apple",
|
||||
target_name="Apple",
|
||||
aliases=["Apple Inc.", "Apple"],
|
||||
domain="Technology",
|
||||
anchors=["iPhone", "MacBook", "iOS", "silicon", "hardware"],
|
||||
negative_anchors=["apple pie", "orchard harvest", "doce de maçã"],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[],
|
||||
)
|
||||
content = (
|
||||
"# Receita Caseira\n\n"
|
||||
"Comprei maçãs frescas no mercado para preparar um doce de maçã com canela e açúcar mascavo."
|
||||
)
|
||||
from src.tools.classifier import InherenceClassifier
|
||||
|
||||
classifier = InherenceClassifier()
|
||||
result = classifier.classify(ecp, content)
|
||||
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
|
||||
assert result.is_inherent is False
|
||||
|
||||
|
||||
def test_adversarial_related_entity_without_scope_context():
|
||||
"""High-weight related entity mentioned in passing without required domain anchors."""
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ent_volkswagen",
|
||||
target_name="Volkswagen",
|
||||
aliases=["Volkswagen AG", "VW"],
|
||||
domain="Automotive & Electric Vehicles",
|
||||
anchors=["Elektrofahrzeuge", "Batteriezellen", "Fahrzeugproduktion"],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[
|
||||
RelatedEntity(
|
||||
entity_id="ent_northvolt",
|
||||
name="Northvolt",
|
||||
relation_type="SUPPLIER_OF",
|
||||
weight=0.95,
|
||||
scope="battery_technology",
|
||||
confidence=0.99,
|
||||
)
|
||||
],
|
||||
)
|
||||
# Content mentions Northvolt in an unrelated/passing architectural context without domain anchors
|
||||
content = (
|
||||
"# Architekturbericht aus Stockholm\n\n"
|
||||
"Während unseres Stadtrundgangs besuchten wir das neue Bürogebäude von Northvolt "
|
||||
"mit moderner Holzfassade und Blick auf den See."
|
||||
)
|
||||
from src.tools.classifier import InherenceClassifier
|
||||
|
||||
classifier = InherenceClassifier()
|
||||
result = classifier.classify(ecp, content)
|
||||
# Must be TANGENTIAL or NOT_RELATED, NEVER CONTEXTUAL_INHERENT
|
||||
assert result.decision in (DecisionCategory.TANGENTIAL, DecisionCategory.NOT_RELATED)
|
||||
assert result.is_inherent is False
|
||||
assert result.decision != DecisionCategory.CONTEXTUAL_INHERENT
|
||||
|
||||
|
||||
def test_adversarial_subprocess_cli_success_stdout(tmp_path):
|
||||
"""Run CLI via subprocess without --output and verify stdout is pure parseable JSON."""
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_petrobras",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text(
|
||||
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
import os
|
||||
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
)
|
||||
|
||||
assert res.returncode == 0
|
||||
# Stdout must be directly parseable as JSON without extraneous log text
|
||||
assert res.stdout is not None and len(res.stdout.strip()) > 0
|
||||
parsed = json.loads(res.stdout)
|
||||
assert parsed["decision"] == "DIRECT_INHERENT"
|
||||
assert parsed["is_inherent"] is True
|
||||
assert parsed["confidence"] >= 0.85
|
||||
assert len(parsed["evidence"]) > 0
|
||||
|
||||
|
||||
def test_adversarial_subprocess_cli_empty_content(tmp_path):
|
||||
"""Run CLI via subprocess with empty content and verify error code and exit code."""
|
||||
import os
|
||||
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Test",
|
||||
"aliases": ["Test"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["tech"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "empty.md"
|
||||
content_file.write_text(" \n\n ", encoding="utf-8")
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
)
|
||||
|
||||
assert res.returncode != 0
|
||||
# Stderr must contain pure parseable error JSON
|
||||
parsed_err = json.loads(res.stderr)
|
||||
assert parsed_err["error_code"] == "empty_content"
|
||||
|
||||
|
||||
def test_adversarial_subprocess_cli_missing_required_field(tmp_path):
|
||||
"""Run CLI via subprocess with missing target_name and verify error payload."""
|
||||
import os
|
||||
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
ecp_file = tmp_path / "ecp_bad.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"aliases": ["Test"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["tech"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Conteúdo de teste válido.", encoding="utf-8")
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
)
|
||||
|
||||
assert res.returncode != 0
|
||||
parsed_err = json.loads(res.stderr)
|
||||
assert parsed_err["error_code"] == "missing_required_field"
|
||||
|
||||
|
||||
def test_adversarial_subprocess_cli_corrupted_json(tmp_path):
|
||||
"""Run CLI via subprocess with corrupted JSON and verify error payload."""
|
||||
import os
|
||||
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
ecp_file = tmp_path / "ecp_corrupted.json"
|
||||
ecp_file.write_text("{ target_entity_id: not_valid_json }", encoding="utf-8")
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Conteúdo de teste válido.", encoding="utf-8")
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
)
|
||||
|
||||
assert res.returncode != 0
|
||||
parsed_err = json.loads(res.stderr)
|
||||
assert parsed_err["error_code"] == "invalid_ecp_json"
|
||||
@@ -0,0 +1,69 @@
|
||||
"""Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence Classifier.
|
||||
|
||||
Matrix: 6 Languages (PT, EN, ES, DE, IT, FR) x 4 Decisions (DIRECT, CONTEXTUAL, TANGENTIAL, NOT_RELATED).
|
||||
Target Success Criterion: Precision >= 90% over the 24 cases.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from src.tools.classifier import InherenceClassifier
|
||||
from src.tools.models import ECPSnapshot
|
||||
|
||||
FIXTURES_DIR = Path(__file__).parent.parent / "fixtures" / "benchmark_24"
|
||||
LANGUAGES = ["pt", "en", "es", "de", "it", "fr"]
|
||||
DECISION_TYPES = ["direct", "contextual", "tangential", "not_related"]
|
||||
|
||||
BENCHMARK_CASES = [(lang, dec_type) for lang in LANGUAGES for dec_type in DECISION_TYPES]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def classifier():
|
||||
return InherenceClassifier()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("lang,dec_type", BENCHMARK_CASES)
|
||||
def test_benchmark_case(classifier, lang: str, dec_type: str):
|
||||
case_dir = FIXTURES_DIR / lang
|
||||
ecp_file = case_dir / "ecp.json"
|
||||
content_file = case_dir / f"{dec_type}.md"
|
||||
expected_file = case_dir / f"{dec_type}_expected.json"
|
||||
|
||||
assert ecp_file.is_file(), f"Missing ECP fixture: {ecp_file}"
|
||||
assert content_file.is_file(), f"Missing Content fixture: {content_file}"
|
||||
assert expected_file.is_file(), f"Missing Expected fixture: {expected_file}"
|
||||
|
||||
ecp = ECPSnapshot.from_json_str(ecp_file.read_text(encoding="utf-8"))
|
||||
content = content_file.read_text(encoding="utf-8")
|
||||
expected = json.loads(expected_file.read_text(encoding="utf-8"))
|
||||
|
||||
result = classifier.classify(ecp, content)
|
||||
|
||||
# 1. Decision category validation
|
||||
assert result.decision.value == expected["expected_decision"], (
|
||||
f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
|
||||
)
|
||||
|
||||
# 2. Derived is_inherent boolean validation
|
||||
assert result.is_inherent == expected["expected_is_inherent"], (
|
||||
f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
|
||||
)
|
||||
|
||||
# 3. Language detection validation
|
||||
assert result.detected_language == expected["expected_language"], (
|
||||
f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
|
||||
)
|
||||
|
||||
# 4. Confidence threshold validation
|
||||
min_conf = expected.get("min_confidence", 0.0)
|
||||
assert result.confidence >= min_conf, (
|
||||
f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
|
||||
)
|
||||
|
||||
# 5. Evidence presence for inherent content
|
||||
if result.is_inherent:
|
||||
assert len(result.evidence) > 0, (
|
||||
f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
|
||||
)
|
||||
@@ -0,0 +1,100 @@
|
||||
"""Unit tests for deterministic classification decision logic."""
|
||||
|
||||
import pytest
|
||||
|
||||
from src.tools.classifier import InherenceClassifier
|
||||
from src.tools.models import DecisionCategory, ECPSnapshot, RelatedEntity
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def petrobras_ecp():
|
||||
return ECPSnapshot(
|
||||
target_entity_id="ent_petrobras",
|
||||
target_name="Petrobras",
|
||||
aliases=["Petróleo Brasileiro S.A.", "Petrobras", "Petrobrás"],
|
||||
domain="Oil & Gas",
|
||||
anchors=["pré-sal", "refinaria", "combustíveis", "petróleo", "exploração"],
|
||||
negative_anchors=["posto de combustíveis pirata"],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[
|
||||
RelatedEntity(
|
||||
entity_id="ent_transpetro",
|
||||
name="Transpetro",
|
||||
relation_type="SUBSIDIARY_OF",
|
||||
weight=0.85,
|
||||
aliases=[],
|
||||
scope="logistics",
|
||||
confidence=1.0,
|
||||
)
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def test_direct_inherent(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Expansão da Produção Nacional\n\n"
|
||||
"A Petrobras anunciou um aumento expressivo na produção de petróleo na camada pré-sal. "
|
||||
"Os investimentos em novas plataformas devem acelerar a exploração offshore."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert result.is_inherent is True
|
||||
assert result.confidence >= 0.85
|
||||
assert result.detected_language == "pt"
|
||||
assert "Petrobras" in result.matched_anchors
|
||||
assert len(result.evidence) > 0
|
||||
|
||||
|
||||
def test_contextual_inherent(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Logística de Combustíveis no Brasil\n\n"
|
||||
"A Transpetro ampliou a sua frota de navios para o transporte de combustíveis e derivados "
|
||||
"pelo litoral brasileiro, reforçando a infraestrutura energética."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.CONTEXTUAL_INHERENT
|
||||
assert result.is_inherent is True
|
||||
assert result.confidence >= 0.70
|
||||
assert len(result.graph_matches) == 1
|
||||
assert result.graph_matches[0]["name"] == "Transpetro"
|
||||
|
||||
|
||||
def test_tangential_inherent(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Crônica de Viagem pelo Interior\n\n"
|
||||
"Passamos perto de um prédio da Petrobras enquanto procurávamos um café na praça central. "
|
||||
"A tarde estava quente e os pássaros cantavam nas árvores antigas."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.TANGENTIAL
|
||||
assert result.is_inherent is False
|
||||
assert result.confidence < 0.60
|
||||
assert len(result.warnings) > 0
|
||||
|
||||
|
||||
def test_not_related(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Como Fazer Bolo de Cenoura com Cobertura de Chocolate\n\n"
|
||||
"Bata as cenouras raladas no liquidificador com os ovos e o óleo. "
|
||||
"Acrescente a farinha de trigo e o açúcar aos poucos até obter uma massa homogênea."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.NOT_RELATED
|
||||
assert result.is_inherent is False
|
||||
assert result.confidence >= 0.85
|
||||
|
||||
|
||||
def test_negative_anchor_suppression(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Operação Policial Fecha Estabelecimento\n\n"
|
||||
"A polícia interditou um posto de combustíveis pirata na rodovia estadual por adulteração."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.NOT_RELATED
|
||||
assert result.is_inherent is False
|
||||
assert len(result.negative_matches) > 0
|
||||
@@ -0,0 +1,806 @@
|
||||
"""
|
||||
Suíte de Testes Exaustiva para o Classificador de Inerência (classify.py e src/).
|
||||
|
||||
Cobre 100% dos caminhos felizes, infelizes, limiares, de ambiguidade,
|
||||
fallback de LLM (OpenAI e Gemini), resiliência de API, erros de contrato CLI
|
||||
e suporte aos 6 idiomas conforme a metodologia da skill-suite-tests.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from classify import main
|
||||
from src.tools.adapters.llm import LLMFallbackAdapter
|
||||
from src.tools.classifier import InherenceClassifier
|
||||
from src.tools.models import (
|
||||
ClassificationResult,
|
||||
DecisionCategory,
|
||||
ECPSnapshot,
|
||||
RelatedEntity,
|
||||
)
|
||||
|
||||
CLASSIFY_CLI = Path(__file__).parent.parent.parent / "classify.py"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Fixtures Universais
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ecp_tech_corp() -> ECPSnapshot:
|
||||
return ECPSnapshot(
|
||||
target_entity_id="ent_tech_corp",
|
||||
target_name="TechCorp Global",
|
||||
aliases=["TechCorp", "TechCorp Global", "TCG"],
|
||||
domain="Tecnologia e Cloud",
|
||||
anchors=[
|
||||
"cloud",
|
||||
"computação em nuvem",
|
||||
"software",
|
||||
"inteligência artificial",
|
||||
"datacenter",
|
||||
],
|
||||
negative_anchors=["TechCorp Calçados", "TechCorp Imóveis", "homônimo"],
|
||||
related_entities=[
|
||||
RelatedEntity(
|
||||
entity_id="ent_cloud_subsidiary",
|
||||
name="CloudPlatform Solutions",
|
||||
relation_type="SUBSIDIARY_OF",
|
||||
weight=0.90,
|
||||
aliases=["CloudPlatform"],
|
||||
scope="cloud_services",
|
||||
),
|
||||
RelatedEntity(
|
||||
entity_id="ent_ceo_tech",
|
||||
name="Alan Turing Silva",
|
||||
relation_type="CEO_OF",
|
||||
weight=0.80,
|
||||
aliases=["Alan Turing"],
|
||||
scope="executive",
|
||||
),
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 1. Casos Felizes (Happy Paths) - NLP Determinístico (Tier 1)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_happy_path_direct_inherent_with_canonical_and_anchors(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.1: Nome canônico + múltiplas âncoras temáticas -> DIRECT_INHERENT com alta confiança."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# TechCorp Global anuncia novo datacenter de computação em nuvem\n\n"
|
||||
"A TechCorp Global investiu 500 milhões para expandir sua infraestrutura de software "
|
||||
"e inteligência artificial na América Latina."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence >= 0.90
|
||||
assert "TechCorp Global" in res.matched_anchors or "TechCorp" in res.matched_anchors
|
||||
assert len(res.evidence) >= 1
|
||||
|
||||
|
||||
def test_happy_path_direct_inherent_via_alias_and_acronym(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.2: Apenas o alias / sigla 'TCG' é mencionado, com âncoras do domínio."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Inovação em Cloud\n\n"
|
||||
"A TCG lançou hoje uma nova plataforma de software baseada em computação em nuvem."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence >= 0.85
|
||||
|
||||
|
||||
def test_happy_path_direct_inherent_by_repetition_without_heavy_anchors(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.3: O nome 'TechCorp' aparece 3 vezes no texto, satisfazendo a regra de menção múltipla."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Relatório Corporativo Trimestral\n\n"
|
||||
"A TechCorp divulgou seus resultados. A TechCorp superou as estimativas de analistas. "
|
||||
"O conselho da TechCorp aprovou dividendos extraordinários."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence >= 0.85
|
||||
|
||||
|
||||
def test_happy_path_contextual_inherent_via_subsidiary_graph_entity(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.4: Menção da subsidiária 'CloudPlatform Solutions' com âncoras de cloud."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Expansão de Infraestrutura de Nuvem\n\n"
|
||||
"A CloudPlatform Solutions ativou novos servidores em seu datacenter de computação em nuvem."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence >= 0.75
|
||||
assert len(res.graph_matches) >= 1
|
||||
assert res.graph_matches[0]["name"] == "CloudPlatform Solutions"
|
||||
|
||||
|
||||
def test_happy_path_contextual_inherent_via_executive_graph_entity(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 1.5: Menção ao CEO no grafo + âncoras de tecnologia."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Discurso na Conferência de Tecnologia\n\n"
|
||||
"O executivo Alan Turing Silva discursou sobre o futuro da inteligência artificial e software."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert any(g["name"] == "Alan Turing Silva" for g in res.graph_matches)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 2. Casos Infelizes e Rejeições (Sad Paths) - NLP Determinístico (Tier 1)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_sad_path_not_related_completely_off_topic(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 2.1: Conteúdo totalmente desvinculado (culinária/jardinagem)."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Receita de Pão Caseiro Fácil\n\n"
|
||||
"Misture a farinha, o fermento biológico seco e a água morna. "
|
||||
"Deixe a massa descansar por 40 minutos em local aquecido."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence >= 0.90
|
||||
assert len(res.matched_anchors) == 0
|
||||
|
||||
|
||||
def test_sad_path_not_related_generic_domain_without_target_or_graph(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 2.2: Artigo cita muitas âncoras ('cloud', 'software'), mas NÃO cita a TechCorp nem o grafo."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# O Mercado Global de Computação em Nuvem\n\n"
|
||||
"O setor de computação em nuvem, datacenter e inteligência artificial cresceu 25% este ano."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
assert (
|
||||
"General domain topics mentioned, but target entity or related entities are absent."
|
||||
in res.rationale
|
||||
)
|
||||
|
||||
|
||||
def test_sad_path_not_related_negative_anchor_dominance(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 2.3: Homônimo 'TechCorp Calçados' dispara âncora negativa dominante."""
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Feira de Moda e Varejo\n\n"
|
||||
"A TechCorp Calçados apresentou sua nova linha de sandálias de couro para o verão."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
assert "TechCorp Calçados" in res.negative_matches
|
||||
|
||||
|
||||
def test_sad_path_not_related_negative_anchor_ties_with_positive_anchor(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 2.4: 1 âncora negativa e 1 positiva -> prioridade de segurança rejeita para NOT_RELATED."""
|
||||
classifier = InherenceClassifier()
|
||||
content = "A TechCorp Calçados adotou um novo software interno de gestão."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 3. Casos Limiares e Ambiguidades (Borderline / Tangential)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_borderline_tangential_single_passing_mention(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 3.1: Menção única isolada sem âncoras temáticas -> TANGENTIAL com baixa confiança."""
|
||||
classifier = InherenceClassifier()
|
||||
content = "Estávamos caminhando pela avenida e vimos a placa da TechCorp ao longe na esquina."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence == 0.40
|
||||
assert any("Low contextual density" in w for w in res.warnings)
|
||||
|
||||
|
||||
def test_borderline_tangential_graph_entity_in_isolation(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 3.2: Entidade do grafo mencionada sem contexto de domínio -> TANGENTIAL."""
|
||||
classifier = InherenceClassifier()
|
||||
content = "Alan Turing Silva participou de uma corrida beneficente no parque no domingo."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence == 0.45
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 4. Suíte Abrangente de Fallback para LLM (Tier 3)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_llm_happy_path_upgrade_tangential_to_direct_inherent(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.1: Caso ambíguo elevado para DIRECT_INHERENT pelo LLM."""
|
||||
mock_resp = json.dumps(
|
||||
{
|
||||
"analysis_summary": "Artigo detalha o projeto estratégico secreto da TechCorp.",
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": 0.95,
|
||||
"rationale": "Embora a redação use linguagem coloquial, o artigo foca inteiramente na estratégia da TechCorp.",
|
||||
}
|
||||
)
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "A diretoria da TechCorp finalizou as negociações confidenciais da rodada."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence == 0.95
|
||||
assert "[Tier 3 LLM]" in res.rationale
|
||||
assert "[Tier 3 LLM Override applied]" in res.warnings
|
||||
|
||||
|
||||
def test_llm_happy_path_upgrade_tangential_to_contextual_inherent(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.2: Caso ambíguo elevado para CONTEXTUAL_INHERENT pelo LLM."""
|
||||
mock_resp = json.dumps(
|
||||
{
|
||||
"analysis_summary": "Matéria sobre fusão de fornecedores onde a TechCorp é impactada diretamente.",
|
||||
"decision": "CONTEXTUAL_INHERENT",
|
||||
"confidence": 0.88,
|
||||
"rationale": "A TechCorp é parte material do ecossistema afetado pela fusão anunciada.",
|
||||
}
|
||||
)
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "O consórcio fornecedor foi reestruturado e envolverá contratos com a TechCorp."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.CONTEXTUAL_INHERENT
|
||||
assert res.is_inherent is True
|
||||
assert res.confidence == 0.88
|
||||
|
||||
|
||||
def test_llm_happy_path_confirmation_of_tangential(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.3: LLM confirma categoricamente que a menção é periférica / irrelevante."""
|
||||
mock_resp = json.dumps(
|
||||
{
|
||||
"analysis_summary": "Crônica sobre trânsito urbano com citação lateral a um outdoor da TechCorp.",
|
||||
"decision": "TANGENTIAL",
|
||||
"confidence": 0.97,
|
||||
"rationale": "A empresa é apenas uma referência visual casual sem relação com a narrativa de trânsito.",
|
||||
}
|
||||
)
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "O tráfego estava parado bem em frente ao painel da TechCorp na autoestrada."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence == 0.97
|
||||
|
||||
|
||||
def test_llm_happy_path_rejection_to_not_related(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.4: LLM identifica homônimo não mapeado nas regras determinísticas e rebaixa para NOT_RELATED."""
|
||||
mock_resp = json.dumps(
|
||||
{
|
||||
"analysis_summary": "Artigo sobre uma banda de rock indie com nome idêntico.",
|
||||
"decision": "NOT_RELATED",
|
||||
"confidence": 0.99,
|
||||
"rationale": "O texto refere-se a um grupo musical e não à empresa de tecnologia.",
|
||||
}
|
||||
)
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda p: mock_resp)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "A banda TechCorp tocou seus novos acordes no festival de música independente."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.NOT_RELATED
|
||||
assert res.is_inherent is False
|
||||
assert res.confidence == 0.99
|
||||
|
||||
|
||||
def test_llm_sad_path_llm_disabled_by_default_never_invokes_adapter(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.5: Quando enable_llm=False (padrão), o LLM NUNCA é chamado mesmo em caso limiar."""
|
||||
called = {"status": False}
|
||||
|
||||
def tracking_fn(p: str) -> str:
|
||||
called["status"] = True
|
||||
return "{}"
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=tracking_fn)
|
||||
classifier = InherenceClassifier(enable_llm=False, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp sem contexto algum."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert called["status"] is False
|
||||
|
||||
|
||||
def test_llm_sad_path_flag_enabled_without_api_key_or_provider(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.6: enable_llm=True mas sem chaves no ambiente -> degrada sem quebrar, retém Tier 1."""
|
||||
with patch.dict("os.environ", {}, clear=True):
|
||||
adapter = LLMFallbackAdapter(api_key="", provider_fn=None)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp em relatório breve."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert res.is_inherent is False
|
||||
|
||||
|
||||
def test_llm_sad_path_network_timeout_graceful_degradation(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.7: API do LLM sofre TimeoutError -> retém Tier 1 e registra aviso em warnings."""
|
||||
|
||||
def timeout_fn(p: str) -> str:
|
||||
raise TimeoutError("Conexão com gateway do LLM excedeu tempo limite de 30s.")
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=timeout_fn)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "A TechCorp esteve presente no evento de premiação."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert any("LLM fallback failed" in w for w in res.warnings)
|
||||
|
||||
|
||||
def test_llm_sad_path_http_500_server_error_graceful_degradation(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.8: API do LLM retorna erro 500 / ConnectionError -> retém Tier 1 com aviso."""
|
||||
|
||||
def error_500_fn(p: str) -> str:
|
||||
raise ConnectionError("HTTP 500: Internal Server Error do provedor de IA.")
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=error_500_fn)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção da TechCorp em comunicado à imprensa."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
assert any("LLM fallback failed" in w for w in res.warnings)
|
||||
|
||||
|
||||
def test_llm_sad_path_malformed_json_and_non_json_strings(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.9: LLM retorna texto livre ou JSON quebrado -> parser ignora com segurança."""
|
||||
|
||||
def make_bad_provider(resp_text: str):
|
||||
def _prov(prompt: str) -> str:
|
||||
return resp_text
|
||||
|
||||
return _prov
|
||||
|
||||
for bad_resp in [
|
||||
"Não tenho certeza sobre este documento.",
|
||||
"{json_quebrado_sem_aspas: true",
|
||||
"```json\n{invalido: 123}\n```",
|
||||
]:
|
||||
adapter = LLMFallbackAdapter(provider_fn=make_bad_provider(bad_resp))
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
|
||||
|
||||
def test_llm_sad_path_missing_decision_key_in_json(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.10: LLM retorna JSON válido mas sem o campo obrigatório 'decision'."""
|
||||
adapter = LLMFallbackAdapter(
|
||||
provider_fn=lambda p: json.dumps({"confidence": 0.90, "rationale": "Faltou a decisao"})
|
||||
)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
|
||||
|
||||
def test_llm_sad_path_unknown_hallucinated_decision_enum(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.11: LLM alucina uma categoria inexistente (ex: 'SUPER_INHERENT')."""
|
||||
adapter = LLMFallbackAdapter(
|
||||
provider_fn=lambda p: json.dumps({"decision": "SUPER_INHERENT", "confidence": 0.99})
|
||||
)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "Menção isolada da TechCorp."
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
assert res.decision == DecisionCategory.TANGENTIAL
|
||||
|
||||
|
||||
def test_llm_resilience_confidence_clipping(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.12: LLM retorna confidence fora do intervalo [0.0, 1.0] -> clippa com segurança."""
|
||||
|
||||
def make_clipping_provider(c_val: float):
|
||||
def _prov(prompt: str) -> str:
|
||||
return json.dumps(
|
||||
{
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": c_val,
|
||||
"rationale": "Teste de clipping.",
|
||||
}
|
||||
)
|
||||
|
||||
return _prov
|
||||
|
||||
for raw_conf, expected_conf in [(1.5, 1.0), (-0.5, 0.0), (0.85432, 0.8543)]:
|
||||
adapter = LLMFallbackAdapter(provider_fn=make_clipping_provider(raw_conf))
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
res = classifier.classify(ecp_tech_corp, "Menção da TechCorp.")
|
||||
assert res.confidence == expected_conf
|
||||
|
||||
|
||||
def test_llm_optimization_clear_case_bypasses_llm(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 4.13: Caso claro de alta densidade NÃO chama LLM mesmo com enable_llm=True."""
|
||||
called = {"status": False}
|
||||
|
||||
def tracking_fn(p: str) -> str:
|
||||
called["status"] = True
|
||||
return json.dumps({"decision": "DIRECT_INHERENT"})
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=tracking_fn)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = (
|
||||
"# TechCorp Global anuncia nova inteligência artificial para computação em nuvem\n\n"
|
||||
"A TechCorp Global ativou hoje novos clusters de datacenter com software avançado."
|
||||
)
|
||||
res = classifier.classify(ecp_tech_corp, content)
|
||||
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert called["status"] is False # LLM NÃO foi acionado
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 5. Provedores Reais de LLM (OpenAI Mock e Gemini REST Mock)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_llm_provider_openai_client_execution(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 5.1: Simula execução bem-sucedida via cliente OpenAI SDK."""
|
||||
mock_chat_completion = MagicMock()
|
||||
mock_choice = MagicMock()
|
||||
mock_choice.message.content = json.dumps(
|
||||
{
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": 0.96,
|
||||
"rationale": "OpenAI validou o contexto corporativo com precisão.",
|
||||
}
|
||||
)
|
||||
mock_chat_completion.choices = [mock_choice]
|
||||
|
||||
mock_openai_instance = MagicMock()
|
||||
mock_openai_instance.chat.completions.create.return_value = mock_chat_completion
|
||||
|
||||
with patch("openai.OpenAI", return_value=mock_openai_instance):
|
||||
adapter = LLMFallbackAdapter(api_key="sk-mock-openai-key")
|
||||
initial_res = ClassificationResult(
|
||||
decision=DecisionCategory.TANGENTIAL,
|
||||
is_inherent=False,
|
||||
confidence=0.40,
|
||||
detected_language="pt",
|
||||
matched_anchors=[],
|
||||
negative_matches=[],
|
||||
graph_matches=[],
|
||||
evidence=[],
|
||||
rationale="Passing.",
|
||||
warnings=[],
|
||||
)
|
||||
res = adapter.disambiguate(ecp_tech_corp, "Artigo sobre TechCorp.", initial_res)
|
||||
assert res is not None
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.confidence == 0.96
|
||||
|
||||
|
||||
def test_llm_provider_gemini_rest_execution(ecp_tech_corp: ECPSnapshot):
|
||||
"""Cenário 5.2: Simula execução bem-sucedida via API REST do Google Gemini."""
|
||||
gemini_payload = {
|
||||
"candidates": [
|
||||
{
|
||||
"content": {
|
||||
"parts": [
|
||||
{
|
||||
"text": json.dumps(
|
||||
{
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": 0.98,
|
||||
"rationale": "Gemini 2.5 Flash confirmou aderência direta ao tópico.",
|
||||
}
|
||||
)
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.read.return_value = json.dumps(gemini_payload).encode("utf-8")
|
||||
mock_response.__enter__.return_value = mock_response
|
||||
|
||||
with patch("urllib.request.urlopen", return_value=mock_response):
|
||||
with patch.dict("os.environ", {"GEMINI_API_KEY": "mock-gemini-key"}):
|
||||
adapter = LLMFallbackAdapter(api_key="")
|
||||
initial_res = ClassificationResult(
|
||||
decision=DecisionCategory.TANGENTIAL,
|
||||
is_inherent=False,
|
||||
confidence=0.40,
|
||||
detected_language="pt",
|
||||
matched_anchors=[],
|
||||
negative_matches=[],
|
||||
graph_matches=[],
|
||||
evidence=[],
|
||||
rationale="Passing.",
|
||||
warnings=[],
|
||||
)
|
||||
res = adapter.disambiguate(ecp_tech_corp, "Artigo sobre TechCorp.", initial_res)
|
||||
assert res is not None
|
||||
assert res.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert res.confidence == 0.98
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 6. Suíte de Contrato e Erros da CLI classify.py
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_cli_error_ecp_file_does_not_exist(tmp_path: Path, capsys):
|
||||
"""Cenário 6.1: Caminho de ECP inexistente -> Exit Code 1, error_code: invalid_ecp_json."""
|
||||
content_file = tmp_path / "valid.md"
|
||||
content_file.write_text("# Conteúdo válido", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(tmp_path / "nao_existe.json"), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "invalid_ecp_json"
|
||||
|
||||
|
||||
def test_cli_error_ecp_corrupted_json_syntax(tmp_path: Path, capsys):
|
||||
"""Cenário 6.2: Arquivo ECP com sintaxe JSON corrompida."""
|
||||
bad_ecp = tmp_path / "corrupt.json"
|
||||
bad_ecp.write_text("{ target_name: 'sem_aspas' ", encoding="utf-8")
|
||||
content_file = tmp_path / "valid.md"
|
||||
content_file.write_text("# Conteúdo válido", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(bad_ecp), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "invalid_ecp_json"
|
||||
|
||||
|
||||
def test_cli_error_ecp_missing_each_required_field(tmp_path: Path, capsys):
|
||||
"""Cenário 6.3: Valida erro para falta de cada um dos campos obrigatórios do ECP."""
|
||||
required_fields = ["target_entity_id", "target_name", "aliases", "domain", "anchors"]
|
||||
|
||||
base_ecp = {
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Nome",
|
||||
"aliases": ["Alias"],
|
||||
"domain": "Domínio",
|
||||
"anchors": ["Âncora"],
|
||||
}
|
||||
content_file = tmp_path / "valid.md"
|
||||
content_file.write_text("# Conteúdo válido", encoding="utf-8")
|
||||
|
||||
for field in required_fields:
|
||||
bad_data = base_ecp.copy()
|
||||
del bad_data[field]
|
||||
bad_file = tmp_path / f"missing_{field}.json"
|
||||
bad_file.write_text(json.dumps(bad_data), encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(bad_file), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "missing_required_field"
|
||||
assert field in err_json["message"]
|
||||
|
||||
|
||||
def test_cli_error_content_file_does_not_exist(tmp_path: Path, capsys):
|
||||
"""Cenário 6.4: Caminho de arquivo Markdown inexistente."""
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "TechCorp",
|
||||
"aliases": ["TechCorp"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["cloud"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(tmp_path / "doc_fantasma.md")])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "invalid_markdown"
|
||||
|
||||
|
||||
def test_cli_error_empty_and_whitespace_content(tmp_path: Path, capsys):
|
||||
"""Cenário 6.5: Arquivo Markdown vazio ou contendo apenas espaços em branco."""
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "TechCorp",
|
||||
"aliases": ["TechCorp"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["cloud"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
for empty_text in ["", " \n\n\t \n "]:
|
||||
empty_file = tmp_path / "empty.md"
|
||||
empty_file.write_text(empty_text, encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(empty_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err_json = json.loads(captured.err)
|
||||
assert err_json["error_code"] == "empty_content"
|
||||
|
||||
|
||||
def test_cli_output_file_creates_nested_directories(tmp_path: Path):
|
||||
"""Cenário 6.6: A flag -o / --output cria diretórios aninhados automaticamente."""
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "TechCorp",
|
||||
"aliases": ["TechCorp"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["cloud"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("# TechCorp\n\nTechCorp cloud computing.", encoding="utf-8")
|
||||
|
||||
nested_out = tmp_path / "deep" / "nested" / "folder" / "resultado.json"
|
||||
|
||||
exit_code = main(
|
||||
["--ecp", str(ecp_file), "--content", str(content_file), "-o", str(nested_out)]
|
||||
)
|
||||
assert exit_code == 0
|
||||
assert nested_out.exists()
|
||||
|
||||
payload = json.loads(nested_out.read_text(encoding="utf-8"))
|
||||
assert payload["decision"] == "DIRECT_INHERENT"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 7. Matriz Multilíngue Completa (6 Idiomas)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"lang,target,aliases,domain,anchors,content,expected_decision,expected_lang",
|
||||
[
|
||||
# Português
|
||||
(
|
||||
"pt",
|
||||
"Petrobras",
|
||||
["Petrobras"],
|
||||
"Energia",
|
||||
["pré-sal", "petróleo", "refinaria"],
|
||||
"# Petrobras bate recorde de produção no pré-sal com novas plataformas.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"pt",
|
||||
),
|
||||
# Inglês
|
||||
(
|
||||
"en",
|
||||
"Apple Inc.",
|
||||
["Apple", "Apple Inc."],
|
||||
"Technology",
|
||||
["iPhone", "MacBook", "iOS", "silicon"],
|
||||
"# Apple unveils new MacBook Pro with M4 silicon and advanced iOS features.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"en",
|
||||
),
|
||||
# Espanhol
|
||||
(
|
||||
"es",
|
||||
"River Plate",
|
||||
["River Plate", "River"],
|
||||
"Fútbol",
|
||||
["Monumental", "Libertadores", "Sudamericana"],
|
||||
"# River Plate se prepara para disputar el torneo continental en el Estadio Monumental.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"es",
|
||||
),
|
||||
# Alemão (Compostos e Diacríticos)
|
||||
(
|
||||
"de",
|
||||
"Volkswagen AG",
|
||||
["Volkswagen", "VW"],
|
||||
"Automobilindustrie",
|
||||
["Elektroauto", "Batteriefabrik", "Produktion"],
|
||||
"# Volkswagen investiert Milliarden in eine neue Batteriefabrik für Elektroautos in Deutschland.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"de",
|
||||
),
|
||||
# Italiano
|
||||
(
|
||||
"it",
|
||||
"Scuderia Ferrari",
|
||||
["Ferrari", "Scuderia Ferrari"],
|
||||
"Automobilismo",
|
||||
["Monza", "Gran Premio", "motore", "pole position"],
|
||||
"# La Ferrari conquista una straordinaria pole position nel Gran Premio di Monza.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"it",
|
||||
),
|
||||
# Francês (Elisão e Apóstrofos)
|
||||
(
|
||||
"fr",
|
||||
"TotalEnergies",
|
||||
["TotalEnergies", "Total"],
|
||||
"Énergie",
|
||||
["énergie solaire", "pétrole", "renouvelable", "électricité"],
|
||||
"# L'entreprise TotalEnergies accélère ses investissements dans l'énergie solaire et l'électricité en France.",
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
"fr",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_multilingual_matrix_6_languages(
|
||||
lang: str,
|
||||
target: str,
|
||||
aliases: list[str],
|
||||
domain: str,
|
||||
anchors: list[str],
|
||||
content: str,
|
||||
expected_decision: DecisionCategory,
|
||||
expected_lang: str,
|
||||
):
|
||||
"""Garante a precisão e robustez do classificador nos 6 idiomas suportados pela POC."""
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id=f"ent_{lang}",
|
||||
target_name=target,
|
||||
aliases=aliases,
|
||||
domain=domain,
|
||||
anchors=anchors,
|
||||
)
|
||||
classifier = InherenceClassifier()
|
||||
res = classifier.classify(ecp, content)
|
||||
|
||||
assert res.decision == expected_decision
|
||||
assert res.is_inherent is True
|
||||
assert res.detected_language == expected_lang
|
||||
assert res.confidence >= 0.85
|
||||
@@ -0,0 +1,126 @@
|
||||
"""CLI execution tests covering flags, arguments, stdout, and error handling."""
|
||||
|
||||
import json
|
||||
|
||||
from classify import main
|
||||
|
||||
|
||||
def test_cli_success_stdout(tmp_path, capsys):
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text(
|
||||
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
|
||||
assert exit_code == 0
|
||||
|
||||
captured = capsys.readouterr()
|
||||
result = json.loads(captured.out)
|
||||
assert result["decision"] == "DIRECT_INHERENT"
|
||||
assert result["is_inherent"] is True
|
||||
assert result["detected_language"] == "pt"
|
||||
|
||||
|
||||
def test_cli_output_file(tmp_path):
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text(
|
||||
"Petrobras anunciou investimentos bilionários em novas refinarias de petróleo.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
output_file = tmp_path / "out" / "result.json"
|
||||
|
||||
exit_code = main(
|
||||
["--ecp", str(ecp_file), "--content", str(content_file), "--output", str(output_file)]
|
||||
)
|
||||
assert exit_code == 0
|
||||
assert output_file.exists()
|
||||
|
||||
result = json.loads(output_file.read_text(encoding="utf-8"))
|
||||
assert result["decision"] == "DIRECT_INHERENT"
|
||||
assert result["is_inherent"] is True
|
||||
|
||||
|
||||
def test_cli_missing_ecp_file(tmp_path, capsys):
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Algum conteúdo válido aqui.", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(tmp_path / "non_existent.json"), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err = json.loads(captured.err)
|
||||
assert err["error_code"] == "invalid_ecp_json"
|
||||
|
||||
|
||||
def test_cli_empty_content_file(tmp_path, capsys):
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "empty.md"
|
||||
content_file.write_text(" \n\n ", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err = json.loads(captured.err)
|
||||
assert err["error_code"] == "empty_content"
|
||||
|
||||
|
||||
def test_cli_missing_required_ecp_field(tmp_path, capsys):
|
||||
ecp_file = tmp_path / "bad_ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps({"target_entity_id": "ent_1", "domain": "Oil & Gas", "anchors": ["petróleo"]}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Algum conteúdo válido para testar o erro.", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err = json.loads(captured.err)
|
||||
assert err["error_code"] == "missing_required_field"
|
||||
@@ -0,0 +1,759 @@
|
||||
"""
|
||||
Suíte de Testes Automatizados para Conversão de Artigo JSON para Markdown.
|
||||
|
||||
Cobre 100% dos Requisitos Funcionais (FR-001 a FR-018), Requisitos Não Funcionais (RNF-001 a RNF-006),
|
||||
Critérios de Aceite (CA-001 a CA-013), Casos de Teste do PRD (CT-001 a CT-012),
|
||||
Matriz de Erros (12 condições), Testes Unitários de Prioridade/Normalização,
|
||||
Testes de Integração de Arquivo e Testes E2E de Pipeline via subprocess.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from scripts.convert_article_to_markdown import (
|
||||
assemble_markdown_document,
|
||||
clean_body_images,
|
||||
convert_article,
|
||||
convert_html_to_markdown,
|
||||
normalize_date,
|
||||
normalize_list,
|
||||
normalize_scalar,
|
||||
parse_arguments,
|
||||
remove_duplicate_initial_h1,
|
||||
resolve_article_body,
|
||||
resolve_article_metadata,
|
||||
validate_url,
|
||||
)
|
||||
|
||||
FIXTURES_DIR = Path(__file__).parent.parent / "fixtures" / "markdown_conversion"
|
||||
SCRIPT_PATH = Path(__file__).parent.parent.parent / "scripts" / "convert_article_to_markdown.py"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 1. Testes Unitários de Normalização e Sanitização Escalar
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_normalize_scalar_complex_html_entities():
|
||||
"""Valida decodificação de entidades HTML nomeadas e numéricas."""
|
||||
assert (
|
||||
normalize_scalar("River & Boca "Superclásico"")
|
||||
== 'River & Boca "Superclásico"'
|
||||
)
|
||||
assert normalize_scalar("Preço: R$ 50,00 €") == "Preço: R$ 50,00 €"
|
||||
|
||||
|
||||
def test_normalize_scalar_whitespace_collapsing():
|
||||
"""Valida colapso de tabs, quebras de linha e espaços múltiplos em um único espaço."""
|
||||
assert (
|
||||
normalize_scalar(" Texto com \t\t múltiplos \n\n espaços ")
|
||||
== "Texto com múltiplos espaços"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"placeholder",
|
||||
[
|
||||
"null",
|
||||
"Null",
|
||||
"NULL",
|
||||
"none",
|
||||
"None",
|
||||
"NONE",
|
||||
"n/a",
|
||||
"N/A",
|
||||
"N/a",
|
||||
"unknown",
|
||||
"Unknown",
|
||||
"UNKNOWN",
|
||||
"[no-author]",
|
||||
"[No-Author]",
|
||||
"[NO-AUTHOR]",
|
||||
"no-author",
|
||||
"No-Author",
|
||||
"NO-AUTHOR",
|
||||
],
|
||||
)
|
||||
def test_normalize_scalar_placeholders_discarded(placeholder: str):
|
||||
"""Garante que todos os placeholders documentados no PRD sejam descartados (retornando None)."""
|
||||
assert normalize_scalar(placeholder) is None
|
||||
assert normalize_scalar(f" {placeholder} ") is None
|
||||
|
||||
|
||||
def test_normalize_scalar_non_string_types():
|
||||
"""Valida que entradas não string retornem None de forma segura."""
|
||||
assert normalize_scalar(None) is None
|
||||
assert normalize_scalar(12345) is None
|
||||
assert normalize_scalar(["lista"]) is None
|
||||
assert normalize_scalar({"chave": "valor"}) is None
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 2. Testes Unitários de Normalização de Listas
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_normalize_list_semicolon_and_comma_split():
|
||||
"""Testa divisão por ponto e vírgula na string e vírgulas em elementos de lista para tags/categorias."""
|
||||
raw_str = "Futebol; Copa Libertadores; Conmebol; Notícias de Hoje"
|
||||
expected_str = ["Futebol", "Copa Libertadores", "Conmebol", "Notícias de Hoje"]
|
||||
assert normalize_list(raw_str) == expected_str
|
||||
|
||||
raw_list = ["Futebol", "Copa Libertadores, Conmebol", "Notícias de Hoje"]
|
||||
expected_list = ["Futebol", "Copa Libertadores", "Conmebol", "Notícias de Hoje"]
|
||||
assert normalize_list(raw_list) == expected_list
|
||||
|
||||
|
||||
def test_normalize_list_author_url_filtering():
|
||||
"""Garante que URLs em campos de autor sejam estritamente descartadas."""
|
||||
raw_authors = [
|
||||
"Ernesto Provitilo",
|
||||
"https://twitter.com/eprovitilo",
|
||||
"http://www.instagram.com/reporter",
|
||||
"www.tycsports.com/autor",
|
||||
"Juan Pablo Varsky",
|
||||
]
|
||||
result = normalize_list(raw_authors, is_author=True)
|
||||
assert result == ["Ernesto Provitilo", "Juan Pablo Varsky"]
|
||||
|
||||
|
||||
def test_normalize_list_deduplication_preserves_case_and_order():
|
||||
"""Testa deduplicação case-insensitive preservando a grafia e ordem da primeira ocorrência."""
|
||||
items = ["River Plate", "Boca Juniors", "river plate", "RIVER PLATE", "boca juniors", "Racing"]
|
||||
assert normalize_list(items) == ["River Plate", "Boca Juniors", "Racing"]
|
||||
|
||||
|
||||
def test_normalize_list_empty_and_invalid():
|
||||
"""Testa comportamento com listas vazias, nulas ou contendo apenas placeholders."""
|
||||
assert normalize_list([]) == []
|
||||
assert normalize_list(None) == []
|
||||
assert normalize_list(["n/a", "unknown", "[no-author]", " "]) == []
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 3. Testes Unitários de Parsing de Datas
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_normalize_date_iso_8601_variants():
|
||||
"""Valida parsing de datas ISO 8601 em múltiplos formatos e fusos."""
|
||||
assert normalize_date("2026-08-20T00:36:33-03:00") == "2026-08-20T00:36:33-03:00"
|
||||
assert normalize_date("2026-08-20T03:36:33+00:00") == "2026-08-20T03:36:33+00:00"
|
||||
# Data pura sem hora
|
||||
assert normalize_date("2026-08-20") == "2026-08-20"
|
||||
|
||||
|
||||
def test_normalize_date_rfc_2822_variants():
|
||||
"""Valida parsing de datas no formato RFC 2822 (usado em feeds RSS e cabeçalhos HTTP)."""
|
||||
d1 = normalize_date("Thu, 20 Aug 2026 03:27:26 GMT")
|
||||
assert d1 is not None and "2026-08-20" in d1
|
||||
|
||||
d2 = normalize_date("Wed, 19 Aug 2026 21:00:00 -0300")
|
||||
assert d2 is not None and "2026-08-19" in d2
|
||||
|
||||
|
||||
def test_normalize_date_invalid_and_placeholders():
|
||||
"""Garante que datas inválidas ou placeholders retornem None sem lançar exceção não tratada."""
|
||||
assert normalize_date("data-invalida") is None
|
||||
assert normalize_date("2026/99/99") is None
|
||||
assert normalize_date("n/a") is None
|
||||
assert normalize_date(None) is None
|
||||
assert normalize_date(123456789) is None
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 4. Testes Unitários de Validação de URLs
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"valid_url",
|
||||
[
|
||||
"https://www.tycsports.com/river-plate/los-puntajes.html",
|
||||
"http://globoesporte.globo.com/futebol/times/flamengo",
|
||||
"https://sub.dominio.co.uk:8080/path/to/resource?param=1&query=test#hash",
|
||||
"https://example.com/noticia-com-acentos-%C3%A1%C3%A9%C3%AD",
|
||||
],
|
||||
)
|
||||
def test_validate_url_valid_schemes(valid_url: str):
|
||||
"""Garante aceitação de URLs absolutas com esquema HTTP e HTTPS válidos."""
|
||||
assert validate_url(valid_url) == valid_url
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"invalid_url",
|
||||
[
|
||||
"ftp://ftp.is.co.za/rfc/rfc1808.txt",
|
||||
"file:///C:/Users/test/file.txt",
|
||||
"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAUA",
|
||||
"javascript:alert('xss')",
|
||||
"/caminho/relativo/artigo.html",
|
||||
"http://",
|
||||
"https://",
|
||||
"",
|
||||
" ",
|
||||
None,
|
||||
],
|
||||
)
|
||||
def test_validate_url_invalid_schemes(invalid_url: str | None):
|
||||
"""Garante rejeição de esquemas não permitidos, URLs relativas e strings vazias."""
|
||||
assert validate_url(invalid_url) is None
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 5. Testes Unitários de Conversão HTML para Markdown
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_convert_html_to_markdown_rich_formatting():
|
||||
"""Valida conversão de elementos HTML estruturados para Markdown com títulos ATX."""
|
||||
html_raw = (
|
||||
"<div>"
|
||||
"<h1>Título H1</h1>"
|
||||
"<h2>Subtítulo H2</h2>"
|
||||
"<h3>Seção H3</h3>"
|
||||
"<p>Parágrafo com <b>negrito</b>, <strong>forte</strong>, <i>itálico</i> e <em>ênfase</em>.</p>"
|
||||
"<blockquote>Uma citação memorável.</blockquote>"
|
||||
"<ul><li>Item 1</li><li>Item 2</li></ul>"
|
||||
"<p>Link para o <a href='https://example.com/fonte'>portal oficial</a>.</p>"
|
||||
"<code>codigo_inline()</code>"
|
||||
"<script>alert('remover');</script>"
|
||||
"<style>.esconder { display: none; }</style>"
|
||||
"</div>"
|
||||
)
|
||||
md = convert_html_to_markdown(html_raw)
|
||||
|
||||
assert "# Título H1" in md
|
||||
assert "## Subtítulo H2" in md
|
||||
assert "### Seção H3" in md
|
||||
assert "**negrito**" in md or "__negrito__" in md
|
||||
assert "> Uma citação memorável." in md
|
||||
assert "* Item 1" in md or "- Item 1" in md
|
||||
assert "[portal oficial](https://example.com/fonte)" in md
|
||||
assert "`codigo_inline()`" in md
|
||||
assert "alert('remover')" not in md
|
||||
assert "display: none" not in md
|
||||
|
||||
|
||||
def test_convert_html_to_markdown_empty_or_whitespace():
|
||||
"""Testa conversão de HTML vazio retornando string vazia."""
|
||||
assert convert_html_to_markdown("") == ""
|
||||
assert convert_html_to_markdown(" \n\t ") == ""
|
||||
assert convert_html_to_markdown(None) == ""
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 6. Testes de Isolamento Estrito do Extrator e Fallback Interno
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_resolve_article_body_trafilatura_primary_and_fallback():
|
||||
"""Testa prioridade trafilatura.markdown sobre trafilatura.text."""
|
||||
# Primário
|
||||
art1 = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"trafilatura": {"markdown": "Corpo primário Trafilatura", "text": "Texto secundário"},
|
||||
}
|
||||
assert resolve_article_body(art1) == "Corpo primário Trafilatura"
|
||||
|
||||
# Fallback
|
||||
art2 = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"trafilatura": {"markdown": None, "text": "Texto secundário Trafilatura"},
|
||||
}
|
||||
assert resolve_article_body(art2) == "Texto secundário Trafilatura"
|
||||
|
||||
|
||||
def test_resolve_article_body_newspaper4k_primary_and_fallback():
|
||||
"""Testa prioridade newspaper4k.article_html sobre newspaper4k.text."""
|
||||
# Primário HTML -> MD
|
||||
art1 = {
|
||||
"selected_extractor": "newspaper4k",
|
||||
"newspaper4k": {
|
||||
"article_html": "<p>Artigo em <b>HTML</b></p>",
|
||||
"text": "Artigo em texto puro",
|
||||
},
|
||||
}
|
||||
assert "**HTML**" in resolve_article_body(art1)
|
||||
|
||||
# Fallback
|
||||
art2 = {
|
||||
"selected_extractor": "newspaper4k",
|
||||
"newspaper4k": {"article_html": None, "text": "Artigo em texto puro Newspaper"},
|
||||
}
|
||||
assert resolve_article_body(art2) == "Artigo em texto puro Newspaper"
|
||||
|
||||
|
||||
def test_resolve_article_body_readability_primary_and_fallback():
|
||||
"""Testa prioridade readability.cleaned_html sobre readability.cleaned_text."""
|
||||
# Primário HTML -> MD
|
||||
art1 = {
|
||||
"selected_extractor": "readability",
|
||||
"readability": {
|
||||
"cleaned_html": "<p>Conteúdo <i>Readability</i></p>",
|
||||
"cleaned_text": "Texto puro Readability",
|
||||
},
|
||||
}
|
||||
body = resolve_article_body(art1)
|
||||
assert "*Readability*" in body or "_Readability_" in body
|
||||
|
||||
# Fallback
|
||||
art2 = {
|
||||
"selected_extractor": "readability",
|
||||
"readability": {"cleaned_html": "", "cleaned_text": "Texto puro Readability Fallback"},
|
||||
}
|
||||
assert resolve_article_body(art2) == "Texto puro Readability Fallback"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("extractor", ["trafilatura", "newspaper4k", "readability"])
|
||||
def test_resolve_article_body_strict_isolation_all_extractors(extractor: str):
|
||||
"""Garante que a ausência de corpo no extrator selecionado NUNCA faça fallback para outro extrator."""
|
||||
article = {
|
||||
"selected_extractor": extractor,
|
||||
"trafilatura": {"markdown": "Texto Trafilatura", "text": "Texto Trafilatura"},
|
||||
"newspaper4k": {"article_html": "<p>Texto Newspaper</p>", "text": "Texto Newspaper"},
|
||||
"readability": {
|
||||
"cleaned_html": "<p>Texto Readability</p>",
|
||||
"cleaned_text": "Texto Readability",
|
||||
},
|
||||
}
|
||||
# Esvazia o corpo do extrator selecionado
|
||||
if extractor == "trafilatura":
|
||||
article["trafilatura"] = {"markdown": None, "text": ""}
|
||||
elif extractor == "newspaper4k":
|
||||
article["newspaper4k"] = {"article_html": "", "text": None}
|
||||
elif extractor == "readability":
|
||||
article["readability"] = {"cleaned_html": None, "cleaned_text": ""}
|
||||
|
||||
with pytest.raises(ValueError, match="Corpo do extrator selecionado.*vazio|indisponível"):
|
||||
resolve_article_body(article)
|
||||
|
||||
|
||||
def test_resolve_article_body_invalid_selected_extractor():
|
||||
"""Garante erro ao receber selected_extractor ausente ou não reconhecido."""
|
||||
with pytest.raises(ValueError, match="selected_extractor inválido ou ausente"):
|
||||
resolve_article_body({"selected_extractor": "extrator_desconhecido"})
|
||||
|
||||
with pytest.raises(ValueError, match="selected_extractor inválido ou ausente"):
|
||||
resolve_article_body({"selected_extractor": None})
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 7. Testes da Matriz Determinística de Resolução de Metadados
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_metadata_priority_title_all_fallbacks():
|
||||
"""Valida a cadeia de fallback completa para o campo TÍTULO (6 níveis)."""
|
||||
# 1. Do selecionado
|
||||
art1 = {"selected_extractor": "trafilatura", "trafilatura": {"title": "Título Selecionado"}}
|
||||
assert (
|
||||
resolve_article_metadata({**art1, "crawled_url": "https://e.com"})["title"]
|
||||
== "Título Selecionado"
|
||||
)
|
||||
|
||||
# 2. input_meta.titulo
|
||||
art2 = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"input_meta": {"titulo": "Título Input Meta"},
|
||||
"crawled_url": "https://e.com",
|
||||
}
|
||||
assert resolve_article_metadata(art2)["title"] == "Título Input Meta"
|
||||
|
||||
# 3. page_title
|
||||
art3 = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"page_title": "Título Page Title",
|
||||
"crawled_url": "https://e.com",
|
||||
}
|
||||
assert resolve_article_metadata(art3)["title"] == "Título Page Title"
|
||||
|
||||
# 4. newspaper4k.title
|
||||
art4 = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"newspaper4k": {"title": "Título Newspaper"},
|
||||
"crawled_url": "https://e.com",
|
||||
}
|
||||
assert resolve_article_metadata(art4)["title"] == "Título Newspaper"
|
||||
|
||||
# 5. readability.title
|
||||
art5 = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"readability": {"title": "Título Readability"},
|
||||
"crawled_url": "https://e.com",
|
||||
}
|
||||
assert resolve_article_metadata(art5)["title"] == "Título Readability"
|
||||
|
||||
|
||||
def test_metadata_priority_original_url_all_fallbacks():
|
||||
"""Valida a cadeia de fallback completa para a URL ORIGINAL (5 níveis)."""
|
||||
# 1. input_meta.url
|
||||
art1 = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"trafilatura": {"title": "T"},
|
||||
"input_meta": {"url": "https://example.com/input-meta"},
|
||||
"crawled_url": "https://example.com/crawled",
|
||||
}
|
||||
assert resolve_article_metadata(art1)["original_url"] == "https://example.com/input-meta"
|
||||
|
||||
# 2. crawled_url
|
||||
art2 = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"trafilatura": {"title": "T"},
|
||||
"crawled_url": "https://example.com/crawled",
|
||||
}
|
||||
assert resolve_article_metadata(art2)["original_url"] == "https://example.com/crawled"
|
||||
|
||||
# 3. Canonical do selecionado
|
||||
art3 = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"trafilatura": {"title": "T", "canonical_url": "https://example.com/canonical-trafilatura"},
|
||||
}
|
||||
assert (
|
||||
resolve_article_metadata(art3)["original_url"]
|
||||
== "https://example.com/canonical-trafilatura"
|
||||
)
|
||||
|
||||
# 4. Canonical do newspaper4k
|
||||
art4 = {
|
||||
"selected_extractor": "readability",
|
||||
"readability": {"title": "T"},
|
||||
"newspaper4k": {"canonical_link": "https://example.com/canonical-newspaper"},
|
||||
}
|
||||
assert (
|
||||
resolve_article_metadata(art4)["original_url"] == "https://example.com/canonical-newspaper"
|
||||
)
|
||||
|
||||
|
||||
def test_metadata_priority_subtitle_omitted_when_equal_to_title():
|
||||
"""Garante que subtítulo idêntico ao título seja automaticamente omitido (None)."""
|
||||
art = {
|
||||
"selected_extractor": "trafilatura",
|
||||
"trafilatura": {
|
||||
"title": "Grande Vitória no Clássico",
|
||||
"description": " grande vitória no clássico ",
|
||||
},
|
||||
"crawled_url": "https://example.com/noticia",
|
||||
}
|
||||
meta = resolve_article_metadata(art)
|
||||
assert meta["title"] == "Grande Vitória no Clássico"
|
||||
assert meta["subtitle"] is None
|
||||
|
||||
|
||||
def test_metadata_priority_first_valid_source_no_cross_merging():
|
||||
"""Garante que listas de autores/tags usem apenas a primeira fonte válida, sem merge cruzado."""
|
||||
art = {
|
||||
"selected_extractor": "readability",
|
||||
"crawled_url": "https://example.com/noticia",
|
||||
"readability": {"title": "Título", "author": "Carlos Bilardo"},
|
||||
"newspaper4k": {"authors": ["Juan Pérez", "María Gómez"]},
|
||||
}
|
||||
meta = resolve_article_metadata(art)
|
||||
# Deve pegar o autor de Readability (selecionado), sem misturar com Newspaper4k
|
||||
assert meta["authors"] == ["Carlos Bilardo"]
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 8. Testes de Higienização de Cabeçalhos e Imagens no Corpo
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_remove_duplicate_initial_h1_exact_and_variations():
|
||||
"""Testa remoção de H1 inicial coincidente com título com variações de espaços e caixa."""
|
||||
title = "River Plate Conquista a Copa"
|
||||
|
||||
# H1 inicial igual
|
||||
body1 = "# river plate conquista a copa\n\nPrimeiro parágrafo do artigo."
|
||||
assert remove_duplicate_initial_h1(body1, title).strip() == "Primeiro parágrafo do artigo."
|
||||
|
||||
# H1 inicial diferente (deve ser preservado)
|
||||
body2 = "# Outro Título Diferente\n\nPrimeiro parágrafo."
|
||||
assert remove_duplicate_initial_h1(body2, title) == body2
|
||||
|
||||
# Sem H1 inicial
|
||||
body3 = "Parágrafo sem nenhum título H1 inicial."
|
||||
assert remove_duplicate_initial_h1(body3, title) == body3
|
||||
|
||||
|
||||
def test_clean_body_images_removes_invalid_and_deduplicates():
|
||||
"""Valida descarte de data:, relativos e deduplicação mantendo a primeira ocorrência."""
|
||||
body = (
|
||||
"\n\n"
|
||||
"\n\n"
|
||||
"\n\n"
|
||||
"\n\n"
|
||||
""
|
||||
)
|
||||
cleaned = clean_body_images(body)
|
||||
|
||||
assert cleaned.count("https://example.com/img1.jpg") == 1
|
||||
assert "https://example.com/img2.webp" in cleaned
|
||||
assert "/imagem.png" not in cleaned
|
||||
assert "data:image" not in cleaned
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 9. Testes de Montagem e Formatação do Documento Markdown
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_assemble_markdown_document_full_and_minimal():
|
||||
"""Testa montagem com todos os campos e apenas com campos obrigatórios."""
|
||||
# Artigo Mínimo (apenas Título e URL Original)
|
||||
min_meta: dict[str, Any] = {
|
||||
"title": "Título Mínimo",
|
||||
"original_url": "https://example.com/minimo",
|
||||
"subtitle": None,
|
||||
"authors": [],
|
||||
"publish_date": None,
|
||||
"site_name": None,
|
||||
"categories": [],
|
||||
"tags": [],
|
||||
"keywords": [],
|
||||
"language": None,
|
||||
"top_image": None,
|
||||
}
|
||||
doc_min = assemble_markdown_document(min_meta, "Corpo do texto simples.")
|
||||
|
||||
assert doc_min.startswith("# Título Mínimo\n\n")
|
||||
assert "**Fonte original:** [https://example.com/minimo](https://example.com/minimo)" in doc_min
|
||||
assert "**Autor:**" not in doc_min
|
||||
assert "**Site:**" not in doc_min
|
||||
assert "![Imagem principal]" not in doc_min
|
||||
assert "\n\n---\n\nCorpo do texto simples.\n" in doc_min
|
||||
assert doc_min.endswith("\n")
|
||||
assert not doc_min.endswith("\n\n")
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 10. Golden Test Fixtures (Conformidade 100% Byte a Byte)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_golden_fixtures_byte_level_precision(tmp_path):
|
||||
"""Garante correspondência exata byte a byte para Trafilatura, Newspaper4k e Readability."""
|
||||
for extractor in ["trafilatura", "newspaper4k", "readability"]:
|
||||
input_json = FIXTURES_DIR / f"valid_{extractor}.json"
|
||||
expected_md = (FIXTURES_DIR / f"valid_{extractor}.md").read_text(encoding="utf-8")
|
||||
output_md = tmp_path / f"valid_{extractor}.md"
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, str(SCRIPT_PATH), "-i", str(input_json), "-o", str(output_md)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res.returncode == 0, f"Erro no extrator {extractor}: {res.stderr}"
|
||||
generated_md = output_md.read_text(encoding="utf-8")
|
||||
assert generated_md == expected_md, f"Divergência byte a byte na fixture {extractor}"
|
||||
|
||||
|
||||
def test_conversion_determinism_sha256_repeatability(tmp_path):
|
||||
"""Garante que múltiplas execuções no mesmo arquivo produzam hashes SHA-256 idênticos."""
|
||||
input_json = FIXTURES_DIR / "valid_trafilatura.json"
|
||||
hashes = set()
|
||||
|
||||
for i in range(5):
|
||||
out_md = tmp_path / f"deterministic_{i}.md"
|
||||
res = subprocess.run(
|
||||
[sys.executable, str(SCRIPT_PATH), "-i", str(input_json), "-o", str(out_md)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res.returncode == 0
|
||||
content_bytes = out_md.read_bytes()
|
||||
hashes.add(hashlib.sha256(content_bytes).hexdigest())
|
||||
|
||||
assert len(hashes) == 1, "A conversão não foi 100% determinística entre execuções repetidas."
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 11. Testes de Integração CLI, Validação e Códigos de Saída
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_cli_exit_codes_and_error_handling(tmp_path):
|
||||
"""Testa toda a matriz de códigos de saída da CLI (0, 1, 2)."""
|
||||
# Código 2: Sintaxe / Argumentos Faltantes
|
||||
res_no_args = subprocess.run([sys.executable, str(SCRIPT_PATH)], capture_output=True, text=True)
|
||||
assert res_no_args.returncode == 2
|
||||
|
||||
# Código 1: Arquivo Inexistente
|
||||
res_missing_file = subprocess.run(
|
||||
[sys.executable, str(SCRIPT_PATH), "-i", "arquivo_que_nao_existe_xyz.json"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res_missing_file.returncode == 1
|
||||
assert "não encontrado" in res_missing_file.stderr.lower()
|
||||
|
||||
# Código 1: Rejeição de Batch com chave 'articles'
|
||||
res_batch = subprocess.run(
|
||||
[sys.executable, str(SCRIPT_PATH), "-i", str(FIXTURES_DIR / "batch_articles_invalid.json")],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res_batch.returncode == 1
|
||||
assert "articles" in res_batch.stderr.lower()
|
||||
|
||||
# Código 1: JSON Corrompido
|
||||
res_corrupt = subprocess.run(
|
||||
[sys.executable, str(SCRIPT_PATH), "-i", str(FIXTURES_DIR / "corrupt_json_invalid.json")],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res_corrupt.returncode == 1
|
||||
assert "json" in res_corrupt.stderr.lower()
|
||||
|
||||
|
||||
def test_cli_json_root_must_be_object(tmp_path):
|
||||
"""Garante encerramento com código 1 caso a raiz do JSON seja lista, número ou string."""
|
||||
for invalid_root in [["item1", "item2"], 12345, "string simples"]:
|
||||
bad_json = tmp_path / "bad_root.json"
|
||||
bad_json.write_text(json.dumps(invalid_root), encoding="utf-8")
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, str(SCRIPT_PATH), "-i", str(bad_json)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res.returncode == 1
|
||||
assert "objeto" in res.stderr.lower() or "dict" in res.stderr.lower()
|
||||
|
||||
|
||||
def test_cli_no_stdout_pollution_and_atomic_preservation(tmp_path):
|
||||
"""Garante que stdout permaneça limpo e gravação atômica preserve arquivos preexistentes em falha."""
|
||||
target_md = tmp_path / "target_document.md"
|
||||
target_md.write_text("Versão Original Preservada", encoding="utf-8")
|
||||
|
||||
# Executa conversão bem-sucedida
|
||||
res_ok = subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
str(SCRIPT_PATH),
|
||||
"-i",
|
||||
str(FIXTURES_DIR / "valid_trafilatura.json"),
|
||||
"-o",
|
||||
str(target_md),
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res_ok.returncode == 0
|
||||
assert res_ok.stdout == "" # Não polui stdout
|
||||
assert "[INFO]" in res_ok.stderr
|
||||
|
||||
# Executa falha direcionada ao mesmo target
|
||||
res_fail = subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
str(SCRIPT_PATH),
|
||||
"-i",
|
||||
str(FIXTURES_DIR / "missing_body_invalid.json"),
|
||||
"-o",
|
||||
str(target_md),
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res_fail.returncode == 1
|
||||
# O arquivo target deve manter o conteúdo do sucesso anterior, não foi apagado/corrompido
|
||||
assert "# Los puntajes de River" in target_md.read_text(encoding="utf-8")
|
||||
|
||||
# Verifica que não há arquivos temporários .tmp no diretório
|
||||
tmp_files = list(tmp_path.glob("*.tmp"))
|
||||
assert len(tmp_files) == 0
|
||||
|
||||
|
||||
def test_cli_default_output_naming(tmp_path):
|
||||
"""Garante geração automática de <input_stem>.md quando -o não é informado."""
|
||||
sample_file = tmp_path / "meu_artigo_editorial.json"
|
||||
sample_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"selected_extractor": "trafilatura",
|
||||
"crawled_url": "https://example.com/editorial",
|
||||
"trafilatura": {"title": "Editorial do Dia", "markdown": "Texto do editorial."},
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
res = subprocess.run(
|
||||
[sys.executable, str(SCRIPT_PATH), "-i", str(sample_file)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res.returncode == 0
|
||||
expected_md = tmp_path / "meu_artigo_editorial.md"
|
||||
assert expected_md.exists()
|
||||
assert "# Editorial do Dia" in expected_md.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 12. Teste E2E de Pipeline Real (Artigo Autêntico)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_e2e_pipeline_with_real_extracted_selected_json(tmp_path):
|
||||
"""Valida a conversão E2E de um artigo real extraído do arquivo out/river_plate_extracted_selected.json."""
|
||||
sample_source = Path(__file__).parent.parent.parent / "out" / "river_plate_extracted_selected.json"
|
||||
if not sample_source.exists():
|
||||
pytest.skip(
|
||||
"Arquivo out/river_plate_extracted_selected.json não encontrado para teste de integração real."
|
||||
)
|
||||
|
||||
data = json.loads(sample_source.read_text(encoding="utf-8"))
|
||||
assert "articles" in data and len(data["articles"]) > 0
|
||||
|
||||
first_article = data["articles"][0]
|
||||
input_json = tmp_path / "river_first_article.json"
|
||||
input_json.write_text(json.dumps(first_article, indent=2, ensure_ascii=False), encoding="utf-8")
|
||||
|
||||
output_md = tmp_path / "river_first_article.md"
|
||||
res = subprocess.run(
|
||||
[sys.executable, str(SCRIPT_PATH), "-i", str(input_json), "-o", str(output_md)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res.returncode == 0, f"Erro na conversão E2E real: {res.stderr}"
|
||||
assert output_md.exists()
|
||||
|
||||
md_content = output_md.read_text(encoding="utf-8")
|
||||
assert md_content.startswith("# ")
|
||||
assert "**Fonte original:** [" in md_content
|
||||
assert "\n\n---\n\n" in md_content
|
||||
assert len(md_content.splitlines()) > 10
|
||||
|
||||
|
||||
def test_convert_article_api_direct(tmp_path):
|
||||
"""Testa a chamada direta da função convert_article em código Python."""
|
||||
in_file = tmp_path / "direct.json"
|
||||
in_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"selected_extractor": "trafilatura",
|
||||
"crawled_url": "https://example.com/direct",
|
||||
"trafilatura": {"title": "Título Direto", "markdown": "Conteúdo direto."},
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
out_file = tmp_path / "direct.md"
|
||||
result = convert_article(in_file, out_file)
|
||||
assert result == out_file
|
||||
assert out_file.exists()
|
||||
assert "# Título Direto" in out_file.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def test_parse_arguments_api_direct():
|
||||
"""Testa a chamada direta do parse_arguments."""
|
||||
args = parse_arguments(["-i", "input_test.json", "-o", "output_test.md"])
|
||||
assert args.input == Path("input_test.json")
|
||||
assert args.output == Path("output_test.md")
|
||||
@@ -0,0 +1,388 @@
|
||||
"""
|
||||
Suíte de Testes E2E e de Integração Completa para Análise de Texto e Classificação de Inerência (QA Sênior).
|
||||
|
||||
Valida todo o funil de análise de texto:
|
||||
1. Sucesso no NLP Determinístico (Tier 1 DIRECT_INHERENT com bypass de LLM).
|
||||
2. Insucesso / Rejeição no NLP (Tier 1 NOT_RELATED por âncoras negativas e homônimos).
|
||||
3. Ambiguidade / Limiar detectada no NLP e resolvida com sucesso no LLM (Tier 3 Upgrade).
|
||||
4. Ambiguidade confirmada pelo LLM como TANGENTIAL (Tier 3 Confirmation).
|
||||
5. Insucesso / Falha de API de LLM com degradação graciosa para Tier 1.
|
||||
6. Cobertura Multilíngue E2E nos 6 idiomas (PT, EN, ES, DE, IT, FR).
|
||||
7. Execução E2E via CLI subprocess com contratos de entrada, saída e flags.
|
||||
8. Chamada real ao vivo a provedores de LLM (OpenAI/Gemini) quando credenciais estiverem disponíveis.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from src.tools.adapters.llm import LLMFallbackAdapter
|
||||
from src.tools.classifier import InherenceClassifier
|
||||
from src.tools.models import (
|
||||
ClassificationResult,
|
||||
DecisionCategory,
|
||||
ECPSnapshot,
|
||||
RelatedEntity,
|
||||
)
|
||||
|
||||
CLASSIFY_CLI = Path(__file__).parent.parent.parent / "classify.py"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Fixtures e Helpers para o Funil de Teste E2E
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ecp_river_plate() -> ECPSnapshot:
|
||||
"""Fixture ECP oficial para o Club Atlético River Plate."""
|
||||
return ECPSnapshot(
|
||||
target_entity_id="ecp_river_plate",
|
||||
target_name="Club Atlético River Plate",
|
||||
aliases=[
|
||||
"Club Atlético River Plate",
|
||||
"River Plate",
|
||||
"River",
|
||||
"El Millonario",
|
||||
"La Banda",
|
||||
"CARP",
|
||||
],
|
||||
domain="Futebol / Esportes",
|
||||
anchors=[
|
||||
"fútbol",
|
||||
"Copa Libertadores",
|
||||
"Libertadores",
|
||||
"Copa Sudamericana",
|
||||
"Sudamericana",
|
||||
"Monumental",
|
||||
"Estadio Monumental",
|
||||
"Eduardo Coudet",
|
||||
"Coudet",
|
||||
"Nicolás Otamendi",
|
||||
"Otamendi",
|
||||
"Rafael Santos Borré",
|
||||
],
|
||||
negative_anchors=[
|
||||
"River Plate de Montevideo",
|
||||
"River Plate de Asunción",
|
||||
"River de Piauí",
|
||||
"Rio da Prata",
|
||||
"Bacia do Rio da Prata",
|
||||
],
|
||||
related_entities=[
|
||||
RelatedEntity(
|
||||
entity_id="estadio_monumental",
|
||||
name="Estadio Mâs Monumental",
|
||||
relation_type="HOME_VENUE_OF",
|
||||
weight=0.95,
|
||||
aliases=["Monumental", "El Monumental"],
|
||||
),
|
||||
RelatedEntity(
|
||||
entity_id="copa_sudamericana",
|
||||
name="Copa Sudamericana",
|
||||
relation_type="COMPETES_IN",
|
||||
weight=0.85,
|
||||
aliases=["Sudamericana"],
|
||||
),
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 1. Funil de Sucesso NLP Determinístico (Tier 1)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_funnel_nlp_deterministic_success(ecp_river_plate: ECPSnapshot):
|
||||
"""
|
||||
Cenário 1: Artigo com alta densidade de âncoras do River Plate.
|
||||
Oráculo: Decisão DIRECT_INHERENT, confiança alta (>=0.95), LLM não é chamado.
|
||||
"""
|
||||
content = """
|
||||
# River Plate vence com autoridade na Copa Sudamericana
|
||||
Em noite histórica no Estadio Monumental, o River dominou a partida sob o comando de Eduardo Coudet.
|
||||
Otamendi e Borré marcaram os gols que garantiram a classificação na Sudamericana.
|
||||
"""
|
||||
llm_called = {"status": False}
|
||||
|
||||
def mock_llm(prompt: str) -> str:
|
||||
llm_called["status"] = True
|
||||
return json.dumps({"decision": "DIRECT_INHERENT", "confidence": 0.99})
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=mock_llm)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
result = classifier.classify(ecp_river_plate, content)
|
||||
|
||||
assert result.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert result.is_inherent is True
|
||||
assert result.confidence >= 0.95
|
||||
assert "River Plate" in result.matched_anchors or "River" in result.matched_anchors
|
||||
assert len(result.graph_matches) >= 1
|
||||
# Otimização de custo: LLM NÃO deve ser chamado em casos determinísticos claros
|
||||
assert llm_called["status"] is False
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 2. Funil de Insucesso / Rejeição NLP (Tier 1 Negativas e Homônimos)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_funnel_nlp_deterministic_rejection_homonym(ecp_river_plate: ECPSnapshot):
|
||||
"""
|
||||
Cenário 2: Artigo sobre a Bacia do Rio da Prata ou clube homônimo do Uruguai.
|
||||
Oráculo: Decisão NOT_RELATED, is_inherent=False, âncoras negativas detectadas.
|
||||
"""
|
||||
content = """
|
||||
# Expedição ambiental navega pela Bacia do Rio da Prata
|
||||
Pesquisadores mapearam a biodiversidade fluvial e os sedimentos do Rio da Prata durante o verão.
|
||||
"""
|
||||
classifier = InherenceClassifier(enable_llm=False)
|
||||
result = classifier.classify(ecp_river_plate, content)
|
||||
|
||||
assert result.decision == DecisionCategory.NOT_RELATED
|
||||
assert result.is_inherent is False
|
||||
assert any("Rio da Prata" in neg for neg in result.negative_matches)
|
||||
assert "Negative anchor" in result.rationale
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 3. Funil de Ambiguidade NLP -> Resolução com Sucesso no LLM (Tier 3)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_funnel_nlp_ambiguity_resolved_by_llm_upgrade(ecp_river_plate: ECPSnapshot):
|
||||
"""
|
||||
Cenário 3: Menção isolada do clube ('River') em contexto com poucas âncoras explícitas.
|
||||
Tier 1 preliminar: TANGENTIAL (baixa densidade).
|
||||
LLM Fallback: Analisa o contexto profundo e eleva para DIRECT_INHERENT.
|
||||
"""
|
||||
ambiguous_content = """
|
||||
# Bastidores do mercado sul-americano
|
||||
A diretoria do River finalizou os últimos detalhes contratuais para a renovação de jovens promessas.
|
||||
"""
|
||||
mock_llm_response = json.dumps(
|
||||
{
|
||||
"analysis_summary": "O artigo trata da gestão de elenco e renovações contratuais do clube River Plate.",
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": 0.94,
|
||||
"rationale": "A análise contextual profunda comprova que a matéria é focada na administração do River Plate.",
|
||||
}
|
||||
)
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda prompt: mock_llm_response)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
result = classifier.classify(ecp_river_plate, ambiguous_content)
|
||||
|
||||
assert result.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert result.is_inherent is True
|
||||
assert result.confidence == 0.94
|
||||
assert "[Tier 3 LLM]" in result.rationale
|
||||
assert "[Tier 3 LLM Override applied]" in result.warnings
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 4. Funil de Ambiguidade NLP -> Confirmação de Tangencial no LLM
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_funnel_nlp_ambiguity_confirmed_tangential_by_llm(ecp_river_plate: ECPSnapshot):
|
||||
"""
|
||||
Cenário 4: Menção metafórica ou turística a um local próximo.
|
||||
Tier 1 preliminar: TANGENTIAL.
|
||||
LLM Fallback: Confirma que é meramente periférico/ilustrativo.
|
||||
"""
|
||||
tangential_content = """
|
||||
# Melhores restaurantes do bairro de Núñez em Buenos Aires
|
||||
Ao visitar a capital portenha, próximo de onde fica o River, você encontra excelentes opções gastronômicas.
|
||||
"""
|
||||
mock_llm_response = json.dumps(
|
||||
{
|
||||
"analysis_summary": "Guia gastronômico sobre o bairro de Núñez com citação geográfica casual ao clube.",
|
||||
"decision": "TANGENTIAL",
|
||||
"confidence": 0.96,
|
||||
"rationale": "A entidade é usada apenas como ponto de referência geográfica em um artigo sobre restaurantes.",
|
||||
}
|
||||
)
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda prompt: mock_llm_response)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
result = classifier.classify(ecp_river_plate, tangential_content)
|
||||
|
||||
assert result.decision == DecisionCategory.TANGENTIAL
|
||||
assert result.is_inherent is False
|
||||
assert result.confidence == 0.96
|
||||
assert "[Tier 3 LLM]" in result.rationale
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 5. Funil de Insucesso / Degradação Graciosa em Falha do LLM
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_funnel_llm_failure_graceful_degradation(ecp_river_plate: ECPSnapshot):
|
||||
"""
|
||||
Cenário 5: LLM configurado, caso ambíguo, mas a API externa sofre timeout/500.
|
||||
Oráculo: Mantém o resultado do Tier 1 determinístico com warning detalhado e sem quebrar.
|
||||
"""
|
||||
|
||||
def broken_llm(prompt: str) -> str:
|
||||
raise TimeoutError("Conexão com serviço de LLM excedeu 30 segundos.")
|
||||
|
||||
adapter = LLMFallbackAdapter(provider_fn=broken_llm)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=adapter)
|
||||
|
||||
content = "O River esteve presente no evento de inauguração da praça."
|
||||
result = classifier.classify(ecp_river_plate, content)
|
||||
|
||||
# Mantém o Tier 1 determinístico
|
||||
assert result.decision == DecisionCategory.TANGENTIAL
|
||||
assert result.is_inherent is False
|
||||
assert any("LLM fallback failed" in w for w in result.warnings)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 6. Cobertura Multilíngue nos 6 Idiomas (PT, EN, ES, DE, IT, FR)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"lang_code,content,expected_lang",
|
||||
[
|
||||
("pt", "# Petrobras anuncia perfuração no pré-sal com tecnologia nacional.", "pt"),
|
||||
("en", "# Apple unveils new generative AI features for upcoming devices.", "en"),
|
||||
(
|
||||
"es",
|
||||
"# River Plate prepara su viaje a Bogotá para disputar el torneo continental.",
|
||||
"es",
|
||||
),
|
||||
(
|
||||
"de",
|
||||
"# Volkswagen investiert Milliarden in neue Batterie-Fabriken in Deutschland.",
|
||||
"de",
|
||||
),
|
||||
("it", "# Ferrari conquista la pole position nel Gran Premio di Monza.", "it"),
|
||||
(
|
||||
"fr",
|
||||
"# L'entreprise TotalEnergies accélère ses investissements solaires en France.",
|
||||
"fr",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_funnel_multilingual_language_detection(
|
||||
lang_code: str, content: str, expected_lang: str, ecp_river_plate: ECPSnapshot
|
||||
):
|
||||
"""Garante a identificação precisa de idioma e integridade nos 6 idiomas suportados."""
|
||||
classifier = InherenceClassifier(enable_llm=False)
|
||||
result = classifier.classify(ecp_river_plate, content)
|
||||
assert result.detected_language == expected_lang
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 7. Execução E2E via CLI Subprocess
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_funnel_cli_subprocess_end_to_end(tmp_path: Path):
|
||||
"""Valida o contrato CLI completo classify.py com saída em arquivo JSON e flags ativas."""
|
||||
ecp_path = tmp_path / "ecp.json"
|
||||
ecp_path.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ecp_test_e2e",
|
||||
"target_name": "Clube Teste",
|
||||
"aliases": ["Clube Teste", "Clube"],
|
||||
"domain": "Esportes",
|
||||
"anchors": ["campeonato", "vitória", "torneio"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
doc_path = tmp_path / "artigo.md"
|
||||
doc_path.write_text(
|
||||
"# Clube Teste comemora vitória histórica no campeonato\n\nEquipe foi campeã do torneio.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
out_path = tmp_path / "resultado.json"
|
||||
|
||||
res = subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
str(CLASSIFY_CLI),
|
||||
"--ecp",
|
||||
str(ecp_path),
|
||||
"--content",
|
||||
str(doc_path),
|
||||
"--output",
|
||||
str(out_path),
|
||||
"--enable-llm",
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
|
||||
assert res.returncode == 0, f"Erro na CLI: {res.stderr}"
|
||||
assert out_path.exists()
|
||||
|
||||
payload = json.loads(out_path.read_text(encoding="utf-8"))
|
||||
assert payload["decision"] == "DIRECT_INHERENT"
|
||||
assert payload["is_inherent"] is True
|
||||
assert payload["confidence"] >= 0.85
|
||||
assert isinstance(payload["matched_anchors"], list)
|
||||
assert isinstance(payload["evidence"], list)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 8. Teste Live Opt-In com API Real (OpenAI / Gemini) se .env Estiver Presente
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_funnel_live_api_execution_if_configured():
|
||||
"""
|
||||
Executa chamada ao vivo contra OpenAI ou Gemini caso OPENAI_API_KEY ou GEMINI_API_KEY
|
||||
esteja configurada no ambiente ou no arquivo .env.
|
||||
"""
|
||||
adapter = LLMFallbackAdapter()
|
||||
if not (adapter.openai_api_key or adapter.gemini_api_key):
|
||||
pytest.skip(
|
||||
"Chaves de API reais (OPENAI_API_KEY ou GEMINI_API_KEY) não configuradas no .env"
|
||||
)
|
||||
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ecp_live_test",
|
||||
target_name="Club Atlético River Plate",
|
||||
aliases=["River Plate", "River"],
|
||||
domain="Futebol",
|
||||
anchors=["Monumental", "Libertadores"],
|
||||
)
|
||||
initial_res = ClassificationResult(
|
||||
decision=DecisionCategory.TANGENTIAL,
|
||||
is_inherent=False,
|
||||
confidence=0.40,
|
||||
detected_language="es",
|
||||
matched_anchors=["River"],
|
||||
negative_matches=[],
|
||||
graph_matches=[],
|
||||
evidence=["River"],
|
||||
rationale="Passing mention detected by Tier 1.",
|
||||
warnings=[],
|
||||
)
|
||||
|
||||
# Texto de teste para a API ao vivo
|
||||
content = "O River Plate empatou em 1 a 1 em Bogotá com gols de Otamendi na Copa Sul-Americana."
|
||||
|
||||
refined = adapter.disambiguate(ecp, content, initial_res)
|
||||
assert refined is not None
|
||||
assert refined.decision in [
|
||||
DecisionCategory.DIRECT_INHERENT,
|
||||
DecisionCategory.CONTEXTUAL_INHERENT,
|
||||
]
|
||||
assert refined.is_inherent is True
|
||||
assert "[Tier 3 LLM]" in refined.rationale
|
||||
@@ -0,0 +1,372 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Testes automatizados para o Extrator e Parser Multimotor de Artigos.
|
||||
Cobre modelos de dados, parsers (Trafilatura, Newspaper4k, Readability),
|
||||
isolamento de falhas, orquestração de lote e interface CLI.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from scripts.extract_article_contents import (
|
||||
ArticleCrawler,
|
||||
ExtractedArticle,
|
||||
ExtractionBatchReport,
|
||||
InputArticle,
|
||||
NewspaperData,
|
||||
NewspaperExtractor,
|
||||
ReadabilityData,
|
||||
ReadabilityExtractor,
|
||||
TrafilaturaData,
|
||||
TrafilaturaExtractor,
|
||||
extract_all_engines,
|
||||
load_search_json,
|
||||
main,
|
||||
process_batch,
|
||||
save_extracted_json,
|
||||
)
|
||||
|
||||
SAMPLE_HTML = """
|
||||
<!DOCTYPE html>
|
||||
<html lang="es">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>River Plate igualó sin goles ante Independiente Santa Fe - Olé</title>
|
||||
<meta name="description" content="El equipo de Núñez empató 0-0 en Bogotá por los octavos de final.">
|
||||
<meta name="author" content="Juan Pérez">
|
||||
<meta property="og:title" content="River Plate igualó sin goles ante Independiente Santa Fe">
|
||||
<meta property="og:image" content="https://media.ole.com.ar/river.jpg">
|
||||
</head>
|
||||
<body>
|
||||
<header><nav><a href="/">Inicio</a></nav></header>
|
||||
<article>
|
||||
<h1>River Plate igualó sin goles ante Independiente Santa Fe</h1>
|
||||
<p class="byline">Por Juan Pérez - 20 de Agosto de 2026</p>
|
||||
<p class="lead">El equipo de Núñez empató 0-0 en Bogotá por la Copa Sudamericana.</p>
|
||||
<p>Franco Armani fue la gran figura del encuentro con tres atajadas espectaculares en el primer tiempo.</p>
|
||||
<p>El partido de vuelta se disputará en el estadio Monumental la próxima semana ante una multitud.</p>
|
||||
</article>
|
||||
<footer><p>Copyright 2026 Olé</p></footer>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_input_json(tmp_path: Path) -> Path:
|
||||
data = {
|
||||
"query": "River Plate",
|
||||
"language": "es",
|
||||
"locale": "AR",
|
||||
"total_itens": 2,
|
||||
"items": [
|
||||
{
|
||||
"titulo": "River Plate igualó sin goles ante Santa Fe",
|
||||
"subtitulo": "Empate en Bogotá",
|
||||
"quando_publicado": "Thu, 20 Aug 2026 03:27:26 GMT",
|
||||
"url": "https://www.ole.com.ar/river-0-0-santa-fe.html",
|
||||
"pagina": 1,
|
||||
},
|
||||
{
|
||||
"titulo": "Armani fue la figura de River",
|
||||
"subtitulo": "Gran actuación del arquero",
|
||||
"quando_publicado": "Thu, 20 Aug 2026 04:00:00 GMT",
|
||||
"url": "https://www.tycsports.com/armani-figura.html",
|
||||
"pagina": 1,
|
||||
},
|
||||
],
|
||||
}
|
||||
input_file = tmp_path / "river_plate.json"
|
||||
input_file.write_text(json.dumps(data, ensure_ascii=False), encoding="utf-8")
|
||||
return input_file
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 1. Testes de Modelos e I/O de JSON
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_input_article_creation():
|
||||
article = InputArticle(
|
||||
titulo="Notícia Teste",
|
||||
url="https://example.com/noticia",
|
||||
subtitulo="Subtítulo",
|
||||
quando_publicado="Thu, 20 Aug 2026",
|
||||
pagina=1,
|
||||
)
|
||||
assert article.titulo == "Notícia Teste"
|
||||
assert article.url == "https://example.com/noticia"
|
||||
assert article.pagina == 1
|
||||
d = article.to_dict()
|
||||
assert d["titulo"] == "Notícia Teste"
|
||||
assert d["url"] == "https://example.com/noticia"
|
||||
|
||||
|
||||
def test_load_search_json_valid(sample_input_json: Path):
|
||||
query, lang, items = load_search_json(sample_input_json)
|
||||
assert query == "River Plate"
|
||||
assert lang == "es"
|
||||
assert len(items) == 2
|
||||
assert items[0].url == "https://www.ole.com.ar/river-0-0-santa-fe.html"
|
||||
|
||||
|
||||
def test_load_search_json_invalid_file(tmp_path: Path):
|
||||
non_existent = tmp_path / "missing.json"
|
||||
with pytest.raises(FileNotFoundError):
|
||||
load_search_json(non_existent)
|
||||
|
||||
|
||||
def test_save_extracted_json(tmp_path: Path):
|
||||
report = ExtractionBatchReport(
|
||||
source_file="test.json",
|
||||
processed_at="2026-08-20T12:00:00Z",
|
||||
total_articles=1,
|
||||
successful_articles=1,
|
||||
failed_articles=0,
|
||||
articles=[
|
||||
ExtractedArticle(
|
||||
input_meta=InputArticle(titulo="Teste", url="https://example.com/noticia"),
|
||||
extraction_status="success",
|
||||
error_message=None,
|
||||
crawled_url="https://example.com/noticia",
|
||||
page_title="Página Teste",
|
||||
http_status=200,
|
||||
trafilatura=TrafilaturaData(
|
||||
title="Teste",
|
||||
author="Autor",
|
||||
date="2026-08-20",
|
||||
description="Desc",
|
||||
categories=[],
|
||||
tags=[],
|
||||
canonical_url=None,
|
||||
text="Texto do teste.",
|
||||
raw_json=None,
|
||||
error=None,
|
||||
),
|
||||
newspaper4k=NewspaperData(
|
||||
title="Teste",
|
||||
authors=["Autor"],
|
||||
publish_date="2026-08-20",
|
||||
text="Texto do teste.",
|
||||
summary="Resumo",
|
||||
keywords=["teste"],
|
||||
top_image=None,
|
||||
images=[],
|
||||
meta_data={},
|
||||
error=None,
|
||||
),
|
||||
readability=ReadabilityData(
|
||||
title="Teste",
|
||||
short_title="Teste",
|
||||
cleaned_html="<p>Texto do teste.</p>",
|
||||
cleaned_text="Texto do teste.",
|
||||
error=None,
|
||||
),
|
||||
)
|
||||
],
|
||||
)
|
||||
out_file = tmp_path / "out" / "result.json"
|
||||
save_extracted_json(report, out_file)
|
||||
assert out_file.exists()
|
||||
content = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert content["total_articles"] == 1
|
||||
assert content["articles"][0]["trafilatura"]["title"] == "Teste"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 2. Testes Unitários dos Parsers (Trafilatura, Newspaper4k, Readability)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_trafilatura_extractor():
|
||||
extractor = TrafilaturaExtractor()
|
||||
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia")
|
||||
assert isinstance(res, TrafilaturaData)
|
||||
assert res.error is None
|
||||
assert "Armani" in res.text or "River Plate" in res.text
|
||||
assert res.title is not None
|
||||
|
||||
|
||||
def test_newspaper_extractor():
|
||||
extractor = NewspaperExtractor()
|
||||
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es")
|
||||
assert isinstance(res, NewspaperData)
|
||||
assert res.error is None
|
||||
assert "Armani" in res.text or "River" in res.text
|
||||
assert isinstance(res.keywords, list)
|
||||
assert len(res.keywords) > 0
|
||||
|
||||
|
||||
def test_readability_extractor():
|
||||
extractor = ReadabilityExtractor()
|
||||
res = extractor.extract(SAMPLE_HTML)
|
||||
assert isinstance(res, ReadabilityData)
|
||||
assert res.error is None
|
||||
assert res.title is not None
|
||||
assert res.cleaned_html is not None
|
||||
assert "Armani" in (res.cleaned_text or "") or "River" in (res.cleaned_text or "")
|
||||
|
||||
|
||||
def test_extract_all_engines():
|
||||
traf, news, read = extract_all_engines(
|
||||
SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es"
|
||||
)
|
||||
assert traf.error is None
|
||||
assert news.error is None
|
||||
assert read.error is None
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 3. Testes de Isolamento de Falhas (Resiliência)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_extractor_error_isolation_on_faulty_engine():
|
||||
with patch(
|
||||
"scripts.extract_article_contents.TrafilaturaExtractor.extract",
|
||||
side_effect=RuntimeError("Trafilatura crash"),
|
||||
):
|
||||
traf, news, read = extract_all_engines(
|
||||
SAMPLE_HTML, url="https://example.com", language="es"
|
||||
)
|
||||
assert traf.error == "Trafilatura crash"
|
||||
assert news.error is None
|
||||
assert read.error is None
|
||||
|
||||
|
||||
def test_crawler_error_isolation(sample_input_json: Path, tmp_path: Path):
|
||||
out_file = tmp_path / "river_plate_extracted.json"
|
||||
|
||||
def mock_crawl(url, timeout_sec=30):
|
||||
if "ole.com.ar" in url:
|
||||
return SAMPLE_HTML, "River Plate Olé", 200
|
||||
raise ConnectionError("Connection refused by tycsports.com")
|
||||
|
||||
with patch.object(ArticleCrawler, "crawl", side_effect=mock_crawl):
|
||||
with patch.object(ArticleCrawler, "start"), patch.object(ArticleCrawler, "close"):
|
||||
report = process_batch(
|
||||
input_path=sample_input_json,
|
||||
output_path=out_file,
|
||||
silent=True,
|
||||
)
|
||||
|
||||
assert report.total_articles == 2
|
||||
assert report.successful_articles == 1
|
||||
assert report.failed_articles == 1
|
||||
assert report.articles[0].extraction_status == "success"
|
||||
assert report.articles[1].extraction_status == "failed"
|
||||
assert "Connection refused" in (report.articles[1].error_message or "")
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 4. Testes de CLI e Limitação (--limit, --silent, --language)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_process_batch_with_limit(sample_input_json: Path, tmp_path: Path):
|
||||
out_file = tmp_path / "limit_extracted.json"
|
||||
|
||||
with (
|
||||
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
|
||||
patch.object(ArticleCrawler, "start"),
|
||||
patch.object(ArticleCrawler, "close"),
|
||||
):
|
||||
report = process_batch(
|
||||
input_path=sample_input_json,
|
||||
output_path=out_file,
|
||||
limit=1,
|
||||
silent=True,
|
||||
)
|
||||
|
||||
assert report.total_articles == 1
|
||||
assert len(report.articles) == 1
|
||||
assert out_file.exists()
|
||||
|
||||
|
||||
def test_cli_main_success(sample_input_json: Path, tmp_path: Path, capsys):
|
||||
out_file = tmp_path / "cli_out.json"
|
||||
|
||||
with (
|
||||
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
|
||||
patch.object(ArticleCrawler, "start"),
|
||||
patch.object(ArticleCrawler, "close"),
|
||||
):
|
||||
exit_code = main(["-i", str(sample_input_json), "-o", str(out_file), "--limit", "1", "-s"])
|
||||
|
||||
assert exit_code == 0
|
||||
assert out_file.exists()
|
||||
|
||||
|
||||
def test_cli_main_missing_input_file(tmp_path: Path, capsys):
|
||||
missing_file = tmp_path / "does_not_exist.json"
|
||||
exit_code = main(["-i", str(missing_file), "-s"])
|
||||
assert exit_code == 1
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 5. Testes End-to-End (E2E) ao Vivo (Live Network)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_e2e_live_article_extraction(tmp_path: Path):
|
||||
"""Valida E2E a extração real ao vivo com Foxcape e os 3 motores em lote."""
|
||||
live_input_file = Path("out/river_plate.json")
|
||||
if not live_input_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E ao vivo.")
|
||||
|
||||
out_file = tmp_path / "e2e_live_extracted.json"
|
||||
|
||||
# Executa o batch real com limite de 1 notícia
|
||||
report = process_batch(
|
||||
input_path=live_input_file,
|
||||
output_path=out_file,
|
||||
limit=1,
|
||||
silent=True,
|
||||
)
|
||||
|
||||
assert report.total_articles == 1
|
||||
assert report.successful_articles == 1
|
||||
assert report.failed_articles == 0
|
||||
assert len(report.articles) == 1
|
||||
|
||||
art = report.articles[0]
|
||||
assert art.extraction_status == "success"
|
||||
assert art.crawled_url.startswith("http")
|
||||
|
||||
# Valida que todos os 3 motores extraíram dados reais
|
||||
assert art.trafilatura is not None and art.trafilatura.error is None
|
||||
assert len(art.trafilatura.text) > 50
|
||||
|
||||
assert art.newspaper4k is not None and art.newspaper4k.error is None
|
||||
assert len(art.newspaper4k.text) > 50
|
||||
assert isinstance(art.newspaper4k.keywords, list)
|
||||
|
||||
assert art.readability is not None and art.readability.error is None
|
||||
assert art.readability.cleaned_html is not None
|
||||
assert len(art.readability.cleaned_text or "") > 50
|
||||
|
||||
# Valida arquivo JSON gravado
|
||||
assert out_file.exists()
|
||||
saved = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert saved["total_articles"] == 1
|
||||
assert saved["articles"][0]["extraction_status"] == "success"
|
||||
|
||||
|
||||
def test_e2e_cli_live_execution(tmp_path: Path):
|
||||
"""Valida E2E a execução do CLI real de ponta a ponta."""
|
||||
live_input_file = Path("out/river_plate.json")
|
||||
if not live_input_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E.")
|
||||
|
||||
out_file = tmp_path / "e2e_cli_live.json"
|
||||
exit_code = main(["-i", str(live_input_file), "-o", str(out_file), "--limit", "1", "-s"])
|
||||
|
||||
assert exit_code == 0
|
||||
assert out_file.exists()
|
||||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert data["successful_articles"] == 1
|
||||
@@ -0,0 +1,345 @@
|
||||
"""
|
||||
Testes unitários e de integração para o Extrator de Manchetes do Google News.
|
||||
|
||||
Cobre validação de entrada, mapeamento de idiomas/locales, parsing e
|
||||
higienização de XML/HTML, orquestração, decodificação de URLs e contrato de execução CLI.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from scripts.extract_google_news import (
|
||||
ExtractionResult,
|
||||
NewsArticle,
|
||||
SearchQuery,
|
||||
extract_google_news,
|
||||
get_hl_gl_ceid,
|
||||
main,
|
||||
parse_google_news_rss,
|
||||
resolve_article_url,
|
||||
resolve_articles_urls,
|
||||
)
|
||||
|
||||
FIXTURE_PATH = Path(__file__).parent.parent / "fixtures" / "google_news_sample.xml"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_rss_xml() -> str:
|
||||
"""Fixture que fornece o conteúdo do XML de exemplo para testes offline."""
|
||||
return FIXTURE_PATH.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def test_get_hl_gl_ceid_default_mappings():
|
||||
"""Valida o mapeamento padrão de idiomas para pares (hl, gl, ceid)."""
|
||||
hl, gl, ceid = get_hl_gl_ceid("pt")
|
||||
assert hl == "pt-BR"
|
||||
assert gl == "BR"
|
||||
assert ceid == "BR:pt-BR"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("en")
|
||||
assert hl == "en-US"
|
||||
assert gl == "US"
|
||||
assert ceid == "US:en-US"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("es")
|
||||
assert hl == "es-419"
|
||||
assert gl == "AR"
|
||||
assert ceid == "AR:es-419"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("de")
|
||||
assert hl == "de"
|
||||
assert gl == "DE"
|
||||
assert ceid == "DE:de"
|
||||
|
||||
|
||||
def test_get_hl_gl_ceid_with_custom_locale():
|
||||
"""Valida a sobrescrita geográfica quando o argumento locale é especificado."""
|
||||
hl, gl, ceid = get_hl_gl_ceid("es", locale="MX")
|
||||
assert hl == "es-419"
|
||||
assert gl == "MX"
|
||||
assert ceid == "MX:es-419"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("en", locale="GB")
|
||||
assert hl == "en-GB"
|
||||
assert gl == "GB"
|
||||
assert ceid == "GB:en-GB"
|
||||
|
||||
hl, gl, ceid = get_hl_gl_ceid("es", locale="ES")
|
||||
assert hl == "es"
|
||||
assert gl == "ES"
|
||||
assert ceid == "ES:es"
|
||||
|
||||
|
||||
def test_get_hl_gl_ceid_dynamic_fallback():
|
||||
"""Valida fallback dinâmico para idiomas regionais não listados explicitamente."""
|
||||
hl, gl, ceid = get_hl_gl_ceid("ja_jp")
|
||||
assert hl == "ja-JP"
|
||||
assert gl == "JP"
|
||||
assert ceid == "JP:ja-JP"
|
||||
|
||||
|
||||
def test_search_query_validation():
|
||||
"""Valida as regras de negócio e limites de SearchQuery."""
|
||||
# Instanciação válida
|
||||
q = SearchQuery(keyword="inteligencia artificial", language="pt", max_pages=1)
|
||||
assert q.clean_keyword == "inteligencia artificial"
|
||||
assert q.clean_language == "pt"
|
||||
assert q.clean_locale is None
|
||||
assert q.max_pages == 1
|
||||
|
||||
# Palavra-chave vazia ou apenas espaços deve lançar ValueError
|
||||
with pytest.raises(ValueError, match="palavra-chave"):
|
||||
SearchQuery(keyword=" ", language="pt")
|
||||
|
||||
# Idioma com menos de 2 caracteres deve lançar ValueError
|
||||
with pytest.raises(ValueError, match="idioma"):
|
||||
SearchQuery(keyword="test", language="p")
|
||||
|
||||
# Intervalo de páginas fora de 1..10 deve lançar ValueError
|
||||
with pytest.raises(ValueError, match="páginas"):
|
||||
SearchQuery(keyword="test", language="pt", max_pages=0)
|
||||
|
||||
with pytest.raises(ValueError, match="páginas"):
|
||||
SearchQuery(keyword="test", language="pt", max_pages=11)
|
||||
|
||||
|
||||
def test_parse_google_news_rss_with_fixture(sample_rss_xml: str):
|
||||
"""Valida o parsing do feed RSS, higienização de tags HTML e deduplicação."""
|
||||
articles = parse_google_news_rss(sample_rss_xml, max_pages=1)
|
||||
|
||||
# 4 itens no fixture, mas 1 não possui link -> exatamente 3 válidos
|
||||
assert len(articles) == 3
|
||||
|
||||
# Artigo 1: InfoMoney com HTML no description que deve ser limpo
|
||||
art1 = articles[0]
|
||||
assert "InfoMoney" in art1.titulo
|
||||
assert art1.url.startswith("https://news.google.com/rss/articles/")
|
||||
assert art1.quando_publicado == "Thu, 20 Aug 2026 10:30:00 GMT"
|
||||
assert art1.pagina == 1
|
||||
assert "<" not in (art1.subtitulo or "")
|
||||
assert ">" not in (art1.subtitulo or "")
|
||||
assert "crescimento expressivo" in (art1.subtitulo or "")
|
||||
|
||||
# Artigo 2: G1 com parágrafos limpos
|
||||
art2 = articles[1]
|
||||
assert "G1" in art2.titulo
|
||||
assert "<p>" not in (art2.subtitulo or "")
|
||||
|
||||
# Artigo 3: Folha com descrição redundante/igual ao título -> subtitulo deve ser None
|
||||
art3 = articles[2]
|
||||
assert art3.subtitulo is None
|
||||
|
||||
|
||||
def test_resolve_article_url_fallback():
|
||||
"""Valida fallback gracioso de URL quando não é link do Google News ou em erro."""
|
||||
direct_url = "https://www.globo.com/noticia/123"
|
||||
assert resolve_article_url(direct_url) == direct_url
|
||||
|
||||
with patch("scripts.extract_google_news.gnewsdecoder", return_value={"status": False}):
|
||||
gn_url = "https://news.google.com/rss/articles/fake_token"
|
||||
assert resolve_article_url(gn_url) == gn_url
|
||||
|
||||
|
||||
def test_resolve_article_url_success():
|
||||
"""Valida resolução bem-sucedida de URL do Google News para o portal destino."""
|
||||
gn_url = "https://news.google.com/rss/articles/valid_token"
|
||||
dest_url = "https://infomoney.com.br/mercados/artigo-ia"
|
||||
|
||||
with patch(
|
||||
"scripts.extract_google_news.gnewsdecoder",
|
||||
return_value={"status": True, "decoded_url": dest_url},
|
||||
):
|
||||
resolved = resolve_article_url(gn_url)
|
||||
assert resolved == dest_url
|
||||
|
||||
|
||||
def test_resolve_articles_urls_batch():
|
||||
"""Valida a resolução concorrente em lote de uma lista de NewsArticle."""
|
||||
articles = [
|
||||
NewsArticle(
|
||||
titulo="Notícia 1",
|
||||
url="https://news.google.com/rss/articles/1",
|
||||
pagina=1,
|
||||
),
|
||||
NewsArticle(
|
||||
titulo="Notícia 2",
|
||||
url="https://news.google.com/rss/articles/2",
|
||||
pagina=1,
|
||||
),
|
||||
]
|
||||
|
||||
with patch(
|
||||
"scripts.extract_google_news.resolve_article_url",
|
||||
side_effect=lambda u: f"https://destinofinal.com/{u.split('/')[-1]}",
|
||||
):
|
||||
resolved = resolve_articles_urls(articles)
|
||||
assert len(resolved) == 2
|
||||
assert resolved[0].url == "https://destinofinal.com/1"
|
||||
assert resolved[1].url == "https://destinofinal.com/2"
|
||||
|
||||
|
||||
def test_extract_google_news_orchestration_mocked(sample_rss_xml: str):
|
||||
"""Valida a consolidação do ExtractionResult a partir da busca mockada com URLs resolvidas."""
|
||||
query = SearchQuery(keyword="inteligência artificial", language="pt", locale="BR", max_pages=1)
|
||||
|
||||
with (
|
||||
patch(
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
return_value=sample_rss_xml,
|
||||
),
|
||||
patch(
|
||||
"scripts.extract_google_news.resolve_article_url",
|
||||
side_effect=lambda u: f"https://resolved.com/{u[-5:]}",
|
||||
),
|
||||
):
|
||||
result = extract_google_news(query, resolve_urls=True)
|
||||
|
||||
assert isinstance(result, ExtractionResult)
|
||||
assert result.query == "inteligência artificial"
|
||||
assert result.language == "pt"
|
||||
assert result.locale == "BR"
|
||||
assert result.total_paginas == 1
|
||||
assert result.total_itens == 3
|
||||
assert len(result.items) == 3
|
||||
assert result.scraped_at is not None
|
||||
assert result.items[0].url.startswith("https://resolved.com/")
|
||||
|
||||
|
||||
def test_cli_execution_stdout(sample_rss_xml: str, capsys: pytest.CaptureFixture[str]):
|
||||
"""Valida execução padrão do CLI com saída JSON no stdout."""
|
||||
with (
|
||||
patch(
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
return_value=sample_rss_xml,
|
||||
),
|
||||
patch("scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u),
|
||||
):
|
||||
exit_code = main(["--query", "inteligencia artificial", "--lang", "pt", "--pretty"])
|
||||
assert exit_code == 0
|
||||
|
||||
captured = capsys.readouterr()
|
||||
data = json.loads(captured.out)
|
||||
assert data["query"] == "inteligencia artificial"
|
||||
assert data["total_itens"] == 3
|
||||
assert len(data["items"]) == 3
|
||||
# Validar indentação presente por causa de --pretty
|
||||
assert "\n " in captured.out
|
||||
|
||||
|
||||
def test_cli_execution_file_output(sample_rss_xml: str, tmp_path: Path):
|
||||
"""Valida gravação em arquivo com criação automática de diretórios pais."""
|
||||
out_file = tmp_path / "sub_dir" / "news_out.json"
|
||||
|
||||
with (
|
||||
patch(
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
return_value=sample_rss_xml,
|
||||
),
|
||||
patch("scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u),
|
||||
):
|
||||
exit_code = main(["-q", "IA", "-p", "1", "-o", str(out_file)])
|
||||
assert exit_code == 0
|
||||
|
||||
assert out_file.exists()
|
||||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert data["total_itens"] == 3
|
||||
|
||||
|
||||
def test_cli_empty_query_error(capsys: pytest.CaptureFixture[str]):
|
||||
"""Valida tratamento de erro e código de saída 1 para parâmetro vazio."""
|
||||
exit_code = main(["--query", " "])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
assert "Erro de validação" in captured.err
|
||||
|
||||
|
||||
def test_cli_network_error_handling(capsys: pytest.CaptureFixture[str]):
|
||||
"""Valida tratamento de erro e código de saída 2 para falhas de rede."""
|
||||
with patch(
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
side_effect=RuntimeError("Connection refused"),
|
||||
):
|
||||
exit_code = main(["--query", "IA"])
|
||||
assert exit_code == 2
|
||||
|
||||
captured = capsys.readouterr()
|
||||
assert "Erro na extração" in captured.err
|
||||
assert "Connection refused" in captured.err
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Testes E2E (End-to-End) com resolução real de rede e validação de URLs finais
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_e2e_resolve_real_google_news_url():
|
||||
"""Valida E2E que o decodificador resolve uma URL real do Google News para o veículo de imprensa."""
|
||||
# URL real de artigo extraída do Google News RSS
|
||||
sample_gn_url = (
|
||||
"https://news.google.com/rss/articles/"
|
||||
"CBMi6wFBVV95cUxQUS0tMGxacUJycDlCWXpWSGR3T0hfR1E0Q0txWjdqek9LWjNZTUo1WEx3"
|
||||
"UEw3Skt1eW1Wa2VYUUE2ajNYMmpEckhsdGFXUGtmQktYX2JWa1hmclhEZEVIa3hNODhpVXNO"
|
||||
"UGN4cDhmWmJMczBEUVFDNS1aX3EzRHh6VmQ3cVY0ZnZmaW9YWDZJSE9UNFJ2dXpyNFlHaVVs"
|
||||
"VlY0V1FiZ0tzZ3FpRVhUYnhPbmdnOFRveW5oOVB3WDAzS3c0eWFjMDBZSERwNmRkRk1MRHZF"
|
||||
"UE92UE9GMmpRcFZ5cUU2Ym1NeDdQU2tn?oc=5"
|
||||
)
|
||||
|
||||
resolved_url = resolve_article_url(sample_gn_url)
|
||||
|
||||
# Não deve mais ser URL do Google News
|
||||
assert "news.google.com" not in resolved_url
|
||||
# Deve ser uma URL absoluta http/https apontando para o portal real (TyC Sports)
|
||||
assert resolved_url.startswith("http")
|
||||
assert "tycsports.com" in resolved_url
|
||||
|
||||
|
||||
def test_e2e_extract_google_news_live_pipeline():
|
||||
"""Valida E2E o fluxo completo de busca, parsing e resolução de URLs reais ao vivo."""
|
||||
query = SearchQuery(keyword="tecnologia", language="pt", locale="BR", max_pages=1)
|
||||
result = extract_google_news(query, resolve_urls=True)
|
||||
|
||||
assert isinstance(result, ExtractionResult)
|
||||
assert result.total_itens > 0
|
||||
assert len(result.items) == result.total_itens
|
||||
|
||||
for article in result.items:
|
||||
assert article.titulo
|
||||
assert article.url.startswith("http")
|
||||
# Garante que as URLs foram decodificadas e não permanecem no formato intermediário
|
||||
assert "news.google.com/rss/articles/" not in article.url
|
||||
|
||||
|
||||
def test_e2e_cli_live_file_output(tmp_path: Path):
|
||||
"""Valida E2E a execução do CLI com saída real em arquivo e URLs decodificadas."""
|
||||
out_file = tmp_path / "e2e_result.json"
|
||||
exit_code = main(
|
||||
[
|
||||
"--query",
|
||||
"economia",
|
||||
"--lang",
|
||||
"pt",
|
||||
"--max-pages",
|
||||
"1",
|
||||
"--output",
|
||||
str(out_file),
|
||||
]
|
||||
)
|
||||
|
||||
assert exit_code == 0
|
||||
assert out_file.exists()
|
||||
|
||||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert data["query"] == "economia"
|
||||
assert data["total_itens"] > 0
|
||||
assert len(data["items"]) > 0
|
||||
|
||||
first_item = data["items"][0]
|
||||
assert first_item["titulo"]
|
||||
assert first_item["url"].startswith("http")
|
||||
assert "news.google.com/rss/articles/" not in first_item["url"]
|
||||
@@ -0,0 +1,61 @@
|
||||
"""Unit tests for language detection and text normalization."""
|
||||
|
||||
from src.tools.language import detect_language, normalize_text
|
||||
|
||||
|
||||
def test_normalize_text():
|
||||
assert normalize_text("São Paulo & Petróleo") == "sao paulo & petroleo"
|
||||
assert normalize_text(
|
||||
"Über große Veränderungen"
|
||||
) == "uber grosse veranderungen" or "uber" in normalize_text("Über")
|
||||
assert normalize_text("Crème brûlée") == "creme brulee"
|
||||
|
||||
|
||||
def test_detect_portuguese():
|
||||
text = "A Petrobras anunciou um novo plano de investimentos para a exploração de petróleo na camada pré-sal."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "pt"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_english():
|
||||
text = "Apple announced its new silicon chip with improved machine learning performance and battery life."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "en"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_spanish():
|
||||
text = (
|
||||
"La empresa petrolera anunció una nueva inversión en el sector energético durante este año."
|
||||
)
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "es"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_german():
|
||||
text = "Volkswagen plant eine umfassende Transformation zur Elektromobilität in den kommenden Jahren."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "de"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_italian():
|
||||
text = "La Ferrari ha presentato la nuova vettura da competizione per il campionato mondiale di Formula 1."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "it"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_french():
|
||||
text = "Le groupe TotalEnergies a confirmé ses nouveaux projets de développement dans les énergies renouvelables."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "fr"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_empty_language():
|
||||
lang, conf = detect_language("")
|
||||
assert lang == "unknown"
|
||||
assert conf == 0.0
|
||||
@@ -0,0 +1,329 @@
|
||||
"""
|
||||
Suíte de Testes para o Adaptador de Fallback para LLM (Tier 3) do Classificador de Inerência.
|
||||
|
||||
Cobre cenários unitários, de integração de pipeline, de parsing estruturado, de desambiguação
|
||||
de casos limiares e de degradação graciosa em falhas de API conforme o requisito FR-004.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from src.tools.adapters.llm import LLMFallbackAdapter
|
||||
from src.tools.classifier import InherenceClassifier
|
||||
from src.tools.models import (
|
||||
ClassificationResult,
|
||||
DecisionCategory,
|
||||
ECPSnapshot,
|
||||
)
|
||||
|
||||
SCRIPT_PATH = Path(__file__).parent.parent.parent / "classify.py"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 1. Testes Unitários do LLMFallbackAdapter
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_llm_adapter_availability_detection():
|
||||
"""Valida detecção de disponibilidade por chave de API ou provider customizado."""
|
||||
# Sem chave e sem provider
|
||||
adapter_empty = LLMFallbackAdapter(api_key="")
|
||||
assert adapter_empty.is_available() is False
|
||||
|
||||
# Com chave de API
|
||||
adapter_with_key = LLMFallbackAdapter(api_key="sk-test-key-12345")
|
||||
assert adapter_with_key.is_available() is True
|
||||
|
||||
# Com provider function
|
||||
adapter_with_fn = LLMFallbackAdapter(
|
||||
api_key="", provider_fn=lambda p: '{"decision": "DIRECT_INHERENT"}'
|
||||
)
|
||||
assert adapter_with_fn.is_available() is True
|
||||
|
||||
|
||||
def test_llm_adapter_build_prompt_structure():
|
||||
"""Valida a montagem do prompt de desambiguação com metadados do ECP e documento."""
|
||||
adapter = LLMFallbackAdapter(api_key="test")
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ecp_river",
|
||||
target_name="River Plate",
|
||||
aliases=["Club Atlético River Plate", "CARP"],
|
||||
domain="Futebol",
|
||||
anchors=["Monumental", "Libertadores"],
|
||||
)
|
||||
initial_res = ClassificationResult(
|
||||
decision=DecisionCategory.TANGENTIAL,
|
||||
is_inherent=False,
|
||||
confidence=0.40,
|
||||
detected_language="es",
|
||||
matched_anchors=["River"],
|
||||
negative_matches=[],
|
||||
graph_matches=[],
|
||||
evidence=["River"],
|
||||
rationale="Passing mention.",
|
||||
warnings=[],
|
||||
)
|
||||
|
||||
prompt = adapter.build_prompt(ecp, "# Título do Artigo\n\nConteúdo sobre o jogo.", initial_res)
|
||||
assert "River Plate" in prompt
|
||||
assert "Futebol" in prompt
|
||||
assert "TANGENTIAL" in prompt
|
||||
assert "Título do Artigo" in prompt
|
||||
|
||||
|
||||
def test_llm_adapter_parsing_valid_json_response():
|
||||
"""Valida o parsing e instanciação correta do ClassificationResult a partir da resposta do LLM."""
|
||||
adapter = LLMFallbackAdapter(
|
||||
provider_fn=lambda p: json.dumps(
|
||||
{
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": 0.95,
|
||||
"rationale": "Artigo detalha o desempenho da equipe no torneio.",
|
||||
}
|
||||
)
|
||||
)
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ecp_test",
|
||||
target_name="Test Entity",
|
||||
aliases=["Test"],
|
||||
domain="Tech",
|
||||
anchors=["cloud"],
|
||||
)
|
||||
initial = ClassificationResult(
|
||||
decision=DecisionCategory.TANGENTIAL,
|
||||
is_inherent=False,
|
||||
confidence=0.40,
|
||||
detected_language="pt",
|
||||
matched_anchors=["Test"],
|
||||
negative_matches=[],
|
||||
graph_matches=[],
|
||||
evidence=["Test"],
|
||||
rationale="Weak match.",
|
||||
warnings=["Low contextual density."],
|
||||
)
|
||||
|
||||
refined = adapter.disambiguate(ecp, "Document content...", initial)
|
||||
assert refined is not None
|
||||
assert refined.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert refined.is_inherent is True
|
||||
assert refined.confidence == 0.95
|
||||
assert "[Tier 3 LLM]" in refined.rationale
|
||||
assert "[Tier 3 LLM Override applied]" in refined.warnings
|
||||
|
||||
|
||||
def test_llm_adapter_parsing_json_wrapped_in_markdown_codeblock():
|
||||
"""Valida extração de JSON quando a resposta do LLM vem formatada em bloco markdown ```json ... ```."""
|
||||
raw_md_json = '```json\n{\n "decision": "CONTEXTUAL_INHERENT",\n "confidence": 0.88,\n "rationale": "Conexão contextual forte através da subsidiária."\n}\n```'
|
||||
adapter = LLMFallbackAdapter(provider_fn=lambda p: raw_md_json)
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ecp_test",
|
||||
target_name="Test Entity",
|
||||
aliases=["Test"],
|
||||
domain="Tech",
|
||||
anchors=["cloud"],
|
||||
)
|
||||
initial = ClassificationResult(
|
||||
decision=DecisionCategory.TANGENTIAL,
|
||||
is_inherent=False,
|
||||
confidence=0.40,
|
||||
detected_language="pt",
|
||||
matched_anchors=[],
|
||||
negative_matches=[],
|
||||
graph_matches=[],
|
||||
evidence=[],
|
||||
rationale="Weak match.",
|
||||
warnings=[],
|
||||
)
|
||||
|
||||
refined = adapter.disambiguate(ecp, "Content...", initial)
|
||||
assert refined is not None
|
||||
assert refined.decision == DecisionCategory.CONTEXTUAL_INHERENT
|
||||
assert refined.is_inherent is True
|
||||
assert refined.confidence == 0.88
|
||||
|
||||
|
||||
def test_llm_adapter_handling_invalid_and_corrupt_responses():
|
||||
"""Valida que respostas corrompidas ou JSONs sem campos obrigatórios retornem None com segurança."""
|
||||
|
||||
def make_bad_provider(resp_str: str):
|
||||
def _prov(prompt: str) -> str:
|
||||
return resp_str
|
||||
|
||||
return _prov
|
||||
|
||||
for bad_response in [
|
||||
"Desculpe, não consegui avaliar o texto.",
|
||||
"{json_invalido_sem_fechamento",
|
||||
json.dumps({"campo_desconhecido": "valor"}),
|
||||
json.dumps({"decision": "DECISAO_INEXISTENTE"}),
|
||||
]:
|
||||
adapter = LLMFallbackAdapter(provider_fn=make_bad_provider(bad_response))
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ecp_test",
|
||||
target_name="Test Entity",
|
||||
aliases=["Test"],
|
||||
domain="Tech",
|
||||
anchors=["cloud"],
|
||||
)
|
||||
initial = ClassificationResult(
|
||||
decision=DecisionCategory.TANGENTIAL,
|
||||
is_inherent=False,
|
||||
confidence=0.40,
|
||||
detected_language="pt",
|
||||
matched_anchors=[],
|
||||
negative_matches=[],
|
||||
graph_matches=[],
|
||||
evidence=[],
|
||||
rationale="Initial.",
|
||||
warnings=[],
|
||||
)
|
||||
assert adapter.disambiguate(ecp, "Content...", initial) is None
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 2. Testes de Integração de Pipeline (InherenceClassifier com Tier 3)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_classifier_triggers_tier3_on_ambiguous_tangential_case():
|
||||
"""
|
||||
Garante que o classificador dispare o Tier 3 LLM para casos ambíguos (TANGENTIAL)
|
||||
e adote o refinamento retornado.
|
||||
"""
|
||||
mock_adapter = LLMFallbackAdapter(
|
||||
provider_fn=lambda prompt: json.dumps(
|
||||
{
|
||||
"decision": "DIRECT_INHERENT",
|
||||
"confidence": 0.92,
|
||||
"rationale": "Análise profunda revelou que o texto é focado na entidade alvo.",
|
||||
}
|
||||
)
|
||||
)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=mock_adapter)
|
||||
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ecp_empresa",
|
||||
target_name="EmpresaAlfa",
|
||||
aliases=["EmpresaAlfa"],
|
||||
domain="Tecnologia",
|
||||
anchors=["software", "computação em nuvem"],
|
||||
)
|
||||
# Texto com menção única sem âncoras temáticas (Tier 1 produziria TANGENTIAL)
|
||||
ambiguous_content = "A EmpresaAlfa esteve presente no evento de encerramento anual da cidade."
|
||||
|
||||
result = classifier.classify(ecp, ambiguous_content)
|
||||
# Como enable_llm=True e o caso era TANGENTIAL, o Tier 3 substitui a decisão
|
||||
assert result.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert result.is_inherent is True
|
||||
assert result.confidence == 0.92
|
||||
assert "[Tier 3 LLM]" in result.rationale
|
||||
|
||||
|
||||
def test_classifier_skips_tier3_on_clear_direct_inherent_case():
|
||||
"""
|
||||
Garante que casos claros (alta confiança e alta densidade de âncoras) NÃO chamem
|
||||
o LLM, economizando chamadas desnecessárias conforme FR-004.
|
||||
"""
|
||||
call_tracker = {"called": False}
|
||||
|
||||
def tracking_provider(prompt: str) -> str:
|
||||
call_tracker["called"] = True
|
||||
return json.dumps({"decision": "DIRECT_INHERENT", "confidence": 0.99})
|
||||
|
||||
mock_adapter = LLMFallbackAdapter(provider_fn=tracking_provider)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=mock_adapter)
|
||||
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ecp_empresa",
|
||||
target_name="EmpresaAlfa",
|
||||
aliases=["EmpresaAlfa"],
|
||||
domain="Tecnologia",
|
||||
anchors=["software", "computação em nuvem", "inteligência artificial"],
|
||||
)
|
||||
# Caso claro com alta densidade de âncoras
|
||||
clear_content = "A EmpresaAlfa desenvolveu uma nova plataforma de software baseada em computação em nuvem e inteligência artificial."
|
||||
|
||||
result = classifier.classify(ecp, clear_content)
|
||||
assert result.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert result.confidence >= 0.85
|
||||
# O LLM NÃO deve ter sido chamado
|
||||
assert call_tracker["called"] is False
|
||||
|
||||
|
||||
def test_classifier_graceful_degradation_when_llm_raises_exception():
|
||||
"""
|
||||
Garante que se o LLM falhar por erro de rede ou timeout, o classificador mantenha
|
||||
o resultado do Tier 1 com degradação graciosa e registre o aviso em warnings.
|
||||
"""
|
||||
|
||||
def failing_provider(prompt: str) -> str:
|
||||
raise ConnectionError("Timeout ao conectar com a API do modelo de linguagem.")
|
||||
|
||||
mock_adapter = LLMFallbackAdapter(provider_fn=failing_provider)
|
||||
classifier = InherenceClassifier(enable_llm=True, llm_adapter=mock_adapter)
|
||||
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ecp_empresa",
|
||||
target_name="EmpresaAlfa",
|
||||
aliases=["EmpresaAlfa"],
|
||||
domain="Tecnologia",
|
||||
anchors=["software"],
|
||||
)
|
||||
ambiguous_content = "A EmpresaAlfa participou da conferência."
|
||||
|
||||
result = classifier.classify(ecp, ambiguous_content)
|
||||
# Retém a decisão original do Tier 1
|
||||
assert result.decision == DecisionCategory.TANGENTIAL
|
||||
assert result.is_inherent is False
|
||||
# Contém aviso sobre a falha do LLM sem quebrar a execução
|
||||
assert any("LLM fallback failed" in w for w in result.warnings)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 3. Teste de Integração CLI com a Flag --enable-llm
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_cli_execution_with_enable_llm_flag(tmp_path):
|
||||
"""Garante que a CLI classify.py aceite e processe a flag --enable-llm sem erros."""
|
||||
ecp_file = tmp_path / "test_ecp.json"
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ecp_test",
|
||||
"target_name": "TestCorp",
|
||||
"aliases": ["TestCorp"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["software", "cloud"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
content_file = tmp_path / "test_doc.md"
|
||||
content_file.write_text("# TestCorp\n\nTestCorp builds cloud software.", encoding="utf-8")
|
||||
output_file = tmp_path / "out.json"
|
||||
|
||||
res = subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
str(SCRIPT_PATH),
|
||||
"--ecp",
|
||||
str(ecp_file),
|
||||
"--content",
|
||||
str(content_file),
|
||||
"--output",
|
||||
str(output_file),
|
||||
"--enable-llm",
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
assert res.returncode == 0, f"Erro na CLI: {res.stderr}"
|
||||
assert output_file.exists()
|
||||
data = json.loads(output_file.read_text(encoding="utf-8"))
|
||||
assert data["decision"] == "DIRECT_INHERENT"
|
||||
assert data["is_inherent"] is True
|
||||
@@ -0,0 +1,114 @@
|
||||
"""Unit tests for ECP models, schema validation, and structured error handling."""
|
||||
|
||||
import pytest
|
||||
|
||||
from src.tools.models import (
|
||||
ClassificationError,
|
||||
ClassificationResult,
|
||||
DecisionCategory,
|
||||
ECPSnapshot,
|
||||
ErrorCode,
|
||||
)
|
||||
from src.tools.parser import extract_evidence_snippets, strip_markdown
|
||||
|
||||
|
||||
def test_ecp_snapshot_valid():
|
||||
data = {
|
||||
"target_entity_id": "ent_123",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petróleo Brasileiro S.A.", "Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["pré-sal", "refinaria", "combustíveis"],
|
||||
"negative_anchors": ["petrobras posto pirata"],
|
||||
"graph_version": "1.0.0",
|
||||
"related_entities": [
|
||||
{
|
||||
"entity_id": "ent_456",
|
||||
"name": "Transpetro",
|
||||
"relation_type": "SUBSIDIARY_OF",
|
||||
"weight": 0.9,
|
||||
"aliases": ["Transpetro Logística"],
|
||||
"scope": "logistics",
|
||||
"confidence": 0.95,
|
||||
}
|
||||
],
|
||||
}
|
||||
snapshot = ECPSnapshot.from_dict(data)
|
||||
assert snapshot.target_entity_id == "ent_123"
|
||||
assert snapshot.target_name == "Petrobras"
|
||||
assert len(snapshot.aliases) == 2
|
||||
assert len(snapshot.related_entities) == 1
|
||||
assert snapshot.related_entities[0].name == "Transpetro"
|
||||
assert snapshot.related_entities[0].weight == 0.9
|
||||
|
||||
|
||||
def test_ecp_snapshot_defaults():
|
||||
data = {
|
||||
"target_entity_id": "ent_123",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["energia"],
|
||||
}
|
||||
snapshot = ECPSnapshot.from_dict(data)
|
||||
assert snapshot.negative_anchors == []
|
||||
assert snapshot.graph_version == "1.0.0"
|
||||
assert snapshot.related_entities == []
|
||||
|
||||
|
||||
def test_ecp_snapshot_missing_required():
|
||||
data = {
|
||||
"target_entity_id": "ent_123",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["energia"],
|
||||
}
|
||||
with pytest.raises(ValueError, match="Missing required field"):
|
||||
ECPSnapshot.from_dict(data)
|
||||
|
||||
|
||||
def test_classification_result_serialization():
|
||||
res = ClassificationResult(
|
||||
decision=DecisionCategory.DIRECT_INHERENT,
|
||||
is_inherent=True,
|
||||
confidence=0.95,
|
||||
detected_language="pt",
|
||||
matched_anchors=["Petrobras"],
|
||||
evidence=["Petrobras anunciou investimentos no pré-sal."],
|
||||
rationale="Match forte da entidade alvo.",
|
||||
)
|
||||
d = res.to_dict()
|
||||
assert d["decision"] == "DIRECT_INHERENT"
|
||||
assert d["is_inherent"] is True
|
||||
assert d["confidence"] == 0.95
|
||||
assert d["detected_language"] == "pt"
|
||||
assert "Petrobras" in d["matched_anchors"]
|
||||
|
||||
|
||||
def test_classification_error_serialization():
|
||||
err = ClassificationError(
|
||||
error_code=ErrorCode.INVALID_ECP_JSON,
|
||||
message="Malformed JSON syntax",
|
||||
details={"path": "snapshot.json"},
|
||||
)
|
||||
d = err.to_dict()
|
||||
assert d["error_code"] == "invalid_ecp_json"
|
||||
assert d["message"] == "Malformed JSON syntax"
|
||||
assert d["details"]["path"] == "snapshot.json"
|
||||
|
||||
|
||||
def test_parser_strip_markdown():
|
||||
md = "# Title\n\nThis is **bold** text and [link](https://example.com).\n- item 1\n- item 2"
|
||||
plain = strip_markdown(md)
|
||||
assert "Title" in plain
|
||||
assert "bold text" in plain
|
||||
assert "link" in plain
|
||||
assert "[" not in plain
|
||||
assert "*" not in plain
|
||||
|
||||
|
||||
def test_extract_evidence_snippets():
|
||||
md = "O pré-sal brasileiro é uma das maiores reservas de petróleo. A Petrobras lidera a exploração técnica."
|
||||
snippets = extract_evidence_snippets(md, ["Petrobras"])
|
||||
assert len(snippets) > 0
|
||||
assert "Petrobras lidera" in snippets[0]
|
||||
@@ -0,0 +1,615 @@
|
||||
"""
|
||||
Suíte de Testes Automatizados para o Seletor Determinístico de Extrator.
|
||||
|
||||
Cobre 100% dos Casos de Teste Obrigatórios do PRD (CT-001 a CT-014), testes unitários
|
||||
de normalização e shingles, testes de integração de lote e testes E2E via subprocess.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from scripts.select_article_extractor import (
|
||||
ExtractorName,
|
||||
generate_shingles,
|
||||
normalize_text,
|
||||
process_batch,
|
||||
select_article_extractor,
|
||||
)
|
||||
|
||||
# ==============================================================================
|
||||
# Testes Unitários de Normalização e Tokenização
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_normalize_text_empty_and_invalid():
|
||||
assert normalize_text(None) == []
|
||||
assert normalize_text("") == []
|
||||
assert normalize_text(" \n\t ") == []
|
||||
assert normalize_text(12345) == []
|
||||
|
||||
|
||||
def test_normalize_text_html_entities_and_tags():
|
||||
raw = "<p>El & <b>futebol</b> mundial "está" mudando.</p>"
|
||||
tokens = normalize_text(raw)
|
||||
assert tokens == ["el", "futebol", "mundial", "está", "mudando"]
|
||||
|
||||
|
||||
def test_normalize_text_markdown_links():
|
||||
raw = "Veja mais no [Portal de Notícias](https://example.com/noticias) hoje."
|
||||
tokens = normalize_text(raw)
|
||||
assert tokens == ["veja", "mais", "no", "portal", "de", "notícias", "hoje"]
|
||||
|
||||
|
||||
def test_normalize_text_markdown_images_stripped_while_links_preserved():
|
||||
"""Garante que marcação de imagem Markdown  seja descartada e link [texto](url) seja preservado."""
|
||||
raw = (
|
||||
"Texto inicial do artigo. "
|
||||
" "
|
||||
"Mais texto com [link importante](https://example.com/pagina) e outra "
|
||||
" informação."
|
||||
)
|
||||
tokens = normalize_text(raw)
|
||||
assert "legenda" not in tokens
|
||||
assert "foto" not in tokens
|
||||
assert "imagem" not in tokens
|
||||
assert "link" in tokens
|
||||
assert "importante" in tokens
|
||||
assert tokens == [
|
||||
"texto",
|
||||
"inicial",
|
||||
"do",
|
||||
"artigo",
|
||||
"mais",
|
||||
"texto",
|
||||
"com",
|
||||
"link",
|
||||
"importante",
|
||||
"e",
|
||||
"outra",
|
||||
"informação",
|
||||
]
|
||||
|
||||
|
||||
def test_normalize_text_nfkc_unicode_and_punctuation():
|
||||
# Caracteres combinados e pontuação
|
||||
raw = "River Plate venceu por 3-0! (Com gol de pênalti & golaço de falta)."
|
||||
tokens = normalize_text(raw)
|
||||
assert tokens == [
|
||||
"river",
|
||||
"plate",
|
||||
"venceu",
|
||||
"por",
|
||||
"3",
|
||||
"0",
|
||||
"com",
|
||||
"gol",
|
||||
"de",
|
||||
"pênalti",
|
||||
"golaço",
|
||||
"de",
|
||||
"falta",
|
||||
]
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Testes Unitários de Shingles
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_generate_shingles_sliding_window():
|
||||
tokens = ["um", "dois", "três", "quatro", "cinco", "seis"]
|
||||
shingles = generate_shingles(tokens, window_size=5)
|
||||
assert len(shingles) == 2
|
||||
assert ("um", "dois", "três", "quatro", "cinco") in shingles
|
||||
assert ("dois", "três", "quatro", "cinco", "seis") in shingles
|
||||
|
||||
|
||||
def test_generate_shingles_short_text():
|
||||
# Entre 1 e 4 tokens deve gerar 1 único shingle com a tupla completa
|
||||
tokens = ["river", "plate", "campeão"]
|
||||
shingles = generate_shingles(tokens, window_size=5)
|
||||
assert len(shingles) == 1
|
||||
assert ("river", "plate", "campeão") in shingles
|
||||
|
||||
|
||||
def test_generate_shingles_empty():
|
||||
assert generate_shingles([]) == set()
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Casos de Teste Obrigatórios do PRD (§12: CT-001 a CT-014)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_ct_001_three_candidates_clear_winner():
|
||||
"""CT-001: Três candidatos com consenso e um vencedor claro -> Selecionar o maior score."""
|
||||
base = "river plate venceu o clássico ontem a noite no estádio monumental"
|
||||
article = {
|
||||
"trafilatura": {"text": base, "error": None},
|
||||
"newspaper4k": {"text": base, "error": None},
|
||||
"readability": {
|
||||
"cleaned_text": "texto completamente diferente sem nenhuma relação",
|
||||
"error": None,
|
||||
},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor in (ExtractorName.NEWSPAPER4K, ExtractorName.TRAFILATURA)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
|
||||
def test_ct_002_technical_tie_smallest_shingles():
|
||||
"""CT-002: Dois ou mais candidatos dentro de 0,03 do maior score -> Selecionar o de menor quantidade de shingles."""
|
||||
tokens_comuns = (
|
||||
"o rio de janeiro continua lindo e sempre maravilhoso em todas as estações do ano"
|
||||
)
|
||||
article = {
|
||||
"trafilatura": {"text": tokens_comuns, "error": None},
|
||||
"readability": {
|
||||
"cleaned_text": tokens_comuns + " propaganda extra adicionada no fim",
|
||||
"error": None,
|
||||
},
|
||||
"newspaper4k": {"text": tokens_comuns, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
|
||||
def test_ct_003_technical_tie_priority_fallback():
|
||||
"""CT-003: Empate técnico e mesma quantidade de shingles -> Aplicar prioridade final (newspaper4k > readability > trafilatura)."""
|
||||
texto = "o time jogou muito bem durante toda a partida de futebol"
|
||||
article = {
|
||||
"trafilatura": {"text": texto, "error": None},
|
||||
"readability": {"cleaned_text": texto, "error": None},
|
||||
"newspaper4k": {"text": texto, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
# Agora sem newspaper4k ativo (somente readability e trafilatura idênticos)
|
||||
article_two = {
|
||||
"trafilatura": {"text": texto, "error": None},
|
||||
"readability": {"cleaned_text": texto, "error": None},
|
||||
"newspaper4k": {"text": None, "error": "Crash"},
|
||||
}
|
||||
result_two = select_article_extractor(article_two)
|
||||
assert result_two.selected_extractor == ExtractorName.READABILITY
|
||||
|
||||
|
||||
def test_ct_004_three_candidates_no_consensus():
|
||||
"""CT-004: Três candidatos sem consenso -> Selecionar a quantidade mediana de shingles."""
|
||||
t1 = "alfa bravo charlie delta echo foxtrot golf hotel india juliet" # 10 tokens -> 6 shingles
|
||||
t2 = "kilo lima mike november oscar papa quebec romeo sierra tango uniform victor" # 12 tokens -> 8 shingles (MEDIANA)
|
||||
t3 = "whiskey xray yankee zulu zero one two three four five six seven eight nine" # 14 tokens -> 10 shingles
|
||||
article = {
|
||||
"trafilatura": {"text": t1, "error": None},
|
||||
"readability": {"cleaned_text": t2, "error": None},
|
||||
"newspaper4k": {"text": t3, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.READABILITY
|
||||
assert result.selection_reason == "no_consensus_median_shingles"
|
||||
|
||||
|
||||
def test_ct_005_two_candidates_no_consensus():
|
||||
"""CT-005: Dois candidatos sem consenso -> Selecionar a maior quantidade de shingles."""
|
||||
t_short = "alfa bravo charlie delta echo foxtrot" # 6 tokens -> 2 shingles
|
||||
t_long = (
|
||||
"kilo lima mike november oscar papa quebec romeo sierra tango" # 10 tokens -> 6 shingles
|
||||
)
|
||||
article = {
|
||||
"trafilatura": {"text": t_short, "error": None},
|
||||
"newspaper4k": {"text": t_long, "error": None},
|
||||
"readability": {"cleaned_text": None, "error": "Not extracted"},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
assert result.selection_reason == "no_consensus_max_shingles"
|
||||
|
||||
|
||||
def test_ct_006_single_usable_candidate():
|
||||
"""CT-006: Somente um candidato utilizável -> Selecionar esse candidato."""
|
||||
article = {
|
||||
"trafilatura": {
|
||||
"text": "conteúdo válido e utilizável extraído com sucesso aqui",
|
||||
"error": None,
|
||||
},
|
||||
"newspaper4k": {"text": "", "error": None},
|
||||
"readability": {"cleaned_text": None, "error": "Timeout error"},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.TRAFILATURA
|
||||
assert result.selection_reason == "single_usable_candidate"
|
||||
|
||||
|
||||
def test_ct_007_degraded_candidates_only():
|
||||
"""CT-007: Nenhum utilizável, mas existe candidato degradado -> Executar o algoritmo somente com os degradados."""
|
||||
texto_comum = "artigo relevante sobre economia global e finanças internacionais com detalhes"
|
||||
article = {
|
||||
"trafilatura": {"text": texto_comum, "error": "Warning: partial parse"},
|
||||
"newspaper4k": {"text": texto_comum, "error": "HTTP 403 partial"},
|
||||
"readability": {"cleaned_text": None, "error": "Fatal exception"},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
|
||||
def test_ct_008_all_candidates_unavailable():
|
||||
"""CT-008: Todos os candidatos indisponíveis -> Selecionar newspaper4k."""
|
||||
article = {
|
||||
"trafilatura": {"text": None, "error": "Error"},
|
||||
"newspaper4k": {"text": "", "error": "Empty"},
|
||||
"readability": {"cleaned_text": " ", "error": "Blank"},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
assert result.selection_reason == "fallback_all_unavailable"
|
||||
|
||||
|
||||
def test_ct_009_small_fragment_loses_due_to_low_coverage():
|
||||
"""CT-009: Readability retorna apenas um fragmento pequeno enquanto os outros concordam -> Perde por baixa cobertura."""
|
||||
full_text = (
|
||||
"o presidente da república anunciou novas medidas econômicas para conter a inflação "
|
||||
"e estimular o crescimento industrial em todo o território nacional durante o pronunciamento oficial"
|
||||
)
|
||||
small_fragment = "o presidente da república anunciou"
|
||||
article = {
|
||||
"trafilatura": {"text": full_text, "error": None},
|
||||
"newspaper4k": {"text": full_text, "error": None},
|
||||
"readability": {"cleaned_text": small_fragment, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor in (ExtractorName.TRAFILATURA, ExtractorName.NEWSPAPER4K)
|
||||
assert result.selected_extractor != ExtractorName.READABILITY
|
||||
|
||||
|
||||
def test_ct_010_excessive_boilerplate_loses_due_to_low_support():
|
||||
"""CT-010: Um candidato contém o conteúdo comum e muito conteúdo excedente -> Perde suporte e reduz pontuação."""
|
||||
common_content = (
|
||||
"notícia oficial com dados apurados sobre a operação policial realizada nesta manhã"
|
||||
)
|
||||
massive_boilerplate = common_content + (
|
||||
" compartilhe no facebook twitter whatsapp veja também esportes receitas horóscopo política "
|
||||
" e assine nossa newsletter diária para receber mais novidades sobre culinária e fofocas"
|
||||
)
|
||||
article = {
|
||||
"trafilatura": {"text": common_content, "error": None},
|
||||
"readability": {"cleaned_text": common_content, "error": None},
|
||||
"newspaper4k": {"text": massive_boilerplate, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.READABILITY
|
||||
|
||||
|
||||
def test_ct_011_partial_content_loses_due_to_low_coverage():
|
||||
"""CT-011: Um candidato contém somente parte do conteúdo comum -> Perde cobertura e reduz pontuação."""
|
||||
full_content = "primeiro parágrafo do artigo completo segundo parágrafo com explicações terceiro parágrafo final"
|
||||
half_content = "primeiro parágrafo do artigo completo"
|
||||
article = {
|
||||
"newspaper4k": {"text": full_content, "error": None},
|
||||
"readability": {"cleaned_text": full_content, "error": None},
|
||||
"trafilatura": {"text": half_content, "error": None},
|
||||
}
|
||||
result = select_article_extractor(article)
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
|
||||
|
||||
def test_ct_012_recalculate_existing_selected_extractor(tmp_path: Path):
|
||||
"""CT-012: A entrada já contém selected_extractor -> Recalcular e substituir somente essa chave."""
|
||||
input_data = {
|
||||
"articles": [
|
||||
{
|
||||
"titulo": "Teste",
|
||||
"selected_extractor": "trafilatura",
|
||||
"trafilatura": {"text": "lixo sem sentido", "error": None},
|
||||
"newspaper4k": {
|
||||
"text": "conteúdo correto compartilhado por dois motores",
|
||||
"error": None,
|
||||
},
|
||||
"readability": {
|
||||
"cleaned_text": "conteúdo correto compartilhado por dois motores",
|
||||
"error": None,
|
||||
},
|
||||
}
|
||||
]
|
||||
}
|
||||
in_file = tmp_path / "artigos.json"
|
||||
in_file.write_text(json.dumps(input_data, ensure_ascii=False), encoding="utf-8")
|
||||
|
||||
res = process_batch(in_file)
|
||||
out_file = Path(res.output_file)
|
||||
assert out_file.exists()
|
||||
|
||||
with open(out_file, "r", encoding="utf-8") as f:
|
||||
out_data = json.load(f)
|
||||
|
||||
assert out_data["articles"][0]["selected_extractor"] == "newspaper4k"
|
||||
|
||||
|
||||
def test_ct_013_empty_articles_list(tmp_path: Path):
|
||||
"""CT-013: articles está vazio -> Gerar saída válida com articles vazio."""
|
||||
input_data = {"metadata": "info", "articles": []}
|
||||
in_file = tmp_path / "empty.json"
|
||||
in_file.write_text(json.dumps(input_data), encoding="utf-8")
|
||||
|
||||
res = process_batch(in_file)
|
||||
out_file = Path(res.output_file)
|
||||
assert out_file.exists()
|
||||
|
||||
with open(out_file, "r", encoding="utf-8") as f:
|
||||
out_data = json.load(f)
|
||||
|
||||
assert out_data["metadata"] == "info"
|
||||
assert out_data["articles"] == []
|
||||
assert res.total_articles == 0
|
||||
assert res.processed_count == 0
|
||||
|
||||
|
||||
def test_ct_014_invalid_json_fails_atomically(tmp_path: Path):
|
||||
"""CT-014: JSON inválido -> Não gerar saída."""
|
||||
in_file = tmp_path / "invalid.json"
|
||||
in_file.write_text("{articles: [ malformed json", encoding="utf-8")
|
||||
expected_out = tmp_path / "invalid_selected.json"
|
||||
|
||||
with pytest.raises(ValueError, match="JSON inválido"):
|
||||
process_batch(in_file)
|
||||
|
||||
assert not expected_out.exists()
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Testes de Integração
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_integration_reference_file(tmp_path: Path):
|
||||
"""Testa o processamento em lote completo sobre o arquivo real out/river_plate_extracted.json."""
|
||||
ref_file = Path("out/river_plate_extracted.json")
|
||||
if not ref_file.exists():
|
||||
pytest.skip(
|
||||
"Arquivo out/river_plate_extracted.json não encontrado para teste de integração."
|
||||
)
|
||||
|
||||
out_file = tmp_path / "river_plate_extracted_selected.json"
|
||||
result = process_batch(ref_file, output_path=out_file, verbose=True)
|
||||
|
||||
assert result.total_articles == 20
|
||||
assert result.processed_count == 20
|
||||
assert out_file.exists()
|
||||
|
||||
with open(out_file, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
assert len(data["articles"]) == 20
|
||||
for art in data["articles"]:
|
||||
assert "selected_extractor" in art
|
||||
assert art["selected_extractor"] in ["trafilatura", "newspaper4k", "readability"]
|
||||
|
||||
# Validar a distribuição exata conforme o algoritmo do PRD
|
||||
assert result.selection_distribution == {
|
||||
"newspaper4k": 9,
|
||||
"readability": 9,
|
||||
"trafilatura": 2,
|
||||
}
|
||||
|
||||
|
||||
def test_article_1_regression_technical_tie_markdown_images():
|
||||
"""
|
||||
Teste de regressão para o Artigo 1 (Los puntajes de River vs. Independiente Santa Fe):
|
||||
Valida que com o descarte de imagens Markdown , Readability e Newspaper4k
|
||||
entram em empate técnico (diff <= 0.03) e Newspaper4k vence por possuir menor quantidade
|
||||
de shingles (1009 vs 1057).
|
||||
"""
|
||||
ref_file = Path("out/river_plate_extracted.json")
|
||||
if not ref_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
|
||||
|
||||
with open(ref_file, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
art1 = data["articles"][0]
|
||||
result = select_article_extractor(art1, article_index=0)
|
||||
|
||||
assert result.selected_extractor == ExtractorName.NEWSPAPER4K
|
||||
assert result.selection_reason == "technical_tie_smallest_shingles"
|
||||
|
||||
cand_news = result.candidates[ExtractorName.NEWSPAPER4K]
|
||||
cand_read = result.candidates[ExtractorName.READABILITY]
|
||||
cand_traf = result.candidates[ExtractorName.TRAFILATURA]
|
||||
|
||||
assert cand_news.shingle_count == 1009
|
||||
assert cand_read.shingle_count == 1057
|
||||
assert cand_traf.shingle_count == 1091
|
||||
|
||||
# Diferença para o maior score <= 0.03 (empate técnico)
|
||||
max_score = max(cand_news.score, cand_read.score, cand_traf.score)
|
||||
assert (max_score - cand_news.score) <= 0.03
|
||||
assert (max_score - cand_read.score) <= 0.03
|
||||
|
||||
|
||||
def test_integration_large_batch_determinism(tmp_path: Path):
|
||||
"""Gera um lote de 100 artigos sintéticos e verifica 100% de repetibilidade determinística entre 2 execuções."""
|
||||
articles = []
|
||||
for i in range(100):
|
||||
if i % 4 == 0:
|
||||
art = {
|
||||
"trafilatura": {
|
||||
"text": f"artigo numero {i} sobre futebol internacional no estadio",
|
||||
"error": None,
|
||||
},
|
||||
"newspaper4k": {
|
||||
"text": f"artigo numero {i} sobre futebol internacional no estadio",
|
||||
"error": None,
|
||||
},
|
||||
"readability": {"cleaned_text": "sem relacao", "error": None},
|
||||
}
|
||||
elif i % 4 == 1:
|
||||
art = {
|
||||
"trafilatura": {"text": None, "error": "timeout"},
|
||||
"newspaper4k": {"text": f"noticia exclusiva {i} com detalhes", "error": None},
|
||||
"readability": {
|
||||
"cleaned_text": f"noticia exclusiva {i} com detalhes",
|
||||
"error": None,
|
||||
},
|
||||
}
|
||||
elif i % 4 == 2:
|
||||
art = {
|
||||
"trafilatura": {"text": f"texto a {i}", "error": None},
|
||||
"newspaper4k": {"text": f"texto b diferente {i}", "error": None},
|
||||
"readability": {"cleaned_text": f"texto c terceiro {i}", "error": None},
|
||||
}
|
||||
else:
|
||||
art = {
|
||||
"trafilatura": {"text": None, "error": "err"},
|
||||
"newspaper4k": {"text": "", "error": "err"},
|
||||
"readability": {"cleaned_text": None, "error": "err"},
|
||||
}
|
||||
articles.append(art)
|
||||
|
||||
batch_payload = {"articles": articles}
|
||||
in_file = tmp_path / "large_batch.json"
|
||||
in_file.write_text(json.dumps(batch_payload, ensure_ascii=False), encoding="utf-8")
|
||||
|
||||
out1 = tmp_path / "large_batch_run1.json"
|
||||
out2 = tmp_path / "large_batch_run2.json"
|
||||
|
||||
res1 = process_batch(in_file, output_path=out1)
|
||||
res2 = process_batch(in_file, output_path=out2)
|
||||
|
||||
assert res1.processed_count == 100
|
||||
assert res2.processed_count == 100
|
||||
|
||||
# Os resultados devem ser 100% idênticos
|
||||
selections1 = [s.selected_extractor for s in res1.selections]
|
||||
selections2 = [s.selected_extractor for s in res2.selections]
|
||||
assert selections1 == selections2
|
||||
|
||||
|
||||
def test_integration_pipeline_downstream_consumer(tmp_path: Path):
|
||||
"""
|
||||
Testa a integração end-to-end do pipeline downstream:
|
||||
Lê o JSON enriquecido com selected_extractor, recupera o conteúdo do extrator vencedor e
|
||||
garante que o texto está higienizado e pronto para os classificadores NLP.
|
||||
"""
|
||||
ref_file = Path("out/river_plate_extracted.json")
|
||||
if not ref_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado.")
|
||||
|
||||
out_file = tmp_path / "downstream_test.json"
|
||||
process_batch(ref_file, output_path=out_file)
|
||||
|
||||
with open(out_file, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
for idx, art in enumerate(data["articles"]):
|
||||
winner = art["selected_extractor"]
|
||||
assert winner in ["trafilatura", "newspaper4k", "readability"]
|
||||
|
||||
# Recuperar texto do extrator vencedor conforme mapeamento do PRD
|
||||
if winner == "trafilatura":
|
||||
chosen_text = art["trafilatura"]["text"]
|
||||
elif winner == "newspaper4k":
|
||||
chosen_text = art["newspaper4k"]["text"]
|
||||
elif winner == "readability":
|
||||
chosen_text = art["readability"]["cleaned_text"]
|
||||
|
||||
assert isinstance(chosen_text, str)
|
||||
assert len(chosen_text.strip()) > 0
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Testes End-to-End (E2E) via Subprocess CLI
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_e2e_cli_subprocess_real_execution(tmp_path: Path):
|
||||
"""E2E: Executa scripts/select_article_extractor.py como subprocesso real na linha de comando."""
|
||||
ref_file = Path("out/river_plate_extracted.json")
|
||||
if not ref_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate_extracted.json não encontrado para teste E2E.")
|
||||
|
||||
out_file = tmp_path / "e2e_river_plate_selected.json"
|
||||
script_path = Path("scripts/select_article_extractor.py").resolve()
|
||||
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(script_path),
|
||||
str(ref_file.resolve()),
|
||||
"-o",
|
||||
str(out_file),
|
||||
"--verbose",
|
||||
"--indent",
|
||||
"2",
|
||||
]
|
||||
|
||||
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||
|
||||
assert proc.returncode == 0
|
||||
assert out_file.exists()
|
||||
|
||||
# Validar saída estruturada JSON do stdout
|
||||
summary = json.loads(proc.stdout)
|
||||
assert summary["status"] == "success"
|
||||
assert summary["total_articles"] == 20
|
||||
assert summary["processed_count"] == 20
|
||||
assert "distribution" in summary
|
||||
|
||||
# Validar que logs de verbose foram emitidos no stderr
|
||||
assert "[Artigo #001]" in proc.stderr
|
||||
assert "[Artigo #020]" in proc.stderr
|
||||
|
||||
|
||||
def test_e2e_cli_subprocess_default_naming(tmp_path: Path):
|
||||
"""E2E: Executa CLI sem a flag -o e valida criação automática de <nome>_selected.json."""
|
||||
sample_data = {
|
||||
"articles": [
|
||||
{
|
||||
"titulo": "Artigo Automático",
|
||||
"trafilatura": {"text": "conteúdo padrão", "error": None},
|
||||
"newspaper4k": {"text": "conteúdo padrão", "error": None},
|
||||
"readability": {"cleaned_text": "conteúdo padrão", "error": None},
|
||||
}
|
||||
]
|
||||
}
|
||||
in_file = tmp_path / "my_news.json"
|
||||
in_file.write_text(json.dumps(sample_data), encoding="utf-8")
|
||||
expected_out = tmp_path / "my_news_selected.json"
|
||||
|
||||
script_path = Path("scripts/select_article_extractor.py").resolve()
|
||||
cmd = [sys.executable, str(script_path), str(in_file)]
|
||||
|
||||
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||
assert proc.returncode == 0
|
||||
assert expected_out.exists()
|
||||
|
||||
with open(expected_out, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
assert data["articles"][0]["selected_extractor"] == "newspaper4k"
|
||||
|
||||
|
||||
def test_e2e_cli_subprocess_invalid_input(tmp_path: Path):
|
||||
"""E2E: Executa CLI com JSON inválido e valida código de saída e erro no stderr."""
|
||||
invalid_file = tmp_path / "broken.json"
|
||||
invalid_file.write_text("not a valid json {", encoding="utf-8")
|
||||
|
||||
script_path = Path("scripts/select_article_extractor.py").resolve()
|
||||
cmd = [sys.executable, str(script_path), str(invalid_file)]
|
||||
|
||||
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||
assert proc.returncode == 2
|
||||
assert "ERRO DE VALIDAÇÃO" in proc.stderr
|
||||
|
||||
|
||||
def test_e2e_cli_subprocess_missing_file():
|
||||
"""E2E: Executa CLI com arquivo inexistente e valida código 1 e mensagem no stderr."""
|
||||
script_path = Path("scripts/select_article_extractor.py").resolve()
|
||||
cmd = [sys.executable, str(script_path), "non_existent_file_12345.json"]
|
||||
|
||||
proc = subprocess.run(cmd, capture_output=True, text=True, check=False)
|
||||
assert proc.returncode == 1
|
||||
assert "ERRO DE ARQUIVO" in proc.stderr
|
||||
Reference in New Issue
Block a user