feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
@@ -27,7 +27,7 @@ def test_classifier_with_adapter_flags():
|
||||
target_name="TestCorp",
|
||||
aliases=["TestCorp"],
|
||||
domain="Tech",
|
||||
anchors=["software"]
|
||||
anchors=["software"],
|
||||
)
|
||||
res = classifier.classify(ecp, "TestCorp builds enterprise cloud software.")
|
||||
assert res.decision.value == "DIRECT_INHERENT"
|
||||
|
||||
+57
-29
@@ -7,9 +7,8 @@ and CLI execution behavior via subprocess (exit codes, stream purity, JSON parsi
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
import pytest
|
||||
from src.models import ECPSnapshot, RelatedEntity, DecisionCategory
|
||||
|
||||
from src.models import DecisionCategory, ECPSnapshot, RelatedEntity
|
||||
|
||||
|
||||
def test_adversarial_sao_paulo_city_vs_fc():
|
||||
@@ -20,9 +19,13 @@ def test_adversarial_sao_paulo_city_vs_fc():
|
||||
aliases=["São Paulo", "SPFC", "Tricolor Paulista"],
|
||||
domain="Futebol e Esportes",
|
||||
anchors=["Morumbi", "futebol", "campeonato", "Copa Libertadores", "elenco", "estádio"],
|
||||
negative_anchors=["prefeitura de são paulo", "governo do estado de são paulo", "trânsito na capital paulista"],
|
||||
negative_anchors=[
|
||||
"prefeitura de são paulo",
|
||||
"governo do estado de são paulo",
|
||||
"trânsito na capital paulista",
|
||||
],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[]
|
||||
related_entities=[],
|
||||
)
|
||||
content = (
|
||||
"# Obras Viárias na Capital\n\n"
|
||||
@@ -30,6 +33,7 @@ def test_adversarial_sao_paulo_city_vs_fc():
|
||||
"para desafogar o fluxo de veículos na região central durante os horários de pico."
|
||||
)
|
||||
from src.classifier import InherenceClassifier
|
||||
|
||||
classifier = InherenceClassifier()
|
||||
result = classifier.classify(ecp, content)
|
||||
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
|
||||
@@ -47,13 +51,14 @@ def test_adversarial_apple_fruit_recipe():
|
||||
anchors=["iPhone", "MacBook", "iOS", "silicon", "hardware"],
|
||||
negative_anchors=["apple pie", "orchard harvest", "doce de maçã"],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[]
|
||||
related_entities=[],
|
||||
)
|
||||
content = (
|
||||
"# Receita Caseira\n\n"
|
||||
"Comprei maçãs frescas no mercado para preparar um doce de maçã com canela e açúcar mascavo."
|
||||
)
|
||||
from src.classifier import InherenceClassifier
|
||||
|
||||
classifier = InherenceClassifier()
|
||||
result = classifier.classify(ecp, content)
|
||||
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
|
||||
@@ -76,9 +81,9 @@ def test_adversarial_related_entity_without_scope_context():
|
||||
relation_type="SUPPLIER_OF",
|
||||
weight=0.95,
|
||||
scope="battery_technology",
|
||||
confidence=0.99
|
||||
confidence=0.99,
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
# Content mentions Northvolt in an unrelated/passing architectural context without domain anchors
|
||||
content = (
|
||||
@@ -87,6 +92,7 @@ def test_adversarial_related_entity_without_scope_context():
|
||||
"mit moderner Holzfassade und Blick auf den See."
|
||||
)
|
||||
from src.classifier import InherenceClassifier
|
||||
|
||||
classifier = InherenceClassifier()
|
||||
result = classifier.classify(ecp, content)
|
||||
# Must be TANGENTIAL or NOT_RELATED, NEVER CONTEXTUAL_INHERENT
|
||||
@@ -98,18 +104,27 @@ def test_adversarial_related_entity_without_scope_context():
|
||||
def test_adversarial_subprocess_cli_success_stdout(tmp_path):
|
||||
"""Run CLI via subprocess without --output and verify stdout is pure parseable JSON."""
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_petrobras",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"]
|
||||
}), encoding="utf-8")
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_petrobras",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.", encoding="utf-8")
|
||||
content_file.write_text(
|
||||
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
import os
|
||||
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
res = subprocess.run(
|
||||
@@ -133,16 +148,22 @@ def test_adversarial_subprocess_cli_success_stdout(tmp_path):
|
||||
def test_adversarial_subprocess_cli_empty_content(tmp_path):
|
||||
"""Run CLI via subprocess with empty content and verify error code and exit code."""
|
||||
import os
|
||||
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Test",
|
||||
"aliases": ["Test"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["tech"]
|
||||
}), encoding="utf-8")
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Test",
|
||||
"aliases": ["Test"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["tech"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "empty.md"
|
||||
content_file.write_text(" \n\n ", encoding="utf-8")
|
||||
@@ -164,15 +185,21 @@ def test_adversarial_subprocess_cli_empty_content(tmp_path):
|
||||
def test_adversarial_subprocess_cli_missing_required_field(tmp_path):
|
||||
"""Run CLI via subprocess with missing target_name and verify error payload."""
|
||||
import os
|
||||
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
ecp_file = tmp_path / "ecp_bad.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"aliases": ["Test"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["tech"]
|
||||
}), encoding="utf-8")
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"aliases": ["Test"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["tech"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Conteúdo de teste válido.", encoding="utf-8")
|
||||
@@ -193,6 +220,7 @@ def test_adversarial_subprocess_cli_missing_required_field(tmp_path):
|
||||
def test_adversarial_subprocess_cli_corrupted_json(tmp_path):
|
||||
"""Run CLI via subprocess with corrupted JSON and verify error payload."""
|
||||
import os
|
||||
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
ecp_file = tmp_path / "ecp_corrupted.json"
|
||||
|
||||
+19
-21
@@ -6,19 +6,17 @@ Target Success Criterion: Precision >= 90% over the 24 cases.
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from src.models import ECPSnapshot, DecisionCategory
|
||||
|
||||
from src.classifier import InherenceClassifier
|
||||
from src.models import ECPSnapshot
|
||||
|
||||
FIXTURES_DIR = Path(__file__).parent / "fixtures" / "benchmark_24"
|
||||
LANGUAGES = ["pt", "en", "es", "de", "it", "fr"]
|
||||
DECISION_TYPES = ["direct", "contextual", "tangential", "not_related"]
|
||||
|
||||
BENCHMARK_CASES = [
|
||||
(lang, dec_type)
|
||||
for lang in LANGUAGES
|
||||
for dec_type in DECISION_TYPES
|
||||
]
|
||||
BENCHMARK_CASES = [(lang, dec_type) for lang in LANGUAGES for dec_type in DECISION_TYPES]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
@@ -44,28 +42,28 @@ def test_benchmark_case(classifier, lang: str, dec_type: str):
|
||||
result = classifier.classify(ecp, content)
|
||||
|
||||
# 1. Decision category validation
|
||||
assert (
|
||||
result.decision.value == expected["expected_decision"]
|
||||
), f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
|
||||
assert result.decision.value == expected["expected_decision"], (
|
||||
f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
|
||||
)
|
||||
|
||||
# 2. Derived is_inherent boolean validation
|
||||
assert (
|
||||
result.is_inherent == expected["expected_is_inherent"]
|
||||
), f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
|
||||
assert result.is_inherent == expected["expected_is_inherent"], (
|
||||
f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
|
||||
)
|
||||
|
||||
# 3. Language detection validation
|
||||
assert (
|
||||
result.detected_language == expected["expected_language"]
|
||||
), f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
|
||||
assert result.detected_language == expected["expected_language"], (
|
||||
f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
|
||||
)
|
||||
|
||||
# 4. Confidence threshold validation
|
||||
min_conf = expected.get("min_confidence", 0.0)
|
||||
assert (
|
||||
result.confidence >= min_conf
|
||||
), f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
|
||||
assert result.confidence >= min_conf, (
|
||||
f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
|
||||
)
|
||||
|
||||
# 5. Evidence presence for inherent content
|
||||
if result.is_inherent:
|
||||
assert (
|
||||
len(result.evidence) > 0
|
||||
), f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
|
||||
assert len(result.evidence) > 0, (
|
||||
f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
|
||||
)
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
"""Unit tests for deterministic classification decision logic."""
|
||||
|
||||
import pytest
|
||||
from src.models import ECPSnapshot, RelatedEntity, DecisionCategory
|
||||
|
||||
from src.classifier import InherenceClassifier
|
||||
from src.models import DecisionCategory, ECPSnapshot, RelatedEntity
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
@@ -23,9 +24,9 @@ def petrobras_ecp():
|
||||
weight=0.85,
|
||||
aliases=[],
|
||||
scope="logistics",
|
||||
confidence=1.0
|
||||
confidence=1.0,
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
|
||||
+52
-31
@@ -1,23 +1,30 @@
|
||||
"""CLI execution tests covering flags, arguments, stdout, and error handling."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
import pytest
|
||||
|
||||
from classify import main
|
||||
|
||||
|
||||
def test_cli_success_stdout(tmp_path, capsys):
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"]
|
||||
}), encoding="utf-8")
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.", encoding="utf-8")
|
||||
content_file.write_text(
|
||||
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
|
||||
assert exit_code == 0
|
||||
@@ -31,20 +38,30 @@ def test_cli_success_stdout(tmp_path, capsys):
|
||||
|
||||
def test_cli_output_file(tmp_path):
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"]
|
||||
}), encoding="utf-8")
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Petrobras anunciou investimentos bilionários em novas refinarias de petróleo.", encoding="utf-8")
|
||||
content_file.write_text(
|
||||
"Petrobras anunciou investimentos bilionários em novas refinarias de petróleo.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
output_file = tmp_path / "out" / "result.json"
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file), "--output", str(output_file)])
|
||||
exit_code = main(
|
||||
["--ecp", str(ecp_file), "--content", str(content_file), "--output", str(output_file)]
|
||||
)
|
||||
assert exit_code == 0
|
||||
assert output_file.exists()
|
||||
|
||||
@@ -67,13 +84,18 @@ def test_cli_missing_ecp_file(tmp_path, capsys):
|
||||
|
||||
def test_cli_empty_content_file(tmp_path, capsys):
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo"]
|
||||
}), encoding="utf-8")
|
||||
ecp_file.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo"],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "empty.md"
|
||||
content_file.write_text(" \n\n ", encoding="utf-8")
|
||||
@@ -88,11 +110,10 @@ def test_cli_empty_content_file(tmp_path, capsys):
|
||||
|
||||
def test_cli_missing_required_ecp_field(tmp_path, capsys):
|
||||
ecp_file = tmp_path / "bad_ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo"]
|
||||
}), encoding="utf-8")
|
||||
ecp_file.write_text(
|
||||
json.dumps({"target_entity_id": "ent_1", "domain": "Oil & Gas", "anchors": ["petróleo"]}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Algum conteúdo válido para testar o erro.", encoding="utf-8")
|
||||
|
||||
@@ -0,0 +1,372 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Testes automatizados para o Extrator e Parser Multimotor de Artigos.
|
||||
Cobre modelos de dados, parsers (Trafilatura, Newspaper4k, Readability),
|
||||
isolamento de falhas, orquestração de lote e interface CLI.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from scripts.extract_article_contents import (
|
||||
ArticleCrawler,
|
||||
ExtractedArticle,
|
||||
ExtractionBatchReport,
|
||||
InputArticle,
|
||||
NewspaperData,
|
||||
NewspaperExtractor,
|
||||
ReadabilityData,
|
||||
ReadabilityExtractor,
|
||||
TrafilaturaData,
|
||||
TrafilaturaExtractor,
|
||||
extract_all_engines,
|
||||
load_search_json,
|
||||
main,
|
||||
process_batch,
|
||||
save_extracted_json,
|
||||
)
|
||||
|
||||
SAMPLE_HTML = """
|
||||
<!DOCTYPE html>
|
||||
<html lang="es">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<title>River Plate igualó sin goles ante Independiente Santa Fe - Olé</title>
|
||||
<meta name="description" content="El equipo de Núñez empató 0-0 en Bogotá por los octavos de final.">
|
||||
<meta name="author" content="Juan Pérez">
|
||||
<meta property="og:title" content="River Plate igualó sin goles ante Independiente Santa Fe">
|
||||
<meta property="og:image" content="https://media.ole.com.ar/river.jpg">
|
||||
</head>
|
||||
<body>
|
||||
<header><nav><a href="/">Inicio</a></nav></header>
|
||||
<article>
|
||||
<h1>River Plate igualó sin goles ante Independiente Santa Fe</h1>
|
||||
<p class="byline">Por Juan Pérez - 20 de Agosto de 2026</p>
|
||||
<p class="lead">El equipo de Núñez empató 0-0 en Bogotá por la Copa Sudamericana.</p>
|
||||
<p>Franco Armani fue la gran figura del encuentro con tres atajadas espectaculares en el primer tiempo.</p>
|
||||
<p>El partido de vuelta se disputará en el estadio Monumental la próxima semana ante una multitud.</p>
|
||||
</article>
|
||||
<footer><p>Copyright 2026 Olé</p></footer>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_input_json(tmp_path: Path) -> Path:
|
||||
data = {
|
||||
"query": "River Plate",
|
||||
"language": "es",
|
||||
"locale": "AR",
|
||||
"total_itens": 2,
|
||||
"items": [
|
||||
{
|
||||
"titulo": "River Plate igualó sin goles ante Santa Fe",
|
||||
"subtitulo": "Empate en Bogotá",
|
||||
"quando_publicado": "Thu, 20 Aug 2026 03:27:26 GMT",
|
||||
"url": "https://www.ole.com.ar/river-0-0-santa-fe.html",
|
||||
"pagina": 1,
|
||||
},
|
||||
{
|
||||
"titulo": "Armani fue la figura de River",
|
||||
"subtitulo": "Gran actuación del arquero",
|
||||
"quando_publicado": "Thu, 20 Aug 2026 04:00:00 GMT",
|
||||
"url": "https://www.tycsports.com/armani-figura.html",
|
||||
"pagina": 1,
|
||||
},
|
||||
],
|
||||
}
|
||||
input_file = tmp_path / "river_plate.json"
|
||||
input_file.write_text(json.dumps(data, ensure_ascii=False), encoding="utf-8")
|
||||
return input_file
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 1. Testes de Modelos e I/O de JSON
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_input_article_creation():
|
||||
article = InputArticle(
|
||||
titulo="Notícia Teste",
|
||||
url="https://example.com/noticia",
|
||||
subtitulo="Subtítulo",
|
||||
quando_publicado="Thu, 20 Aug 2026",
|
||||
pagina=1,
|
||||
)
|
||||
assert article.titulo == "Notícia Teste"
|
||||
assert article.url == "https://example.com/noticia"
|
||||
assert article.pagina == 1
|
||||
d = article.to_dict()
|
||||
assert d["titulo"] == "Notícia Teste"
|
||||
assert d["url"] == "https://example.com/noticia"
|
||||
|
||||
|
||||
def test_load_search_json_valid(sample_input_json: Path):
|
||||
query, lang, items = load_search_json(sample_input_json)
|
||||
assert query == "River Plate"
|
||||
assert lang == "es"
|
||||
assert len(items) == 2
|
||||
assert items[0].url == "https://www.ole.com.ar/river-0-0-santa-fe.html"
|
||||
|
||||
|
||||
def test_load_search_json_invalid_file(tmp_path: Path):
|
||||
non_existent = tmp_path / "missing.json"
|
||||
with pytest.raises(FileNotFoundError):
|
||||
load_search_json(non_existent)
|
||||
|
||||
|
||||
def test_save_extracted_json(tmp_path: Path):
|
||||
report = ExtractionBatchReport(
|
||||
source_file="test.json",
|
||||
processed_at="2026-08-20T12:00:00Z",
|
||||
total_articles=1,
|
||||
successful_articles=1,
|
||||
failed_articles=0,
|
||||
articles=[
|
||||
ExtractedArticle(
|
||||
input_meta=InputArticle(titulo="Teste", url="https://example.com/noticia"),
|
||||
extraction_status="success",
|
||||
error_message=None,
|
||||
crawled_url="https://example.com/noticia",
|
||||
page_title="Página Teste",
|
||||
http_status=200,
|
||||
trafilatura=TrafilaturaData(
|
||||
title="Teste",
|
||||
author="Autor",
|
||||
date="2026-08-20",
|
||||
description="Desc",
|
||||
categories=[],
|
||||
tags=[],
|
||||
canonical_url=None,
|
||||
text="Texto do teste.",
|
||||
raw_json=None,
|
||||
error=None,
|
||||
),
|
||||
newspaper4k=NewspaperData(
|
||||
title="Teste",
|
||||
authors=["Autor"],
|
||||
publish_date="2026-08-20",
|
||||
text="Texto do teste.",
|
||||
summary="Resumo",
|
||||
keywords=["teste"],
|
||||
top_image=None,
|
||||
images=[],
|
||||
meta_data={},
|
||||
error=None,
|
||||
),
|
||||
readability=ReadabilityData(
|
||||
title="Teste",
|
||||
short_title="Teste",
|
||||
cleaned_html="<p>Texto do teste.</p>",
|
||||
cleaned_text="Texto do teste.",
|
||||
error=None,
|
||||
),
|
||||
)
|
||||
],
|
||||
)
|
||||
out_file = tmp_path / "out" / "result.json"
|
||||
save_extracted_json(report, out_file)
|
||||
assert out_file.exists()
|
||||
content = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert content["total_articles"] == 1
|
||||
assert content["articles"][0]["trafilatura"]["title"] == "Teste"
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 2. Testes Unitários dos Parsers (Trafilatura, Newspaper4k, Readability)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_trafilatura_extractor():
|
||||
extractor = TrafilaturaExtractor()
|
||||
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia")
|
||||
assert isinstance(res, TrafilaturaData)
|
||||
assert res.error is None
|
||||
assert "Armani" in res.text or "River Plate" in res.text
|
||||
assert res.title is not None
|
||||
|
||||
|
||||
def test_newspaper_extractor():
|
||||
extractor = NewspaperExtractor()
|
||||
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es")
|
||||
assert isinstance(res, NewspaperData)
|
||||
assert res.error is None
|
||||
assert "Armani" in res.text or "River" in res.text
|
||||
assert isinstance(res.keywords, list)
|
||||
assert len(res.keywords) > 0
|
||||
|
||||
|
||||
def test_readability_extractor():
|
||||
extractor = ReadabilityExtractor()
|
||||
res = extractor.extract(SAMPLE_HTML)
|
||||
assert isinstance(res, ReadabilityData)
|
||||
assert res.error is None
|
||||
assert res.title is not None
|
||||
assert res.cleaned_html is not None
|
||||
assert "Armani" in (res.cleaned_text or "") or "River" in (res.cleaned_text or "")
|
||||
|
||||
|
||||
def test_extract_all_engines():
|
||||
traf, news, read = extract_all_engines(
|
||||
SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es"
|
||||
)
|
||||
assert traf.error is None
|
||||
assert news.error is None
|
||||
assert read.error is None
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 3. Testes de Isolamento de Falhas (Resiliência)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_extractor_error_isolation_on_faulty_engine():
|
||||
with patch(
|
||||
"scripts.extract_article_contents.TrafilaturaExtractor.extract",
|
||||
side_effect=RuntimeError("Trafilatura crash"),
|
||||
):
|
||||
traf, news, read = extract_all_engines(
|
||||
SAMPLE_HTML, url="https://example.com", language="es"
|
||||
)
|
||||
assert traf.error == "Trafilatura crash"
|
||||
assert news.error is None
|
||||
assert read.error is None
|
||||
|
||||
|
||||
def test_crawler_error_isolation(sample_input_json: Path, tmp_path: Path):
|
||||
out_file = tmp_path / "river_plate_extracted.json"
|
||||
|
||||
def mock_crawl(url, timeout_sec=30):
|
||||
if "ole.com.ar" in url:
|
||||
return SAMPLE_HTML, "River Plate Olé", 200
|
||||
raise ConnectionError("Connection refused by tycsports.com")
|
||||
|
||||
with patch.object(ArticleCrawler, "crawl", side_effect=mock_crawl):
|
||||
with patch.object(ArticleCrawler, "start"), patch.object(ArticleCrawler, "close"):
|
||||
report = process_batch(
|
||||
input_path=sample_input_json,
|
||||
output_path=out_file,
|
||||
silent=True,
|
||||
)
|
||||
|
||||
assert report.total_articles == 2
|
||||
assert report.successful_articles == 1
|
||||
assert report.failed_articles == 1
|
||||
assert report.articles[0].extraction_status == "success"
|
||||
assert report.articles[1].extraction_status == "failed"
|
||||
assert "Connection refused" in (report.articles[1].error_message or "")
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 4. Testes de CLI e Limitação (--limit, --silent, --language)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_process_batch_with_limit(sample_input_json: Path, tmp_path: Path):
|
||||
out_file = tmp_path / "limit_extracted.json"
|
||||
|
||||
with (
|
||||
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
|
||||
patch.object(ArticleCrawler, "start"),
|
||||
patch.object(ArticleCrawler, "close"),
|
||||
):
|
||||
report = process_batch(
|
||||
input_path=sample_input_json,
|
||||
output_path=out_file,
|
||||
limit=1,
|
||||
silent=True,
|
||||
)
|
||||
|
||||
assert report.total_articles == 1
|
||||
assert len(report.articles) == 1
|
||||
assert out_file.exists()
|
||||
|
||||
|
||||
def test_cli_main_success(sample_input_json: Path, tmp_path: Path, capsys):
|
||||
out_file = tmp_path / "cli_out.json"
|
||||
|
||||
with (
|
||||
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
|
||||
patch.object(ArticleCrawler, "start"),
|
||||
patch.object(ArticleCrawler, "close"),
|
||||
):
|
||||
exit_code = main(["-i", str(sample_input_json), "-o", str(out_file), "--limit", "1", "-s"])
|
||||
|
||||
assert exit_code == 0
|
||||
assert out_file.exists()
|
||||
|
||||
|
||||
def test_cli_main_missing_input_file(tmp_path: Path, capsys):
|
||||
missing_file = tmp_path / "does_not_exist.json"
|
||||
exit_code = main(["-i", str(missing_file), "-s"])
|
||||
assert exit_code == 1
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 5. Testes End-to-End (E2E) ao Vivo (Live Network)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_e2e_live_article_extraction(tmp_path: Path):
|
||||
"""Valida E2E a extração real ao vivo com Foxcape e os 3 motores em lote."""
|
||||
live_input_file = Path("out/river_plate.json")
|
||||
if not live_input_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E ao vivo.")
|
||||
|
||||
out_file = tmp_path / "e2e_live_extracted.json"
|
||||
|
||||
# Executa o batch real com limite de 1 notícia
|
||||
report = process_batch(
|
||||
input_path=live_input_file,
|
||||
output_path=out_file,
|
||||
limit=1,
|
||||
silent=True,
|
||||
)
|
||||
|
||||
assert report.total_articles == 1
|
||||
assert report.successful_articles == 1
|
||||
assert report.failed_articles == 0
|
||||
assert len(report.articles) == 1
|
||||
|
||||
art = report.articles[0]
|
||||
assert art.extraction_status == "success"
|
||||
assert art.crawled_url.startswith("http")
|
||||
|
||||
# Valida que todos os 3 motores extraíram dados reais
|
||||
assert art.trafilatura is not None and art.trafilatura.error is None
|
||||
assert len(art.trafilatura.text) > 50
|
||||
|
||||
assert art.newspaper4k is not None and art.newspaper4k.error is None
|
||||
assert len(art.newspaper4k.text) > 50
|
||||
assert isinstance(art.newspaper4k.keywords, list)
|
||||
|
||||
assert art.readability is not None and art.readability.error is None
|
||||
assert art.readability.cleaned_html is not None
|
||||
assert len(art.readability.cleaned_text or "") > 50
|
||||
|
||||
# Valida arquivo JSON gravado
|
||||
assert out_file.exists()
|
||||
saved = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert saved["total_articles"] == 1
|
||||
assert saved["articles"][0]["extraction_status"] == "success"
|
||||
|
||||
|
||||
def test_e2e_cli_live_execution(tmp_path: Path):
|
||||
"""Valida E2E a execução do CLI real de ponta a ponta."""
|
||||
live_input_file = Path("out/river_plate.json")
|
||||
if not live_input_file.exists():
|
||||
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E.")
|
||||
|
||||
out_file = tmp_path / "e2e_cli_live.json"
|
||||
exit_code = main(["-i", str(live_input_file), "-o", str(out_file), "--limit", "1", "-s"])
|
||||
|
||||
assert exit_code == 0
|
||||
assert out_file.exists()
|
||||
data = json.loads(out_file.read_text(encoding="utf-8"))
|
||||
assert data["successful_articles"] == 1
|
||||
@@ -140,9 +140,7 @@ def test_resolve_article_url_fallback():
|
||||
direct_url = "https://www.globo.com/noticia/123"
|
||||
assert resolve_article_url(direct_url) == direct_url
|
||||
|
||||
with patch(
|
||||
"scripts.extract_google_news.gnewsdecoder", return_value={"status": False}
|
||||
):
|
||||
with patch("scripts.extract_google_news.gnewsdecoder", return_value={"status": False}):
|
||||
gn_url = "https://news.google.com/rss/articles/fake_token"
|
||||
assert resolve_article_url(gn_url) == gn_url
|
||||
|
||||
@@ -187,9 +185,7 @@ def test_resolve_articles_urls_batch():
|
||||
|
||||
def test_extract_google_news_orchestration_mocked(sample_rss_xml: str):
|
||||
"""Valida a consolidação do ExtractionResult a partir da busca mockada com URLs resolvidas."""
|
||||
query = SearchQuery(
|
||||
keyword="inteligência artificial", language="pt", locale="BR", max_pages=1
|
||||
)
|
||||
query = SearchQuery(keyword="inteligência artificial", language="pt", locale="BR", max_pages=1)
|
||||
|
||||
with (
|
||||
patch(
|
||||
@@ -221,13 +217,9 @@ def test_cli_execution_stdout(sample_rss_xml: str, capsys: pytest.CaptureFixture
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
return_value=sample_rss_xml,
|
||||
),
|
||||
patch(
|
||||
"scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u
|
||||
),
|
||||
patch("scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u),
|
||||
):
|
||||
exit_code = main(
|
||||
["--query", "inteligencia artificial", "--lang", "pt", "--pretty"]
|
||||
)
|
||||
exit_code = main(["--query", "inteligencia artificial", "--lang", "pt", "--pretty"])
|
||||
assert exit_code == 0
|
||||
|
||||
captured = capsys.readouterr()
|
||||
@@ -248,9 +240,7 @@ def test_cli_execution_file_output(sample_rss_xml: str, tmp_path: Path):
|
||||
"scripts.extract_google_news._fetch_rss_content",
|
||||
return_value=sample_rss_xml,
|
||||
),
|
||||
patch(
|
||||
"scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u
|
||||
),
|
||||
patch("scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u),
|
||||
):
|
||||
exit_code = main(["-q", "IA", "-p", "1", "-o", str(out_file)])
|
||||
assert exit_code == 0
|
||||
|
||||
@@ -1,11 +1,13 @@
|
||||
"""Unit tests for language detection and text normalization."""
|
||||
|
||||
from src.language import detect_language, normalize_text, SUPPORTED_LANGUAGES
|
||||
from src.language import detect_language, normalize_text
|
||||
|
||||
|
||||
def test_normalize_text():
|
||||
assert normalize_text("São Paulo & Petróleo") == "sao paulo & petroleo"
|
||||
assert normalize_text("Über große Veränderungen") == "uber grosse veranderungen" or "uber" in normalize_text("Über")
|
||||
assert normalize_text(
|
||||
"Über große Veränderungen"
|
||||
) == "uber grosse veranderungen" or "uber" in normalize_text("Über")
|
||||
assert normalize_text("Crème brûlée") == "creme brulee"
|
||||
|
||||
|
||||
@@ -24,7 +26,9 @@ def test_detect_english():
|
||||
|
||||
|
||||
def test_detect_spanish():
|
||||
text = "La empresa petrolera anunció una nueva inversión en el sector energético durante este año."
|
||||
text = (
|
||||
"La empresa petrolera anunció una nueva inversión en el sector energético durante este año."
|
||||
)
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "es"
|
||||
assert conf > 0.5
|
||||
|
||||
@@ -1,15 +1,15 @@
|
||||
"""Unit tests for ECP models, schema validation, and structured error handling."""
|
||||
|
||||
import pytest
|
||||
|
||||
from src.models import (
|
||||
ECPSnapshot,
|
||||
RelatedEntity,
|
||||
ClassificationResult,
|
||||
ClassificationError,
|
||||
ClassificationResult,
|
||||
DecisionCategory,
|
||||
ECPSnapshot,
|
||||
ErrorCode,
|
||||
)
|
||||
from src.parser import strip_markdown, extract_sentences, extract_evidence_snippets
|
||||
from src.parser import extract_evidence_snippets, strip_markdown
|
||||
|
||||
|
||||
def test_ecp_snapshot_valid():
|
||||
@@ -29,9 +29,9 @@ def test_ecp_snapshot_valid():
|
||||
"weight": 0.9,
|
||||
"aliases": ["Transpetro Logística"],
|
||||
"scope": "logistics",
|
||||
"confidence": 0.95
|
||||
"confidence": 0.95,
|
||||
}
|
||||
]
|
||||
],
|
||||
}
|
||||
snapshot = ECPSnapshot.from_dict(data)
|
||||
assert snapshot.target_entity_id == "ent_123"
|
||||
@@ -48,7 +48,7 @@ def test_ecp_snapshot_defaults():
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["energia"]
|
||||
"anchors": ["energia"],
|
||||
}
|
||||
snapshot = ECPSnapshot.from_dict(data)
|
||||
assert snapshot.negative_anchors == []
|
||||
@@ -61,7 +61,7 @@ def test_ecp_snapshot_missing_required():
|
||||
"target_entity_id": "ent_123",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["energia"]
|
||||
"anchors": ["energia"],
|
||||
}
|
||||
with pytest.raises(ValueError, match="Missing required field"):
|
||||
ECPSnapshot.from_dict(data)
|
||||
@@ -89,7 +89,7 @@ def test_classification_error_serialization():
|
||||
err = ClassificationError(
|
||||
error_code=ErrorCode.INVALID_ECP_JSON,
|
||||
message="Malformed JSON syntax",
|
||||
details={"path": "snapshot.json"}
|
||||
details={"path": "snapshot.json"},
|
||||
)
|
||||
d = err.to_dict()
|
||||
assert d["error_code"] == "invalid_ecp_json"
|
||||
|
||||
Reference in New Issue
Block a user