feat(extractor): implement multi-engine article content extractor

- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability)
- Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing)
- Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts
- Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
2026-08-20 19:22:20 -03:00
parent 6e3d57619b
commit 6a45368cb0
85 changed files with 18345 additions and 3897 deletions
+1 -1
View File
@@ -27,7 +27,7 @@ def test_classifier_with_adapter_flags():
target_name="TestCorp",
aliases=["TestCorp"],
domain="Tech",
anchors=["software"]
anchors=["software"],
)
res = classifier.classify(ecp, "TestCorp builds enterprise cloud software.")
assert res.decision.value == "DIRECT_INHERENT"
+57 -29
View File
@@ -7,9 +7,8 @@ and CLI execution behavior via subprocess (exit codes, stream purity, JSON parsi
import json
import subprocess
import sys
from pathlib import Path
import pytest
from src.models import ECPSnapshot, RelatedEntity, DecisionCategory
from src.models import DecisionCategory, ECPSnapshot, RelatedEntity
def test_adversarial_sao_paulo_city_vs_fc():
@@ -20,9 +19,13 @@ def test_adversarial_sao_paulo_city_vs_fc():
aliases=["São Paulo", "SPFC", "Tricolor Paulista"],
domain="Futebol e Esportes",
anchors=["Morumbi", "futebol", "campeonato", "Copa Libertadores", "elenco", "estádio"],
negative_anchors=["prefeitura de são paulo", "governo do estado de são paulo", "trânsito na capital paulista"],
negative_anchors=[
"prefeitura de são paulo",
"governo do estado de são paulo",
"trânsito na capital paulista",
],
graph_version="1.0.0",
related_entities=[]
related_entities=[],
)
content = (
"# Obras Viárias na Capital\n\n"
@@ -30,6 +33,7 @@ def test_adversarial_sao_paulo_city_vs_fc():
"para desafogar o fluxo de veículos na região central durante os horários de pico."
)
from src.classifier import InherenceClassifier
classifier = InherenceClassifier()
result = classifier.classify(ecp, content)
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
@@ -47,13 +51,14 @@ def test_adversarial_apple_fruit_recipe():
anchors=["iPhone", "MacBook", "iOS", "silicon", "hardware"],
negative_anchors=["apple pie", "orchard harvest", "doce de maçã"],
graph_version="1.0.0",
related_entities=[]
related_entities=[],
)
content = (
"# Receita Caseira\n\n"
"Comprei maçãs frescas no mercado para preparar um doce de maçã com canela e açúcar mascavo."
)
from src.classifier import InherenceClassifier
classifier = InherenceClassifier()
result = classifier.classify(ecp, content)
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
@@ -76,9 +81,9 @@ def test_adversarial_related_entity_without_scope_context():
relation_type="SUPPLIER_OF",
weight=0.95,
scope="battery_technology",
confidence=0.99
confidence=0.99,
)
]
],
)
# Content mentions Northvolt in an unrelated/passing architectural context without domain anchors
content = (
@@ -87,6 +92,7 @@ def test_adversarial_related_entity_without_scope_context():
"mit moderner Holzfassade und Blick auf den See."
)
from src.classifier import InherenceClassifier
classifier = InherenceClassifier()
result = classifier.classify(ecp, content)
# Must be TANGENTIAL or NOT_RELATED, NEVER CONTEXTUAL_INHERENT
@@ -98,18 +104,27 @@ def test_adversarial_related_entity_without_scope_context():
def test_adversarial_subprocess_cli_success_stdout(tmp_path):
"""Run CLI via subprocess without --output and verify stdout is pure parseable JSON."""
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_petrobras",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_petrobras",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.", encoding="utf-8")
content_file.write_text(
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
encoding="utf-8",
)
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
res = subprocess.run(
@@ -133,16 +148,22 @@ def test_adversarial_subprocess_cli_success_stdout(tmp_path):
def test_adversarial_subprocess_cli_empty_content(tmp_path):
"""Run CLI via subprocess with empty content and verify error code and exit code."""
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"target_name": "Test",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Test",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "empty.md"
content_file.write_text(" \n\n ", encoding="utf-8")
@@ -164,15 +185,21 @@ def test_adversarial_subprocess_cli_empty_content(tmp_path):
def test_adversarial_subprocess_cli_missing_required_field(tmp_path):
"""Run CLI via subprocess with missing target_name and verify error payload."""
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
ecp_file = tmp_path / "ecp_bad.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("Conteúdo de teste válido.", encoding="utf-8")
@@ -193,6 +220,7 @@ def test_adversarial_subprocess_cli_missing_required_field(tmp_path):
def test_adversarial_subprocess_cli_corrupted_json(tmp_path):
"""Run CLI via subprocess with corrupted JSON and verify error payload."""
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
ecp_file = tmp_path / "ecp_corrupted.json"
+19 -21
View File
@@ -6,19 +6,17 @@ Target Success Criterion: Precision >= 90% over the 24 cases.
import json
from pathlib import Path
import pytest
from src.models import ECPSnapshot, DecisionCategory
from src.classifier import InherenceClassifier
from src.models import ECPSnapshot
FIXTURES_DIR = Path(__file__).parent / "fixtures" / "benchmark_24"
LANGUAGES = ["pt", "en", "es", "de", "it", "fr"]
DECISION_TYPES = ["direct", "contextual", "tangential", "not_related"]
BENCHMARK_CASES = [
(lang, dec_type)
for lang in LANGUAGES
for dec_type in DECISION_TYPES
]
BENCHMARK_CASES = [(lang, dec_type) for lang in LANGUAGES for dec_type in DECISION_TYPES]
@pytest.fixture(scope="module")
@@ -44,28 +42,28 @@ def test_benchmark_case(classifier, lang: str, dec_type: str):
result = classifier.classify(ecp, content)
# 1. Decision category validation
assert (
result.decision.value == expected["expected_decision"]
), f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
assert result.decision.value == expected["expected_decision"], (
f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
)
# 2. Derived is_inherent boolean validation
assert (
result.is_inherent == expected["expected_is_inherent"]
), f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
assert result.is_inherent == expected["expected_is_inherent"], (
f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
)
# 3. Language detection validation
assert (
result.detected_language == expected["expected_language"]
), f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
assert result.detected_language == expected["expected_language"], (
f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
)
# 4. Confidence threshold validation
min_conf = expected.get("min_confidence", 0.0)
assert (
result.confidence >= min_conf
), f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
assert result.confidence >= min_conf, (
f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
)
# 5. Evidence presence for inherent content
if result.is_inherent:
assert (
len(result.evidence) > 0
), f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
assert len(result.evidence) > 0, (
f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
)
+4 -3
View File
@@ -1,8 +1,9 @@
"""Unit tests for deterministic classification decision logic."""
import pytest
from src.models import ECPSnapshot, RelatedEntity, DecisionCategory
from src.classifier import InherenceClassifier
from src.models import DecisionCategory, ECPSnapshot, RelatedEntity
@pytest.fixture
@@ -23,9 +24,9 @@ def petrobras_ecp():
weight=0.85,
aliases=[],
scope="logistics",
confidence=1.0
confidence=1.0,
)
]
],
)
+52 -31
View File
@@ -1,23 +1,30 @@
"""CLI execution tests covering flags, arguments, stdout, and error handling."""
import json
from pathlib import Path
import pytest
from classify import main
def test_cli_success_stdout(tmp_path, capsys):
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.", encoding="utf-8")
content_file.write_text(
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
encoding="utf-8",
)
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
assert exit_code == 0
@@ -31,20 +38,30 @@ def test_cli_success_stdout(tmp_path, capsys):
def test_cli_output_file(tmp_path):
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("Petrobras anunciou investimentos bilionários em novas refinarias de petróleo.", encoding="utf-8")
content_file.write_text(
"Petrobras anunciou investimentos bilionários em novas refinarias de petróleo.",
encoding="utf-8",
)
output_file = tmp_path / "out" / "result.json"
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file), "--output", str(output_file)])
exit_code = main(
["--ecp", str(ecp_file), "--content", str(content_file), "--output", str(output_file)]
)
assert exit_code == 0
assert output_file.exists()
@@ -67,13 +84,18 @@ def test_cli_missing_ecp_file(tmp_path, capsys):
def test_cli_empty_content_file(tmp_path, capsys):
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "empty.md"
content_file.write_text(" \n\n ", encoding="utf-8")
@@ -88,11 +110,10 @@ def test_cli_empty_content_file(tmp_path, capsys):
def test_cli_missing_required_ecp_field(tmp_path, capsys):
ecp_file = tmp_path / "bad_ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"domain": "Oil & Gas",
"anchors": ["petróleo"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps({"target_entity_id": "ent_1", "domain": "Oil & Gas", "anchors": ["petróleo"]}),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("Algum conteúdo válido para testar o erro.", encoding="utf-8")
+372
View File
@@ -0,0 +1,372 @@
#!/usr/bin/env python3
"""
Testes automatizados para o Extrator e Parser Multimotor de Artigos.
Cobre modelos de dados, parsers (Trafilatura, Newspaper4k, Readability),
isolamento de falhas, orquestração de lote e interface CLI.
"""
from __future__ import annotations
import json
from pathlib import Path
from unittest.mock import patch
import pytest
from scripts.extract_article_contents import (
ArticleCrawler,
ExtractedArticle,
ExtractionBatchReport,
InputArticle,
NewspaperData,
NewspaperExtractor,
ReadabilityData,
ReadabilityExtractor,
TrafilaturaData,
TrafilaturaExtractor,
extract_all_engines,
load_search_json,
main,
process_batch,
save_extracted_json,
)
SAMPLE_HTML = """
<!DOCTYPE html>
<html lang="es">
<head>
<meta charset="utf-8">
<title>River Plate igualó sin goles ante Independiente Santa Fe - Olé</title>
<meta name="description" content="El equipo de Núñez empató 0-0 en Bogotá por los octavos de final.">
<meta name="author" content="Juan Pérez">
<meta property="og:title" content="River Plate igualó sin goles ante Independiente Santa Fe">
<meta property="og:image" content="https://media.ole.com.ar/river.jpg">
</head>
<body>
<header><nav><a href="/">Inicio</a></nav></header>
<article>
<h1>River Plate igualó sin goles ante Independiente Santa Fe</h1>
<p class="byline">Por Juan Pérez - 20 de Agosto de 2026</p>
<p class="lead">El equipo de Núñez empató 0-0 en Bogotá por la Copa Sudamericana.</p>
<p>Franco Armani fue la gran figura del encuentro con tres atajadas espectaculares en el primer tiempo.</p>
<p>El partido de vuelta se disputará en el estadio Monumental la próxima semana ante una multitud.</p>
</article>
<footer><p>Copyright 2026 Olé</p></footer>
</body>
</html>
"""
@pytest.fixture
def sample_input_json(tmp_path: Path) -> Path:
data = {
"query": "River Plate",
"language": "es",
"locale": "AR",
"total_itens": 2,
"items": [
{
"titulo": "River Plate igualó sin goles ante Santa Fe",
"subtitulo": "Empate en Bogotá",
"quando_publicado": "Thu, 20 Aug 2026 03:27:26 GMT",
"url": "https://www.ole.com.ar/river-0-0-santa-fe.html",
"pagina": 1,
},
{
"titulo": "Armani fue la figura de River",
"subtitulo": "Gran actuación del arquero",
"quando_publicado": "Thu, 20 Aug 2026 04:00:00 GMT",
"url": "https://www.tycsports.com/armani-figura.html",
"pagina": 1,
},
],
}
input_file = tmp_path / "river_plate.json"
input_file.write_text(json.dumps(data, ensure_ascii=False), encoding="utf-8")
return input_file
# ==============================================================================
# 1. Testes de Modelos e I/O de JSON
# ==============================================================================
def test_input_article_creation():
article = InputArticle(
titulo="Notícia Teste",
url="https://example.com/noticia",
subtitulo="Subtítulo",
quando_publicado="Thu, 20 Aug 2026",
pagina=1,
)
assert article.titulo == "Notícia Teste"
assert article.url == "https://example.com/noticia"
assert article.pagina == 1
d = article.to_dict()
assert d["titulo"] == "Notícia Teste"
assert d["url"] == "https://example.com/noticia"
def test_load_search_json_valid(sample_input_json: Path):
query, lang, items = load_search_json(sample_input_json)
assert query == "River Plate"
assert lang == "es"
assert len(items) == 2
assert items[0].url == "https://www.ole.com.ar/river-0-0-santa-fe.html"
def test_load_search_json_invalid_file(tmp_path: Path):
non_existent = tmp_path / "missing.json"
with pytest.raises(FileNotFoundError):
load_search_json(non_existent)
def test_save_extracted_json(tmp_path: Path):
report = ExtractionBatchReport(
source_file="test.json",
processed_at="2026-08-20T12:00:00Z",
total_articles=1,
successful_articles=1,
failed_articles=0,
articles=[
ExtractedArticle(
input_meta=InputArticle(titulo="Teste", url="https://example.com/noticia"),
extraction_status="success",
error_message=None,
crawled_url="https://example.com/noticia",
page_title="Página Teste",
http_status=200,
trafilatura=TrafilaturaData(
title="Teste",
author="Autor",
date="2026-08-20",
description="Desc",
categories=[],
tags=[],
canonical_url=None,
text="Texto do teste.",
raw_json=None,
error=None,
),
newspaper4k=NewspaperData(
title="Teste",
authors=["Autor"],
publish_date="2026-08-20",
text="Texto do teste.",
summary="Resumo",
keywords=["teste"],
top_image=None,
images=[],
meta_data={},
error=None,
),
readability=ReadabilityData(
title="Teste",
short_title="Teste",
cleaned_html="<p>Texto do teste.</p>",
cleaned_text="Texto do teste.",
error=None,
),
)
],
)
out_file = tmp_path / "out" / "result.json"
save_extracted_json(report, out_file)
assert out_file.exists()
content = json.loads(out_file.read_text(encoding="utf-8"))
assert content["total_articles"] == 1
assert content["articles"][0]["trafilatura"]["title"] == "Teste"
# ==============================================================================
# 2. Testes Unitários dos Parsers (Trafilatura, Newspaper4k, Readability)
# ==============================================================================
def test_trafilatura_extractor():
extractor = TrafilaturaExtractor()
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia")
assert isinstance(res, TrafilaturaData)
assert res.error is None
assert "Armani" in res.text or "River Plate" in res.text
assert res.title is not None
def test_newspaper_extractor():
extractor = NewspaperExtractor()
res = extractor.extract(SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es")
assert isinstance(res, NewspaperData)
assert res.error is None
assert "Armani" in res.text or "River" in res.text
assert isinstance(res.keywords, list)
assert len(res.keywords) > 0
def test_readability_extractor():
extractor = ReadabilityExtractor()
res = extractor.extract(SAMPLE_HTML)
assert isinstance(res, ReadabilityData)
assert res.error is None
assert res.title is not None
assert res.cleaned_html is not None
assert "Armani" in (res.cleaned_text or "") or "River" in (res.cleaned_text or "")
def test_extract_all_engines():
traf, news, read = extract_all_engines(
SAMPLE_HTML, url="https://www.ole.com.ar/noticia", language="es"
)
assert traf.error is None
assert news.error is None
assert read.error is None
# ==============================================================================
# 3. Testes de Isolamento de Falhas (Resiliência)
# ==============================================================================
def test_extractor_error_isolation_on_faulty_engine():
with patch(
"scripts.extract_article_contents.TrafilaturaExtractor.extract",
side_effect=RuntimeError("Trafilatura crash"),
):
traf, news, read = extract_all_engines(
SAMPLE_HTML, url="https://example.com", language="es"
)
assert traf.error == "Trafilatura crash"
assert news.error is None
assert read.error is None
def test_crawler_error_isolation(sample_input_json: Path, tmp_path: Path):
out_file = tmp_path / "river_plate_extracted.json"
def mock_crawl(url, timeout_sec=30):
if "ole.com.ar" in url:
return SAMPLE_HTML, "River Plate Olé", 200
raise ConnectionError("Connection refused by tycsports.com")
with patch.object(ArticleCrawler, "crawl", side_effect=mock_crawl):
with patch.object(ArticleCrawler, "start"), patch.object(ArticleCrawler, "close"):
report = process_batch(
input_path=sample_input_json,
output_path=out_file,
silent=True,
)
assert report.total_articles == 2
assert report.successful_articles == 1
assert report.failed_articles == 1
assert report.articles[0].extraction_status == "success"
assert report.articles[1].extraction_status == "failed"
assert "Connection refused" in (report.articles[1].error_message or "")
# ==============================================================================
# 4. Testes de CLI e Limitação (--limit, --silent, --language)
# ==============================================================================
def test_process_batch_with_limit(sample_input_json: Path, tmp_path: Path):
out_file = tmp_path / "limit_extracted.json"
with (
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
patch.object(ArticleCrawler, "start"),
patch.object(ArticleCrawler, "close"),
):
report = process_batch(
input_path=sample_input_json,
output_path=out_file,
limit=1,
silent=True,
)
assert report.total_articles == 1
assert len(report.articles) == 1
assert out_file.exists()
def test_cli_main_success(sample_input_json: Path, tmp_path: Path, capsys):
out_file = tmp_path / "cli_out.json"
with (
patch.object(ArticleCrawler, "crawl", return_value=(SAMPLE_HTML, "Page Title", 200)),
patch.object(ArticleCrawler, "start"),
patch.object(ArticleCrawler, "close"),
):
exit_code = main(["-i", str(sample_input_json), "-o", str(out_file), "--limit", "1", "-s"])
assert exit_code == 0
assert out_file.exists()
def test_cli_main_missing_input_file(tmp_path: Path, capsys):
missing_file = tmp_path / "does_not_exist.json"
exit_code = main(["-i", str(missing_file), "-s"])
assert exit_code == 1
# ==============================================================================
# 5. Testes End-to-End (E2E) ao Vivo (Live Network)
# ==============================================================================
def test_e2e_live_article_extraction(tmp_path: Path):
"""Valida E2E a extração real ao vivo com Foxcape e os 3 motores em lote."""
live_input_file = Path("out/river_plate.json")
if not live_input_file.exists():
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E ao vivo.")
out_file = tmp_path / "e2e_live_extracted.json"
# Executa o batch real com limite de 1 notícia
report = process_batch(
input_path=live_input_file,
output_path=out_file,
limit=1,
silent=True,
)
assert report.total_articles == 1
assert report.successful_articles == 1
assert report.failed_articles == 0
assert len(report.articles) == 1
art = report.articles[0]
assert art.extraction_status == "success"
assert art.crawled_url.startswith("http")
# Valida que todos os 3 motores extraíram dados reais
assert art.trafilatura is not None and art.trafilatura.error is None
assert len(art.trafilatura.text) > 50
assert art.newspaper4k is not None and art.newspaper4k.error is None
assert len(art.newspaper4k.text) > 50
assert isinstance(art.newspaper4k.keywords, list)
assert art.readability is not None and art.readability.error is None
assert art.readability.cleaned_html is not None
assert len(art.readability.cleaned_text or "") > 50
# Valida arquivo JSON gravado
assert out_file.exists()
saved = json.loads(out_file.read_text(encoding="utf-8"))
assert saved["total_articles"] == 1
assert saved["articles"][0]["extraction_status"] == "success"
def test_e2e_cli_live_execution(tmp_path: Path):
"""Valida E2E a execução do CLI real de ponta a ponta."""
live_input_file = Path("out/river_plate.json")
if not live_input_file.exists():
pytest.skip("Arquivo out/river_plate.json não encontrado para teste E2E.")
out_file = tmp_path / "e2e_cli_live.json"
exit_code = main(["-i", str(live_input_file), "-o", str(out_file), "--limit", "1", "-s"])
assert exit_code == 0
assert out_file.exists()
data = json.loads(out_file.read_text(encoding="utf-8"))
assert data["successful_articles"] == 1
+5 -15
View File
@@ -140,9 +140,7 @@ def test_resolve_article_url_fallback():
direct_url = "https://www.globo.com/noticia/123"
assert resolve_article_url(direct_url) == direct_url
with patch(
"scripts.extract_google_news.gnewsdecoder", return_value={"status": False}
):
with patch("scripts.extract_google_news.gnewsdecoder", return_value={"status": False}):
gn_url = "https://news.google.com/rss/articles/fake_token"
assert resolve_article_url(gn_url) == gn_url
@@ -187,9 +185,7 @@ def test_resolve_articles_urls_batch():
def test_extract_google_news_orchestration_mocked(sample_rss_xml: str):
"""Valida a consolidação do ExtractionResult a partir da busca mockada com URLs resolvidas."""
query = SearchQuery(
keyword="inteligência artificial", language="pt", locale="BR", max_pages=1
)
query = SearchQuery(keyword="inteligência artificial", language="pt", locale="BR", max_pages=1)
with (
patch(
@@ -221,13 +217,9 @@ def test_cli_execution_stdout(sample_rss_xml: str, capsys: pytest.CaptureFixture
"scripts.extract_google_news._fetch_rss_content",
return_value=sample_rss_xml,
),
patch(
"scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u
),
patch("scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u),
):
exit_code = main(
["--query", "inteligencia artificial", "--lang", "pt", "--pretty"]
)
exit_code = main(["--query", "inteligencia artificial", "--lang", "pt", "--pretty"])
assert exit_code == 0
captured = capsys.readouterr()
@@ -248,9 +240,7 @@ def test_cli_execution_file_output(sample_rss_xml: str, tmp_path: Path):
"scripts.extract_google_news._fetch_rss_content",
return_value=sample_rss_xml,
),
patch(
"scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u
),
patch("scripts.extract_google_news.resolve_article_url", side_effect=lambda u: u),
):
exit_code = main(["-q", "IA", "-p", "1", "-o", str(out_file)])
assert exit_code == 0
+7 -3
View File
@@ -1,11 +1,13 @@
"""Unit tests for language detection and text normalization."""
from src.language import detect_language, normalize_text, SUPPORTED_LANGUAGES
from src.language import detect_language, normalize_text
def test_normalize_text():
assert normalize_text("São Paulo & Petróleo") == "sao paulo & petroleo"
assert normalize_text("Über große Veränderungen") == "uber grosse veranderungen" or "uber" in normalize_text("Über")
assert normalize_text(
"Über große Veränderungen"
) == "uber grosse veranderungen" or "uber" in normalize_text("Über")
assert normalize_text("Crème brûlée") == "creme brulee"
@@ -24,7 +26,9 @@ def test_detect_english():
def test_detect_spanish():
text = "La empresa petrolera anunció una nueva inversión en el sector energético durante este año."
text = (
"La empresa petrolera anunció una nueva inversión en el sector energético durante este año."
)
lang, conf = detect_language(text)
assert lang == "es"
assert conf > 0.5
+9 -9
View File
@@ -1,15 +1,15 @@
"""Unit tests for ECP models, schema validation, and structured error handling."""
import pytest
from src.models import (
ECPSnapshot,
RelatedEntity,
ClassificationResult,
ClassificationError,
ClassificationResult,
DecisionCategory,
ECPSnapshot,
ErrorCode,
)
from src.parser import strip_markdown, extract_sentences, extract_evidence_snippets
from src.parser import extract_evidence_snippets, strip_markdown
def test_ecp_snapshot_valid():
@@ -29,9 +29,9 @@ def test_ecp_snapshot_valid():
"weight": 0.9,
"aliases": ["Transpetro Logística"],
"scope": "logistics",
"confidence": 0.95
"confidence": 0.95,
}
]
],
}
snapshot = ECPSnapshot.from_dict(data)
assert snapshot.target_entity_id == "ent_123"
@@ -48,7 +48,7 @@ def test_ecp_snapshot_defaults():
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["energia"]
"anchors": ["energia"],
}
snapshot = ECPSnapshot.from_dict(data)
assert snapshot.negative_anchors == []
@@ -61,7 +61,7 @@ def test_ecp_snapshot_missing_required():
"target_entity_id": "ent_123",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["energia"]
"anchors": ["energia"],
}
with pytest.raises(ValueError, match="Missing required field"):
ECPSnapshot.from_dict(data)
@@ -89,7 +89,7 @@ def test_classification_error_serialization():
err = ClassificationError(
error_code=ErrorCode.INVALID_ECP_JSON,
message="Malformed JSON syntax",
details={"path": "snapshot.json"}
details={"path": "snapshot.json"},
)
d = err.to_dict()
assert d["error_code"] == "invalid_ecp_json"