feat(extractor): implement multi-engine article content extractor

- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability)
- Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing)
- Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts
- Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
2026-08-20 19:22:20 -03:00
parent 6e3d57619b
commit 6a45368cb0
85 changed files with 18345 additions and 3897 deletions
+57 -29
View File
@@ -7,9 +7,8 @@ and CLI execution behavior via subprocess (exit codes, stream purity, JSON parsi
import json
import subprocess
import sys
from pathlib import Path
import pytest
from src.models import ECPSnapshot, RelatedEntity, DecisionCategory
from src.models import DecisionCategory, ECPSnapshot, RelatedEntity
def test_adversarial_sao_paulo_city_vs_fc():
@@ -20,9 +19,13 @@ def test_adversarial_sao_paulo_city_vs_fc():
aliases=["São Paulo", "SPFC", "Tricolor Paulista"],
domain="Futebol e Esportes",
anchors=["Morumbi", "futebol", "campeonato", "Copa Libertadores", "elenco", "estádio"],
negative_anchors=["prefeitura de são paulo", "governo do estado de são paulo", "trânsito na capital paulista"],
negative_anchors=[
"prefeitura de são paulo",
"governo do estado de são paulo",
"trânsito na capital paulista",
],
graph_version="1.0.0",
related_entities=[]
related_entities=[],
)
content = (
"# Obras Viárias na Capital\n\n"
@@ -30,6 +33,7 @@ def test_adversarial_sao_paulo_city_vs_fc():
"para desafogar o fluxo de veículos na região central durante os horários de pico."
)
from src.classifier import InherenceClassifier
classifier = InherenceClassifier()
result = classifier.classify(ecp, content)
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
@@ -47,13 +51,14 @@ def test_adversarial_apple_fruit_recipe():
anchors=["iPhone", "MacBook", "iOS", "silicon", "hardware"],
negative_anchors=["apple pie", "orchard harvest", "doce de maçã"],
graph_version="1.0.0",
related_entities=[]
related_entities=[],
)
content = (
"# Receita Caseira\n\n"
"Comprei maçãs frescas no mercado para preparar um doce de maçã com canela e açúcar mascavo."
)
from src.classifier import InherenceClassifier
classifier = InherenceClassifier()
result = classifier.classify(ecp, content)
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
@@ -76,9 +81,9 @@ def test_adversarial_related_entity_without_scope_context():
relation_type="SUPPLIER_OF",
weight=0.95,
scope="battery_technology",
confidence=0.99
confidence=0.99,
)
]
],
)
# Content mentions Northvolt in an unrelated/passing architectural context without domain anchors
content = (
@@ -87,6 +92,7 @@ def test_adversarial_related_entity_without_scope_context():
"mit moderner Holzfassade und Blick auf den See."
)
from src.classifier import InherenceClassifier
classifier = InherenceClassifier()
result = classifier.classify(ecp, content)
# Must be TANGENTIAL or NOT_RELATED, NEVER CONTEXTUAL_INHERENT
@@ -98,18 +104,27 @@ def test_adversarial_related_entity_without_scope_context():
def test_adversarial_subprocess_cli_success_stdout(tmp_path):
"""Run CLI via subprocess without --output and verify stdout is pure parseable JSON."""
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_petrobras",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_petrobras",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.", encoding="utf-8")
content_file.write_text(
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
encoding="utf-8",
)
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
res = subprocess.run(
@@ -133,16 +148,22 @@ def test_adversarial_subprocess_cli_success_stdout(tmp_path):
def test_adversarial_subprocess_cli_empty_content(tmp_path):
"""Run CLI via subprocess with empty content and verify error code and exit code."""
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"target_name": "Test",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Test",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "empty.md"
content_file.write_text(" \n\n ", encoding="utf-8")
@@ -164,15 +185,21 @@ def test_adversarial_subprocess_cli_empty_content(tmp_path):
def test_adversarial_subprocess_cli_missing_required_field(tmp_path):
"""Run CLI via subprocess with missing target_name and verify error payload."""
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
ecp_file = tmp_path / "ecp_bad.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"aliases": ["Test"],
"domain": "Tech",
"anchors": ["tech"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("Conteúdo de teste válido.", encoding="utf-8")
@@ -193,6 +220,7 @@ def test_adversarial_subprocess_cli_missing_required_field(tmp_path):
def test_adversarial_subprocess_cli_corrupted_json(tmp_path):
"""Run CLI via subprocess with corrupted JSON and verify error payload."""
import os
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
ecp_file = tmp_path / "ecp_corrupted.json"