feat(extractor): implement multi-engine article content extractor

- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability)
- Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing)
- Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts
- Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
2026-08-20 19:22:20 -03:00
parent 6e3d57619b
commit 6a45368cb0
85 changed files with 18345 additions and 3897 deletions
+52 -31
View File
@@ -1,23 +1,30 @@
"""CLI execution tests covering flags, arguments, stdout, and error handling."""
import json
from pathlib import Path
import pytest
from classify import main
def test_cli_success_stdout(tmp_path, capsys):
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.", encoding="utf-8")
content_file.write_text(
"# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.",
encoding="utf-8",
)
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
assert exit_code == 0
@@ -31,20 +38,30 @@ def test_cli_success_stdout(tmp_path, capsys):
def test_cli_output_file(tmp_path):
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo", "pré-sal"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("Petrobras anunciou investimentos bilionários em novas refinarias de petróleo.", encoding="utf-8")
content_file.write_text(
"Petrobras anunciou investimentos bilionários em novas refinarias de petróleo.",
encoding="utf-8",
)
output_file = tmp_path / "out" / "result.json"
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file), "--output", str(output_file)])
exit_code = main(
["--ecp", str(ecp_file), "--content", str(content_file), "--output", str(output_file)]
)
assert exit_code == 0
assert output_file.exists()
@@ -67,13 +84,18 @@ def test_cli_missing_ecp_file(tmp_path, capsys):
def test_cli_empty_content_file(tmp_path, capsys):
ecp_file = tmp_path / "ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps(
{
"target_entity_id": "ent_1",
"target_name": "Petrobras",
"aliases": ["Petrobras"],
"domain": "Oil & Gas",
"anchors": ["petróleo"],
}
),
encoding="utf-8",
)
content_file = tmp_path / "empty.md"
content_file.write_text(" \n\n ", encoding="utf-8")
@@ -88,11 +110,10 @@ def test_cli_empty_content_file(tmp_path, capsys):
def test_cli_missing_required_ecp_field(tmp_path, capsys):
ecp_file = tmp_path / "bad_ecp.json"
ecp_file.write_text(json.dumps({
"target_entity_id": "ent_1",
"domain": "Oil & Gas",
"anchors": ["petróleo"]
}), encoding="utf-8")
ecp_file.write_text(
json.dumps({"target_entity_id": "ent_1", "domain": "Oil & Gas", "anchors": ["petróleo"]}),
encoding="utf-8",
)
content_file = tmp_path / "content.md"
content_file.write_text("Algum conteúdo válido para testar o erro.", encoding="utf-8")