feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
+19
-21
@@ -6,19 +6,17 @@ Target Success Criterion: Precision >= 90% over the 24 cases.
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from src.models import ECPSnapshot, DecisionCategory
|
||||
|
||||
from src.classifier import InherenceClassifier
|
||||
from src.models import ECPSnapshot
|
||||
|
||||
FIXTURES_DIR = Path(__file__).parent / "fixtures" / "benchmark_24"
|
||||
LANGUAGES = ["pt", "en", "es", "de", "it", "fr"]
|
||||
DECISION_TYPES = ["direct", "contextual", "tangential", "not_related"]
|
||||
|
||||
BENCHMARK_CASES = [
|
||||
(lang, dec_type)
|
||||
for lang in LANGUAGES
|
||||
for dec_type in DECISION_TYPES
|
||||
]
|
||||
BENCHMARK_CASES = [(lang, dec_type) for lang in LANGUAGES for dec_type in DECISION_TYPES]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
@@ -44,28 +42,28 @@ def test_benchmark_case(classifier, lang: str, dec_type: str):
|
||||
result = classifier.classify(ecp, content)
|
||||
|
||||
# 1. Decision category validation
|
||||
assert (
|
||||
result.decision.value == expected["expected_decision"]
|
||||
), f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
|
||||
assert result.decision.value == expected["expected_decision"], (
|
||||
f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
|
||||
)
|
||||
|
||||
# 2. Derived is_inherent boolean validation
|
||||
assert (
|
||||
result.is_inherent == expected["expected_is_inherent"]
|
||||
), f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
|
||||
assert result.is_inherent == expected["expected_is_inherent"], (
|
||||
f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
|
||||
)
|
||||
|
||||
# 3. Language detection validation
|
||||
assert (
|
||||
result.detected_language == expected["expected_language"]
|
||||
), f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
|
||||
assert result.detected_language == expected["expected_language"], (
|
||||
f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
|
||||
)
|
||||
|
||||
# 4. Confidence threshold validation
|
||||
min_conf = expected.get("min_confidence", 0.0)
|
||||
assert (
|
||||
result.confidence >= min_conf
|
||||
), f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
|
||||
assert result.confidence >= min_conf, (
|
||||
f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
|
||||
)
|
||||
|
||||
# 5. Evidence presence for inherent content
|
||||
if result.is_inherent:
|
||||
assert (
|
||||
len(result.evidence) > 0
|
||||
), f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
|
||||
assert len(result.evidence) > 0, (
|
||||
f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user