feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
@@ -1,8 +1,9 @@
|
||||
"""Unit tests for deterministic classification decision logic."""
|
||||
|
||||
import pytest
|
||||
from src.models import ECPSnapshot, RelatedEntity, DecisionCategory
|
||||
|
||||
from src.classifier import InherenceClassifier
|
||||
from src.models import DecisionCategory, ECPSnapshot, RelatedEntity
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
@@ -23,9 +24,9 @@ def petrobras_ecp():
|
||||
weight=0.85,
|
||||
aliases=[],
|
||||
scope="logistics",
|
||||
confidence=1.0
|
||||
confidence=1.0,
|
||||
)
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user