feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
@@ -27,7 +27,7 @@ def test_classifier_with_adapter_flags():
|
||||
target_name="TestCorp",
|
||||
aliases=["TestCorp"],
|
||||
domain="Tech",
|
||||
anchors=["software"]
|
||||
anchors=["software"],
|
||||
)
|
||||
res = classifier.classify(ecp, "TestCorp builds enterprise cloud software.")
|
||||
assert res.decision.value == "DIRECT_INHERENT"
|
||||
|
||||
Reference in New Issue
Block a user