- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
35 lines
1.1 KiB
Python
35 lines
1.1 KiB
Python
"""Unit tests for optional adapter interfaces (Tier 2 / Tier 3)."""
|
|
|
|
from src.adapters.embeddings import LocalEmbeddingsAdapter
|
|
from src.adapters.llm import LLMFallbackAdapter
|
|
from src.classifier import InherenceClassifier
|
|
from src.models import ECPSnapshot
|
|
|
|
|
|
def test_embeddings_adapter_interface():
|
|
adapter = LocalEmbeddingsAdapter()
|
|
assert isinstance(adapter.is_available(), bool)
|
|
assert adapter.evaluate_similarity("test text", ["term1", "term2"]) == 0.0
|
|
|
|
|
|
def test_llm_adapter_interface():
|
|
adapter = LLMFallbackAdapter()
|
|
assert isinstance(adapter.is_available(), bool)
|
|
|
|
|
|
def test_classifier_with_adapter_flags():
|
|
classifier = InherenceClassifier(enable_embeddings=True, enable_llm=True)
|
|
assert classifier._embeddings_adapter is not None
|
|
assert classifier._llm_adapter is not None
|
|
|
|
ecp = ECPSnapshot(
|
|
target_entity_id="ent_test",
|
|
target_name="TestCorp",
|
|
aliases=["TestCorp"],
|
|
domain="Tech",
|
|
anchors=["software"],
|
|
)
|
|
res = classifier.classify(ecp, "TestCorp builds enterprise cloud software.")
|
|
assert res.decision.value == "DIRECT_INHERENT"
|
|
assert res.is_inherent is True
|