70 lines
2.7 KiB
Python
70 lines
2.7 KiB
Python
"""Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence Classifier.
|
|
|
|
Matrix: 6 Languages (PT, EN, ES, DE, IT, FR) x 4 Decisions (DIRECT, CONTEXTUAL, TANGENTIAL, NOT_RELATED).
|
|
Target Success Criterion: Precision >= 90% over the 24 cases.
|
|
"""
|
|
|
|
import json
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from src.tools.classifier import InherenceClassifier
|
|
from src.tools.models import ECPSnapshot
|
|
|
|
FIXTURES_DIR = Path(__file__).parent.parent / "fixtures" / "benchmark_24"
|
|
LANGUAGES = ["pt", "en", "es", "de", "it", "fr"]
|
|
DECISION_TYPES = ["direct", "contextual", "tangential", "not_related"]
|
|
|
|
BENCHMARK_CASES = [(lang, dec_type) for lang in LANGUAGES for dec_type in DECISION_TYPES]
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
def classifier():
|
|
return InherenceClassifier()
|
|
|
|
|
|
@pytest.mark.parametrize("lang,dec_type", BENCHMARK_CASES)
|
|
def test_benchmark_case(classifier, lang: str, dec_type: str):
|
|
case_dir = FIXTURES_DIR / lang
|
|
ecp_file = case_dir / "ecp.json"
|
|
content_file = case_dir / f"{dec_type}.md"
|
|
expected_file = case_dir / f"{dec_type}_expected.json"
|
|
|
|
assert ecp_file.is_file(), f"Missing ECP fixture: {ecp_file}"
|
|
assert content_file.is_file(), f"Missing Content fixture: {content_file}"
|
|
assert expected_file.is_file(), f"Missing Expected fixture: {expected_file}"
|
|
|
|
ecp = ECPSnapshot.from_json_str(ecp_file.read_text(encoding="utf-8"))
|
|
content = content_file.read_text(encoding="utf-8")
|
|
expected = json.loads(expected_file.read_text(encoding="utf-8"))
|
|
|
|
result = classifier.classify(ecp, content)
|
|
|
|
# 1. Decision category validation
|
|
assert result.decision.value == expected["expected_decision"], (
|
|
f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
|
|
)
|
|
|
|
# 2. Derived is_inherent boolean validation
|
|
assert result.is_inherent == expected["expected_is_inherent"], (
|
|
f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
|
|
)
|
|
|
|
# 3. Language detection validation
|
|
assert result.detected_language == expected["expected_language"], (
|
|
f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
|
|
)
|
|
|
|
# 4. Confidence threshold validation
|
|
min_conf = expected.get("min_confidence", 0.0)
|
|
assert result.confidence >= min_conf, (
|
|
f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
|
|
)
|
|
|
|
# 5. Evidence presence for inherent content
|
|
if result.is_inherent:
|
|
assert len(result.evidence) > 0, (
|
|
f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
|
|
)
|