"""Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence Classifier. Matrix: 6 Languages (PT, EN, ES, DE, IT, FR) x 4 Decisions (DIRECT, CONTEXTUAL, TANGENTIAL, NOT_RELATED). Target Success Criterion: Precision >= 90% over the 24 cases. """ import json from pathlib import Path import pytest from src.classifier import InherenceClassifier from src.models import ECPSnapshot FIXTURES_DIR = Path(__file__).parent / "fixtures" / "benchmark_24" LANGUAGES = ["pt", "en", "es", "de", "it", "fr"] DECISION_TYPES = ["direct", "contextual", "tangential", "not_related"] BENCHMARK_CASES = [(lang, dec_type) for lang in LANGUAGES for dec_type in DECISION_TYPES] @pytest.fixture(scope="module") def classifier(): return InherenceClassifier() @pytest.mark.parametrize("lang,dec_type", BENCHMARK_CASES) def test_benchmark_case(classifier, lang: str, dec_type: str): case_dir = FIXTURES_DIR / lang ecp_file = case_dir / "ecp.json" content_file = case_dir / f"{dec_type}.md" expected_file = case_dir / f"{dec_type}_expected.json" assert ecp_file.is_file(), f"Missing ECP fixture: {ecp_file}" assert content_file.is_file(), f"Missing Content fixture: {content_file}" assert expected_file.is_file(), f"Missing Expected fixture: {expected_file}" ecp = ECPSnapshot.from_json_str(ecp_file.read_text(encoding="utf-8")) content = content_file.read_text(encoding="utf-8") expected = json.loads(expected_file.read_text(encoding="utf-8")) result = classifier.classify(ecp, content) # 1. Decision category validation assert result.decision.value == expected["expected_decision"], ( f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}" ) # 2. Derived is_inherent boolean validation assert result.is_inherent == expected["expected_is_inherent"], ( f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}" ) # 3. Language detection validation assert result.detected_language == expected["expected_language"], ( f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'" ) # 4. Confidence threshold validation min_conf = expected.get("min_confidence", 0.0) assert result.confidence >= min_conf, ( f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}" ) # 5. Evidence presence for inherent content if result.is_inherent: assert len(result.evidence) > 0, ( f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets" )