feat(classifier): add multilingual ECP inherence classifier POC
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Test package for Multilingual NLP Entity Inherence Classifier."""
|
||||
@@ -0,0 +1,2 @@
|
||||
# Batteriefabrik in Europa
|
||||
Northvolt steigert die Produktion von Batteriezellen für führende Automobilhersteller auf dem europäischen Markt.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "CONTEXTUAL_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "de",
|
||||
"min_confidence": 0.70
|
||||
}
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
# Investitionsoffensive in Wolfsburg
|
||||
Volkswagen investiert Milliarden in die Fahrzeugproduktion und neue Plattformen für moderne Elektrofahrzeuge.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "DIRECT_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "de",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"target_entity_id": "ent_volkswagen_de",
|
||||
"target_name": "Volkswagen",
|
||||
"aliases": ["Volkswagen AG", "VW", "Volkswagen Group"],
|
||||
"domain": "Automobilindustrie",
|
||||
"anchors": ["Elektrofahrzeuge", "Batteriezellen", "Fahrzeugproduktion", "Automobilhersteller", "Elektromobilität"],
|
||||
"negative_anchors": ["Modellautosammlung", "Spielzeugautos"],
|
||||
"graph_version": "1.0.0",
|
||||
"related_entities": [
|
||||
{
|
||||
"entity_id": "ent_northvolt_de",
|
||||
"name": "Northvolt",
|
||||
"relation_type": "SUPPLIER_OF",
|
||||
"weight": 0.85,
|
||||
"scope": "battery_supply"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Backrezept für Apfelstrudel
|
||||
Den Teig dünn ausrollen, mit Rosinen und Zimt bestreuen und bei mittlerer Hitze im Ofen goldgelb backen.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "NOT_RELATED",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "de",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Reisebericht aus Niedersachsen
|
||||
Beim Spaziergang durch die Fußgängerzone sahen wir in der Ferne ein Werbeschild von Volkswagen neben dem Parkhaus.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "TANGENTIAL",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "de",
|
||||
"min_confidence": 0.30
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Electronics Assembly Expansion
|
||||
Foxconn is expanding its major manufacturing plants to boost production capacity for consumer electronics devices.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "CONTEXTUAL_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "en",
|
||||
"min_confidence": 0.70
|
||||
}
|
||||
+3
@@ -0,0 +1,3 @@
|
||||
# Apple Unveils Next-Generation iPhone
|
||||
|
||||
Apple announced its flagship iPhone today, featuring an advanced custom silicon chip and upgraded camera hardware.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "DIRECT_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "en",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
{
|
||||
"target_entity_id": "ent_apple_en",
|
||||
"target_name": "Apple",
|
||||
"aliases": ["Apple Inc.", "Apple"],
|
||||
"domain": "Technology & Consumer Electronics",
|
||||
"anchors": ["iPhone", "MacBook", "iOS", "silicon", "hardware", "smartphone", "consumer electronics", "manufacturing"],
|
||||
"negative_anchors": [
|
||||
"apple pie",
|
||||
"apple pie recipe",
|
||||
"orchard harvest"
|
||||
],
|
||||
"graph_version": "1.0.0",
|
||||
"related_entities": [
|
||||
{
|
||||
"entity_id": "ent_foxconn_en",
|
||||
"name": "Foxconn",
|
||||
"relation_type": "MANUFACTURER_FOR",
|
||||
"weight": 0.85,
|
||||
"scope": "manufacturing"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Classic Apple Pie Baking Guide
|
||||
Mix sliced cinnamon apples with brown sugar and butter before placing inside the homemade flaky apple pie crust.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "NOT_RELATED",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "en",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Morning City Walk
|
||||
While walking through the park downtown, we noticed an Apple poster near the transit stop before heading into the coffee shop.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "TANGENTIAL",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "en",
|
||||
"min_confidence": 0.30
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Nuevas Soluciones de Pagos Electrónicos
|
||||
La compañía PagoNxt implementó un nuevo sistema de liquidación de pagos y financiación para comercios internacionales.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "CONTEXTUAL_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "es",
|
||||
"min_confidence": 0.70
|
||||
}
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
# Expansión de Servicios Financieros en España
|
||||
El Banco Santander anunció hoy un aumento en la concesión de créditos e hipotecas a través de su plataforma de banca digital.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "DIRECT_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "es",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"target_entity_id": "ent_santander_es",
|
||||
"target_name": "Banco Santander",
|
||||
"aliases": ["Santander", "Grupo Santander", "Banco Santander S.A."],
|
||||
"domain": "Banca y Finanzas",
|
||||
"anchors": ["créditos", "sucursales", "cuentas bancarias", "hipotecas", "financiación", "banca digital"],
|
||||
"negative_anchors": ["playa de santander", "bahía de santander turismo"],
|
||||
"graph_version": "1.0.0",
|
||||
"related_entities": [
|
||||
{
|
||||
"entity_id": "ent_pagonxt_es",
|
||||
"name": "PagoNxt",
|
||||
"relation_type": "SUBSIDIARY_OF",
|
||||
"weight": 0.85,
|
||||
"scope": "payments"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Turismo en la Costa Norte
|
||||
La hermosa bahía de santander turismo ofrece paseos marítimos y playas con vistas espectaculares del mar Cantábrico.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "NOT_RELATED",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "es",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Guía Cultural de la Ciudad
|
||||
Caminando cerca del ayuntamiento pasamos frente a un edificio del Santander en una tarde soleada de primavera.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "TANGENTIAL",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "es",
|
||||
"min_confidence": 0.30
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Batteries Industrielles de Pointe
|
||||
L'entreprise Saft a inauguré une ligne de fabrication de batteries avancées destinées au stockage d'énergie et aux transports.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "CONTEXTUAL_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "fr",
|
||||
"min_confidence": 0.70
|
||||
}
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
# Développement Énergétique en France
|
||||
TotalEnergies a annoncé de nouveaux investissements massifs dans les énergies renouvelables et le raffinage durable.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "DIRECT_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "fr",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"target_entity_id": "ent_totalenergies_fr",
|
||||
"target_name": "TotalEnergies",
|
||||
"aliases": ["TotalEnergies SE", "Total"],
|
||||
"domain": "Énergie et Pétrole",
|
||||
"anchors": ["raffinage", "énergies renouvelables", "carburants", "gaz naturel", "production pétrolière", "stockage d'énergie", "batteries"],
|
||||
"negative_anchors": ["total look mode", "somme totale facture"],
|
||||
"graph_version": "1.0.0",
|
||||
"related_entities": [
|
||||
{
|
||||
"entity_id": "ent_saft_fr",
|
||||
"name": "Saft",
|
||||
"relation_type": "SUBSIDIARY_OF",
|
||||
"weight": 0.85,
|
||||
"scope": "battery_technology"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Recette Traditionnelle de la Quiche Lorraine
|
||||
Mélanger la crème fraîche avec les œufs battus, ajouter les lardons dorés et verser sur la pâte brisée avant de cuire au four.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "NOT_RELATED",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "fr",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Promenade dans Paris
|
||||
En marchant le long du boulevard Haussmann, nous avons aperçu un panneau publicitaire de TotalEnergies près de la station de métro.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "TANGENTIAL",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "fr",
|
||||
"min_confidence": 0.30
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Innovazione negli Impianti Frenanti
|
||||
Brembo ha sviluppato una nuova generazione di freni carboceramici ad alte prestazioni per le supercar sportive.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "CONTEXTUAL_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "it",
|
||||
"min_confidence": 0.70
|
||||
}
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
# Nuova Monoposto a Maranello
|
||||
La Ferrari ha presentato la nuova vettura di Formula 1 dotata di motori ibridi potenziati e aerodinamica avanzata.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "DIRECT_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "it",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"target_entity_id": "ent_ferrari_it",
|
||||
"target_name": "Ferrari",
|
||||
"aliases": ["Scuderia Ferrari", "Ferrari N.V."],
|
||||
"domain": "Supercar e Motorsport",
|
||||
"anchors": ["Maranello", "motori", "Formula 1", "supercar", "aerodinamica", "velocità"],
|
||||
"negative_anchors": ["Ferrari Trento vino spumante"],
|
||||
"graph_version": "1.0.0",
|
||||
"related_entities": [
|
||||
{
|
||||
"entity_id": "ent_brembo_it",
|
||||
"name": "Brembo",
|
||||
"relation_type": "SUPPLIER_OF",
|
||||
"weight": 0.85,
|
||||
"scope": "braking_systems"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Ricetta Tradizionale del Risotto
|
||||
Tostare il riso Carnaroli con burro e cipolla, sfumare con brodo caldo e mantecare con formaggio Parmigiano Reggiano.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "NOT_RELATED",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "it",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Passeggiata Pomeridiana a Modena
|
||||
Camminando nel centro storico di Modena abbiamo notato una maglietta della Ferrari esposta nella vetrina di un negozio.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "TANGENTIAL",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "it",
|
||||
"min_confidence": 0.30
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Logística de Derivados
|
||||
A Transpetro modernizou os dutos e navios petroleiros para transporte de combustíveis pelo país.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "CONTEXTUAL_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "pt",
|
||||
"min_confidence": 0.70
|
||||
}
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
# Produção de Petróleo no Brasil
|
||||
A Petrobras registrou um aumento na extração de petróleo no pré-sal com novas tecnologias offshore.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "DIRECT_INHERENT",
|
||||
"expected_is_inherent": true,
|
||||
"expected_language": "pt",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"target_entity_id": "ent_petrobras_pt",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petróleo Brasileiro S.A.", "Petrobras"],
|
||||
"domain": "Energia e Petróleo",
|
||||
"anchors": ["pré-sal", "refinaria", "combustíveis", "petróleo"],
|
||||
"negative_anchors": ["posto clandestino"],
|
||||
"graph_version": "1.0.0",
|
||||
"related_entities": [
|
||||
{
|
||||
"entity_id": "ent_transpetro_pt",
|
||||
"name": "Transpetro",
|
||||
"relation_type": "SUBSIDIARY_OF",
|
||||
"weight": 0.9,
|
||||
"scope": "logistics"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Dicas de Jardinagem
|
||||
Cultivar orquídeas requer rega moderada e ambiente com luz solar indireta para florescer com saúde.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "NOT_RELATED",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "pt",
|
||||
"min_confidence": 0.85
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Turismo no Rio de Janeiro
|
||||
Durante a caminhada pela orla da praia, avistamos ao longe uma placa da Petrobras perto da avenida movimentada.
|
||||
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"expected_decision": "TANGENTIAL",
|
||||
"expected_is_inherent": false,
|
||||
"expected_language": "pt",
|
||||
"min_confidence": 0.30
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
"""Unit tests for optional adapter interfaces (Tier 2 / Tier 3)."""
|
||||
|
||||
from src.adapters.embeddings import LocalEmbeddingsAdapter
|
||||
from src.adapters.llm import LLMFallbackAdapter
|
||||
from src.classifier import InherenceClassifier
|
||||
from src.models import ECPSnapshot
|
||||
|
||||
|
||||
def test_embeddings_adapter_interface():
|
||||
adapter = LocalEmbeddingsAdapter()
|
||||
assert isinstance(adapter.is_available(), bool)
|
||||
assert adapter.evaluate_similarity("test text", ["term1", "term2"]) == 0.0
|
||||
|
||||
|
||||
def test_llm_adapter_interface():
|
||||
adapter = LLMFallbackAdapter()
|
||||
assert isinstance(adapter.is_available(), bool)
|
||||
|
||||
|
||||
def test_classifier_with_adapter_flags():
|
||||
classifier = InherenceClassifier(enable_embeddings=True, enable_llm=True)
|
||||
assert classifier._embeddings_adapter is not None
|
||||
assert classifier._llm_adapter is not None
|
||||
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ent_test",
|
||||
target_name="TestCorp",
|
||||
aliases=["TestCorp"],
|
||||
domain="Tech",
|
||||
anchors=["software"]
|
||||
)
|
||||
res = classifier.classify(ecp, "TestCorp builds enterprise cloud software.")
|
||||
assert res.decision.value == "DIRECT_INHERENT"
|
||||
assert res.is_inherent is True
|
||||
@@ -0,0 +1,214 @@
|
||||
"""Adversarial and robustness test suite for Multilingual NLP Entity Inherence Classifier.
|
||||
|
||||
Validates homonym disambiguation, isolated related entities, edge cases,
|
||||
and CLI execution behavior via subprocess (exit codes, stream purity, JSON parsing).
|
||||
"""
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
import pytest
|
||||
from src.models import ECPSnapshot, RelatedEntity, DecisionCategory
|
||||
|
||||
|
||||
def test_adversarial_sao_paulo_city_vs_fc():
|
||||
"""Content about city/state governance of São Paulo against ECP for São Paulo FC."""
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ent_spfc",
|
||||
target_name="São Paulo Futebol Clube",
|
||||
aliases=["São Paulo", "SPFC", "Tricolor Paulista"],
|
||||
domain="Futebol e Esportes",
|
||||
anchors=["Morumbi", "futebol", "campeonato", "Copa Libertadores", "elenco", "estádio"],
|
||||
negative_anchors=["prefeitura de são paulo", "governo do estado de são paulo", "trânsito na capital paulista"],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[]
|
||||
)
|
||||
content = (
|
||||
"# Obras Viárias na Capital\n\n"
|
||||
"A prefeitura de São Paulo anunciou novas intervenções no trânsito na capital paulista "
|
||||
"para desafogar o fluxo de veículos na região central durante os horários de pico."
|
||||
)
|
||||
from src.classifier import InherenceClassifier
|
||||
classifier = InherenceClassifier()
|
||||
result = classifier.classify(ecp, content)
|
||||
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
|
||||
assert result.is_inherent is False
|
||||
assert result.decision != DecisionCategory.DIRECT_INHERENT
|
||||
|
||||
|
||||
def test_adversarial_apple_fruit_recipe():
|
||||
"""Content about apple fruit/culinary recipe against Apple Inc. tech entity."""
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ent_apple",
|
||||
target_name="Apple",
|
||||
aliases=["Apple Inc.", "Apple"],
|
||||
domain="Technology",
|
||||
anchors=["iPhone", "MacBook", "iOS", "silicon", "hardware"],
|
||||
negative_anchors=["apple pie", "orchard harvest", "doce de maçã"],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[]
|
||||
)
|
||||
content = (
|
||||
"# Receita Caseira\n\n"
|
||||
"Comprei maçãs frescas no mercado para preparar um doce de maçã com canela e açúcar mascavo."
|
||||
)
|
||||
from src.classifier import InherenceClassifier
|
||||
classifier = InherenceClassifier()
|
||||
result = classifier.classify(ecp, content)
|
||||
assert result.decision in (DecisionCategory.NOT_RELATED, DecisionCategory.TANGENTIAL)
|
||||
assert result.is_inherent is False
|
||||
|
||||
|
||||
def test_adversarial_related_entity_without_scope_context():
|
||||
"""High-weight related entity mentioned in passing without required domain anchors."""
|
||||
ecp = ECPSnapshot(
|
||||
target_entity_id="ent_volkswagen",
|
||||
target_name="Volkswagen",
|
||||
aliases=["Volkswagen AG", "VW"],
|
||||
domain="Automotive & Electric Vehicles",
|
||||
anchors=["Elektrofahrzeuge", "Batteriezellen", "Fahrzeugproduktion"],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[
|
||||
RelatedEntity(
|
||||
entity_id="ent_northvolt",
|
||||
name="Northvolt",
|
||||
relation_type="SUPPLIER_OF",
|
||||
weight=0.95,
|
||||
scope="battery_technology",
|
||||
confidence=0.99
|
||||
)
|
||||
]
|
||||
)
|
||||
# Content mentions Northvolt in an unrelated/passing architectural context without domain anchors
|
||||
content = (
|
||||
"# Architekturbericht aus Stockholm\n\n"
|
||||
"Während unseres Stadtrundgangs besuchten wir das neue Bürogebäude von Northvolt "
|
||||
"mit moderner Holzfassade und Blick auf den See."
|
||||
)
|
||||
from src.classifier import InherenceClassifier
|
||||
classifier = InherenceClassifier()
|
||||
result = classifier.classify(ecp, content)
|
||||
# Must be TANGENTIAL or NOT_RELATED, NEVER CONTEXTUAL_INHERENT
|
||||
assert result.decision in (DecisionCategory.TANGENTIAL, DecisionCategory.NOT_RELATED)
|
||||
assert result.is_inherent is False
|
||||
assert result.decision != DecisionCategory.CONTEXTUAL_INHERENT
|
||||
|
||||
|
||||
def test_adversarial_subprocess_cli_success_stdout(tmp_path):
|
||||
"""Run CLI via subprocess without --output and verify stdout is pure parseable JSON."""
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_petrobras",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"]
|
||||
}), encoding="utf-8")
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.", encoding="utf-8")
|
||||
|
||||
import os
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
)
|
||||
|
||||
assert res.returncode == 0
|
||||
# Stdout must be directly parseable as JSON without extraneous log text
|
||||
assert res.stdout is not None and len(res.stdout.strip()) > 0
|
||||
parsed = json.loads(res.stdout)
|
||||
assert parsed["decision"] == "DIRECT_INHERENT"
|
||||
assert parsed["is_inherent"] is True
|
||||
assert parsed["confidence"] >= 0.85
|
||||
assert len(parsed["evidence"]) > 0
|
||||
|
||||
|
||||
def test_adversarial_subprocess_cli_empty_content(tmp_path):
|
||||
"""Run CLI via subprocess with empty content and verify error code and exit code."""
|
||||
import os
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Test",
|
||||
"aliases": ["Test"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["tech"]
|
||||
}), encoding="utf-8")
|
||||
|
||||
content_file = tmp_path / "empty.md"
|
||||
content_file.write_text(" \n\n ", encoding="utf-8")
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
)
|
||||
|
||||
assert res.returncode != 0
|
||||
# Stderr must contain pure parseable error JSON
|
||||
parsed_err = json.loads(res.stderr)
|
||||
assert parsed_err["error_code"] == "empty_content"
|
||||
|
||||
|
||||
def test_adversarial_subprocess_cli_missing_required_field(tmp_path):
|
||||
"""Run CLI via subprocess with missing target_name and verify error payload."""
|
||||
import os
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
ecp_file = tmp_path / "ecp_bad.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"aliases": ["Test"],
|
||||
"domain": "Tech",
|
||||
"anchors": ["tech"]
|
||||
}), encoding="utf-8")
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Conteúdo de teste válido.", encoding="utf-8")
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
)
|
||||
|
||||
assert res.returncode != 0
|
||||
parsed_err = json.loads(res.stderr)
|
||||
assert parsed_err["error_code"] == "missing_required_field"
|
||||
|
||||
|
||||
def test_adversarial_subprocess_cli_corrupted_json(tmp_path):
|
||||
"""Run CLI via subprocess with corrupted JSON and verify error payload."""
|
||||
import os
|
||||
env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
|
||||
|
||||
ecp_file = tmp_path / "ecp_corrupted.json"
|
||||
ecp_file.write_text("{ target_entity_id: not_valid_json }", encoding="utf-8")
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Conteúdo de teste válido.", encoding="utf-8")
|
||||
|
||||
res = subprocess.run(
|
||||
[sys.executable, "classify.py", "--ecp", str(ecp_file), "--content", str(content_file)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
env=env,
|
||||
)
|
||||
|
||||
assert res.returncode != 0
|
||||
parsed_err = json.loads(res.stderr)
|
||||
assert parsed_err["error_code"] == "invalid_ecp_json"
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Controlled 24-case benchmark suite for Multilingual NLP Entity Inherence Classifier.
|
||||
|
||||
Matrix: 6 Languages (PT, EN, ES, DE, IT, FR) x 4 Decisions (DIRECT, CONTEXTUAL, TANGENTIAL, NOT_RELATED).
|
||||
Target Success Criterion: Precision >= 90% over the 24 cases.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
import pytest
|
||||
from src.models import ECPSnapshot, DecisionCategory
|
||||
from src.classifier import InherenceClassifier
|
||||
|
||||
FIXTURES_DIR = Path(__file__).parent / "fixtures" / "benchmark_24"
|
||||
LANGUAGES = ["pt", "en", "es", "de", "it", "fr"]
|
||||
DECISION_TYPES = ["direct", "contextual", "tangential", "not_related"]
|
||||
|
||||
BENCHMARK_CASES = [
|
||||
(lang, dec_type)
|
||||
for lang in LANGUAGES
|
||||
for dec_type in DECISION_TYPES
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def classifier():
|
||||
return InherenceClassifier()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("lang,dec_type", BENCHMARK_CASES)
|
||||
def test_benchmark_case(classifier, lang: str, dec_type: str):
|
||||
case_dir = FIXTURES_DIR / lang
|
||||
ecp_file = case_dir / "ecp.json"
|
||||
content_file = case_dir / f"{dec_type}.md"
|
||||
expected_file = case_dir / f"{dec_type}_expected.json"
|
||||
|
||||
assert ecp_file.is_file(), f"Missing ECP fixture: {ecp_file}"
|
||||
assert content_file.is_file(), f"Missing Content fixture: {content_file}"
|
||||
assert expected_file.is_file(), f"Missing Expected fixture: {expected_file}"
|
||||
|
||||
ecp = ECPSnapshot.from_json_str(ecp_file.read_text(encoding="utf-8"))
|
||||
content = content_file.read_text(encoding="utf-8")
|
||||
expected = json.loads(expected_file.read_text(encoding="utf-8"))
|
||||
|
||||
result = classifier.classify(ecp, content)
|
||||
|
||||
# 1. Decision category validation
|
||||
assert (
|
||||
result.decision.value == expected["expected_decision"]
|
||||
), f"[{lang.upper()} - {dec_type}] Expected {expected['expected_decision']}, got {result.decision.value}. Rationale: {result.rationale}"
|
||||
|
||||
# 2. Derived is_inherent boolean validation
|
||||
assert (
|
||||
result.is_inherent == expected["expected_is_inherent"]
|
||||
), f"[{lang.upper()} - {dec_type}] Expected is_inherent={expected['expected_is_inherent']}, got {result.is_inherent}"
|
||||
|
||||
# 3. Language detection validation
|
||||
assert (
|
||||
result.detected_language == expected["expected_language"]
|
||||
), f"[{lang.upper()} - {dec_type}] Expected language '{expected['expected_language']}', got '{result.detected_language}'"
|
||||
|
||||
# 4. Confidence threshold validation
|
||||
min_conf = expected.get("min_confidence", 0.0)
|
||||
assert (
|
||||
result.confidence >= min_conf
|
||||
), f"[{lang.upper()} - {dec_type}] Expected confidence >= {min_conf}, got {result.confidence}"
|
||||
|
||||
# 5. Evidence presence for inherent content
|
||||
if result.is_inherent:
|
||||
assert (
|
||||
len(result.evidence) > 0
|
||||
), f"[{lang.upper()} - {dec_type}] Inherent decision must have non-empty evidence snippets"
|
||||
@@ -0,0 +1,99 @@
|
||||
"""Unit tests for deterministic classification decision logic."""
|
||||
|
||||
import pytest
|
||||
from src.models import ECPSnapshot, RelatedEntity, DecisionCategory
|
||||
from src.classifier import InherenceClassifier
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def petrobras_ecp():
|
||||
return ECPSnapshot(
|
||||
target_entity_id="ent_petrobras",
|
||||
target_name="Petrobras",
|
||||
aliases=["Petróleo Brasileiro S.A.", "Petrobras", "Petrobrás"],
|
||||
domain="Oil & Gas",
|
||||
anchors=["pré-sal", "refinaria", "combustíveis", "petróleo", "exploração"],
|
||||
negative_anchors=["posto de combustíveis pirata"],
|
||||
graph_version="1.0.0",
|
||||
related_entities=[
|
||||
RelatedEntity(
|
||||
entity_id="ent_transpetro",
|
||||
name="Transpetro",
|
||||
relation_type="SUBSIDIARY_OF",
|
||||
weight=0.85,
|
||||
aliases=[],
|
||||
scope="logistics",
|
||||
confidence=1.0
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def test_direct_inherent(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Expansão da Produção Nacional\n\n"
|
||||
"A Petrobras anunciou um aumento expressivo na produção de petróleo na camada pré-sal. "
|
||||
"Os investimentos em novas plataformas devem acelerar a exploração offshore."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.DIRECT_INHERENT
|
||||
assert result.is_inherent is True
|
||||
assert result.confidence >= 0.85
|
||||
assert result.detected_language == "pt"
|
||||
assert "Petrobras" in result.matched_anchors
|
||||
assert len(result.evidence) > 0
|
||||
|
||||
|
||||
def test_contextual_inherent(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Logística de Combustíveis no Brasil\n\n"
|
||||
"A Transpetro ampliou a sua frota de navios para o transporte de combustíveis e derivados "
|
||||
"pelo litoral brasileiro, reforçando a infraestrutura energética."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.CONTEXTUAL_INHERENT
|
||||
assert result.is_inherent is True
|
||||
assert result.confidence >= 0.70
|
||||
assert len(result.graph_matches) == 1
|
||||
assert result.graph_matches[0]["name"] == "Transpetro"
|
||||
|
||||
|
||||
def test_tangential_inherent(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Crônica de Viagem pelo Interior\n\n"
|
||||
"Passamos perto de um prédio da Petrobras enquanto procurávamos um café na praça central. "
|
||||
"A tarde estava quente e os pássaros cantavam nas árvores antigas."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.TANGENTIAL
|
||||
assert result.is_inherent is False
|
||||
assert result.confidence < 0.60
|
||||
assert len(result.warnings) > 0
|
||||
|
||||
|
||||
def test_not_related(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Como Fazer Bolo de Cenoura com Cobertura de Chocolate\n\n"
|
||||
"Bata as cenouras raladas no liquidificador com os ovos e o óleo. "
|
||||
"Acrescente a farinha de trigo e o açúcar aos poucos até obter uma massa homogênea."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.NOT_RELATED
|
||||
assert result.is_inherent is False
|
||||
assert result.confidence >= 0.85
|
||||
|
||||
|
||||
def test_negative_anchor_suppression(petrobras_ecp):
|
||||
classifier = InherenceClassifier()
|
||||
content = (
|
||||
"# Operação Policial Fecha Estabelecimento\n\n"
|
||||
"A polícia interditou um posto de combustíveis pirata na rodovia estadual por adulteração."
|
||||
)
|
||||
result = classifier.classify(petrobras_ecp, content)
|
||||
assert result.decision == DecisionCategory.NOT_RELATED
|
||||
assert result.is_inherent is False
|
||||
assert len(result.negative_matches) > 0
|
||||
@@ -0,0 +1,105 @@
|
||||
"""CLI execution tests covering flags, arguments, stdout, and error handling."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
import pytest
|
||||
from classify import main
|
||||
|
||||
|
||||
def test_cli_success_stdout(tmp_path, capsys):
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"]
|
||||
}), encoding="utf-8")
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("# Notícia\n\nA Petrobras bateu recorde de extração de petróleo no pré-sal este mês.", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
|
||||
assert exit_code == 0
|
||||
|
||||
captured = capsys.readouterr()
|
||||
result = json.loads(captured.out)
|
||||
assert result["decision"] == "DIRECT_INHERENT"
|
||||
assert result["is_inherent"] is True
|
||||
assert result["detected_language"] == "pt"
|
||||
|
||||
|
||||
def test_cli_output_file(tmp_path):
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo", "pré-sal"]
|
||||
}), encoding="utf-8")
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Petrobras anunciou investimentos bilionários em novas refinarias de petróleo.", encoding="utf-8")
|
||||
|
||||
output_file = tmp_path / "out" / "result.json"
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file), "--output", str(output_file)])
|
||||
assert exit_code == 0
|
||||
assert output_file.exists()
|
||||
|
||||
result = json.loads(output_file.read_text(encoding="utf-8"))
|
||||
assert result["decision"] == "DIRECT_INHERENT"
|
||||
assert result["is_inherent"] is True
|
||||
|
||||
|
||||
def test_cli_missing_ecp_file(tmp_path, capsys):
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Algum conteúdo válido aqui.", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(tmp_path / "non_existent.json"), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err = json.loads(captured.err)
|
||||
assert err["error_code"] == "invalid_ecp_json"
|
||||
|
||||
|
||||
def test_cli_empty_content_file(tmp_path, capsys):
|
||||
ecp_file = tmp_path / "ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo"]
|
||||
}), encoding="utf-8")
|
||||
|
||||
content_file = tmp_path / "empty.md"
|
||||
content_file.write_text(" \n\n ", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err = json.loads(captured.err)
|
||||
assert err["error_code"] == "empty_content"
|
||||
|
||||
|
||||
def test_cli_missing_required_ecp_field(tmp_path, capsys):
|
||||
ecp_file = tmp_path / "bad_ecp.json"
|
||||
ecp_file.write_text(json.dumps({
|
||||
"target_entity_id": "ent_1",
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["petróleo"]
|
||||
}), encoding="utf-8")
|
||||
|
||||
content_file = tmp_path / "content.md"
|
||||
content_file.write_text("Algum conteúdo válido para testar o erro.", encoding="utf-8")
|
||||
|
||||
exit_code = main(["--ecp", str(ecp_file), "--content", str(content_file)])
|
||||
assert exit_code == 1
|
||||
|
||||
captured = capsys.readouterr()
|
||||
err = json.loads(captured.err)
|
||||
assert err["error_code"] == "missing_required_field"
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Unit tests for language detection and text normalization."""
|
||||
|
||||
from src.language import detect_language, normalize_text, SUPPORTED_LANGUAGES
|
||||
|
||||
|
||||
def test_normalize_text():
|
||||
assert normalize_text("São Paulo & Petróleo") == "sao paulo & petroleo"
|
||||
assert normalize_text("Über große Veränderungen") == "uber grosse veranderungen" or "uber" in normalize_text("Über")
|
||||
assert normalize_text("Crème brûlée") == "creme brulee"
|
||||
|
||||
|
||||
def test_detect_portuguese():
|
||||
text = "A Petrobras anunciou um novo plano de investimentos para a exploração de petróleo na camada pré-sal."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "pt"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_english():
|
||||
text = "Apple announced its new silicon chip with improved machine learning performance and battery life."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "en"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_spanish():
|
||||
text = "La empresa petrolera anunció una nueva inversión en el sector energético durante este año."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "es"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_german():
|
||||
text = "Volkswagen plant eine umfassende Transformation zur Elektromobilität in den kommenden Jahren."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "de"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_italian():
|
||||
text = "La Ferrari ha presentato la nuova vettura da competizione per il campionato mondiale di Formula 1."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "it"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_detect_french():
|
||||
text = "Le groupe TotalEnergies a confirmé ses nouveaux projets de développement dans les énergies renouvelables."
|
||||
lang, conf = detect_language(text)
|
||||
assert lang == "fr"
|
||||
assert conf > 0.5
|
||||
|
||||
|
||||
def test_empty_language():
|
||||
lang, conf = detect_language("")
|
||||
assert lang == "unknown"
|
||||
assert conf == 0.0
|
||||
@@ -0,0 +1,114 @@
|
||||
"""Unit tests for ECP models, schema validation, and structured error handling."""
|
||||
|
||||
import pytest
|
||||
from src.models import (
|
||||
ECPSnapshot,
|
||||
RelatedEntity,
|
||||
ClassificationResult,
|
||||
ClassificationError,
|
||||
DecisionCategory,
|
||||
ErrorCode,
|
||||
)
|
||||
from src.parser import strip_markdown, extract_sentences, extract_evidence_snippets
|
||||
|
||||
|
||||
def test_ecp_snapshot_valid():
|
||||
data = {
|
||||
"target_entity_id": "ent_123",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petróleo Brasileiro S.A.", "Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["pré-sal", "refinaria", "combustíveis"],
|
||||
"negative_anchors": ["petrobras posto pirata"],
|
||||
"graph_version": "1.0.0",
|
||||
"related_entities": [
|
||||
{
|
||||
"entity_id": "ent_456",
|
||||
"name": "Transpetro",
|
||||
"relation_type": "SUBSIDIARY_OF",
|
||||
"weight": 0.9,
|
||||
"aliases": ["Transpetro Logística"],
|
||||
"scope": "logistics",
|
||||
"confidence": 0.95
|
||||
}
|
||||
]
|
||||
}
|
||||
snapshot = ECPSnapshot.from_dict(data)
|
||||
assert snapshot.target_entity_id == "ent_123"
|
||||
assert snapshot.target_name == "Petrobras"
|
||||
assert len(snapshot.aliases) == 2
|
||||
assert len(snapshot.related_entities) == 1
|
||||
assert snapshot.related_entities[0].name == "Transpetro"
|
||||
assert snapshot.related_entities[0].weight == 0.9
|
||||
|
||||
|
||||
def test_ecp_snapshot_defaults():
|
||||
data = {
|
||||
"target_entity_id": "ent_123",
|
||||
"target_name": "Petrobras",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["energia"]
|
||||
}
|
||||
snapshot = ECPSnapshot.from_dict(data)
|
||||
assert snapshot.negative_anchors == []
|
||||
assert snapshot.graph_version == "1.0.0"
|
||||
assert snapshot.related_entities == []
|
||||
|
||||
|
||||
def test_ecp_snapshot_missing_required():
|
||||
data = {
|
||||
"target_entity_id": "ent_123",
|
||||
"aliases": ["Petrobras"],
|
||||
"domain": "Oil & Gas",
|
||||
"anchors": ["energia"]
|
||||
}
|
||||
with pytest.raises(ValueError, match="Missing required field"):
|
||||
ECPSnapshot.from_dict(data)
|
||||
|
||||
|
||||
def test_classification_result_serialization():
|
||||
res = ClassificationResult(
|
||||
decision=DecisionCategory.DIRECT_INHERENT,
|
||||
is_inherent=True,
|
||||
confidence=0.95,
|
||||
detected_language="pt",
|
||||
matched_anchors=["Petrobras"],
|
||||
evidence=["Petrobras anunciou investimentos no pré-sal."],
|
||||
rationale="Match forte da entidade alvo.",
|
||||
)
|
||||
d = res.to_dict()
|
||||
assert d["decision"] == "DIRECT_INHERENT"
|
||||
assert d["is_inherent"] is True
|
||||
assert d["confidence"] == 0.95
|
||||
assert d["detected_language"] == "pt"
|
||||
assert "Petrobras" in d["matched_anchors"]
|
||||
|
||||
|
||||
def test_classification_error_serialization():
|
||||
err = ClassificationError(
|
||||
error_code=ErrorCode.INVALID_ECP_JSON,
|
||||
message="Malformed JSON syntax",
|
||||
details={"path": "snapshot.json"}
|
||||
)
|
||||
d = err.to_dict()
|
||||
assert d["error_code"] == "invalid_ecp_json"
|
||||
assert d["message"] == "Malformed JSON syntax"
|
||||
assert d["details"]["path"] == "snapshot.json"
|
||||
|
||||
|
||||
def test_parser_strip_markdown():
|
||||
md = "# Title\n\nThis is **bold** text and [link](https://example.com).\n- item 1\n- item 2"
|
||||
plain = strip_markdown(md)
|
||||
assert "Title" in plain
|
||||
assert "bold text" in plain
|
||||
assert "link" in plain
|
||||
assert "[" not in plain
|
||||
assert "*" not in plain
|
||||
|
||||
|
||||
def test_extract_evidence_snippets():
|
||||
md = "O pré-sal brasileiro é uma das maiores reservas de petróleo. A Petrobras lidera a exploração técnica."
|
||||
snippets = extract_evidence_snippets(md, ["Petrobras"])
|
||||
assert len(snippets) > 0
|
||||
assert "Petrobras lidera" in snippets[0]
|
||||
Reference in New Issue
Block a user