43 lines
1.2 KiB
Python
43 lines
1.2 KiB
Python
"""Unit tests for sequence equivalence mapping covering scenarios CAN-001 to CAN-010."""
|
|
|
|
from src.runtime.candidate.equivalence import (
|
|
compute_sequence_similarity,
|
|
map_candidate_equivalences,
|
|
normalize_text_for_comparison,
|
|
)
|
|
from src.runtime.candidate.models import CandidateObject
|
|
|
|
|
|
def test_text_normalization():
|
|
text = " São Paulo Futebol Clube\n\t "
|
|
norm = normalize_text_for_comparison(text)
|
|
assert norm == "são paulo futebol clube"
|
|
|
|
|
|
def test_sequence_similarity():
|
|
t1 = "River Plate empató sin goles ante Independiente Santa Fe."
|
|
t2 = "River Plate empató 0-0 con Independiente Santa Fe."
|
|
sim = compute_sequence_similarity(t1, t2)
|
|
assert sim > 0.6
|
|
|
|
|
|
def test_map_candidate_equivalences():
|
|
c1 = CandidateObject(
|
|
id="traf_01",
|
|
type="paragraph",
|
|
text="El partido finalizó 0 a 0 en Bogotá.",
|
|
extractor="trafilatura",
|
|
position=1,
|
|
)
|
|
c2 = CandidateObject(
|
|
id="news_01",
|
|
type="paragraph",
|
|
text="El partido finalizó 0 a 0 en Bogotá.",
|
|
extractor="newspaper4k",
|
|
position=1,
|
|
)
|
|
|
|
map_candidate_equivalences([c1], [c2], similarity_threshold=0.9)
|
|
assert "news_01" in c1.equivalent_ids
|
|
assert "traf_01" in c2.equivalent_ids
|