feat(classifier): add multilingual ECP inherence classifier POC

This commit is contained in:
2026-08-20 00:51:02 -03:00
parent d371b81aa4
commit 67cc40f91a
175 changed files with 30399 additions and 703 deletions
+78
View File
@@ -0,0 +1,78 @@
"""Markdown content parser and excerpt extraction utilities."""
from __future__ import annotations
import re
from typing import List, Tuple
def strip_markdown(markdown_text: str) -> str:
"""Remove markdown syntax markers (headers, bold, italics, links, code blocks) to obtain plain text."""
if not markdown_text:
return ""
text = markdown_text
# Remove code blocks
text = re.sub(r"```[\s\S]*?```", " ", text)
text = re.sub(r"`[^`]*`", " ", text)
# Remove headers (# Header)
text = re.sub(r"^#+\s+", " ", text, flags=re.MULTILINE)
# Replace markdown links [anchor](url) with just anchor
text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text)
# Remove image links ![alt](url)
text = re.sub(r"!\[[^\]]*\]\([^)]+\)", " ", text)
# Remove bold/italics (*, _, **, __)
text = re.sub(r"(\*\*|__)(.*?)\1", r"\2", text)
text = re.sub(r"(\*|_)(.*?)\1", r"\2", text)
# Remove blockquotes and list markers
text = re.sub(r"^\s*[-*+]\s+", " ", text, flags=re.MULTILINE)
text = re.sub(r"^\s*\d+\.\s+", " ", text, flags=re.MULTILINE)
text = re.sub(r"^\s*>\s*", " ", text, flags=re.MULTILINE)
# Normalize whitespace
text = re.sub(r"\s+", " ", text).strip()
return text
def extract_sentences(text: str) -> List[str]:
"""Split text into individual sentences."""
# Split by period, exclamation, question mark followed by space or newline
raw_sentences = re.split(r"(?<=[.!?])\s+", text.strip())
sentences = [s.strip() for s in raw_sentences if len(s.strip()) > 3]
return sentences
def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_snippets: int = 3) -> List[str]:
"""
Extract relevant sentence excerpts from Markdown text that contain any of the given match terms.
Preserves original phrasing and formats as clean evidence.
"""
if not markdown_text or not match_terms:
return []
plain_text = strip_markdown(markdown_text)
sentences = extract_sentences(plain_text)
if not sentences:
sentences = [plain_text]
lower_terms = [t.lower() for t in match_terms if t]
evidence: List[str] = []
for sentence in sentences:
lower_sent = sentence.lower()
for term in lower_terms:
if re.search(r"\b" + re.escape(term) + r"\b", lower_sent) or term in lower_sent:
clean_snippet = sentence.strip()
if clean_snippet and clean_snippet not in evidence:
evidence.append(clean_snippet)
if len(evidence) >= max_snippets:
return evidence
break
return evidence