feat(classifier): add multilingual ECP inherence classifier POC
This commit is contained in:
@@ -0,0 +1,78 @@
|
||||
"""Markdown content parser and excerpt extraction utilities."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import List, Tuple
|
||||
|
||||
|
||||
def strip_markdown(markdown_text: str) -> str:
|
||||
"""Remove markdown syntax markers (headers, bold, italics, links, code blocks) to obtain plain text."""
|
||||
if not markdown_text:
|
||||
return ""
|
||||
|
||||
text = markdown_text
|
||||
|
||||
# Remove code blocks
|
||||
text = re.sub(r"```[\s\S]*?```", " ", text)
|
||||
text = re.sub(r"`[^`]*`", " ", text)
|
||||
|
||||
# Remove headers (# Header)
|
||||
text = re.sub(r"^#+\s+", " ", text, flags=re.MULTILINE)
|
||||
|
||||
# Replace markdown links [anchor](url) with just anchor
|
||||
text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text)
|
||||
|
||||
# Remove image links 
|
||||
text = re.sub(r"!\[[^\]]*\]\([^)]+\)", " ", text)
|
||||
|
||||
# Remove bold/italics (*, _, **, __)
|
||||
text = re.sub(r"(\*\*|__)(.*?)\1", r"\2", text)
|
||||
text = re.sub(r"(\*|_)(.*?)\1", r"\2", text)
|
||||
|
||||
# Remove blockquotes and list markers
|
||||
text = re.sub(r"^\s*[-*+]\s+", " ", text, flags=re.MULTILINE)
|
||||
text = re.sub(r"^\s*\d+\.\s+", " ", text, flags=re.MULTILINE)
|
||||
text = re.sub(r"^\s*>\s*", " ", text, flags=re.MULTILINE)
|
||||
|
||||
# Normalize whitespace
|
||||
text = re.sub(r"\s+", " ", text).strip()
|
||||
return text
|
||||
|
||||
|
||||
def extract_sentences(text: str) -> List[str]:
|
||||
"""Split text into individual sentences."""
|
||||
# Split by period, exclamation, question mark followed by space or newline
|
||||
raw_sentences = re.split(r"(?<=[.!?])\s+", text.strip())
|
||||
sentences = [s.strip() for s in raw_sentences if len(s.strip()) > 3]
|
||||
return sentences
|
||||
|
||||
|
||||
def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_snippets: int = 3) -> List[str]:
|
||||
"""
|
||||
Extract relevant sentence excerpts from Markdown text that contain any of the given match terms.
|
||||
Preserves original phrasing and formats as clean evidence.
|
||||
"""
|
||||
if not markdown_text or not match_terms:
|
||||
return []
|
||||
|
||||
plain_text = strip_markdown(markdown_text)
|
||||
sentences = extract_sentences(plain_text)
|
||||
if not sentences:
|
||||
sentences = [plain_text]
|
||||
|
||||
lower_terms = [t.lower() for t in match_terms if t]
|
||||
evidence: List[str] = []
|
||||
|
||||
for sentence in sentences:
|
||||
lower_sent = sentence.lower()
|
||||
for term in lower_terms:
|
||||
if re.search(r"\b" + re.escape(term) + r"\b", lower_sent) or term in lower_sent:
|
||||
clean_snippet = sentence.strip()
|
||||
if clean_snippet and clean_snippet not in evidence:
|
||||
evidence.append(clean_snippet)
|
||||
if len(evidence) >= max_snippets:
|
||||
return evidence
|
||||
break
|
||||
|
||||
return evidence
|
||||
Reference in New Issue
Block a user