feat(extractor): implement multi-engine article content extractor

- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability)
- Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing)
- Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts
- Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
2026-08-20 19:22:20 -03:00
parent 6e3d57619b
commit 6a45368cb0
85 changed files with 18345 additions and 3897 deletions
+4 -7
View File
@@ -3,8 +3,8 @@
from __future__ import annotations
from abc import ABC, abstractmethod
from typing import Any, Dict, List, Optional
from src.models import ECPSnapshot, ClassificationResult
from src.models import ClassificationResult, ECPSnapshot
class BaseNLPAdapter(ABC):
@@ -13,12 +13,10 @@ class BaseNLPAdapter(ABC):
@abstractmethod
def is_available(self) -> bool:
"""Return True if the underlying provider or model is installed and configured."""
pass
@abstractmethod
def evaluate_similarity(self, text: str, terms: List[str]) -> float:
def evaluate_similarity(self, text: str, terms: list[str]) -> float:
"""Compute semantic similarity score between text and a set of candidate terms."""
pass
@abstractmethod
def disambiguate(
@@ -26,6 +24,5 @@ class BaseNLPAdapter(ABC):
ecp: ECPSnapshot,
content_md: str,
initial_result: ClassificationResult,
) -> Optional[ClassificationResult]:
) -> ClassificationResult | None:
"""Optionally refine an ambiguous classification result."""
pass
+7 -5
View File
@@ -6,15 +6,16 @@ without requiring sentence-transformers to be pre-installed in the core environm
from __future__ import annotations
from typing import List, Optional
from src.models import ECPSnapshot, ClassificationResult
from src.adapters.base import BaseNLPAdapter
from src.models import ClassificationResult, ECPSnapshot
class LocalEmbeddingsAdapter(BaseNLPAdapter):
"""Optional adapter for local multilingual semantic vector embeddings."""
def __init__(self, model_name: str = "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2") -> None:
def __init__(
self, model_name: str = "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2"
) -> None:
self.model_name = model_name
self._model = None
self._initialized = False
@@ -22,11 +23,12 @@ class LocalEmbeddingsAdapter(BaseNLPAdapter):
def is_available(self) -> bool:
try:
import sentence_transformers # noqa: F401
return True
except ImportError:
return False
def evaluate_similarity(self, text: str, terms: List[str]) -> float:
def evaluate_similarity(self, text: str, terms: list[str]) -> float:
if not self.is_available() or not terms:
return 0.0
# Placeholder stub for local embedding computation
@@ -37,6 +39,6 @@ class LocalEmbeddingsAdapter(BaseNLPAdapter):
ecp: ECPSnapshot,
content_md: str,
initial_result: ClassificationResult,
) -> Optional[ClassificationResult]:
) -> ClassificationResult | None:
# Embeddings adapter does not alter decisions in POC unless explicitly wired
return None
+5 -5
View File
@@ -7,22 +7,22 @@ without requiring OpenAI/Anthropic/Gemini API keys for core POC execution.
from __future__ import annotations
import os
from typing import List, Optional
from src.models import ECPSnapshot, ClassificationResult
from src.adapters.base import BaseNLPAdapter
from src.models import ClassificationResult, ECPSnapshot
class LLMFallbackAdapter(BaseNLPAdapter):
"""Optional adapter for LLM fallback boundary disambiguation."""
def __init__(self, model_name: str = "gpt-4o-mini", api_key: Optional[str] = None) -> None:
def __init__(self, model_name: str = "gpt-4o-mini", api_key: str | None = None) -> None:
self.model_name = model_name
self.api_key = api_key or os.environ.get("OPENAI_API_KEY")
def is_available(self) -> bool:
return bool(self.api_key)
def evaluate_similarity(self, text: str, terms: List[str]) -> float:
def evaluate_similarity(self, text: str, terms: list[str]) -> float:
return 0.0
def disambiguate(
@@ -30,7 +30,7 @@ class LLMFallbackAdapter(BaseNLPAdapter):
ecp: ECPSnapshot,
content_md: str,
initial_result: ClassificationResult,
) -> Optional[ClassificationResult]:
) -> ClassificationResult | None:
# If API key is not configured or case is already clear, skip
if not self.is_available():
return None
+56 -37
View File
@@ -3,18 +3,15 @@
from __future__ import annotations
import re
from typing import Any, Dict, List, Optional, Tuple
from typing import Any
from src.models import (
ECPSnapshot,
RelatedEntity,
ClassificationResult,
ClassificationError,
DecisionCategory,
ErrorCode,
)
from src.language import detect_language, normalize_text
from src.parser import strip_markdown, extract_evidence_snippets
from src.models import (
ClassificationResult,
DecisionCategory,
ECPSnapshot,
)
from src.parser import extract_evidence_snippets, strip_markdown
def match_phrase_in_text(phrase: str, normalized_text: str) -> bool:
@@ -24,7 +21,7 @@ def match_phrase_in_text(phrase: str, normalized_text: str) -> bool:
norm_phrase = normalize_text(phrase)
if not norm_phrase:
return False
# Word boundary regex pattern
pattern = r"\b" + re.escape(norm_phrase) + r"\b"
return bool(re.search(pattern, normalized_text))
@@ -52,10 +49,12 @@ class InherenceClassifier:
if enable_embeddings:
from src.adapters.embeddings import LocalEmbeddingsAdapter
self._embeddings_adapter = LocalEmbeddingsAdapter()
if enable_llm:
from src.adapters.llm import LLMFallbackAdapter
self._llm_adapter = LLMFallbackAdapter()
def classify(self, ecp: ECPSnapshot, content_md: str) -> ClassificationResult:
@@ -70,7 +69,7 @@ class InherenceClassifier:
# 2. Match Target Entity & Aliases (deduplicate normalized terms)
all_target_terms = [ecp.target_name] + [a for a in ecp.aliases if a != ecp.target_name]
matched_target_terms: List[str] = []
matched_target_terms: list[str] = []
target_mention_count = 0
seen_norm_terms: set[str] = set()
@@ -85,27 +84,35 @@ class InherenceClassifier:
target_mention_count += count
# 3. Match Context Anchors
matched_anchors: List[str] = []
matched_anchors: list[str] = []
for anchor in ecp.anchors:
if match_phrase_in_text(anchor, norm_text):
matched_anchors.append(anchor)
# 4. Match Negative Anchors (homonym disambiguators)
matched_negative_anchors: List[str] = []
matched_negative_anchors: list[str] = []
for neg in ecp.negative_anchors:
if match_phrase_in_text(neg, norm_text):
matched_negative_anchors.append(neg)
# 5. Match Related Graph Entities
matched_graph_entities: List[Dict[str, Any]] = []
matched_graph_entities: list[dict[str, Any]] = []
highest_graph_weight = 0.0
for rel in ecp.related_entities:
rel_name = rel.name if hasattr(rel, "name") else rel.get("name", "")
rel_id = rel.entity_id if hasattr(rel, "entity_id") else rel.get("entity_id", "")
rel_type = rel.relation_type if hasattr(rel, "relation_type") else rel.get("relation_type", "")
rel_weight = float(rel.weight if hasattr(rel, "weight") else rel.get("weight", 1.0))
rel_scope = str(rel.scope if hasattr(rel, "scope") else rel.get("scope", "general"))
rel_aliases = rel.aliases if hasattr(rel, "aliases") else rel.get("aliases", [])
if isinstance(rel, dict):
rel_name = rel.get("name", "")
rel_id = rel.get("entity_id", "")
rel_type = rel.get("relation_type", "")
rel_weight = float(rel.get("weight", 1.0))
rel_scope = str(rel.get("scope", "general"))
rel_aliases = rel.get("aliases", [])
else:
rel_name = rel.name
rel_id = rel.entity_id
rel_type = rel.relation_type
rel_weight = float(rel.weight)
rel_scope = str(rel.scope)
rel_aliases = rel.aliases
rel_terms = [rel_name] + list(rel_aliases)
rel_matched = False
@@ -114,40 +121,47 @@ class InherenceClassifier:
rel_matched = True
break
if rel_matched:
matched_graph_entities.append({
"entity_id": rel_id,
"name": rel_name,
"relation_type": rel_type,
"weight": rel_weight,
"scope": rel_scope,
})
if rel_weight > highest_graph_weight:
highest_graph_weight = rel_weight
matched_graph_entities.append(
{
"entity_id": rel_id,
"name": rel_name,
"relation_type": rel_type,
"weight": rel_weight,
"scope": rel_scope,
}
)
highest_graph_weight = max(highest_graph_weight, rel_weight)
# 6. Evaluate Decision Rules
warnings: List[str] = []
warnings: list[str] = []
has_direct_match = len(matched_target_terms) > 0
has_negative_match = len(matched_negative_anchors) > 0
has_graph_match = len(matched_graph_entities) > 0
has_anchor_match = len(matched_anchors) > 0
# Term collection for evidence extraction
evidence_terms = matched_target_terms + [g["name"] for g in matched_graph_entities] + matched_anchors
evidence_terms = (
matched_target_terms + [g["name"] for g in matched_graph_entities] + matched_anchors
)
# Decision 1: Dominant Negative Anchors (overrides passing mentions)
if has_negative_match and (not has_anchor_match or len(matched_negative_anchors) >= len(matched_anchors)):
if has_negative_match and (
not has_anchor_match or len(matched_negative_anchors) >= len(matched_anchors)
):
decision = DecisionCategory.NOT_RELATED
is_inherent = False
confidence = 0.90
rationale = f"Negative anchor '{matched_negative_anchors[0]}' detected indicating irrelevant context or homonym."
elif has_direct_match:
# Check context density
if has_anchor_match or target_mention_count >= 2:
# Strong direct match with supporting context
decision = DecisionCategory.DIRECT_INHERENT
is_inherent = True
confidence = min(0.98, 0.85 + (0.04 * len(matched_anchors)) + (0.02 * target_mention_count))
confidence = min(
0.98, 0.85 + (0.04 * len(matched_anchors)) + (0.02 * target_mention_count)
)
rationale = (
f"Direct match of target entity '{ecp.target_name}' with strong contextual anchor density "
f"({len(matched_anchors)} anchor(s) matched)."
@@ -168,7 +182,10 @@ class InherenceClassifier:
# Contextual inherence via connected graph entity with domain anchor alignment
decision = DecisionCategory.CONTEXTUAL_INHERENT
is_inherent = True
confidence = round(min(0.95, 0.70 + (highest_graph_weight * 0.20) + (0.03 * len(matched_anchors))), 4)
confidence = round(
min(0.95, 0.70 + (highest_graph_weight * 0.20) + (0.03 * len(matched_anchors))),
4,
)
top_rel = matched_graph_entities[0]
rationale = (
f"Matched connected entity '{top_rel['name']}' ({top_rel['relation_type']}) "
@@ -191,7 +208,9 @@ class InherenceClassifier:
decision = DecisionCategory.NOT_RELATED
is_inherent = False
confidence = 0.85
rationale = "General domain topics mentioned, but target entity or related entities are absent."
rationale = (
"General domain topics mentioned, but target entity or related entities are absent."
)
else:
# Completely unrelated
+492 -55
View File
@@ -4,66 +4,458 @@ from __future__ import annotations
import re
import unicodedata
from typing import Dict, List, Set, Tuple
# Supported ISO 639-1 language codes
SUPPORTED_LANGUAGES: Set[str] = {"pt", "en", "es", "de", "it", "fr"}
SUPPORTED_LANGUAGES: set[str] = {"pt", "en", "es", "de", "it", "fr"}
# Characteristic function words / stopwords for deterministic language identification
LANGUAGE_STOPWORDS: Dict[str, Set[str]] = {
LANGUAGE_STOPWORDS: dict[str, set[str]] = {
"pt": {
"de", "a", "o", "que", "e", "do", "da", "em", "um", "para", "é", "com", "não",
"uma", "os", "no", "se", "na", "por", "mais", "as", "dos", "como", "mas", "foi",
"ao", "ele", "das", "tem", "à", "seu", "sua", "ou", "ser", "quando", "muito",
"nos", "já", "está", "eu", "também", "só", "pelo", "pela", "até", "isso", "ela",
"entre", "depois", "sem", "mesmo", "aos", "ter", "seus", "quem", "nas", "me",
"esse", "eles", "estão", "você", "tinha", "foram", "essa", "num", "nem", "suas",
"anunciou", "produção", "empresa", "mercado", "setor", "governo", "ano"
"de",
"a",
"o",
"que",
"e",
"do",
"da",
"em",
"um",
"para",
"é",
"com",
"não",
"uma",
"os",
"no",
"se",
"na",
"por",
"mais",
"as",
"dos",
"como",
"mas",
"foi",
"ao",
"ele",
"das",
"tem",
"à",
"seu",
"sua",
"ou",
"ser",
"quando",
"muito",
"nos",
"já",
"está",
"eu",
"também",
"só",
"pelo",
"pela",
"até",
"isso",
"ela",
"entre",
"depois",
"sem",
"mesmo",
"aos",
"ter",
"seus",
"quem",
"nas",
"me",
"esse",
"eles",
"estão",
"você",
"tinha",
"foram",
"essa",
"num",
"nem",
"suas",
"anunciou",
"produção",
"empresa",
"mercado",
"setor",
"governo",
"ano",
},
"en": {
"the", "be", "to", "of", "and", "a", "in", "that", "have", "i", "it", "for",
"not", "on", "with", "he", "as", "you", "do", "at", "this", "but", "his", "by",
"from", "they", "we", "say", "her", "she", "or", "an", "will", "my", "one",
"all", "would", "there", "their", "what", "so", "up", "out", "if", "about",
"who", "get", "which", "go", "me", "when", "make", "can", "like", "time", "no",
"just", "him", "know", "take", "people", "into", "year", "your", "good", "some",
"could", "them", "see", "other", "than", "then", "now", "look", "only", "come"
"the",
"be",
"to",
"of",
"and",
"a",
"in",
"that",
"have",
"i",
"it",
"for",
"not",
"on",
"with",
"he",
"as",
"you",
"do",
"at",
"this",
"but",
"his",
"by",
"from",
"they",
"we",
"say",
"her",
"she",
"or",
"an",
"will",
"my",
"one",
"all",
"would",
"there",
"their",
"what",
"so",
"up",
"out",
"if",
"about",
"who",
"get",
"which",
"go",
"me",
"when",
"make",
"can",
"like",
"time",
"no",
"just",
"him",
"know",
"take",
"people",
"into",
"year",
"your",
"good",
"some",
"could",
"them",
"see",
"other",
"than",
"then",
"now",
"look",
"only",
"come",
},
"es": {
"de", "la", "que", "el", "en", "y", "a", "los", "del", "se", "las", "por", "un",
"para", "con", "no", "una", "su", "al", "lo", "como", "más", "pero", "sus", "le",
"ya", "o", "este", "sí", "porque", "esta", "entre", "cuando", "muy", "sin", "sobre",
"también", "me", "hasta", "hay", "donde", "quien", "desde", "todo", "nos", "durante",
"todos", "uno", "les", "ni", "contra", "otros", "ese", "eso", "ante", "ellos",
"e", "esto", "mí", "antes", "algunos", "qué", "unos", "yo", "otro", "otras",
"anunció", "producción", "empresa", "mercado", "sector", "año", "gobierno"
"de",
"la",
"que",
"el",
"en",
"y",
"a",
"los",
"del",
"se",
"las",
"por",
"un",
"para",
"con",
"no",
"una",
"su",
"al",
"lo",
"como",
"más",
"pero",
"sus",
"le",
"ya",
"o",
"este",
"sí",
"porque",
"esta",
"entre",
"cuando",
"muy",
"sin",
"sobre",
"también",
"me",
"hasta",
"hay",
"donde",
"quien",
"desde",
"todo",
"nos",
"durante",
"todos",
"uno",
"les",
"ni",
"contra",
"otros",
"ese",
"eso",
"ante",
"ellos",
"e",
"esto",
"mí",
"antes",
"algunos",
"qué",
"unos",
"yo",
"otro",
"otras",
"anunció",
"producción",
"empresa",
"mercado",
"sector",
"año",
"gobierno",
},
"de": {
"der", "die", "und", "in", "den", "von", "zu", "das", "mit", "sich", "des", "auf",
"für", "ist", "im", "dem", "nicht", "ein", "eine", "als", "auch", "es", "an",
"werden", "aus", "er", "hat", "dass", "sie", "nach", "wird", "bei", "einer", "um",
"am", "sind", "noch", "wie", "einem", "über", "einen", "so", "zum", "war", "haben",
"nur", "oder", "aber", "vor", "zur", "bis", "mehr", "durch", "man", "sein", "wurde",
"sei", "prozent", "hatte", "kann", "gegen", "vom", "können", "schon", "wenn", "habe",
"seine", "ihre", "unter", "wir", "sollen", "neue", "neuen", "batteriezellen", "unternehmen"
"der",
"die",
"und",
"in",
"den",
"von",
"zu",
"das",
"mit",
"sich",
"des",
"auf",
"für",
"ist",
"im",
"dem",
"nicht",
"ein",
"eine",
"als",
"auch",
"es",
"an",
"werden",
"aus",
"er",
"hat",
"dass",
"sie",
"nach",
"wird",
"bei",
"einer",
"um",
"am",
"sind",
"noch",
"wie",
"einem",
"über",
"einen",
"so",
"zum",
"war",
"haben",
"nur",
"oder",
"aber",
"vor",
"zur",
"bis",
"mehr",
"durch",
"man",
"sein",
"wurde",
"sei",
"prozent",
"hatte",
"kann",
"gegen",
"vom",
"können",
"schon",
"wenn",
"habe",
"seine",
"ihre",
"unter",
"wir",
"sollen",
"neue",
"neuen",
"batteriezellen",
"unternehmen",
},
"it": {
"di", "e", "il", "che", "la", "a", "in", "per", "un", "del", "non", "i", "si", "da",
"le", "con", "sono", "della", "dei", "degli", "una", "al", "ma", "più", "delle",
"questo", "nel", "alla", "anche", "ha", "gli", "come", "dall", "dalla", "ed",
"se", "ci", "lo", "su", "loro", "dopo", "qualche", "nella", "uno", "mio", "tuo",
"suo", "nostro", "vostro", "loro", "stato", "stata", "tra", "fra", "mentre", "prima",
"quando", "molto", "tutto", "tutti", "tutte", "tutta", "senza", "ancora", "solo",
"azienda", "mercato", "settore", "anno", "governo", "produzione", "motori"
"di",
"e",
"il",
"che",
"la",
"a",
"in",
"per",
"un",
"del",
"non",
"i",
"si",
"da",
"le",
"con",
"sono",
"della",
"dei",
"degli",
"una",
"al",
"ma",
"più",
"delle",
"questo",
"nel",
"alla",
"anche",
"ha",
"gli",
"come",
"dall",
"dalla",
"ed",
"se",
"ci",
"lo",
"su",
"loro",
"dopo",
"qualche",
"nella",
"uno",
"mio",
"tuo",
"suo",
"nostro",
"vostro",
"stato",
"stata",
"tra",
"fra",
"mentre",
"prima",
"quando",
"molto",
"tutto",
"tutti",
"tutte",
"tutta",
"senza",
"ancora",
"solo",
"azienda",
"mercato",
"settore",
"anno",
"governo",
"produzione",
"motori",
},
"fr": {
"de", "la", "le", "et", "les", "des", "en", "un", "du", "une", "que", "est", "pour",
"qui", "dans", "a", "par", "sur", "pas", "plus", "au", "avec", "ce", "il", "sont",
"se", "ne", "son", "sa", "ses", "aux", "ou", "comme", "mais", "nous", "vous", "ils",
"leur", "y", "tout", "faire", "été", "aussi", "ces", "ont", "si", "fait", "même",
"très", "après", "sans", "sous", "entre", "deux", "bien", "chez", "autre", "autres",
"entreprise", "marché", "secteur", "année", "gouvernement", "production", "véhicules"
}
"de",
"la",
"le",
"et",
"les",
"des",
"en",
"un",
"du",
"une",
"que",
"est",
"pour",
"qui",
"dans",
"a",
"par",
"sur",
"pas",
"plus",
"au",
"avec",
"ce",
"il",
"sont",
"se",
"ne",
"son",
"sa",
"ses",
"aux",
"ou",
"comme",
"mais",
"nous",
"vous",
"ils",
"leur",
"y",
"tout",
"faire",
"été",
"aussi",
"ces",
"ont",
"si",
"fait",
"même",
"très",
"après",
"sans",
"sous",
"entre",
"deux",
"bien",
"chez",
"autre",
"autres",
"entreprise",
"marché",
"secteur",
"année",
"gouvernement",
"production",
"véhicules",
},
}
@@ -77,12 +469,12 @@ def normalize_text(text: str) -> str:
return "".join(c for c in nfd if unicodedata.category(c) != "Mn")
def extract_words(text: str) -> List[str]:
def extract_words(text: str) -> list[str]:
"""Tokenize text into lowercase alphanumeric words."""
return re.findall(r"\b\w+\b", text.lower())
def detect_language(text: str) -> Tuple[str, float]:
def detect_language(text: str) -> tuple[str, float]:
"""
Detect the ISO-639-1 language code of text among supported languages (pt, en, es, de, it, fr).
Returns (detected_language, confidence_score).
@@ -94,11 +486,10 @@ def detect_language(text: str) -> Tuple[str, float]:
if not words:
return "unknown", 0.0
total_words = len(words)
word_set = set(words)
# Score languages based on matched stopword counts
scores: Dict[str, int] = {}
scores: dict[str, int] = {}
for lang, stopwords in LANGUAGE_STOPWORDS.items():
matched = word_set.intersection(stopwords)
scores[lang] = len(matched)
@@ -110,13 +501,55 @@ def detect_language(text: str) -> Tuple[str, float]:
# Specific disambiguation rules for closely related languages (PT vs ES)
pt_exclusive = {
"não", "do", "da", "no", "na", "nos", "nas", "em", "um", "uma", "você", "são", "é", "dos",
"das", "foi", "está", "estão", "com", "pelo", "pela", "pelos", "pelas", "notícia",
"extração", "mês", "ano", "produção", "bateu"
"não",
"do",
"da",
"no",
"na",
"nos",
"nas",
"em",
"um",
"uma",
"você",
"são",
"é",
"dos",
"das",
"foi",
"está",
"estão",
"com",
"pelo",
"pela",
"pelos",
"pelas",
"notícia",
"extração",
"mês",
"ano",
"produção",
"bateu",
}
es_exclusive = {
"el", "la", "y", "del", "al", "los", "las", "su", "sus", "con", "más", "pero",
"durante", "noticia", "extracción", "mes", "año", "producción"
"el",
"la",
"y",
"del",
"al",
"los",
"las",
"su",
"sus",
"con",
"más",
"pero",
"durante",
"noticia",
"extracción",
"mes",
"año",
"producción",
}
# Normalize words to match accents cleanly
@@ -133,7 +566,11 @@ def detect_language(text: str) -> Tuple[str, float]:
elif "e" in word_set and "y" not in word_set:
pt_score += 1
if top_lang in ("pt", "es") or (top_lang == "pt" and es_score > pt_score) or (top_lang == "es" and pt_score > es_score):
if (
top_lang in ("pt", "es")
or (top_lang == "pt" and es_score > pt_score)
or (top_lang == "es" and pt_score > es_score)
):
if es_score > pt_score:
top_lang = "es"
top_matches = max(top_matches, es_score)
+30 -26
View File
@@ -3,9 +3,9 @@
from __future__ import annotations
import json
from dataclasses import dataclass, field, asdict
from dataclasses import dataclass, field
from enum import Enum
from typing import Any, Dict, List, Optional
from typing import Any
class DecisionCategory(str, Enum):
@@ -29,20 +29,20 @@ class RelatedEntity:
name: str
relation_type: str
weight: float
aliases: List[str] = field(default_factory=list)
aliases: list[str] = field(default_factory=list)
scope: str = "general"
confidence: float = 1.0
@classmethod
def from_dict(cls, data: Dict[str, Any]) -> RelatedEntity:
def from_dict(cls, data: dict[str, Any]) -> RelatedEntity:
if not isinstance(data, dict):
raise ValueError("Related entity must be a JSON object")
required = ["entity_id", "name", "relation_type", "weight"]
for req in required:
if req not in data or data[req] is None:
raise ValueError(f"Missing required field in related entity: '{req}'")
return cls(
entity_id=str(data["entity_id"]),
name=str(data["name"]),
@@ -58,36 +58,36 @@ class RelatedEntity:
class ECPSnapshot:
target_entity_id: str
target_name: str
aliases: List[str]
aliases: list[str]
domain: str
anchors: List[str]
negative_anchors: List[str] = field(default_factory=list)
anchors: list[str]
negative_anchors: list[str] = field(default_factory=list)
graph_version: str = "1.0.0"
related_entities: List[RelatedEntity] = field(default_factory=list)
related_entities: list[RelatedEntity] = field(default_factory=list)
@classmethod
def from_dict(cls, data: Dict[str, Any]) -> ECPSnapshot:
def from_dict(cls, data: dict[str, Any]) -> ECPSnapshot:
if not isinstance(data, dict):
raise ValueError("ECP Snapshot payload must be a JSON object")
required_fields = ["target_entity_id", "target_name", "aliases", "domain", "anchors"]
for field_name in required_fields:
if field_name not in data or data[field_name] is None:
raise ValueError(f"Missing required field in ECP Snapshot: '{field_name}'")
if not isinstance(data["aliases"], list):
raise ValueError("Field 'aliases' must be a list of strings")
if not isinstance(data["anchors"], list):
raise ValueError("Field 'anchors' must be a list of strings")
neg_anchors = data.get("negative_anchors", [])
if neg_anchors is not None and not isinstance(neg_anchors, list):
raise ValueError("Field 'negative_anchors' must be a list of strings if provided")
related_data = data.get("related_entities", [])
if related_data is not None and not isinstance(related_data, list):
raise ValueError("Field 'related_entities' must be a list if provided")
related_objs = [RelatedEntity.from_dict(item) for item in (related_data or [])]
return cls(
@@ -124,16 +124,18 @@ class ClassificationResult:
is_inherent: bool
confidence: float
detected_language: str
matched_anchors: List[str] = field(default_factory=list)
negative_matches: List[str] = field(default_factory=list)
graph_matches: List[Dict[str, Any]] = field(default_factory=list)
evidence: List[str] = field(default_factory=list)
matched_anchors: list[str] = field(default_factory=list)
negative_matches: list[str] = field(default_factory=list)
graph_matches: list[dict[str, Any]] = field(default_factory=list)
evidence: list[str] = field(default_factory=list)
rationale: str = ""
warnings: List[str] = field(default_factory=list)
warnings: list[str] = field(default_factory=list)
def to_dict(self) -> Dict[str, Any]:
def to_dict(self) -> dict[str, Any]:
return {
"decision": self.decision.value if isinstance(self.decision, DecisionCategory) else str(self.decision),
"decision": self.decision.value
if isinstance(self.decision, DecisionCategory)
else str(self.decision),
"is_inherent": bool(self.is_inherent),
"confidence": round(float(self.confidence), 4),
"detected_language": str(self.detected_language),
@@ -153,11 +155,13 @@ class ClassificationResult:
class ClassificationError:
error_code: ErrorCode
message: str
details: Dict[str, Any] = field(default_factory=dict)
details: dict[str, Any] = field(default_factory=dict)
def to_dict(self) -> Dict[str, Any]:
def to_dict(self) -> dict[str, Any]:
return {
"error_code": self.error_code.value if isinstance(self.error_code, ErrorCode) else str(self.error_code),
"error_code": self.error_code.value
if isinstance(self.error_code, ErrorCode)
else str(self.error_code),
"message": str(self.message),
"details": dict(self.details),
}
+5 -4
View File
@@ -3,7 +3,6 @@
from __future__ import annotations
import re
from typing import List, Tuple
def strip_markdown(markdown_text: str) -> str:
@@ -40,7 +39,7 @@ def strip_markdown(markdown_text: str) -> str:
return text
def extract_sentences(text: str) -> List[str]:
def extract_sentences(text: str) -> list[str]:
"""Split text into individual sentences."""
# Split by period, exclamation, question mark followed by space or newline
raw_sentences = re.split(r"(?<=[.!?])\s+", text.strip())
@@ -48,7 +47,9 @@ def extract_sentences(text: str) -> List[str]:
return sentences
def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_snippets: int = 3) -> List[str]:
def extract_evidence_snippets(
markdown_text: str, match_terms: list[str], max_snippets: int = 3
) -> list[str]:
"""
Extract relevant sentence excerpts from Markdown text that contain any of the given match terms.
Preserves original phrasing and formats as clean evidence.
@@ -62,7 +63,7 @@ def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_sn
sentences = [plain_text]
lower_terms = [t.lower() for t in match_terms if t]
evidence: List[str] = []
evidence: list[str] = []
for sentence in sentences:
lower_sent = sentence.lower()