feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
@@ -3,8 +3,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import Any, Dict, List, Optional
|
||||
from src.models import ECPSnapshot, ClassificationResult
|
||||
|
||||
from src.models import ClassificationResult, ECPSnapshot
|
||||
|
||||
|
||||
class BaseNLPAdapter(ABC):
|
||||
@@ -13,12 +13,10 @@ class BaseNLPAdapter(ABC):
|
||||
@abstractmethod
|
||||
def is_available(self) -> bool:
|
||||
"""Return True if the underlying provider or model is installed and configured."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def evaluate_similarity(self, text: str, terms: List[str]) -> float:
|
||||
def evaluate_similarity(self, text: str, terms: list[str]) -> float:
|
||||
"""Compute semantic similarity score between text and a set of candidate terms."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def disambiguate(
|
||||
@@ -26,6 +24,5 @@ class BaseNLPAdapter(ABC):
|
||||
ecp: ECPSnapshot,
|
||||
content_md: str,
|
||||
initial_result: ClassificationResult,
|
||||
) -> Optional[ClassificationResult]:
|
||||
) -> ClassificationResult | None:
|
||||
"""Optionally refine an ambiguous classification result."""
|
||||
pass
|
||||
|
||||
@@ -6,15 +6,16 @@ without requiring sentence-transformers to be pre-installed in the core environm
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List, Optional
|
||||
from src.models import ECPSnapshot, ClassificationResult
|
||||
from src.adapters.base import BaseNLPAdapter
|
||||
from src.models import ClassificationResult, ECPSnapshot
|
||||
|
||||
|
||||
class LocalEmbeddingsAdapter(BaseNLPAdapter):
|
||||
"""Optional adapter for local multilingual semantic vector embeddings."""
|
||||
|
||||
def __init__(self, model_name: str = "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2") -> None:
|
||||
def __init__(
|
||||
self, model_name: str = "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2"
|
||||
) -> None:
|
||||
self.model_name = model_name
|
||||
self._model = None
|
||||
self._initialized = False
|
||||
@@ -22,11 +23,12 @@ class LocalEmbeddingsAdapter(BaseNLPAdapter):
|
||||
def is_available(self) -> bool:
|
||||
try:
|
||||
import sentence_transformers # noqa: F401
|
||||
|
||||
return True
|
||||
except ImportError:
|
||||
return False
|
||||
|
||||
def evaluate_similarity(self, text: str, terms: List[str]) -> float:
|
||||
def evaluate_similarity(self, text: str, terms: list[str]) -> float:
|
||||
if not self.is_available() or not terms:
|
||||
return 0.0
|
||||
# Placeholder stub for local embedding computation
|
||||
@@ -37,6 +39,6 @@ class LocalEmbeddingsAdapter(BaseNLPAdapter):
|
||||
ecp: ECPSnapshot,
|
||||
content_md: str,
|
||||
initial_result: ClassificationResult,
|
||||
) -> Optional[ClassificationResult]:
|
||||
) -> ClassificationResult | None:
|
||||
# Embeddings adapter does not alter decisions in POC unless explicitly wired
|
||||
return None
|
||||
|
||||
+5
-5
@@ -7,22 +7,22 @@ without requiring OpenAI/Anthropic/Gemini API keys for core POC execution.
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from typing import List, Optional
|
||||
from src.models import ECPSnapshot, ClassificationResult
|
||||
|
||||
from src.adapters.base import BaseNLPAdapter
|
||||
from src.models import ClassificationResult, ECPSnapshot
|
||||
|
||||
|
||||
class LLMFallbackAdapter(BaseNLPAdapter):
|
||||
"""Optional adapter for LLM fallback boundary disambiguation."""
|
||||
|
||||
def __init__(self, model_name: str = "gpt-4o-mini", api_key: Optional[str] = None) -> None:
|
||||
def __init__(self, model_name: str = "gpt-4o-mini", api_key: str | None = None) -> None:
|
||||
self.model_name = model_name
|
||||
self.api_key = api_key or os.environ.get("OPENAI_API_KEY")
|
||||
|
||||
def is_available(self) -> bool:
|
||||
return bool(self.api_key)
|
||||
|
||||
def evaluate_similarity(self, text: str, terms: List[str]) -> float:
|
||||
def evaluate_similarity(self, text: str, terms: list[str]) -> float:
|
||||
return 0.0
|
||||
|
||||
def disambiguate(
|
||||
@@ -30,7 +30,7 @@ class LLMFallbackAdapter(BaseNLPAdapter):
|
||||
ecp: ECPSnapshot,
|
||||
content_md: str,
|
||||
initial_result: ClassificationResult,
|
||||
) -> Optional[ClassificationResult]:
|
||||
) -> ClassificationResult | None:
|
||||
# If API key is not configured or case is already clear, skip
|
||||
if not self.is_available():
|
||||
return None
|
||||
|
||||
+56
-37
@@ -3,18 +3,15 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
from typing import Any
|
||||
|
||||
from src.models import (
|
||||
ECPSnapshot,
|
||||
RelatedEntity,
|
||||
ClassificationResult,
|
||||
ClassificationError,
|
||||
DecisionCategory,
|
||||
ErrorCode,
|
||||
)
|
||||
from src.language import detect_language, normalize_text
|
||||
from src.parser import strip_markdown, extract_evidence_snippets
|
||||
from src.models import (
|
||||
ClassificationResult,
|
||||
DecisionCategory,
|
||||
ECPSnapshot,
|
||||
)
|
||||
from src.parser import extract_evidence_snippets, strip_markdown
|
||||
|
||||
|
||||
def match_phrase_in_text(phrase: str, normalized_text: str) -> bool:
|
||||
@@ -24,7 +21,7 @@ def match_phrase_in_text(phrase: str, normalized_text: str) -> bool:
|
||||
norm_phrase = normalize_text(phrase)
|
||||
if not norm_phrase:
|
||||
return False
|
||||
|
||||
|
||||
# Word boundary regex pattern
|
||||
pattern = r"\b" + re.escape(norm_phrase) + r"\b"
|
||||
return bool(re.search(pattern, normalized_text))
|
||||
@@ -52,10 +49,12 @@ class InherenceClassifier:
|
||||
|
||||
if enable_embeddings:
|
||||
from src.adapters.embeddings import LocalEmbeddingsAdapter
|
||||
|
||||
self._embeddings_adapter = LocalEmbeddingsAdapter()
|
||||
|
||||
if enable_llm:
|
||||
from src.adapters.llm import LLMFallbackAdapter
|
||||
|
||||
self._llm_adapter = LLMFallbackAdapter()
|
||||
|
||||
def classify(self, ecp: ECPSnapshot, content_md: str) -> ClassificationResult:
|
||||
@@ -70,7 +69,7 @@ class InherenceClassifier:
|
||||
|
||||
# 2. Match Target Entity & Aliases (deduplicate normalized terms)
|
||||
all_target_terms = [ecp.target_name] + [a for a in ecp.aliases if a != ecp.target_name]
|
||||
matched_target_terms: List[str] = []
|
||||
matched_target_terms: list[str] = []
|
||||
target_mention_count = 0
|
||||
seen_norm_terms: set[str] = set()
|
||||
|
||||
@@ -85,27 +84,35 @@ class InherenceClassifier:
|
||||
target_mention_count += count
|
||||
|
||||
# 3. Match Context Anchors
|
||||
matched_anchors: List[str] = []
|
||||
matched_anchors: list[str] = []
|
||||
for anchor in ecp.anchors:
|
||||
if match_phrase_in_text(anchor, norm_text):
|
||||
matched_anchors.append(anchor)
|
||||
|
||||
# 4. Match Negative Anchors (homonym disambiguators)
|
||||
matched_negative_anchors: List[str] = []
|
||||
matched_negative_anchors: list[str] = []
|
||||
for neg in ecp.negative_anchors:
|
||||
if match_phrase_in_text(neg, norm_text):
|
||||
matched_negative_anchors.append(neg)
|
||||
|
||||
# 5. Match Related Graph Entities
|
||||
matched_graph_entities: List[Dict[str, Any]] = []
|
||||
matched_graph_entities: list[dict[str, Any]] = []
|
||||
highest_graph_weight = 0.0
|
||||
for rel in ecp.related_entities:
|
||||
rel_name = rel.name if hasattr(rel, "name") else rel.get("name", "")
|
||||
rel_id = rel.entity_id if hasattr(rel, "entity_id") else rel.get("entity_id", "")
|
||||
rel_type = rel.relation_type if hasattr(rel, "relation_type") else rel.get("relation_type", "")
|
||||
rel_weight = float(rel.weight if hasattr(rel, "weight") else rel.get("weight", 1.0))
|
||||
rel_scope = str(rel.scope if hasattr(rel, "scope") else rel.get("scope", "general"))
|
||||
rel_aliases = rel.aliases if hasattr(rel, "aliases") else rel.get("aliases", [])
|
||||
if isinstance(rel, dict):
|
||||
rel_name = rel.get("name", "")
|
||||
rel_id = rel.get("entity_id", "")
|
||||
rel_type = rel.get("relation_type", "")
|
||||
rel_weight = float(rel.get("weight", 1.0))
|
||||
rel_scope = str(rel.get("scope", "general"))
|
||||
rel_aliases = rel.get("aliases", [])
|
||||
else:
|
||||
rel_name = rel.name
|
||||
rel_id = rel.entity_id
|
||||
rel_type = rel.relation_type
|
||||
rel_weight = float(rel.weight)
|
||||
rel_scope = str(rel.scope)
|
||||
rel_aliases = rel.aliases
|
||||
|
||||
rel_terms = [rel_name] + list(rel_aliases)
|
||||
rel_matched = False
|
||||
@@ -114,40 +121,47 @@ class InherenceClassifier:
|
||||
rel_matched = True
|
||||
break
|
||||
if rel_matched:
|
||||
matched_graph_entities.append({
|
||||
"entity_id": rel_id,
|
||||
"name": rel_name,
|
||||
"relation_type": rel_type,
|
||||
"weight": rel_weight,
|
||||
"scope": rel_scope,
|
||||
})
|
||||
if rel_weight > highest_graph_weight:
|
||||
highest_graph_weight = rel_weight
|
||||
matched_graph_entities.append(
|
||||
{
|
||||
"entity_id": rel_id,
|
||||
"name": rel_name,
|
||||
"relation_type": rel_type,
|
||||
"weight": rel_weight,
|
||||
"scope": rel_scope,
|
||||
}
|
||||
)
|
||||
highest_graph_weight = max(highest_graph_weight, rel_weight)
|
||||
|
||||
# 6. Evaluate Decision Rules
|
||||
warnings: List[str] = []
|
||||
warnings: list[str] = []
|
||||
has_direct_match = len(matched_target_terms) > 0
|
||||
has_negative_match = len(matched_negative_anchors) > 0
|
||||
has_graph_match = len(matched_graph_entities) > 0
|
||||
has_anchor_match = len(matched_anchors) > 0
|
||||
|
||||
# Term collection for evidence extraction
|
||||
evidence_terms = matched_target_terms + [g["name"] for g in matched_graph_entities] + matched_anchors
|
||||
evidence_terms = (
|
||||
matched_target_terms + [g["name"] for g in matched_graph_entities] + matched_anchors
|
||||
)
|
||||
|
||||
# Decision 1: Dominant Negative Anchors (overrides passing mentions)
|
||||
if has_negative_match and (not has_anchor_match or len(matched_negative_anchors) >= len(matched_anchors)):
|
||||
if has_negative_match and (
|
||||
not has_anchor_match or len(matched_negative_anchors) >= len(matched_anchors)
|
||||
):
|
||||
decision = DecisionCategory.NOT_RELATED
|
||||
is_inherent = False
|
||||
confidence = 0.90
|
||||
rationale = f"Negative anchor '{matched_negative_anchors[0]}' detected indicating irrelevant context or homonym."
|
||||
|
||||
|
||||
elif has_direct_match:
|
||||
# Check context density
|
||||
if has_anchor_match or target_mention_count >= 2:
|
||||
# Strong direct match with supporting context
|
||||
decision = DecisionCategory.DIRECT_INHERENT
|
||||
is_inherent = True
|
||||
confidence = min(0.98, 0.85 + (0.04 * len(matched_anchors)) + (0.02 * target_mention_count))
|
||||
confidence = min(
|
||||
0.98, 0.85 + (0.04 * len(matched_anchors)) + (0.02 * target_mention_count)
|
||||
)
|
||||
rationale = (
|
||||
f"Direct match of target entity '{ecp.target_name}' with strong contextual anchor density "
|
||||
f"({len(matched_anchors)} anchor(s) matched)."
|
||||
@@ -168,7 +182,10 @@ class InherenceClassifier:
|
||||
# Contextual inherence via connected graph entity with domain anchor alignment
|
||||
decision = DecisionCategory.CONTEXTUAL_INHERENT
|
||||
is_inherent = True
|
||||
confidence = round(min(0.95, 0.70 + (highest_graph_weight * 0.20) + (0.03 * len(matched_anchors))), 4)
|
||||
confidence = round(
|
||||
min(0.95, 0.70 + (highest_graph_weight * 0.20) + (0.03 * len(matched_anchors))),
|
||||
4,
|
||||
)
|
||||
top_rel = matched_graph_entities[0]
|
||||
rationale = (
|
||||
f"Matched connected entity '{top_rel['name']}' ({top_rel['relation_type']}) "
|
||||
@@ -191,7 +208,9 @@ class InherenceClassifier:
|
||||
decision = DecisionCategory.NOT_RELATED
|
||||
is_inherent = False
|
||||
confidence = 0.85
|
||||
rationale = "General domain topics mentioned, but target entity or related entities are absent."
|
||||
rationale = (
|
||||
"General domain topics mentioned, but target entity or related entities are absent."
|
||||
)
|
||||
|
||||
else:
|
||||
# Completely unrelated
|
||||
|
||||
+492
-55
@@ -4,66 +4,458 @@ from __future__ import annotations
|
||||
|
||||
import re
|
||||
import unicodedata
|
||||
from typing import Dict, List, Set, Tuple
|
||||
|
||||
# Supported ISO 639-1 language codes
|
||||
SUPPORTED_LANGUAGES: Set[str] = {"pt", "en", "es", "de", "it", "fr"}
|
||||
SUPPORTED_LANGUAGES: set[str] = {"pt", "en", "es", "de", "it", "fr"}
|
||||
|
||||
# Characteristic function words / stopwords for deterministic language identification
|
||||
LANGUAGE_STOPWORDS: Dict[str, Set[str]] = {
|
||||
LANGUAGE_STOPWORDS: dict[str, set[str]] = {
|
||||
"pt": {
|
||||
"de", "a", "o", "que", "e", "do", "da", "em", "um", "para", "é", "com", "não",
|
||||
"uma", "os", "no", "se", "na", "por", "mais", "as", "dos", "como", "mas", "foi",
|
||||
"ao", "ele", "das", "tem", "à", "seu", "sua", "ou", "ser", "quando", "muito",
|
||||
"nos", "já", "está", "eu", "também", "só", "pelo", "pela", "até", "isso", "ela",
|
||||
"entre", "depois", "sem", "mesmo", "aos", "ter", "seus", "quem", "nas", "me",
|
||||
"esse", "eles", "estão", "você", "tinha", "foram", "essa", "num", "nem", "suas",
|
||||
"anunciou", "produção", "empresa", "mercado", "setor", "governo", "ano"
|
||||
"de",
|
||||
"a",
|
||||
"o",
|
||||
"que",
|
||||
"e",
|
||||
"do",
|
||||
"da",
|
||||
"em",
|
||||
"um",
|
||||
"para",
|
||||
"é",
|
||||
"com",
|
||||
"não",
|
||||
"uma",
|
||||
"os",
|
||||
"no",
|
||||
"se",
|
||||
"na",
|
||||
"por",
|
||||
"mais",
|
||||
"as",
|
||||
"dos",
|
||||
"como",
|
||||
"mas",
|
||||
"foi",
|
||||
"ao",
|
||||
"ele",
|
||||
"das",
|
||||
"tem",
|
||||
"à",
|
||||
"seu",
|
||||
"sua",
|
||||
"ou",
|
||||
"ser",
|
||||
"quando",
|
||||
"muito",
|
||||
"nos",
|
||||
"já",
|
||||
"está",
|
||||
"eu",
|
||||
"também",
|
||||
"só",
|
||||
"pelo",
|
||||
"pela",
|
||||
"até",
|
||||
"isso",
|
||||
"ela",
|
||||
"entre",
|
||||
"depois",
|
||||
"sem",
|
||||
"mesmo",
|
||||
"aos",
|
||||
"ter",
|
||||
"seus",
|
||||
"quem",
|
||||
"nas",
|
||||
"me",
|
||||
"esse",
|
||||
"eles",
|
||||
"estão",
|
||||
"você",
|
||||
"tinha",
|
||||
"foram",
|
||||
"essa",
|
||||
"num",
|
||||
"nem",
|
||||
"suas",
|
||||
"anunciou",
|
||||
"produção",
|
||||
"empresa",
|
||||
"mercado",
|
||||
"setor",
|
||||
"governo",
|
||||
"ano",
|
||||
},
|
||||
"en": {
|
||||
"the", "be", "to", "of", "and", "a", "in", "that", "have", "i", "it", "for",
|
||||
"not", "on", "with", "he", "as", "you", "do", "at", "this", "but", "his", "by",
|
||||
"from", "they", "we", "say", "her", "she", "or", "an", "will", "my", "one",
|
||||
"all", "would", "there", "their", "what", "so", "up", "out", "if", "about",
|
||||
"who", "get", "which", "go", "me", "when", "make", "can", "like", "time", "no",
|
||||
"just", "him", "know", "take", "people", "into", "year", "your", "good", "some",
|
||||
"could", "them", "see", "other", "than", "then", "now", "look", "only", "come"
|
||||
"the",
|
||||
"be",
|
||||
"to",
|
||||
"of",
|
||||
"and",
|
||||
"a",
|
||||
"in",
|
||||
"that",
|
||||
"have",
|
||||
"i",
|
||||
"it",
|
||||
"for",
|
||||
"not",
|
||||
"on",
|
||||
"with",
|
||||
"he",
|
||||
"as",
|
||||
"you",
|
||||
"do",
|
||||
"at",
|
||||
"this",
|
||||
"but",
|
||||
"his",
|
||||
"by",
|
||||
"from",
|
||||
"they",
|
||||
"we",
|
||||
"say",
|
||||
"her",
|
||||
"she",
|
||||
"or",
|
||||
"an",
|
||||
"will",
|
||||
"my",
|
||||
"one",
|
||||
"all",
|
||||
"would",
|
||||
"there",
|
||||
"their",
|
||||
"what",
|
||||
"so",
|
||||
"up",
|
||||
"out",
|
||||
"if",
|
||||
"about",
|
||||
"who",
|
||||
"get",
|
||||
"which",
|
||||
"go",
|
||||
"me",
|
||||
"when",
|
||||
"make",
|
||||
"can",
|
||||
"like",
|
||||
"time",
|
||||
"no",
|
||||
"just",
|
||||
"him",
|
||||
"know",
|
||||
"take",
|
||||
"people",
|
||||
"into",
|
||||
"year",
|
||||
"your",
|
||||
"good",
|
||||
"some",
|
||||
"could",
|
||||
"them",
|
||||
"see",
|
||||
"other",
|
||||
"than",
|
||||
"then",
|
||||
"now",
|
||||
"look",
|
||||
"only",
|
||||
"come",
|
||||
},
|
||||
"es": {
|
||||
"de", "la", "que", "el", "en", "y", "a", "los", "del", "se", "las", "por", "un",
|
||||
"para", "con", "no", "una", "su", "al", "lo", "como", "más", "pero", "sus", "le",
|
||||
"ya", "o", "este", "sí", "porque", "esta", "entre", "cuando", "muy", "sin", "sobre",
|
||||
"también", "me", "hasta", "hay", "donde", "quien", "desde", "todo", "nos", "durante",
|
||||
"todos", "uno", "les", "ni", "contra", "otros", "ese", "eso", "ante", "ellos",
|
||||
"e", "esto", "mí", "antes", "algunos", "qué", "unos", "yo", "otro", "otras",
|
||||
"anunció", "producción", "empresa", "mercado", "sector", "año", "gobierno"
|
||||
"de",
|
||||
"la",
|
||||
"que",
|
||||
"el",
|
||||
"en",
|
||||
"y",
|
||||
"a",
|
||||
"los",
|
||||
"del",
|
||||
"se",
|
||||
"las",
|
||||
"por",
|
||||
"un",
|
||||
"para",
|
||||
"con",
|
||||
"no",
|
||||
"una",
|
||||
"su",
|
||||
"al",
|
||||
"lo",
|
||||
"como",
|
||||
"más",
|
||||
"pero",
|
||||
"sus",
|
||||
"le",
|
||||
"ya",
|
||||
"o",
|
||||
"este",
|
||||
"sí",
|
||||
"porque",
|
||||
"esta",
|
||||
"entre",
|
||||
"cuando",
|
||||
"muy",
|
||||
"sin",
|
||||
"sobre",
|
||||
"también",
|
||||
"me",
|
||||
"hasta",
|
||||
"hay",
|
||||
"donde",
|
||||
"quien",
|
||||
"desde",
|
||||
"todo",
|
||||
"nos",
|
||||
"durante",
|
||||
"todos",
|
||||
"uno",
|
||||
"les",
|
||||
"ni",
|
||||
"contra",
|
||||
"otros",
|
||||
"ese",
|
||||
"eso",
|
||||
"ante",
|
||||
"ellos",
|
||||
"e",
|
||||
"esto",
|
||||
"mí",
|
||||
"antes",
|
||||
"algunos",
|
||||
"qué",
|
||||
"unos",
|
||||
"yo",
|
||||
"otro",
|
||||
"otras",
|
||||
"anunció",
|
||||
"producción",
|
||||
"empresa",
|
||||
"mercado",
|
||||
"sector",
|
||||
"año",
|
||||
"gobierno",
|
||||
},
|
||||
"de": {
|
||||
"der", "die", "und", "in", "den", "von", "zu", "das", "mit", "sich", "des", "auf",
|
||||
"für", "ist", "im", "dem", "nicht", "ein", "eine", "als", "auch", "es", "an",
|
||||
"werden", "aus", "er", "hat", "dass", "sie", "nach", "wird", "bei", "einer", "um",
|
||||
"am", "sind", "noch", "wie", "einem", "über", "einen", "so", "zum", "war", "haben",
|
||||
"nur", "oder", "aber", "vor", "zur", "bis", "mehr", "durch", "man", "sein", "wurde",
|
||||
"sei", "prozent", "hatte", "kann", "gegen", "vom", "können", "schon", "wenn", "habe",
|
||||
"seine", "ihre", "unter", "wir", "sollen", "neue", "neuen", "batteriezellen", "unternehmen"
|
||||
"der",
|
||||
"die",
|
||||
"und",
|
||||
"in",
|
||||
"den",
|
||||
"von",
|
||||
"zu",
|
||||
"das",
|
||||
"mit",
|
||||
"sich",
|
||||
"des",
|
||||
"auf",
|
||||
"für",
|
||||
"ist",
|
||||
"im",
|
||||
"dem",
|
||||
"nicht",
|
||||
"ein",
|
||||
"eine",
|
||||
"als",
|
||||
"auch",
|
||||
"es",
|
||||
"an",
|
||||
"werden",
|
||||
"aus",
|
||||
"er",
|
||||
"hat",
|
||||
"dass",
|
||||
"sie",
|
||||
"nach",
|
||||
"wird",
|
||||
"bei",
|
||||
"einer",
|
||||
"um",
|
||||
"am",
|
||||
"sind",
|
||||
"noch",
|
||||
"wie",
|
||||
"einem",
|
||||
"über",
|
||||
"einen",
|
||||
"so",
|
||||
"zum",
|
||||
"war",
|
||||
"haben",
|
||||
"nur",
|
||||
"oder",
|
||||
"aber",
|
||||
"vor",
|
||||
"zur",
|
||||
"bis",
|
||||
"mehr",
|
||||
"durch",
|
||||
"man",
|
||||
"sein",
|
||||
"wurde",
|
||||
"sei",
|
||||
"prozent",
|
||||
"hatte",
|
||||
"kann",
|
||||
"gegen",
|
||||
"vom",
|
||||
"können",
|
||||
"schon",
|
||||
"wenn",
|
||||
"habe",
|
||||
"seine",
|
||||
"ihre",
|
||||
"unter",
|
||||
"wir",
|
||||
"sollen",
|
||||
"neue",
|
||||
"neuen",
|
||||
"batteriezellen",
|
||||
"unternehmen",
|
||||
},
|
||||
"it": {
|
||||
"di", "e", "il", "che", "la", "a", "in", "per", "un", "del", "non", "i", "si", "da",
|
||||
"le", "con", "sono", "della", "dei", "degli", "una", "al", "ma", "più", "delle",
|
||||
"questo", "nel", "alla", "anche", "ha", "gli", "come", "dall", "dalla", "ed",
|
||||
"se", "ci", "lo", "su", "loro", "dopo", "qualche", "nella", "uno", "mio", "tuo",
|
||||
"suo", "nostro", "vostro", "loro", "stato", "stata", "tra", "fra", "mentre", "prima",
|
||||
"quando", "molto", "tutto", "tutti", "tutte", "tutta", "senza", "ancora", "solo",
|
||||
"azienda", "mercato", "settore", "anno", "governo", "produzione", "motori"
|
||||
"di",
|
||||
"e",
|
||||
"il",
|
||||
"che",
|
||||
"la",
|
||||
"a",
|
||||
"in",
|
||||
"per",
|
||||
"un",
|
||||
"del",
|
||||
"non",
|
||||
"i",
|
||||
"si",
|
||||
"da",
|
||||
"le",
|
||||
"con",
|
||||
"sono",
|
||||
"della",
|
||||
"dei",
|
||||
"degli",
|
||||
"una",
|
||||
"al",
|
||||
"ma",
|
||||
"più",
|
||||
"delle",
|
||||
"questo",
|
||||
"nel",
|
||||
"alla",
|
||||
"anche",
|
||||
"ha",
|
||||
"gli",
|
||||
"come",
|
||||
"dall",
|
||||
"dalla",
|
||||
"ed",
|
||||
"se",
|
||||
"ci",
|
||||
"lo",
|
||||
"su",
|
||||
"loro",
|
||||
"dopo",
|
||||
"qualche",
|
||||
"nella",
|
||||
"uno",
|
||||
"mio",
|
||||
"tuo",
|
||||
"suo",
|
||||
"nostro",
|
||||
"vostro",
|
||||
"stato",
|
||||
"stata",
|
||||
"tra",
|
||||
"fra",
|
||||
"mentre",
|
||||
"prima",
|
||||
"quando",
|
||||
"molto",
|
||||
"tutto",
|
||||
"tutti",
|
||||
"tutte",
|
||||
"tutta",
|
||||
"senza",
|
||||
"ancora",
|
||||
"solo",
|
||||
"azienda",
|
||||
"mercato",
|
||||
"settore",
|
||||
"anno",
|
||||
"governo",
|
||||
"produzione",
|
||||
"motori",
|
||||
},
|
||||
"fr": {
|
||||
"de", "la", "le", "et", "les", "des", "en", "un", "du", "une", "que", "est", "pour",
|
||||
"qui", "dans", "a", "par", "sur", "pas", "plus", "au", "avec", "ce", "il", "sont",
|
||||
"se", "ne", "son", "sa", "ses", "aux", "ou", "comme", "mais", "nous", "vous", "ils",
|
||||
"leur", "y", "tout", "faire", "été", "aussi", "ces", "ont", "si", "fait", "même",
|
||||
"très", "après", "sans", "sous", "entre", "deux", "bien", "chez", "autre", "autres",
|
||||
"entreprise", "marché", "secteur", "année", "gouvernement", "production", "véhicules"
|
||||
}
|
||||
"de",
|
||||
"la",
|
||||
"le",
|
||||
"et",
|
||||
"les",
|
||||
"des",
|
||||
"en",
|
||||
"un",
|
||||
"du",
|
||||
"une",
|
||||
"que",
|
||||
"est",
|
||||
"pour",
|
||||
"qui",
|
||||
"dans",
|
||||
"a",
|
||||
"par",
|
||||
"sur",
|
||||
"pas",
|
||||
"plus",
|
||||
"au",
|
||||
"avec",
|
||||
"ce",
|
||||
"il",
|
||||
"sont",
|
||||
"se",
|
||||
"ne",
|
||||
"son",
|
||||
"sa",
|
||||
"ses",
|
||||
"aux",
|
||||
"ou",
|
||||
"comme",
|
||||
"mais",
|
||||
"nous",
|
||||
"vous",
|
||||
"ils",
|
||||
"leur",
|
||||
"y",
|
||||
"tout",
|
||||
"faire",
|
||||
"été",
|
||||
"aussi",
|
||||
"ces",
|
||||
"ont",
|
||||
"si",
|
||||
"fait",
|
||||
"même",
|
||||
"très",
|
||||
"après",
|
||||
"sans",
|
||||
"sous",
|
||||
"entre",
|
||||
"deux",
|
||||
"bien",
|
||||
"chez",
|
||||
"autre",
|
||||
"autres",
|
||||
"entreprise",
|
||||
"marché",
|
||||
"secteur",
|
||||
"année",
|
||||
"gouvernement",
|
||||
"production",
|
||||
"véhicules",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
@@ -77,12 +469,12 @@ def normalize_text(text: str) -> str:
|
||||
return "".join(c for c in nfd if unicodedata.category(c) != "Mn")
|
||||
|
||||
|
||||
def extract_words(text: str) -> List[str]:
|
||||
def extract_words(text: str) -> list[str]:
|
||||
"""Tokenize text into lowercase alphanumeric words."""
|
||||
return re.findall(r"\b\w+\b", text.lower())
|
||||
|
||||
|
||||
def detect_language(text: str) -> Tuple[str, float]:
|
||||
def detect_language(text: str) -> tuple[str, float]:
|
||||
"""
|
||||
Detect the ISO-639-1 language code of text among supported languages (pt, en, es, de, it, fr).
|
||||
Returns (detected_language, confidence_score).
|
||||
@@ -94,11 +486,10 @@ def detect_language(text: str) -> Tuple[str, float]:
|
||||
if not words:
|
||||
return "unknown", 0.0
|
||||
|
||||
total_words = len(words)
|
||||
word_set = set(words)
|
||||
|
||||
# Score languages based on matched stopword counts
|
||||
scores: Dict[str, int] = {}
|
||||
scores: dict[str, int] = {}
|
||||
for lang, stopwords in LANGUAGE_STOPWORDS.items():
|
||||
matched = word_set.intersection(stopwords)
|
||||
scores[lang] = len(matched)
|
||||
@@ -110,13 +501,55 @@ def detect_language(text: str) -> Tuple[str, float]:
|
||||
|
||||
# Specific disambiguation rules for closely related languages (PT vs ES)
|
||||
pt_exclusive = {
|
||||
"não", "do", "da", "no", "na", "nos", "nas", "em", "um", "uma", "você", "são", "é", "dos",
|
||||
"das", "foi", "está", "estão", "com", "pelo", "pela", "pelos", "pelas", "notícia",
|
||||
"extração", "mês", "ano", "produção", "bateu"
|
||||
"não",
|
||||
"do",
|
||||
"da",
|
||||
"no",
|
||||
"na",
|
||||
"nos",
|
||||
"nas",
|
||||
"em",
|
||||
"um",
|
||||
"uma",
|
||||
"você",
|
||||
"são",
|
||||
"é",
|
||||
"dos",
|
||||
"das",
|
||||
"foi",
|
||||
"está",
|
||||
"estão",
|
||||
"com",
|
||||
"pelo",
|
||||
"pela",
|
||||
"pelos",
|
||||
"pelas",
|
||||
"notícia",
|
||||
"extração",
|
||||
"mês",
|
||||
"ano",
|
||||
"produção",
|
||||
"bateu",
|
||||
}
|
||||
es_exclusive = {
|
||||
"el", "la", "y", "del", "al", "los", "las", "su", "sus", "con", "más", "pero",
|
||||
"durante", "noticia", "extracción", "mes", "año", "producción"
|
||||
"el",
|
||||
"la",
|
||||
"y",
|
||||
"del",
|
||||
"al",
|
||||
"los",
|
||||
"las",
|
||||
"su",
|
||||
"sus",
|
||||
"con",
|
||||
"más",
|
||||
"pero",
|
||||
"durante",
|
||||
"noticia",
|
||||
"extracción",
|
||||
"mes",
|
||||
"año",
|
||||
"producción",
|
||||
}
|
||||
|
||||
# Normalize words to match accents cleanly
|
||||
@@ -133,7 +566,11 @@ def detect_language(text: str) -> Tuple[str, float]:
|
||||
elif "e" in word_set and "y" not in word_set:
|
||||
pt_score += 1
|
||||
|
||||
if top_lang in ("pt", "es") or (top_lang == "pt" and es_score > pt_score) or (top_lang == "es" and pt_score > es_score):
|
||||
if (
|
||||
top_lang in ("pt", "es")
|
||||
or (top_lang == "pt" and es_score > pt_score)
|
||||
or (top_lang == "es" and pt_score > es_score)
|
||||
):
|
||||
if es_score > pt_score:
|
||||
top_lang = "es"
|
||||
top_matches = max(top_matches, es_score)
|
||||
|
||||
+30
-26
@@ -3,9 +3,9 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass, field, asdict
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import Any
|
||||
|
||||
|
||||
class DecisionCategory(str, Enum):
|
||||
@@ -29,20 +29,20 @@ class RelatedEntity:
|
||||
name: str
|
||||
relation_type: str
|
||||
weight: float
|
||||
aliases: List[str] = field(default_factory=list)
|
||||
aliases: list[str] = field(default_factory=list)
|
||||
scope: str = "general"
|
||||
confidence: float = 1.0
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, data: Dict[str, Any]) -> RelatedEntity:
|
||||
def from_dict(cls, data: dict[str, Any]) -> RelatedEntity:
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError("Related entity must be a JSON object")
|
||||
|
||||
|
||||
required = ["entity_id", "name", "relation_type", "weight"]
|
||||
for req in required:
|
||||
if req not in data or data[req] is None:
|
||||
raise ValueError(f"Missing required field in related entity: '{req}'")
|
||||
|
||||
|
||||
return cls(
|
||||
entity_id=str(data["entity_id"]),
|
||||
name=str(data["name"]),
|
||||
@@ -58,36 +58,36 @@ class RelatedEntity:
|
||||
class ECPSnapshot:
|
||||
target_entity_id: str
|
||||
target_name: str
|
||||
aliases: List[str]
|
||||
aliases: list[str]
|
||||
domain: str
|
||||
anchors: List[str]
|
||||
negative_anchors: List[str] = field(default_factory=list)
|
||||
anchors: list[str]
|
||||
negative_anchors: list[str] = field(default_factory=list)
|
||||
graph_version: str = "1.0.0"
|
||||
related_entities: List[RelatedEntity] = field(default_factory=list)
|
||||
related_entities: list[RelatedEntity] = field(default_factory=list)
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, data: Dict[str, Any]) -> ECPSnapshot:
|
||||
def from_dict(cls, data: dict[str, Any]) -> ECPSnapshot:
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError("ECP Snapshot payload must be a JSON object")
|
||||
|
||||
|
||||
required_fields = ["target_entity_id", "target_name", "aliases", "domain", "anchors"]
|
||||
for field_name in required_fields:
|
||||
if field_name not in data or data[field_name] is None:
|
||||
raise ValueError(f"Missing required field in ECP Snapshot: '{field_name}'")
|
||||
|
||||
|
||||
if not isinstance(data["aliases"], list):
|
||||
raise ValueError("Field 'aliases' must be a list of strings")
|
||||
if not isinstance(data["anchors"], list):
|
||||
raise ValueError("Field 'anchors' must be a list of strings")
|
||||
|
||||
|
||||
neg_anchors = data.get("negative_anchors", [])
|
||||
if neg_anchors is not None and not isinstance(neg_anchors, list):
|
||||
raise ValueError("Field 'negative_anchors' must be a list of strings if provided")
|
||||
|
||||
|
||||
related_data = data.get("related_entities", [])
|
||||
if related_data is not None and not isinstance(related_data, list):
|
||||
raise ValueError("Field 'related_entities' must be a list if provided")
|
||||
|
||||
|
||||
related_objs = [RelatedEntity.from_dict(item) for item in (related_data or [])]
|
||||
|
||||
return cls(
|
||||
@@ -124,16 +124,18 @@ class ClassificationResult:
|
||||
is_inherent: bool
|
||||
confidence: float
|
||||
detected_language: str
|
||||
matched_anchors: List[str] = field(default_factory=list)
|
||||
negative_matches: List[str] = field(default_factory=list)
|
||||
graph_matches: List[Dict[str, Any]] = field(default_factory=list)
|
||||
evidence: List[str] = field(default_factory=list)
|
||||
matched_anchors: list[str] = field(default_factory=list)
|
||||
negative_matches: list[str] = field(default_factory=list)
|
||||
graph_matches: list[dict[str, Any]] = field(default_factory=list)
|
||||
evidence: list[str] = field(default_factory=list)
|
||||
rationale: str = ""
|
||||
warnings: List[str] = field(default_factory=list)
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"decision": self.decision.value if isinstance(self.decision, DecisionCategory) else str(self.decision),
|
||||
"decision": self.decision.value
|
||||
if isinstance(self.decision, DecisionCategory)
|
||||
else str(self.decision),
|
||||
"is_inherent": bool(self.is_inherent),
|
||||
"confidence": round(float(self.confidence), 4),
|
||||
"detected_language": str(self.detected_language),
|
||||
@@ -153,11 +155,13 @@ class ClassificationResult:
|
||||
class ClassificationError:
|
||||
error_code: ErrorCode
|
||||
message: str
|
||||
details: Dict[str, Any] = field(default_factory=dict)
|
||||
details: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"error_code": self.error_code.value if isinstance(self.error_code, ErrorCode) else str(self.error_code),
|
||||
"error_code": self.error_code.value
|
||||
if isinstance(self.error_code, ErrorCode)
|
||||
else str(self.error_code),
|
||||
"message": str(self.message),
|
||||
"details": dict(self.details),
|
||||
}
|
||||
|
||||
+5
-4
@@ -3,7 +3,6 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import List, Tuple
|
||||
|
||||
|
||||
def strip_markdown(markdown_text: str) -> str:
|
||||
@@ -40,7 +39,7 @@ def strip_markdown(markdown_text: str) -> str:
|
||||
return text
|
||||
|
||||
|
||||
def extract_sentences(text: str) -> List[str]:
|
||||
def extract_sentences(text: str) -> list[str]:
|
||||
"""Split text into individual sentences."""
|
||||
# Split by period, exclamation, question mark followed by space or newline
|
||||
raw_sentences = re.split(r"(?<=[.!?])\s+", text.strip())
|
||||
@@ -48,7 +47,9 @@ def extract_sentences(text: str) -> List[str]:
|
||||
return sentences
|
||||
|
||||
|
||||
def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_snippets: int = 3) -> List[str]:
|
||||
def extract_evidence_snippets(
|
||||
markdown_text: str, match_terms: list[str], max_snippets: int = 3
|
||||
) -> list[str]:
|
||||
"""
|
||||
Extract relevant sentence excerpts from Markdown text that contain any of the given match terms.
|
||||
Preserves original phrasing and formats as clean evidence.
|
||||
@@ -62,7 +63,7 @@ def extract_evidence_snippets(markdown_text: str, match_terms: List[str], max_sn
|
||||
sentences = [plain_text]
|
||||
|
||||
lower_terms = [t.lower() for t in match_terms if t]
|
||||
evidence: List[str] = []
|
||||
evidence: list[str] = []
|
||||
|
||||
for sentence in sentences:
|
||||
lower_sent = sentence.lower()
|
||||
|
||||
Reference in New Issue
Block a user