feat: add deterministic content extractor selector engine with F1 consensus
This commit is contained in:
@@ -0,0 +1,603 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Seletor Determinístico de Extrator de Conteúdo de Artigos.
|
||||
|
||||
Compara deterministicamente as saídas de Trafilatura, Newspaper4k e Readability
|
||||
a partir de um arquivo JSON consolidado de extrações, calculando métricas de consenso
|
||||
(shingles de 5-tokens, cobertura, suporte, F1-score) e aplicando regras rígidas de
|
||||
desempate técnico e hierárquico, enriquecendo o JSON exclusivamente com a chave
|
||||
`selected_extractor` de forma não-destrutiva e atômica.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import html
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import tempfile
|
||||
import unicodedata
|
||||
from collections import Counter
|
||||
from dataclasses import dataclass, field
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
# ==============================================================================
|
||||
# Modelos de Dados e Enumerações
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
class ExtractorName(str, Enum):
|
||||
"""Nomes canônicos e catálogo fechado dos motores de extração."""
|
||||
|
||||
TRAFILATURA = "trafilatura"
|
||||
NEWSPAPER4K = "newspaper4k"
|
||||
READABILITY = "readability"
|
||||
|
||||
|
||||
class CandidateStatus(str, Enum):
|
||||
"""Estado de viabilidade do candidato para formação do conjunto ativo."""
|
||||
|
||||
USABLE = "usable"
|
||||
DEGRADED = "degraded"
|
||||
UNAVAILABLE = "unavailable"
|
||||
|
||||
|
||||
# Ordem estrita de prioridade de desempate final (PRD §7.1 item 4, §7.5 item 6, §7.6 item 4)
|
||||
FALLBACK_PRIORITY: list[ExtractorName] = [
|
||||
ExtractorName.NEWSPAPER4K,
|
||||
ExtractorName.READABILITY,
|
||||
ExtractorName.TRAFILATURA,
|
||||
]
|
||||
|
||||
# Margem de empate técnico (PRD §7.5 item 3)
|
||||
TECHNICAL_TIE_THRESHOLD: float = 0.03
|
||||
FLOAT_EPSILON: float = 1e-9
|
||||
|
||||
|
||||
@dataclass
|
||||
class ExtractorCandidate:
|
||||
"""Representação e métricas de um candidato a extrator em um artigo."""
|
||||
|
||||
name: ExtractorName
|
||||
raw_text: str | None = None
|
||||
error: str | None = None
|
||||
status: CandidateStatus = CandidateStatus.UNAVAILABLE
|
||||
tokens: list[str] = field(default_factory=list)
|
||||
shingles: set[tuple[str, ...]] = field(default_factory=set)
|
||||
shingle_count: int = 0
|
||||
coverage: float = 0.0
|
||||
support: float = 0.0
|
||||
score: float = 0.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class ArticleSelectionResult:
|
||||
"""Resultado detalhado da seleção para um artigo individual."""
|
||||
|
||||
article_index: int
|
||||
selected_extractor: ExtractorName
|
||||
selection_reason: str
|
||||
active_candidates_count: int
|
||||
consensus_shingles_count: int
|
||||
candidates: dict[ExtractorName, ExtractorCandidate] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass
|
||||
class BatchProcessingResult:
|
||||
"""Resultado consolidado do processamento em lote."""
|
||||
|
||||
total_articles: int
|
||||
processed_count: int
|
||||
selection_distribution: dict[str, int]
|
||||
input_file: str
|
||||
output_file: str
|
||||
selections: list[ArticleSelectionResult] = field(default_factory=list)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Pipeline de Normalização Textual e Shingles
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def normalize_text(text: Any) -> list[str]:
|
||||
"""
|
||||
Executa a normalização determinística para comparação:
|
||||
1. Decodificar entidades HTML.
|
||||
2. Remover marcação HTML e Markdown, preservando o texto visível.
|
||||
3. Em links, preservar o texto e remover o endereço.
|
||||
4. Aplicar normalização Unicode NFKC.
|
||||
5. Converter o texto para minúsculas.
|
||||
6. Substituir toda sequência de espaços, tabulações ou quebras de linha por um único espaço.
|
||||
7. Tokenizar mantendo letras e números Unicode.
|
||||
8. Desconsiderar pontuação.
|
||||
"""
|
||||
if not text or not isinstance(text, str):
|
||||
return []
|
||||
|
||||
# 1. Decodificar entidades HTML
|
||||
s = html.unescape(text)
|
||||
|
||||
# 2. Remover imagens Markdown  -> '' (marcação de mídia não é texto visível)
|
||||
s = re.sub(r"!\s*\[[^\]]*\]\([^)]*\)", " ", s)
|
||||
|
||||
# 3. Em links Markdown [texto](url), preservar texto âncora e remover endereço
|
||||
s = re.sub(r"\[([^\]]+)\]\([^)]+\)", r" \1 ", s)
|
||||
|
||||
# 2. Remover tags HTML mantendo espaço entre palavras
|
||||
s = re.sub(r"<[^>]+>", " ", s)
|
||||
|
||||
# 4. Normalização Unicode NFKC
|
||||
s = unicodedata.normalize("NFKC", s)
|
||||
|
||||
# 5. Minúsculas
|
||||
s = s.lower()
|
||||
|
||||
# 6. Colapso de múltiplos espaços em branco
|
||||
s = re.sub(r"\s+", " ", s).strip()
|
||||
|
||||
# 7 & 8. Tokenizar mantendo letras e números Unicode, ignorando pontuações
|
||||
tokens = re.findall(r"[\w]+", s, flags=re.UNICODE)
|
||||
return tokens
|
||||
|
||||
|
||||
def generate_shingles(tokens: list[str], window_size: int = 5) -> set[tuple[str, ...]]:
|
||||
"""
|
||||
Gera conjunto de shingles ordenados de tamanho window_size (padrão 5).
|
||||
- Se len(tokens) >= window_size: todas as janelas consecutivas de 5 tokens.
|
||||
- Se 1 <= len(tokens) < window_size: sequência completa como um único shingle.
|
||||
- Se tokens vazio: conjunto vazio.
|
||||
"""
|
||||
n = len(tokens)
|
||||
if n == 0:
|
||||
return set()
|
||||
if n < window_size:
|
||||
return {tuple(tokens)}
|
||||
return {tuple(tokens[i : i + window_size]) for i in range(n - window_size + 1)}
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Classificação de Candidatos e Formação do Conjunto Ativo
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def extract_candidate_data(article_dict: dict[str, Any], name: ExtractorName) -> ExtractorCandidate:
|
||||
"""Extrai campo de texto, erro e calcula tokens/shingles para um motor."""
|
||||
lib_data = article_dict.get(name.value)
|
||||
if not isinstance(lib_data, dict):
|
||||
return ExtractorCandidate(name=name, status=CandidateStatus.UNAVAILABLE)
|
||||
|
||||
# Campo usado na comparação (PRD §5.2)
|
||||
if name == ExtractorName.TRAFILATURA:
|
||||
raw_text = lib_data.get("text")
|
||||
elif name == ExtractorName.NEWSPAPER4K:
|
||||
raw_text = lib_data.get("text")
|
||||
elif name == ExtractorName.READABILITY:
|
||||
raw_text = lib_data.get("cleaned_text")
|
||||
else:
|
||||
raw_text = None
|
||||
|
||||
raw_error = lib_data.get("error")
|
||||
# Tratar erro vazio/nulo
|
||||
error_val = str(raw_error) if raw_error is not None and str(raw_error).strip() else None
|
||||
|
||||
# Normalizar texto para obter tokens
|
||||
tokens = normalize_text(raw_text)
|
||||
shingles = generate_shingles(tokens)
|
||||
shingle_count = len(shingles)
|
||||
|
||||
# Determinar status do candidato (PRD §5.3)
|
||||
if not tokens or shingle_count == 0:
|
||||
status = CandidateStatus.UNAVAILABLE
|
||||
elif error_val is None:
|
||||
status = CandidateStatus.USABLE
|
||||
else:
|
||||
status = CandidateStatus.DEGRADED
|
||||
|
||||
return ExtractorCandidate(
|
||||
name=name,
|
||||
raw_text=raw_text if isinstance(raw_text, str) else None,
|
||||
error=error_val,
|
||||
status=status,
|
||||
tokens=tokens,
|
||||
shingles=shingles,
|
||||
shingle_count=shingle_count,
|
||||
)
|
||||
|
||||
|
||||
def form_active_set(
|
||||
candidates: dict[ExtractorName, ExtractorCandidate],
|
||||
) -> list[ExtractorCandidate]:
|
||||
"""
|
||||
Forma o conjunto ativo de candidatos conforme PRD §7.1:
|
||||
1. Se existir pelo menos um utilizável, considerar somente os utilizáveis.
|
||||
2. Se não existir utilizável, considerar os candidatos degradados.
|
||||
3. Se não existir utilizável nem degradado, retorna lista vazia.
|
||||
"""
|
||||
usables = [c for c in candidates.values() if c.status == CandidateStatus.USABLE]
|
||||
if usables:
|
||||
return usables
|
||||
|
||||
degradeds = [c for c in candidates.values() if c.status == CandidateStatus.DEGRADED]
|
||||
if degradeds:
|
||||
return degradeds
|
||||
|
||||
return []
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Cálculo de Consenso e Métricas F1
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def calculate_consensus_metrics(
|
||||
active_candidates: list[ExtractorCandidate],
|
||||
) -> set[tuple[str, ...]]:
|
||||
"""
|
||||
Constrói o conjunto de consenso (shingles presentes em >= 2 candidatos ativos)
|
||||
e calcula Cobertura, Suporte e F1-score para cada candidato ativo (PRD §7.4).
|
||||
"""
|
||||
if len(active_candidates) < 2:
|
||||
for c in active_candidates:
|
||||
c.coverage = 1.0 if c.shingle_count > 0 else 0.0
|
||||
c.support = 1.0 if c.shingle_count > 0 else 0.0
|
||||
c.score = 1.0 if c.shingle_count > 0 else 0.0
|
||||
return set()
|
||||
|
||||
# Contar frequência de cada shingle entre os candidatos ativos
|
||||
shingle_counts: Counter[tuple[str, ...]] = Counter()
|
||||
for c in active_candidates:
|
||||
for s in c.shingles:
|
||||
shingle_counts[s] += 1
|
||||
|
||||
consensus_shingles = {s for s, cnt in shingle_counts.items() if cnt >= 2}
|
||||
total_consensus = len(consensus_shingles)
|
||||
|
||||
for c in active_candidates:
|
||||
if total_consensus == 0 or c.shingle_count == 0:
|
||||
c.coverage = 0.0
|
||||
c.support = 0.0
|
||||
c.score = 0.0
|
||||
continue
|
||||
|
||||
common_shingles = len(c.shingles & consensus_shingles)
|
||||
c.coverage = common_shingles / total_consensus
|
||||
c.support = common_shingles / c.shingle_count
|
||||
|
||||
denominator = c.coverage + c.support
|
||||
if denominator > 0:
|
||||
c.score = (2.0 * c.coverage * c.support) / denominator
|
||||
else:
|
||||
c.score = 0.0
|
||||
|
||||
return consensus_shingles
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Algoritmos de Seleção e Desempate
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def break_priority_tie(candidates: list[ExtractorCandidate]) -> ExtractorCandidate:
|
||||
"""Aplica a prioridade final estrita: newspaper4k > readability > trafilatura."""
|
||||
for priority_name in FALLBACK_PRIORITY:
|
||||
for c in candidates:
|
||||
if c.name == priority_name:
|
||||
return c
|
||||
return candidates[0]
|
||||
|
||||
|
||||
def select_with_consensus(
|
||||
active_candidates: list[ExtractorCandidate], consensus_shingles: set[tuple[str, ...]]
|
||||
) -> tuple[ExtractorName, str]:
|
||||
"""
|
||||
Seleciona o melhor candidato quando existe consenso (PRD §7.5):
|
||||
1. Ordenar por score decrescente.
|
||||
2. Identificar candidatos no empate técnico (diferença para maior score <= 0.03).
|
||||
3. Se houver 1 candidato no empate técnico, selecioná-lo.
|
||||
4. Se houver empate técnico, selecionar o candidato com menor quantidade de shingles.
|
||||
5. Se empatar na quantidade de shingles, aplicar prioridade final newspaper4k > readability > trafilatura.
|
||||
"""
|
||||
# Ordenar por score decrescente
|
||||
sorted_by_score = sorted(active_candidates, key=lambda c: c.score, reverse=True)
|
||||
max_score = sorted_by_score[0].score
|
||||
|
||||
# Grupo de empate técnico (score >= max_score - 0.03 com tolerância de precisão float)
|
||||
technical_tie_pool = [
|
||||
c
|
||||
for c in sorted_by_score
|
||||
if (max_score - c.score) <= (TECHNICAL_TIE_THRESHOLD + FLOAT_EPSILON)
|
||||
]
|
||||
|
||||
if len(technical_tie_pool) == 1:
|
||||
return technical_tie_pool[0].name, "highest_score"
|
||||
|
||||
# Selecionar o candidato com menor quantidade de shingles no grupo de empate
|
||||
min_shingles = min(c.shingle_count for c in technical_tie_pool)
|
||||
shingle_tie_pool = [c for c in technical_tie_pool if c.shingle_count == min_shingles]
|
||||
|
||||
if len(shingle_tie_pool) == 1:
|
||||
return shingle_tie_pool[0].name, "technical_tie_smallest_shingles"
|
||||
|
||||
# Desempate final de prioridade
|
||||
winner = break_priority_tie(shingle_tie_pool)
|
||||
return winner.name, "technical_tie_priority_fallback"
|
||||
|
||||
|
||||
def select_without_consensus(
|
||||
active_candidates: list[ExtractorCandidate],
|
||||
) -> tuple[ExtractorName, str]:
|
||||
"""
|
||||
Seleciona o candidato quando NÃO existe consenso (PRD §7.6):
|
||||
- Com 3 candidatos ativos: selecionar o candidato com a quantidade mediana de shingles.
|
||||
- Com 2 candidatos ativos: selecionar o candidato com a maior quantidade de shingles.
|
||||
- Com 1 candidato ativo: selecionar o único candidato.
|
||||
- Em empate de quantidade: aplicar prioridade newspaper4k > readability > trafilatura.
|
||||
- Sem candidato ativo: selecionar newspaper4k.
|
||||
"""
|
||||
k = len(active_candidates)
|
||||
if k == 0:
|
||||
return ExtractorName.NEWSPAPER4K, "fallback_all_unavailable"
|
||||
if k == 1:
|
||||
return active_candidates[0].name, "single_active_candidate"
|
||||
|
||||
if k == 2:
|
||||
c1, c2 = active_candidates[0], active_candidates[1]
|
||||
if c1.shingle_count > c2.shingle_count:
|
||||
return c1.name, "no_consensus_max_shingles"
|
||||
elif c2.shingle_count > c1.shingle_count:
|
||||
return c2.name, "no_consensus_max_shingles"
|
||||
else:
|
||||
winner = break_priority_tie([c1, c2])
|
||||
return winner.name, "no_consensus_tie_priority_fallback"
|
||||
|
||||
if k == 3:
|
||||
# 3 candidatos ativos -> quantidade mediana de shingles
|
||||
# Ordenar por shingle_count crescente
|
||||
sorted_by_shingles = sorted(active_candidates, key=lambda c: c.shingle_count)
|
||||
s0, s1, s2 = (
|
||||
sorted_by_shingles[0].shingle_count,
|
||||
sorted_by_shingles[1].shingle_count,
|
||||
sorted_by_shingles[2].shingle_count,
|
||||
)
|
||||
|
||||
# Se todos tiverem contagens distintas (ex: 10, 20, 30), a mediana é o elemento do meio (20)
|
||||
if s0 < s1 < s2:
|
||||
return sorted_by_shingles[1].name, "no_consensus_median_shingles"
|
||||
|
||||
# Se houver empate na mediana (ex: [10, 20, 20] ou [20, 20, 30] ou [20, 20, 20])
|
||||
# Os candidatos cujo shingle_count é igual ao valor mediano (s1) entram no pool de desempate
|
||||
median_value = s1
|
||||
median_candidates = [c for c in active_candidates if c.shingle_count == median_value]
|
||||
|
||||
if len(median_candidates) == 1:
|
||||
return median_candidates[0].name, "no_consensus_median_shingles"
|
||||
else:
|
||||
winner = break_priority_tie(median_candidates)
|
||||
return winner.name, "no_consensus_median_priority_fallback"
|
||||
|
||||
# Fallback genérico para listas maiores (se houver)
|
||||
winner = break_priority_tie(active_candidates)
|
||||
return winner.name, "priority_fallback"
|
||||
|
||||
|
||||
def select_article_extractor(
|
||||
article_dict: dict[str, Any], article_index: int = 0
|
||||
) -> ArticleSelectionResult:
|
||||
"""
|
||||
Orquestrador determinístico completo para um único artigo.
|
||||
Classifica candidatos, forma conjunto ativo, calcula métricas e aplica regras de decisão.
|
||||
"""
|
||||
candidates: dict[ExtractorName, ExtractorCandidate] = {
|
||||
name: extract_candidate_data(article_dict, name) for name in ExtractorName
|
||||
}
|
||||
|
||||
active_set = form_active_set(candidates)
|
||||
|
||||
if not active_set:
|
||||
# Nenhum candidato utilizável nem degradado -> Fallback compulsório para newspaper4k (PRD §7.1 item 4)
|
||||
return ArticleSelectionResult(
|
||||
article_index=article_index,
|
||||
selected_extractor=ExtractorName.NEWSPAPER4K,
|
||||
selection_reason="fallback_all_unavailable",
|
||||
active_candidates_count=0,
|
||||
consensus_shingles_count=0,
|
||||
candidates=candidates,
|
||||
)
|
||||
|
||||
if len(active_set) == 1:
|
||||
# Apenas 1 candidato ativo -> selecioná-lo imediatamente (PRD §7.1 item 5)
|
||||
winner = active_set[0]
|
||||
return ArticleSelectionResult(
|
||||
article_index=article_index,
|
||||
selected_extractor=winner.name,
|
||||
selection_reason="single_usable_candidate"
|
||||
if winner.status == CandidateStatus.USABLE
|
||||
else "single_degraded_candidate",
|
||||
active_candidates_count=1,
|
||||
consensus_shingles_count=0,
|
||||
candidates=candidates,
|
||||
)
|
||||
|
||||
# Calcular consenso e métricas F1
|
||||
consensus_shingles = calculate_consensus_metrics(active_set)
|
||||
|
||||
if consensus_shingles:
|
||||
winner_name, reason = select_with_consensus(active_set, consensus_shingles)
|
||||
else:
|
||||
winner_name, reason = select_without_consensus(active_set)
|
||||
|
||||
return ArticleSelectionResult(
|
||||
article_index=article_index,
|
||||
selected_extractor=winner_name,
|
||||
selection_reason=reason,
|
||||
active_candidates_count=len(active_set),
|
||||
consensus_shingles_count=len(consensus_shingles),
|
||||
candidates=candidates,
|
||||
)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Processamento em Lote e I/O Atômico
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def atomic_save_json(data: Any, target_path: Path, indent: int = 2) -> None:
|
||||
"""Salva dados em JSON de forma atômica utilizando arquivo temporário e rename."""
|
||||
target_path = Path(target_path).resolve()
|
||||
target_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
temp_fd, temp_file_path = tempfile.mkstemp(
|
||||
dir=target_path.parent, prefix=f".{target_path.name}.tmp_", text=True
|
||||
)
|
||||
try:
|
||||
with open(temp_fd, "w", encoding="utf-8") as f:
|
||||
if indent > 0:
|
||||
json.dump(data, f, ensure_ascii=False, indent=indent)
|
||||
else:
|
||||
json.dump(data, f, ensure_ascii=False, separators=(",", ":"))
|
||||
f.write("\n")
|
||||
os.replace(temp_file_path, target_path)
|
||||
except Exception:
|
||||
if os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path)
|
||||
raise
|
||||
|
||||
|
||||
def process_batch(
|
||||
input_path: Path | str,
|
||||
output_path: Path | str | None = None,
|
||||
indent: int = 2,
|
||||
verbose: bool = False,
|
||||
) -> BatchProcessingResult:
|
||||
"""
|
||||
Lê o JSON de entrada, valida a estrutura, processa todos os artigos e grava o arquivo de saída.
|
||||
Preserva 100% dos dados originais e a ordem dos artigos.
|
||||
"""
|
||||
in_file = Path(input_path).resolve()
|
||||
if not in_file.is_file():
|
||||
raise FileNotFoundError(f"Arquivo de entrada não encontrado: {in_file}")
|
||||
|
||||
try:
|
||||
with open(in_file, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(f"JSON inválido em '{in_file}': {exc}") from exc
|
||||
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError("A raiz do JSON de entrada deve ser um objeto.")
|
||||
|
||||
articles = data.get("articles")
|
||||
if articles is None or not isinstance(articles, list):
|
||||
raise ValueError("A chave 'articles' é obrigatória e deve ser uma lista.")
|
||||
|
||||
if output_path is None:
|
||||
# Padrão: <nome_original_sem_extensao>_selected.json
|
||||
out_file = in_file.parent / f"{in_file.stem}_selected.json"
|
||||
else:
|
||||
out_file = Path(output_path).resolve()
|
||||
|
||||
selections: list[ArticleSelectionResult] = []
|
||||
distribution: Counter[str] = Counter()
|
||||
|
||||
for idx, article in enumerate(articles):
|
||||
if not isinstance(article, dict):
|
||||
# Tratar artigo malformado
|
||||
article = {}
|
||||
articles[idx] = article
|
||||
|
||||
res = select_article_extractor(article, article_index=idx)
|
||||
selections.append(res)
|
||||
distribution[res.selected_extractor.value] += 1
|
||||
|
||||
# Enriquecer ou recalcular chave no artigo (PRD §6.2 e §6.3)
|
||||
article["selected_extractor"] = res.selected_extractor.value
|
||||
|
||||
if verbose:
|
||||
sys.stderr.write(
|
||||
f"[Artigo #{idx + 1:03d}] Extrator: {res.selected_extractor.value:<12} | "
|
||||
f"Motivo: {res.selection_reason:<32} | Ativos: {res.active_candidates_count} | "
|
||||
f"Consenso: {res.consensus_shingles_count}\n"
|
||||
)
|
||||
|
||||
# Gravar arquivo de saída atomicamente
|
||||
atomic_save_json(data, out_file, indent=indent)
|
||||
|
||||
return BatchProcessingResult(
|
||||
total_articles=len(articles),
|
||||
processed_count=len(selections),
|
||||
selection_distribution=dict(distribution),
|
||||
input_file=str(in_file),
|
||||
output_file=str(out_file),
|
||||
selections=selections,
|
||||
)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Interface CLI
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Seletor determinístico da melhor extração de conteúdo (Trafilatura / Newspaper4k / Readability)."
|
||||
)
|
||||
parser.add_argument(
|
||||
"input_file", type=Path, help="Caminho do arquivo JSON consolidado de extrações."
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output",
|
||||
type=Path,
|
||||
default=None,
|
||||
help="Caminho do arquivo JSON de saída (padrão: <nome>_selected.json).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--indent", type=int, default=2, help="Indentação do arquivo JSON de saída (padrão: 2)."
|
||||
)
|
||||
parser.add_argument(
|
||||
"-v",
|
||||
"--verbose",
|
||||
action="store_true",
|
||||
help="Exibe detalhes da seleção e pontuações no stderr.",
|
||||
)
|
||||
return parser.parse_args(argv)
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
args = parse_args(argv)
|
||||
try:
|
||||
result = process_batch(
|
||||
input_path=args.input_file,
|
||||
output_path=args.output,
|
||||
indent=args.indent,
|
||||
verbose=args.verbose,
|
||||
)
|
||||
|
||||
output_summary = {
|
||||
"status": "success",
|
||||
"input_file": result.input_file,
|
||||
"output_file": result.output_file,
|
||||
"total_articles": result.total_articles,
|
||||
"processed_count": result.processed_count,
|
||||
"distribution": result.selection_distribution,
|
||||
}
|
||||
sys.stdout.write(json.dumps(output_summary, indent=2, ensure_ascii=False) + "\n")
|
||||
return 0
|
||||
|
||||
except FileNotFoundError as e:
|
||||
sys.stderr.write(f"ERRO DE ARQUIVO: {e}\n")
|
||||
return 1
|
||||
except ValueError as e:
|
||||
sys.stderr.write(f"ERRO DE VALIDAÇÃO: {e}\n")
|
||||
return 2
|
||||
except Exception as e:
|
||||
sys.stderr.write(f"ERRO INESPERADO: {e}\n")
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user