721 lines
21 KiB
Python
721 lines
21 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Convert Article JSON to Markdown CLI.
|
|
|
|
Converte o JSON de um único artigo extraído (com selected_extractor) para um documento
|
|
Markdown (.md) limpo, padronizado e com seleção determinística de metadados.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import datetime
|
|
import email.utils
|
|
import html
|
|
import json
|
|
import re
|
|
import sys
|
|
import urllib.parse
|
|
from pathlib import Path
|
|
from typing import Any, Dict, List, Optional, Set
|
|
|
|
import markdownify
|
|
|
|
KNOWN_PLACEHOLDERS: Set[str] = {
|
|
"null",
|
|
"none",
|
|
"n/a",
|
|
"unknown",
|
|
"[no-author]",
|
|
"no-author",
|
|
}
|
|
|
|
VALID_EXTRACTORS: Set[str] = {
|
|
"trafilatura",
|
|
"newspaper4k",
|
|
"readability",
|
|
}
|
|
|
|
|
|
def normalize_scalar(value: Any) -> Optional[str]:
|
|
"""
|
|
Decodifica entidades HTML, remove espaços no início/fim, colapsa espaços internos
|
|
e descarta placeholders conhecidos.
|
|
"""
|
|
if not isinstance(value, str):
|
|
return None
|
|
unescaped = html.unescape(value).strip()
|
|
if not unescaped:
|
|
return None
|
|
collapsed = re.sub(r"\s+", " ", unescaped)
|
|
if collapsed.lower() in KNOWN_PLACEHOLDERS:
|
|
return None
|
|
return collapsed
|
|
|
|
|
|
def normalize_list(value: Any, is_author: bool = False) -> List[str]:
|
|
"""
|
|
Normaliza listas ou strings separadas por ponto e vírgula, descartando placeholders,
|
|
URLs em autores e deduplicando sem diferenciar maiúsculas/minúsculas.
|
|
"""
|
|
if not value:
|
|
return []
|
|
|
|
raw_items: List[str] = []
|
|
if isinstance(value, list):
|
|
for item in value:
|
|
if isinstance(item, str):
|
|
# Se um elemento da lista contiver ponto e vírgula, divide
|
|
if ";" in item:
|
|
raw_items.extend(item.split(";"))
|
|
elif "," in item and not is_author:
|
|
# Trafilatura às vezes emite tags separadas por vírgula em string única
|
|
raw_items.extend(item.split(","))
|
|
else:
|
|
raw_items.append(item)
|
|
elif isinstance(value, str):
|
|
raw_items.extend(value.split(";"))
|
|
else:
|
|
return []
|
|
|
|
normalized_items: List[str] = []
|
|
seen_lower: Set[str] = set()
|
|
|
|
for item in raw_items:
|
|
norm = normalize_scalar(item)
|
|
if not norm:
|
|
continue
|
|
if is_author:
|
|
norm_lower = norm.lower()
|
|
if (
|
|
norm_lower.startswith("http://")
|
|
or norm_lower.startswith("https://")
|
|
or norm_lower.startswith("www.")
|
|
):
|
|
continue
|
|
lower_key = norm.lower()
|
|
if lower_key not in seen_lower:
|
|
seen_lower.add(lower_key)
|
|
normalized_items.append(norm)
|
|
|
|
return normalized_items
|
|
|
|
|
|
def normalize_date(value: Any) -> Optional[str]:
|
|
"""
|
|
Interpreta datas ISO 8601 e RFC 2822 preservando fuso horário ou YYYY-MM-DD para datas puras.
|
|
"""
|
|
if not isinstance(value, str):
|
|
return None
|
|
s = value.strip()
|
|
if not s or s.lower() in KNOWN_PLACEHOLDERS:
|
|
return None
|
|
|
|
# Se for apenas data YYYY-MM-DD
|
|
if re.match(r"^\d{4}-\d{2}-\d{2}$", s):
|
|
return s
|
|
|
|
# Tenta ISO 8601
|
|
try:
|
|
dt = datetime.datetime.fromisoformat(s)
|
|
return dt.isoformat()
|
|
except (ValueError, TypeError):
|
|
pass
|
|
|
|
# Tenta RFC 2822
|
|
try:
|
|
dt = email.utils.parsedate_to_datetime(s)
|
|
return dt.isoformat()
|
|
except (ValueError, TypeError):
|
|
pass
|
|
|
|
return None
|
|
|
|
|
|
def validate_url(value: Any) -> Optional[str]:
|
|
"""Valida se a URL é absoluta com protocolo http ou https e hostname não vazio."""
|
|
norm = normalize_scalar(value)
|
|
if not norm:
|
|
return None
|
|
try:
|
|
parsed = urllib.parse.urlparse(norm)
|
|
if parsed.scheme.lower() in ("http", "https") and parsed.netloc:
|
|
return norm
|
|
except Exception:
|
|
pass
|
|
return None
|
|
|
|
|
|
def convert_html_to_markdown(html_content: str) -> str:
|
|
"""Converte HTML para Markdown usando títulos ATX, removendo scripts e estilos."""
|
|
if not html_content or not isinstance(html_content, str) or not html_content.strip():
|
|
return ""
|
|
# Remove blocos completos de <script> e <style> incluindo conteúdo
|
|
sanitized_html = re.sub(
|
|
r"<(script|style)[^>]*>.*?</\1>",
|
|
"",
|
|
html_content,
|
|
flags=re.DOTALL | re.IGNORECASE,
|
|
)
|
|
md = markdownify.markdownify(
|
|
sanitized_html,
|
|
heading_style=markdownify.ATX,
|
|
)
|
|
return md.strip()
|
|
|
|
|
|
def resolve_article_body(article: Dict[str, Any]) -> str:
|
|
"""
|
|
Obtém o corpo do artigo exclusivamente do selected_extractor com fallback interno
|
|
(markdown/html -> text). Falha se o extrator selecionado não contiver corpo.
|
|
"""
|
|
selected = article.get("selected_extractor")
|
|
if not selected or selected not in VALID_EXTRACTORS:
|
|
raise ValueError(
|
|
f"selected_extractor inválido ou ausente: '{selected}'. "
|
|
f"Valores permitidos: {', '.join(sorted(VALID_EXTRACTORS))}"
|
|
)
|
|
|
|
extractor_data = article.get(selected)
|
|
if not isinstance(extractor_data, dict):
|
|
raise ValueError(f"Objeto do extrator selecionado '{selected}' ausente na entrada.")
|
|
|
|
body: Optional[str] = None
|
|
|
|
if selected == "trafilatura":
|
|
primary = extractor_data.get("markdown")
|
|
if isinstance(primary, str) and primary.strip():
|
|
body = primary.strip()
|
|
else:
|
|
fallback = extractor_data.get("text")
|
|
if isinstance(fallback, str) and fallback.strip():
|
|
body = fallback.strip()
|
|
|
|
elif selected == "newspaper4k":
|
|
primary = extractor_data.get("article_html")
|
|
if isinstance(primary, str) and primary.strip():
|
|
converted = convert_html_to_markdown(primary)
|
|
if converted:
|
|
body = converted
|
|
if not body:
|
|
fallback = extractor_data.get("text")
|
|
if isinstance(fallback, str) and fallback.strip():
|
|
body = fallback.strip()
|
|
|
|
elif selected == "readability":
|
|
primary = extractor_data.get("cleaned_html")
|
|
if isinstance(primary, str) and primary.strip():
|
|
converted = convert_html_to_markdown(primary)
|
|
if converted:
|
|
body = converted
|
|
if not body:
|
|
fallback = extractor_data.get("cleaned_text")
|
|
if isinstance(fallback, str) and fallback.strip():
|
|
body = fallback.strip()
|
|
|
|
if not body or not body.strip():
|
|
raise ValueError(
|
|
f"Corpo do extrator selecionado '{selected}' está vazio ou indisponível. "
|
|
"Proibido fallback para outro extrator."
|
|
)
|
|
|
|
return body.strip()
|
|
|
|
|
|
def _get_dict(data: Dict[str, Any], key: str) -> Dict[str, Any]:
|
|
"""Retorna o dicionário associado à chave ou um dicionário vazio caso não seja dict."""
|
|
val = data.get(key)
|
|
return val if isinstance(val, dict) else {}
|
|
|
|
|
|
def resolve_article_metadata(article: Dict[str, Any]) -> Dict[str, Any]:
|
|
"""
|
|
Resolve todos os metadados do artigo seguindo a matriz estrita de prioridades do PRD.
|
|
"""
|
|
selected = str(article.get("selected_extractor", ""))
|
|
input_meta = _get_dict(article, "input_meta")
|
|
trafilatura = _get_dict(article, "trafilatura")
|
|
newspaper = _get_dict(article, "newspaper4k")
|
|
readability = _get_dict(article, "readability")
|
|
sel_data = _get_dict(article, selected)
|
|
|
|
# 1. TÍTULO
|
|
title_candidates: List[Any] = []
|
|
if selected in ("trafilatura", "newspaper4k", "readability"):
|
|
title_candidates.append(sel_data.get("title"))
|
|
|
|
title_candidates.extend(
|
|
[
|
|
input_meta.get("titulo"),
|
|
article.get("page_title"),
|
|
newspaper.get("title"),
|
|
trafilatura.get("title"),
|
|
readability.get("title"),
|
|
]
|
|
)
|
|
|
|
resolved_title: Optional[str] = None
|
|
for cand in title_candidates:
|
|
val = normalize_scalar(cand)
|
|
if val:
|
|
resolved_title = val
|
|
break
|
|
|
|
if not resolved_title:
|
|
raise ValueError("Título do artigo não pôde ser resolvido a partir de nenhuma fonte.")
|
|
|
|
# 2. URL ORIGINAL
|
|
canonical_sel = None
|
|
if selected == "trafilatura":
|
|
canonical_sel = sel_data.get("canonical_url")
|
|
elif selected == "newspaper4k":
|
|
canonical_sel = sel_data.get("canonical_link")
|
|
|
|
url_candidates = [
|
|
input_meta.get("url"),
|
|
article.get("crawled_url"),
|
|
canonical_sel,
|
|
trafilatura.get("canonical_url"),
|
|
newspaper.get("canonical_link"),
|
|
]
|
|
|
|
resolved_url: Optional[str] = None
|
|
for cand in url_candidates:
|
|
val = validate_url(cand)
|
|
if val:
|
|
resolved_url = val
|
|
break
|
|
|
|
if not resolved_url:
|
|
raise ValueError(
|
|
"URL original válida (http/https) não pôde ser resolvida a partir de nenhuma fonte."
|
|
)
|
|
|
|
# 3. SUBTÍTULO / DESCRIÇÃO
|
|
desc_sel = None
|
|
if selected == "trafilatura":
|
|
desc_sel = sel_data.get("description")
|
|
elif selected == "newspaper4k":
|
|
desc_sel = sel_data.get("meta_description")
|
|
|
|
desc_candidates = [
|
|
desc_sel,
|
|
trafilatura.get("description"),
|
|
newspaper.get("meta_description"),
|
|
input_meta.get("subtitulo"),
|
|
]
|
|
|
|
resolved_subtitle: Optional[str] = None
|
|
for cand in desc_candidates:
|
|
val = normalize_scalar(cand)
|
|
if val:
|
|
# Omitir quando for igual ao título após normalização
|
|
if val.lower() != resolved_title.lower():
|
|
resolved_subtitle = val
|
|
break
|
|
|
|
# 4. AUTORES
|
|
author_sel = None
|
|
if selected == "trafilatura":
|
|
author_sel = sel_data.get("author")
|
|
elif selected == "newspaper4k":
|
|
author_sel = sel_data.get("authors")
|
|
elif selected == "readability":
|
|
author_sel = sel_data.get("author")
|
|
|
|
author_candidates = [
|
|
author_sel,
|
|
newspaper.get("authors"),
|
|
trafilatura.get("author"),
|
|
readability.get("author"),
|
|
]
|
|
|
|
resolved_authors: List[str] = []
|
|
for cand in author_candidates:
|
|
lst = normalize_list(cand, is_author=True)
|
|
if lst:
|
|
resolved_authors = lst
|
|
break
|
|
|
|
# 5. DATA DE PUBLICAÇÃO
|
|
date_sel = None
|
|
if selected == "trafilatura":
|
|
date_sel = sel_data.get("date")
|
|
elif selected == "newspaper4k":
|
|
date_sel = sel_data.get("publish_date")
|
|
|
|
date_candidates = [
|
|
date_sel,
|
|
newspaper.get("publish_date"),
|
|
trafilatura.get("date"),
|
|
input_meta.get("quando_publicado"),
|
|
]
|
|
|
|
resolved_date: Optional[str] = None
|
|
for cand in date_candidates:
|
|
val = normalize_date(cand)
|
|
if val:
|
|
resolved_date = val
|
|
break
|
|
|
|
# 6. SITE
|
|
site_sel = None
|
|
if selected == "trafilatura":
|
|
site_sel = sel_data.get("sitename")
|
|
elif selected == "newspaper4k":
|
|
site_sel = sel_data.get("meta_site_name")
|
|
|
|
url_hostname = urllib.parse.urlparse(resolved_url).netloc if resolved_url else None
|
|
|
|
site_candidates = [
|
|
site_sel,
|
|
trafilatura.get("sitename"),
|
|
newspaper.get("meta_site_name"),
|
|
trafilatura.get("hostname"),
|
|
url_hostname,
|
|
]
|
|
|
|
resolved_site: Optional[str] = None
|
|
for cand in site_candidates:
|
|
val = normalize_scalar(cand)
|
|
if val:
|
|
resolved_site = val
|
|
break
|
|
|
|
# 7. CATEGORIAS
|
|
cat_sel = sel_data.get("categories") if selected == "trafilatura" else None
|
|
cat_candidates = [
|
|
cat_sel,
|
|
trafilatura.get("categories"),
|
|
]
|
|
resolved_categories: List[str] = []
|
|
for cand in cat_candidates:
|
|
lst = normalize_list(cand)
|
|
if lst:
|
|
resolved_categories = lst
|
|
break
|
|
|
|
# 8. TAGS
|
|
tag_sel = None
|
|
if selected == "trafilatura":
|
|
tag_sel = sel_data.get("tags")
|
|
elif selected == "newspaper4k":
|
|
tag_sel = sel_data.get("tags")
|
|
|
|
tag_candidates = [
|
|
tag_sel,
|
|
trafilatura.get("tags"),
|
|
newspaper.get("tags"),
|
|
newspaper.get("meta_keywords"),
|
|
]
|
|
resolved_tags: List[str] = []
|
|
for cand in tag_candidates:
|
|
lst = normalize_list(cand)
|
|
if lst:
|
|
resolved_tags = lst
|
|
break
|
|
|
|
# 9. PALAVRAS-CHAVE
|
|
kw_candidates = [
|
|
newspaper.get("keywords"),
|
|
newspaper.get("meta_keywords"),
|
|
]
|
|
resolved_keywords: List[str] = []
|
|
for cand in kw_candidates:
|
|
lst = normalize_list(cand)
|
|
if lst:
|
|
resolved_keywords = lst
|
|
break
|
|
|
|
# 10. IDIOMA
|
|
lang_sel = None
|
|
if selected == "trafilatura":
|
|
lang_sel = sel_data.get("language")
|
|
elif selected == "newspaper4k":
|
|
lang_sel = sel_data.get("meta_lang")
|
|
|
|
lang_candidates = [
|
|
lang_sel,
|
|
trafilatura.get("language"),
|
|
newspaper.get("meta_lang"),
|
|
]
|
|
resolved_language: Optional[str] = None
|
|
for cand in lang_candidates:
|
|
val = normalize_scalar(cand)
|
|
if val:
|
|
resolved_language = val
|
|
break
|
|
|
|
# 11. IMAGEM PRINCIPAL
|
|
img_sel = None
|
|
if selected == "trafilatura":
|
|
img_sel = sel_data.get("image")
|
|
elif selected == "newspaper4k":
|
|
img_sel = sel_data.get("top_image")
|
|
|
|
img_candidates = [
|
|
img_sel,
|
|
newspaper.get("top_image"),
|
|
trafilatura.get("image"),
|
|
]
|
|
resolved_top_image: Optional[str] = None
|
|
for cand in img_candidates:
|
|
val = validate_url(cand)
|
|
if val:
|
|
resolved_top_image = val
|
|
break
|
|
|
|
return {
|
|
"title": resolved_title,
|
|
"original_url": resolved_url,
|
|
"subtitle": resolved_subtitle,
|
|
"authors": resolved_authors,
|
|
"publish_date": resolved_date,
|
|
"site_name": resolved_site,
|
|
"categories": resolved_categories,
|
|
"tags": resolved_tags,
|
|
"keywords": resolved_keywords,
|
|
"language": resolved_language,
|
|
"top_image": resolved_top_image,
|
|
}
|
|
|
|
|
|
def remove_duplicate_initial_h1(body: str, resolved_title: str) -> str:
|
|
"""
|
|
Remove o primeiro título H1 do corpo somente quando ele for igual ao título resolvido
|
|
(comparação case-insensitive após decodificação HTML e colapso de espaços).
|
|
"""
|
|
if not body:
|
|
return ""
|
|
|
|
lines = body.splitlines()
|
|
first_h1_idx: Optional[int] = None
|
|
|
|
for i, line in enumerate(lines):
|
|
stripped = line.strip()
|
|
if not stripped:
|
|
continue
|
|
if stripped.startswith("# "):
|
|
h1_text = stripped[2:].strip()
|
|
norm_h1 = normalize_scalar(h1_text)
|
|
norm_title = normalize_scalar(resolved_title)
|
|
if norm_h1 and norm_title and norm_h1.lower() == norm_title.lower():
|
|
first_h1_idx = i
|
|
break
|
|
else:
|
|
# Encontrou outro conteúdo antes de qualquer H1
|
|
break
|
|
|
|
if first_h1_idx is not None:
|
|
lines.pop(first_h1_idx)
|
|
# Remove linhas em branco residuais no início
|
|
while lines and not lines[0].strip():
|
|
lines.pop(0)
|
|
|
|
return "\n".join(lines)
|
|
|
|
|
|
def clean_body_images(body: str) -> str:
|
|
"""
|
|
Preserva imagens com URL absoluta http/https, remove relativas/data:/vazias e
|
|
deduplica repetições exatas da mesma URL de imagem.
|
|
"""
|
|
if not body:
|
|
return ""
|
|
|
|
seen_images: Set[str] = set()
|
|
|
|
def replace_image(match: re.Match) -> str:
|
|
alt_text = match.group(1)
|
|
raw_url = match.group(2).strip()
|
|
|
|
# Extrai URL se tiver atributos extras como '<url 960w>' ou srcset
|
|
clean_url = raw_url.split()[0].strip() if raw_url else ""
|
|
valid = validate_url(clean_url)
|
|
if not valid:
|
|
return ""
|
|
|
|
if valid in seen_images:
|
|
return ""
|
|
|
|
seen_images.add(valid)
|
|
return f""
|
|
|
|
# Expressão regular para imagem Markdown 
|
|
pattern = r"!\[(.*?)\]\((.*?)\)"
|
|
cleaned = re.sub(pattern, replace_image, body)
|
|
return cleaned
|
|
|
|
|
|
def assemble_markdown_document(meta: Dict[str, Any], body: str) -> str:
|
|
"""
|
|
Monta a estrutura final do documento Markdown respeitando a ordem estrita do PRD:
|
|
# Título
|
|
Subtítulo (se houver)
|
|
Bloco de metadados
|
|
 (se houver)
|
|
---
|
|
Conteúdo do corpo
|
|
"""
|
|
sections: List[str] = []
|
|
|
|
# 1. Título
|
|
sections.append(f"# {meta['title']}")
|
|
|
|
# 2. Subtítulo (quando disponível e diferente do título)
|
|
if meta.get("subtitle"):
|
|
sections.append(meta["subtitle"])
|
|
|
|
# 3. Metadados
|
|
meta_lines: List[str] = []
|
|
if meta.get("authors"):
|
|
meta_lines.append(f"**Autor:** {', '.join(meta['authors'])}")
|
|
if meta.get("publish_date"):
|
|
meta_lines.append(f"**Publicado em:** {meta['publish_date']}")
|
|
if meta.get("site_name"):
|
|
meta_lines.append(f"**Site:** {meta['site_name']}")
|
|
if meta.get("categories"):
|
|
meta_lines.append(f"**Categoria:** {', '.join(meta['categories'])}")
|
|
if meta.get("tags"):
|
|
meta_lines.append(f"**Tags:** {', '.join(meta['tags'])}")
|
|
if meta.get("keywords"):
|
|
meta_lines.append(f"**Palavras-chave:** {', '.join(meta['keywords'])}")
|
|
if meta.get("language"):
|
|
meta_lines.append(f"**Idioma:** {meta['language']}")
|
|
if meta.get("original_url"):
|
|
meta_lines.append(f"**Fonte original:** [{meta['original_url']}]({meta['original_url']})")
|
|
|
|
if meta_lines:
|
|
sections.append("\n".join(meta_lines))
|
|
|
|
# 4. Imagem principal
|
|
if meta.get("top_image"):
|
|
sections.append(f"")
|
|
|
|
# 5. Separador
|
|
sections.append("---")
|
|
|
|
# 6. Corpo
|
|
sections.append(body.strip())
|
|
|
|
# Junção com 2 quebras de linha
|
|
raw_doc = "\n\n".join(sections)
|
|
|
|
# Formatação final:
|
|
# 1. Quebras LF
|
|
raw_doc = raw_doc.replace("\r\n", "\n").replace("\r", "\n")
|
|
# 2. Remover espaços no fim de linha
|
|
lines = [line.rstrip() for line in raw_doc.split("\n")]
|
|
formatted_doc = "\n".join(lines)
|
|
# 3. Limitar linhas em branco consecutivas a no máximo 2 (\n\n\n -> \n\n)
|
|
formatted_doc = re.sub(r"\n{3,}", "\n\n", formatted_doc)
|
|
# 4. Terminar com exatamente 1 quebra de linha
|
|
formatted_doc = formatted_doc.strip() + "\n"
|
|
|
|
return formatted_doc
|
|
|
|
|
|
def convert_article(input_path: Path, output_path: Optional[Path] = None) -> Path:
|
|
"""
|
|
Executa a leitura do JSON, validação, conversão e escrita atômica do arquivo Markdown.
|
|
"""
|
|
if not input_path.exists() or not input_path.is_file():
|
|
raise FileNotFoundError(f"Arquivo de entrada não encontrado ou ilegível: '{input_path}'")
|
|
|
|
try:
|
|
content = input_path.read_text(encoding="utf-8")
|
|
data = json.loads(content)
|
|
except UnicodeDecodeError as e:
|
|
raise ValueError(f"Arquivo '{input_path}' não está codificado em UTF-8 válido: {e}")
|
|
except json.JSONDecodeError as e:
|
|
raise ValueError(f"Entrada não é um JSON válido: {e}")
|
|
|
|
if not isinstance(data, dict):
|
|
raise ValueError(f"A raiz do JSON deve ser um objeto, mas recebeu '{type(data).__name__}'.")
|
|
|
|
if "articles" in data:
|
|
raise ValueError(
|
|
"O arquivo JSON contém uma coleção 'articles'. O CLI aceita apenas um único artigo por execução."
|
|
)
|
|
|
|
# Resolução de corpo e metadados
|
|
body_raw = resolve_article_body(data)
|
|
metadata = resolve_article_metadata(data)
|
|
|
|
# Limpezas no corpo
|
|
body_no_dup_h1 = remove_duplicate_initial_h1(body_raw, metadata["title"])
|
|
body_clean_images = clean_body_images(body_no_dup_h1)
|
|
|
|
# Montagem final
|
|
final_markdown = assemble_markdown_document(metadata, body_clean_images)
|
|
|
|
# Definição do caminho de saída
|
|
if output_path is None:
|
|
target_path = input_path.with_suffix(".md")
|
|
else:
|
|
target_path = output_path
|
|
|
|
# Garantir que o diretório de destino exista
|
|
target_path.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
# Gravação atômica: arquivo temporário no mesmo diretório + replace
|
|
temp_file = target_path.parent / f".{target_path.name}.tmp"
|
|
try:
|
|
temp_file.write_text(final_markdown, encoding="utf-8", newline="\n")
|
|
temp_file.replace(target_path)
|
|
except Exception as e:
|
|
if temp_file.exists():
|
|
try:
|
|
temp_file.unlink()
|
|
except OSError:
|
|
pass
|
|
raise IOError(f"Falha na gravação do arquivo de saída '{target_path}': {e}")
|
|
|
|
return target_path
|
|
|
|
|
|
def parse_arguments(args: Optional[List[str]] = None) -> argparse.Namespace:
|
|
"""Configura o parser de argumentos do CLI."""
|
|
parser = argparse.ArgumentParser(
|
|
description="Converte JSON de artigo selecionado para Markdown estruturado e determinístico."
|
|
)
|
|
parser.add_argument(
|
|
"-i",
|
|
"--input",
|
|
required=True,
|
|
type=Path,
|
|
help="Caminho para o arquivo JSON contendo exatamente um único artigo.",
|
|
)
|
|
parser.add_argument(
|
|
"-o",
|
|
"--output",
|
|
required=False,
|
|
type=Path,
|
|
default=None,
|
|
help="Caminho do arquivo Markdown de destino (padrão: <input_stem>.md).",
|
|
)
|
|
return parser.parse_args(args)
|
|
|
|
|
|
def main() -> int:
|
|
"""Ponto de entrada do CLI."""
|
|
try:
|
|
args = parse_arguments()
|
|
except SystemExit as e:
|
|
return e.code if isinstance(e.code, int) else 2
|
|
|
|
try:
|
|
out_file = convert_article(args.input, args.output)
|
|
sys.stderr.write(f"[INFO] Artigo convertido com sucesso: '{out_file}'\n")
|
|
return 0
|
|
except (FileNotFoundError, ValueError, IOError) as e:
|
|
sys.stderr.write(f"[ERRO] {e}\n")
|
|
return 1
|
|
except Exception as e:
|
|
sys.stderr.write(f"[ERRO INESPERADO] {type(e).__name__}: {e}\n")
|
|
return 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|