feat(converter): implement deterministic JSON to Markdown article converter (spec 005)
This commit is contained in:
@@ -0,0 +1,720 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Convert Article JSON to Markdown CLI.
|
||||
|
||||
Converte o JSON de um único artigo extraído (com selected_extractor) para um documento
|
||||
Markdown (.md) limpo, padronizado e com seleção determinística de metadados.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import datetime
|
||||
import email.utils
|
||||
import html
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import urllib.parse
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Set
|
||||
|
||||
import markdownify
|
||||
|
||||
KNOWN_PLACEHOLDERS: Set[str] = {
|
||||
"null",
|
||||
"none",
|
||||
"n/a",
|
||||
"unknown",
|
||||
"[no-author]",
|
||||
"no-author",
|
||||
}
|
||||
|
||||
VALID_EXTRACTORS: Set[str] = {
|
||||
"trafilatura",
|
||||
"newspaper4k",
|
||||
"readability",
|
||||
}
|
||||
|
||||
|
||||
def normalize_scalar(value: Any) -> Optional[str]:
|
||||
"""
|
||||
Decodifica entidades HTML, remove espaços no início/fim, colapsa espaços internos
|
||||
e descarta placeholders conhecidos.
|
||||
"""
|
||||
if not isinstance(value, str):
|
||||
return None
|
||||
unescaped = html.unescape(value).strip()
|
||||
if not unescaped:
|
||||
return None
|
||||
collapsed = re.sub(r"\s+", " ", unescaped)
|
||||
if collapsed.lower() in KNOWN_PLACEHOLDERS:
|
||||
return None
|
||||
return collapsed
|
||||
|
||||
|
||||
def normalize_list(value: Any, is_author: bool = False) -> List[str]:
|
||||
"""
|
||||
Normaliza listas ou strings separadas por ponto e vírgula, descartando placeholders,
|
||||
URLs em autores e deduplicando sem diferenciar maiúsculas/minúsculas.
|
||||
"""
|
||||
if not value:
|
||||
return []
|
||||
|
||||
raw_items: List[str] = []
|
||||
if isinstance(value, list):
|
||||
for item in value:
|
||||
if isinstance(item, str):
|
||||
# Se um elemento da lista contiver ponto e vírgula, divide
|
||||
if ";" in item:
|
||||
raw_items.extend(item.split(";"))
|
||||
elif "," in item and not is_author:
|
||||
# Trafilatura às vezes emite tags separadas por vírgula em string única
|
||||
raw_items.extend(item.split(","))
|
||||
else:
|
||||
raw_items.append(item)
|
||||
elif isinstance(value, str):
|
||||
raw_items.extend(value.split(";"))
|
||||
else:
|
||||
return []
|
||||
|
||||
normalized_items: List[str] = []
|
||||
seen_lower: Set[str] = set()
|
||||
|
||||
for item in raw_items:
|
||||
norm = normalize_scalar(item)
|
||||
if not norm:
|
||||
continue
|
||||
if is_author:
|
||||
norm_lower = norm.lower()
|
||||
if (
|
||||
norm_lower.startswith("http://")
|
||||
or norm_lower.startswith("https://")
|
||||
or norm_lower.startswith("www.")
|
||||
):
|
||||
continue
|
||||
lower_key = norm.lower()
|
||||
if lower_key not in seen_lower:
|
||||
seen_lower.add(lower_key)
|
||||
normalized_items.append(norm)
|
||||
|
||||
return normalized_items
|
||||
|
||||
|
||||
def normalize_date(value: Any) -> Optional[str]:
|
||||
"""
|
||||
Interpreta datas ISO 8601 e RFC 2822 preservando fuso horário ou YYYY-MM-DD para datas puras.
|
||||
"""
|
||||
if not isinstance(value, str):
|
||||
return None
|
||||
s = value.strip()
|
||||
if not s or s.lower() in KNOWN_PLACEHOLDERS:
|
||||
return None
|
||||
|
||||
# Se for apenas data YYYY-MM-DD
|
||||
if re.match(r"^\d{4}-\d{2}-\d{2}$", s):
|
||||
return s
|
||||
|
||||
# Tenta ISO 8601
|
||||
try:
|
||||
dt = datetime.datetime.fromisoformat(s)
|
||||
return dt.isoformat()
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
# Tenta RFC 2822
|
||||
try:
|
||||
dt = email.utils.parsedate_to_datetime(s)
|
||||
return dt.isoformat()
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def validate_url(value: Any) -> Optional[str]:
|
||||
"""Valida se a URL é absoluta com protocolo http ou https e hostname não vazio."""
|
||||
norm = normalize_scalar(value)
|
||||
if not norm:
|
||||
return None
|
||||
try:
|
||||
parsed = urllib.parse.urlparse(norm)
|
||||
if parsed.scheme.lower() in ("http", "https") and parsed.netloc:
|
||||
return norm
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def convert_html_to_markdown(html_content: str) -> str:
|
||||
"""Converte HTML para Markdown usando títulos ATX, removendo scripts e estilos."""
|
||||
if not html_content or not isinstance(html_content, str) or not html_content.strip():
|
||||
return ""
|
||||
# Remove blocos completos de <script> e <style> incluindo conteúdo
|
||||
sanitized_html = re.sub(
|
||||
r"<(script|style)[^>]*>.*?</\1>",
|
||||
"",
|
||||
html_content,
|
||||
flags=re.DOTALL | re.IGNORECASE,
|
||||
)
|
||||
md = markdownify.markdownify(
|
||||
sanitized_html,
|
||||
heading_style=markdownify.ATX,
|
||||
)
|
||||
return md.strip()
|
||||
|
||||
|
||||
def resolve_article_body(article: Dict[str, Any]) -> str:
|
||||
"""
|
||||
Obtém o corpo do artigo exclusivamente do selected_extractor com fallback interno
|
||||
(markdown/html -> text). Falha se o extrator selecionado não contiver corpo.
|
||||
"""
|
||||
selected = article.get("selected_extractor")
|
||||
if not selected or selected not in VALID_EXTRACTORS:
|
||||
raise ValueError(
|
||||
f"selected_extractor inválido ou ausente: '{selected}'. "
|
||||
f"Valores permitidos: {', '.join(sorted(VALID_EXTRACTORS))}"
|
||||
)
|
||||
|
||||
extractor_data = article.get(selected)
|
||||
if not isinstance(extractor_data, dict):
|
||||
raise ValueError(f"Objeto do extrator selecionado '{selected}' ausente na entrada.")
|
||||
|
||||
body: Optional[str] = None
|
||||
|
||||
if selected == "trafilatura":
|
||||
primary = extractor_data.get("markdown")
|
||||
if isinstance(primary, str) and primary.strip():
|
||||
body = primary.strip()
|
||||
else:
|
||||
fallback = extractor_data.get("text")
|
||||
if isinstance(fallback, str) and fallback.strip():
|
||||
body = fallback.strip()
|
||||
|
||||
elif selected == "newspaper4k":
|
||||
primary = extractor_data.get("article_html")
|
||||
if isinstance(primary, str) and primary.strip():
|
||||
converted = convert_html_to_markdown(primary)
|
||||
if converted:
|
||||
body = converted
|
||||
if not body:
|
||||
fallback = extractor_data.get("text")
|
||||
if isinstance(fallback, str) and fallback.strip():
|
||||
body = fallback.strip()
|
||||
|
||||
elif selected == "readability":
|
||||
primary = extractor_data.get("cleaned_html")
|
||||
if isinstance(primary, str) and primary.strip():
|
||||
converted = convert_html_to_markdown(primary)
|
||||
if converted:
|
||||
body = converted
|
||||
if not body:
|
||||
fallback = extractor_data.get("cleaned_text")
|
||||
if isinstance(fallback, str) and fallback.strip():
|
||||
body = fallback.strip()
|
||||
|
||||
if not body or not body.strip():
|
||||
raise ValueError(
|
||||
f"Corpo do extrator selecionado '{selected}' está vazio ou indisponível. "
|
||||
"Proibido fallback para outro extrator."
|
||||
)
|
||||
|
||||
return body.strip()
|
||||
|
||||
|
||||
def _get_dict(data: Dict[str, Any], key: str) -> Dict[str, Any]:
|
||||
"""Retorna o dicionário associado à chave ou um dicionário vazio caso não seja dict."""
|
||||
val = data.get(key)
|
||||
return val if isinstance(val, dict) else {}
|
||||
|
||||
|
||||
def resolve_article_metadata(article: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""
|
||||
Resolve todos os metadados do artigo seguindo a matriz estrita de prioridades do PRD.
|
||||
"""
|
||||
selected = str(article.get("selected_extractor", ""))
|
||||
input_meta = _get_dict(article, "input_meta")
|
||||
trafilatura = _get_dict(article, "trafilatura")
|
||||
newspaper = _get_dict(article, "newspaper4k")
|
||||
readability = _get_dict(article, "readability")
|
||||
sel_data = _get_dict(article, selected)
|
||||
|
||||
# 1. TÍTULO
|
||||
title_candidates: List[Any] = []
|
||||
if selected in ("trafilatura", "newspaper4k", "readability"):
|
||||
title_candidates.append(sel_data.get("title"))
|
||||
|
||||
title_candidates.extend(
|
||||
[
|
||||
input_meta.get("titulo"),
|
||||
article.get("page_title"),
|
||||
newspaper.get("title"),
|
||||
trafilatura.get("title"),
|
||||
readability.get("title"),
|
||||
]
|
||||
)
|
||||
|
||||
resolved_title: Optional[str] = None
|
||||
for cand in title_candidates:
|
||||
val = normalize_scalar(cand)
|
||||
if val:
|
||||
resolved_title = val
|
||||
break
|
||||
|
||||
if not resolved_title:
|
||||
raise ValueError("Título do artigo não pôde ser resolvido a partir de nenhuma fonte.")
|
||||
|
||||
# 2. URL ORIGINAL
|
||||
canonical_sel = None
|
||||
if selected == "trafilatura":
|
||||
canonical_sel = sel_data.get("canonical_url")
|
||||
elif selected == "newspaper4k":
|
||||
canonical_sel = sel_data.get("canonical_link")
|
||||
|
||||
url_candidates = [
|
||||
input_meta.get("url"),
|
||||
article.get("crawled_url"),
|
||||
canonical_sel,
|
||||
trafilatura.get("canonical_url"),
|
||||
newspaper.get("canonical_link"),
|
||||
]
|
||||
|
||||
resolved_url: Optional[str] = None
|
||||
for cand in url_candidates:
|
||||
val = validate_url(cand)
|
||||
if val:
|
||||
resolved_url = val
|
||||
break
|
||||
|
||||
if not resolved_url:
|
||||
raise ValueError(
|
||||
"URL original válida (http/https) não pôde ser resolvida a partir de nenhuma fonte."
|
||||
)
|
||||
|
||||
# 3. SUBTÍTULO / DESCRIÇÃO
|
||||
desc_sel = None
|
||||
if selected == "trafilatura":
|
||||
desc_sel = sel_data.get("description")
|
||||
elif selected == "newspaper4k":
|
||||
desc_sel = sel_data.get("meta_description")
|
||||
|
||||
desc_candidates = [
|
||||
desc_sel,
|
||||
trafilatura.get("description"),
|
||||
newspaper.get("meta_description"),
|
||||
input_meta.get("subtitulo"),
|
||||
]
|
||||
|
||||
resolved_subtitle: Optional[str] = None
|
||||
for cand in desc_candidates:
|
||||
val = normalize_scalar(cand)
|
||||
if val:
|
||||
# Omitir quando for igual ao título após normalização
|
||||
if val.lower() != resolved_title.lower():
|
||||
resolved_subtitle = val
|
||||
break
|
||||
|
||||
# 4. AUTORES
|
||||
author_sel = None
|
||||
if selected == "trafilatura":
|
||||
author_sel = sel_data.get("author")
|
||||
elif selected == "newspaper4k":
|
||||
author_sel = sel_data.get("authors")
|
||||
elif selected == "readability":
|
||||
author_sel = sel_data.get("author")
|
||||
|
||||
author_candidates = [
|
||||
author_sel,
|
||||
newspaper.get("authors"),
|
||||
trafilatura.get("author"),
|
||||
readability.get("author"),
|
||||
]
|
||||
|
||||
resolved_authors: List[str] = []
|
||||
for cand in author_candidates:
|
||||
lst = normalize_list(cand, is_author=True)
|
||||
if lst:
|
||||
resolved_authors = lst
|
||||
break
|
||||
|
||||
# 5. DATA DE PUBLICAÇÃO
|
||||
date_sel = None
|
||||
if selected == "trafilatura":
|
||||
date_sel = sel_data.get("date")
|
||||
elif selected == "newspaper4k":
|
||||
date_sel = sel_data.get("publish_date")
|
||||
|
||||
date_candidates = [
|
||||
date_sel,
|
||||
newspaper.get("publish_date"),
|
||||
trafilatura.get("date"),
|
||||
input_meta.get("quando_publicado"),
|
||||
]
|
||||
|
||||
resolved_date: Optional[str] = None
|
||||
for cand in date_candidates:
|
||||
val = normalize_date(cand)
|
||||
if val:
|
||||
resolved_date = val
|
||||
break
|
||||
|
||||
# 6. SITE
|
||||
site_sel = None
|
||||
if selected == "trafilatura":
|
||||
site_sel = sel_data.get("sitename")
|
||||
elif selected == "newspaper4k":
|
||||
site_sel = sel_data.get("meta_site_name")
|
||||
|
||||
url_hostname = urllib.parse.urlparse(resolved_url).netloc if resolved_url else None
|
||||
|
||||
site_candidates = [
|
||||
site_sel,
|
||||
trafilatura.get("sitename"),
|
||||
newspaper.get("meta_site_name"),
|
||||
trafilatura.get("hostname"),
|
||||
url_hostname,
|
||||
]
|
||||
|
||||
resolved_site: Optional[str] = None
|
||||
for cand in site_candidates:
|
||||
val = normalize_scalar(cand)
|
||||
if val:
|
||||
resolved_site = val
|
||||
break
|
||||
|
||||
# 7. CATEGORIAS
|
||||
cat_sel = sel_data.get("categories") if selected == "trafilatura" else None
|
||||
cat_candidates = [
|
||||
cat_sel,
|
||||
trafilatura.get("categories"),
|
||||
]
|
||||
resolved_categories: List[str] = []
|
||||
for cand in cat_candidates:
|
||||
lst = normalize_list(cand)
|
||||
if lst:
|
||||
resolved_categories = lst
|
||||
break
|
||||
|
||||
# 8. TAGS
|
||||
tag_sel = None
|
||||
if selected == "trafilatura":
|
||||
tag_sel = sel_data.get("tags")
|
||||
elif selected == "newspaper4k":
|
||||
tag_sel = sel_data.get("tags")
|
||||
|
||||
tag_candidates = [
|
||||
tag_sel,
|
||||
trafilatura.get("tags"),
|
||||
newspaper.get("tags"),
|
||||
newspaper.get("meta_keywords"),
|
||||
]
|
||||
resolved_tags: List[str] = []
|
||||
for cand in tag_candidates:
|
||||
lst = normalize_list(cand)
|
||||
if lst:
|
||||
resolved_tags = lst
|
||||
break
|
||||
|
||||
# 9. PALAVRAS-CHAVE
|
||||
kw_candidates = [
|
||||
newspaper.get("keywords"),
|
||||
newspaper.get("meta_keywords"),
|
||||
]
|
||||
resolved_keywords: List[str] = []
|
||||
for cand in kw_candidates:
|
||||
lst = normalize_list(cand)
|
||||
if lst:
|
||||
resolved_keywords = lst
|
||||
break
|
||||
|
||||
# 10. IDIOMA
|
||||
lang_sel = None
|
||||
if selected == "trafilatura":
|
||||
lang_sel = sel_data.get("language")
|
||||
elif selected == "newspaper4k":
|
||||
lang_sel = sel_data.get("meta_lang")
|
||||
|
||||
lang_candidates = [
|
||||
lang_sel,
|
||||
trafilatura.get("language"),
|
||||
newspaper.get("meta_lang"),
|
||||
]
|
||||
resolved_language: Optional[str] = None
|
||||
for cand in lang_candidates:
|
||||
val = normalize_scalar(cand)
|
||||
if val:
|
||||
resolved_language = val
|
||||
break
|
||||
|
||||
# 11. IMAGEM PRINCIPAL
|
||||
img_sel = None
|
||||
if selected == "trafilatura":
|
||||
img_sel = sel_data.get("image")
|
||||
elif selected == "newspaper4k":
|
||||
img_sel = sel_data.get("top_image")
|
||||
|
||||
img_candidates = [
|
||||
img_sel,
|
||||
newspaper.get("top_image"),
|
||||
trafilatura.get("image"),
|
||||
]
|
||||
resolved_top_image: Optional[str] = None
|
||||
for cand in img_candidates:
|
||||
val = validate_url(cand)
|
||||
if val:
|
||||
resolved_top_image = val
|
||||
break
|
||||
|
||||
return {
|
||||
"title": resolved_title,
|
||||
"original_url": resolved_url,
|
||||
"subtitle": resolved_subtitle,
|
||||
"authors": resolved_authors,
|
||||
"publish_date": resolved_date,
|
||||
"site_name": resolved_site,
|
||||
"categories": resolved_categories,
|
||||
"tags": resolved_tags,
|
||||
"keywords": resolved_keywords,
|
||||
"language": resolved_language,
|
||||
"top_image": resolved_top_image,
|
||||
}
|
||||
|
||||
|
||||
def remove_duplicate_initial_h1(body: str, resolved_title: str) -> str:
|
||||
"""
|
||||
Remove o primeiro título H1 do corpo somente quando ele for igual ao título resolvido
|
||||
(comparação case-insensitive após decodificação HTML e colapso de espaços).
|
||||
"""
|
||||
if not body:
|
||||
return ""
|
||||
|
||||
lines = body.splitlines()
|
||||
first_h1_idx: Optional[int] = None
|
||||
|
||||
for i, line in enumerate(lines):
|
||||
stripped = line.strip()
|
||||
if not stripped:
|
||||
continue
|
||||
if stripped.startswith("# "):
|
||||
h1_text = stripped[2:].strip()
|
||||
norm_h1 = normalize_scalar(h1_text)
|
||||
norm_title = normalize_scalar(resolved_title)
|
||||
if norm_h1 and norm_title and norm_h1.lower() == norm_title.lower():
|
||||
first_h1_idx = i
|
||||
break
|
||||
else:
|
||||
# Encontrou outro conteúdo antes de qualquer H1
|
||||
break
|
||||
|
||||
if first_h1_idx is not None:
|
||||
lines.pop(first_h1_idx)
|
||||
# Remove linhas em branco residuais no início
|
||||
while lines and not lines[0].strip():
|
||||
lines.pop(0)
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def clean_body_images(body: str) -> str:
|
||||
"""
|
||||
Preserva imagens com URL absoluta http/https, remove relativas/data:/vazias e
|
||||
deduplica repetições exatas da mesma URL de imagem.
|
||||
"""
|
||||
if not body:
|
||||
return ""
|
||||
|
||||
seen_images: Set[str] = set()
|
||||
|
||||
def replace_image(match: re.Match) -> str:
|
||||
alt_text = match.group(1)
|
||||
raw_url = match.group(2).strip()
|
||||
|
||||
# Extrai URL se tiver atributos extras como '<url 960w>' ou srcset
|
||||
clean_url = raw_url.split()[0].strip() if raw_url else ""
|
||||
valid = validate_url(clean_url)
|
||||
if not valid:
|
||||
return ""
|
||||
|
||||
if valid in seen_images:
|
||||
return ""
|
||||
|
||||
seen_images.add(valid)
|
||||
return f""
|
||||
|
||||
# Expressão regular para imagem Markdown 
|
||||
pattern = r"!\[(.*?)\]\((.*?)\)"
|
||||
cleaned = re.sub(pattern, replace_image, body)
|
||||
return cleaned
|
||||
|
||||
|
||||
def assemble_markdown_document(meta: Dict[str, Any], body: str) -> str:
|
||||
"""
|
||||
Monta a estrutura final do documento Markdown respeitando a ordem estrita do PRD:
|
||||
# Título
|
||||
Subtítulo (se houver)
|
||||
Bloco de metadados
|
||||
 (se houver)
|
||||
---
|
||||
Conteúdo do corpo
|
||||
"""
|
||||
sections: List[str] = []
|
||||
|
||||
# 1. Título
|
||||
sections.append(f"# {meta['title']}")
|
||||
|
||||
# 2. Subtítulo (quando disponível e diferente do título)
|
||||
if meta.get("subtitle"):
|
||||
sections.append(meta["subtitle"])
|
||||
|
||||
# 3. Metadados
|
||||
meta_lines: List[str] = []
|
||||
if meta.get("authors"):
|
||||
meta_lines.append(f"**Autor:** {', '.join(meta['authors'])}")
|
||||
if meta.get("publish_date"):
|
||||
meta_lines.append(f"**Publicado em:** {meta['publish_date']}")
|
||||
if meta.get("site_name"):
|
||||
meta_lines.append(f"**Site:** {meta['site_name']}")
|
||||
if meta.get("categories"):
|
||||
meta_lines.append(f"**Categoria:** {', '.join(meta['categories'])}")
|
||||
if meta.get("tags"):
|
||||
meta_lines.append(f"**Tags:** {', '.join(meta['tags'])}")
|
||||
if meta.get("keywords"):
|
||||
meta_lines.append(f"**Palavras-chave:** {', '.join(meta['keywords'])}")
|
||||
if meta.get("language"):
|
||||
meta_lines.append(f"**Idioma:** {meta['language']}")
|
||||
if meta.get("original_url"):
|
||||
meta_lines.append(f"**Fonte original:** [{meta['original_url']}]({meta['original_url']})")
|
||||
|
||||
if meta_lines:
|
||||
sections.append("\n".join(meta_lines))
|
||||
|
||||
# 4. Imagem principal
|
||||
if meta.get("top_image"):
|
||||
sections.append(f"")
|
||||
|
||||
# 5. Separador
|
||||
sections.append("---")
|
||||
|
||||
# 6. Corpo
|
||||
sections.append(body.strip())
|
||||
|
||||
# Junção com 2 quebras de linha
|
||||
raw_doc = "\n\n".join(sections)
|
||||
|
||||
# Formatação final:
|
||||
# 1. Quebras LF
|
||||
raw_doc = raw_doc.replace("\r\n", "\n").replace("\r", "\n")
|
||||
# 2. Remover espaços no fim de linha
|
||||
lines = [line.rstrip() for line in raw_doc.split("\n")]
|
||||
formatted_doc = "\n".join(lines)
|
||||
# 3. Limitar linhas em branco consecutivas a no máximo 2 (\n\n\n -> \n\n)
|
||||
formatted_doc = re.sub(r"\n{3,}", "\n\n", formatted_doc)
|
||||
# 4. Terminar com exatamente 1 quebra de linha
|
||||
formatted_doc = formatted_doc.strip() + "\n"
|
||||
|
||||
return formatted_doc
|
||||
|
||||
|
||||
def convert_article(input_path: Path, output_path: Optional[Path] = None) -> Path:
|
||||
"""
|
||||
Executa a leitura do JSON, validação, conversão e escrita atômica do arquivo Markdown.
|
||||
"""
|
||||
if not input_path.exists() or not input_path.is_file():
|
||||
raise FileNotFoundError(f"Arquivo de entrada não encontrado ou ilegível: '{input_path}'")
|
||||
|
||||
try:
|
||||
content = input_path.read_text(encoding="utf-8")
|
||||
data = json.loads(content)
|
||||
except UnicodeDecodeError as e:
|
||||
raise ValueError(f"Arquivo '{input_path}' não está codificado em UTF-8 válido: {e}")
|
||||
except json.JSONDecodeError as e:
|
||||
raise ValueError(f"Entrada não é um JSON válido: {e}")
|
||||
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError(f"A raiz do JSON deve ser um objeto, mas recebeu '{type(data).__name__}'.")
|
||||
|
||||
if "articles" in data:
|
||||
raise ValueError(
|
||||
"O arquivo JSON contém uma coleção 'articles'. O CLI aceita apenas um único artigo por execução."
|
||||
)
|
||||
|
||||
# Resolução de corpo e metadados
|
||||
body_raw = resolve_article_body(data)
|
||||
metadata = resolve_article_metadata(data)
|
||||
|
||||
# Limpezas no corpo
|
||||
body_no_dup_h1 = remove_duplicate_initial_h1(body_raw, metadata["title"])
|
||||
body_clean_images = clean_body_images(body_no_dup_h1)
|
||||
|
||||
# Montagem final
|
||||
final_markdown = assemble_markdown_document(metadata, body_clean_images)
|
||||
|
||||
# Definição do caminho de saída
|
||||
if output_path is None:
|
||||
target_path = input_path.with_suffix(".md")
|
||||
else:
|
||||
target_path = output_path
|
||||
|
||||
# Garantir que o diretório de destino exista
|
||||
target_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Gravação atômica: arquivo temporário no mesmo diretório + replace
|
||||
temp_file = target_path.parent / f".{target_path.name}.tmp"
|
||||
try:
|
||||
temp_file.write_text(final_markdown, encoding="utf-8", newline="\n")
|
||||
temp_file.replace(target_path)
|
||||
except Exception as e:
|
||||
if temp_file.exists():
|
||||
try:
|
||||
temp_file.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
raise IOError(f"Falha na gravação do arquivo de saída '{target_path}': {e}")
|
||||
|
||||
return target_path
|
||||
|
||||
|
||||
def parse_arguments(args: Optional[List[str]] = None) -> argparse.Namespace:
|
||||
"""Configura o parser de argumentos do CLI."""
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Converte JSON de artigo selecionado para Markdown estruturado e determinístico."
|
||||
)
|
||||
parser.add_argument(
|
||||
"-i",
|
||||
"--input",
|
||||
required=True,
|
||||
type=Path,
|
||||
help="Caminho para o arquivo JSON contendo exatamente um único artigo.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output",
|
||||
required=False,
|
||||
type=Path,
|
||||
default=None,
|
||||
help="Caminho do arquivo Markdown de destino (padrão: <input_stem>.md).",
|
||||
)
|
||||
return parser.parse_args(args)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
"""Ponto de entrada do CLI."""
|
||||
try:
|
||||
args = parse_arguments()
|
||||
except SystemExit as e:
|
||||
return e.code if isinstance(e.code, int) else 2
|
||||
|
||||
try:
|
||||
out_file = convert_article(args.input, args.output)
|
||||
sys.stderr.write(f"[INFO] Artigo convertido com sucesso: '{out_file}'\n")
|
||||
return 0
|
||||
except (FileNotFoundError, ValueError, IOError) as e:
|
||||
sys.stderr.write(f"[ERRO] {e}\n")
|
||||
return 1
|
||||
except Exception as e:
|
||||
sys.stderr.write(f"[ERRO INESPERADO] {type(e).__name__}: {e}\n")
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user