- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
771 lines
26 KiB
Python
771 lines
26 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability).
|
|
|
|
Lê listagens JSON de notícias (ex: out/river_plate.json), acessa e renderiza as páginas
|
|
em modo stealth headless utilizando Foxcape reutilizando a mesma sessão de navegador,
|
|
executa a extração em paralelo/sequência com 3 motores de conteúdo (Trafilatura,
|
|
Newspaper4k e Readability) e salva o resultado enriquecido e higienizado em JSON.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import sys
|
|
import time
|
|
from dataclasses import dataclass, field
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any, Literal
|
|
|
|
import trafilatura
|
|
from bs4 import BeautifulSoup
|
|
from foxcape import Foxcape, FoxcapeConfig
|
|
from newspaper import Article
|
|
from readability import Document
|
|
|
|
# ==============================================================================
|
|
# Modelos de Dados e Dataclasses
|
|
# ==============================================================================
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class InputArticle:
|
|
"""Metadados originais da notícia contida no JSON de entrada."""
|
|
|
|
titulo: str
|
|
url: str
|
|
subtitulo: str | None = None
|
|
quando_publicado: str | None = None
|
|
pagina: int = 1
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"titulo": self.titulo,
|
|
"subtitulo": self.subtitulo,
|
|
"quando_publicado": self.quando_publicado,
|
|
"url": self.url,
|
|
"pagina": self.pagina,
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class TrafilaturaData:
|
|
"""Dados completos extraídos pelo motor Trafilatura."""
|
|
|
|
title: str | None = None
|
|
author: str | None = None
|
|
date: str | None = None
|
|
description: str | None = None
|
|
sitename: str | None = None
|
|
hostname: str | None = None
|
|
language: str | None = None
|
|
categories: list[str] = field(default_factory=list)
|
|
tags: list[str] = field(default_factory=list)
|
|
canonical_url: str | None = None
|
|
image: str | None = None
|
|
pagetype: str | None = None
|
|
fingerprint: str | None = None
|
|
license: str | None = None
|
|
comments: str | None = None
|
|
text: str = ""
|
|
markdown: str | None = None
|
|
raw_json: dict[str, Any] | None = None
|
|
error: str | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"title": self.title,
|
|
"author": self.author,
|
|
"date": self.date,
|
|
"description": self.description,
|
|
"sitename": self.sitename,
|
|
"hostname": self.hostname,
|
|
"language": self.language,
|
|
"categories": self.categories,
|
|
"tags": self.tags,
|
|
"canonical_url": self.canonical_url,
|
|
"image": self.image,
|
|
"pagetype": self.pagetype,
|
|
"fingerprint": self.fingerprint,
|
|
"license": self.license,
|
|
"comments": self.comments,
|
|
"text": self.text,
|
|
"markdown": self.markdown,
|
|
"raw_json": self.raw_json,
|
|
"error": self.error,
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class NewspaperData:
|
|
"""Dados completos extraídos e enriquecidos com NLP pelo motor Newspaper4k."""
|
|
|
|
title: str | None = None
|
|
authors: list[str] = field(default_factory=list)
|
|
publish_date: str | None = None
|
|
text: str = ""
|
|
summary: str | None = None
|
|
keywords: list[str] = field(default_factory=list)
|
|
keyword_scores: dict[str, float] = field(default_factory=dict)
|
|
top_image: str | None = None
|
|
images: list[str] = field(default_factory=list)
|
|
movies: list[str] = field(default_factory=list)
|
|
tags: list[str] = field(default_factory=list)
|
|
canonical_link: str | None = None
|
|
article_html: str | None = None
|
|
meta_description: str | None = None
|
|
meta_keywords: list[str] = field(default_factory=list)
|
|
meta_favicon: str | None = None
|
|
meta_site_name: str | None = None
|
|
meta_lang: str | None = None
|
|
meta_data: dict[str, Any] = field(default_factory=dict)
|
|
error: str | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"title": self.title,
|
|
"authors": self.authors,
|
|
"publish_date": self.publish_date,
|
|
"text": self.text,
|
|
"summary": self.summary,
|
|
"keywords": self.keywords,
|
|
"keyword_scores": self.keyword_scores,
|
|
"top_image": self.top_image,
|
|
"images": self.images,
|
|
"movies": self.movies,
|
|
"tags": self.tags,
|
|
"canonical_link": self.canonical_link,
|
|
"article_html": self.article_html,
|
|
"meta_description": self.meta_description,
|
|
"meta_keywords": self.meta_keywords,
|
|
"meta_favicon": self.meta_favicon,
|
|
"meta_site_name": self.meta_site_name,
|
|
"meta_lang": self.meta_lang,
|
|
"meta_data": self.meta_data,
|
|
"error": self.error,
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ReadabilityData:
|
|
"""Dados completos higienizados pelo algoritmo Readability."""
|
|
|
|
title: str | None = None
|
|
short_title: str | None = None
|
|
author: str | None = None
|
|
cleaned_html: str | None = None
|
|
cleaned_text: str | None = None
|
|
error: str | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"title": self.title,
|
|
"short_title": self.short_title,
|
|
"author": self.author,
|
|
"cleaned_html": self.cleaned_html,
|
|
"cleaned_text": self.cleaned_text,
|
|
"error": self.error,
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ExtractedArticle:
|
|
"""Resultado consolidado da extração de um artigo."""
|
|
|
|
input_meta: InputArticle
|
|
extraction_status: Literal["success", "failed"]
|
|
error_message: str | None
|
|
crawled_url: str
|
|
page_title: str | None
|
|
http_status: int | None
|
|
trafilatura: TrafilaturaData | None = None
|
|
newspaper4k: NewspaperData | None = None
|
|
readability: ReadabilityData | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"input_meta": self.input_meta.to_dict(),
|
|
"extraction_status": self.extraction_status,
|
|
"error_message": self.error_message,
|
|
"crawled_url": self.crawled_url,
|
|
"page_title": self.page_title,
|
|
"http_status": self.http_status,
|
|
"trafilatura": self.trafilatura.to_dict() if self.trafilatura else None,
|
|
"newspaper4k": self.newspaper4k.to_dict() if self.newspaper4k else None,
|
|
"readability": self.readability.to_dict() if self.readability else None,
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ExtractionBatchReport:
|
|
"""Relatório consolidado de saída do processamento de um lote."""
|
|
|
|
source_file: str
|
|
processed_at: str
|
|
total_articles: int
|
|
successful_articles: int
|
|
failed_articles: int
|
|
articles: list[ExtractedArticle] = field(default_factory=list)
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"source_file": self.source_file,
|
|
"processed_at": self.processed_at,
|
|
"total_articles": self.total_articles,
|
|
"successful_articles": self.successful_articles,
|
|
"failed_articles": self.failed_articles,
|
|
"articles": [a.to_dict() for a in self.articles],
|
|
}
|
|
|
|
|
|
# ==============================================================================
|
|
# Parsers / Extratores Especializados
|
|
# ==============================================================================
|
|
|
|
|
|
class TrafilaturaExtractor:
|
|
"""Motor de extração baseado na biblioteca Trafilatura."""
|
|
|
|
@staticmethod
|
|
def extract(html: str, url: str | None = None) -> TrafilaturaData:
|
|
try:
|
|
# Extração bare document completa
|
|
doc = trafilatura.bare_extraction(
|
|
html,
|
|
url=url,
|
|
include_comments=True,
|
|
include_tables=True,
|
|
include_images=True,
|
|
include_links=True,
|
|
include_formatting=True,
|
|
with_metadata=True,
|
|
)
|
|
|
|
# Extração em markdown
|
|
markdown_text = trafilatura.extract(
|
|
html,
|
|
output_format="markdown",
|
|
include_comments=True,
|
|
include_tables=True,
|
|
include_images=True,
|
|
include_links=True,
|
|
include_formatting=True,
|
|
url=url,
|
|
)
|
|
|
|
# Extração em JSON nativo
|
|
json_output_str = trafilatura.extract(
|
|
html,
|
|
output_format="json",
|
|
include_comments=True,
|
|
include_tables=True,
|
|
include_images=True,
|
|
include_links=True,
|
|
url=url,
|
|
)
|
|
raw_json = json.loads(json_output_str) if json_output_str else None
|
|
|
|
if doc:
|
|
if isinstance(doc, dict):
|
|
categories = list(doc.get("categories", [])) if doc.get("categories") else []
|
|
tags = list(doc.get("tags", [])) if doc.get("tags") else []
|
|
return TrafilaturaData(
|
|
title=doc.get("title"),
|
|
author=doc.get("author"),
|
|
date=doc.get("date"),
|
|
description=doc.get("description"),
|
|
sitename=doc.get("sitename"),
|
|
hostname=doc.get("hostname"),
|
|
language=doc.get("language"),
|
|
categories=categories,
|
|
tags=tags,
|
|
canonical_url=doc.get("url") or url,
|
|
image=doc.get("image"),
|
|
pagetype=doc.get("pagetype"),
|
|
fingerprint=doc.get("fingerprint"),
|
|
license=doc.get("license"),
|
|
comments=doc.get("comments"),
|
|
text=(doc.get("text") or "").strip(),
|
|
markdown=(markdown_text or "").strip() if markdown_text else None,
|
|
raw_json=raw_json,
|
|
error=None,
|
|
)
|
|
else:
|
|
categories = list(doc.categories) if doc.categories else []
|
|
tags = list(doc.tags) if doc.tags else []
|
|
return TrafilaturaData(
|
|
title=doc.title,
|
|
author=doc.author,
|
|
date=doc.date,
|
|
description=doc.description,
|
|
sitename=doc.sitename,
|
|
hostname=doc.hostname,
|
|
language=doc.language,
|
|
categories=categories,
|
|
tags=tags,
|
|
canonical_url=doc.url or url,
|
|
image=doc.image,
|
|
pagetype=doc.pagetype,
|
|
fingerprint=doc.fingerprint,
|
|
license=doc.license,
|
|
comments=doc.comments,
|
|
text=(doc.text or "").strip(),
|
|
markdown=(markdown_text or "").strip() if markdown_text else None,
|
|
raw_json=raw_json,
|
|
error=None,
|
|
)
|
|
else:
|
|
raw_text = trafilatura.extract(html, output_format="txt", url=url) or ""
|
|
return TrafilaturaData(
|
|
title=raw_json.get("title") if raw_json else None,
|
|
text=raw_text.strip(),
|
|
markdown=markdown_text.strip() if markdown_text else None,
|
|
raw_json=raw_json,
|
|
error=None,
|
|
)
|
|
except Exception as e:
|
|
return TrafilaturaData(error=str(e))
|
|
|
|
|
|
class NewspaperExtractor:
|
|
"""Motor de extração baseado no Newspaper4k com NLP."""
|
|
|
|
@staticmethod
|
|
def extract(html: str, url: str = "", language: str = "en") -> NewspaperData:
|
|
try:
|
|
lang_code = language.split("-")[0].lower() if language else "en"
|
|
|
|
article = Article(url=url, language=lang_code)
|
|
article.download(input_html=html)
|
|
article.parse()
|
|
|
|
# Executar NLP para summary e keywords com fallback gracioso
|
|
try:
|
|
article.nlp()
|
|
summary = article.summary
|
|
keywords = list(article.keywords) if article.keywords else []
|
|
keyword_scores = getattr(article, "keyword_scores", {}) or {}
|
|
except Exception:
|
|
summary = None
|
|
keywords = []
|
|
keyword_scores = {}
|
|
|
|
publish_date_str = (
|
|
article.publish_date.isoformat()
|
|
if article.publish_date and hasattr(article.publish_date, "isoformat")
|
|
else str(article.publish_date)
|
|
if article.publish_date
|
|
else None
|
|
)
|
|
|
|
# Metadados e tags adicionais
|
|
meta_keywords = list(article.meta_keywords) if article.meta_keywords else []
|
|
tags = list(article.tags) if getattr(article, "tags", None) else []
|
|
movies = list(article.movies) if getattr(article, "movies", None) else []
|
|
images = list(article.images) if article.images else []
|
|
|
|
return NewspaperData(
|
|
title=article.title or None,
|
|
authors=list(article.authors) if article.authors else [],
|
|
publish_date=publish_date_str,
|
|
text=article.text or "",
|
|
summary=summary,
|
|
keywords=keywords,
|
|
keyword_scores=dict(keyword_scores),
|
|
top_image=article.top_image or getattr(article, "meta_img", None) or None,
|
|
images=images,
|
|
movies=movies,
|
|
tags=tags,
|
|
canonical_link=getattr(article, "canonical_link", None) or None,
|
|
article_html=getattr(article, "article_html", None) or None,
|
|
meta_description=getattr(article, "meta_description", None) or None,
|
|
meta_keywords=meta_keywords,
|
|
meta_favicon=getattr(article, "meta_favicon", None) or None,
|
|
meta_site_name=getattr(article, "meta_site_name", None) or None,
|
|
meta_lang=getattr(article, "meta_lang", None) or None,
|
|
meta_data=dict(article.meta_data) if article.meta_data else {},
|
|
error=None,
|
|
)
|
|
except Exception as e:
|
|
return NewspaperData(error=str(e))
|
|
|
|
|
|
class ReadabilityExtractor:
|
|
"""Motor de extração baseado no algoritmo Readability (readability-lxml)."""
|
|
|
|
@staticmethod
|
|
def extract(html: str) -> ReadabilityData:
|
|
try:
|
|
doc = Document(html)
|
|
title = doc.title()
|
|
short_title = doc.short_title()
|
|
cleaned_html = doc.summary()
|
|
author = None
|
|
try:
|
|
author = doc.author()
|
|
except Exception:
|
|
pass
|
|
|
|
# Extração de texto limpo a partir do HTML higienizado
|
|
soup = BeautifulSoup(cleaned_html, "html.parser")
|
|
cleaned_text = soup.get_text(separator="\n\n", strip=True)
|
|
|
|
return ReadabilityData(
|
|
title=title or None,
|
|
short_title=short_title or None,
|
|
author=author or None,
|
|
cleaned_html=cleaned_html or None,
|
|
cleaned_text=cleaned_text or None,
|
|
error=None,
|
|
)
|
|
except Exception as e:
|
|
return ReadabilityData(error=str(e))
|
|
|
|
|
|
def extract_all_engines(
|
|
html: str, url: str = "", language: str = "en"
|
|
) -> tuple[TrafilaturaData, NewspaperData, ReadabilityData]:
|
|
"""Executa a extração simultânea pelos três motores de conteúdo com isolamento defensivo."""
|
|
try:
|
|
traf_data = TrafilaturaExtractor.extract(html, url=url)
|
|
except Exception as exc:
|
|
traf_data = TrafilaturaData(
|
|
title=None,
|
|
author=None,
|
|
date=None,
|
|
description=None,
|
|
categories=[],
|
|
tags=[],
|
|
canonical_url=None,
|
|
text="",
|
|
raw_json=None,
|
|
error=str(exc),
|
|
)
|
|
|
|
try:
|
|
newspaper_data = NewspaperExtractor.extract(html, url=url, language=language)
|
|
except Exception as exc:
|
|
newspaper_data = NewspaperData(
|
|
title=None,
|
|
authors=[],
|
|
publish_date=None,
|
|
text="",
|
|
summary=None,
|
|
keywords=[],
|
|
top_image=None,
|
|
images=[],
|
|
meta_data={},
|
|
error=str(exc),
|
|
)
|
|
|
|
try:
|
|
readability_data = ReadabilityExtractor.extract(html)
|
|
except Exception as exc:
|
|
readability_data = ReadabilityData(
|
|
title=None,
|
|
short_title=None,
|
|
cleaned_html=None,
|
|
cleaned_text=None,
|
|
error=str(exc),
|
|
)
|
|
|
|
return traf_data, newspaper_data, readability_data
|
|
|
|
|
|
# ==============================================================================
|
|
# Motor de Navegação Stealth Headless com Foxcape
|
|
# ==============================================================================
|
|
|
|
|
|
class ArticleCrawler:
|
|
"""Gerenciador de ciclo de vida e requisições via Foxcape Headless."""
|
|
|
|
def __init__(self, timeout_sec: int = 30) -> None:
|
|
self.timeout_sec = timeout_sec
|
|
self.timeout_ms = timeout_sec * 1000
|
|
self._scraper: Foxcape | None = None
|
|
|
|
def start(self) -> None:
|
|
if self._scraper is None:
|
|
config = FoxcapeConfig(headless=True, humanize=False)
|
|
self._scraper = Foxcape(config=config)
|
|
self._scraper.start()
|
|
|
|
def close(self) -> None:
|
|
if self._scraper is not None:
|
|
try:
|
|
self._scraper.close()
|
|
except Exception:
|
|
pass
|
|
self._scraper = None
|
|
|
|
def __enter__(self) -> ArticleCrawler:
|
|
self.start()
|
|
return self
|
|
|
|
def __exit__(self, exc_type: Any, exc_val: Any, exc_tb: Any) -> None:
|
|
self.close()
|
|
|
|
def crawl(self, url: str) -> tuple[str, str | None, int | None]:
|
|
"""
|
|
Navega até a URL, aguarda o carregamento do DOM e retorna (html, page_title, http_status).
|
|
"""
|
|
if self._scraper is None:
|
|
self.start()
|
|
|
|
assert self._scraper is not None
|
|
result = self._scraper.get(
|
|
url,
|
|
wait_until="domcontentloaded",
|
|
timeout_ms=self.timeout_ms,
|
|
human_delay=False,
|
|
)
|
|
return result.html, result.title, result.status_code
|
|
|
|
|
|
# ==============================================================================
|
|
# Helpers de I/O e Orquestrador de Lote
|
|
# ==============================================================================
|
|
|
|
|
|
def log_info(message: str, silent: bool = False) -> None:
|
|
"""Escreve mensagem informativa no stderr."""
|
|
if not silent:
|
|
sys.stderr.write(f"[INFO] {message}\n")
|
|
sys.stderr.flush()
|
|
|
|
|
|
def load_search_json(file_path: Path) -> tuple[str | None, str, list[InputArticle]]:
|
|
"""Carrega o arquivo JSON gerado pelo extrator de notícias."""
|
|
if not file_path.exists():
|
|
raise FileNotFoundError(f"Arquivo de entrada não encontrado: {file_path}")
|
|
|
|
with file_path.open("r", encoding="utf-8") as f:
|
|
data = json.load(f)
|
|
|
|
query = data.get("query")
|
|
language = data.get("language", "en")
|
|
raw_items = data.get("items", [])
|
|
|
|
articles = []
|
|
for item in raw_items:
|
|
if isinstance(item, dict) and "url" in item and "titulo" in item:
|
|
articles.append(
|
|
InputArticle(
|
|
titulo=item["titulo"],
|
|
url=item["url"],
|
|
subtitulo=item.get("subtitulo"),
|
|
quando_publicado=item.get("quando_publicado"),
|
|
pagina=item.get("pagina", 1),
|
|
)
|
|
)
|
|
|
|
return query, language, articles
|
|
|
|
|
|
def save_extracted_json(report: ExtractionBatchReport, output_path: Path) -> None:
|
|
"""Salva o relatório consolidado em formato JSON com UTF-8."""
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
with output_path.open("w", encoding="utf-8") as f:
|
|
json.dump(report.to_dict(), f, ensure_ascii=False, indent=2)
|
|
|
|
|
|
def process_batch(
|
|
input_path: Path,
|
|
output_path: Path | None = None,
|
|
limit: int | None = None,
|
|
language_override: str | None = None,
|
|
timeout: int = 30,
|
|
silent: bool = False,
|
|
) -> ExtractionBatchReport:
|
|
"""
|
|
Executa o pipeline completo de extração em lote para o arquivo de entrada.
|
|
"""
|
|
query, search_lang, input_articles = load_search_json(input_path)
|
|
effective_lang = language_override or search_lang or "en"
|
|
|
|
if limit is not None and limit > 0:
|
|
input_articles = input_articles[:limit]
|
|
|
|
total = len(input_articles)
|
|
log_info(
|
|
f"🚀 Iniciando extração de {total} artigo(s) a partir de '{input_path}' (Idioma NLP: '{effective_lang}')...",
|
|
silent=silent,
|
|
)
|
|
|
|
# Determinar caminho de saída padrão se não especificado
|
|
if output_path is None:
|
|
output_path = input_path.parent / f"{input_path.stem}_extracted.json"
|
|
|
|
extracted_list: list[ExtractedArticle] = []
|
|
successful_count = 0
|
|
failed_count = 0
|
|
start_time = time.time()
|
|
|
|
with ArticleCrawler(timeout_sec=timeout) as crawler:
|
|
for idx, article in enumerate(input_articles, start=1):
|
|
url = article.url
|
|
log_info(f"🌐 [{idx}/{total}] Navegando com Foxcape: {url}", silent=silent)
|
|
|
|
try:
|
|
html, page_title, http_status = crawler.crawl(url)
|
|
|
|
log_info(
|
|
f"⚙️ [{idx}/{total}] Processando extratores (Trafilatura, Newspaper4k, Readability)...",
|
|
silent=silent,
|
|
)
|
|
traf_data, newspaper_data, readability_data = extract_all_engines(
|
|
html=html, url=url, language=effective_lang
|
|
)
|
|
|
|
extracted_article = ExtractedArticle(
|
|
input_meta=article,
|
|
extraction_status="success",
|
|
error_message=None,
|
|
crawled_url=url,
|
|
page_title=page_title,
|
|
http_status=http_status,
|
|
trafilatura=traf_data,
|
|
newspaper4k=newspaper_data,
|
|
readability=readability_data,
|
|
)
|
|
successful_count += 1
|
|
log_info(
|
|
f'✅ [{idx}/{total}] Sucesso (Título: "{article.titulo[:50]}...")',
|
|
silent=silent,
|
|
)
|
|
|
|
except Exception as exc:
|
|
failed_count += 1
|
|
error_msg = str(exc)
|
|
log_info(
|
|
f"⚠️ [{idx}/{total}] Falha ao processar URL '{url}': {error_msg}",
|
|
silent=silent,
|
|
)
|
|
extracted_article = ExtractedArticle(
|
|
input_meta=article,
|
|
extraction_status="failed",
|
|
error_message=error_msg,
|
|
crawled_url=url,
|
|
page_title=None,
|
|
http_status=None,
|
|
trafilatura=None,
|
|
newspaper4k=None,
|
|
readability=None,
|
|
)
|
|
|
|
extracted_list.append(extracted_article)
|
|
|
|
elapsed = time.time() - start_time
|
|
now_iso = datetime.now(timezone.utc).isoformat()
|
|
|
|
report = ExtractionBatchReport(
|
|
source_file=str(input_path),
|
|
processed_at=now_iso,
|
|
total_articles=total,
|
|
successful_articles=successful_count,
|
|
failed_articles=failed_count,
|
|
articles=extracted_list,
|
|
)
|
|
|
|
save_extracted_json(report, output_path)
|
|
log_info(f"💾 Relatório final gravado com sucesso em: '{output_path}'", silent=silent)
|
|
log_info(
|
|
f"📊 Resumo: {total} total | {successful_count} sucessos | {failed_count} falhas | Tempo: {elapsed:.2f}s",
|
|
silent=silent,
|
|
)
|
|
|
|
return report
|
|
|
|
|
|
# ==============================================================================
|
|
# Interface CLI
|
|
# ==============================================================================
|
|
|
|
|
|
def parse_arguments(args: list[str] | None = None) -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability)",
|
|
formatter_class=argparse.RawTextHelpFormatter,
|
|
)
|
|
parser.add_argument(
|
|
"-i",
|
|
"--input",
|
|
required=True,
|
|
type=str,
|
|
help="Caminho para o arquivo JSON de busca de notícias (ex: out/river_plate.json)",
|
|
)
|
|
parser.add_argument(
|
|
"-o",
|
|
"--output",
|
|
required=False,
|
|
type=str,
|
|
default=None,
|
|
help="Caminho do arquivo JSON de destino (padrão: <input_stem>_extracted.json)",
|
|
)
|
|
parser.add_argument(
|
|
"-l",
|
|
"--limit",
|
|
required=False,
|
|
type=int,
|
|
default=None,
|
|
help="Limita a quantidade máxima de artigos a serem processados",
|
|
)
|
|
parser.add_argument(
|
|
"--lang",
|
|
"--language",
|
|
dest="language",
|
|
required=False,
|
|
type=str,
|
|
default=None,
|
|
help="Sobrescreve o código de idioma para o NLP do Newspaper4k (ex: pt, es, en)",
|
|
)
|
|
parser.add_argument(
|
|
"-t",
|
|
"--timeout",
|
|
required=False,
|
|
type=int,
|
|
default=30,
|
|
help="Timeout em segundos para carregamento do DOM de cada página no Foxcape (padrão: 30)",
|
|
)
|
|
parser.add_argument(
|
|
"-s",
|
|
"--silent",
|
|
action="store_true",
|
|
help="Suprime mensagens de log e progresso no stderr",
|
|
)
|
|
return parser.parse_args(args)
|
|
|
|
|
|
def main(args: list[str] | None = None) -> int:
|
|
try:
|
|
parsed = parse_arguments(args)
|
|
input_path = Path(parsed.input)
|
|
output_path = Path(parsed.output) if parsed.output else None
|
|
|
|
if not input_path.exists():
|
|
sys.stderr.write(f"Erro: Arquivo de entrada '{input_path}' não existe.\n")
|
|
return 1
|
|
|
|
process_batch(
|
|
input_path=input_path,
|
|
output_path=output_path,
|
|
limit=parsed.limit,
|
|
language_override=parsed.language,
|
|
timeout=parsed.timeout,
|
|
silent=parsed.silent,
|
|
)
|
|
return 0
|
|
except KeyboardInterrupt:
|
|
sys.stderr.write("\nExecução cancelada pelo usuário.\n")
|
|
return 130
|
|
except Exception as exc:
|
|
sys.stderr.write(f"Erro fatal durante a execução: {exc}\n")
|
|
return 2
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|