feat(extractor): implement multi-engine article content extractor
- Added scripts/extract_article_contents.py for batch scraping with stealth Foxcape and triple extraction (Trafilatura, Newspaper4k, Readability) - Created unit, integration, and E2E test suite in tests/test_extract_article_contents.py (90/90 passing) - Updated specs/003-article-content-extractor and README.md with usage documentation and CLI contracts - Passed ruff linting/formatting and mypy type checking cleanly
This commit is contained in:
@@ -0,0 +1,770 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability).
|
||||
|
||||
Lê listagens JSON de notícias (ex: out/river_plate.json), acessa e renderiza as páginas
|
||||
em modo stealth headless utilizando Foxcape reutilizando a mesma sessão de navegador,
|
||||
executa a extração em paralelo/sequência com 3 motores de conteúdo (Trafilatura,
|
||||
Newspaper4k e Readability) e salva o resultado enriquecido e higienizado em JSON.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Literal
|
||||
|
||||
import trafilatura
|
||||
from bs4 import BeautifulSoup
|
||||
from foxcape import Foxcape, FoxcapeConfig
|
||||
from newspaper import Article
|
||||
from readability import Document
|
||||
|
||||
# ==============================================================================
|
||||
# Modelos de Dados e Dataclasses
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class InputArticle:
|
||||
"""Metadados originais da notícia contida no JSON de entrada."""
|
||||
|
||||
titulo: str
|
||||
url: str
|
||||
subtitulo: str | None = None
|
||||
quando_publicado: str | None = None
|
||||
pagina: int = 1
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"titulo": self.titulo,
|
||||
"subtitulo": self.subtitulo,
|
||||
"quando_publicado": self.quando_publicado,
|
||||
"url": self.url,
|
||||
"pagina": self.pagina,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TrafilaturaData:
|
||||
"""Dados completos extraídos pelo motor Trafilatura."""
|
||||
|
||||
title: str | None = None
|
||||
author: str | None = None
|
||||
date: str | None = None
|
||||
description: str | None = None
|
||||
sitename: str | None = None
|
||||
hostname: str | None = None
|
||||
language: str | None = None
|
||||
categories: list[str] = field(default_factory=list)
|
||||
tags: list[str] = field(default_factory=list)
|
||||
canonical_url: str | None = None
|
||||
image: str | None = None
|
||||
pagetype: str | None = None
|
||||
fingerprint: str | None = None
|
||||
license: str | None = None
|
||||
comments: str | None = None
|
||||
text: str = ""
|
||||
markdown: str | None = None
|
||||
raw_json: dict[str, Any] | None = None
|
||||
error: str | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"title": self.title,
|
||||
"author": self.author,
|
||||
"date": self.date,
|
||||
"description": self.description,
|
||||
"sitename": self.sitename,
|
||||
"hostname": self.hostname,
|
||||
"language": self.language,
|
||||
"categories": self.categories,
|
||||
"tags": self.tags,
|
||||
"canonical_url": self.canonical_url,
|
||||
"image": self.image,
|
||||
"pagetype": self.pagetype,
|
||||
"fingerprint": self.fingerprint,
|
||||
"license": self.license,
|
||||
"comments": self.comments,
|
||||
"text": self.text,
|
||||
"markdown": self.markdown,
|
||||
"raw_json": self.raw_json,
|
||||
"error": self.error,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NewspaperData:
|
||||
"""Dados completos extraídos e enriquecidos com NLP pelo motor Newspaper4k."""
|
||||
|
||||
title: str | None = None
|
||||
authors: list[str] = field(default_factory=list)
|
||||
publish_date: str | None = None
|
||||
text: str = ""
|
||||
summary: str | None = None
|
||||
keywords: list[str] = field(default_factory=list)
|
||||
keyword_scores: dict[str, float] = field(default_factory=dict)
|
||||
top_image: str | None = None
|
||||
images: list[str] = field(default_factory=list)
|
||||
movies: list[str] = field(default_factory=list)
|
||||
tags: list[str] = field(default_factory=list)
|
||||
canonical_link: str | None = None
|
||||
article_html: str | None = None
|
||||
meta_description: str | None = None
|
||||
meta_keywords: list[str] = field(default_factory=list)
|
||||
meta_favicon: str | None = None
|
||||
meta_site_name: str | None = None
|
||||
meta_lang: str | None = None
|
||||
meta_data: dict[str, Any] = field(default_factory=dict)
|
||||
error: str | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"title": self.title,
|
||||
"authors": self.authors,
|
||||
"publish_date": self.publish_date,
|
||||
"text": self.text,
|
||||
"summary": self.summary,
|
||||
"keywords": self.keywords,
|
||||
"keyword_scores": self.keyword_scores,
|
||||
"top_image": self.top_image,
|
||||
"images": self.images,
|
||||
"movies": self.movies,
|
||||
"tags": self.tags,
|
||||
"canonical_link": self.canonical_link,
|
||||
"article_html": self.article_html,
|
||||
"meta_description": self.meta_description,
|
||||
"meta_keywords": self.meta_keywords,
|
||||
"meta_favicon": self.meta_favicon,
|
||||
"meta_site_name": self.meta_site_name,
|
||||
"meta_lang": self.meta_lang,
|
||||
"meta_data": self.meta_data,
|
||||
"error": self.error,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ReadabilityData:
|
||||
"""Dados completos higienizados pelo algoritmo Readability."""
|
||||
|
||||
title: str | None = None
|
||||
short_title: str | None = None
|
||||
author: str | None = None
|
||||
cleaned_html: str | None = None
|
||||
cleaned_text: str | None = None
|
||||
error: str | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"title": self.title,
|
||||
"short_title": self.short_title,
|
||||
"author": self.author,
|
||||
"cleaned_html": self.cleaned_html,
|
||||
"cleaned_text": self.cleaned_text,
|
||||
"error": self.error,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ExtractedArticle:
|
||||
"""Resultado consolidado da extração de um artigo."""
|
||||
|
||||
input_meta: InputArticle
|
||||
extraction_status: Literal["success", "failed"]
|
||||
error_message: str | None
|
||||
crawled_url: str
|
||||
page_title: str | None
|
||||
http_status: int | None
|
||||
trafilatura: TrafilaturaData | None = None
|
||||
newspaper4k: NewspaperData | None = None
|
||||
readability: ReadabilityData | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"input_meta": self.input_meta.to_dict(),
|
||||
"extraction_status": self.extraction_status,
|
||||
"error_message": self.error_message,
|
||||
"crawled_url": self.crawled_url,
|
||||
"page_title": self.page_title,
|
||||
"http_status": self.http_status,
|
||||
"trafilatura": self.trafilatura.to_dict() if self.trafilatura else None,
|
||||
"newspaper4k": self.newspaper4k.to_dict() if self.newspaper4k else None,
|
||||
"readability": self.readability.to_dict() if self.readability else None,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ExtractionBatchReport:
|
||||
"""Relatório consolidado de saída do processamento de um lote."""
|
||||
|
||||
source_file: str
|
||||
processed_at: str
|
||||
total_articles: int
|
||||
successful_articles: int
|
||||
failed_articles: int
|
||||
articles: list[ExtractedArticle] = field(default_factory=list)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"source_file": self.source_file,
|
||||
"processed_at": self.processed_at,
|
||||
"total_articles": self.total_articles,
|
||||
"successful_articles": self.successful_articles,
|
||||
"failed_articles": self.failed_articles,
|
||||
"articles": [a.to_dict() for a in self.articles],
|
||||
}
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Parsers / Extratores Especializados
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
class TrafilaturaExtractor:
|
||||
"""Motor de extração baseado na biblioteca Trafilatura."""
|
||||
|
||||
@staticmethod
|
||||
def extract(html: str, url: str | None = None) -> TrafilaturaData:
|
||||
try:
|
||||
# Extração bare document completa
|
||||
doc = trafilatura.bare_extraction(
|
||||
html,
|
||||
url=url,
|
||||
include_comments=True,
|
||||
include_tables=True,
|
||||
include_images=True,
|
||||
include_links=True,
|
||||
include_formatting=True,
|
||||
with_metadata=True,
|
||||
)
|
||||
|
||||
# Extração em markdown
|
||||
markdown_text = trafilatura.extract(
|
||||
html,
|
||||
output_format="markdown",
|
||||
include_comments=True,
|
||||
include_tables=True,
|
||||
include_images=True,
|
||||
include_links=True,
|
||||
include_formatting=True,
|
||||
url=url,
|
||||
)
|
||||
|
||||
# Extração em JSON nativo
|
||||
json_output_str = trafilatura.extract(
|
||||
html,
|
||||
output_format="json",
|
||||
include_comments=True,
|
||||
include_tables=True,
|
||||
include_images=True,
|
||||
include_links=True,
|
||||
url=url,
|
||||
)
|
||||
raw_json = json.loads(json_output_str) if json_output_str else None
|
||||
|
||||
if doc:
|
||||
if isinstance(doc, dict):
|
||||
categories = list(doc.get("categories", [])) if doc.get("categories") else []
|
||||
tags = list(doc.get("tags", [])) if doc.get("tags") else []
|
||||
return TrafilaturaData(
|
||||
title=doc.get("title"),
|
||||
author=doc.get("author"),
|
||||
date=doc.get("date"),
|
||||
description=doc.get("description"),
|
||||
sitename=doc.get("sitename"),
|
||||
hostname=doc.get("hostname"),
|
||||
language=doc.get("language"),
|
||||
categories=categories,
|
||||
tags=tags,
|
||||
canonical_url=doc.get("url") or url,
|
||||
image=doc.get("image"),
|
||||
pagetype=doc.get("pagetype"),
|
||||
fingerprint=doc.get("fingerprint"),
|
||||
license=doc.get("license"),
|
||||
comments=doc.get("comments"),
|
||||
text=(doc.get("text") or "").strip(),
|
||||
markdown=(markdown_text or "").strip() if markdown_text else None,
|
||||
raw_json=raw_json,
|
||||
error=None,
|
||||
)
|
||||
else:
|
||||
categories = list(doc.categories) if doc.categories else []
|
||||
tags = list(doc.tags) if doc.tags else []
|
||||
return TrafilaturaData(
|
||||
title=doc.title,
|
||||
author=doc.author,
|
||||
date=doc.date,
|
||||
description=doc.description,
|
||||
sitename=doc.sitename,
|
||||
hostname=doc.hostname,
|
||||
language=doc.language,
|
||||
categories=categories,
|
||||
tags=tags,
|
||||
canonical_url=doc.url or url,
|
||||
image=doc.image,
|
||||
pagetype=doc.pagetype,
|
||||
fingerprint=doc.fingerprint,
|
||||
license=doc.license,
|
||||
comments=doc.comments,
|
||||
text=(doc.text or "").strip(),
|
||||
markdown=(markdown_text or "").strip() if markdown_text else None,
|
||||
raw_json=raw_json,
|
||||
error=None,
|
||||
)
|
||||
else:
|
||||
raw_text = trafilatura.extract(html, output_format="txt", url=url) or ""
|
||||
return TrafilaturaData(
|
||||
title=raw_json.get("title") if raw_json else None,
|
||||
text=raw_text.strip(),
|
||||
markdown=markdown_text.strip() if markdown_text else None,
|
||||
raw_json=raw_json,
|
||||
error=None,
|
||||
)
|
||||
except Exception as e:
|
||||
return TrafilaturaData(error=str(e))
|
||||
|
||||
|
||||
class NewspaperExtractor:
|
||||
"""Motor de extração baseado no Newspaper4k com NLP."""
|
||||
|
||||
@staticmethod
|
||||
def extract(html: str, url: str = "", language: str = "en") -> NewspaperData:
|
||||
try:
|
||||
lang_code = language.split("-")[0].lower() if language else "en"
|
||||
|
||||
article = Article(url=url, language=lang_code)
|
||||
article.download(input_html=html)
|
||||
article.parse()
|
||||
|
||||
# Executar NLP para summary e keywords com fallback gracioso
|
||||
try:
|
||||
article.nlp()
|
||||
summary = article.summary
|
||||
keywords = list(article.keywords) if article.keywords else []
|
||||
keyword_scores = getattr(article, "keyword_scores", {}) or {}
|
||||
except Exception:
|
||||
summary = None
|
||||
keywords = []
|
||||
keyword_scores = {}
|
||||
|
||||
publish_date_str = (
|
||||
article.publish_date.isoformat()
|
||||
if article.publish_date and hasattr(article.publish_date, "isoformat")
|
||||
else str(article.publish_date)
|
||||
if article.publish_date
|
||||
else None
|
||||
)
|
||||
|
||||
# Metadados e tags adicionais
|
||||
meta_keywords = list(article.meta_keywords) if article.meta_keywords else []
|
||||
tags = list(article.tags) if getattr(article, "tags", None) else []
|
||||
movies = list(article.movies) if getattr(article, "movies", None) else []
|
||||
images = list(article.images) if article.images else []
|
||||
|
||||
return NewspaperData(
|
||||
title=article.title or None,
|
||||
authors=list(article.authors) if article.authors else [],
|
||||
publish_date=publish_date_str,
|
||||
text=article.text or "",
|
||||
summary=summary,
|
||||
keywords=keywords,
|
||||
keyword_scores=dict(keyword_scores),
|
||||
top_image=article.top_image or getattr(article, "meta_img", None) or None,
|
||||
images=images,
|
||||
movies=movies,
|
||||
tags=tags,
|
||||
canonical_link=getattr(article, "canonical_link", None) or None,
|
||||
article_html=getattr(article, "article_html", None) or None,
|
||||
meta_description=getattr(article, "meta_description", None) or None,
|
||||
meta_keywords=meta_keywords,
|
||||
meta_favicon=getattr(article, "meta_favicon", None) or None,
|
||||
meta_site_name=getattr(article, "meta_site_name", None) or None,
|
||||
meta_lang=getattr(article, "meta_lang", None) or None,
|
||||
meta_data=dict(article.meta_data) if article.meta_data else {},
|
||||
error=None,
|
||||
)
|
||||
except Exception as e:
|
||||
return NewspaperData(error=str(e))
|
||||
|
||||
|
||||
class ReadabilityExtractor:
|
||||
"""Motor de extração baseado no algoritmo Readability (readability-lxml)."""
|
||||
|
||||
@staticmethod
|
||||
def extract(html: str) -> ReadabilityData:
|
||||
try:
|
||||
doc = Document(html)
|
||||
title = doc.title()
|
||||
short_title = doc.short_title()
|
||||
cleaned_html = doc.summary()
|
||||
author = None
|
||||
try:
|
||||
author = doc.author()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Extração de texto limpo a partir do HTML higienizado
|
||||
soup = BeautifulSoup(cleaned_html, "html.parser")
|
||||
cleaned_text = soup.get_text(separator="\n\n", strip=True)
|
||||
|
||||
return ReadabilityData(
|
||||
title=title or None,
|
||||
short_title=short_title or None,
|
||||
author=author or None,
|
||||
cleaned_html=cleaned_html or None,
|
||||
cleaned_text=cleaned_text or None,
|
||||
error=None,
|
||||
)
|
||||
except Exception as e:
|
||||
return ReadabilityData(error=str(e))
|
||||
|
||||
|
||||
def extract_all_engines(
|
||||
html: str, url: str = "", language: str = "en"
|
||||
) -> tuple[TrafilaturaData, NewspaperData, ReadabilityData]:
|
||||
"""Executa a extração simultânea pelos três motores de conteúdo com isolamento defensivo."""
|
||||
try:
|
||||
traf_data = TrafilaturaExtractor.extract(html, url=url)
|
||||
except Exception as exc:
|
||||
traf_data = TrafilaturaData(
|
||||
title=None,
|
||||
author=None,
|
||||
date=None,
|
||||
description=None,
|
||||
categories=[],
|
||||
tags=[],
|
||||
canonical_url=None,
|
||||
text="",
|
||||
raw_json=None,
|
||||
error=str(exc),
|
||||
)
|
||||
|
||||
try:
|
||||
newspaper_data = NewspaperExtractor.extract(html, url=url, language=language)
|
||||
except Exception as exc:
|
||||
newspaper_data = NewspaperData(
|
||||
title=None,
|
||||
authors=[],
|
||||
publish_date=None,
|
||||
text="",
|
||||
summary=None,
|
||||
keywords=[],
|
||||
top_image=None,
|
||||
images=[],
|
||||
meta_data={},
|
||||
error=str(exc),
|
||||
)
|
||||
|
||||
try:
|
||||
readability_data = ReadabilityExtractor.extract(html)
|
||||
except Exception as exc:
|
||||
readability_data = ReadabilityData(
|
||||
title=None,
|
||||
short_title=None,
|
||||
cleaned_html=None,
|
||||
cleaned_text=None,
|
||||
error=str(exc),
|
||||
)
|
||||
|
||||
return traf_data, newspaper_data, readability_data
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Motor de Navegação Stealth Headless com Foxcape
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
class ArticleCrawler:
|
||||
"""Gerenciador de ciclo de vida e requisições via Foxcape Headless."""
|
||||
|
||||
def __init__(self, timeout_sec: int = 30) -> None:
|
||||
self.timeout_sec = timeout_sec
|
||||
self.timeout_ms = timeout_sec * 1000
|
||||
self._scraper: Foxcape | None = None
|
||||
|
||||
def start(self) -> None:
|
||||
if self._scraper is None:
|
||||
config = FoxcapeConfig(headless=True, humanize=False)
|
||||
self._scraper = Foxcape(config=config)
|
||||
self._scraper.start()
|
||||
|
||||
def close(self) -> None:
|
||||
if self._scraper is not None:
|
||||
try:
|
||||
self._scraper.close()
|
||||
except Exception:
|
||||
pass
|
||||
self._scraper = None
|
||||
|
||||
def __enter__(self) -> ArticleCrawler:
|
||||
self.start()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type: Any, exc_val: Any, exc_tb: Any) -> None:
|
||||
self.close()
|
||||
|
||||
def crawl(self, url: str) -> tuple[str, str | None, int | None]:
|
||||
"""
|
||||
Navega até a URL, aguarda o carregamento do DOM e retorna (html, page_title, http_status).
|
||||
"""
|
||||
if self._scraper is None:
|
||||
self.start()
|
||||
|
||||
assert self._scraper is not None
|
||||
result = self._scraper.get(
|
||||
url,
|
||||
wait_until="domcontentloaded",
|
||||
timeout_ms=self.timeout_ms,
|
||||
human_delay=False,
|
||||
)
|
||||
return result.html, result.title, result.status_code
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Helpers de I/O e Orquestrador de Lote
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def log_info(message: str, silent: bool = False) -> None:
|
||||
"""Escreve mensagem informativa no stderr."""
|
||||
if not silent:
|
||||
sys.stderr.write(f"[INFO] {message}\n")
|
||||
sys.stderr.flush()
|
||||
|
||||
|
||||
def load_search_json(file_path: Path) -> tuple[str | None, str, list[InputArticle]]:
|
||||
"""Carrega o arquivo JSON gerado pelo extrator de notícias."""
|
||||
if not file_path.exists():
|
||||
raise FileNotFoundError(f"Arquivo de entrada não encontrado: {file_path}")
|
||||
|
||||
with file_path.open("r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
|
||||
query = data.get("query")
|
||||
language = data.get("language", "en")
|
||||
raw_items = data.get("items", [])
|
||||
|
||||
articles = []
|
||||
for item in raw_items:
|
||||
if isinstance(item, dict) and "url" in item and "titulo" in item:
|
||||
articles.append(
|
||||
InputArticle(
|
||||
titulo=item["titulo"],
|
||||
url=item["url"],
|
||||
subtitulo=item.get("subtitulo"),
|
||||
quando_publicado=item.get("quando_publicado"),
|
||||
pagina=item.get("pagina", 1),
|
||||
)
|
||||
)
|
||||
|
||||
return query, language, articles
|
||||
|
||||
|
||||
def save_extracted_json(report: ExtractionBatchReport, output_path: Path) -> None:
|
||||
"""Salva o relatório consolidado em formato JSON com UTF-8."""
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with output_path.open("w", encoding="utf-8") as f:
|
||||
json.dump(report.to_dict(), f, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
def process_batch(
|
||||
input_path: Path,
|
||||
output_path: Path | None = None,
|
||||
limit: int | None = None,
|
||||
language_override: str | None = None,
|
||||
timeout: int = 30,
|
||||
silent: bool = False,
|
||||
) -> ExtractionBatchReport:
|
||||
"""
|
||||
Executa o pipeline completo de extração em lote para o arquivo de entrada.
|
||||
"""
|
||||
query, search_lang, input_articles = load_search_json(input_path)
|
||||
effective_lang = language_override or search_lang or "en"
|
||||
|
||||
if limit is not None and limit > 0:
|
||||
input_articles = input_articles[:limit]
|
||||
|
||||
total = len(input_articles)
|
||||
log_info(
|
||||
f"🚀 Iniciando extração de {total} artigo(s) a partir de '{input_path}' (Idioma NLP: '{effective_lang}')...",
|
||||
silent=silent,
|
||||
)
|
||||
|
||||
# Determinar caminho de saída padrão se não especificado
|
||||
if output_path is None:
|
||||
output_path = input_path.parent / f"{input_path.stem}_extracted.json"
|
||||
|
||||
extracted_list: list[ExtractedArticle] = []
|
||||
successful_count = 0
|
||||
failed_count = 0
|
||||
start_time = time.time()
|
||||
|
||||
with ArticleCrawler(timeout_sec=timeout) as crawler:
|
||||
for idx, article in enumerate(input_articles, start=1):
|
||||
url = article.url
|
||||
log_info(f"🌐 [{idx}/{total}] Navegando com Foxcape: {url}", silent=silent)
|
||||
|
||||
try:
|
||||
html, page_title, http_status = crawler.crawl(url)
|
||||
|
||||
log_info(
|
||||
f"⚙️ [{idx}/{total}] Processando extratores (Trafilatura, Newspaper4k, Readability)...",
|
||||
silent=silent,
|
||||
)
|
||||
traf_data, newspaper_data, readability_data = extract_all_engines(
|
||||
html=html, url=url, language=effective_lang
|
||||
)
|
||||
|
||||
extracted_article = ExtractedArticle(
|
||||
input_meta=article,
|
||||
extraction_status="success",
|
||||
error_message=None,
|
||||
crawled_url=url,
|
||||
page_title=page_title,
|
||||
http_status=http_status,
|
||||
trafilatura=traf_data,
|
||||
newspaper4k=newspaper_data,
|
||||
readability=readability_data,
|
||||
)
|
||||
successful_count += 1
|
||||
log_info(
|
||||
f'✅ [{idx}/{total}] Sucesso (Título: "{article.titulo[:50]}...")',
|
||||
silent=silent,
|
||||
)
|
||||
|
||||
except Exception as exc:
|
||||
failed_count += 1
|
||||
error_msg = str(exc)
|
||||
log_info(
|
||||
f"⚠️ [{idx}/{total}] Falha ao processar URL '{url}': {error_msg}",
|
||||
silent=silent,
|
||||
)
|
||||
extracted_article = ExtractedArticle(
|
||||
input_meta=article,
|
||||
extraction_status="failed",
|
||||
error_message=error_msg,
|
||||
crawled_url=url,
|
||||
page_title=None,
|
||||
http_status=None,
|
||||
trafilatura=None,
|
||||
newspaper4k=None,
|
||||
readability=None,
|
||||
)
|
||||
|
||||
extracted_list.append(extracted_article)
|
||||
|
||||
elapsed = time.time() - start_time
|
||||
now_iso = datetime.now(timezone.utc).isoformat()
|
||||
|
||||
report = ExtractionBatchReport(
|
||||
source_file=str(input_path),
|
||||
processed_at=now_iso,
|
||||
total_articles=total,
|
||||
successful_articles=successful_count,
|
||||
failed_articles=failed_count,
|
||||
articles=extracted_list,
|
||||
)
|
||||
|
||||
save_extracted_json(report, output_path)
|
||||
log_info(f"💾 Relatório final gravado com sucesso em: '{output_path}'", silent=silent)
|
||||
log_info(
|
||||
f"📊 Resumo: {total} total | {successful_count} sucessos | {failed_count} falhas | Tempo: {elapsed:.2f}s",
|
||||
silent=silent,
|
||||
)
|
||||
|
||||
return report
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# Interface CLI
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def parse_arguments(args: list[str] | None = None) -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability)",
|
||||
formatter_class=argparse.RawTextHelpFormatter,
|
||||
)
|
||||
parser.add_argument(
|
||||
"-i",
|
||||
"--input",
|
||||
required=True,
|
||||
type=str,
|
||||
help="Caminho para o arquivo JSON de busca de notícias (ex: out/river_plate.json)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output",
|
||||
required=False,
|
||||
type=str,
|
||||
default=None,
|
||||
help="Caminho do arquivo JSON de destino (padrão: <input_stem>_extracted.json)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-l",
|
||||
"--limit",
|
||||
required=False,
|
||||
type=int,
|
||||
default=None,
|
||||
help="Limita a quantidade máxima de artigos a serem processados",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--lang",
|
||||
"--language",
|
||||
dest="language",
|
||||
required=False,
|
||||
type=str,
|
||||
default=None,
|
||||
help="Sobrescreve o código de idioma para o NLP do Newspaper4k (ex: pt, es, en)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-t",
|
||||
"--timeout",
|
||||
required=False,
|
||||
type=int,
|
||||
default=30,
|
||||
help="Timeout em segundos para carregamento do DOM de cada página no Foxcape (padrão: 30)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-s",
|
||||
"--silent",
|
||||
action="store_true",
|
||||
help="Suprime mensagens de log e progresso no stderr",
|
||||
)
|
||||
return parser.parse_args(args)
|
||||
|
||||
|
||||
def main(args: list[str] | None = None) -> int:
|
||||
try:
|
||||
parsed = parse_arguments(args)
|
||||
input_path = Path(parsed.input)
|
||||
output_path = Path(parsed.output) if parsed.output else None
|
||||
|
||||
if not input_path.exists():
|
||||
sys.stderr.write(f"Erro: Arquivo de entrada '{input_path}' não existe.\n")
|
||||
return 1
|
||||
|
||||
process_batch(
|
||||
input_path=input_path,
|
||||
output_path=output_path,
|
||||
limit=parsed.limit,
|
||||
language_override=parsed.language,
|
||||
timeout=parsed.timeout,
|
||||
silent=parsed.silent,
|
||||
)
|
||||
return 0
|
||||
except KeyboardInterrupt:
|
||||
sys.stderr.write("\nExecução cancelada pelo usuário.\n")
|
||||
return 130
|
||||
except Exception as exc:
|
||||
sys.stderr.write(f"Erro fatal durante a execução: {exc}\n")
|
||||
return 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user