1257 lines
46 KiB
Python
1257 lines
46 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability).
|
|
|
|
Lê listagens JSON de notícias (ex: out/river_plate.json), acessa e renderiza as páginas
|
|
em modo stealth headless utilizando Foxcape reutilizando a mesma sessão de navegador,
|
|
executa a extração em paralelo/sequência com 3 motores de conteúdo (Trafilatura,
|
|
Newspaper4k e Readability) e salva o resultado enriquecido e higienizado em JSON.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import sys
|
|
import time
|
|
import urllib.error
|
|
import urllib.request
|
|
from dataclasses import dataclass, field
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any, Literal
|
|
|
|
import trafilatura
|
|
from bs4 import BeautifulSoup
|
|
from foxcape import Foxcape, FoxcapeConfig
|
|
from newspaper import Article
|
|
from readability import Document
|
|
|
|
# ==============================================================================
|
|
# Modelos de Dados e Dataclasses
|
|
# ==============================================================================
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class MediaCandidateInfo:
|
|
"""Informações estruturais da DOM sobre mídias candidatas identificadas."""
|
|
|
|
has_candidate_media: bool
|
|
has_video: bool = False
|
|
image_count: int = 0
|
|
has_embed: bool = False
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class MediaClassification:
|
|
"""Classificação estruturada emitida pelo classificador semântico."""
|
|
|
|
content_type: Literal["text", "media"]
|
|
media_type: Literal["video", "image", "images", "embed", "mixed"] | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"content_type": self.content_type,
|
|
"media_type": self.media_type,
|
|
}
|
|
|
|
|
|
MEDIA_CLASSIFIER_SCHEMA: dict[str, Any] = {
|
|
"type": "object",
|
|
"properties": {
|
|
"content_type": {
|
|
"type": "string",
|
|
"enum": ["text", "media"],
|
|
"description": "Classification: 'text' for substantive journalistic text, 'media' for predominantly media.",
|
|
},
|
|
"media_type": {
|
|
"type": ["string", "null"],
|
|
"enum": ["video", "image", "images", "embed", "mixed", None],
|
|
"description": "Specific media category when content_type is 'media', or null when content_type is 'text'.",
|
|
},
|
|
},
|
|
"required": ["content_type", "media_type"],
|
|
"additionalProperties": False,
|
|
}
|
|
|
|
|
|
def validate_classifier_response(data: Any) -> MediaClassification | None:
|
|
"""Valida estritamente o contrato de 2 campos da resposta do classificador."""
|
|
if not isinstance(data, dict):
|
|
return None
|
|
if set(data.keys()) != {"content_type", "media_type"}:
|
|
return None
|
|
|
|
content_type = data.get("content_type")
|
|
media_type = data.get("media_type")
|
|
|
|
if content_type not in ("text", "media"):
|
|
return None
|
|
|
|
if content_type == "text":
|
|
if media_type is not None:
|
|
return None
|
|
return MediaClassification(content_type="text", media_type=None)
|
|
|
|
# content_type == "media"
|
|
if media_type not in ("video", "image", "images", "embed", "mixed"):
|
|
return None
|
|
return MediaClassification(content_type="media", media_type=media_type)
|
|
|
|
|
|
def _http_post_json(
|
|
url: str,
|
|
payload: dict[str, Any],
|
|
headers: dict[str, str],
|
|
timeout: int,
|
|
) -> tuple[int, str]:
|
|
"""Helper de baixo nível para envio de requisições POST JSON via urllib.request."""
|
|
data_bytes = json.dumps(payload).encode("utf-8")
|
|
req = urllib.request.Request(url, data=data_bytes, headers=headers, method="POST")
|
|
with urllib.request.urlopen(req, timeout=timeout) as response:
|
|
status = getattr(response, "status", response.getcode())
|
|
body = response.read().decode("utf-8")
|
|
return status, body
|
|
|
|
|
|
MEDIA_CLASSIFIER_PROMPT: str = (
|
|
"You are an editorial news classifier. Classify if this news publication is predominantly media or substantive journalistic text.\n\n"
|
|
"Publication Title: {title}\n"
|
|
"Structural Media Present: Video={has_video}, ImagesCount={image_count}, Embed={has_embed}\n"
|
|
"Text Content:\n"
|
|
"{text_content}\n\n"
|
|
"Definitions:\n"
|
|
"- \"media\": The primary informative content is in the media (video, single image, multiple images/gallery, social embed, or mixed), and the text functions essentially as a brief introduction, caption, contextualization, or description.\n"
|
|
"- \"text\": The publication contains substantive journalistic text on its own, even if accompanied by illustrative media.\n\n"
|
|
"Respond ONLY with a JSON object matching this exact schema:\n"
|
|
"{{\"content_type\": \"text\" | \"media\", \"media_type\": \"video\" | \"image\" | \"images\" | \"embed\" | \"mixed\" | null}}\n"
|
|
"Rules:\n"
|
|
"- If content_type is \"text\", media_type MUST be null.\n"
|
|
"- If content_type is \"media\", media_type MUST be one of: \"video\", \"image\", \"images\", \"embed\", \"mixed\"."
|
|
)
|
|
|
|
|
|
def _find_editorial_region(soup: BeautifulSoup) -> Any:
|
|
"""Localiza a região editorial da DOM respeitando a ordem de precedência."""
|
|
article = soup.find("article")
|
|
if article:
|
|
return article
|
|
main = soup.find("main")
|
|
if main:
|
|
return main
|
|
role_main = soup.find("div", attrs={"role": "main"})
|
|
if role_main:
|
|
return role_main
|
|
if soup.body:
|
|
return soup.body
|
|
return soup
|
|
|
|
|
|
def detect_candidate_media(soup: BeautifulSoup) -> MediaCandidateInfo:
|
|
"""
|
|
Analisa estruturalmente a DOM carregada para identificar elementos candidatos a mídia.
|
|
Executa exclusivamente via navegação DOM (Zero-Regex).
|
|
"""
|
|
region = _find_editorial_region(soup)
|
|
if not region:
|
|
return MediaCandidateInfo(has_candidate_media=False)
|
|
|
|
# Identifica vídeos: tags <video>
|
|
videos = region.find_all("video")
|
|
valid_videos = 0
|
|
for v in videos:
|
|
if v.find_parent(["header", "nav", "footer", "aside"]):
|
|
continue
|
|
valid_videos += 1
|
|
has_video = valid_videos > 0
|
|
|
|
# Contagem de imagens reais: tags <img>
|
|
images = region.find_all("img")
|
|
valid_images = 0
|
|
for img in images:
|
|
if img.find_parent(["header", "nav", "footer", "aside"]):
|
|
continue
|
|
valid_images += 1
|
|
|
|
# Identifica embeds: <iframe>, <embed>, <object>
|
|
embed_tags = region.find_all(["iframe", "embed", "object"])
|
|
valid_embeds = 0
|
|
for emb in embed_tags:
|
|
if emb.find_parent(["header", "nav", "footer", "aside"]):
|
|
continue
|
|
valid_embeds += 1
|
|
has_embed = valid_embeds > 0
|
|
|
|
has_candidate_media = has_video or (valid_images > 0) or has_embed
|
|
|
|
return MediaCandidateInfo(
|
|
has_candidate_media=has_candidate_media,
|
|
has_video=has_video,
|
|
image_count=valid_images,
|
|
has_embed=has_embed,
|
|
)
|
|
|
|
|
|
def build_compact_payload(soup: BeautifulSoup, candidate_info: MediaCandidateInfo) -> str:
|
|
"""Monta o payload compacto sem marcações HTML para envio ao classificador semântico."""
|
|
region = _find_editorial_region(soup)
|
|
|
|
title = ""
|
|
if soup.title and soup.title.string:
|
|
title = soup.title.string.strip()
|
|
elif region:
|
|
h1 = region.find("h1")
|
|
if h1:
|
|
title = h1.get_text(strip=True)
|
|
if not title and soup.find("h1"):
|
|
h1 = soup.find("h1")
|
|
if h1:
|
|
title = h1.get_text(strip=True)
|
|
|
|
paragraphs: list[str] = []
|
|
if region:
|
|
for p in region.find_all("p"):
|
|
if p.find_parent(["header", "nav", "footer", "aside"]):
|
|
continue
|
|
text = " ".join(p.get_text().split())
|
|
if text:
|
|
paragraphs.append(text)
|
|
|
|
text_content = "\n\n".join(paragraphs)
|
|
|
|
return MEDIA_CLASSIFIER_PROMPT.format(
|
|
title=title,
|
|
has_video=candidate_info.has_video,
|
|
image_count=candidate_info.image_count,
|
|
has_embed=candidate_info.has_embed,
|
|
text_content=text_content,
|
|
)
|
|
|
|
|
|
def classify_media_content(
|
|
payload: str,
|
|
metrics: dict[str, int],
|
|
silent: bool = False,
|
|
) -> tuple[MediaClassification | None, str | None]:
|
|
"""
|
|
Classifica a publicação usando a cadeia sequencial de provedores LLM:
|
|
Ollama (qwen3.5:2b) -> Groq (openai/gpt-oss-20b) -> OmniRoute (cgpt-web/gpt-5.5).
|
|
Retorna (MediaClassification, None) na primeira resposta válida ou (None, error_message) em caso de falha cumulativa.
|
|
"""
|
|
# --------------------------------------------------------------------------
|
|
# 1. Provedor Primário: Ollama
|
|
# --------------------------------------------------------------------------
|
|
ollama_endpoint = os.environ.get("OLLAMA_ENDPOINT", "http://localhost:11434").rstrip("/")
|
|
ollama_model = os.environ.get("OLLAMA_MODEL", "qwen3.5:2b")
|
|
ollama_timeout = int(os.environ.get("OLLAMA_TIMEOUT", "10"))
|
|
|
|
ollama_url = f"{ollama_endpoint}/api/chat"
|
|
ollama_payload = {
|
|
"model": ollama_model,
|
|
"messages": [{"role": "user", "content": payload}],
|
|
"stream": False,
|
|
"format": MEDIA_CLASSIFIER_SCHEMA,
|
|
"options": {
|
|
"temperature": 0.0,
|
|
},
|
|
"think": False,
|
|
}
|
|
|
|
try:
|
|
status, body = _http_post_json(
|
|
ollama_url,
|
|
ollama_payload,
|
|
{"Content-Type": "application/json"},
|
|
ollama_timeout,
|
|
)
|
|
if status == 200:
|
|
parsed = json.loads(body)
|
|
content_str = parsed.get("message", {}).get("content", "")
|
|
data = json.loads(content_str) if isinstance(content_str, str) else content_str
|
|
classification = validate_classifier_response(data)
|
|
if classification is not None:
|
|
if not silent:
|
|
m_label = f" ({classification.media_type})" if classification.media_type else ""
|
|
sys.stderr.write(
|
|
f"[MEDIA] Provedor: Ollama ({ollama_model}) | Classificação: {classification.content_type}{m_label}\n"
|
|
)
|
|
sys.stderr.flush()
|
|
return classification, None
|
|
except Exception:
|
|
pass
|
|
|
|
# --------------------------------------------------------------------------
|
|
# 2. Primeiro Fallback: Groq
|
|
# --------------------------------------------------------------------------
|
|
metrics["fallback_groq"] += 1
|
|
if not silent:
|
|
sys.stderr.write("[MEDIA] Acionando fallback 1: Groq\n")
|
|
sys.stderr.flush()
|
|
|
|
groq_endpoint = os.environ.get("GROQ_ENDPOINT", "https://api.groq.com/openai/v1/chat/completions")
|
|
groq_api_key = os.environ.get("GROQ_API_KEY", "")
|
|
groq_model = os.environ.get("GROQ_MODEL", "openai/gpt-oss-20b")
|
|
groq_timeout = int(os.environ.get("GROQ_TIMEOUT", "15"))
|
|
|
|
if groq_endpoint and groq_api_key:
|
|
groq_payload = {
|
|
"model": groq_model,
|
|
"messages": [{"role": "user", "content": payload}],
|
|
"temperature": 0.0,
|
|
"reasoning_effort": "low",
|
|
"response_format": {
|
|
"type": "json_schema",
|
|
"json_schema": {
|
|
"name": "media_classifier",
|
|
"strict": True,
|
|
"schema": MEDIA_CLASSIFIER_SCHEMA,
|
|
},
|
|
},
|
|
}
|
|
groq_headers = {
|
|
"Content-Type": "application/json",
|
|
"Authorization": f"Bearer {groq_api_key}",
|
|
}
|
|
try:
|
|
status, body = _http_post_json(groq_endpoint, groq_payload, groq_headers, groq_timeout)
|
|
if status == 200:
|
|
parsed = json.loads(body)
|
|
choices = parsed.get("choices", [])
|
|
if choices:
|
|
content_str = choices[0].get("message", {}).get("content", "")
|
|
data = json.loads(content_str) if isinstance(content_str, str) else content_str
|
|
classification = validate_classifier_response(data)
|
|
if classification is not None:
|
|
if not silent:
|
|
m_label = f" ({classification.media_type})" if classification.media_type else ""
|
|
sys.stderr.write(
|
|
f"[MEDIA] Provedor: Groq ({groq_model}) | Classificação: {classification.content_type}{m_label}\n"
|
|
)
|
|
sys.stderr.flush()
|
|
return classification, None
|
|
except Exception:
|
|
pass
|
|
|
|
# --------------------------------------------------------------------------
|
|
# 3. Segundo Fallback: OmniRoute
|
|
# --------------------------------------------------------------------------
|
|
metrics["fallback_omniroute"] += 1
|
|
if not silent:
|
|
sys.stderr.write("[MEDIA] Acionando fallback 2: OmniRoute\n")
|
|
sys.stderr.flush()
|
|
|
|
omniroute_endpoint = os.environ.get("OMNIROUTE_ENDPOINT", "")
|
|
omniroute_api_key = os.environ.get("OMNIROUTE_API_KEY", "")
|
|
omniroute_model = os.environ.get("OMNIROUTE_MODEL", "cgpt-web/gpt-5.5")
|
|
omniroute_timeout = int(os.environ.get("OMNIROUTE_TIMEOUT", "20"))
|
|
|
|
if omniroute_endpoint:
|
|
omniroute_payload = {
|
|
"model": omniroute_model,
|
|
"messages": [{"role": "user", "content": payload}],
|
|
"temperature": 0.0,
|
|
"response_format": {
|
|
"type": "json_schema",
|
|
"json_schema": {
|
|
"name": "media_classifier",
|
|
"strict": True,
|
|
"schema": MEDIA_CLASSIFIER_SCHEMA,
|
|
},
|
|
},
|
|
}
|
|
omniroute_headers = {
|
|
"Content-Type": "application/json",
|
|
}
|
|
if omniroute_api_key:
|
|
omniroute_headers["Authorization"] = f"Bearer {omniroute_api_key}"
|
|
|
|
try:
|
|
status, body = _http_post_json(
|
|
omniroute_endpoint, omniroute_payload, omniroute_headers, omniroute_timeout
|
|
)
|
|
if status == 200:
|
|
parsed = json.loads(body)
|
|
choices = parsed.get("choices", [])
|
|
if choices:
|
|
content_str = choices[0].get("message", {}).get("content", "")
|
|
data = json.loads(content_str) if isinstance(content_str, str) else content_str
|
|
classification = validate_classifier_response(data)
|
|
if classification is not None:
|
|
if not silent:
|
|
m_label = f" ({classification.media_type})" if classification.media_type else ""
|
|
sys.stderr.write(
|
|
f"[MEDIA] Provedor: OmniRoute ({omniroute_model}) | Classificação: {classification.content_type}{m_label}\n"
|
|
)
|
|
sys.stderr.flush()
|
|
return classification, None
|
|
except Exception:
|
|
pass
|
|
|
|
# --------------------------------------------------------------------------
|
|
# Falha Total
|
|
# --------------------------------------------------------------------------
|
|
error_msg = "All classification providers failed (Ollama, Groq, OmniRoute)."
|
|
return None, error_msg
|
|
|
|
|
|
def save_media_json(articles: list[dict[str, Any]], output_path: Path) -> None:
|
|
"""Salva o arquivo de mídia dedicado com envelope mínimo."""
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
payload = {"articles": articles}
|
|
with output_path.open("w", encoding="utf-8") as f:
|
|
json.dump(payload, f, ensure_ascii=False, indent=2)
|
|
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class InputArticle:
|
|
"""Metadados originais da notícia contida no JSON de entrada."""
|
|
|
|
titulo: str
|
|
url: str
|
|
subtitulo: str | None = None
|
|
quando_publicado: str | None = None
|
|
pagina: int = 1
|
|
raw_data: dict[str, Any] = field(default_factory=dict)
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
result = dict(self.raw_data)
|
|
result["titulo"] = self.titulo
|
|
result["url"] = self.url
|
|
if self.subtitulo is not None and "subtitulo" not in result:
|
|
result["subtitulo"] = self.subtitulo
|
|
if self.quando_publicado is not None and "quando_publicado" not in result:
|
|
result["quando_publicado"] = self.quando_publicado
|
|
if "pagina" not in result:
|
|
result["pagina"] = self.pagina
|
|
return result
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class TrafilaturaData:
|
|
"""Dados completos extraídos pelo motor Trafilatura."""
|
|
|
|
title: str | None = None
|
|
author: str | None = None
|
|
date: str | None = None
|
|
description: str | None = None
|
|
sitename: str | None = None
|
|
hostname: str | None = None
|
|
language: str | None = None
|
|
categories: list[str] = field(default_factory=list)
|
|
tags: list[str] = field(default_factory=list)
|
|
canonical_url: str | None = None
|
|
image: str | None = None
|
|
pagetype: str | None = None
|
|
fingerprint: str | None = None
|
|
license: str | None = None
|
|
comments: str | None = None
|
|
text: str = ""
|
|
markdown: str | None = None
|
|
raw_json: dict[str, Any] | None = None
|
|
error: str | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"title": self.title,
|
|
"author": self.author,
|
|
"date": self.date,
|
|
"description": self.description,
|
|
"sitename": self.sitename,
|
|
"hostname": self.hostname,
|
|
"language": self.language,
|
|
"categories": self.categories,
|
|
"tags": self.tags,
|
|
"canonical_url": self.canonical_url,
|
|
"image": self.image,
|
|
"pagetype": self.pagetype,
|
|
"fingerprint": self.fingerprint,
|
|
"license": self.license,
|
|
"comments": self.comments,
|
|
"text": self.text,
|
|
"markdown": self.markdown,
|
|
"raw_json": self.raw_json,
|
|
"error": self.error,
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class NewspaperData:
|
|
"""Dados completos extraídos e enriquecidos com NLP pelo motor Newspaper4k."""
|
|
|
|
title: str | None = None
|
|
authors: list[str] = field(default_factory=list)
|
|
publish_date: str | None = None
|
|
text: str = ""
|
|
summary: str | None = None
|
|
keywords: list[str] = field(default_factory=list)
|
|
keyword_scores: dict[str, float] = field(default_factory=dict)
|
|
top_image: str | None = None
|
|
images: list[str] = field(default_factory=list)
|
|
movies: list[str] = field(default_factory=list)
|
|
tags: list[str] = field(default_factory=list)
|
|
canonical_link: str | None = None
|
|
article_html: str | None = None
|
|
meta_description: str | None = None
|
|
meta_keywords: list[str] = field(default_factory=list)
|
|
meta_favicon: str | None = None
|
|
meta_site_name: str | None = None
|
|
meta_lang: str | None = None
|
|
meta_data: dict[str, Any] = field(default_factory=dict)
|
|
error: str | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"title": self.title,
|
|
"authors": self.authors,
|
|
"publish_date": self.publish_date,
|
|
"text": self.text,
|
|
"summary": self.summary,
|
|
"keywords": self.keywords,
|
|
"keyword_scores": self.keyword_scores,
|
|
"top_image": self.top_image,
|
|
"images": self.images,
|
|
"movies": self.movies,
|
|
"tags": self.tags,
|
|
"canonical_link": self.canonical_link,
|
|
"article_html": self.article_html,
|
|
"meta_description": self.meta_description,
|
|
"meta_keywords": self.meta_keywords,
|
|
"meta_favicon": self.meta_favicon,
|
|
"meta_site_name": self.meta_site_name,
|
|
"meta_lang": self.meta_lang,
|
|
"meta_data": self.meta_data,
|
|
"error": self.error,
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ReadabilityData:
|
|
"""Dados completos higienizados pelo algoritmo Readability."""
|
|
|
|
title: str | None = None
|
|
short_title: str | None = None
|
|
author: str | None = None
|
|
cleaned_html: str | None = None
|
|
cleaned_text: str | None = None
|
|
error: str | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"title": self.title,
|
|
"short_title": self.short_title,
|
|
"author": self.author,
|
|
"cleaned_html": self.cleaned_html,
|
|
"cleaned_text": self.cleaned_text,
|
|
"error": self.error,
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ExtractedArticle:
|
|
"""Resultado consolidado da extração de um artigo."""
|
|
|
|
input_meta: InputArticle
|
|
extraction_status: Literal["success", "failed"] | None = None
|
|
classification_status: Literal["failed"] | None = None
|
|
error_message: str | None = None
|
|
crawled_url: str = ""
|
|
page_title: str | None = None
|
|
http_status: int | None = None
|
|
trafilatura: TrafilaturaData | None = None
|
|
newspaper4k: NewspaperData | None = None
|
|
readability: ReadabilityData | None = None
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
res: dict[str, Any] = {
|
|
"input_meta": self.input_meta.to_dict(),
|
|
}
|
|
if self.classification_status is not None:
|
|
res["classification_status"] = self.classification_status
|
|
if self.extraction_status is not None:
|
|
res["extraction_status"] = self.extraction_status
|
|
res["error_message"] = self.error_message
|
|
res["crawled_url"] = self.crawled_url
|
|
res["page_title"] = self.page_title
|
|
res["http_status"] = self.http_status
|
|
if self.classification_status is None:
|
|
res["trafilatura"] = self.trafilatura.to_dict() if self.trafilatura else None
|
|
res["newspaper4k"] = self.newspaper4k.to_dict() if self.newspaper4k else None
|
|
res["readability"] = self.readability.to_dict() if self.readability else None
|
|
return res
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ExtractionBatchReport:
|
|
"""Relatório consolidado de saída do processamento de um lote."""
|
|
|
|
source_file: str
|
|
processed_at: str
|
|
total_articles: int
|
|
successful_articles: int
|
|
failed_articles: int
|
|
articles: list[ExtractedArticle] = field(default_factory=list)
|
|
|
|
def to_dict(self) -> dict[str, Any]:
|
|
return {
|
|
"source_file": self.source_file,
|
|
"processed_at": self.processed_at,
|
|
"total_articles": self.total_articles,
|
|
"successful_articles": self.successful_articles,
|
|
"failed_articles": self.failed_articles,
|
|
"articles": [a.to_dict() for a in self.articles],
|
|
}
|
|
|
|
|
|
# ==============================================================================
|
|
# Parsers / Extratores Especializados
|
|
# ==============================================================================
|
|
|
|
|
|
class TrafilaturaExtractor:
|
|
"""Motor de extração baseado na biblioteca Trafilatura."""
|
|
|
|
@staticmethod
|
|
def extract(html: str, url: str | None = None) -> TrafilaturaData:
|
|
try:
|
|
# Extração bare document completa
|
|
doc = trafilatura.bare_extraction(
|
|
html,
|
|
url=url,
|
|
include_comments=True,
|
|
include_tables=True,
|
|
include_images=True,
|
|
include_links=True,
|
|
include_formatting=True,
|
|
with_metadata=True,
|
|
)
|
|
|
|
# Extração em markdown
|
|
markdown_text = trafilatura.extract(
|
|
html,
|
|
output_format="markdown",
|
|
include_comments=True,
|
|
include_tables=True,
|
|
include_images=True,
|
|
include_links=True,
|
|
include_formatting=True,
|
|
url=url,
|
|
)
|
|
|
|
# Extração em JSON nativo
|
|
json_output_str = trafilatura.extract(
|
|
html,
|
|
output_format="json",
|
|
include_comments=True,
|
|
include_tables=True,
|
|
include_images=True,
|
|
include_links=True,
|
|
url=url,
|
|
)
|
|
raw_json = json.loads(json_output_str) if json_output_str else None
|
|
|
|
if doc:
|
|
if isinstance(doc, dict):
|
|
categories = list(doc.get("categories", [])) if doc.get("categories") else []
|
|
tags = list(doc.get("tags", [])) if doc.get("tags") else []
|
|
return TrafilaturaData(
|
|
title=doc.get("title"),
|
|
author=doc.get("author"),
|
|
date=doc.get("date"),
|
|
description=doc.get("description"),
|
|
sitename=doc.get("sitename"),
|
|
hostname=doc.get("hostname"),
|
|
language=doc.get("language"),
|
|
categories=categories,
|
|
tags=tags,
|
|
canonical_url=doc.get("url") or url,
|
|
image=doc.get("image"),
|
|
pagetype=doc.get("pagetype"),
|
|
fingerprint=doc.get("fingerprint"),
|
|
license=doc.get("license"),
|
|
comments=doc.get("comments"),
|
|
text=(doc.get("text") or "").strip(),
|
|
markdown=(markdown_text or "").strip() if markdown_text else None,
|
|
raw_json=raw_json,
|
|
error=None,
|
|
)
|
|
else:
|
|
categories = list(doc.categories) if doc.categories else []
|
|
tags = list(doc.tags) if doc.tags else []
|
|
return TrafilaturaData(
|
|
title=doc.title,
|
|
author=doc.author,
|
|
date=doc.date,
|
|
description=doc.description,
|
|
sitename=doc.sitename,
|
|
hostname=doc.hostname,
|
|
language=doc.language,
|
|
categories=categories,
|
|
tags=tags,
|
|
canonical_url=doc.url or url,
|
|
image=doc.image,
|
|
pagetype=doc.pagetype,
|
|
fingerprint=doc.fingerprint,
|
|
license=doc.license,
|
|
comments=doc.comments,
|
|
text=(doc.text or "").strip(),
|
|
markdown=(markdown_text or "").strip() if markdown_text else None,
|
|
raw_json=raw_json,
|
|
error=None,
|
|
)
|
|
else:
|
|
raw_text = trafilatura.extract(html, output_format="txt", url=url) or ""
|
|
return TrafilaturaData(
|
|
title=raw_json.get("title") if raw_json else None,
|
|
text=raw_text.strip(),
|
|
markdown=markdown_text.strip() if markdown_text else None,
|
|
raw_json=raw_json,
|
|
error=None,
|
|
)
|
|
except Exception as e:
|
|
return TrafilaturaData(error=str(e))
|
|
|
|
|
|
class NewspaperExtractor:
|
|
"""Motor de extração baseado no Newspaper4k com NLP."""
|
|
|
|
@staticmethod
|
|
def extract(html: str, url: str = "", language: str = "en") -> NewspaperData:
|
|
try:
|
|
lang_code = language.split("-")[0].lower() if language else "en"
|
|
|
|
article = Article(url=url, language=lang_code)
|
|
article.download(input_html=html)
|
|
article.parse()
|
|
|
|
# Executar NLP para summary e keywords com fallback gracioso
|
|
try:
|
|
article.nlp()
|
|
summary = article.summary
|
|
keywords = list(article.keywords) if article.keywords else []
|
|
keyword_scores = getattr(article, "keyword_scores", {}) or {}
|
|
except Exception:
|
|
summary = None
|
|
keywords = []
|
|
keyword_scores = {}
|
|
|
|
publish_date_str = (
|
|
article.publish_date.isoformat()
|
|
if article.publish_date and hasattr(article.publish_date, "isoformat")
|
|
else str(article.publish_date)
|
|
if article.publish_date
|
|
else None
|
|
)
|
|
|
|
# Metadados e tags adicionais
|
|
meta_keywords = list(article.meta_keywords) if article.meta_keywords else []
|
|
tags = list(article.tags) if getattr(article, "tags", None) else []
|
|
movies = list(article.movies) if getattr(article, "movies", None) else []
|
|
images = list(article.images) if article.images else []
|
|
|
|
return NewspaperData(
|
|
title=article.title or None,
|
|
authors=list(article.authors) if article.authors else [],
|
|
publish_date=publish_date_str,
|
|
text=article.text or "",
|
|
summary=summary,
|
|
keywords=keywords,
|
|
keyword_scores=dict(keyword_scores),
|
|
top_image=article.top_image or getattr(article, "meta_img", None) or None,
|
|
images=images,
|
|
movies=movies,
|
|
tags=tags,
|
|
canonical_link=getattr(article, "canonical_link", None) or None,
|
|
article_html=getattr(article, "article_html", None) or None,
|
|
meta_description=getattr(article, "meta_description", None) or None,
|
|
meta_keywords=meta_keywords,
|
|
meta_favicon=getattr(article, "meta_favicon", None) or None,
|
|
meta_site_name=getattr(article, "meta_site_name", None) or None,
|
|
meta_lang=getattr(article, "meta_lang", None) or None,
|
|
meta_data=dict(article.meta_data) if article.meta_data else {},
|
|
error=None,
|
|
)
|
|
except Exception as e:
|
|
return NewspaperData(error=str(e))
|
|
|
|
|
|
class ReadabilityExtractor:
|
|
"""Motor de extração baseado no algoritmo Readability (readability-lxml)."""
|
|
|
|
@staticmethod
|
|
def extract(html: str) -> ReadabilityData:
|
|
try:
|
|
doc = Document(html)
|
|
title = doc.title()
|
|
short_title = doc.short_title()
|
|
cleaned_html = doc.summary()
|
|
author = None
|
|
try:
|
|
author = doc.author()
|
|
except Exception:
|
|
pass
|
|
|
|
# Extração de texto limpo a partir do HTML higienizado
|
|
soup = BeautifulSoup(cleaned_html, "html.parser")
|
|
cleaned_text = soup.get_text(separator="\n\n", strip=True)
|
|
|
|
return ReadabilityData(
|
|
title=title or None,
|
|
short_title=short_title or None,
|
|
author=author or None,
|
|
cleaned_html=cleaned_html or None,
|
|
cleaned_text=cleaned_text or None,
|
|
error=None,
|
|
)
|
|
except Exception as e:
|
|
return ReadabilityData(error=str(e))
|
|
|
|
|
|
def extract_all_engines(
|
|
html: str, url: str = "", language: str = "en"
|
|
) -> tuple[TrafilaturaData, NewspaperData, ReadabilityData]:
|
|
"""Executa a extração simultânea pelos três motores de conteúdo com isolamento defensivo."""
|
|
try:
|
|
traf_data = TrafilaturaExtractor.extract(html, url=url)
|
|
except Exception as exc:
|
|
traf_data = TrafilaturaData(
|
|
title=None,
|
|
author=None,
|
|
date=None,
|
|
description=None,
|
|
categories=[],
|
|
tags=[],
|
|
canonical_url=None,
|
|
text="",
|
|
raw_json=None,
|
|
error=str(exc),
|
|
)
|
|
|
|
try:
|
|
newspaper_data = NewspaperExtractor.extract(html, url=url, language=language)
|
|
except Exception as exc:
|
|
newspaper_data = NewspaperData(
|
|
title=None,
|
|
authors=[],
|
|
publish_date=None,
|
|
text="",
|
|
summary=None,
|
|
keywords=[],
|
|
top_image=None,
|
|
images=[],
|
|
meta_data={},
|
|
error=str(exc),
|
|
)
|
|
|
|
try:
|
|
readability_data = ReadabilityExtractor.extract(html)
|
|
except Exception as exc:
|
|
readability_data = ReadabilityData(
|
|
title=None,
|
|
short_title=None,
|
|
cleaned_html=None,
|
|
cleaned_text=None,
|
|
error=str(exc),
|
|
)
|
|
|
|
return traf_data, newspaper_data, readability_data
|
|
|
|
|
|
# ==============================================================================
|
|
# Motor de Navegação Stealth Headless com Foxcape
|
|
# ==============================================================================
|
|
|
|
|
|
class ArticleCrawler:
|
|
"""Gerenciador de ciclo de vida e requisições via Foxcape Headless."""
|
|
|
|
def __init__(self, timeout_sec: int = 30) -> None:
|
|
self.timeout_sec = timeout_sec
|
|
self.timeout_ms = timeout_sec * 1000
|
|
self._scraper: Foxcape | None = None
|
|
|
|
def start(self) -> None:
|
|
if self._scraper is None:
|
|
config = FoxcapeConfig(headless=True, humanize=False)
|
|
self._scraper = Foxcape(config=config)
|
|
self._scraper.start()
|
|
|
|
def close(self) -> None:
|
|
if self._scraper is not None:
|
|
try:
|
|
self._scraper.close()
|
|
except Exception:
|
|
pass
|
|
self._scraper = None
|
|
|
|
def __enter__(self) -> ArticleCrawler:
|
|
self.start()
|
|
return self
|
|
|
|
def __exit__(self, exc_type: Any, exc_val: Any, exc_tb: Any) -> None:
|
|
self.close()
|
|
|
|
def crawl(self, url: str) -> tuple[str, str | None, int | None]:
|
|
"""
|
|
Navega até a URL, aguarda o carregamento do DOM e retorna (html, page_title, http_status).
|
|
"""
|
|
if self._scraper is None:
|
|
self.start()
|
|
|
|
assert self._scraper is not None
|
|
result = self._scraper.get(
|
|
url,
|
|
wait_until="domcontentloaded",
|
|
timeout_ms=self.timeout_ms,
|
|
human_delay=False,
|
|
)
|
|
return result.html, result.title, result.status_code
|
|
|
|
|
|
# ==============================================================================
|
|
# Helpers de I/O e Orquestrador de Lote
|
|
# ==============================================================================
|
|
|
|
|
|
def log_info(message: str, silent: bool = False) -> None:
|
|
"""Escreve mensagem informativa no stderr."""
|
|
if not silent:
|
|
sys.stderr.write(f"[INFO] {message}\n")
|
|
sys.stderr.flush()
|
|
|
|
|
|
def load_search_json(file_path: Path) -> tuple[str | None, str, list[InputArticle]]:
|
|
"""Carrega o arquivo JSON gerado pelo extrator de notícias preservando 100% dos metadados."""
|
|
if not file_path.exists():
|
|
raise FileNotFoundError(f"Arquivo de entrada não encontrado: {file_path}")
|
|
|
|
with file_path.open("r", encoding="utf-8") as f:
|
|
data = json.load(f)
|
|
|
|
query = data.get("query")
|
|
language = data.get("language", "en")
|
|
raw_items = data.get("items", [])
|
|
|
|
articles = []
|
|
for item in raw_items:
|
|
if isinstance(item, dict) and "url" in item and "titulo" in item:
|
|
articles.append(
|
|
InputArticle(
|
|
titulo=item["titulo"],
|
|
url=item["url"],
|
|
raw_data=dict(item),
|
|
)
|
|
)
|
|
|
|
return query, language, articles
|
|
|
|
|
|
def save_extracted_json(report: ExtractionBatchReport, output_path: Path) -> None:
|
|
"""Salva o relatório consolidado em formato JSON com UTF-8."""
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
with output_path.open("w", encoding="utf-8") as f:
|
|
json.dump(report.to_dict(), f, ensure_ascii=False, indent=2)
|
|
|
|
|
|
def process_batch(
|
|
input_path: Path,
|
|
output_path: Path | None = None,
|
|
limit: int | None = None,
|
|
language_override: str | None = None,
|
|
timeout: int = 30,
|
|
silent: bool = False,
|
|
) -> ExtractionBatchReport:
|
|
"""
|
|
Executa o pipeline completo de extração em lote para o arquivo de entrada.
|
|
"""
|
|
query, search_lang, input_articles = load_search_json(input_path)
|
|
effective_lang = language_override or search_lang or "en"
|
|
|
|
if limit is not None and limit > 0:
|
|
input_articles = input_articles[:limit]
|
|
|
|
total = len(input_articles)
|
|
log_info(
|
|
f"🚀 Iniciando extração de {total} artigo(s) a partir de '{input_path}' (Idioma NLP: '{effective_lang}')...",
|
|
silent=silent,
|
|
)
|
|
|
|
# Determinar caminho de saída textual e caminho do arquivo de mídia
|
|
if output_path is None:
|
|
text_output_path = input_path.parent / f"{input_path.stem}_extracted.json"
|
|
media_output_path = input_path.parent / f"{input_path.stem}_media.json"
|
|
else:
|
|
text_output_path = output_path
|
|
media_output_path = output_path.with_name(f"{output_path.stem}_media{output_path.suffix}")
|
|
|
|
# Inicializar contadores operacionais das 11 métricas
|
|
metrics: dict[str, int] = {
|
|
"total_evaluated": 0,
|
|
"text": 0,
|
|
"media": 0,
|
|
"media/video": 0,
|
|
"media/image": 0,
|
|
"media/images": 0,
|
|
"media/embed": 0,
|
|
"media/mixed": 0,
|
|
"fallback_groq": 0,
|
|
"fallback_omniroute": 0,
|
|
"classification_failed": 0,
|
|
}
|
|
|
|
extracted_list: list[ExtractedArticle] = []
|
|
media_articles: list[dict[str, Any]] = []
|
|
successful_count = 0
|
|
failed_count = 0
|
|
start_time = time.time()
|
|
|
|
with ArticleCrawler(timeout_sec=timeout) as crawler:
|
|
for idx, article in enumerate(input_articles, start=1):
|
|
url = article.url
|
|
log_info(f"🌐 [{idx}/{total}] Navegando com Foxcape: {url}", silent=silent)
|
|
|
|
try:
|
|
html, page_title, http_status = crawler.crawl(url)
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
|
|
metrics["total_evaluated"] += 1
|
|
candidate_info = detect_candidate_media(soup)
|
|
|
|
# Gate estrutural prévio: se houver mídia candidata, envia ao classificador
|
|
if candidate_info.has_candidate_media:
|
|
payload = build_compact_payload(soup, candidate_info)
|
|
classification, error_msg = classify_media_content(
|
|
payload, metrics, silent=silent
|
|
)
|
|
|
|
if classification is not None and classification.content_type == "media":
|
|
metrics["media"] += 1
|
|
if classification.media_type:
|
|
m_key = f"media/{classification.media_type}"
|
|
if m_key in metrics:
|
|
metrics[m_key] += 1
|
|
|
|
media_article = {
|
|
"input_meta": article.to_dict(),
|
|
"crawled_url": url,
|
|
"page_title": page_title,
|
|
"http_status": http_status,
|
|
"content_type": "media",
|
|
"media_type": classification.media_type,
|
|
}
|
|
media_articles.append(media_article)
|
|
log_info(
|
|
f'📹 [{idx}/{total}] Publicação predominantemente de mídia ({classification.media_type}) desviada para *_media.json',
|
|
silent=silent,
|
|
)
|
|
continue
|
|
elif classification is not None and classification.content_type == "text":
|
|
metrics["text"] += 1
|
|
elif classification is None:
|
|
# Falha total na cadeia de provedores
|
|
metrics["classification_failed"] += 1
|
|
failed_count += 1
|
|
log_info(
|
|
f"⚠️ [{idx}/{total}] Falha de classificação para URL '{url}': {error_msg}",
|
|
silent=silent,
|
|
)
|
|
if not silent:
|
|
sys.stderr.write(f"[MEDIA] Falha total da cadeia de classificação: {error_msg}\n")
|
|
sys.stderr.flush()
|
|
failed_article = ExtractedArticle(
|
|
input_meta=article,
|
|
classification_status="failed",
|
|
error_message=error_msg,
|
|
crawled_url=url,
|
|
page_title=page_title,
|
|
http_status=http_status,
|
|
trafilatura=None,
|
|
newspaper4k=None,
|
|
readability=None,
|
|
)
|
|
extracted_list.append(failed_article)
|
|
continue
|
|
else:
|
|
# Bypass direto do gate estrutural (sem mídia candidata)
|
|
metrics["text"] += 1
|
|
|
|
log_info(
|
|
f"⚙️ [{idx}/{total}] Processando extratores (Trafilatura, Newspaper4k, Readability)...",
|
|
silent=silent,
|
|
)
|
|
traf_data, newspaper_data, readability_data = extract_all_engines(
|
|
html=html, url=url, language=effective_lang
|
|
)
|
|
|
|
extracted_article = ExtractedArticle(
|
|
input_meta=article,
|
|
extraction_status="success",
|
|
error_message=None,
|
|
crawled_url=url,
|
|
page_title=page_title,
|
|
http_status=http_status,
|
|
trafilatura=traf_data,
|
|
newspaper4k=newspaper_data,
|
|
readability=readability_data,
|
|
)
|
|
successful_count += 1
|
|
log_info(
|
|
f'✅ [{idx}/{total}] Sucesso (Título: "{article.titulo[:50]}...")',
|
|
silent=silent,
|
|
)
|
|
|
|
except Exception as exc:
|
|
failed_count += 1
|
|
error_msg = str(exc)
|
|
log_info(
|
|
f"⚠️ [{idx}/{total}] Falha ao processar URL '{url}': {error_msg}",
|
|
silent=silent,
|
|
)
|
|
extracted_article = ExtractedArticle(
|
|
input_meta=article,
|
|
extraction_status="failed",
|
|
error_message=error_msg,
|
|
crawled_url=url,
|
|
page_title=None,
|
|
http_status=None,
|
|
trafilatura=None,
|
|
newspaper4k=None,
|
|
readability=None,
|
|
)
|
|
|
|
extracted_list.append(extracted_article)
|
|
|
|
# Emissão incondicional do arquivo de mídia *_media.json
|
|
save_media_json(media_articles, media_output_path)
|
|
|
|
elapsed = time.time() - start_time
|
|
now_iso = datetime.now(timezone.utc).isoformat()
|
|
|
|
# Contadores do relatório textual refletem estritamente os itens presentes em extracted_list
|
|
report = ExtractionBatchReport(
|
|
source_file=str(input_path),
|
|
processed_at=now_iso,
|
|
total_articles=len(extracted_list),
|
|
successful_articles=successful_count,
|
|
failed_articles=failed_count,
|
|
articles=extracted_list,
|
|
)
|
|
|
|
save_extracted_json(report, text_output_path)
|
|
log_info(f"💾 Relatório final gravado com sucesso em: '{text_output_path}'", silent=silent)
|
|
log_info(f"💾 Arquivo de mídia gravado com sucesso em: '{media_output_path}' ({len(media_articles)} artigo(s))", silent=silent)
|
|
log_info(
|
|
f"📊 Resumo: {len(extracted_list)} no JSON textual | {successful_count} sucessos | {failed_count} falhas | Tempo: {elapsed:.2f}s",
|
|
silent=silent,
|
|
)
|
|
|
|
if not silent:
|
|
media_subtypes = ", ".join(
|
|
f"{k.split('/')[1]}={v}"
|
|
for k, v in metrics.items()
|
|
if k.startswith("media/") and v > 0
|
|
)
|
|
subtypes_str = f" ({media_subtypes})" if media_subtypes else ""
|
|
sys.stderr.write(
|
|
f"[MEDIA] Métricas de Roteamento:\n"
|
|
f" - Total avaliados: {metrics['total_evaluated']}\n"
|
|
f" - Texto: {metrics['text']}\n"
|
|
f" - Mídia: {metrics['media']}{subtypes_str}\n"
|
|
f" - Fallbacks: Groq={metrics['fallback_groq']}, OmniRoute={metrics['fallback_omniroute']}\n"
|
|
f" - Falhas de classificação: {metrics['classification_failed']}\n"
|
|
)
|
|
sys.stderr.flush()
|
|
|
|
return report
|
|
|
|
|
|
# ==============================================================================
|
|
# Interface CLI
|
|
# ==============================================================================
|
|
|
|
|
|
def parse_arguments(args: list[str] | None = None) -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability)",
|
|
formatter_class=argparse.RawTextHelpFormatter,
|
|
)
|
|
parser.add_argument(
|
|
"-i",
|
|
"--input",
|
|
required=True,
|
|
type=str,
|
|
help="Caminho para o arquivo JSON de busca de notícias (ex: out/river_plate.json)",
|
|
)
|
|
parser.add_argument(
|
|
"-o",
|
|
"--output",
|
|
required=False,
|
|
type=str,
|
|
default=None,
|
|
help="Caminho do arquivo JSON de destino (padrão: <input_stem>_extracted.json)",
|
|
)
|
|
parser.add_argument(
|
|
"-l",
|
|
"--limit",
|
|
required=False,
|
|
type=int,
|
|
default=None,
|
|
help="Limita a quantidade máxima de artigos a serem processados",
|
|
)
|
|
parser.add_argument(
|
|
"--lang",
|
|
"--language",
|
|
dest="language",
|
|
required=False,
|
|
type=str,
|
|
default=None,
|
|
help="Sobrescreve o código de idioma para o NLP do Newspaper4k (ex: pt, es, en)",
|
|
)
|
|
parser.add_argument(
|
|
"-t",
|
|
"--timeout",
|
|
required=False,
|
|
type=int,
|
|
default=30,
|
|
help="Timeout em segundos para carregamento do DOM de cada página no Foxcape (padrão: 30)",
|
|
)
|
|
parser.add_argument(
|
|
"-s",
|
|
"--silent",
|
|
action="store_true",
|
|
help="Suprime mensagens de log e progresso no stderr",
|
|
)
|
|
return parser.parse_args(args)
|
|
|
|
|
|
def main(args: list[str] | None = None) -> int:
|
|
try:
|
|
parsed = parse_arguments(args)
|
|
input_path = Path(parsed.input)
|
|
output_path = Path(parsed.output) if parsed.output else None
|
|
|
|
if not input_path.exists():
|
|
sys.stderr.write(f"Erro: Arquivo de entrada '{input_path}' não existe.\n")
|
|
return 1
|
|
|
|
process_batch(
|
|
input_path=input_path,
|
|
output_path=output_path,
|
|
limit=parsed.limit,
|
|
language_override=parsed.language,
|
|
timeout=parsed.timeout,
|
|
silent=parsed.silent,
|
|
)
|
|
return 0
|
|
except KeyboardInterrupt:
|
|
sys.stderr.write("\nExecução cancelada pelo usuário.\n")
|
|
return 130
|
|
except Exception as exc:
|
|
sys.stderr.write(f"Erro fatal durante a execução: {exc}\n")
|
|
return 2
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|