#!/usr/bin/env python3 """ Extrator e Parser de Artigos Multimotor (Foxcape + Trafilatura + Newspaper4k + Readability). Lê listagens JSON de notícias (ex: out/river_plate.json), acessa e renderiza as páginas em modo stealth headless utilizando Foxcape reutilizando a mesma sessão de navegador, executa a extração em paralelo/sequência com 3 motores de conteúdo (Trafilatura, Newspaper4k e Readability) e salva o resultado enriquecido e higienizado em JSON. """ from __future__ import annotations import argparse import json import os import sys import time import urllib.error import urllib.request from dataclasses import dataclass, field from datetime import datetime, timezone from pathlib import Path from typing import Any, Literal import trafilatura from bs4 import BeautifulSoup from foxcape import Foxcape, FoxcapeConfig from newspaper import Article from readability import Document # ============================================================================== # Modelos de Dados e Dataclasses # ============================================================================== @dataclass(frozen=True) class MediaCandidateInfo: """Informações estruturais da DOM sobre mídias candidatas identificadas.""" has_candidate_media: bool has_video: bool = False image_count: int = 0 has_embed: bool = False @dataclass(frozen=True) class MediaClassification: """Classificação estruturada emitida pelo classificador semântico.""" content_type: Literal["text", "media"] media_type: Literal["video", "image", "images", "embed", "mixed"] | None = None def to_dict(self) -> dict[str, Any]: return { "content_type": self.content_type, "media_type": self.media_type, } MEDIA_CLASSIFIER_SCHEMA: dict[str, Any] = { "type": "object", "properties": { "content_type": { "type": "string", "enum": ["text", "media"], "description": "Classification: 'text' for substantive journalistic text, 'media' for predominantly media.", }, "media_type": { "type": ["string", "null"], "enum": ["video", "image", "images", "embed", "mixed", None], "description": "Specific media category when content_type is 'media', or null when content_type is 'text'.", }, }, "required": ["content_type", "media_type"], "additionalProperties": False, } def validate_classifier_response(data: Any) -> MediaClassification | None: """Valida estritamente o contrato de 2 campos da resposta do classificador.""" if not isinstance(data, dict): return None if set(data.keys()) != {"content_type", "media_type"}: return None content_type = data.get("content_type") media_type = data.get("media_type") if content_type not in ("text", "media"): return None if content_type == "text": if media_type is not None: return None return MediaClassification(content_type="text", media_type=None) # content_type == "media" if media_type not in ("video", "image", "images", "embed", "mixed"): return None return MediaClassification(content_type="media", media_type=media_type) def _http_post_json( url: str, payload: dict[str, Any], headers: dict[str, str], timeout: int, ) -> tuple[int, str]: """Helper de baixo nível para envio de requisições POST JSON via urllib.request.""" data_bytes = json.dumps(payload).encode("utf-8") req = urllib.request.Request(url, data=data_bytes, headers=headers, method="POST") with urllib.request.urlopen(req, timeout=timeout) as response: status = getattr(response, "status", response.getcode()) body = response.read().decode("utf-8") return status, body MEDIA_CLASSIFIER_PROMPT: str = ( "You are an editorial news classifier. Classify if this news publication is predominantly media or substantive journalistic text.\n\n" "Publication Title: {title}\n" "Structural Media Present: Video={has_video}, ImagesCount={image_count}, Embed={has_embed}\n" "Text Content:\n" "{text_content}\n\n" "Definitions:\n" "- \"media\": The primary informative content is in the media (video, single image, multiple images/gallery, social embed, or mixed), and the text functions essentially as a brief introduction, caption, contextualization, or description.\n" "- \"text\": The publication contains substantive journalistic text on its own, even if accompanied by illustrative media.\n\n" "Respond ONLY with a JSON object matching this exact schema:\n" "{{\"content_type\": \"text\" | \"media\", \"media_type\": \"video\" | \"image\" | \"images\" | \"embed\" | \"mixed\" | null}}\n" "Rules:\n" "- If content_type is \"text\", media_type MUST be null.\n" "- If content_type is \"media\", media_type MUST be one of: \"video\", \"image\", \"images\", \"embed\", \"mixed\"." ) def _find_editorial_region(soup: BeautifulSoup) -> Any: """Localiza a região editorial da DOM respeitando a ordem de precedência.""" article = soup.find("article") if article: return article main = soup.find("main") if main: return main role_main = soup.find("div", attrs={"role": "main"}) if role_main: return role_main if soup.body: return soup.body return soup def detect_candidate_media(soup: BeautifulSoup) -> MediaCandidateInfo: """ Analisa estruturalmente a DOM carregada para identificar elementos candidatos a mídia. Executa exclusivamente via navegação DOM (Zero-Regex). """ region = _find_editorial_region(soup) if not region: return MediaCandidateInfo(has_candidate_media=False) # Identifica vídeos: tags