feat(media-routing): implement 007 media article routing, runtime architecture diagram and update graphify knowledge graph
This commit is contained in:
@@ -12,8 +12,11 @@ from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
@@ -30,6 +33,377 @@ from readability import Document
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MediaCandidateInfo:
|
||||
"""Informações estruturais da DOM sobre mídias candidatas identificadas."""
|
||||
|
||||
has_candidate_media: bool
|
||||
has_video: bool = False
|
||||
image_count: int = 0
|
||||
has_embed: bool = False
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MediaClassification:
|
||||
"""Classificação estruturada emitida pelo classificador semântico."""
|
||||
|
||||
content_type: Literal["text", "media"]
|
||||
media_type: Literal["video", "image", "images", "embed", "mixed"] | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"content_type": self.content_type,
|
||||
"media_type": self.media_type,
|
||||
}
|
||||
|
||||
|
||||
MEDIA_CLASSIFIER_SCHEMA: dict[str, Any] = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"content_type": {
|
||||
"type": "string",
|
||||
"enum": ["text", "media"],
|
||||
"description": "Classification: 'text' for substantive journalistic text, 'media' for predominantly media.",
|
||||
},
|
||||
"media_type": {
|
||||
"type": ["string", "null"],
|
||||
"enum": ["video", "image", "images", "embed", "mixed", None],
|
||||
"description": "Specific media category when content_type is 'media', or null when content_type is 'text'.",
|
||||
},
|
||||
},
|
||||
"required": ["content_type", "media_type"],
|
||||
"additionalProperties": False,
|
||||
}
|
||||
|
||||
|
||||
def validate_classifier_response(data: Any) -> MediaClassification | None:
|
||||
"""Valida estritamente o contrato de 2 campos da resposta do classificador."""
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
if set(data.keys()) != {"content_type", "media_type"}:
|
||||
return None
|
||||
|
||||
content_type = data.get("content_type")
|
||||
media_type = data.get("media_type")
|
||||
|
||||
if content_type not in ("text", "media"):
|
||||
return None
|
||||
|
||||
if content_type == "text":
|
||||
if media_type is not None:
|
||||
return None
|
||||
return MediaClassification(content_type="text", media_type=None)
|
||||
|
||||
# content_type == "media"
|
||||
if media_type not in ("video", "image", "images", "embed", "mixed"):
|
||||
return None
|
||||
return MediaClassification(content_type="media", media_type=media_type)
|
||||
|
||||
|
||||
def _http_post_json(
|
||||
url: str,
|
||||
payload: dict[str, Any],
|
||||
headers: dict[str, str],
|
||||
timeout: int,
|
||||
) -> tuple[int, str]:
|
||||
"""Helper de baixo nível para envio de requisições POST JSON via urllib.request."""
|
||||
data_bytes = json.dumps(payload).encode("utf-8")
|
||||
req = urllib.request.Request(url, data=data_bytes, headers=headers, method="POST")
|
||||
with urllib.request.urlopen(req, timeout=timeout) as response:
|
||||
status = getattr(response, "status", response.getcode())
|
||||
body = response.read().decode("utf-8")
|
||||
return status, body
|
||||
|
||||
|
||||
MEDIA_CLASSIFIER_PROMPT: str = (
|
||||
"You are an editorial news classifier. Classify if this news publication is predominantly media or substantive journalistic text.\n\n"
|
||||
"Publication Title: {title}\n"
|
||||
"Structural Media Present: Video={has_video}, ImagesCount={image_count}, Embed={has_embed}\n"
|
||||
"Text Content:\n"
|
||||
"{text_content}\n\n"
|
||||
"Definitions:\n"
|
||||
"- \"media\": The primary informative content is in the media (video, single image, multiple images/gallery, social embed, or mixed), and the text functions essentially as a brief introduction, caption, contextualization, or description.\n"
|
||||
"- \"text\": The publication contains substantive journalistic text on its own, even if accompanied by illustrative media.\n\n"
|
||||
"Respond ONLY with a JSON object matching this exact schema:\n"
|
||||
"{{\"content_type\": \"text\" | \"media\", \"media_type\": \"video\" | \"image\" | \"images\" | \"embed\" | \"mixed\" | null}}\n"
|
||||
"Rules:\n"
|
||||
"- If content_type is \"text\", media_type MUST be null.\n"
|
||||
"- If content_type is \"media\", media_type MUST be one of: \"video\", \"image\", \"images\", \"embed\", \"mixed\"."
|
||||
)
|
||||
|
||||
|
||||
def _find_editorial_region(soup: BeautifulSoup) -> Any:
|
||||
"""Localiza a região editorial da DOM respeitando a ordem de precedência."""
|
||||
article = soup.find("article")
|
||||
if article:
|
||||
return article
|
||||
main = soup.find("main")
|
||||
if main:
|
||||
return main
|
||||
role_main = soup.find("div", attrs={"role": "main"})
|
||||
if role_main:
|
||||
return role_main
|
||||
if soup.body:
|
||||
return soup.body
|
||||
return soup
|
||||
|
||||
|
||||
def detect_candidate_media(soup: BeautifulSoup) -> MediaCandidateInfo:
|
||||
"""
|
||||
Analisa estruturalmente a DOM carregada para identificar elementos candidatos a mídia.
|
||||
Executa exclusivamente via navegação DOM (Zero-Regex).
|
||||
"""
|
||||
region = _find_editorial_region(soup)
|
||||
if not region:
|
||||
return MediaCandidateInfo(has_candidate_media=False)
|
||||
|
||||
# Identifica vídeos: tags <video>
|
||||
videos = region.find_all("video")
|
||||
valid_videos = 0
|
||||
for v in videos:
|
||||
if v.find_parent(["header", "nav", "footer", "aside"]):
|
||||
continue
|
||||
valid_videos += 1
|
||||
has_video = valid_videos > 0
|
||||
|
||||
# Contagem de imagens reais: tags <img>
|
||||
images = region.find_all("img")
|
||||
valid_images = 0
|
||||
for img in images:
|
||||
if img.find_parent(["header", "nav", "footer", "aside"]):
|
||||
continue
|
||||
valid_images += 1
|
||||
|
||||
# Identifica embeds: <iframe>, <embed>, <object>
|
||||
embed_tags = region.find_all(["iframe", "embed", "object"])
|
||||
valid_embeds = 0
|
||||
for emb in embed_tags:
|
||||
if emb.find_parent(["header", "nav", "footer", "aside"]):
|
||||
continue
|
||||
valid_embeds += 1
|
||||
has_embed = valid_embeds > 0
|
||||
|
||||
has_candidate_media = has_video or (valid_images > 0) or has_embed
|
||||
|
||||
return MediaCandidateInfo(
|
||||
has_candidate_media=has_candidate_media,
|
||||
has_video=has_video,
|
||||
image_count=valid_images,
|
||||
has_embed=has_embed,
|
||||
)
|
||||
|
||||
|
||||
def build_compact_payload(soup: BeautifulSoup, candidate_info: MediaCandidateInfo) -> str:
|
||||
"""Monta o payload compacto sem marcações HTML para envio ao classificador semântico."""
|
||||
region = _find_editorial_region(soup)
|
||||
|
||||
title = ""
|
||||
if soup.title and soup.title.string:
|
||||
title = soup.title.string.strip()
|
||||
elif region:
|
||||
h1 = region.find("h1")
|
||||
if h1:
|
||||
title = h1.get_text(strip=True)
|
||||
if not title and soup.find("h1"):
|
||||
h1 = soup.find("h1")
|
||||
if h1:
|
||||
title = h1.get_text(strip=True)
|
||||
|
||||
paragraphs: list[str] = []
|
||||
if region:
|
||||
for p in region.find_all("p"):
|
||||
if p.find_parent(["header", "nav", "footer", "aside"]):
|
||||
continue
|
||||
text = " ".join(p.get_text().split())
|
||||
if text:
|
||||
paragraphs.append(text)
|
||||
|
||||
text_content = "\n\n".join(paragraphs)
|
||||
|
||||
return MEDIA_CLASSIFIER_PROMPT.format(
|
||||
title=title,
|
||||
has_video=candidate_info.has_video,
|
||||
image_count=candidate_info.image_count,
|
||||
has_embed=candidate_info.has_embed,
|
||||
text_content=text_content,
|
||||
)
|
||||
|
||||
|
||||
def classify_media_content(
|
||||
payload: str,
|
||||
metrics: dict[str, int],
|
||||
silent: bool = False,
|
||||
) -> tuple[MediaClassification | None, str | None]:
|
||||
"""
|
||||
Classifica a publicação usando a cadeia sequencial de provedores LLM:
|
||||
Ollama (qwen3.5:2b) -> Groq (openai/gpt-oss-20b) -> OmniRoute (cgpt-web/gpt-5.5).
|
||||
Retorna (MediaClassification, None) na primeira resposta válida ou (None, error_message) em caso de falha cumulativa.
|
||||
"""
|
||||
# --------------------------------------------------------------------------
|
||||
# 1. Provedor Primário: Ollama
|
||||
# --------------------------------------------------------------------------
|
||||
ollama_endpoint = os.environ.get("OLLAMA_ENDPOINT", "http://localhost:11434").rstrip("/")
|
||||
ollama_model = os.environ.get("OLLAMA_MODEL", "qwen3.5:2b")
|
||||
ollama_timeout = int(os.environ.get("OLLAMA_TIMEOUT", "10"))
|
||||
|
||||
ollama_url = f"{ollama_endpoint}/api/chat"
|
||||
ollama_payload = {
|
||||
"model": ollama_model,
|
||||
"messages": [{"role": "user", "content": payload}],
|
||||
"stream": False,
|
||||
"format": MEDIA_CLASSIFIER_SCHEMA,
|
||||
"options": {
|
||||
"temperature": 0.0,
|
||||
},
|
||||
"think": False,
|
||||
}
|
||||
|
||||
try:
|
||||
status, body = _http_post_json(
|
||||
ollama_url,
|
||||
ollama_payload,
|
||||
{"Content-Type": "application/json"},
|
||||
ollama_timeout,
|
||||
)
|
||||
if status == 200:
|
||||
parsed = json.loads(body)
|
||||
content_str = parsed.get("message", {}).get("content", "")
|
||||
data = json.loads(content_str) if isinstance(content_str, str) else content_str
|
||||
classification = validate_classifier_response(data)
|
||||
if classification is not None:
|
||||
if not silent:
|
||||
m_label = f" ({classification.media_type})" if classification.media_type else ""
|
||||
sys.stderr.write(
|
||||
f"[MEDIA] Provedor: Ollama ({ollama_model}) | Classificação: {classification.content_type}{m_label}\n"
|
||||
)
|
||||
sys.stderr.flush()
|
||||
return classification, None
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# 2. Primeiro Fallback: Groq
|
||||
# --------------------------------------------------------------------------
|
||||
metrics["fallback_groq"] += 1
|
||||
if not silent:
|
||||
sys.stderr.write("[MEDIA] Acionando fallback 1: Groq\n")
|
||||
sys.stderr.flush()
|
||||
|
||||
groq_endpoint = os.environ.get("GROQ_ENDPOINT", "https://api.groq.com/openai/v1/chat/completions")
|
||||
groq_api_key = os.environ.get("GROQ_API_KEY", "")
|
||||
groq_model = os.environ.get("GROQ_MODEL", "openai/gpt-oss-20b")
|
||||
groq_timeout = int(os.environ.get("GROQ_TIMEOUT", "15"))
|
||||
|
||||
if groq_endpoint and groq_api_key:
|
||||
groq_payload = {
|
||||
"model": groq_model,
|
||||
"messages": [{"role": "user", "content": payload}],
|
||||
"temperature": 0.0,
|
||||
"reasoning_effort": "low",
|
||||
"response_format": {
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "media_classifier",
|
||||
"strict": True,
|
||||
"schema": MEDIA_CLASSIFIER_SCHEMA,
|
||||
},
|
||||
},
|
||||
}
|
||||
groq_headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {groq_api_key}",
|
||||
}
|
||||
try:
|
||||
status, body = _http_post_json(groq_endpoint, groq_payload, groq_headers, groq_timeout)
|
||||
if status == 200:
|
||||
parsed = json.loads(body)
|
||||
choices = parsed.get("choices", [])
|
||||
if choices:
|
||||
content_str = choices[0].get("message", {}).get("content", "")
|
||||
data = json.loads(content_str) if isinstance(content_str, str) else content_str
|
||||
classification = validate_classifier_response(data)
|
||||
if classification is not None:
|
||||
if not silent:
|
||||
m_label = f" ({classification.media_type})" if classification.media_type else ""
|
||||
sys.stderr.write(
|
||||
f"[MEDIA] Provedor: Groq ({groq_model}) | Classificação: {classification.content_type}{m_label}\n"
|
||||
)
|
||||
sys.stderr.flush()
|
||||
return classification, None
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# 3. Segundo Fallback: OmniRoute
|
||||
# --------------------------------------------------------------------------
|
||||
metrics["fallback_omniroute"] += 1
|
||||
if not silent:
|
||||
sys.stderr.write("[MEDIA] Acionando fallback 2: OmniRoute\n")
|
||||
sys.stderr.flush()
|
||||
|
||||
omniroute_endpoint = os.environ.get("OMNIROUTE_ENDPOINT", "")
|
||||
omniroute_api_key = os.environ.get("OMNIROUTE_API_KEY", "")
|
||||
omniroute_model = os.environ.get("OMNIROUTE_MODEL", "cgpt-web/gpt-5.5")
|
||||
omniroute_timeout = int(os.environ.get("OMNIROUTE_TIMEOUT", "20"))
|
||||
|
||||
if omniroute_endpoint:
|
||||
omniroute_payload = {
|
||||
"model": omniroute_model,
|
||||
"messages": [{"role": "user", "content": payload}],
|
||||
"temperature": 0.0,
|
||||
"response_format": {
|
||||
"type": "json_schema",
|
||||
"json_schema": {
|
||||
"name": "media_classifier",
|
||||
"strict": True,
|
||||
"schema": MEDIA_CLASSIFIER_SCHEMA,
|
||||
},
|
||||
},
|
||||
}
|
||||
omniroute_headers = {
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
if omniroute_api_key:
|
||||
omniroute_headers["Authorization"] = f"Bearer {omniroute_api_key}"
|
||||
|
||||
try:
|
||||
status, body = _http_post_json(
|
||||
omniroute_endpoint, omniroute_payload, omniroute_headers, omniroute_timeout
|
||||
)
|
||||
if status == 200:
|
||||
parsed = json.loads(body)
|
||||
choices = parsed.get("choices", [])
|
||||
if choices:
|
||||
content_str = choices[0].get("message", {}).get("content", "")
|
||||
data = json.loads(content_str) if isinstance(content_str, str) else content_str
|
||||
classification = validate_classifier_response(data)
|
||||
if classification is not None:
|
||||
if not silent:
|
||||
m_label = f" ({classification.media_type})" if classification.media_type else ""
|
||||
sys.stderr.write(
|
||||
f"[MEDIA] Provedor: OmniRoute ({omniroute_model}) | Classificação: {classification.content_type}{m_label}\n"
|
||||
)
|
||||
sys.stderr.flush()
|
||||
return classification, None
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# --------------------------------------------------------------------------
|
||||
# Falha Total
|
||||
# --------------------------------------------------------------------------
|
||||
error_msg = "All classification providers failed (Ollama, Groq, OmniRoute)."
|
||||
return None, error_msg
|
||||
|
||||
|
||||
def save_media_json(articles: list[dict[str, Any]], output_path: Path) -> None:
|
||||
"""Salva o arquivo de mídia dedicado com envelope mínimo."""
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
payload = {"articles": articles}
|
||||
with output_path.open("w", encoding="utf-8") as f:
|
||||
json.dump(payload, f, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class InputArticle:
|
||||
"""Metadados originais da notícia contida no JSON de entrada."""
|
||||
@@ -39,15 +413,19 @@ class InputArticle:
|
||||
subtitulo: str | None = None
|
||||
quando_publicado: str | None = None
|
||||
pagina: int = 1
|
||||
raw_data: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"titulo": self.titulo,
|
||||
"subtitulo": self.subtitulo,
|
||||
"quando_publicado": self.quando_publicado,
|
||||
"url": self.url,
|
||||
"pagina": self.pagina,
|
||||
}
|
||||
result = dict(self.raw_data)
|
||||
result["titulo"] = self.titulo
|
||||
result["url"] = self.url
|
||||
if self.subtitulo is not None and "subtitulo" not in result:
|
||||
result["subtitulo"] = self.subtitulo
|
||||
if self.quando_publicado is not None and "quando_publicado" not in result:
|
||||
result["quando_publicado"] = self.quando_publicado
|
||||
if "pagina" not in result:
|
||||
result["pagina"] = self.pagina
|
||||
return result
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -175,27 +553,33 @@ class ExtractedArticle:
|
||||
"""Resultado consolidado da extração de um artigo."""
|
||||
|
||||
input_meta: InputArticle
|
||||
extraction_status: Literal["success", "failed"]
|
||||
error_message: str | None
|
||||
crawled_url: str
|
||||
page_title: str | None
|
||||
http_status: int | None
|
||||
extraction_status: Literal["success", "failed"] | None = None
|
||||
classification_status: Literal["failed"] | None = None
|
||||
error_message: str | None = None
|
||||
crawled_url: str = ""
|
||||
page_title: str | None = None
|
||||
http_status: int | None = None
|
||||
trafilatura: TrafilaturaData | None = None
|
||||
newspaper4k: NewspaperData | None = None
|
||||
readability: ReadabilityData | None = None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
res: dict[str, Any] = {
|
||||
"input_meta": self.input_meta.to_dict(),
|
||||
"extraction_status": self.extraction_status,
|
||||
"error_message": self.error_message,
|
||||
"crawled_url": self.crawled_url,
|
||||
"page_title": self.page_title,
|
||||
"http_status": self.http_status,
|
||||
"trafilatura": self.trafilatura.to_dict() if self.trafilatura else None,
|
||||
"newspaper4k": self.newspaper4k.to_dict() if self.newspaper4k else None,
|
||||
"readability": self.readability.to_dict() if self.readability else None,
|
||||
}
|
||||
if self.classification_status is not None:
|
||||
res["classification_status"] = self.classification_status
|
||||
if self.extraction_status is not None:
|
||||
res["extraction_status"] = self.extraction_status
|
||||
res["error_message"] = self.error_message
|
||||
res["crawled_url"] = self.crawled_url
|
||||
res["page_title"] = self.page_title
|
||||
res["http_status"] = self.http_status
|
||||
if self.classification_status is None:
|
||||
res["trafilatura"] = self.trafilatura.to_dict() if self.trafilatura else None
|
||||
res["newspaper4k"] = self.newspaper4k.to_dict() if self.newspaper4k else None
|
||||
res["readability"] = self.readability.to_dict() if self.readability else None
|
||||
return res
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -538,7 +922,7 @@ def log_info(message: str, silent: bool = False) -> None:
|
||||
|
||||
|
||||
def load_search_json(file_path: Path) -> tuple[str | None, str, list[InputArticle]]:
|
||||
"""Carrega o arquivo JSON gerado pelo extrator de notícias."""
|
||||
"""Carrega o arquivo JSON gerado pelo extrator de notícias preservando 100% dos metadados."""
|
||||
if not file_path.exists():
|
||||
raise FileNotFoundError(f"Arquivo de entrada não encontrado: {file_path}")
|
||||
|
||||
@@ -556,9 +940,7 @@ def load_search_json(file_path: Path) -> tuple[str | None, str, list[InputArticl
|
||||
InputArticle(
|
||||
titulo=item["titulo"],
|
||||
url=item["url"],
|
||||
subtitulo=item.get("subtitulo"),
|
||||
quando_publicado=item.get("quando_publicado"),
|
||||
pagina=item.get("pagina", 1),
|
||||
raw_data=dict(item),
|
||||
)
|
||||
)
|
||||
|
||||
@@ -595,11 +977,31 @@ def process_batch(
|
||||
silent=silent,
|
||||
)
|
||||
|
||||
# Determinar caminho de saída padrão se não especificado
|
||||
# Determinar caminho de saída textual e caminho do arquivo de mídia
|
||||
if output_path is None:
|
||||
output_path = input_path.parent / f"{input_path.stem}_extracted.json"
|
||||
text_output_path = input_path.parent / f"{input_path.stem}_extracted.json"
|
||||
media_output_path = input_path.parent / f"{input_path.stem}_media.json"
|
||||
else:
|
||||
text_output_path = output_path
|
||||
media_output_path = output_path.with_name(f"{output_path.stem}_media{output_path.suffix}")
|
||||
|
||||
# Inicializar contadores operacionais das 11 métricas
|
||||
metrics: dict[str, int] = {
|
||||
"total_evaluated": 0,
|
||||
"text": 0,
|
||||
"media": 0,
|
||||
"media/video": 0,
|
||||
"media/image": 0,
|
||||
"media/images": 0,
|
||||
"media/embed": 0,
|
||||
"media/mixed": 0,
|
||||
"fallback_groq": 0,
|
||||
"fallback_omniroute": 0,
|
||||
"classification_failed": 0,
|
||||
}
|
||||
|
||||
extracted_list: list[ExtractedArticle] = []
|
||||
media_articles: list[dict[str, Any]] = []
|
||||
successful_count = 0
|
||||
failed_count = 0
|
||||
start_time = time.time()
|
||||
@@ -611,6 +1013,68 @@ def process_batch(
|
||||
|
||||
try:
|
||||
html, page_title, http_status = crawler.crawl(url)
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
|
||||
metrics["total_evaluated"] += 1
|
||||
candidate_info = detect_candidate_media(soup)
|
||||
|
||||
# Gate estrutural prévio: se houver mídia candidata, envia ao classificador
|
||||
if candidate_info.has_candidate_media:
|
||||
payload = build_compact_payload(soup, candidate_info)
|
||||
classification, error_msg = classify_media_content(
|
||||
payload, metrics, silent=silent
|
||||
)
|
||||
|
||||
if classification is not None and classification.content_type == "media":
|
||||
metrics["media"] += 1
|
||||
if classification.media_type:
|
||||
m_key = f"media/{classification.media_type}"
|
||||
if m_key in metrics:
|
||||
metrics[m_key] += 1
|
||||
|
||||
media_article = {
|
||||
"input_meta": article.to_dict(),
|
||||
"crawled_url": url,
|
||||
"page_title": page_title,
|
||||
"http_status": http_status,
|
||||
"content_type": "media",
|
||||
"media_type": classification.media_type,
|
||||
}
|
||||
media_articles.append(media_article)
|
||||
log_info(
|
||||
f'📹 [{idx}/{total}] Publicação predominantemente de mídia ({classification.media_type}) desviada para *_media.json',
|
||||
silent=silent,
|
||||
)
|
||||
continue
|
||||
elif classification is not None and classification.content_type == "text":
|
||||
metrics["text"] += 1
|
||||
elif classification is None:
|
||||
# Falha total na cadeia de provedores
|
||||
metrics["classification_failed"] += 1
|
||||
failed_count += 1
|
||||
log_info(
|
||||
f"⚠️ [{idx}/{total}] Falha de classificação para URL '{url}': {error_msg}",
|
||||
silent=silent,
|
||||
)
|
||||
if not silent:
|
||||
sys.stderr.write(f"[MEDIA] Falha total da cadeia de classificação: {error_msg}\n")
|
||||
sys.stderr.flush()
|
||||
failed_article = ExtractedArticle(
|
||||
input_meta=article,
|
||||
classification_status="failed",
|
||||
error_message=error_msg,
|
||||
crawled_url=url,
|
||||
page_title=page_title,
|
||||
http_status=http_status,
|
||||
trafilatura=None,
|
||||
newspaper4k=None,
|
||||
readability=None,
|
||||
)
|
||||
extracted_list.append(failed_article)
|
||||
continue
|
||||
else:
|
||||
# Bypass direto do gate estrutural (sem mídia candidata)
|
||||
metrics["text"] += 1
|
||||
|
||||
log_info(
|
||||
f"⚙️ [{idx}/{total}] Processando extratores (Trafilatura, Newspaper4k, Readability)...",
|
||||
@@ -658,25 +1122,47 @@ def process_batch(
|
||||
|
||||
extracted_list.append(extracted_article)
|
||||
|
||||
# Emissão incondicional do arquivo de mídia *_media.json
|
||||
save_media_json(media_articles, media_output_path)
|
||||
|
||||
elapsed = time.time() - start_time
|
||||
now_iso = datetime.now(timezone.utc).isoformat()
|
||||
|
||||
# Contadores do relatório textual refletem estritamente os itens presentes em extracted_list
|
||||
report = ExtractionBatchReport(
|
||||
source_file=str(input_path),
|
||||
processed_at=now_iso,
|
||||
total_articles=total,
|
||||
total_articles=len(extracted_list),
|
||||
successful_articles=successful_count,
|
||||
failed_articles=failed_count,
|
||||
articles=extracted_list,
|
||||
)
|
||||
|
||||
save_extracted_json(report, output_path)
|
||||
log_info(f"💾 Relatório final gravado com sucesso em: '{output_path}'", silent=silent)
|
||||
save_extracted_json(report, text_output_path)
|
||||
log_info(f"💾 Relatório final gravado com sucesso em: '{text_output_path}'", silent=silent)
|
||||
log_info(f"💾 Arquivo de mídia gravado com sucesso em: '{media_output_path}' ({len(media_articles)} artigo(s))", silent=silent)
|
||||
log_info(
|
||||
f"📊 Resumo: {total} total | {successful_count} sucessos | {failed_count} falhas | Tempo: {elapsed:.2f}s",
|
||||
f"📊 Resumo: {len(extracted_list)} no JSON textual | {successful_count} sucessos | {failed_count} falhas | Tempo: {elapsed:.2f}s",
|
||||
silent=silent,
|
||||
)
|
||||
|
||||
if not silent:
|
||||
media_subtypes = ", ".join(
|
||||
f"{k.split('/')[1]}={v}"
|
||||
for k, v in metrics.items()
|
||||
if k.startswith("media/") and v > 0
|
||||
)
|
||||
subtypes_str = f" ({media_subtypes})" if media_subtypes else ""
|
||||
sys.stderr.write(
|
||||
f"[MEDIA] Métricas de Roteamento:\n"
|
||||
f" - Total avaliados: {metrics['total_evaluated']}\n"
|
||||
f" - Texto: {metrics['text']}\n"
|
||||
f" - Mídia: {metrics['media']}{subtypes_str}\n"
|
||||
f" - Fallbacks: Groq={metrics['fallback_groq']}, OmniRoute={metrics['fallback_omniroute']}\n"
|
||||
f" - Falhas de classificação: {metrics['classification_failed']}\n"
|
||||
)
|
||||
sys.stderr.flush()
|
||||
|
||||
return report
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user