feat(media-routing): implement 007 media article routing, runtime architecture diagram and update graphify knowledge graph
This commit is contained in:
@@ -0,0 +1,365 @@
|
||||
"""
|
||||
Testes unitários do classificador de mídia e gate estrutural DOM.
|
||||
|
||||
Valida schema de 2 campos, navegação estrutural na DOM (Zero-Regex)
|
||||
e montagem do payload compacto.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
import urllib.error
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
# Adiciona o diretório raiz ao sys.path para import dos scripts
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent.parent))
|
||||
|
||||
from scripts.extract_article_contents import (
|
||||
MEDIA_CLASSIFIER_PROMPT,
|
||||
MEDIA_CLASSIFIER_SCHEMA,
|
||||
MediaCandidateInfo,
|
||||
MediaClassification,
|
||||
build_compact_payload,
|
||||
classify_media_content,
|
||||
detect_candidate_media,
|
||||
validate_classifier_response,
|
||||
)
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 1. Testes de Validação do Contrato Estruturado (validate_classifier_response)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_validate_classifier_response_text_valid() -> None:
|
||||
data = {"content_type": "text", "media_type": None}
|
||||
result = validate_classifier_response(data)
|
||||
assert result == MediaClassification(content_type="text", media_type=None)
|
||||
|
||||
|
||||
def test_validate_classifier_response_text_with_media_type_invalid() -> None:
|
||||
data = {"content_type": "text", "media_type": "image"}
|
||||
assert validate_classifier_response(data) is None
|
||||
|
||||
|
||||
def test_validate_classifier_response_media_valid() -> None:
|
||||
for m_type in ("video", "image", "images", "embed", "mixed"):
|
||||
data = {"content_type": "media", "media_type": m_type}
|
||||
result = validate_classifier_response(data)
|
||||
assert result == MediaClassification(content_type="media", media_type=m_type)
|
||||
|
||||
|
||||
def test_validate_classifier_response_media_with_null_invalid() -> None:
|
||||
data = {"content_type": "media", "media_type": None}
|
||||
assert validate_classifier_response(data) is None
|
||||
|
||||
|
||||
def test_validate_classifier_response_unknown_values() -> None:
|
||||
assert validate_classifier_response({"content_type": "audio", "media_type": None}) is None
|
||||
assert validate_classifier_response({"content_type": "media", "media_type": "podcast"}) is None
|
||||
|
||||
|
||||
def test_validate_classifier_response_missing_or_extra_fields() -> None:
|
||||
assert validate_classifier_response({"content_type": "text"}) is None
|
||||
assert validate_classifier_response({"content_type": "text", "media_type": None, "extra": 123}) is None
|
||||
assert validate_classifier_response("not_a_dict") is None
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 2. Testes do Gate Estrutural DOM (detect_candidate_media)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_detect_candidate_media_video() -> None:
|
||||
html = """
|
||||
<html>
|
||||
<body>
|
||||
<article>
|
||||
<h1>Título com Vídeo</h1>
|
||||
<video controls><source src="movie.mp4" type="video/mp4"></video>
|
||||
<p>Texto curto da matéria.</p>
|
||||
</article>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
info = detect_candidate_media(soup)
|
||||
assert info.has_candidate_media is True
|
||||
assert info.has_video is True
|
||||
assert info.image_count == 0
|
||||
assert info.has_embed is False
|
||||
|
||||
|
||||
def test_detect_candidate_media_isolated_source_not_video() -> None:
|
||||
html = """
|
||||
<html>
|
||||
<body>
|
||||
<article>
|
||||
<h1>Artigo com source isolado</h1>
|
||||
<audio><source src="podcast.mp3"></audio>
|
||||
<p>Apenas texto e áudio.</p>
|
||||
</article>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
info = detect_candidate_media(soup)
|
||||
assert info.has_video is False
|
||||
assert info.has_candidate_media is False
|
||||
|
||||
|
||||
def test_detect_candidate_media_image_wrappers_single_count() -> None:
|
||||
html = """
|
||||
<html>
|
||||
<body>
|
||||
<article>
|
||||
<h1>Notícia com Imagens em Wrappers</h1>
|
||||
<figure>
|
||||
<img src="foto1.jpg" alt="Foto 1">
|
||||
<figcaption>Legenda 1</figcaption>
|
||||
</figure>
|
||||
<picture>
|
||||
<source srcset="foto2.webp">
|
||||
<img src="foto2.jpg" alt="Foto 2">
|
||||
</picture>
|
||||
</article>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
info = detect_candidate_media(soup)
|
||||
assert info.has_candidate_media is True
|
||||
assert info.image_count == 2
|
||||
assert info.has_video is False
|
||||
|
||||
|
||||
def test_detect_candidate_media_multiple_images() -> None:
|
||||
html = """
|
||||
<html>
|
||||
<body>
|
||||
<main>
|
||||
<h1>Galeria de Fotos</h1>
|
||||
<img src="img1.jpg">
|
||||
<img src="img2.jpg">
|
||||
<img src="img3.jpg">
|
||||
<p>Galeria completa do evento.</p>
|
||||
</main>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
info = detect_candidate_media(soup)
|
||||
assert info.has_candidate_media is True
|
||||
assert info.image_count == 3
|
||||
|
||||
|
||||
def test_detect_candidate_media_embeds() -> None:
|
||||
html = """
|
||||
<html>
|
||||
<body>
|
||||
<div role="main">
|
||||
<h1>Post Incorporado</h1>
|
||||
<iframe src="https://platform.twitter.com/embed/Tweet.html"></iframe>
|
||||
<p>Veja o post acima.</p>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
info = detect_candidate_media(soup)
|
||||
assert info.has_candidate_media is True
|
||||
assert info.has_embed is True
|
||||
|
||||
|
||||
def test_detect_candidate_media_in_page_furniture_ignored() -> None:
|
||||
html = """
|
||||
<html>
|
||||
<body>
|
||||
<header>
|
||||
<img src="logo.png" alt="Logo">
|
||||
<nav><img src="icon.png"></nav>
|
||||
</header>
|
||||
<article>
|
||||
<h1>Matéria Textual Pura</h1>
|
||||
<p>Texto jornalístico longo sem mídia interna no corpo.</p>
|
||||
<p>Segundo parágrafo informativo.</p>
|
||||
</article>
|
||||
<aside>
|
||||
<iframe src="banner.html"></iframe>
|
||||
<img src="ad.jpg">
|
||||
</aside>
|
||||
<footer>
|
||||
<img src="partner.png">
|
||||
</footer>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
info = detect_candidate_media(soup)
|
||||
assert info.has_candidate_media is False
|
||||
assert info.image_count == 0
|
||||
assert info.has_video is False
|
||||
assert info.has_embed is False
|
||||
|
||||
|
||||
def test_detect_candidate_media_structural_region_precedence() -> None:
|
||||
# <article> deve ter precedência sobre <main> e <body>
|
||||
html = """
|
||||
<html>
|
||||
<body>
|
||||
<main>
|
||||
<img src="outside_article.jpg">
|
||||
<article>
|
||||
<h1>Dentro do Article</h1>
|
||||
<video src="article_video.mp4"></video>
|
||||
<p>Parágrafo do artigo.</p>
|
||||
</article>
|
||||
</main>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
info = detect_candidate_media(soup)
|
||||
assert info.has_candidate_media is True
|
||||
assert info.has_video is True
|
||||
assert info.image_count == 0
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 3. Testes de Montagem do Compact Payload (build_compact_payload)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_build_compact_payload_preserves_text_and_unicode() -> None:
|
||||
html = """
|
||||
<html>
|
||||
<head><title>Título da Publicação — Jornal Exemplo</title></head>
|
||||
<body>
|
||||
<article>
|
||||
<h1>Título da Publicação</h1>
|
||||
<p>Primeiro parágrafo com acentuação: <em>ação, notícia & análise</em>.</p>
|
||||
<p>Segundo parágrafo com texto em espanhol: ¿Cómo estás? Fútbol y pasión.</p>
|
||||
<video src="clip.mp4"></video>
|
||||
</article>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
candidate_info = MediaCandidateInfo(has_candidate_media=True, has_video=True, image_count=0, has_embed=False)
|
||||
payload = build_compact_payload(soup, candidate_info)
|
||||
|
||||
assert "Título da Publicação" in payload
|
||||
assert "ação, notícia & análise" in payload
|
||||
assert "¿Cómo estás? Fútbol y pasión." in payload
|
||||
assert "Video=True" in payload
|
||||
assert "ImagesCount=0" in payload
|
||||
assert "Embed=False" in payload
|
||||
assert "<p>" not in payload
|
||||
assert "<article>" not in payload
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 4. Testes de Bypass Direto do Gate Sem LLM (User Story 2 / T010)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_detect_candidate_media_pure_text_bypasses_llm() -> None:
|
||||
html = """
|
||||
<html>
|
||||
<body>
|
||||
<article>
|
||||
<h1>Editorial Textual Pura</h1>
|
||||
<p>Primeiro parágrafo de notícia densa.</p>
|
||||
<p>Segundo parágrafo detalhando o acontecimento político e econômico.</p>
|
||||
<p>Terceiro parágrafo com as conclusões e declarações oficiais.</p>
|
||||
</article>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
info = detect_candidate_media(soup)
|
||||
assert info.has_candidate_media is False
|
||||
assert info.has_video is False
|
||||
assert info.image_count == 0
|
||||
assert info.has_embed is False
|
||||
|
||||
|
||||
# ==============================================================================
|
||||
# 5. Testes da Cadeia Sequencial de Provedores e Invariância de Prompt (User Story 3 / T013)
|
||||
# ==============================================================================
|
||||
|
||||
|
||||
def test_classify_media_content_first_valid_wins_ollama(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
# Valida que prompt não solicita reasoning, keywords, summary, etc.
|
||||
forbidden = ["reasoning", "confidence", "rationale", "summary", "keywords", "evidence", "tradução"]
|
||||
for word in forbidden:
|
||||
assert word not in MEDIA_CLASSIFIER_PROMPT.lower()
|
||||
|
||||
metrics = {"fallback_groq": 0, "fallback_omniroute": 0, "classification_failed": 0}
|
||||
call_log: list[str] = []
|
||||
|
||||
def mock_http(url: str, payload: dict, headers: dict, timeout: int) -> tuple[int, str]:
|
||||
call_log.append(url)
|
||||
# Ollama responde válido
|
||||
return 200, json.dumps({"message": {"content": json.dumps({"content_type": "media", "media_type": "video"})}})
|
||||
|
||||
monkeypatch.setattr("scripts.extract_article_contents._http_post_json", mock_http)
|
||||
|
||||
res, err = classify_media_content("payload", metrics, silent=True)
|
||||
assert res == MediaClassification(content_type="media", media_type="video")
|
||||
assert err is None
|
||||
assert len(call_log) == 1
|
||||
assert "11434" in call_log[0]
|
||||
assert metrics["fallback_groq"] == 0
|
||||
assert metrics["fallback_omniroute"] == 0
|
||||
|
||||
|
||||
def test_classify_media_content_ollama_fail_groq_success(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("GROQ_API_KEY", "test-key")
|
||||
|
||||
metrics = {"fallback_groq": 0, "fallback_omniroute": 0, "classification_failed": 0}
|
||||
call_log: list[str] = []
|
||||
|
||||
def mock_http(url: str, payload: dict, headers: dict, timeout: int) -> tuple[int, str]:
|
||||
call_log.append(url)
|
||||
if "11434" in url:
|
||||
# Ollama falha
|
||||
raise urllib.error.URLError("Connection refused")
|
||||
# Groq responde válido
|
||||
assert payload["model"] == "openai/gpt-oss-20b"
|
||||
assert payload["reasoning_effort"] == "low"
|
||||
assert payload["temperature"] == 0.0
|
||||
return 200, json.dumps({"choices": [{"message": {"content": json.dumps({"content_type": "text", "media_type": None})}}]})
|
||||
|
||||
monkeypatch.setattr("scripts.extract_article_contents._http_post_json", mock_http)
|
||||
|
||||
res, err = classify_media_content("payload", metrics, silent=True)
|
||||
assert res == MediaClassification(content_type="text", media_type=None)
|
||||
assert err is None
|
||||
assert len(call_log) == 2
|
||||
assert metrics["fallback_groq"] == 1
|
||||
assert metrics["fallback_omniroute"] == 0
|
||||
|
||||
|
||||
def test_classify_media_content_all_fail(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("GROQ_API_KEY", "test-key")
|
||||
monkeypatch.setenv("OMNIROUTE_ENDPOINT", "http://omniroute.local")
|
||||
|
||||
metrics = {"fallback_groq": 0, "fallback_omniroute": 0, "classification_failed": 0}
|
||||
|
||||
def mock_http(url: str, payload: dict, headers: dict, timeout: int) -> tuple[int, str]:
|
||||
raise urllib.error.URLError("Server unreachable")
|
||||
|
||||
monkeypatch.setattr("scripts.extract_article_contents._http_post_json", mock_http)
|
||||
|
||||
res, err = classify_media_content("payload", metrics, silent=True)
|
||||
assert res is None
|
||||
assert err is not None
|
||||
assert metrics["fallback_groq"] == 1
|
||||
assert metrics["fallback_omniroute"] == 1
|
||||
|
||||
|
||||
Reference in New Issue
Block a user