568 lines
23 KiB
Python
568 lines
23 KiB
Python
"""
|
|
Testes de integração para roteamento de artigos de mídia e texto.
|
|
|
|
Cobre os Cenários B a F (User Story 1), resolução de caminhos padrão e com -o,
|
|
e preservação integral de metadados em input_meta com campos customizados.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
from unittest.mock import patch, MagicMock
|
|
|
|
import pytest
|
|
|
|
# Adiciona o diretório raiz ao sys.path
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent.parent))
|
|
|
|
from scripts.extract_article_contents import process_batch, ArticleCrawler
|
|
|
|
|
|
@pytest.fixture
|
|
def sample_input_json(tmp_path: Path) -> Path:
|
|
"""Cria um arquivo de entrada de busca sintético para testes."""
|
|
input_file = tmp_path / "search_sample.json"
|
|
data = {
|
|
"query": "noticias futebol",
|
|
"language": "pt",
|
|
"items": [
|
|
{
|
|
"titulo": "Gol Histórico do River Plate",
|
|
"url": "https://example.com/video-gol",
|
|
"subtitulo": "Veja o lance em vídeo",
|
|
"quando_publicado": "2026-08-24T12:00:00Z",
|
|
"pagina": 1,
|
|
"custom_field": "preserve-me",
|
|
"custom_number": 42,
|
|
}
|
|
],
|
|
}
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump(data, f, ensure_ascii=False, indent=2)
|
|
return input_file
|
|
|
|
|
|
def mock_crawl_html(html_body: str, title: str = "Página Notícia", status_code: int = 200) -> MagicMock:
|
|
mock = MagicMock()
|
|
mock.__enter__.return_value = mock
|
|
mock.__exit__.return_value = None
|
|
mock.crawl.return_value = (html_body, title, status_code)
|
|
return mock
|
|
|
|
|
|
def test_media_routing_scenario_b_video(tmp_path: Path) -> None:
|
|
"""Cenário B: Publicação com vídeo e texto curto descritivo -> roteada para *_media.json."""
|
|
input_file = tmp_path / "river_plate.json"
|
|
input_data = {
|
|
"query": "river plate",
|
|
"language": "es",
|
|
"items": [
|
|
{
|
|
"titulo": "Video del Golazo",
|
|
"url": "https://example.com/video",
|
|
"custom_tracker_id": "trk-999",
|
|
}
|
|
],
|
|
}
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump(input_data, f)
|
|
|
|
html = """
|
|
<html>
|
|
<body>
|
|
<article>
|
|
<h1>Video del Golazo</h1>
|
|
<video src="gol.mp4"></video>
|
|
<p>Mira el resumen del partido.</p>
|
|
</article>
|
|
</body>
|
|
</html>
|
|
"""
|
|
|
|
# Mock crawler e mock LLM (Ollama respondendo content_type=media, media_type=video)
|
|
mock_llm_response = (
|
|
200,
|
|
json.dumps(
|
|
{
|
|
"message": {
|
|
"content": json.dumps({"content_type": "media", "media_type": "video"})
|
|
}
|
|
}
|
|
),
|
|
)
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json", return_value=mock_llm_response), \
|
|
patch("scripts.extract_article_contents.extract_all_engines") as mock_extractors:
|
|
|
|
crawler_instance = mock_crawl_html(html, title="Video del Golazo", status_code=200)
|
|
MockCrawler.return_value = crawler_instance
|
|
|
|
report = process_batch(input_file, silent=True)
|
|
|
|
# Multimotor NÃO deve ter sido executado
|
|
assert mock_extractors.call_count == 0
|
|
|
|
# Relatório textual deve ter 0 artigos no JSON principal
|
|
assert report.total_articles == 0
|
|
assert report.successful_articles == 0
|
|
assert len(report.articles) == 0
|
|
|
|
# Arquivo de mídia river_plate_media.json deve ter sido criado
|
|
media_file = tmp_path / "river_plate_media.json"
|
|
assert media_file.exists()
|
|
with open(media_file, "r", encoding="utf-8") as f:
|
|
media_data = json.load(f)
|
|
|
|
assert "articles" in media_data
|
|
assert len(media_data["articles"]) == 1
|
|
article = media_data["articles"][0]
|
|
assert article["content_type"] == "media"
|
|
assert article["media_type"] == "video"
|
|
assert article["crawled_url"] == "https://example.com/video"
|
|
assert article["http_status"] == 200
|
|
assert article["input_meta"]["titulo"] == "Video del Golazo"
|
|
assert article["input_meta"]["custom_tracker_id"] == "trk-999"
|
|
|
|
|
|
def test_media_routing_scenarios_c_d_e_f(tmp_path: Path) -> None:
|
|
"""Cenários C, D, E, F: image, images, embed, mixed."""
|
|
cases = [
|
|
("image", "<article><img src='foto.jpg'><p>Legenda curta.</p></article>"),
|
|
("images", "<article><img src='1.jpg'><img src='2.jpg'><p>Galeria.</p></article>"),
|
|
("embed", "<article><iframe src='insta.com/p/123'></iframe><p>Post.</p></article>"),
|
|
("mixed", "<article><video src='v.mp4'></video><img src='f.jpg'><p>Misto.</p></article>"),
|
|
]
|
|
|
|
for media_type, html in cases:
|
|
input_file = tmp_path / f"test_{media_type}.json"
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump(
|
|
{
|
|
"items": [
|
|
{
|
|
"titulo": f"Titulo {media_type}",
|
|
"url": f"https://example.com/{media_type}",
|
|
}
|
|
]
|
|
},
|
|
f,
|
|
)
|
|
|
|
mock_llm_response = (
|
|
200,
|
|
json.dumps(
|
|
{
|
|
"message": {
|
|
"content": json.dumps({"content_type": "media", "media_type": media_type})
|
|
}
|
|
}
|
|
),
|
|
)
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json", return_value=mock_llm_response), \
|
|
patch("scripts.extract_article_contents.extract_all_engines") as mock_extractors:
|
|
|
|
MockCrawler.return_value = mock_crawl_html(html)
|
|
process_batch(input_file, silent=True)
|
|
|
|
assert mock_extractors.call_count == 0
|
|
|
|
media_file = tmp_path / f"test_{media_type}_media.json"
|
|
assert media_file.exists()
|
|
with open(media_file, "r", encoding="utf-8") as f:
|
|
media_data = json.load(f)
|
|
|
|
assert len(media_data["articles"]) == 1
|
|
assert media_data["articles"][0]["media_type"] == media_type
|
|
|
|
|
|
def test_media_routing_custom_output_path_resolution(tmp_path: Path) -> None:
|
|
"""Valida resolução de caminho com -o customizado: out/processados/resultado.json -> resultado_media.json."""
|
|
input_file = tmp_path / "raw_input.json"
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump({"items": [{"titulo": "T", "url": "https://example.com/item"}]}, f)
|
|
|
|
custom_out = tmp_path / "custom_dir" / "saida_final.json"
|
|
|
|
html = "<article><video src='v.mp4'></video><p>Texto curto.</p></article>"
|
|
mock_llm_response = (
|
|
200,
|
|
json.dumps(
|
|
{"message": {"content": json.dumps({"content_type": "media", "media_type": "video"})}}
|
|
),
|
|
)
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json", return_value=mock_llm_response):
|
|
|
|
MockCrawler.return_value = mock_crawl_html(html)
|
|
process_batch(input_file, output_path=custom_out, silent=True)
|
|
|
|
assert custom_out.exists()
|
|
expected_media_file = tmp_path / "custom_dir" / "saida_final_media.json"
|
|
assert expected_media_file.exists()
|
|
|
|
|
|
def test_textual_routing_scenario_a_no_media_bypasses_llm(tmp_path: Path) -> None:
|
|
"""Cenário A: Artigo sem mídia candidata segue direto ao multimotor sem chamar LLM."""
|
|
input_file = tmp_path / "pure_text.json"
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump(
|
|
{
|
|
"items": [
|
|
{
|
|
"titulo": "Notícia Textual Sem Mídia",
|
|
"url": "https://example.com/texto-puro",
|
|
"custom_field": "preserve-me",
|
|
"custom_number": 42,
|
|
}
|
|
]
|
|
},
|
|
f,
|
|
)
|
|
|
|
html = "<article><h1>Texto Puro</h1><p>Notícia densa com múltiplos parágrafos.</p></article>"
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json") as mock_http, \
|
|
patch("scripts.extract_article_contents.extract_all_engines") as mock_extractors:
|
|
|
|
MockCrawler.return_value = mock_crawl_html(html)
|
|
mock_extractors.return_value = (None, None, None)
|
|
|
|
report = process_batch(input_file, silent=True)
|
|
|
|
# LLM NÃO deve ser chamado (bypass)
|
|
assert mock_http.call_count == 0
|
|
|
|
# Multimotor DEVE ser chamado
|
|
assert mock_extractors.call_count == 1
|
|
|
|
# JSON textual principal contém o artigo
|
|
assert report.total_articles == 1
|
|
assert report.successful_articles == 1
|
|
assert report.failed_articles == 0
|
|
assert len(report.articles) == 1
|
|
assert report.articles[0].input_meta.to_dict()["custom_field"] == "preserve-me"
|
|
assert report.articles[0].input_meta.to_dict()["custom_number"] == 42
|
|
|
|
# *_media.json MUST ser gerado com {"articles": []}
|
|
media_file = tmp_path / "pure_text_media.json"
|
|
assert media_file.exists()
|
|
with open(media_file, "r", encoding="utf-8") as f:
|
|
media_data = json.load(f)
|
|
assert media_data == {"articles": []}
|
|
|
|
|
|
def test_textual_routing_scenarios_g_and_h_with_illustrative_media(tmp_path: Path) -> None:
|
|
"""Cenários G e H: Notícias longas com imagem/vídeo ilustrativo recebem content_type=text."""
|
|
cases = [
|
|
("cenario_g_image", "<article><img src='foto.jpg'><p>Texto jornalístico longo e substancial 1.</p><p>Texto longo 2.</p></article>"),
|
|
("cenario_h_video", "<article><video src='v.mp4'></video><p>Texto jornalístico longo e substancial 1.</p><p>Texto longo 2.</p></article>"),
|
|
]
|
|
|
|
mock_llm_text_response = (
|
|
200,
|
|
json.dumps(
|
|
{"message": {"content": json.dumps({"content_type": "text", "media_type": None})}}
|
|
),
|
|
)
|
|
|
|
for case_name, html in cases:
|
|
input_file = tmp_path / f"{case_name}.json"
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump(
|
|
{
|
|
"items": [
|
|
{
|
|
"titulo": f"Matéria {case_name}",
|
|
"url": f"https://example.com/{case_name}",
|
|
"custom_author": "Repórter Especial",
|
|
}
|
|
]
|
|
},
|
|
f,
|
|
)
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json", return_value=mock_llm_text_response) as mock_http, \
|
|
patch("scripts.extract_article_contents.extract_all_engines") as mock_extractors:
|
|
|
|
MockCrawler.return_value = mock_crawl_html(html)
|
|
mock_extractors.return_value = (None, None, None)
|
|
|
|
report = process_batch(input_file, silent=True)
|
|
|
|
# LLM foi chamado pois havia mídia candidata
|
|
assert mock_http.call_count == 1
|
|
# Multimotor foi executado pois a classificação retornou text
|
|
assert mock_extractors.call_count == 1
|
|
|
|
assert report.total_articles == 1
|
|
assert report.successful_articles == 1
|
|
assert report.articles[0].input_meta.to_dict()["custom_author"] == "Repórter Especial"
|
|
|
|
# *_media.json gerado incondicionalmente vazio
|
|
media_file = tmp_path / f"{case_name}_media.json"
|
|
assert media_file.exists()
|
|
with open(media_file, "r", encoding="utf-8") as f:
|
|
media_data = json.load(f)
|
|
assert media_data == {"articles": []}
|
|
|
|
|
|
# ==============================================================================
|
|
# Testes de Integração de Fallback e Falha Cumulativa (User Story 3 / T016)
|
|
# ==============================================================================
|
|
|
|
|
|
def test_scenario_k_cumulative_failure(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
"""Cenário K: Falha em todos os 3 provedores -> registro inline de falha no JSON principal."""
|
|
monkeypatch.setenv("GROQ_API_KEY", "test-key")
|
|
monkeypatch.setenv("OMNIROUTE_ENDPOINT", "http://omniroute.test")
|
|
|
|
input_file = tmp_path / "fail_input.json"
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump(
|
|
{
|
|
"items": [
|
|
{
|
|
"titulo": "Artigo com Falha de LLM",
|
|
"url": "https://example.com/fail-item",
|
|
"custom_id": "cust-77",
|
|
}
|
|
]
|
|
},
|
|
f,
|
|
)
|
|
|
|
html = "<article><video src='v.mp4'></video><p>Texto.</p></article>"
|
|
|
|
def mock_http_fail(*args: Any, **kwargs: Any) -> tuple[int, str]:
|
|
raise ConnectionResetError("Connection reset")
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json", side_effect=mock_http_fail), \
|
|
patch("scripts.extract_article_contents.extract_all_engines") as mock_extractors:
|
|
|
|
MockCrawler.return_value = mock_crawl_html(html, title="Artigo com Falha de LLM", status_code=200)
|
|
|
|
report = process_batch(input_file, silent=True)
|
|
|
|
# Multimotor NÃO deve ter sido executado
|
|
assert mock_extractors.call_count == 0
|
|
|
|
# Artigo presente no JSON textual como falha
|
|
assert report.total_articles == 1
|
|
assert report.successful_articles == 0
|
|
assert report.failed_articles == 1
|
|
assert len(report.articles) == 1
|
|
|
|
failed_art = report.articles[0]
|
|
assert failed_art.classification_status == "failed"
|
|
assert failed_art.error_message is not None
|
|
assert failed_art.crawled_url == "https://example.com/fail-item"
|
|
assert failed_art.page_title == "Artigo com Falha de LLM"
|
|
assert failed_art.http_status == 200
|
|
assert failed_art.input_meta.to_dict()["custom_id"] == "cust-77"
|
|
assert failed_art.trafilatura is None
|
|
assert failed_art.newspaper4k is None
|
|
assert failed_art.readability is None
|
|
|
|
# NÃO deve estar em *_media.json
|
|
media_file = tmp_path / "fail_input_media.json"
|
|
assert media_file.exists()
|
|
with open(media_file, "r", encoding="utf-8") as f:
|
|
media_data = json.load(f)
|
|
assert media_data == {"articles": []}
|
|
|
|
|
|
def test_scenario_i_j_l_fallbacks_and_multilingual(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
"""Cenários I, J, L e M: Transição para Groq/OmniRoute e suporte multilíngue (pt, es, en)."""
|
|
monkeypatch.setenv("GROQ_API_KEY", "groq-key")
|
|
monkeypatch.setenv("OMNIROUTE_ENDPOINT", "http://omniroute.test")
|
|
|
|
input_file = tmp_path / "multilingual.json"
|
|
items = [
|
|
{"titulo": "Notícia em Português", "url": "https://example.com/pt", "lang": "pt"},
|
|
{"titulo": "Noticia en Español", "url": "https://example.com/es", "lang": "es"},
|
|
{"titulo": "English News Article", "url": "https://example.com/en", "lang": "en"},
|
|
]
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump({"items": items}, f)
|
|
|
|
html = "<article><video src='v.mp4'></video><p>Texto.</p></article>"
|
|
|
|
# 1º item: Ollama falha, Groq retorna media
|
|
# 2º item: Ollama falha, Groq falha, OmniRoute retorna media
|
|
# 3º item: Ollama falha, Groq retorna text -> vai para multimotor
|
|
step_state = {"call_idx": 0}
|
|
|
|
def mock_http_chain(url: str, payload: dict, headers: dict, timeout: int) -> tuple[int, str]:
|
|
if "11434" in url:
|
|
# Ollama sempre falha
|
|
raise TimeoutError("Ollama timeout")
|
|
if "groq.com" in url:
|
|
step_state["call_idx"] += 1
|
|
if step_state["call_idx"] == 1:
|
|
# 1º item via Groq: media
|
|
return 200, json.dumps({"choices": [{"message": {"content": json.dumps({"content_type": "media", "media_type": "video"})}}]})
|
|
if step_state["call_idx"] == 2:
|
|
# 2º item: Groq falha
|
|
raise RuntimeError("Groq rate limit")
|
|
if step_state["call_idx"] >= 3:
|
|
# 3º item via Groq: text
|
|
return 200, json.dumps({"choices": [{"message": {"content": json.dumps({"content_type": "text", "media_type": None})}}]})
|
|
if "omniroute.test" in url:
|
|
# OmniRoute responde media para o 2º item
|
|
return 200, json.dumps({"choices": [{"message": {"content": json.dumps({"content_type": "media", "media_type": "embed"})}}]})
|
|
return 500, "error"
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json", side_effect=mock_http_chain), \
|
|
patch("scripts.extract_article_contents.extract_all_engines") as mock_extractors:
|
|
|
|
MockCrawler.return_value = mock_crawl_html(html)
|
|
mock_extractors.return_value = (None, None, None)
|
|
|
|
report = process_batch(input_file, silent=True)
|
|
|
|
# 3º item foi para multimotor
|
|
assert mock_extractors.call_count == 1
|
|
assert report.total_articles == 1
|
|
assert report.successful_articles == 1
|
|
assert report.articles[0].input_meta.to_dict()["lang"] == "en"
|
|
|
|
# 1º e 2º itens foram para *_media.json
|
|
media_file = tmp_path / "multilingual_media.json"
|
|
assert media_file.exists()
|
|
with open(media_file, "r", encoding="utf-8") as f:
|
|
media_data = json.load(f)
|
|
|
|
assert len(media_data["articles"]) == 2
|
|
assert media_data["articles"][0]["input_meta"]["lang"] == "pt"
|
|
assert media_data["articles"][0]["media_type"] == "video"
|
|
assert media_data["articles"][1]["input_meta"]["lang"] == "es"
|
|
assert media_data["articles"][1]["media_type"] == "embed"
|
|
|
|
|
|
# ==============================================================================
|
|
# Testes de Observabilidade, Segurança e CLI (Phase 5 / T019)
|
|
# ==============================================================================
|
|
|
|
|
|
def test_metrics_values_and_stderr_summary_logging(tmp_path: Path, capsys: pytest.CaptureFixture[str]) -> None:
|
|
"""Valida valores das métricas e emissão formatada no stderr quando silent=False."""
|
|
input_file = tmp_path / "metrics_test.json"
|
|
items = [
|
|
{"titulo": "Texto Puro", "url": "https://example.com/puro"},
|
|
{"titulo": "Video Direct", "url": "https://example.com/video"},
|
|
]
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump({"items": items}, f)
|
|
|
|
def mock_crawl(url: str) -> tuple[str, str, int]:
|
|
if "puro" in url:
|
|
return "<article><p>Densa matéria pura.</p></article>", "Puro", 200
|
|
return "<article><video src='v.mp4'></video><p>Vídeo.</p></article>", "Vídeo", 200
|
|
|
|
def mock_http(url: str, payload: dict, headers: dict, timeout: int) -> tuple[int, str]:
|
|
return 200, json.dumps({"message": {"content": json.dumps({"content_type": "media", "media_type": "video"})}})
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json", side_effect=mock_http), \
|
|
patch("scripts.extract_article_contents.extract_all_engines", return_value=(None, None, None)):
|
|
|
|
crawler = MagicMock()
|
|
crawler.__enter__.return_value = crawler
|
|
crawler.__exit__.return_value = None
|
|
crawler.crawl.side_effect = mock_crawl
|
|
MockCrawler.return_value = crawler
|
|
|
|
process_batch(input_file, silent=False)
|
|
|
|
captured = capsys.readouterr()
|
|
assert "[MEDIA] Métricas de Roteamento:" in captured.err
|
|
assert "Total avaliados: 2" in captured.err
|
|
assert "Texto: 1" in captured.err
|
|
assert "Mídia: 1" in captured.err
|
|
assert "video=1" in captured.err
|
|
|
|
|
|
def test_silent_mode_suppresses_all_media_logs(tmp_path: Path, capsys: pytest.CaptureFixture[str]) -> None:
|
|
"""Valida que flag silent=True suprime todos os logs em stderr."""
|
|
input_file = tmp_path / "silent_test.json"
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump({"items": [{"titulo": "T", "url": "https://example.com/item"}]}, f)
|
|
|
|
html = "<article><video src='v.mp4'></video><p>Texto.</p></article>"
|
|
mock_resp = (200, json.dumps({"message": {"content": json.dumps({"content_type": "media", "media_type": "video"})}}))
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json", return_value=mock_resp):
|
|
|
|
MockCrawler.return_value = mock_crawl_html(html)
|
|
process_batch(input_file, silent=True)
|
|
|
|
captured = capsys.readouterr()
|
|
assert "[MEDIA]" not in captured.err
|
|
|
|
|
|
def test_api_key_security_not_leaked_in_logs(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]
|
|
) -> None:
|
|
"""Valida que chaves secretas de API (Groq/OmniRoute) nunca são expostas em logs/stderr."""
|
|
secret_groq = "gsk_secret_token_12345"
|
|
secret_omni = "omni_secret_token_67890"
|
|
monkeypatch.setenv("GROQ_API_KEY", secret_groq)
|
|
monkeypatch.setenv("OMNIROUTE_API_KEY", secret_omni)
|
|
monkeypatch.setenv("OMNIROUTE_ENDPOINT", "http://omniroute.test")
|
|
|
|
input_file = tmp_path / "sec_test.json"
|
|
with open(input_file, "w", encoding="utf-8") as f:
|
|
json.dump({"items": [{"titulo": "T", "url": "https://example.com/sec"}]}, f)
|
|
|
|
html = "<article><video src='v.mp4'></video><p>Texto de teste confidencial.</p></article>"
|
|
|
|
# Ollama falha -> Groq falha -> OmniRoute responde
|
|
def mock_http(url: str, payload: dict, headers: dict, timeout: int) -> tuple[int, str]:
|
|
if "11434" in url:
|
|
raise ConnectionRefusedError()
|
|
if "groq.com" in url:
|
|
raise RuntimeError("Groq rate limit")
|
|
return 200, json.dumps({"choices": [{"message": {"content": json.dumps({"content_type": "media", "media_type": "mixed"})}}]})
|
|
|
|
with patch("scripts.extract_article_contents.ArticleCrawler") as MockCrawler, \
|
|
patch("scripts.extract_article_contents._http_post_json", side_effect=mock_http):
|
|
|
|
MockCrawler.return_value = mock_crawl_html(html)
|
|
process_batch(input_file, silent=False)
|
|
|
|
captured = capsys.readouterr()
|
|
assert secret_groq not in captured.err
|
|
assert secret_omni not in captured.err
|
|
assert secret_groq not in captured.out
|
|
assert secret_omni not in captured.out
|
|
# Garante que payload textual não foi vazado no log
|
|
assert "Texto de teste confidencial." not in captured.err
|
|
|
|
|
|
def test_cli_argument_compatibility() -> None:
|
|
"""Valida que argumentos de linha de comando existentes continuam totalmente compatíveis."""
|
|
from scripts.extract_article_contents import parse_arguments
|
|
|
|
args = parse_arguments(["-i", "input.json", "-o", "custom_out.json", "--limit", "10", "--lang", "es", "--silent"])
|
|
assert args.input == "input.json"
|
|
assert args.output == "custom_out.json"
|
|
assert args.limit == 10
|
|
assert args.language == "es"
|
|
assert args.silent is True
|
|
|
|
|
|
|
|
|