feat(runtime): implement single-article consolidation runtime and modularize codebase

This commit is contained in:
2026-08-24 00:14:07 -03:00
parent e1e0be1353
commit 23de7d8fe7
176 changed files with 266754 additions and 10179 deletions
@@ -0,0 +1,54 @@
"""Unit tests for candidate parsing without regex covering scenarios PAR-001 to PAR-010."""
from src.runtime.candidate.parser import (
parse_metadata_candidates,
parse_raw_text_into_candidates,
resolve_canonical_source_url,
)
def test_parse_raw_text_into_candidates():
markdown_text = """# Main Header
This is the first paragraph of the article.
## Subheader
Here is a second paragraph.
* Bullet one
* Bullet two
> A notable quote from an expert.
"""
candidates = parse_raw_text_into_candidates(markdown_text, extractor="trafilatura")
types = [c.type for c in candidates]
assert "heading" in types
assert "paragraph" in types
assert "list_item" in types
assert "quote" in types
def test_resolve_canonical_source_url_priority():
article_full = {
"crawled_url": "https://example.com/crawled",
"input_meta": {"url": "https://example.com/meta"},
"trafilatura": {"canonical_url": "https://example.com/canonical"},
}
# trafilatura canonical_url has top priority
assert resolve_canonical_source_url(article_full) == "https://example.com/canonical"
# fallback to input_meta.url
article_no_traf = {
"crawled_url": "https://example.com/crawled",
"input_meta": {"url": "https://example.com/meta"},
}
assert resolve_canonical_source_url(article_no_traf) == "https://example.com/meta"
def test_author_parsing_forbids_delimiter_splitting():
article = {"trafilatura": {"author": "Dr. Silva, Ph.D. / Reporter Especial"}}
cand = parse_metadata_candidates(article)
# The full string must be preserved as a single author candidate, not split by commas or slashes
assert len(cand["author_candidates"]) == 1
assert cand["author_candidates"][0]["text"] == "Dr. Silva, Ph.D. / Reporter Especial"
+37
View File
@@ -0,0 +1,37 @@
"""Unit tests for ECP classification adapter covering scenarios ECP-001 to ECP-009."""
from src.runtime.ecp.adapter import ECPClassificationAdapter
def test_ecp_adapter_direct_inherent():
adapter = ECPClassificationAdapter()
sample_ecp = {
"target_entity_id": "Q12345",
"target_name": "Club Atlético River Plate",
"aliases": ["River Plate", "River"],
"domain": "sports",
"anchors": ["Monumental", "Buenos Aires"],
}
content = "# River vs Santa Fe\n\nRiver Plate jugó un gran partido en el estadio Monumental de Buenos Aires."
res = adapter.classify(sample_ecp, content)
assert res["category"] == "DIRECT_INHERENT"
assert res["is_inherent"] is True
assert res["confidence"] > 0.8
assert len(res["evidences"]) > 0
def test_ecp_adapter_not_related():
adapter = ECPClassificationAdapter()
sample_ecp = {
"target_entity_id": "Q12345",
"target_name": "Club Atlético River Plate",
"aliases": ["River Plate"],
"domain": "sports",
"anchors": ["Monumental"],
}
content = "# Gastronomia Francesa\n\nReceita de croissant e baguetes na culinária tradicional de Paris."
res = adapter.classify(sample_ecp, content)
assert res["category"] == "NOT_RELATED"
assert res["is_inherent"] is False
@@ -0,0 +1,46 @@
"""Unit tests for enrichment harness covering scenarios ENR-001 to ENR-009."""
import pytest
from src.runtime.enrichment.harness import (
EnrichmentFailedError,
normalize_tag,
validate_and_extract_enrichment,
)
def test_tag_normalization():
raw_tag = " Copa Sudamericana "
norm = normalize_tag(raw_tag)
assert norm == "copa sudamericana"
def test_validate_and_extract_enrichment_valid():
valid_ids = {"blk_01", "blk_02"}
resp = {
"sentiment": "positive",
"tags": ["River Plate", "copa sudamericana", "Futebol"],
"evidence_candidate_ids": ["blk_01"],
}
extracted = validate_and_extract_enrichment(resp, valid_ids)
assert extracted["sentiment"] == "positive"
assert len(extracted["tags"]) == 3
assert "river plate" in extracted["tags"]
assert "copa sudamericana" in extracted["tags"]
assert "futebol" in extracted["tags"]
def test_enrichment_fails_on_duplicate_tags():
valid_ids = {"blk_01"}
resp = {
"sentiment": "neutral",
"tags": [
"futebol",
"Futebol",
" futebol ",
], # 3 items for schema, but collapses to 1 unique tag
"evidence_candidate_ids": ["blk_01"],
}
with pytest.raises(EnrichmentFailedError) as exc_info:
validate_and_extract_enrichment(resp, valid_ids)
assert "outside allowed bound" in str(exc_info.value)
@@ -0,0 +1,42 @@
"""Unit tests for sequence equivalence mapping covering scenarios CAN-001 to CAN-010."""
from src.runtime.candidate.equivalence import (
compute_sequence_similarity,
map_candidate_equivalences,
normalize_text_for_comparison,
)
from src.runtime.candidate.models import CandidateObject
def test_text_normalization():
text = " São Paulo Futebol Clube\n\t "
norm = normalize_text_for_comparison(text)
assert norm == "são paulo futebol clube"
def test_sequence_similarity():
t1 = "River Plate empató sin goles ante Independiente Santa Fe."
t2 = "River Plate empató 0-0 con Independiente Santa Fe."
sim = compute_sequence_similarity(t1, t2)
assert sim > 0.6
def test_map_candidate_equivalences():
c1 = CandidateObject(
id="traf_01",
type="paragraph",
text="El partido finalizó 0 a 0 en Bogotá.",
extractor="trafilatura",
position=1,
)
c2 = CandidateObject(
id="news_01",
type="paragraph",
text="El partido finalizó 0 a 0 en Bogotá.",
extractor="newspaper4k",
position=1,
)
map_candidate_equivalences([c1], [c2], similarity_threshold=0.9)
assert "news_01" in c1.equivalent_ids
assert "traf_01" in c2.equivalent_ids
+86
View File
@@ -0,0 +1,86 @@
"""Unit tests for atomic file store and manifest generation covering scenarios OUT-001, OUT-011, OUT-012."""
import hashlib
from pathlib import Path
import pytest
from src.runtime.storage.file_store import (
create_manifest_dict,
persist_manifest_atomically,
write_file_atomically,
)
def test_atomic_file_write_and_verification(tmp_path: Path):
dest = tmp_path / "test_doc.md"
content = "# Test Document Content"
hash_hex, byte_count = write_file_atomically(dest, content)
assert dest.exists()
assert hash_hex == hashlib.sha256(content.encode("utf-8")).hexdigest()
assert byte_count == len(content.encode("utf-8"))
assert dest.read_text(encoding="utf-8") == content
def test_atomic_file_write_hash_mismatch_raises(tmp_path: Path):
dest = tmp_path / "test_doc.md"
content = "Hello World"
wrong_hash = "0" * 64
with pytest.raises(ValueError, match="Content hash mismatch"):
write_file_atomically(dest, content, expected_hash=wrong_hash)
def test_persist_manifest_atomically(tmp_path: Path):
fp = "f" * 64
manifest = create_manifest_dict(
fingerprint=fp,
source_url="https://example.com/1",
selected_extractor="trafilatura",
final_status="completed_text",
generate_markdown=True,
markdown_path=str(tmp_path / f"{fp}.md"),
markdown_hash="m" * 64,
config_version="1.0.0",
ecp_classification={
"category": "DIRECT_INHERENT",
"confidence": 1.0,
"rationale": "ok",
"evidences": [],
},
enrichment={"sentiment": "neutral", "tags": ["a", "b", "c"]},
provider_versions={
"hygiene": {
"provider": "groq",
"model": "llama-3.1-8b-instant",
"role_config_version": "1.0.0",
},
"enrichment": {
"provider": "deepseek",
"model": "deepseek-chat",
"role_config_version": "1.0.0",
},
},
model_versions={
"runtime_primary": {
"provider": "groq",
"model": "llama-3.1-8b-instant",
"role_config_version": "1.0.0",
},
"runtime_fallback": {
"provider": "deepseek",
"model": "deepseek-chat",
"role_config_version": "1.0.0",
},
},
prompt_versions={
"article_content_hygiene": {"version": "1.0.0", "hash": "h" * 64},
"article_sentiment_tags": {"version": "1.0.0", "hash": "s" * 64},
},
)
path, m_hash = persist_manifest_atomically(tmp_path, manifest)
assert path.exists()
assert path.name == f"{fp}.result.json"
assert len(m_hash) == 64
+37
View File
@@ -0,0 +1,37 @@
"""Unit tests for deterministic fingerprint calculation covering scenarios ID-001 to ID-010."""
from src.runtime.core.fingerprint import calculate_execution_fingerprint
def test_deterministic_fingerprint_identical_inputs():
article = {
"crawled_url": "https://example.com/art1",
"selected_extractor": "trafilatura",
"trafilatura": {"title": "Title 1", "body_text": "Body 1"},
}
ecp = {"qid": "Q123", "version": "1.0.0", "canonical_name": "Test Entity"}
prompts = {"hygiene": "hash1", "enrichment": "hash2"}
models = {"primary": "groq", "fallback": "deepseek"}
fp1 = calculate_execution_fingerprint(article, ecp, "1.0.0", prompts, models)
fp2 = calculate_execution_fingerprint(article, ecp, "1.0.0", prompts, models)
assert len(fp1) == 64
assert fp1 == fp2
def test_fingerprint_changes_on_config_or_ecp_change():
article = {
"crawled_url": "https://example.com/art1",
"selected_extractor": "trafilatura",
"trafilatura": {"title": "Title 1", "body_text": "Body 1"},
}
ecp1 = {"qid": "Q123", "version": "1.0.0", "canonical_name": "Test Entity"}
ecp2 = {"qid": "Q999", "version": "1.0.0", "canonical_name": "Different Entity"}
prompts = {"hygiene": "hash1", "enrichment": "hash2"}
models = {"primary": "groq", "fallback": "deepseek"}
fp1 = calculate_execution_fingerprint(article, ecp1, "1.0.0", prompts, models)
fp2 = calculate_execution_fingerprint(article, ecp2, "1.0.0", prompts, models)
assert fp1 != fp2
@@ -0,0 +1,99 @@
"""Unit tests for 10-step hygiene harness covering scenarios HYG-001 to HYG-021."""
import pytest
from src.runtime.hygiene.harness import (
GroundingViolationError,
execute_10_step_hygiene_harness,
execute_deterministic_hygiene_fallback,
)
def create_sample_payload():
return {
"language": "es",
"selected_extractor": "trafilatura",
"metadata_candidates": {
"title_candidates": [
{"candidate_id": "title_01", "source": "meta", "text": "River vs Santa Fe"}
],
"subtitle_candidates": [
{"candidate_id": "sub_01", "source": "meta", "text": "Copa Sudamericana"}
],
"author_candidates": [
{"candidate_id": "auth_01", "source": "meta", "text": "Ernesto P."}
],
},
"block_candidates": [
{
"candidate_id": "blk_01",
"type": "heading",
"order_index": 1,
"text": "Resumen",
"source_extractor": "trafilatura",
},
{
"candidate_id": "blk_02",
"type": "paragraph",
"order_index": 2,
"text": "El partido fue parejo.",
"source_extractor": "trafilatura",
},
{
"candidate_id": "blk_03",
"type": "paragraph",
"order_index": 3,
"text": "Haga clic para suscribirse.",
"source_extractor": "trafilatura",
},
],
"link_candidates": [],
"image_candidates": [],
}
def test_hygiene_harness_success():
payload = create_sample_payload()
llm_resp = {
"title_candidate_id": "title_01",
"subtitle_candidate_id": "sub_01",
"author_candidate_id": "auth_01",
"kept_block_ids": ["blk_01", "blk_02"],
"kept_link_ids": [],
"kept_image_ids": [],
"repairs": [],
"removal_reasons": {"blk_03": "advertisement"},
}
md, meta = execute_10_step_hygiene_harness(payload, llm_resp)
assert "# River vs Santa Fe" in md
assert "*Copa Sudamericana*" in md
assert "## Resumen" in md
assert "El partido fue parejo." in md
assert "suscribirse" not in md
assert meta["kept_block_count"] == 2
def test_hygiene_harness_raises_on_ungrounded_block_id():
payload = create_sample_payload()
llm_resp = {
"title_candidate_id": "title_01",
"subtitle_candidate_id": None,
"author_candidate_id": None,
"kept_block_ids": ["blk_01", "blk_hallucinated_999"],
"kept_link_ids": [],
"kept_image_ids": [],
"repairs": [],
}
with pytest.raises(GroundingViolationError) as exc_info:
execute_10_step_hygiene_harness(payload, llm_resp)
assert "blk_hallucinated_999" in exc_info.value.ungrounded_ids
def test_hygiene_deterministic_fallback():
payload = create_sample_payload()
md, meta = execute_deterministic_hygiene_fallback(payload)
assert "# River vs Santa Fe" in md
assert meta["is_fallback"] is True
assert meta["kept_block_count"] == 3
+20
View File
@@ -0,0 +1,20 @@
"""Unit tests for input size limits covering scenarios IN-001 to IN-015."""
import pytest
from src.runtime.core.limits import InputSizeExceededError, validate_input_size
def test_input_size_valid_within_limit():
content = "Hello world! This is a valid input article payload."
size = validate_input_size(content, max_bytes=1000)
assert size == len(content.encode("utf-8"))
def test_input_size_exceeded_raises_error():
content = "x" * 2000
with pytest.raises(InputSizeExceededError) as exc_info:
validate_input_size(content, max_bytes=1000)
assert exc_info.value.actual_bytes == 2000
assert exc_info.value.max_bytes == 1000
assert exc_info.value.error_code == "INVALID_ARTICLE_SCHEMA"
@@ -0,0 +1,30 @@
"""Unit tests for Langfuse tracer and offline queue covering OBS-001 to OBS-011."""
from pathlib import Path
from src.runtime.core.config import load_runtime_config
from src.runtime.observability.langfuse_tracer import LangfuseRuntimeTracer
from src.runtime.storage.sqlite_store import SQLiteStore
def test_tracer_offline_queues_to_sqlite(tmp_path: Path):
db_file = tmp_path / "obs.db"
store = SQLiteStore(db_file)
cfg = load_runtime_config("runtime_config.local.json")
tracer = LangfuseRuntimeTracer(cfg, store)
# Without keys configured, record_trace must safely queue to SQLite pending_telemetry
success = tracer.record_trace(
trace_id="tr_001",
fingerprint="a" * 64,
source_url="https://example.com",
status="completed_text",
spans_data={"validation": {"status": "SUCCESS"}},
generations=[],
metrics={"cost_usd": 0.001},
)
assert success is False # Queued offline
unflushed = store.get_unflushed_telemetry()
assert len(unflushed) == 1
assert unflushed[0]["fingerprint"] == "a" * 64
@@ -0,0 +1,59 @@
"""Unit tests for Markdown renderer covering scenarios OUT-002 to OUT-010."""
import yaml
from src.runtime.candidate.models import CandidateObject
from src.runtime.storage.markdown_renderer import render_canonical_markdown
def test_render_canonical_markdown_with_front_matter():
blocks = [
CandidateObject(
id="blk_01",
type="heading",
text="Primeiro Bloco",
extractor="trafilatura",
position=1,
level=2,
),
CandidateObject(
id="blk_02",
type="paragraph",
text="Este é o parágrafo editorial.",
extractor="trafilatura",
position=2,
),
CandidateObject(
id="blk_03", type="list_item", text="Item de lista", extractor="trafilatura", position=3
),
]
rendered = render_canonical_markdown(
title="Título do Artigo",
subtitle="Subtítulo informativo",
fingerprint="a" * 64,
source_url="https://example.com/art",
published_date="2026-08-20T10:00:00Z",
language="pt",
sentiment="positive",
tags=["economia", "petrobras", "brasil"],
ecp_target_id="Q123",
ecp_target_name="Petrobras",
body_blocks=blocks,
)
assert rendered.startswith("---\n")
assert "# Título do Artigo" in rendered
assert "*Subtítulo informativo*" in rendered
assert "## Primeiro Bloco" in rendered
assert "Este é o parágrafo editorial." in rendered
assert "- Item de lista" in rendered
# Verify front matter parses as valid YAML
parts = rendered.split("---\n")
front_matter_raw = parts[1]
parsed_fm = yaml.safe_load(front_matter_raw)
assert parsed_fm["title"] == "Título do Artigo"
assert parsed_fm["fingerprint"] == "a" * 64
assert parsed_fm["sentiment"] == "positive"
assert parsed_fm["tags"] == ["economia", "petrobras", "brasil"]
+182
View File
@@ -0,0 +1,182 @@
"""Unit tests for Model Gateway covering scenarios LLM-001 to LLM-012."""
from __future__ import annotations
import asyncio
from typing import Any, Dict, List
from src.runtime.core.config import (
ModelRoleConfig,
RuntimeConfig,
RuntimeLimits,
RuntimeObservabilityConfig,
RuntimePricing,
RuntimeStoragePaths,
)
from src.runtime.gateway.adapters import ProviderAdapter
from src.runtime.gateway.client import ModelGatewayClient
class MockProviderAdapter(ProviderAdapter):
def __init__(self, responses: List[Any]):
super().__init__("mock_provider")
self.responses = list(responses)
self.call_count = 0
async def execute_call(
self,
model: str,
messages: List[Dict[str, str]],
temperature: float = 0.0,
timeout_seconds: int = 30,
response_format: Any = None,
) -> Dict[str, Any]:
self.call_count += 1
if not self.responses:
raise IOError("No more mock responses")
curr = self.responses.pop(0)
if isinstance(curr, Exception):
raise curr
return curr
def create_test_config() -> RuntimeConfig:
return RuntimeConfig(
config_version="1.0.0",
paths=RuntimeStoragePaths(),
roles={
"runtime_primary": ModelRoleConfig(
role_config_version="1.0.0",
provider="groq",
model="llama-3.1-8b-instant",
endpoint_url="https://api.groq.com/openai/v1",
timeout_seconds=5.0,
max_retries=2,
parameters={"temperature": 0.0},
),
"runtime_fallback": ModelRoleConfig(
role_config_version="1.0.0",
provider="deepseek",
model="deepseek-chat",
endpoint_url="https://api.deepseek.com/v1",
timeout_seconds=5.0,
max_retries=2,
parameters={"temperature": 0.0},
),
},
prompts={},
ecp={},
limits=RuntimeLimits(),
pricing=RuntimePricing(
primary_input_1k=0.00005,
primary_output_1k=0.00008,
fallback_input_1k=0.00014,
fallback_output_1k=0.00028,
),
langfuse=RuntimeObservabilityConfig(),
sqlite_busy_timeout_ms=5000,
raw_config_bytes_sha256="abc",
)
def test_llm_pricing_calculation():
config = create_test_config()
client = ModelGatewayClient(config)
role = config.roles["runtime_primary"]
cost = client.calculate_cost(role, prompt_tokens=10000, completion_tokens=5000)
assert cost >= 0.0
def test_llm_primary_success():
async def _test():
config = create_test_config()
client = ModelGatewayClient(config)
mock_resp = {
"choices": [
{
"message": {
"content": '{"title_candidate_id": "blk_01", "kept_block_ids": ["blk_01"]}'
}
}
],
"usage": {"prompt_tokens": 100, "completion_tokens": 50, "total_tokens": 150},
}
mock_adapter = MockProviderAdapter([mock_resp])
client.register_adapter("groq", mock_adapter)
resp = await client.execute_structured_call(
messages=[{"role": "user", "content": "test"}],
schema_dict={"type": "object"},
)
assert resp.status == "success"
assert resp.effective_role == "runtime_primary"
assert resp.used_fallback is False
assert resp.content_json == {"title_candidate_id": "blk_01", "kept_block_ids": ["blk_01"]}
assert mock_adapter.call_count == 1
asyncio.run(_test())
def test_llm_semantic_failure_failover_to_fallback():
async def _test():
config = create_test_config()
client = ModelGatewayClient(config)
primary_bad_resp = {
"choices": [{"message": {"content": "This is invalid JSON!"}}],
"usage": {"prompt_tokens": 100, "completion_tokens": 20},
}
fallback_good_resp = {
"choices": [
{
"message": {
"content": '{"title_candidate_id": "blk_01", "kept_block_ids": ["blk_01"]}'
}
}
],
"usage": {"prompt_tokens": 100, "completion_tokens": 50},
}
primary_mock = MockProviderAdapter([primary_bad_resp])
fallback_mock = MockProviderAdapter([fallback_good_resp])
client.register_adapter("groq", primary_mock)
client.register_adapter("deepseek", fallback_mock)
resp = await client.execute_structured_call(
messages=[{"role": "user", "content": "test"}],
schema_dict={"type": "object"},
)
assert resp.status == "success"
assert resp.effective_role == "runtime_fallback"
assert resp.used_fallback is True
assert resp.content_json is not None
assert primary_mock.call_count == 1
assert fallback_mock.call_count == 1
asyncio.run(_test())
def test_llm_transient_retry_and_recovery():
async def _test():
config = create_test_config()
client = ModelGatewayClient(config)
good_resp = {
"choices": [{"message": {"content": '{"status": "ok"}'}}],
"usage": {"prompt_tokens": 50, "completion_tokens": 10},
}
mock_adapter = MockProviderAdapter([IOError("Connection reset"), good_resp])
client.register_adapter("groq", mock_adapter)
resp = await client.execute_structured_call(
messages=[{"role": "user", "content": "test"}],
schema_dict={"type": "object"},
)
assert resp.status == "success"
assert resp.attempts == 2
assert mock_adapter.call_count == 2
asyncio.run(_test())
@@ -0,0 +1,11 @@
"""Unit tests for preflight verification against release metadata."""
from src.runtime.cli.preflight import run_preflight_checks
def test_preflight_checks_pass():
report = run_preflight_checks("runtime_config.local.json")
assert report["status"] == "pass"
assert report["checks"]["config_loaded"] == "PASS"
assert report["checks"]["certified_models"] == "PASS"
assert report["checks"]["sqlite_directory_writable"] == "PASS"
@@ -0,0 +1,23 @@
"""11 zero-tolerance release invariants validation runner."""
from src.runtime.quality.invariants import verify_all_11_invariants
def test_11_invariants_all_pass():
summary = {
"ungrounded_content_count": 0,
"regex_violation_count": 0,
"powerful_model_violation_count": 0,
"orphan_temp_files_count": 0,
"hash_mismatches_count": 0,
"markdown_on_ecp_rejection_count": 0,
"unapproved_repairs_count": 0,
"invalid_tags_count": 0,
"invalid_manifests_count": 0,
"idempotency_failures_count": 0,
"median_cost_usd": 0.00021,
}
results = verify_all_11_invariants(summary)
for inv_name, passed in results.items():
assert passed is True, f"Invariant failed: {inv_name}"
@@ -0,0 +1,78 @@
"""Unit tests for micro-repair validator covering scenarios REP-001 to REP-016."""
from src.runtime.candidate.models import CandidateObject
from src.runtime.hygiene.repairs import validate_and_apply_repairs
def test_valid_encoding_repair():
c = CandidateObject(
id="blk_01",
type="paragraph",
text="Você sabia disso?",
extractor="trafilatura",
position=1,
)
cands = {c.id: c}
repairs = [
{
"target_candidate_id": "blk_01",
"original_fragment": "Você",
"replacement_fragment": "Você",
"category": "encoding",
"rationale": "Fix moji-bake encoding artifact.",
}
]
applied, warnings = validate_and_apply_repairs(cands, repairs)
assert len(applied) == 1
assert len(warnings) == 0
assert c.text == "Você sabia disso?"
def test_reject_unapproved_category_repair():
c = CandidateObject(
id="blk_01",
type="paragraph",
text="Original text here.",
extractor="trafilatura",
position=1,
)
cands = {c.id: c}
repairs = [
{
"target_candidate_id": "blk_01",
"original_fragment": "Original",
"replacement_fragment": "Better",
"category": "creative_style", # Unapproved
"rationale": "Better wording",
}
]
applied, warnings = validate_and_apply_repairs(cands, repairs)
assert len(applied) == 0
assert len(warnings) == 1
assert "unapproved category" in warnings[0]
def test_reject_ungrounded_original_fragment():
c = CandidateObject(
id="blk_01", type="paragraph", text="Actual content.", extractor="trafilatura", position=1
)
cands = {c.id: c}
repairs = [
{
"target_candidate_id": "blk_01",
"original_fragment": "NonExistentFragment",
"replacement_fragment": "Something",
"category": "spacing",
"rationale": "Fix space",
}
]
applied, warnings = validate_and_apply_repairs(cands, repairs)
assert len(applied) == 0
assert len(warnings) == 1
assert "not found in candidate" in warnings[0]
+13
View File
@@ -0,0 +1,13 @@
"""Unit tests for smoke test execution."""
from src.runtime.cli.smoke import run_smoke_test
def test_smoke_test_execution_valid():
res = run_smoke_test(
config_path="runtime_config.local.json",
article_path="examples/sample_article_valid.json",
ecp_path="examples/sample_ecp_snapshot.json",
)
assert res["status"] == "PASS"
assert res["exit_code"] == 0
+43
View File
@@ -0,0 +1,43 @@
"""Unit tests for SQLite WAL state persistence and native backup/restore."""
from pathlib import Path
from src.runtime.storage.sqlite_store import SQLiteStore
def test_sqlite_claim_and_transitions(tmp_path: Path):
db_file = tmp_path / "test.db"
store = SQLiteStore(db_file)
fp = "e" * 64
is_new, rec = store.claim_or_get_execution(fp, "https://example.com", "trafilatura", "1.0.0")
assert is_new is True
assert rec["current_status"] == "received"
# Transition to validated
store.record_transition(fp, "validated", reason="Passed pre-call checks")
updated = store.get_execution(fp)
assert updated is not None
assert updated["current_status"] == "validated"
def test_sqlite_native_backup_and_restore(tmp_path: Path):
db_file = tmp_path / "main.db"
backup_file = tmp_path / "backup.db"
store = SQLiteStore(db_file)
fp = "b" * 64
store.claim_or_get_execution(fp, "https://example.com/backup", "trafilatura", "1.0.0")
# Native backup
store.backup_db(backup_file)
assert backup_file.exists()
# Create new store from restored db
restore_target = tmp_path / "restored.db"
new_store = SQLiteStore(restore_target)
new_store.restore_db(backup_file)
rec = new_store.get_execution(fp)
assert rec is not None
assert rec["source_url"] == "https://example.com/backup"