feat(runtime): implement single-article consolidation runtime and modularize codebase
This commit is contained in:
@@ -0,0 +1,45 @@
|
||||
"""Contract tests for article-input.schema.json evaluated against all 20 reference units."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import jsonschema
|
||||
|
||||
from src.runtime.core.config import create_schema_registry, load_schema
|
||||
|
||||
|
||||
def test_article_input_schema_against_all_20_reference_units():
|
||||
schema = load_schema("article-input.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
ref_dir = Path("evals/reference_20")
|
||||
article_files = sorted(ref_dir.glob("article_*.json"))
|
||||
assert len(article_files) == 20, f"Expected 20 reference unit files, found {len(article_files)}"
|
||||
|
||||
for art_file in article_files:
|
||||
data = json.loads(art_file.read_text(encoding="utf-8"))
|
||||
errors = list(validator.iter_errors(data))
|
||||
assert len(errors) == 0, (
|
||||
f"Article {art_file.name} failed contract validation: {[e.message for e in errors]}"
|
||||
)
|
||||
|
||||
|
||||
def test_article_input_rejects_batch_wrapper():
|
||||
schema = load_schema("article-input.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
# Batch wrapper containing "articles" key must fail contract validation
|
||||
batch_data = {
|
||||
"articles": [
|
||||
{"source_url": "https://example.com/1"},
|
||||
{"source_url": "https://example.com/2"},
|
||||
]
|
||||
}
|
||||
errors = list(validator.iter_errors(batch_data))
|
||||
assert len(errors) > 0, (
|
||||
"Batch wrapper containing 'articles' key must be rejected by contract schema"
|
||||
)
|
||||
@@ -0,0 +1,26 @@
|
||||
"""Contract tests for candidates-payload.schema.json."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import jsonschema
|
||||
|
||||
from src.runtime.candidate.parser import build_candidates_payload
|
||||
from src.runtime.core.config import create_schema_registry, load_schema
|
||||
|
||||
|
||||
def test_candidates_payload_against_reference_articles():
|
||||
schema = load_schema("candidates-payload.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
ref_dir = Path("evals/reference_20")
|
||||
for art_file in ref_dir.glob("article_*.json"):
|
||||
article_data = json.loads(art_file.read_text(encoding="utf-8"))
|
||||
payload = build_candidates_payload(article_data)
|
||||
errors = list(validator.iter_errors(payload))
|
||||
assert len(errors) == 0, (
|
||||
f"Payload for {art_file.name} failed schema: {[e.message for e in errors]}"
|
||||
)
|
||||
@@ -0,0 +1,16 @@
|
||||
"""Contract parity tests checking that all schema contracts match version 1.0.0."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def test_all_contract_schemas_version_1_0_0():
|
||||
contracts_dir = Path("specs/006-article-consolidation-runtime/contracts")
|
||||
schema_files = list(contracts_dir.glob("*.schema.json"))
|
||||
assert len(schema_files) >= 5
|
||||
|
||||
for sf in schema_files:
|
||||
data = json.loads(sf.read_text(encoding="utf-8"))
|
||||
version = data.get("x-contract-version") or data.get("version")
|
||||
# Assert each schema declares version 1.0.0
|
||||
assert version == "1.0.0", f"Schema {sf.name} version is {version}, expected 1.0.0"
|
||||
@@ -0,0 +1,41 @@
|
||||
"""Contract tests for ecp-snapshot.schema.json and local referencing.Registry resolution."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from src.runtime.ecp.adapter import validate_ecp_snapshot
|
||||
|
||||
|
||||
def test_ecp_snapshot_schema_valid():
|
||||
sample_ecp = {
|
||||
"target_entity_id": "Q12345",
|
||||
"target_name": "Club Atlético River Plate",
|
||||
"aliases": ["River", "El Millonario", "CARP"],
|
||||
"domain": "sports",
|
||||
"anchors": ["Monumental", "Buenos Aires", "Copa Libertadores"],
|
||||
"negative_anchors": ["River Plate Uruguay"],
|
||||
"graph_version": "1.0.0",
|
||||
"related_entities": [
|
||||
{
|
||||
"entity_id": "Q54321",
|
||||
"name": "Boca Juniors",
|
||||
"relation_type": "rival",
|
||||
"weight": 0.9,
|
||||
"aliases": ["Xeneize"],
|
||||
"scope": "derby",
|
||||
"confidence": 1.0,
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
validate_ecp_snapshot(sample_ecp)
|
||||
|
||||
|
||||
def test_ecp_snapshot_invalid_schema():
|
||||
invalid_ecp = {
|
||||
"target_name": "Missing target entity id",
|
||||
"domain": "sports",
|
||||
}
|
||||
with pytest.raises(ValueError, match="ECP Snapshot schema validation failed"):
|
||||
validate_ecp_snapshot(invalid_ecp)
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Contract tests for enrichment-response.schema.json."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import jsonschema
|
||||
|
||||
from src.runtime.core.config import create_schema_registry, load_schema
|
||||
|
||||
|
||||
def test_enrichment_response_schema_valid():
|
||||
schema = load_schema("enrichment-response.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
valid_response = {
|
||||
"sentiment": "positive",
|
||||
"tags": ["river plate", "futebol argentino", "copa sudamericana"],
|
||||
"evidence_candidate_ids": ["trafilatura_blk_001", "trafilatura_blk_002"],
|
||||
}
|
||||
|
||||
errors = list(validator.iter_errors(valid_response))
|
||||
assert len(errors) == 0, f"Schema errors: {[e.message for e in errors]}"
|
||||
|
||||
|
||||
def test_enrichment_response_schema_invalid_bounds():
|
||||
schema = load_schema("enrichment-response.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
# Less than 3 tags
|
||||
invalid_response = {
|
||||
"sentiment": "neutral",
|
||||
"tags": ["only_one_tag"],
|
||||
"evidence_candidate_ids": ["blk_01"],
|
||||
}
|
||||
errors = list(validator.iter_errors(invalid_response))
|
||||
assert len(errors) > 0
|
||||
@@ -0,0 +1,48 @@
|
||||
"""Contract tests for hygiene-response.schema.json."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import jsonschema
|
||||
|
||||
from src.runtime.core.config import create_schema_registry, load_schema
|
||||
|
||||
|
||||
def test_hygiene_response_schema_valid():
|
||||
schema = load_schema("hygiene-response.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
valid_response = {
|
||||
"title_candidate_id": "title_meta",
|
||||
"subtitle_candidate_id": "subtitle_meta",
|
||||
"author_candidate_id": "author_trafilatura",
|
||||
"kept_block_ids": ["trafilatura_blk_001", "trafilatura_blk_002"],
|
||||
"kept_link_ids": [],
|
||||
"kept_image_ids": [],
|
||||
"repairs": [
|
||||
{
|
||||
"target_candidate_id": "trafilatura_blk_001",
|
||||
"original_fragment": "River Plate empató",
|
||||
"replacement_fragment": "River Plate empató",
|
||||
"category": "encoding",
|
||||
"rationale": "Fix moji-bake encoding artifact.",
|
||||
}
|
||||
],
|
||||
"removal_reasons": {"trafilatura_blk_003": "advertisement"},
|
||||
}
|
||||
|
||||
errors = list(validator.iter_errors(valid_response))
|
||||
assert len(errors) == 0, f"Schema errors: {[e.message for e in errors]}"
|
||||
|
||||
|
||||
def test_hygiene_response_schema_missing_required():
|
||||
schema = load_schema("hygiene-response.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
invalid_response = {
|
||||
"kept_block_ids": ["blk_01"]
|
||||
# Missing title_candidate_id, repairs, etc.
|
||||
}
|
||||
errors = list(validator.iter_errors(invalid_response))
|
||||
assert len(errors) > 0
|
||||
@@ -0,0 +1,121 @@
|
||||
"""Contract tests for manifest-output.schema.json."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import jsonschema
|
||||
|
||||
from src.runtime.core.config import create_schema_registry, load_schema
|
||||
from src.runtime.storage.file_store import create_manifest_dict
|
||||
|
||||
|
||||
def test_manifest_output_schema_completed_text_valid():
|
||||
schema = load_schema("manifest-output.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
valid_manifest = create_manifest_dict(
|
||||
fingerprint="a" * 64,
|
||||
source_url="https://example.com/article/1",
|
||||
selected_extractor="trafilatura",
|
||||
final_status="completed_text",
|
||||
generate_markdown=True,
|
||||
markdown_path="out/articles/" + "a" * 64 + ".md",
|
||||
markdown_hash="b" * 64,
|
||||
config_version="1.0.0",
|
||||
trace_id="trace_001",
|
||||
ecp_classification={
|
||||
"category": "DIRECT_INHERENT",
|
||||
"confidence": 0.95,
|
||||
"rationale": "High direct entity relevance.",
|
||||
"evidences": ["Direct entity mentioned."],
|
||||
},
|
||||
enrichment={
|
||||
"sentiment": "positive",
|
||||
"tags": ["river plate", "futebol", "argentina"],
|
||||
},
|
||||
provider_versions={
|
||||
"hygiene": {
|
||||
"provider": "groq",
|
||||
"model": "llama-3.1-8b-instant",
|
||||
"role_config_version": "1.0.0",
|
||||
},
|
||||
"enrichment": {
|
||||
"provider": "deepseek",
|
||||
"model": "deepseek-chat",
|
||||
"role_config_version": "1.0.0",
|
||||
},
|
||||
},
|
||||
model_versions={
|
||||
"runtime_primary": {
|
||||
"provider": "groq",
|
||||
"model": "llama-3.1-8b-instant",
|
||||
"role_config_version": "1.0.0",
|
||||
},
|
||||
"runtime_fallback": {
|
||||
"provider": "deepseek",
|
||||
"model": "deepseek-chat",
|
||||
"role_config_version": "1.0.0",
|
||||
},
|
||||
},
|
||||
prompt_versions={
|
||||
"article_content_hygiene": {"version": "1.0.0", "hash": "c" * 64},
|
||||
"article_sentiment_tags": {"version": "1.0.0", "hash": "d" * 64},
|
||||
},
|
||||
error_codes=[],
|
||||
)
|
||||
|
||||
errors = list(validator.iter_errors(valid_manifest))
|
||||
assert len(errors) == 0, f"Schema errors: {[e.message for e in errors]}"
|
||||
|
||||
|
||||
def test_manifest_output_schema_rejected_ecp_valid():
|
||||
schema = load_schema("manifest-output.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
rejected_manifest = create_manifest_dict(
|
||||
fingerprint="a" * 64,
|
||||
source_url="https://example.com/article/2",
|
||||
selected_extractor="newspaper4k",
|
||||
final_status="rejected_ecp",
|
||||
generate_markdown=False,
|
||||
markdown_path=None,
|
||||
markdown_hash=None,
|
||||
config_version="1.0.0",
|
||||
trace_id="trace_002",
|
||||
ecp_classification={
|
||||
"category": "TANGENTIAL",
|
||||
"confidence": 0.88,
|
||||
"rationale": "Only brief tangential reference.",
|
||||
"evidences": ["Brief reference."],
|
||||
},
|
||||
enrichment=None,
|
||||
provider_versions={
|
||||
"hygiene": {
|
||||
"provider": "groq",
|
||||
"model": "llama-3.1-8b-instant",
|
||||
"role_config_version": "1.0.0",
|
||||
},
|
||||
"enrichment": None,
|
||||
},
|
||||
model_versions={
|
||||
"runtime_primary": {
|
||||
"provider": "groq",
|
||||
"model": "llama-3.1-8b-instant",
|
||||
"role_config_version": "1.0.0",
|
||||
},
|
||||
"runtime_fallback": {
|
||||
"provider": "deepseek",
|
||||
"model": "deepseek-chat",
|
||||
"role_config_version": "1.0.0",
|
||||
},
|
||||
},
|
||||
prompt_versions={
|
||||
"article_content_hygiene": {"version": "1.0.0", "hash": "c" * 64},
|
||||
"article_sentiment_tags": {"version": "1.0.0", "hash": "d" * 64},
|
||||
},
|
||||
error_codes=["ECP_REJECTED"],
|
||||
)
|
||||
|
||||
errors = list(validator.iter_errors(rejected_manifest))
|
||||
assert len(errors) == 0, f"Schema errors: {[e.message for e in errors]}"
|
||||
@@ -0,0 +1,26 @@
|
||||
"""Contract tests for versioned prompts verifying 6-block sequence and parity."""
|
||||
|
||||
import hashlib
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def test_prompts_6_block_architecture():
|
||||
prompts_dir = Path("prompts")
|
||||
prompt_files = list(prompts_dir.glob("*.txt"))
|
||||
assert len(prompt_files) >= 2
|
||||
|
||||
for p_file in prompt_files:
|
||||
content = p_file.read_text(encoding="utf-8")
|
||||
assert "# BLOCK 1: SYSTEM ROLE & OBJECTIVE" in content
|
||||
assert "# BLOCK 2: TASK INSTRUCTIONS" in content
|
||||
assert "# BLOCK 3:" in content
|
||||
assert "# BLOCK 4: OUTPUT CONTRACT SPECIFICATION" in content
|
||||
assert "# BLOCK 5: QUALITY GUARDRAILS" in content
|
||||
assert "# BLOCK 6: INPUT DATA PAYLOAD" in content
|
||||
|
||||
|
||||
def test_prompts_sha256_calculation():
|
||||
prompts_dir = Path("prompts")
|
||||
for p_file in prompts_dir.glob("*.txt"):
|
||||
sha = hashlib.sha256(p_file.read_bytes()).hexdigest()
|
||||
assert len(sha) == 64
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Contract tests for repair-operations.schema.json."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import jsonschema
|
||||
|
||||
from src.runtime.core.config import create_schema_registry, load_schema
|
||||
|
||||
|
||||
def test_repair_operations_schema_valid():
|
||||
schema = load_schema("repair-operations.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
valid_repairs = [
|
||||
{
|
||||
"target_candidate_id": "blk_001",
|
||||
"original_fragment": "São Paulo F.C.",
|
||||
"replacement_fragment": "São Paulo FC",
|
||||
"category": "punctuation_corruption",
|
||||
"rationale": "Normalize acronym dots.",
|
||||
},
|
||||
{
|
||||
"target_candidate_id": "blk_002",
|
||||
"original_fragment": "artigo com espacos",
|
||||
"replacement_fragment": "artigo com espacos",
|
||||
"category": "spacing",
|
||||
"rationale": "Collapse multiple spaces.",
|
||||
},
|
||||
]
|
||||
|
||||
errors = list(validator.iter_errors(valid_repairs))
|
||||
assert len(errors) == 0, f"Schema errors: {[e.message for e in errors]}"
|
||||
|
||||
|
||||
def test_repair_operations_rejects_unapproved_category():
|
||||
schema = load_schema("repair-operations.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
invalid_repairs = [
|
||||
{
|
||||
"target_candidate_id": "blk_001",
|
||||
"original_fragment": "old",
|
||||
"replacement_fragment": "new",
|
||||
"category": "editorial_rephrasing", # Unapproved category
|
||||
"rationale": "Rewriting paragraph style.",
|
||||
}
|
||||
]
|
||||
errors = list(validator.iter_errors(invalid_repairs))
|
||||
assert len(errors) > 0
|
||||
@@ -0,0 +1,96 @@
|
||||
"""Contract tests for runtime-config.schema.json."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import jsonschema
|
||||
import pytest
|
||||
|
||||
from src.runtime.core.config import create_schema_registry, load_runtime_config, load_schema
|
||||
|
||||
|
||||
def test_runtime_config_schema_validation_valid():
|
||||
schema = load_schema("runtime-config.schema.json")
|
||||
registry = create_schema_registry()
|
||||
validator = jsonschema.Draft202012Validator(schema, registry=registry)
|
||||
|
||||
valid_config = {
|
||||
"config_version": "1.0.0",
|
||||
"paths": {"output_dir": "out/articles", "sqlite_db": "out/runtime.db"},
|
||||
"roles": {
|
||||
"runtime_primary": {
|
||||
"role_config_version": "1.0.0",
|
||||
"provider": "groq",
|
||||
"model": "llama-3.1-8b-instant",
|
||||
"endpoint_url": "https://api.groq.com/openai/v1",
|
||||
"timeout_seconds": 30,
|
||||
"max_retries": 3,
|
||||
"parameters": {"temperature": 0.0},
|
||||
"hygiene_prompt_version": "1.0.0",
|
||||
"hygiene_schema_version": "1.0.0",
|
||||
"enrichment_prompt_version": "1.0.0",
|
||||
"enrichment_schema_version": "1.0.0",
|
||||
},
|
||||
"runtime_fallback": {
|
||||
"role_config_version": "1.0.0",
|
||||
"provider": "deepseek",
|
||||
"model": "deepseek-chat",
|
||||
"endpoint_url": "https://api.deepseek.com/v1",
|
||||
"timeout_seconds": 30,
|
||||
"max_retries": 3,
|
||||
"parameters": {"temperature": 0.0},
|
||||
"hygiene_prompt_version": "1.0.0",
|
||||
"hygiene_schema_version": "1.0.0",
|
||||
"enrichment_prompt_version": "1.0.0",
|
||||
"enrichment_schema_version": "1.0.0",
|
||||
},
|
||||
},
|
||||
"prompts": {
|
||||
"article_content_hygiene": {
|
||||
"path": "prompts/article_content_hygiene.v1.txt",
|
||||
"version": "1.0.0",
|
||||
"hash": "0000000000000000000000000000000000000000000000000000000000000000",
|
||||
},
|
||||
"article_sentiment_tags": {
|
||||
"path": "prompts/article_sentiment_tags.v1.txt",
|
||||
"version": "1.0.0",
|
||||
"hash": "0000000000000000000000000000000000000000000000000000000000000000",
|
||||
},
|
||||
},
|
||||
"ecp": {
|
||||
"canonical_schema_reference": "specs/006-article-consolidation-runtime/contracts/ecp-snapshot.schema.json",
|
||||
"classifier_module": "src.classifier.InherenceClassifier",
|
||||
},
|
||||
"limits": {"max_input_bytes": 1048576, "context_strategy": "fail_before_provider"},
|
||||
"pricing": {
|
||||
"primary_input_1k": 0.00005,
|
||||
"primary_output_1k": 0.00008,
|
||||
"fallback_input_1k": 0.00014,
|
||||
"fallback_output_1k": 0.00028,
|
||||
},
|
||||
"langfuse": {"environment": "local", "trace_content_policy": "metadata_only"},
|
||||
"sqlite": {"busy_timeout_ms": 5000},
|
||||
}
|
||||
|
||||
errors = list(validator.iter_errors(valid_config))
|
||||
assert len(errors) == 0, f"Schema validation errors: {[e.message for e in errors]}"
|
||||
|
||||
|
||||
def test_runtime_config_fixture_loads_successfully():
|
||||
config = load_runtime_config("runtime_config.local.json")
|
||||
assert config.config_version == "1.0.0"
|
||||
assert "runtime_primary" in config.roles
|
||||
assert "runtime_fallback" in config.roles
|
||||
assert config.roles["runtime_primary"].model == "llama-3.1-8b-instant"
|
||||
|
||||
|
||||
def test_runtime_config_rejects_powerful_models(tmp_path: Path):
|
||||
valid_base = json.loads(Path("runtime_config.local.json").read_text(encoding="utf-8"))
|
||||
valid_base["roles"]["runtime_primary"]["model"] = "gpt-4o" # Forbidden powerful model
|
||||
cfg_file = tmp_path / "invalid_cfg.json"
|
||||
cfg_file.write_text(json.dumps(valid_base), encoding="utf-8")
|
||||
|
||||
with pytest.raises(ValueError, match="Forbidden powerful model"):
|
||||
load_runtime_config(cfg_file)
|
||||
Reference in New Issue
Block a user