feat(runtime): implement single-article consolidation runtime and modularize codebase

This commit is contained in:
2026-08-24 00:14:07 -03:00
parent e1e0be1353
commit 23de7d8fe7
176 changed files with 266754 additions and 10179 deletions
+48
View File
@@ -0,0 +1,48 @@
"""Release packaging script generating src/core/release-metadata.json."""
from __future__ import annotations
import hashlib
import json
from pathlib import Path
def build_release_metadata() -> Path:
cfg_file = Path("runtime_config.local.json")
cfg_bytes = cfg_file.read_bytes()
cfg_sha = hashlib.sha256(cfg_bytes).hexdigest()
metadata = {
"release_version": "1.0.0",
"runtime_config_sha256": cfg_sha,
"certified_models": [
"llama-3.1-8b-instant",
"llama-3.3-70b-versatile",
"deepseek-chat",
"deepseek-reasoner",
"gpt-4o-mini",
"claude-3-haiku-20240307",
],
"schemas": {
"article_input": "1.0.0",
"ecp_snapshot": "1.0.0",
"candidates_payload": "1.0.0",
"hygiene_response": "1.0.0",
"repair_operations": "1.0.0",
"enrichment_response": "1.0.0",
"manifest_output": "1.0.0",
},
"prompts": {
"article_content_hygiene": "1.0.0",
"article_sentiment_tags": "1.0.0",
},
}
out_file = Path("src/runtime/core/release-metadata.json")
out_file.write_text(json.dumps(metadata, indent=2), encoding="utf-8")
return out_file
if __name__ == "__main__":
p = build_release_metadata()
print(f"Generated release metadata at {p}")
+25
View File
@@ -0,0 +1,25 @@
"""CI multi-tier trigger runner and verification script."""
from __future__ import annotations
import subprocess
import sys
def run_ci_checks() -> int:
print("=== STEP 1: Static Zero Regex and Schema Policy Check ===")
res_static = subprocess.run([sys.executable, "tests/scripts/check_zero_regex.py"])
if res_static.returncode != 0:
return 1
print("\n=== STEP 2: Pytest Automated Suites ===")
res_pytest = subprocess.run([sys.executable, "-m", "pytest", "tests/", "-v"])
if res_pytest.returncode != 0:
return 1
print("\n=== ALL CI QUALITY GATES PASSED SUCCESSFULLY ===")
return 0
if __name__ == "__main__":
sys.exit(run_ci_checks())
+50
View File
@@ -0,0 +1,50 @@
"""Helper script to extract the 20 reference articles into individual unit files for evals/reference_20/."""
import json
from pathlib import Path
def main() -> None:
source_file = Path("out/river_plate_extracted.json")
ecp_source = Path("examples/ecp_club_atletico_river_plate.json")
target_dir = Path("evals/reference_20")
target_dir.mkdir(parents=True, exist_ok=True)
if not source_file.exists():
print(f"Source file {source_file} does not exist.")
return
data = json.loads(source_file.read_text(encoding="utf-8"))
articles = data.get("articles", [])
print(f"Found {len(articles)} articles in {source_file}")
ecp_data = json.loads(ecp_source.read_text(encoding="utf-8")) if ecp_source.exists() else {}
(target_dir / "ecp_snapshot.json").write_text(
json.dumps(ecp_data, indent=2, ensure_ascii=False),
encoding="utf-8"
)
for idx, article in enumerate(articles, start=1):
# Ensure selected_extractor is populated if missing or null in raw crawl
if not article.get("selected_extractor"):
if article.get("trafilatura") and article["trafilatura"].get("body_text"):
article["selected_extractor"] = "trafilatura"
elif article.get("newspaper4k") and article["newspaper4k"].get("body_text"):
article["selected_extractor"] = "newspaper4k"
elif article.get("readability") and article["readability"].get("body_text"):
article["selected_extractor"] = "readability"
else:
article["selected_extractor"] = "trafilatura"
out_path = target_dir / f"article_{idx:02d}.json"
out_path.write_text(
json.dumps(article, indent=2, ensure_ascii=False),
encoding="utf-8"
)
print(f"Wrote {out_path}")
print("Reference 20 dataset creation completed.")
if __name__ == "__main__":
main()