"""Helper script to extract the 20 reference articles into individual unit files for evals/reference_20/.""" import json from pathlib import Path def main() -> None: source_file = Path("out/river_plate_extracted.json") ecp_source = Path("examples/ecp_club_atletico_river_plate.json") target_dir = Path("evals/reference_20") target_dir.mkdir(parents=True, exist_ok=True) if not source_file.exists(): print(f"Source file {source_file} does not exist.") return data = json.loads(source_file.read_text(encoding="utf-8")) articles = data.get("articles", []) print(f"Found {len(articles)} articles in {source_file}") ecp_data = json.loads(ecp_source.read_text(encoding="utf-8")) if ecp_source.exists() else {} (target_dir / "ecp_snapshot.json").write_text( json.dumps(ecp_data, indent=2, ensure_ascii=False), encoding="utf-8" ) for idx, article in enumerate(articles, start=1): # Ensure selected_extractor is populated if missing or null in raw crawl if not article.get("selected_extractor"): if article.get("trafilatura") and article["trafilatura"].get("body_text"): article["selected_extractor"] = "trafilatura" elif article.get("newspaper4k") and article["newspaper4k"].get("body_text"): article["selected_extractor"] = "newspaper4k" elif article.get("readability") and article["readability"].get("body_text"): article["selected_extractor"] = "readability" else: article["selected_extractor"] = "trafilatura" out_path = target_dir / f"article_{idx:02d}.json" out_path.write_text( json.dumps(article, indent=2, ensure_ascii=False), encoding="utf-8" ) print(f"Wrote {out_path}") print("Reference 20 dataset creation completed.") if __name__ == "__main__": main()