Files

51 lines
1.9 KiB
Python

"""Helper script to extract the 20 reference articles into individual unit files for evals/reference_20/."""
import json
from pathlib import Path
def main() -> None:
source_file = Path("out/river_plate_extracted.json")
ecp_source = Path("examples/ecp_club_atletico_river_plate.json")
target_dir = Path("evals/reference_20")
target_dir.mkdir(parents=True, exist_ok=True)
if not source_file.exists():
print(f"Source file {source_file} does not exist.")
return
data = json.loads(source_file.read_text(encoding="utf-8"))
articles = data.get("articles", [])
print(f"Found {len(articles)} articles in {source_file}")
ecp_data = json.loads(ecp_source.read_text(encoding="utf-8")) if ecp_source.exists() else {}
(target_dir / "ecp_snapshot.json").write_text(
json.dumps(ecp_data, indent=2, ensure_ascii=False),
encoding="utf-8"
)
for idx, article in enumerate(articles, start=1):
# Ensure selected_extractor is populated if missing or null in raw crawl
if not article.get("selected_extractor"):
if article.get("trafilatura") and article["trafilatura"].get("body_text"):
article["selected_extractor"] = "trafilatura"
elif article.get("newspaper4k") and article["newspaper4k"].get("body_text"):
article["selected_extractor"] = "newspaper4k"
elif article.get("readability") and article["readability"].get("body_text"):
article["selected_extractor"] = "readability"
else:
article["selected_extractor"] = "trafilatura"
out_path = target_dir / f"article_{idx:02d}.json"
out_path.write_text(
json.dumps(article, indent=2, ensure_ascii=False),
encoding="utf-8"
)
print(f"Wrote {out_path}")
print("Reference 20 dataset creation completed.")
if __name__ == "__main__":
main()