51 lines
1.9 KiB
Python
51 lines
1.9 KiB
Python
"""Helper script to extract the 20 reference articles into individual unit files for evals/reference_20/."""
|
|
|
|
import json
|
|
from pathlib import Path
|
|
|
|
|
|
def main() -> None:
|
|
source_file = Path("out/river_plate_extracted.json")
|
|
ecp_source = Path("examples/ecp_club_atletico_river_plate.json")
|
|
target_dir = Path("evals/reference_20")
|
|
target_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
if not source_file.exists():
|
|
print(f"Source file {source_file} does not exist.")
|
|
return
|
|
|
|
data = json.loads(source_file.read_text(encoding="utf-8"))
|
|
articles = data.get("articles", [])
|
|
print(f"Found {len(articles)} articles in {source_file}")
|
|
|
|
ecp_data = json.loads(ecp_source.read_text(encoding="utf-8")) if ecp_source.exists() else {}
|
|
(target_dir / "ecp_snapshot.json").write_text(
|
|
json.dumps(ecp_data, indent=2, ensure_ascii=False),
|
|
encoding="utf-8"
|
|
)
|
|
|
|
for idx, article in enumerate(articles, start=1):
|
|
# Ensure selected_extractor is populated if missing or null in raw crawl
|
|
if not article.get("selected_extractor"):
|
|
if article.get("trafilatura") and article["trafilatura"].get("body_text"):
|
|
article["selected_extractor"] = "trafilatura"
|
|
elif article.get("newspaper4k") and article["newspaper4k"].get("body_text"):
|
|
article["selected_extractor"] = "newspaper4k"
|
|
elif article.get("readability") and article["readability"].get("body_text"):
|
|
article["selected_extractor"] = "readability"
|
|
else:
|
|
article["selected_extractor"] = "trafilatura"
|
|
|
|
out_path = target_dir / f"article_{idx:02d}.json"
|
|
out_path.write_text(
|
|
json.dumps(article, indent=2, ensure_ascii=False),
|
|
encoding="utf-8"
|
|
)
|
|
print(f"Wrote {out_path}")
|
|
|
|
print("Reference 20 dataset creation completed.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|