"""Phase 3: build data/studies.csv and data/excluded.csv from the screened pools, the automated
exclusion check (data/harvest/exclusion_auto_*.json) and the manual review (data/harvest/review_decisions.json).
Refuses to run if any automated REVIEW / EXCLUDE decision has not been reviewed.
usage: python3 scripts/build_studies.py"""
import csv, json, sys
CUTOFF = "2026-07"  # forecaster training cutoff is June 2026: anything whose primary completion is before July 2026 is out
rev = json.load(open("data/harvest/review_decisions.json"))
pilot = {c["id"]: c["pilot_role"] for c in json.load(open("data/harvest/pilot_candidates.json"))}
studies, excluded, missing = [], [], []
for src in ("ct", "osf", "rr"):
    pool = json.load(open(f"data/harvest/pool_{src}.json"))
    auto = {o["id"]: o for o in json.load(open(f"data/harvest/exclusion_auto_{src}.json"))}
    for c in pool:
        sid = c["id"]; a = auto.get(sid)
        if a is None:
            missing.append(sid); continue
        s = c.get("screen", {}) if src != "rr" else {
            "primary_hypothesis": c["hypotheses"], "hypothesis_source": c["hypotheses_form"], "primary_outcome": c["primary_outcome"],
            "design_short": c["design"], "topic": ""}
        flags = []
        pc = c.get("primary_completion") or ""
        if a.get("sources_skipped"): flags.append("openalex not searched")
        if src == "osf" and s.get("design_short", "").startswith("N=1"): flags.append("single-participant design")
        if src == "rr":
            d = c.get("pci_listing_date") or c.get("ipa_date") or ""
            flags.append(f"completion date unknown (IPA {d})")
        decision, reason, url = "KEEP", a["decision"], ""
        if sid in rev:
            decision, reason, url = rev[sid]["decision"], "reviewed: " + rev[sid]["note"], rev[sid]["evidence_url"]
        elif not a["decision"].startswith("KEEP"):
            missing.append(sid); continue
        if src == "ct" and pc and pc < CUTOFF:
            decision, reason = "EXCLUDE", f"primary completion {pc} is before the forecaster's June 2026 training cutoff"
        row = {"study_id": sid, "source": {"ct": "ClinicalTrials.gov", "osf": "OSF Registries", "rr": c.get("source")}[src],
               "registry_url": c.get("url") if src != "rr" else (c.get("ipa_registration_url") or c.get("url")),
               "title": c["title"], "topic": s.get("topic", ""), "design_short": s.get("design_short", ""),
               "primary_hypothesis": s.get("primary_hypothesis", ""), "hypothesis_source": s.get("hypothesis_source", ""),
               "primary_outcome": s.get("primary_outcome", ""), "registry_status": a["registry"].get("status", ""),
               "primary_completion": pc, "results_check_on": a["checked_on"], "results_check": reason,
               "evidence_url": url, "flags": "; ".join(flags), "pilot": pilot.get(sid, "")}
        if decision == "KEEP":
            studies.append(row)
        else:
            excluded.append({"study_id": sid, "source": row["source"], "registry_url": row["registry_url"], "title": row["title"],
                             "reason": reason, "evidence_url": url, "decided_on": a["checked_on"]})
if missing:
    sys.exit(f"not ready: {len(missing)} studies lack a check or a review decision, e.g. {missing[:5]}")
for path, rows in (("data/studies.csv", studies), ("data/excluded.csv", excluded)):
    cols = list(rows[0].keys()) if rows else ["study_id", "source", "registry_url", "title", "reason", "evidence_url", "decided_on"]
    w = csv.DictWriter(open(path, "w", newline=""), fieldnames=cols); w.writeheader(); w.writerows(rows)
print(len(studies), "studies,", len(excluded), "excluded")
