"""Phase 3 harvest: pull candidate studies from ClinicalTrials.gov API v2 and OSF registrations API.
Writes data/harvest/ctgov_raw.json and data/harvest/osf_raw.json, plus a log in data/harvest/log.md.
Eligibility pre-filter (the rest is screened later):
  CT.gov: status RECRUITING / ACTIVE_NOT_RECRUITING / COMPLETED, hasResults false,
          primary completion date >= 2026-07-01 (after the forecaster's June 2026 training cutoff).
  OSF:    registration created >= 2026-04-01, not withdrawn."""
import json, os, time, urllib.parse, urllib.request, datetime
os.makedirs("data/harvest", exist_ok=True)
TERMS = ["placebo effect", "placebo response", "open-label placebo", "nocebo", "expectancy", "expectation",
         "hypnosis", "hypnotherapy", "psilocybin", "MDMA", "mindset", "targeted memory reactivation",
         "sleep memory consolidation", "persuasion", "belief change"]
LOG = []
def get(url):
    for i in range(4):
        try:
            return json.load(urllib.request.urlopen(urllib.request.Request(url, headers={"User-Agent": "pm-forecast-harvest"}), timeout=60))
        except Exception as e:
            err = e; time.sleep(2 * (i + 1))
    raise err
def ctgov():
    out = {}
    for term in TERMS:
        q = {"query.term": f'"{term}"', "filter.overallStatus": "RECRUITING,ACTIVE_NOT_RECRUITING,COMPLETED",
             "filter.advanced": "AREA[PrimaryCompletionDate]RANGE[2026-07-01,MAX]", "pageSize": "200", "countTotal": "true"}
        tok, n = None, 0
        while True:
            if tok: q["pageToken"] = tok
            try: d = get("https://clinicaltrials.gov/api/v2/studies?" + urllib.parse.urlencode(q))
            except Exception as e:
                LOG.append(f"CT.gov term '{term}': FAILED {e}"); break
            for s in d.get("studies", []):
                if s.get("hasResults"): continue
                nct = s["protocolSection"]["identificationModule"]["nctId"]
                out.setdefault(nct, {"study": s, "terms": []})["terms"].append(term); n += 1
            tok = d.get("nextPageToken")
            if not tok: break
        LOG.append(f"CT.gov term '{term}': {n} hits without posted results")
    return out
def osf():
    out = {}
    for term in TERMS:
        url = "https://api.osf.io/v2/registrations/?" + urllib.parse.urlencode({
            "filter[title][icontains]": term, "filter[date_created][gte]": "2026-04-01", "page[size]": "100"})
        n = 0
        while url:
            try: d = get(url)
            except Exception as e:
                LOG.append(f"OSF term '{term}': FAILED {e}"); break
            for r in d.get("data", []):
                a = r["attributes"]
                if a.get("withdrawn"): continue
                out.setdefault(r["id"], {"reg": r, "terms": []})["terms"].append(term); n += 1
            url = d.get("links", {}).get("next")
        LOG.append(f"OSF term '{term}' (title contains, created >= 2026-04-01): {n} hits")
    return out
c = ctgov(); json.dump(c, open("data/harvest/ctgov_raw.json", "w"))
o = osf(); json.dump(o, open("data/harvest/osf_raw.json", "w"))
open("data/harvest/log.md", "w").write(f"# Harvest log {datetime.date.today()}\n\n" + "\n".join("- " + l for l in LOG) +
    f"\n\nUnique CT.gov: {len(c)}\nUnique OSF: {len(o)}\n")
print("\n".join(LOG)); print("CT.gov unique", len(c), "OSF unique", len(o))
