"""Phase 1 crawl: fetch the site pages named in the brief and save their visible text to corpus/."""
import re, html, urllib.request, json, datetime, os
BASE = "https://pistomechanics.org"
def get(u):
    return urllib.request.urlopen(urllib.request.Request(u, headers={"User-Agent": "pm-forecast-crawl"}), timeout=30).read().decode("utf-8")
def text(h):
    h = re.sub(r"<!--.*?-->", "", h, flags=re.S)
    h = re.sub(r"<(script|style|nav|footer|noscript|svg)\b.*?</\1>", "", h, flags=re.S | re.I)
    h = re.sub(r"<(br|/p|/h[1-6]|/li|/div|/section|/blockquote|/summary|/tr)[^>]*>", "\n", h, flags=re.I)
    t = html.unescape(re.sub(r"<[^>]+>", "", h))
    return re.sub(r"\n\s*\n+", "\n\n", re.sub(r"[ \t]+", " ", t)).strip()
idx = get(BASE + "/essays/")
essays = sorted({m for m in re.findall(r'href="(/essays/[^"#?]+)"', idx) if m.rstrip("/") != "/essays"})
sm = re.findall(r"<loc>https://pistomechanics.org(/essays/[^<]+)</loc>", get(BASE + "/sitemap.xml"))
paths = ["/", "/belief", "/lab", "/framework", "/history", "/essays/"] + sorted(set(essays) | set(sm))
manifest = []
for p in paths:
    try:
        t = text(get(BASE + p))
        name = (p.strip("/").replace("/", "__") or "home") + ".txt"
        open(f"corpus/{name}", "w").write(f"SOURCE: {BASE}{p}\n\n{t}\n")
        manifest.append({"url": BASE + p, "file": name, "chars": len(t), "status": "ok"})
    except Exception as e:
        manifest.append({"url": BASE + p, "file": None, "status": f"error: {e}"})
json.dump({"crawled": datetime.datetime.utcnow().isoformat() + "Z", "pages": manifest}, open("corpus/manifest.json", "w"), indent=1)
print(len(manifest), "pages;", sum(m["status"] == "ok" for m in manifest), "ok")
