"""Check a claims CSV: header, required fields, level/testable values, excerpt length, and that each
verbatim_excerpt occurs in the corpus file for its source_url (whitespace- and quote-normalised)."""
import csv, json, re, sys
man = {p["url"]: p["file"] for p in json.load(open("corpus/manifest.json"))["pages"]}
def norm(s):
    s = s.replace("’", "'").replace("‘", "'").replace("“", '"').replace("”", '"').replace(" ", " ")
    return re.sub(r"\s+", " ", s).strip().lower()
cache = {}
bad = 0; n = 0
for path in sys.argv[1:]:
    for r in csv.DictReader(open(path, newline="")):
        n += 1; errs = []
        f = man.get(r["source_url"].rstrip("/") if r["source_url"].rstrip("/") in man else r["source_url"])
        if not f: errs.append("unknown source_url")
        else:
            if f not in cache: cache[f] = norm(open("corpus/" + f).read())
            if norm(r["verbatim_excerpt"]) not in cache[f]: errs.append("excerpt not found verbatim")
        if len(r["verbatim_excerpt"].split()) > 25: errs.append("excerpt > 25 words")
        if r["level"] not in {"0", "1", "2", "3", "4"}: errs.append("bad level")
        if r["testable"] not in {"yes", "only with operational definition", "no"}: errs.append("bad testable")
        for k in ("claim_in_plain_terms", "operational_definition", "what_would_refute_it"):
            if not r[k].strip(): errs.append("empty " + k)
        if errs:
            bad += 1; print(path, r.get("local_id") or r.get("claim_id"), "; ".join(errs))
print(f"{n} rows, {bad} with errors")
