import json
inp = "/var/www/html/peta/storage/app/propertylab-catalogue-match/review/r2_batch_05_in.jsonl"
cases = [json.loads(l) for l in open(inp) if l.strip()]
rows = {}
for ln in open("r2b05_verdicts_b.txt"):
    ln = ln.rstrip("\n")
    if not ln: continue
    parts = ln.split("|")
    assert len(parts) == 6, ln
    cid, verdict, cand, conf, reason, web = parts
    assert cid not in rows, "dup " + cid
    rows[cid] = dict(case=cid, verdict=verdict, cand=None if cand == "null" else int(cand),
                     confidence=conf, reason=reason, web=(web == "true"))
problems = []
out = []
for c in cases:
    cid = c["case"]
    if cid not in rows:
        problems.append("missing " + cid); continue
    r = rows[cid]
    ks = {cd["k"] for cd in c["catalogue_candidates"]}
    if r["verdict"] not in ("SAME", "PART_OF", "DIFFERENT", "UNSURE"): problems.append(cid + " bad verdict")
    if r["confidence"] not in ("high", "medium", "low"): problems.append(cid + " bad conf")
    if r["verdict"] in ("SAME", "PART_OF") and r["cand"] is None: problems.append(cid + " needs cand")
    if r["verdict"] == "DIFFERENT" and r["cand"] is not None: problems.append(cid + " DIFFERENT with cand")
    if r["cand"] is not None and r["cand"] not in ks: problems.append(cid + " cand not in candidates")
    nw = len(r["reason"].split())
    if nw > 20: problems.append(f"{cid} reason {nw} words")
    out.append(r)
extra = set(rows) - {c["case"] for c in cases}
if extra: problems.append("extra " + str(extra))
print("cases", len(cases), "rows", len(out))
print("\n".join(problems) if problems else "NO PROBLEMS")
json.dump(out, open("r2b05_final.json", "w"))
