import json, sys, collections
sys.path.insert(0, "/tmp/claude-1000/-var-www-html-peta/310517d2-8f28-442b-9907-0e3560add2cb/scratchpad")
from vrest04_decisions import D

IN = "/var/www/html/peta/storage/app/propertylab-catalogue-match/review/v_rest_04_in.jsonl"
OUT = "/var/www/html/peta/storage/app/propertylab-catalogue-match/review/v_rest_04_out.jsonl"

cases = [json.loads(l) for l in open(IN)]
errors = []
if len(cases) != len(D):
    errors.append(f"count mismatch: {len(cases)} input vs {len(D)} decisions")
long_reasons = []
for i, (c, d) in enumerate(zip(cases, D)):
    cid, verdict, cand, conf, rel, reason, web = d
    if c["case"] != cid:
        errors.append(f"#{i}: order mismatch {c['case']} vs {cid}")
    ks = {x["k"] for x in c["catalogue_candidates"]}
    if verdict not in ("SAME", "PART_OF", "DIFFERENT", "UNSURE"):
        errors.append(f"{cid}: bad verdict")
    if verdict == "DIFFERENT" and cand is not None:
        errors.append(f"{cid}: DIFFERENT with cand")
    if verdict in ("SAME", "PART_OF") and cand is None:
        errors.append(f"{cid}: {verdict} without cand")
    if cand is not None and cand not in ks:
        errors.append(f"{cid}: cand {cand} not in {ks}")
    if verdict == "PART_OF" and rel not in ("R_in_C", "C_in_R"):
        errors.append(f"{cid}: PART_OF bad relation")
    if verdict != "PART_OF" and rel is not None:
        errors.append(f"{cid}: relation on non-PART_OF")
    if not (0 <= conf <= 100):
        errors.append(f"{cid}: bad conf")
    if len(reason.split()) > 20:
        long_reasons.append((cid, len(reason.split()), reason))
ids = [d[0] for d in D]
dups = [k for k, v in collections.Counter(ids).items() if v > 1]
if dups:
    errors.append(f"duplicate ids: {dups}")
webn = sum(1 for d in D if d[6])
if webn > 15:
    errors.append(f"web cases {webn} > 15")
print("errors:", errors)
print("long reasons:", long_reasons)
if errors or long_reasons:
    sys.exit(1)
with open(OUT, "w") as f:
    for cid, verdict, cand, conf, rel, reason, web in D:
        f.write(json.dumps({"case": cid, "verdict": verdict, "cand": cand, "confidence_pct": conf,
                            "relation": rel, "reason": reason, "web": web}, ensure_ascii=False) + "\n")
cnt = collections.Counter(d[1] for d in D)
print("written", len(D), "lines")
print("verdicts:", dict(cnt))
print(">=90:", sum(1 for d in D if d[3] >= 90))
print(">=90 by verdict:", dict(collections.Counter(d[1] for d in D if d[3] >= 90)))
print("web cases:", webn)
