import json, random, os, re, collections, sys
sys.path.insert(0, '/tmp/claude-1000/-var-www-html-peta/310517d2-8f28-442b-9907-0e3560add2cb/scratchpad/match')
from normalize import GENERIC, variants, core, PREFIX
D = '/var/www/html/peta/storage/app/propertylab-catalogue-match'
rc = {r['scheme_id']: r for r in json.load(open(f'{D}/realtycheck_schemes.json'))}
cat = json.load(open(f'{D}/catalogue_my_projects.json'))
out = json.load(open(f'{D}/match_v3.json'))
key1 = json.load(open(f'{D}/review/_key.json'))
done = {k['scheme_id'] for k in key1.values()}

def f(x):
    try:
        v = float(x); return v if v > 0 else None
    except (TypeError, ValueError):
        return None
def main_key(name):
    m = [core(v) for k, v in variants(name) if k == 'main']
    return ' '.join(sorted(t for t in (m[0] if m else []) if t not in PREFIX))
def weak(tks):
    t = [x for x in tks if x not in PREFIX]
    return all(x.isdigit() or x in GENERIC or len(x) <= 2 for x in t)
catkey = collections.Counter(main_key(p['project_name']) for p in cat)
COMM_T = {'Shop/Shoplot', 'Office', 'SoHo', 'Factory'}
COMM_W = re.compile(r'\b(arked|plaza|business|commercial|biz|hub|kedai|shop|office|soho|sovo|sofo|industri|industrial|perindustrian|centre|center|square|galleria|mall|complex|kompleks|wisma|menara)\b', re.I)

picked = collections.Counter(); rows = []
for o in out:
    sid = o['scheme_id']
    if sid in done or not o['top']: continue
    r = rc[sid]; t = o['tier']; b = o['top'][0]; p = cat[b['i']]
    pr = (f(r['median_rm']) / f(p['price_median'])) if (f(r['median_rm']) and f(p['price_median'])) else None
    price_ok = pr is not None and 0.85 <= pr <= 1.15
    why = None
    if t == 'A_SAME':
        if weak(b['rc_variant'].split()) and main_key(r['display_name']) != main_key(p['project_name']): why = 'A: weak core, names differ'
        elif not r['district'] and catkey[main_key(r['display_name'])] >= 2 and not price_ok: why = 'A: no district, common name, no price support'
    elif t == 'E5_WEAK': why = 'E5'
    elif t == 'E1_DIFFERENT_PROPERTY_TYPE':
        if r['category'] in ('Landed', 'Condo/Apartment', 'Serviced Apartment', 'Flat'): why = 'E1 residential'
        elif any((set(x.strip() for x in (cat[c['i']]['property_type'] or '').split(',')) & COMM_T) or COMM_W.search(cat[c['i']]['project_name']) for c in o['top'][:4]):
            why = 'E1 commercial with a commercial candidate'
    elif t == 'E3_SAME_NAME_OTHER_PLACE':
        rs = (r['state'] or '').lower().replace('wp ', '').strip()
        same_state = (not rs) or rs in (p['state'] or '').lower()
        if (same_state and b['dist_m'] <= 60000) or (pr is not None and 0.9 <= pr <= 1.1 and same_state): why = 'E3 coords suspect'
    if why:
        picked[why] += 1; rows.append((o, why))
print(dict(picked), 'total', len(rows))

random.seed(7); random.shuffle(rows)
def cand(c, k):
    p = cat[c['i']]
    return {'k': k, 'name': p['project_name'], 'area': p['area'], 'state': p['state'], 'type': p['property_type'],
            'dist_m': c['dist_m'], 'psf_median': p['psf_median'], 'price_median': p['price_median'],
            'completion_year': p['completion_year'], 'sale_status': p['sale_status'], 'tenure': p['tenure']}
cases, key = [], {}
for n, (o, why) in enumerate(rows, 1):
    r = rc[o['scheme_id']]; top = o['top']
    main = [c for c in top if not c.get('twin')][:4]
    cands = []
    for k, c in enumerate(main):
        x = cand(c, k)
        if k == 0:
            tw = [cat[t['i']]['project_name'] for t in top if t.get('twin')]
            if tw: x['same_listing_also_as'] = tw
        cands.append(x)
    cid = f'd{n:04d}'
    cases.append({'case': cid, 'realtycheck': {'name': r['display_name'], 'category': r['category'], 'state': r['state'],
                  'district': r['district'], 'mukim': r['mukim'], 'coord_precision': r['precision'], 'psf': r['reported_psf'],
                  'median_price_rm': r['median_rm']}, 'catalogue_candidates': cands})
    key[cid] = {'scheme_id': o['scheme_id'], 'tier': o['tier'], 'why': why, 'audit': False, 'cand_i': [c['i'] for c in main]}
NB = 6
for b in range(NB):
    with open(f'{D}/review/r2_batch_{b+1:02d}_in.jsonl', 'w') as fh:
        for c in cases[b::NB]: fh.write(json.dumps(c, ensure_ascii=False) + '\n')
json.dump(key, open(f'{D}/review/_key2.json', 'w'))
print('batches', NB, 'of ~', len(cases) // NB)
