"""Final decision per realtycheck scheme: LINK (auto, >=90%) / HOLD (same, <90%) / RELATED (part of) / NONE."""
import json, glob, collections, sys
sys.path.insert(0, '/tmp/claude-1000/-var-www-html-peta/310517d2-8f28-442b-9907-0e3560add2cb/scratchpad/match')
from adj_decisions import ADJ
D = '/var/www/html/peta/storage/app/propertylab-catalogue-match'; R = f'{D}/review'
rc = {r['scheme_id']: r for r in json.load(open(f'{D}/realtycheck_schemes.json'))}
cat = json.load(open(f'{D}/catalogue_my_projects.json'))
final = json.load(open(f'{D}/final.json'))
key3 = json.load(open(f'{R}/_key3.json'))
by_scheme = {k['scheme_id']: (c, k) for c, k in key3.items()}
v = {}
for fn in glob.glob(f'{R}/v_*_out.jsonl'):
    for l in open(fn):
        if l.strip(): x = json.loads(l); v[x['case']] = x
outcome = json.load(open(f'{R}/_verify_outcome.json'))
HR = {'Condo/Apartment', 'Serviced Apartment', 'Flat'}
def seg(r): return 'high-rise' if r['category'] in HR else 'landed' if r['category'] == 'Landed' else 'commercial'
def fnum(x):
    try:
        y = float(x); return y if y > 0 else None
    except (TypeError, ValueError):
        return None
def price_ok(r, i):
    a, b = fnum(r['median_rm']), fnum(cat[i]['price_median']) if i is not None else None
    return bool(a and b and 0.85 <= a / b <= 1.15)
PRIOR = {'SAME - confident': 'SAME', 'SAME - probable': 'SAME', 'PART OF (phase / block / sub-estate)': 'PART_OF', 'NEEDS HUMAN CHECK': 'UNSURE'}

out = []
for f in final:
    sid = f['scheme_id']; r = rc[sid]; s = seg(r)
    rec = {'scheme_id': sid, 'segment': s, 'link': 'NONE', 'cat_i': None, 'confidence': None, 'path': '', 'note': ''}
    if sid not in by_scheme:
        if f['status'] == 'SAME - confident':
            rec.update(link='LINK', cat_i=f['cat_i'], confidence=95 if f['method'] == 'rules' else 92,
                       path='rules' if f['method'] == 'rules' else 'AI review', note=f['reason'])
        elif f['status'].startswith('NOT SAME'):
            rec.update(link='NONE', path='rules' if f['method'] == 'rules' else 'AI review', note=f['status'] + ' — ' + (f['reason'] or ''))
        else:
            rec.update(link='NONE', path='rules' if f['method'] == 'rules' else 'AI review', note='not in catalogue — ' + (f['reason'] or ''))
        out.append(rec); continue
    cid, k = by_scheme[sid]
    x = v.get(cid)
    def cand_i(kk):
        if kk == 'prior': return f['cat_i']
        return k['cand_i'][kk] if isinstance(kk, int) and 0 <= kk < len(k['cand_i']) else None
    if cid in ADJ:
        fin, kk, conf, note = ADJ[cid]
        i = cand_i(kk)
        if fin == 'SAME' and i is not None:
            rec.update(link='LINK' if conf >= 90 else 'HOLD', cat_i=i, confidence=conf, path='2 AI reviews + Claude adjudication', note=note)
        elif fin == 'PART_OF' and i is not None:
            rec.update(link='RELATED', cat_i=i, confidence=conf, path='2 AI reviews + Claude adjudication', note=note)
        else:
            rec.update(link='NONE', path='2 AI reviews + Claude adjudication', note=note)
        out.append(rec); continue
    if x is None:
        rec.update(link='PENDING', cat_i=f['cat_i'], path='awaiting second review', note=f['reason'] or '')
        out.append(rec); continue
    o = outcome.get(cid, ('adjudicate', ''))
    pv = PRIOR[f['status']]; vv = x['verdict']; conf2 = x.get('confidence_pct') or 0
    vi = cand_i(x.get('cand'))
    two = '2 independent AI reviews agree'
    if o[0] == 'agree_same':
        rec.update(link='LINK', cat_i=o[1], confidence=max(90, conf2), path=two, note=x.get('reason', ''))
    elif o[0] == 'agree_part':
        rec.update(link='RELATED', cat_i=o[1], confidence=conf2, path=two, note=x.get('reason', ''))
    elif o[0] == 'no_link':
        rec.update(link='NONE', path=two, note=x.get('reason', ''))
    elif s == 'high-rise':
        rec.update(link='PENDING', cat_i=f['cat_i'], path='needs Claude adjudication', note=x.get('reason', ''))
    else:
        # landed / commercial disagreements: objective tie-breakers, never a link on one opinion alone
        if pv == 'SAME' and vv == 'SAME' and vi is not None:
            if price_ok(r, vi): rec.update(link='LINK', cat_i=vi, confidence=90, path=two + ' + median price within ±15%', note=x.get('reason', ''))
            else: rec.update(link='HOLD', cat_i=vi, confidence=conf2, path=two + ', no price confirmation', note=x.get('reason', ''))
        elif pv == 'UNSURE' and vv == 'SAME' and vi is not None:
            if conf2 >= 90 and price_ok(r, vi): rec.update(link='LINK', cat_i=vi, confidence=90, path='second review ≥90% + median price within ±15%', note=x.get('reason', ''))
            else: rec.update(link='HOLD', cat_i=vi, confidence=conf2, path='first review unsure, second says same', note=x.get('reason', ''))
        elif 'PART_OF' in (pv, vv) and vv != 'DIFFERENT' and (vi is not None or f['cat_i'] is not None):
            rec.update(link='RELATED', cat_i=vi if vi is not None else f['cat_i'], confidence=conf2, path='reviews split between same / part-of', note=x.get('reason', ''))
        elif vv == 'UNSURE' and pv == 'SAME':
            rec.update(link='HOLD', cat_i=f['cat_i'], confidence=conf2, path='second review unsure', note=x.get('reason', ''))
        else:
            rec.update(link='NONE', path='reviews disagree', note=x.get('reason', ''))
    out.append(rec)

# High-rise scheme linked to a mixed estate / landed row: keep the link only when the row's own numbers
# are clearly this high-rise's (median price within ±15%, and PSF too when both sides have it).
import re
LANDED_T = {'Terrace House', 'Semi-Detached House', 'Detached House', 'Bungalow', 'Cluster House', 'Town House', 'Low-Cost House', 'Land'}
HWORD = re.compile(r'\b(apartment|apartments|flat|flats|condo|condominium|kondominium|pangsapuri|suites?|residences?|residensi|residency|tower|towers|court|mansion|heights|soho|sovo|sofo)\b', re.I)
estate_downgrades = 0
for rec in out:
    if rec['segment'] != 'high-rise' or rec['link'] != 'LINK': continue
    r = rc[rec['scheme_id']]; p = cat[rec['cat_i']]
    parts = {x.strip() for x in (p['property_type'] or '').split(',') if x.strip()}
    if not parts & LANDED_T: continue
    a, b = fnum(r['median_rm']), fnum(p['price_median']); pa, pb = fnum(r['reported_psf']), fnum(p['psf_median'])
    price_ok_ = bool(a and b and 0.85 <= a / b <= 1.15)
    psf_ok_ = (not (pa and pb)) or 0.85 <= pa / pb <= 1.15
    named = bool(HWORD.search(p['project_name'])) and not re.match(r'(?i)^(taman|bandar|kampung|desa|seksyen|section)\b', p['project_name'])
    if (price_ok_ and psf_ok_) or (named and not (parts - LANDED_T)):
        continue
    rec.update(link='RELATED', path=rec['path'] + ' → estate-level check',
               note='catalogue row is a mixed estate / landed row whose prices are not this high-rise\'s — ' + (rec['note'] or ''))
    estate_downgrades += 1
print('high-rise links moved to RELATED by the estate-level check:', estate_downgrades)
json.dump(out, open(f'{D}/final_links.json', 'w'))
t = collections.Counter((o['segment'], o['link']) for o in out)
print(f"{'segment':12s} " + ' '.join(f'{l:>8s}' for l in ('LINK', 'HOLD', 'RELATED', 'NONE', 'PENDING')))
for sg in ('high-rise', 'landed', 'commercial'):
    print(f'{sg:12s} ' + ' '.join(f'{t[(sg, l)]:8d}' for l in ('LINK', 'HOLD', 'RELATED', 'NONE', 'PENDING')))
print('total', len(out))
