"""Match realtycheck schemes (propertylab_*) to master-catalogue MY projects. Read-only; writes match_v3.json."""
import json, math, collections, sys, re
from rapidfuzz.distance import Levenshtein
from normalize import (variants, core, identity_numbers, name_kind, tokens, clean_full, type_class,
                       GENERIC, BUILDING_WORDS, SUFFIX_ADJ, NEIGHBOURHOOD, PREFIX)

D = '/var/www/html/peta/storage/app/propertylab-catalogue-match'
rc = json.load(open(f'{D}/realtycheck_schemes.json'))
cat = json.load(open(f'{D}/catalogue_my_projects.json'))
HR = {'Condo/Apartment', 'Serviced Apartment', 'Flat', 'Office/SOHO'}

def fnum(x):
    try:
        v = float(x); return v if v != 0 else None
    except (TypeError, ValueError):
        return None

def hav(a, b, c, d):
    if None in (a, b, c, d): return None
    a, b, c, d = map(math.radians, (a, b, c, d))
    return 6371000 * 2 * math.asin(math.sqrt(math.sin((c - a) / 2) ** 2 + math.cos(a) * math.cos(c) * math.sin((d - b) / 2) ** 2))

def prep(rec, name):
    rec['V'] = [(k, c) for k, c in ((k, core(v)) for k, v in variants(name)) if c]
    rec['ids'], rec['locn'] = identity_numbers(name)
    rec['kind'] = name_kind(name)
    rec['raw'] = set(tokens(clean_full(name)))
    rec['tclass'] = type_class(rec['raw'])
    mains = [v for k, v in rec['V'] if k == 'main']
    rec['full'] = mains[0] if mains else []
    rec['key'] = ' '.join(sorted(t for t in rec['full'] if t not in PREFIX))

for r in rc:
    prep(r, r['display_name']); r['lat'], r['lng'] = fnum(r['latitude']), fnum(r['longitude'])
for p in cat:
    prep(p, p['project_name']); p['lat'], p['lng'] = fnum(p['latitude']), fnum(p['longitude'])

df = collections.Counter()
for rec in rc + cat:
    for t in set(t for _, v in rec['V'] for t in v): df[t] += 1
NDOC = len(rc) + len(cat)
places = set()
for rec, fields in [(r, ('state', 'district', 'mukim')) for r in rc] + [(p, ('state', 'area')) for p in cat]:
    for f in fields:
        places.update(tokens(clean_full(rec.get(f) or '')))
LOCATION = places - SUFFIX_ADJ - BUILDING_WORDS
W = {}
def wt(t):
    if t not in W:
        w = math.log(NDOC / (1 + df[t])) + 0.5
        if t in GENERIC: w *= 0.3
        elif t in places: w *= 0.5
        if any(c.isdigit() for c in t): w = max(w, 1.5)
        W[t] = w
    return W[t]

def tok_eq(a, b):
    if a == b: return 1.0
    if a.isdigit() or b.isdigit() or len(a) < 5 or len(b) < 5: return 0.0
    s = Levenshtein.normalized_similarity(a, b)
    return s if s >= 0.84 else 0.0

def soft_jaccard(A, B):
    A, B = list(dict.fromkeys(A)), list(dict.fromkeys(B))
    if not A or not B: return 0.0
    mw, used = 0.0, set()
    for a in A:
        best, bj = 0.0, None
        for j, b in enumerate(B):
            if j in used: continue
            e = tok_eq(a, b)
            if e > best: best, bj = e, j
        if bj is not None:
            used.add(bj); mw += best * (wt(a) + wt(B[bj])) / 2
    tot = sum(map(wt, A)) + sum(map(wt, B)) - mw
    return mw / tot if tot else 0.0

def names_equal(a, b):
    """Same name apart from spelling slips and a missing Taman/Bandar/Kampung. Taman vs Kampung is NOT the same."""
    pa, pb = set(a) & PREFIX, set(b) & PREFIX
    if pa and pb and not (pa & pb): return False
    a = [t for t in a if t not in PREFIX]; b = [t for t in b if t not in PREFIX]
    if not a or len(a) != len(b): return False
    left = list(b)
    for t in a:
        hit = next((u for u in left if tok_eq(t, u) > 0), None)
        if hit is None: return False
        left.remove(hit)
    return True

PEN = {'main': 0.0, 'pre': 0.02, 'inner': 0.10}
def compare(x, y):
    """Best reading pair between two records: (sim, a, b, kinds, exact_kind) — exact_kind 'main' | 'inner' | None."""
    best, exact = (0.0, None, None, None), None
    for ka, a in x['V']:
        for kb, b in y['V']:
            s = soft_jaccard(a, b) - max(PEN[ka], PEN[kb])
            eq = names_equal(a, b)
            if eq:
                ek = 'inner' if 'inner' in (ka, kb) else 'main'
                if exact != 'main':
                    exact = ek
                    if ek == 'main' or best[0] < s + 0.2: best = (max(s, 0.9), a, b, (ka, kb))
            if s > best[0] and not (exact == 'main' and not eq): best = (s, a, b, (ka, kb))
    return best + (exact,)

def dropped_qualifier(rec, matched):
    """Words of the full name that the matched reading left out and that are NOT just location."""
    return [t for t in rec['full'] if t not in matched and not (t in LOCATION or t in NEIGHBOURHOOD or len(t) <= 3 or t in rec['locn'])]

post = collections.defaultdict(set)
for i, p in enumerate(cat):
    for _, v in p['V']:
        for t in v: post[t].add(i)
by_len = collections.defaultdict(list)
for t in post:
    if len(t) >= 5 and not t.isdigit(): by_len[len(t)].append(t)
fz = {}
def near_tokens(t):
    if t not in fz:
        out = {t} if t in post else set()
        if len(t) >= 5 and not t.isdigit():
            for L in range(len(t) - 2, len(t) + 3):
                out.update(u for u in by_len.get(L, ()) if u[0] == t[0] and tok_eq(t, u) > 0)
        fz[t] = out
    return fz[t]

def thresholds(r):
    if r['category'] in HR and r['precision'] != 'road': return 300, 1000, 2500
    return 1500, 3000, 6000

LANDED_T = {'Terrace House', 'Semi-Detached House', 'Detached House', 'Bungalow', 'Cluster House', 'Town House', 'Low-Cost House', 'Land'}
HIGH_T = {'Condominium/Apartment', 'Flat', 'Hotel/Service Apartment', 'Serviced Apartment', 'Low-Cost Flat'}
SHOP_T, IND_T, OFF_T = {'Shop/Shoplot'}, {'Factory'}, {'Office', 'SoHo'}
def type_fit(r, p):
    parts = {x.strip() for x in (p.get('property_type') or '').split(',') if x.strip()}
    known = parts & (LANDED_T | HIGH_T | SHOP_T | IND_T | OFF_T)
    c, nk = r['category'], p['kind']
    if c == 'Landed':
        if nk == 'highrise': return 'conflict'
        return 'match' if known & LANDED_T else ('conflict' if known else 'unknown')
    if c in ('Condo/Apartment', 'Serviced Apartment', 'Flat'):
        if nk == 'landed': return 'conflict'
        return 'match' if known & HIGH_T else ('conflict' if known else ('match' if nk == 'highrise' else 'unknown'))
    if c == 'Office/SOHO':
        if known & OFF_T: return 'match'
        return 'partial' if (known & {'Hotel/Service Apartment', 'Serviced Apartment', 'Condominium/Apartment'} or not known) else 'conflict'
    if c == 'Shop': return 'match' if known & SHOP_T else 'conflict'
    if c == 'Industrial': return 'match' if known & IND_T else 'conflict'
    return 'unknown'

def psf_ratio(r, p):
    a, b = fnum(r.get('reported_psf')), fnum(p.get('psf_median'))
    return round(a / b, 2) if a and b else None

def num_relation(x, y):
    a, b = x['ids'], y['ids']
    rel = ('equal' if a == b else 'overlap' if a & b else 'conflict') if (a and b) else ('one_side' if (a or b) else 'none')
    if x['locn'] and y['locn'] and not (x['locn'] & y['locn']): rel = 'conflict'
    return rel

def building_conflict(x, y):
    a, b = x['raw'] & BUILDING_WORDS, y['raw'] & BUILDING_WORDS
    if a and b and not (a & b): return True
    ta, tb = x['tclass'], y['tclass']
    return bool(ta and tb and ta != tb and 'flat' in (ta, tb))

state_norm = lambda s: re.sub(r'^(wp|w\.p\.)\s*', '', (s or '').lower()).strip()
cat_key_state = collections.Counter((p['key'], state_norm(p['state'])) for p in cat)
rc_key_state = collections.Counter((r['key'], state_norm(r['state']), r['category'] in HR) for r in rc)
TYPE_RANK = {'match': 0, 'partial': 1, 'unknown': 2, 'conflict': 3}
NUM_RANK = {'equal': 0, 'none': 1, 'overlap': 2, 'one_side': 3, 'conflict': 4}

def candidates(r):
    cand = set()
    for _, v in r['V']:
        strong_t = [t for t in v if wt(t) >= 1.0]
        if strong_t or len(v) == 1:
            for t in (strong_t or v):
                for u in near_tokens(t): cand |= post[u]
        else:
            sets = [post.get(t, set()) for t in v]
            cand |= set.intersection(*sets) if sets else set()
    return cand

results = []
for n, r in enumerate(rc):
    strong, ok, far = thresholds(r)
    scored = []
    for i in candidates(r):
        p = cat[i]
        s, a, b, kinds, exact = compare(r, p)
        if s < 0.45 and not exact: continue
        d = hav(r['lat'], r['lng'], p['lat'], p['lng'])
        tf, nr, bc = type_fit(r, p), num_relation(r, p), building_conflict(r, p)
        dpen = 0.05 if d is None else (0 if d <= strong else 0.1 if d <= ok else 0.25 if d <= far else 0.6)
        rank = (s + (0.15 if exact == 'main' else 0.05 if exact else 0) - dpen - (0.15 if tf == 'conflict' else 0)
                - (0.1 if nr == 'conflict' else 0) + (0.03 if nr == 'equal' else -0.03 if nr == 'one_side' else 0) - (0.1 if bc else 0))
        scored.append({'i': i, 'rank': round(rank, 3), 'name_sim': round(s, 3), 'exact': exact, 'dist_m': None if d is None else round(d),
                       'num_rel': nr, 'rc_ids': sorted(r['ids']), 'cat_ids': sorted(p['ids']), 'type_fit': tf, 'bconf': bc,
                       'psf_ratio': psf_ratio(r, p), 'via': kinds, 'rc_variant': ' '.join(a or []), 'cat_variant': ' '.join(b or []),
                       'q_rc': dropped_qualifier(r, a or []), 'q_cat': dropped_qualifier(p, b or [])})
    scored.sort(key=lambda c: c['rank'], reverse=True)
    top = scored[:6]
    if top:
        b0 = cat[top[0]['i']]
        tw_r = 800 if r['category'] in HR else 3000
        group = [top[0]]
        for c in top[1:]:
            pc = cat[c['i']]
            dd = hav(b0['lat'], b0['lng'], pc['lat'], pc['lng'])
            if (c['type_fit'] != 'conflict' and b0['ids'] == pc['ids'] and not building_conflict(b0, pc)
                    and any(names_equal(x, y) for kx, x in b0['V'] if kx != 'inner' for ky, y in pc['V'] if ky != 'inner')
                    and (dd is None or dd <= tw_r)):
                group.append(c)
        if len(group) > 1:
            def raw_sim(c): return soft_jaccard(sorted(r['raw']), sorted(cat[c['i']]['raw']))
            group.sort(key=lambda c: (TYPE_RANK[c['type_fit']], NUM_RANK[c['num_rel']], c['exact'] != 'main', c['bconf'], -raw_sim(c),
                                      'edgeprop' not in cat[c['i']]['providers'], cat[c['i']]['psf_median'] is None,
                                      c['dist_m'] if c['dist_m'] is not None else 1e9))
            gi = {c['i'] for c in group}
            top = group + [c for c in top if c['i'] not in gi]
            for c in top[1:]: c['twin'] = c['i'] in gi
    results.append({'scheme_id': r['scheme_id'], 'top': top})
    if n % 3000 == 0: print(n, file=sys.stderr)

def verdict(r, top):
    if not top: return 'E4_NOT_IN_CATALOGUE', 'no catalogue project shares a distinctive name word'
    strong, ok, far = thresholds(r)
    b = top[0]; p = cat[b['i']]; d = b['dist_m']; hr = r['category'] in HR
    if b['bconf']: nq = 'weak'
    elif b['exact'] == 'main' and not b['q_rc'] and not b['q_cat']: nq = 'strong'
    elif b['exact']: nq = 'strong_q'
    elif b['name_sim'] >= 0.75: nq = 'good'
    elif b['name_sim'] >= 0.6: nq = 'weak'
    else: nq = 'none'
    dq = 'unknown' if d is None else 'near' if d <= strong else 'mid' if d <= ok else 'farish' if d <= far else 'far'
    pr = b['psf_ratio']
    price_ok = hr and pr is not None and 0.8 <= pr <= 1.25
    price_tight = hr and pr is not None and 0.9 <= pr <= 1.1
    price_bad = hr and pr is not None and (pr < 0.6 or pr > 1.6)
    unique_state = bool(state_norm(r['state'])) and cat_key_state[(p['key'], state_norm(p['state']))] == 1 and rc_key_state[(r['key'], state_norm(r['state']), hr)] == 1
    if nq == 'none': return 'E4_NOT_IN_CATALOGUE', f'closest name only {b["name_sim"]} similar ({p["project_name"]})'
    if dq == 'far':
        if nq in ('strong', 'strong_q'):
            same_state = state_norm(r['state']) and state_norm(r['state']) == state_norm(p['state'])
            if same_state and unique_state and d <= 60000 and b['type_fit'] != 'conflict' and b['num_rel'] != 'conflict':
                if nq == 'strong' and price_tight:
                    return 'A_SAME', f'coordinates differ by {round(d/1000,1)} km, but the name is unique in the state and PSF matches ({pr}x)'
                return 'B2_LIKELY_COORDS_DISAGREE', f'same name, the only one in {r["state"]} on each side, but {round(d/1000,1)} km apart'
            return 'E3_SAME_NAME_OTHER_PLACE', f'same name, {round(d/1000,1)} km away'
        return 'E4_NOT_IN_CATALOGUE', f'closest similar name is {round(d/1000,1)} km away ({p["project_name"]})'
    if b['type_fit'] == 'conflict':
        return 'E1_DIFFERENT_PROPERTY_TYPE', f'{r["category"]} vs catalogue {p["property_type"] or p["kind"]}'
    if b['num_rel'] == 'conflict':
        return 'E2_DIFFERENT_PHASE', f'numbers {b["rc_ids"] or sorted(r["locn"])} vs {b["cat_ids"] or sorted(p["locn"])}'
    if b['num_rel'] in ('one_side', 'overlap') and nq in ('strong', 'strong_q', 'good') and dq in ('near', 'mid', 'unknown'):
        return 'C_PART_OF', f'phase/number on one side only: {b["rc_ids"]} vs {b["cat_ids"]}'
    if nq == 'strong':
        rivals = [c for c in top[1:] if not c.get('twin') and c['exact'] == 'main' and c['dist_m'] is not None and c['dist_m'] <= ok
                  and c['type_fit'] != 'conflict' and c['num_rel'] != 'conflict' and not c['bconf']]
        if rivals:
            return 'D_AMBIGUOUS', 'more than one catalogue project fits: ' + ' / '.join(cat[c['i']]['project_name'] for c in [b] + rivals[:2])
        if price_bad:
            return 'B1a_LIKELY', f'same name and place, but PSF differs {pr}x'
        if r['category'] == 'Office/SOHO' and b['type_fit'] == 'partial' and not price_ok:
            return 'B1a_LIKELY', 'office/SOHO scheme vs residential catalogue row'
        if dq == 'near':
            return 'A_SAME', 'same name, same place' + (f', PSF agrees ({pr}x)' if price_ok else '')
        if dq == 'mid' and (price_ok or (not hr and b['type_fit'] == 'match' and unique_state)):
            return 'A_SAME', f'same name, {d} m apart, ' + (f'PSF agrees ({pr}x)' if price_ok else 'the only one of this name in the state')
        return 'B1a_LIKELY', f'same name, {dq} ({d} m)' + (f', PSF {pr}x' if pr else '')
    if nq == 'strong_q' and dq in ('near', 'mid', 'farish', 'unknown'):
        extra = b['q_rc'] or b['q_cat']
        return 'B1a_LIKELY', ('matched on the bracketed name' if b['exact'] == 'inner' else 'same name but one side adds: ' + ' '.join(extra)) + f' ({d} m)'
    if nq == 'good' and dq in ('near', 'mid', 'farish', 'unknown'):
        return 'B1b_DOUBTFUL', f'similar but not identical names ({b["name_sim"]}), {dq} ({d} m)'
    return 'E5_WEAK', f'weak name match ({b["name_sim"]}), {dq} ({d} m)'

rcmap = {r['scheme_id']: r for r in rc}
out = []
for x in results:
    t, why = verdict(rcmap[x['scheme_id']], x['top'])
    out.append({'scheme_id': x['scheme_id'], 'tier': t, 'why': why, 'top': x['top']})
json.dump(out, open(f'{D}/match_v3.json', 'w'))
cnt = collections.Counter(o['tier'] for o in out)
for k in sorted(cnt): print(f'{k:28s} {cnt[k]:6d}')
