"""Match realtycheck schemes (propertylab_*) to master-catalogue MY projects. Read-only; writes JSON next to the inputs."""
import json, math, collections, sys, re
from rapidfuzz.distance import Levenshtein
from normalize import variants, core, identity_numbers, name_kind, tokens, clean_full, GENERIC, BUILDING_WORDS

D = '/var/www/html/peta/storage/app/propertylab-catalogue-match'
rc = json.load(open(f'{D}/realtycheck_schemes.json'))
cat = json.load(open(f'{D}/catalogue_my_projects.json'))

def fnum(x):
    try:
        v = float(x); return v if v != 0 else None
    except (TypeError, ValueError):
        return None

def hav(a, b, c, d):
    if None in (a, b, c, d): return None
    a, b, c, d = map(math.radians, (a, b, c, d))
    return 6371000 * 2 * math.asin(math.sqrt(math.sin((c - a) / 2) ** 2 + math.cos(a) * math.cos(c) * math.sin((d - b) / 2) ** 2))

def prep(rec, name):
    vs = [(k, core(v)) for k, v in variants(name)]
    rec['V'] = [(k, v) for k, v in vs if v]
    rec['ids'], rec['locn'] = identity_numbers(name)
    rec['kind'] = name_kind(name)
    rec['raw'] = set(tokens(clean_full(name)))
    main = [v for k, v in rec['V'] if k == 'main']
    rec['key'] = ' '.join(sorted(main[-1])) if main else ''

for r in rc:
    prep(r, r['display_name']); r['lat'], r['lng'] = fnum(r['latitude']), fnum(r['longitude'])
for p in cat:
    prep(p, p['project_name']); p['lat'], p['lng'] = fnum(p['latitude']), fnum(p['longitude'])

df = collections.Counter()
for rec in rc + cat:
    for t in set(t for _, v in rec['V'] for t in v): df[t] += 1
N = len(rc) + len(cat)
places = set()
for rec, fields in [(r, ('state', 'district', 'mukim')) for r in rc] + [(p, ('state', 'area')) for p in cat]:
    for f in fields:
        places.update(tokens(clean_full(rec.get(f) or '')))
W = {}
def wt(t):
    if t not in W:
        w = math.log(N / (1 + df[t])) + 0.5
        if t in GENERIC: w *= 0.3
        elif t in places: w *= 0.5
        if any(c.isdigit() for c in t): w = max(w, 1.5)
        W[t] = w
    return W[t]

def tok_eq(a, b):
    if a == b: return 1.0
    if a.isdigit() or b.isdigit() or len(a) < 5 or len(b) < 5: return 0.0
    s = Levenshtein.normalized_similarity(a, b)
    return s if s >= 0.84 else 0.0

def soft_jaccard(A, B):
    A, B = list(dict.fromkeys(A)), list(dict.fromkeys(B))
    if not A or not B: return 0.0
    mw, used = 0.0, set()
    for a in A:
        best, bj = 0.0, None
        for j, b in enumerate(B):
            if j in used: continue
            e = tok_eq(a, b)
            if e > best: best, bj = e, j
        if bj is not None:
            used.add(bj); mw += best * (wt(a) + wt(B[bj])) / 2
    tot = sum(map(wt, A)) + sum(map(wt, B)) - mw
    return mw / tot if tot else 0.0


PREFIX = {'taman', 'bandar', 'kampung'}
def fuzzy_equal(a, b):
    a = [t for t in a if t not in PREFIX]; b = [t for t in b if t not in PREFIX]
    if not a or len(a) != len(b): return False
    left = list(b)
    for t in a:
        hit = next((u for u in left if tok_eq(t, u) > 0), None)
        if hit is None: return False
        left.remove(hit)
    return True

def same_core(x, y):
    return any(fuzzy_equal(a, b) for ka, a in x['V'] if ka != 'inner' for kb, b in y['V'] if kb != 'inner')

def full_core(rec):
    mains = [v for k, v in rec['V'] if k == 'main']
    return mains[0] if mains else []

def dropped_is_location(rec, matched):
    """Tokens of the full name that the matched reading left out are all place words / codes / neighbourhood words."""
    left = [t for t in full_core(rec) if t not in matched]
    return all(t in places or t in GENERIC or len(t) <= 3 or t in rec['locn'] for t in left), left

PEN = {'main': 0.0, 'pre': 0.02, 'inner': 0.10}
def name_score(x, y):
    best = (0.0, None, None, None)
    for ka, a in x['V']:
        for kb, b in y['V']:
            s = soft_jaccard(a, b) - max(PEN[ka], PEN[kb])
            if s > best[0]: best = (s, a, b, (ka, kb))
    return best

post = collections.defaultdict(set)
for i, p in enumerate(cat):
    for _, v in p['V']:
        for t in v: post[t].add(i)
by_len = collections.defaultdict(list)
for t in post:
    if len(t) >= 5 and not t.isdigit(): by_len[len(t)].append(t)
fz = {}
def near_tokens(t):
    if t not in fz:
        out = {t} if t in post else set()
        if len(t) >= 5 and not t.isdigit():
            for L in range(len(t) - 2, len(t) + 3):
                out.update(u for u in by_len.get(L, ()) if u[0] == t[0] and tok_eq(t, u) > 0)
        fz[t] = out
    return fz[t]

HR = {'Condo/Apartment', 'Serviced Apartment', 'Flat', 'Office/SOHO'}
def thresholds(r):
    if r['category'] in HR and r['precision'] != 'road': return 300, 1000, 2500
    return 1500, 3000, 6000

LANDED_T = {'Terrace House', 'Semi-Detached House', 'Detached House', 'Bungalow', 'Cluster House', 'Town House', 'Low-Cost House', 'Land'}
HIGH_T = {'Condominium/Apartment', 'Flat', 'Hotel/Service Apartment', 'Serviced Apartment', 'Low-Cost Flat'}
SHOP_T, IND_T, OFF_T = {'Shop/Shoplot'}, {'Factory'}, {'Office', 'SoHo'}
def type_fit(r, p):
    parts = {x.strip() for x in (p.get('property_type') or '').split(',') if x.strip()}
    known = parts & (LANDED_T | HIGH_T | SHOP_T | IND_T | OFF_T)
    c, nk = r['category'], p['kind']
    if c == 'Landed':
        if nk == 'highrise': return 'conflict'
        return 'match' if known & LANDED_T else ('conflict' if known else 'unknown')
    if c in ('Condo/Apartment', 'Serviced Apartment', 'Flat'):
        if nk == 'landed': return 'conflict'
        return 'match' if known & HIGH_T else ('conflict' if known else ('match' if nk == 'highrise' else 'unknown'))
    if c == 'Office/SOHO':
        if known & OFF_T: return 'match'
        return 'partial' if (known & {'Hotel/Service Apartment', 'Serviced Apartment', 'Condominium/Apartment'} or not known) else 'conflict'
    if c == 'Shop': return 'match' if known & SHOP_T else 'conflict'
    if c == 'Industrial': return 'match' if known & IND_T else 'conflict'
    return 'unknown'

def psf_ratio(r, p):
    a, b = fnum(r.get('reported_psf')), fnum(p.get('psf_median'))
    return round(a / b, 2) if a and b else None

def num_relation(r, p):
    a, b = r['ids'], p['ids']
    if a and b:
        if a == b: rel = 'equal'
        elif a & b: rel = 'overlap'
        else: rel = 'conflict'
    elif a or b: rel = 'one_side'
    else: rel = 'none'
    la, lb = r['locn'], p['locn']
    if la and lb and not (la & lb): rel = 'conflict'
    return rel

def building_conflict(r, p):
    a, b = r['raw'] & BUILDING_WORDS, p['raw'] & BUILDING_WORDS
    return bool(a and b and not (a & b))

state_norm = lambda s: re.sub(r'^(wp|w\.p\.)\s*', '', (s or '').lower()).strip()
cat_key_state = collections.Counter((p['key'], state_norm(p['state'])) for p in cat)
rc_key_state = collections.Counter((r['key'], state_norm(r['state']), r['category'] in HR) for r in rc)

TYPE_RANK = {'match': 0, 'partial': 1, 'unknown': 2, 'conflict': 3}
results = []
for n, r in enumerate(rc):
    cand = set()
    for _, v in r['V']:
        strong_t = [t for t in v if wt(t) >= 1.0]
        if strong_t or len(v) == 1:
            for t in (strong_t or v):
                for u in near_tokens(t): cand |= post[u]
        else:
            sets = [post.get(t, set()) for t in v]
            cand |= set.intersection(*sets) if sets else set()
    strong, ok, far = thresholds(r)
    scored = []
    for i in cand:
        p = cat[i]
        s, a, b, kinds = name_score(r, p)
        if s < 0.45: continue
        d = hav(r['lat'], r['lng'], p['lat'], p['lng'])
        tf = type_fit(r, p)
        dpen = 0.05 if d is None else (0 if d <= strong else 0.1 if d <= ok else 0.25 if d <= far else 0.6)
        rank = s - dpen - (0.15 if tf == 'conflict' else 0) - (0.1 if num_relation(r, p) == 'conflict' else 0)
        scored.append((rank, s, d, i, a, b, kinds, tf))
    scored.sort(key=lambda x: x[0], reverse=True)
    top = []
    for rank, s, d, i, a, b, kinds, tf in scored[:6]:
        p = cat[i]
        top.append({'i': i, 'rank': round(rank, 3), 'name_sim': round(s, 3), 'dist_m': None if d is None else round(d),
                    'exact_core': sorted(a) == sorted(b), 'num_rel': num_relation(r, p), 'rc_ids': sorted(r['ids']), 'cat_ids': sorted(p['ids']),
                    'type_fit': tf, 'psf_ratio': psf_ratio(r, p), 'bconf': building_conflict(r, p), 'via': kinds,
                    'rc_variant': ' '.join(a), 'cat_variant': ' '.join(b), '_a': a, '_b': b})
    # twins: catalogue rows that are the same development listed twice (identical core name, close together)
    if top:
        b0 = cat[top[0]['i']]
        twin_r = 500 if r['category'] in HR else 1500
        for c in top[1:]:
            pc = cat[c['i']]
            dd = hav(b0['lat'], b0['lng'], pc['lat'], pc['lng'])
            c['twin'] = bool(c['type_fit'] != 'conflict' and same_core(b0, pc) and (dd is None or dd <= twin_r))
        group = [top[0]] + [c for c in top[1:] if c['twin']]
        if len(group) > 1:
            group.sort(key=lambda c: (TYPE_RANK[c['type_fit']], c['num_rel'] == 'conflict', 'edgeprop' not in cat[c['i']]['providers'],
                                      cat[c['i']]['psf_median'] is None, -c['name_sim'], c['dist_m'] if c['dist_m'] is not None else 1e9))
            gi = {c['i'] for c in group}
            top = group + [c for c in top if c['i'] not in gi]
            for c in top: c['twin'] = c['i'] in gi and c is not top[0]
    results.append({'scheme_id': r['scheme_id'], 'top': top, 'n_cand': len(scored)})
    if n % 3000 == 0: print(n, file=sys.stderr)

# ---- verdicts
def verdict(r, x):
    top = x['top']
    strong, ok, far = thresholds(r)
    if not top: return 'E4_NOT_IN_CATALOGUE', 'no catalogue project shares a distinctive name word'
    b = top[0]; p = cat[b['i']]
    d = b['dist_m']
    nq = 'strong' if ((b['exact_core'] or b['name_sim'] >= 0.9) and not b['bconf']) else 'good' if b['name_sim'] >= 0.75 else 'weak' if b['name_sim'] >= 0.6 else 'none'
    dq = 'unknown' if d is None else 'near' if d <= strong else 'mid' if d <= ok else 'farish' if d <= far else 'far'
    pr = b['psf_ratio']; hr = r['category'] in HR
    price_ok = hr and pr is not None and 0.8 <= pr <= 1.25
    price_bad = hr and pr is not None and (pr < 0.6 or pr > 1.6)
    if nq == 'none': return 'E4_NOT_IN_CATALOGUE', f'closest name only {b["name_sim"]} similar'
    if dq == 'far':
        if nq == 'strong':
            same_state = state_norm(r['state']) and state_norm(r['state']) == state_norm(p['state'])
            unique = cat_key_state[(p['key'], state_norm(p['state']))] == 1 and rc_key_state[(r['key'], state_norm(r['state']), hr)] == 1
            if same_state and unique and d <= 60000 and b['type_fit'] != 'conflict' and b['num_rel'] != 'conflict':
                return 'B2_LIKELY_COORDS_DISAGREE', f'same name, only one of it in {r["state"]} on each side, but {round(d/1000,1)} km apart'
            return 'E3_SAME_NAME_OTHER_PLACE', f'same name but {round(d/1000,1)} km away'
        return 'E4_NOT_IN_CATALOGUE', f'closest similar name is {round(d/1000,1)} km away'
    if b['type_fit'] == 'conflict':
        return 'E1_DIFFERENT_PROPERTY_TYPE', f'{r["category"]} vs catalogue {p["property_type"] or p["kind"]}'
    if b['num_rel'] == 'conflict':
        return 'E2_DIFFERENT_PHASE', f'numbers {b["rc_ids"] or sorted(r["locn"])} vs {b["cat_ids"] or sorted(p["locn"])}'
    rivals = [c for c in top[1:] if not c.get('twin') and c['name_sim'] >= b['name_sim'] - 0.05 and c['dist_m'] is not None
              and c['dist_m'] <= ok and c['type_fit'] != 'conflict' and c['num_rel'] != 'conflict'
              and not (b['exact_core'] and not c['exact_core'])]
    if b['num_rel'] in ('one_side', 'overlap') and nq in ('strong', 'good') and dq in ('near', 'mid', 'unknown'):
        return 'C_PART_OF', f'phase/number only on one side: {b["rc_ids"]} vs {b["cat_ids"]}'
    if rivals and nq in ('strong', 'good'):
        return 'D_AMBIGUOUS', 'two different catalogue projects fit: ' + ' / '.join(cat[c['i']]['project_name'] for c in [b] + rivals[:2])
    if nq == 'strong' and not price_bad:
        loc_a, left_a = dropped_is_location(r, b['_a'])
        loc_b, left_b = dropped_is_location(p, b['_b'])
        if 'inner' in b['via']:
            return 'B1_LIKELY', 'matched on the name in brackets: ' + (r['display_name'] if b['via'][0] == 'inner' else p['project_name'])
        if not (loc_a and loc_b):
            return 'B1_LIKELY', 'same core name, but one side adds: ' + ' '.join(left_a if not loc_a else left_b)
        if dq == 'near':
            return 'A_SAME', 'same name, same place' + (', price agrees' if price_ok else '')
        if dq == 'mid' and (price_ok or (not hr and b['type_fit'] == 'match' and cat_key_state[(p['key'], state_norm(p['state']))] == 1)):
            return 'A_SAME', 'same name, close by, ' + ('price agrees' if price_ok else 'only one of this name in the state')
    if nq == 'strong' and price_bad and dq in ('near', 'mid'):
        return 'B1_LIKELY', f'same name and place but PSF differs {pr}x'
    if nq in ('strong', 'good') and dq in ('near', 'mid', 'farish', 'unknown'):
        return 'B1_LIKELY', f'name {nq} ({b["name_sim"]}), {dq} ({d} m)' + (f', psf ratio {pr}' if pr else '')
    return 'E5_WEAK', f'name {nq} ({b["name_sim"]}), {dq} ({d} m)'

rcmap = {r['scheme_id']: r for r in rc}
out = []
for x in results:
    r = rcmap[x['scheme_id']]
    t, why = verdict(r, x)
    for c in x['top']: c.pop('_a', None); c.pop('_b', None)
    out.append({'scheme_id': x['scheme_id'], 'tier': t, 'why': why, 'top': x['top']})
json.dump(out, open(f'{D}/match_v2.json', 'w'))
cnt = collections.Counter(o['tier'] for o in out)
for k in sorted(cnt): print(f'{k:28s} {cnt[k]:6d}')
