import json, collections, random, sys
D = '/var/www/html/peta/storage/app/propertylab-catalogue-match'
rc = {r['scheme_id']: r for r in json.load(open(f'{D}/realtycheck_schemes.json'))}
cat = json.load(open(f'{D}/catalogue_my_projects.json'))
raw = json.load(open(f'{D}/match_raw.json'))
HR = {'Condo/Apartment', 'Serviced Apartment', 'Flat', 'Office/SOHO'}

def thresholds(r):
    if r['category'] in HR and r['precision'] != 'road': return 300, 1000, 2500
    return 1500, 3000, 6000

def classify(x):
    r = rc[x['scheme_id']]
    top = x['top']
    if not top: return 'E_NO_CANDIDATE', 'no catalogue name shares a distinctive word', None
    b = top[0]; c = top[1] if len(top) > 1 else None
    strong, ok, far = thresholds(r)
    d = b['dist_m']
    nq = 'strong' if (b['exact_core'] or b['name_sim'] >= 0.9) else 'good' if b['name_sim'] >= 0.75 else 'weak' if b['name_sim'] >= 0.6 else 'none'
    dq = 'unknown' if d is None else 'near' if d <= strong else 'mid' if d <= ok else 'farish' if d <= far else 'far'
    pr = b['psf_ratio']
    price_ok = pr is not None and 0.67 <= pr <= 1.5
    price_bad = pr is not None and (pr < 0.5 or pr > 2.0)
    if nq == 'none': return 'E_NO_CANDIDATE', f'best name similarity {b["name_sim"]}', b
    if dq == 'far':
        return ('E_SAME_NAME_ELSEWHERE' if nq == 'strong' else 'E_NO_CANDIDATE'), f'best candidate {d} m away', b
    if b['type_fit'] == 'conflict':
        return 'E_DIFFERENT_TYPE', f'{r["category"]} vs {cat[b["i"]]["property_type"]}', b
    if b['num_conflict']:
        return 'E_DIFFERENT_PHASE', f'numbers {b["num_a"]} vs {b["num_b"]}', b
    rival = None
    if c and c['name_sim'] >= b['name_sim'] - 0.05 and c['dist_m'] is not None and c['dist_m'] <= ok \
            and not (b['exact_core'] and not c['exact_core']) and c['type_fit'] != 'conflict' and not c['num_conflict']:
        rival = c
    if b['num_one_side'] and nq in ('strong', 'good') and dq in ('near', 'mid', 'unknown'):
        return 'C_PART_OF', f'one side carries a phase/number: {b["num_a"]} vs {b["num_b"]}', b
    if rival:
        return 'D_AMBIGUOUS', f'two candidates: {cat[b["i"]]["project_name"]} / {cat[rival["i"]]["project_name"]}', b
    if nq == 'strong' and dq == 'near' and not price_bad:
        return 'A_SAME', 'name equal, location agrees', b
    if nq == 'strong' and dq == 'mid' and b['type_fit'] == 'match' and price_ok:
        return 'A_SAME', 'name equal, location close, price agrees', b
    if nq in ('strong', 'good') and dq in ('near', 'mid', 'farish', 'unknown'):
        return 'B_LIKELY', f'name {nq} ({b["name_sim"]}), distance {dq} ({d} m), psf ratio {pr}', b
    return 'E_WEAK', f'name {nq} ({b["name_sim"]}), distance {dq} ({d} m)', b

out = []
for x in raw:
    tier, why, b = classify(x)
    out.append((tier, why, x, b))
cnt = collections.Counter(t for t, *_ in out)
for k in sorted(cnt): print(f'{k:24s} {cnt[k]:6d}')
json.dump([{'scheme_id': x['scheme_id'], 'tier': t, 'why': w, 'best': b, 'top': x['top']} for t, w, x, b in out], open(f'{D}/match_tiers.json', 'w'))

if len(sys.argv) > 1:
    tier = sys.argv[1]; k = int(sys.argv[2]) if len(sys.argv) > 2 else 30
    rows = [o for o in out if o[0] == tier]
    random.seed(7)
    for t, w, x, b in random.sample(rows, min(k, len(rows))):
        r = rc[x['scheme_id']]
        p = cat[b['i']] if b else None
        print(f"- {r['display_name']} [{r['category']}, {r['district']}/{r['state']}, {r['precision']}, psf {r['reported_psf']}]"
              f"  =>  {p['project_name'] if p else '-'} [{p['property_type'] if p else ''}, {p['area'] if p else ''}, psf {p['psf_median'] if p else ''}]"
              f"  sim={b['name_sim'] if b else ''} d={b['dist_m'] if b else ''} | {w}")
