"""Second round for high-rise NONE (2026-09-25): candidate generation + priority order.

For each high-rise scheme decided NONE, list catalogue candidates the first round could not see:
 A) same-state projects sharing a distinctive name word, ranked by word overlap (pins ignored — many are wrong);
 B) projects within 1.2 km of realtycheck's pin, any name, high-rise or untyped (catches renamed buildings).
Priority: Klang Valley first, then everything else; within a tier, most sales first.
Writes review/n_candidates.json.
"""
import csv, json, math, sys, collections
sys.path.insert(0, 'scripts/pipeline')
from normalize import tokens, clean_full, GENERIC, TYPE_WORDS
rc = {r['scheme_id']: r for r in json.load(open('realtycheck_schemes.json'))}
cat = json.load(open('catalogue_my_projects.json'))
ST = {'WP Kuala Lumpur': 'Kuala Lumpur', 'Pulau Pinang': 'Penang', 'WP Putrajaya': 'Putrajaya', 'WP Labuan': 'Labuan'}
KV_SEL = {'Petaling', 'Klang', 'Gombak', 'Hulu Langat', 'Sepang'}
HR_T = {'Condominium/Apartment', 'Flat', 'Hotel/Service Apartment', 'Serviced Apartment', 'Low-Cost Flat', 'SOHO', 'Service Residence'}
def f(x):
    try: return float(x)
    except (TypeError, ValueError): return None
def dist(a, b):
    la1, lo1, la2, lo2 = f(a.get('latitude')), f(a.get('longitude')), f(b.get('latitude')), f(b.get('longitude'))
    if None in (la1, lo1, la2, lo2) or la1 == 0: return None
    x = math.radians(lo2 - lo1) * math.cos(math.radians((la1 + la2) / 2)); y = math.radians(la2 - la1)
    return int(6371000 * math.hypot(x, y))
def dt(n): return {t for t in tokens(clean_full(n or '')) if t not in GENERIC and t not in TYPE_WORDS and len(t) > 2 and not t.isdigit()}
def is_hr(c):
    parts = {p.strip() for p in (c['property_type'] or '').split(',') if p.strip()}
    return not parts or bool(parts & HR_T) or parts <= {'0', '1', '2', '3'}
ctok = [(c, dt(c['project_name'])) for c in cat]
rows = [r for r in csv.DictReader(open('FINAL-crosswalk.csv')) if r['segment'] == 'high-rise' and r['decision'] == 'NONE']
out = []
for r in rows:
    s = rc[r['scheme_id']]; st = ST.get(s['state'] or '', s['state'] or '')
    kv = st in ('Kuala Lumpur', 'Putrajaya') or (st == 'Selangor' and (s['district'] or '') in KV_SEL)
    toks = dt(s['display_name']); cands = {}
    for c, ct in ctok:
        if st and (c['state'] or '').strip() != st: continue
        ov = len(toks & ct)
        if ov:
            score = ov / max(len(toks | ct), 1)
            cands[c['uuid']] = (c, round(score, 2), 'name')
    ranked = sorted(cands.values(), key=lambda x: -x[1])[:6]
    near = []
    for c, ct in ctok:
        d = dist(s, c)
        if d is not None and d <= 1200 and is_hr(c) and c['uuid'] not in {x[0]['uuid'] for x in ranked}:
            near.append((c, d))
    near = sorted(near, key=lambda x: x[1])[:5]
    cl = [{'uuid': c['uuid'], 'name': c['project_name'], 'area': c['area'], 'state': c['state'], 'type': c['property_type'],
           'dist_m': dist(s, c), 'price_median': c['price_median'] or None, 'psf_median': c['psf_median'],
           'completion_year': c['completion_year'], 'why': w if w == 'name' else w, 'name_overlap': sc} for c, sc, w in ranked]
    cl += [{'uuid': c['uuid'], 'name': c['project_name'], 'area': c['area'], 'state': c['state'], 'type': c['property_type'],
            'dist_m': d, 'price_median': c['price_median'] or None, 'psf_median': c['psf_median'],
            'completion_year': c['completion_year'], 'why': 'near', 'name_overlap': 0} for c, d in near]
    out.append({'scheme_id': r['scheme_id'], 'name': s['display_name'], 'category': s['category'], 'state': s['state'], 'district': s['district'],
                'mukim': s['mukim'], 'lat': s['latitude'], 'lng': s['longitude'], 'coord_precision': s['precision'],
                'median_price_rm': s['median_rm'], 'psf': s['reported_psf'], 'sales_count': s['source_n'],
                'klang_valley': kv, 'first_round_reason': r['reason'], 'candidates': cl})
out.sort(key=lambda x: (not x['klang_valley'], -(x['sales_count'] or 0)))
json.dump(out, open('review/n_candidates.json', 'w'), ensure_ascii=False)
n = len(out); kv = sum(1 for x in out if x['klang_valley'])
print('cases', n, '| Klang Valley', kv, '| with name candidates', sum(1 for x in out if any(c['why'] == 'name' for c in x['candidates'])),
      '| with near high-rise', sum(1 for x in out if any(c['why'] == 'near' for c in x['candidates'])), '| no candidate at all', sum(1 for x in out if not x['candidates']))
print('KV: no candidate', sum(1 for x in out if x['klang_valley'] and not x['candidates']), '| KV sales', sum(x['sales_count'] or 0 for x in out if x['klang_valley']))
