"""Build review packets for the high-rise NONE second round from review/n_candidates.json.

Adds realtycheck's own page (address line + recent sales with floors), the scheme's sales from the
database, each candidate's recorded highest/lowest sale, transaction window and source URLs, and a
`same_sale` flag when a realtycheck sale equals a candidate's recorded high/low (same month, price
within 2%). Cases with no candidate outside Klang Valley stay NONE and get no packet.
Writes review/n_{kv|rest}_NN_in.jsonl in priority order (Klang Valley first, most sales first).
"""
import json, os, re, sys

S = 'review'  # n_extra.json (candidate history + realtycheck sales), copied from the session scratchpad
R = 'review'; BATCH = int(sys.argv[1]) if len(sys.argv) > 1 else 24
cases = json.load(open(f'{R}/n_candidates.json'))
extra = json.load(open(f'{S}/n_extra.json'))
urls = json.load(open('realtycheck_urls.json'))


def page(sid):
    u = urls.get(sid)
    fn = f"{R}/_rc_pages/{u.rstrip('/').split('/')[-1]}.txt" if u else None
    if not fn or not os.path.exists(fn):
        return u, '', ''
    t = open(fn).read()
    a = re.search(r'transacted prices to \d{4}-\d{2}-\d{2} (.+?) (?:Nearest station|Median transacted)', t)
    s = re.search(r'Recent registered sales (.+?) (?:Buying or selling|Get your full)', t)
    return u, (a.group(1)[:220] if a else ''), (s.group(1)[:700] if s else '')


def num(x):
    try:
        return float(x)
    except (TypeError, ValueError):
        return None


packets = []
for i, c in enumerate(cases, 1):
    if not c['candidates'] and not c['klang_valley']:
        continue
    sales = extra['sales'].get(c['scheme_id'], [])
    url, addr, table = page(c['scheme_id'])
    cands = []
    for k, cd in enumerate(c['candidates']):
        p = extra['proj'].get(cd['uuid'], {}) or {}
        hist = []
        for side in ('high', 'low'):
            psf, sqft, date = num(p.get(f'historical_{side}_psf')), num(p.get(f'historical_{side}_sqft')), p.get(f'historical_{side}_date')
            if psf and sqft and date:
                hist.append({'side': side, 'date': str(date)[:10], 'sqft': int(sqft), 'psf': psf, 'price': round(psf * sqft)})
        same = [h for h in hist for sm in sales
                if sm[0] == h['date'][:7] and sm[2] and abs(sm[2] - h['price']) / sm[2] <= 0.02]
        cands.append({'k': k, **cd, 'address': p.get('address'), 'total_transactions': p.get('total_transactions'),
                      'transaction_period': p.get('transaction_period'), 'recorded_high_low': hist,
                      'same_sale_as_realtycheck': same, 'sources': extra['src'].get(cd['uuid'], [])})
    m = extra['months'].get(c['scheme_id'], {})
    packets.append({'case': f'n{i:03d}', 'scheme_id': c['scheme_id'], 'klang_valley': c['klang_valley'],
                    'realtycheck': {k: c[k] for k in ('name', 'category', 'state', 'district', 'mukim', 'lat', 'lng', 'coord_precision',
                                                      'median_price_rm', 'psf', 'sales_count')} | {
                        'months': [m.get('a'), m.get('b'), m.get('n')], 'url': url, 'page_address': addr,
                        'page_recent_sales_with_floors': table, 'recent_sales_[month,sqft,price,psf]': sales},
                    'first_round_reason': c['first_round_reason'], 'catalogue_candidates': cands})

kv = [p for p in packets if p['klang_valley']]; rest = [p for p in packets if not p['klang_valley']]
for tag, group in (('kv', kv), ('rest', rest)):
    for j in range(0, len(group), BATCH):
        with open(f'{R}/n_{tag}_{j // BATCH + 1:02d}_in.jsonl', 'w') as fh:
            for p in group[j:j + BATCH]:
                fh.write(json.dumps(p, ensure_ascii=False) + '\n')
print('packets', len(packets), '| kv', len(kv), 'in', -(-len(kv) // BATCH), 'batches | rest', len(rest), 'in', -(-len(rest) // BATCH),
      'batches | same-sale hits', sum(1 for p in packets if any(c['same_sale_as_realtycheck'] for c in p['catalogue_candidates'])),
      '| with page address', sum(1 for p in packets if p['realtycheck']['page_address']))
