"""Second check for HOLD re-check LINKs: read each realtycheck scheme page's own address line
(postcode + town, reverse-geocoded by realtycheck from its pin) and compare it with the linked
catalogue project's area / address / state. Writes review/_rc_page_check.json.

Usage: python3 rc_page_check.py [batch ...]   (default: every review/h_*_out.jsonl)
Fetches are cached in review/_rc_pages/ and spaced 1 s apart.
"""
import glob, html, json, os, re, subprocess, sys, time

D = '/var/www/html/peta/storage/app/propertylab-catalogue-match'; R = f'{D}/review'; C = f'{R}/_rc_pages'
os.makedirs(C, exist_ok=True)
cat = {c['uuid']: c for c in json.load(open(f'{D}/catalogue_my_projects.json'))}


def page(url):
    fn = f"{C}/{url.rstrip('/').split('/')[-1]}.txt"
    if os.path.exists(fn):
        return open(fn).read()
    raw = subprocess.run(['curl', '-s', '-m', '25', '-A', 'Mozilla/5.0', url], capture_output=True, text=True, errors='ignore').stdout
    t = re.sub(r'<script[\s\S]*?</script>|<style[\s\S]*?</style>', ' ', raw)
    t = re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', t)))
    open(fn, 'w').write(t); time.sleep(1)
    return t


def address(t):
    m = re.search(r'transacted prices to \d{4}-\d{2}-\d{2} (.+?) (?:Nearest station|Median transacted)', t)
    return m.group(1).strip() if m else ''


def norm(s):
    return re.sub(r'[^a-z0-9 ]', ' ', (s or '').lower())


batches = sys.argv[1:] or [os.path.basename(f)[:-10] for f in sorted(glob.glob(f'{R}/h_*_out.jsonl'))]
out = json.load(open(f'{R}/_rc_page_check.json')) if os.path.exists(f'{R}/_rc_page_check.json') else {}
for b in batches:
    P = {json.loads(l)['case']: json.loads(l) for f in glob.glob(f'{R}/h_*_in.jsonl') for l in open(f) if l.strip()}
    for line in open(f'{R}/{b}_out.jsonl'):
        if not line.strip():
            continue
        v = json.loads(line); p = P[v['case']]
        if v['decision'] != 'LINK':
            continue
        uuid = v.get('uuid') or (p['catalogue_candidates'][v['cand']]['uuid'] if v.get('cand') is not None else None)
        c = cat.get(uuid, {})
        url = p['realtycheck'].get('url')
        addr = address(page(url)) if url else ''
        pc = re.findall(r'\b(\d{5})\b', addr)
        town = re.sub(r'.*\d{5}\s+', '', addr.split('·')[0]).strip() if pc else ''
        hay = norm(' '.join([c.get('area') or '', c.get('address') or '', c.get('project_name') or '']))
        towns_match = bool(town) and all(w in hay.split() for w in norm(town).split())
        pc_match = any(x in (c.get('address') or '') for x in pc)
        out[v['case']] = {'batch': b, 'scheme': p['realtycheck']['name'], 'rc_address': addr, 'rc_postcode': pc[:1], 'rc_town': town,
                          'catalogue': c.get('project_name'), 'catalogue_area': c.get('area'), 'catalogue_state': c.get('state'),
                          'catalogue_address': c.get('address'), 'town_match': towns_match, 'postcode_match': pc_match,
                          'confidence': v['confidence_pct']}
json.dump(out, open(f'{R}/_rc_page_check.json', 'w'), indent=1, ensure_ascii=False)
n = [x for x in out.values() if x['batch'] in batches]
print(len(n), 'LINKs checked |', sum(1 for x in n if x['town_match'] or x['postcode_match']), 'town/postcode agree |',
      sum(1 for x in n if not x['rc_address']), 'no address on page')
for k, x in sorted(out.items()):
    if x['batch'] in batches and not (x['town_match'] or x['postcode_match']):
        print(f"  {k} {x['scheme']} | RC: {x['rc_address'][:90]} | C: {x['catalogue']} / {x['catalogue_area']} / {x['catalogue_state']}")
