#!/usr/bin/env python3
import os
"""Crawler for 贵池区人民政府 — 环评审批"""
import sys, os, re, time, sqlite3, subprocess
from datetime import datetime, timedelta

BASE = 'https://www.ahgc.gov.cn'
LIST_TMPL = '/OpennessTarget/183/101924/page_{}.html'
DETAIL_TMPL = '/OpennessContent/show/{}.html'
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
SITE_NAME = '贵池区环评审批'
MAX_PAGES = 43

def fetch(url, retries=3):
    for i in range(retries):
        try:
            r = subprocess.run(['curl', '-sL', '--max-time', '15', url],
                capture_output=True, timeout=20)
            if r.returncode == 0 and r.stdout:
                return r.stdout.decode('utf-8', errors='replace')
        except: pass
        time.sleep(2)
    return ''

def extract_list(html):
    items = re.findall(r'<td class="bt"><a href="/OpennessContent/show/(\d+)\.html"[^>]*>(.*?)</a>', html, re.DOTALL)
    dates = re.findall(r'<td class="cwrq">([^<]+)</td>', html)
    return [(item_id, dates[i] if i < len(dates) else '') for i, (item_id, _) in enumerate(items)]

def extract_detail(html):
    title = pub_date = ''
    m = re.search(r'<meta name="ArticleTitle"[^>]*content="([^"]*)"', html)
    if m: title = m.group(1).strip()
    m = re.search(r'<meta name="PubDate"[^>]*content="([^"]*)"', html)
    if m: pub_date = m.group(1).strip()
    content = ''
    m = re.search(r'<div[^>]*id="zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m: content = m.group(1).strip()
    for fm in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|docx?|xlsx?|zip))"[^>]*>([^<]+)</a>', html, re.DOTALL):
        furl = fm.group(1)
        fname = re.sub(r'<[^>]+>', '', fm.group(2)).strip()
        if not furl.startswith('http'): furl = BASE + furl
        content += '\n<p><a href="{}" target="_blank">[文件] {}</a></p>'.format(furl, fname)
    return title, pub_date, content

def main():
    total = 0
    db = sqlite3.connect(DB_PATH, timeout=60)
    for page in range(1, MAX_PAGES + 1):
        url = BASE + LIST_TMPL.format(page)
        print('  Page {}...'.format(page), end=' ', flush=True)
        html = fetch(url)
        if not html: print('empty'); continue
        items = extract_list(html)
        if not items: print('no items, done'); break
        all_before = all(d < CUTOFF for _, d in items)
        for item_id, date_str in items:
            if date_str < CUTOFF: continue
            detail_html = fetch(BASE + DETAIL_TMPL.format(item_id))
            if not detail_html: continue
            title, pub_date, content = extract_detail(detail_html)
            if not title: continue
            summary = re.sub(r'<[^>]+>', '', content)[:200].strip()
            sql = "INSERT OR REPLACE INTO gov_raw (id, title, content, publish_date, source_url, page_url, site_name, summary) VALUES (?, ?, ?, ?, ?, ?, ?, ?)"
            detail_url = BASE + DETAIL_TMPL.format(item_id)
            db.execute(sql, (int(item_id), title, content, pub_date[:10], detail_url, detail_url, SITE_NAME, summary))
            total += 1
        db.commit()
        print('{} items, {} total'.format(len(items), total))
        if all_before: print('  All before cutoff, stopping'); break
        time.sleep(0.3)
    db.close()
    print('\nDone! {} records'.format(total))

if __name__ == '__main__':
    main()
