#!/usr/bin/env python3
"""Crawl 奉新县人民政府 - 公告公示 via API
http://www.fengxin.gov.cn/fxxrmzf/gggs3a30/pc/list.html
"""
import requests, re, sqlite3, sys, time, hashlib, json
from datetime import datetime, timedelta
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '奉新县人民政府 - 公告公示'
API_URL = 'https://www.fengxin.gov.cn/queryList'
CHANNEL_CODE = 'gggs3a30'
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
PAGE_SIZE = 15

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Content-Type': 'application/json',
}
session = requests.Session()
session.headers.update(HEADERS)

def fetch_page(page):
    payload = {"current": page, "pageSize": PAGE_SIZE, "channelCode": CHANNEL_CODE}
    resp = session.post(API_URL, json=payload, timeout=30)
    data = resp.json()
    if 'data' not in data:
        return [], 0
    total = data['data'].get('total', 0)
    items = []
    for r in data['data'].get('results', []):
        s = r['source']
        title = s.get('title', '').strip()
        pub_date = s.get('pubDate', '')[:10]
        content_obj = s.get('content', {})
        content = content_obj.get('content', '') if isinstance(content_obj, dict) else (content_obj if isinstance(content_obj, str) else '')
        urls = s.get('urls', '{}')
        if isinstance(urls, str):
            try: urls = json.loads(urls)
            except: urls = {}
        url = urls.get('pc', '') if isinstance(urls, dict) else ''
        if url and not url.startswith('http'):
            url = 'http://www.fengxin.gov.cn' + url
        if title and url:
            items.append({'title': title, 'url': url, 'date': pub_date, 'content': content})
    return items, total

def main():
    is_incremental = 'incremental' in sys.argv
    print(f'=== {SITE_NAME} ===')
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = skipped = old_total = 0
    start = time.time()

    _, total = fetch_page(1)
    total_pages = max(1, (total + PAGE_SIZE - 1) // PAGE_SIZE)
    max_pages = 1 if is_incremental else total_pages
    print(f'  Total: {total}, pages: {total_pages}')

    consecutive_old = 0
    for pg in range(1, max_pages + 1):
        try:
            items, _ = fetch_page(pg)
        except Exception as e:
            print(f'  [ERR] Page {pg}: {e}')
            time.sleep(2)
            continue
        if not items:
            break
        all_old = all(it['date'] and it['date'] < CUTOFF_DATE for it in items if it['date'])
        if all_old:
            consecutive_old += 1
            if consecutive_old >= 2 and pg >= 5:
                print(f'  Page {pg}: all old, stop')
                break
        else:
            consecutive_old = 0
        for it in items:
            if it['date'] and it['date'] < CUTOFF_DATE:
                old_total += 1
                continue
            has = c.execute("SELECT 1 FROM gov_raw WHERE page_url=? AND content IS NOT NULL AND content!=''", (it['url'],)).fetchone()
            if has:
                skipped += 1
                continue
            content = re.sub(r'<script[^>]*>.*?</script>', '', it['content'], flags=re.DOTALL|re.I)
            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
            fid = int(hashlib.md5(it['url'].encode()).hexdigest()[:15], 16) % (2**63)
            dr = int(it['date'].replace('-','')) if it['date'] else 0
            summary = re.sub(r'<[^>]+>', '', content)[:200] if content else it['title']
            summary = re.sub(r'\s+', ' ', summary).strip()
            c.execute('INSERT OR IGNORE INTO gov_raw(id,site_name,source_url,page_url,title,publish_date,content,summary,date_rank) VALUES(?,?,?,?,?,?,?,?,?)',
                      (fid, SITE_NAME, it['url'], it['url'], it['title'], it['date'], content, summary, dr))
            if c.rowcount > 0:
                inserted += 1
        print(f'  Page {pg}/{max_pages}: +{inserted} (skip {skipped}, old {old_total}) [{time.time()-start:.0f}s]')
    conn.commit()
    c2 = conn.cursor()
    c2.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
    rows = c2.execute("SELECT id,title,site_name,summary FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall()
    for r in rows:
        c2.execute("INSERT OR IGNORE INTO gov_search(rowid,title,site_name,summary) VALUES(?,?,?,?)", r)
    conn.commit()
    conn.close()
    print(f'\nDone ({time.time()-start:.0f}s). +{inserted}, skip {skipped}, old {old_total}')

if __name__ == '__main__':
    main()
