#!/usr/bin/env python3
"""
江苏淮安工业园区 - 信息公开 爬虫
URL: https://hipac.huaian.gov.cn/cmsweb/zwgk/gyyq/index.html
API: POST /articleCommonController/lists.do
"""
import os, re, json, time, sqlite3
from datetime import datetime, date
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
API_URL = "https://hipac.huaian.gov.cn/articleCommonController/lists.do"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Content-Type": "application/x-www-form-urlencoded"
}
SITE_NAME = "淮安工业园区信息公开"
CATEGORY = "xxgk"
THRESHOLD = date(2023, 6, 18)
R_DEPTID = "0000000064a8f16d0164ae1f25dd00ff"
PAGE_SIZE = 10

COUNT_NEW = COUNT_SKIP = COUNT_ERR = 0

def parse_date(s):
    s = s.strip().replace('.', '-')
    try:
        return datetime.strptime(s[:10], '%Y-%m-%d').date()
    except:
        return None

def date_rank(d):
    return int(d.strftime('%Y%m%d'))

def fetch_api_page(page):
    data = {
        'page': page,
        'pagesize': PAGE_SIZE,
        'rdeptid': R_DEPTID,
        'topic': ''
    }
    r = requests.post(API_URL, headers=HEADERS, data=data, timeout=20)
    return r.json()

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'html.parser')
    except Exception as e:
        return None, None, None, str(e)
    
    # Title
    title = ''
    ct = soup.select_one('div.content_title')
    if ct:
        title = ct.get_text(strip=True)
    if not title:
        t = soup.find('title')
        if t:
            title = t.get_text(strip=True).split('-')[-1].strip()
    
    # Date: PubDate meta
    pubdate = None
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta and meta.get('content'):
        pubdate = parse_date(meta['content'])
    
    # Content: div.content_d
    content_html = ''
    cd = soup.select_one('div.content_d')
    if cd:
        content_html = str(cd)
    
    return title, pubdate, content_html, None

def safe_insert(c, conn, values, max_retries=10):
    for attempt in range(max_retries):
        try:
            c.execute("""INSERT OR IGNORE INTO gov_raw 
                (site_name, title, page_url, source_url, publish_date, date_rank, summary, status, category, content, visits, tags)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 0, '')""", values)
            return c.rowcount > 0, None
        except sqlite3.OperationalError as e:
            if 'locked' in str(e):
                time.sleep(1)
                continue
            return False, str(e)
    return False, "DB locked after %d retries" % max_retries

def main():
    global COUNT_NEW, COUNT_SKIP, COUNT_ERR
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    # Get first page to determine total
    resp = fetch_api_page(1)
    if not resp.get('value') or not resp['value'].get('list'):
        print("API error:", resp.get('message', 'unknown'))
        return
    
    total = resp['value']['total']
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    print("[%s] 共 %d 条, %d 页" % (SITE_NAME, total, total_pages))
    
    for page in range(1, total_pages + 1):
        try:
            resp = fetch_api_page(page)
        except Exception as e:
            COUNT_ERR += 1
            print("  [ERR] 第%d页API: %s" % (page, e))
            continue
        
        if not resp.get('value') or not resp['value'].get('list'):
            COUNT_ERR += 1
            print("  [ERR] 第%d页API返回异常" % page)
            continue
        
        items = resp['value']['list']
        if not items:
            break
        
        for item in items:
            title = item.get('title', '')
            release_time = item.get('releaseTime', '')
            item_date = parse_date(release_time)
            
            # URL
            domain = item.get('domain', '')
            path = item.get('path', '')
            link = item.get('link', '')
            href = link if link else (domain + path)
            
            if not href:
                COUNT_ERR += 1
                continue
            
            if item_date and item_date < THRESHOLD:
                COUNT_SKIP += 1
                continue
            
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
            if c.fetchone():
                COUNT_SKIP += 1
                continue
            
            det_title, det_date, content_html, err = fetch_detail(href)
            if err:
                COUNT_ERR += 1
                if COUNT_ERR <= 5:
                    print("  [ERR] %s: %s" % (href[:60], err))
                continue
            
            final_title = det_title or title
            final_date = det_date or item_date
            if not final_date:
                COUNT_SKIP += 1
                continue
            
            if not content_html or len(content_html.strip()) < 50:
                COUNT_SKIP += 1
                continue
            
            date_str_final = final_date.strftime('%Y-%m-%d')
            dr = date_rank(final_date)
            summary = BeautifulSoup(content_html, 'html.parser').get_text(strip=True)[:200]
            
            values = (SITE_NAME, final_title, href, href, date_str_final, dr, summary, 'published', CATEGORY, content_html)
            ok, e = safe_insert(c, conn, values)
            if ok:
                COUNT_NEW += 1
                if COUNT_NEW <= 3:
                    print("  + %s | %s" % (final_title[:50], date_str_final))
            elif e:
                COUNT_ERR += 1
                if COUNT_ERR <= 5:
                    print("  [ERR] 入库: %s" % e)
    
        if page % 20 == 0:
            conn.commit()
            print("  ... %d/%d页 新增%d 跳过%d 错误%d" % (page, total_pages, COUNT_NEW, COUNT_SKIP, COUNT_ERR))
        
        time.sleep(0.2)
    
    conn.commit()
    conn.close()
    
    print("\n===== %s 完成 =====" % SITE_NAME)
    print("新增: %d" % COUNT_NEW)
    print("跳过: %d" % COUNT_SKIP)
    print("错误: %d" % COUNT_ERR)

if __name__ == '__main__':
    main()
