#!/usr/bin/env python3
"""
江阴市人民政府 - 乡镇专栏通知公告 爬虫
URL: https://www.jiangyin.gov.cn/zt/xzzl/tzgg/index.shtml
CMS: 自定义 (20页, 每页20条)
"""
import os, re, json, time, sqlite3
from datetime import datetime, date
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE = "https://www.jiangyin.gov.cn/zt/xzzl/tzgg/index.shtml"
BASE_DIR = "https://www.jiangyin.gov.cn/zt/xzzl/tzgg/"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
SITE_NAME = "江阴乡镇通知公告"
CATEGORY = "tzgg"
THRESHOLD = date(2023, 6, 18)

COUNT_NEW = COUNT_SKIP = COUNT_ERR = 0

def get_soup(url, timeout=20):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = 'utf-8'
    return BeautifulSoup(r.text, 'html.parser')

def parse_date(s):
    s = s.strip().replace('.', '-')
    try:
        return datetime.strptime(s[:10], '%Y-%m-%d').date()
    except:
        return None

def date_rank(d):
    return int(d.strftime('%Y%m%d'))

def fetch_detail(url):
    try:
        soup = get_soup(url, timeout=20)
    except Exception as e:
        return None, None, None, str(e)
    
    title = ''
    t_div = soup.select_one('div.sdgjz_art_title')
    if t_div:
        title = t_div.get_text(strip=True)
    
    pubdate = None
    time_div = soup.select_one('div.sdgjz_art_time')
    if time_div:
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', time_div.get_text())
        if m:
            pubdate = parse_date(m.group(1))
    
    content_html = ''
    cont_div = soup.select_one('div.sdgjz_art_cont')
    if cont_div:
        content_html = str(cont_div)
    
    if not title:
        t = soup.find('title')
        if t:
            title = t.get_text(strip=True).split('|')[0].strip()
    
    return title, pubdate, content_html, None

def safe_insert(c, conn, values, max_retries=10):
    for attempt in range(max_retries):
        try:
            c.execute("""INSERT OR IGNORE INTO gov_raw 
                (site_name, title, page_url, source_url, publish_date, date_rank, summary, status, category, content, visits, tags)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 0, '')""", values)
            return c.rowcount > 0, None
        except sqlite3.OperationalError as e:
            if 'locked' in str(e):
                time.sleep(1)
                continue
            return False, str(e)
    return False, "DB locked after %d retries" % max_retries

def main():
    global COUNT_NEW, COUNT_SKIP, COUNT_ERR
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    soup = get_soup(BASE)
    total_pages = 0
    for s in soup.find_all('script'):
        if s.string and 'pageCount' in s.string:
            m = re.search(r'"pageCount":"(\d+)"', s.string)
            if m:
                total_pages = int(m.group(1))
                break
    
    print("[%s] 共 %d 页" % (SITE_NAME, total_pages))
    
    for page in range(1, total_pages + 1):
        if page == 1:
            url = BASE
        else:
            url = "%sindex_%d.shtml" % (BASE_DIR, page)
        
        try:
            soup = get_soup(url, timeout=20)
        except Exception as e:
            COUNT_ERR += 1
            print("  [ERR] 第%d页: %s" % (page, e))
            continue
        
        items = soup.select('div.jczw-list > ul > li')
        if not items:
            continue
        
        for li in items:
            a = li.find('a')
            span = li.find('span')
            if not a or not span:
                continue
            
            href = a.get('href', '').strip()
            # Clean up title - remove <!--[镇名]--> comments
            raw_title = a.get('title', '') or a.get_text(strip=True)
            title = re.sub(r'<!--.*?-->', '', raw_title).strip()
            date_str = span.get_text(strip=True)
            item_date = parse_date(date_str)
            
            if not href:
                continue
            href = urljoin(BASE, href) if not href.startswith('http') else href
            
            if item_date and item_date < THRESHOLD:
                COUNT_SKIP += 1
                continue
            
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
            if c.fetchone():
                COUNT_SKIP += 1
                continue
            
            det_title, det_date, content_html, err = fetch_detail(href)
            if err:
                COUNT_ERR += 1
                print("  [ERR] %s: %s" % (href[:60], err))
                continue
            
            final_title = det_title or title
            final_date = det_date or item_date
            if not final_date:
                COUNT_SKIP += 1
                continue
            
            if not content_html or len(content_html.strip()) < 50:
                COUNT_SKIP += 1
                continue
            
            date_str_final = final_date.strftime('%Y-%m-%d')
            dr = date_rank(final_date)
            summary = BeautifulSoup(content_html, 'html.parser').get_text(strip=True)[:200]
            
            values = (SITE_NAME, final_title, href, href, date_str_final, dr, summary, 'published', CATEGORY, content_html)
            ok, e = safe_insert(c, conn, values)
            if ok:
                COUNT_NEW += 1
                if COUNT_NEW <= 3:
                    print("  + %s | %s" % (final_title[:50], date_str_final))
            elif e:
                COUNT_ERR += 1
                print("  [ERR] 入库: %s" % e)
        
        time.sleep(0.3)
    
    conn.commit()
    conn.close()
    
    print("\n===== %s 完成 =====" % SITE_NAME)
    print("新增: %d" % COUNT_NEW)
    print("跳过: %d" % COUNT_SKIP)
    print("错误: %d" % COUNT_ERR)

if __name__ == '__main__':
    main()
