#!/usr/bin/env python3
"""
镇宁县人民政府 - 通知公告 爬虫
URL: https://www.gzzn.gov.cn/xw/tzgg/
CMS: TRS (createPageHTML 247页 3680条)
"""
import os, sys, re, json, time, hashlib, sqlite3
from datetime import datetime, date
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE = "https://www.gzzn.gov.cn/xw/tzgg/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
SITE_NAME = "镇宁县通知公告"
CATEGORY = "tzgg"
THRESHOLD = date(2023, 6, 18)

COUNT_NEW = 0
COUNT_SKIP = 0
COUNT_ERR = 0

def get_soup(url, timeout=20):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = 'utf-8'
    return BeautifulSoup(r.text, 'html.parser')

def parse_date(s):
    s = s.strip().replace('.', '-')
    try:
        return datetime.strptime(s[:10], '%Y-%m-%d').date()
    except:
        return None

def date_rank(d):
    return int(d.strftime('%Y%m%d'))

def fetch_detail(url):
    try:
        soup = get_soup(url, timeout=20)
    except Exception as e:
        return None, None, None, str(e)
    
    title = soup.find('title')
    title_text = title.get_text(strip=True) if title else ''
    
    pubdate = None
    meta = soup.find('meta', attrs={'name': 'PubDate'}) or soup.find('meta', attrs={'name': 'pubdate'})
    if meta and meta.get('content'):
        pubdate = parse_date(meta['content'])
    
    content_div = soup.find('div', class_='TRS_UEDITOR')
    if not content_div:
        content_div = soup.find('div', class_='trs_editor_view')
    if not content_div:
        content_div = soup.find('div', class_='view')
    if not content_div:
        content_div = soup.find('div', class_='Content')
    
    content_html = str(content_div) if content_div else ''
    return title_text, pubdate, content_html, None

def safe_insert(c, conn, values, max_retries=10):
    """带重试的INSERT"""
    for attempt in range(max_retries):
        try:
            c.execute("""INSERT OR IGNORE INTO gov_raw 
                (site_name, title, page_url, source_url, publish_date, date_rank, summary, status, category, content, visits, tags)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 0, '')""", values)
            if c.rowcount > 0:
                return True, None
            return False, None  # 重复
        except sqlite3.OperationalError as e:
            if 'locked' in str(e):
                time.sleep(1)
                continue
            return False, str(e)
    return False, "DB locked after %d retries" % max_retries

def main():
    global COUNT_NEW, COUNT_SKIP, COUNT_ERR
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    soup = get_soup(BASE)
    total_pages = 0
    for s in soup.find_all('script'):
        if s.string and 'createPageHTML' in s.string:
            m = re.search(r'createPageHTML\((\d+)', s.string)
            if m:
                total_pages = int(m.group(1))
                break
    
    print("[%s] 共 %d 页" % (SITE_NAME, total_pages + 1))
    
    for page in range(total_pages + 1):
        url = BASE if page == 0 else "%sindex_%d.html" % (BASE, page)
        
        try:
            soup = get_soup(url, timeout=20)
        except Exception as e:
            COUNT_ERR += 1
            print("  [ERR] 第%d页: %s" % (page, e))
            continue
        
        items = soup.select('ul.NewsList > li')
        if not items:
            continue
        
        for li in items:
            a = li.find('a')
            span = li.find('span')
            if not a or not span:
                continue
            
            href = a.get('href', '').strip()
            title = a.get('title', '') or a.get_text(strip=True)
            date_str = span.get_text(strip=True)
            item_date = parse_date(date_str)
            
            if not href:
                continue
            href = urljoin(BASE, href) if not href.startswith('http') else href
            
            if item_date and item_date < THRESHOLD:
                COUNT_SKIP += 1
                continue
            
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
            if c.fetchone():
                COUNT_SKIP += 1
                continue
            
            det_title, det_date, content_html, err = fetch_detail(href)
            if err:
                COUNT_ERR += 1
                print("  [ERR] %s: %s" % (href[:60], err))
                continue
            
            final_title = det_title or title
            final_date = det_date or item_date
            
            if not final_date:
                COUNT_SKIP += 1
                continue
            
            if not content_html or len(content_html.strip()) < 50:
                COUNT_SKIP += 1
                continue
            
            date_str_final = final_date.strftime('%Y-%m-%d')
            dr = date_rank(final_date)
            summary = BeautifulSoup(content_html, 'html.parser').get_text(strip=True)[:200]
            
            values = (SITE_NAME, final_title, href, href, date_str_final, dr, summary, 'published', CATEGORY, content_html)
            ok, e = safe_insert(c, conn, values)
            if ok:
                COUNT_NEW += 1
                if COUNT_NEW <= 3:
                    print("  + %s | %s" % (final_title[:50], date_str_final))
            elif e:
                COUNT_ERR += 1
                print("  [ERR] 入库: %s" % e)
        
        if page % 20 == 0 and page > 0:
            conn.commit()
            print("  ... %d/%d页 新增%d 跳过%d 错误%d" % (page + 1, total_pages + 1, COUNT_NEW, COUNT_SKIP, COUNT_ERR))
        
        time.sleep(0.3)
    
    conn.commit()
    conn.close()
    
    print("\n===== %s 完成 =====" % SITE_NAME)
    print("新增: %d" % COUNT_NEW)
    print("跳过: %d" % COUNT_SKIP)
    print("错误: %d" % COUNT_ERR)

if __name__ == '__main__':
    main()
