#!/usr/bin/env python3
"""
信丰县人民政府 - 通知公告 爬虫
URL: http://www.jxxf.gov.cn/xfxrmzfyyh/c104550/list.shtml
CMS: TRS (createPageHTML, 14页, 164条)
"""
import os, sys, re, json, time, sqlite3
from datetime import datetime, date
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE = "http://www.jxxf.gov.cn/xfxrmzfyyh/c104550/list.shtml"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
SITE_NAME = "信丰县通知公告"
CATEGORY = "tzgg"
THRESHOLD = date(2023, 6, 18)

COUNT_NEW = COUNT_SKIP = COUNT_ERR = 0

def get_soup(url, timeout=20):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = 'utf-8'
    return BeautifulSoup(r.text, 'html.parser')

def parse_date(s):
    s = s.strip().replace('.', '-')
    try:
        return datetime.strptime(s[:10], '%Y-%m-%d').date()
    except:
        return None

def date_rank(d):
    return int(d.strftime('%Y%m%d'))

def fetch_detail(url):
    try:
        soup = get_soup(url, timeout=20)
    except Exception as e:
        return None, None, None, str(e)
    
    # Title: ArticleTitle meta or UCAPTITLE or <title>
    title = ''
    meta = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta and meta.get('content'):
        title = meta['content'].strip()
    if not title:
        ucap = soup.find('ucaptitle')
        if ucap:
            title = ucap.get_text(strip=True)
    if not title:
        t = soup.find('title')
        if t:
            title = t.get_text(strip=True).split('|')[0].strip()
    
    # Date: PubDate meta
    pubdate = None
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta and meta.get('content'):
        pubdate = parse_date(meta['content'])
    
    # Content: div#zoomcon.article-content-body or div.article-content
    content_div = soup.find('div', id='zoomcon')
    if not content_div:
        content_div = soup.find('div', class_='article-content')
    if not content_div:
        content_div = soup.find('div', class_='article-content-body')
    
    content_html = str(content_div) if content_div else ''
    return title, pubdate, content_html, None

def safe_insert(c, conn, values, max_retries=10):
    for attempt in range(max_retries):
        try:
            c.execute("""INSERT OR IGNORE INTO gov_raw 
                (site_name, title, page_url, source_url, publish_date, date_rank, summary, status, category, content, visits, tags)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 0, '')""", values)
            return c.rowcount > 0, None
        except sqlite3.OperationalError as e:
            if 'locked' in str(e):
                time.sleep(1)
                continue
            return False, str(e)
    return False, "DB locked after %d retries" % max_retries

def main():
    global COUNT_NEW, COUNT_SKIP, COUNT_ERR
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    soup = get_soup(BASE)
    total_pages = 0
    for s in soup.find_all('script'):
        if s.string and 'createPageHTML' in s.string:
            m = re.search(r'createPageHTML\([^,]+,\s*(\d+)', s.string)
            if m:
                total_pages = int(m.group(1))
                break
    
    print("[%s] 共 %d 页" % (SITE_NAME, total_pages))
    
    for page in range(1, total_pages + 1):
        url = BASE if page == 1 else "http://www.jxxf.gov.cn/xfxrmzfyyh/c104550/list_%d.shtml" % page
        
        try:
            soup = get_soup(url, timeout=20)
        except Exception as e:
            COUNT_ERR += 1
            print("  [ERR] 第%d页: %s" % (page, e))
            continue
        
        items = soup.select('div.pageList > ul > li')
        if not items:
            print("  [WARN] 第%d页无列表项" % page)
            continue
        
        for li in items:
            a = li.find('a')
            span = li.find('span', class_='time')
            if not a or not span:
                continue
            
            href = a.get('href', '').strip()
            title = a.get('title', '') or a.get_text(strip=True)
            date_str = span.get_text(strip=True)
            item_date = parse_date(date_str)
            
            if not href:
                continue
            href = urljoin(BASE, href) if not href.startswith('http') else href
            
            if item_date and item_date < THRESHOLD:
                COUNT_SKIP += 1
                continue
            
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
            if c.fetchone():
                COUNT_SKIP += 1
                continue
            
            det_title, det_date, content_html, err = fetch_detail(href)
            if err:
                COUNT_ERR += 1
                print("  [ERR] %s: %s" % (href[:60], err))
                continue
            
            final_title = det_title or title
            final_date = det_date or item_date
            if not final_date:
                COUNT_SKIP += 1
                continue
            
            if not content_html or len(content_html.strip()) < 50:
                COUNT_SKIP += 1
                continue
            
            date_str_final = final_date.strftime('%Y-%m-%d')
            dr = date_rank(final_date)
            summary = BeautifulSoup(content_html, 'html.parser').get_text(strip=True)[:200]
            
            values = (SITE_NAME, final_title, href, href, date_str_final, dr, summary, 'published', CATEGORY, content_html)
            ok, e = safe_insert(c, conn, values)
            if ok:
                COUNT_NEW += 1
                if COUNT_NEW <= 3:
                    print("  + %s | %s" % (final_title[:50], date_str_final))
            elif e:
                COUNT_ERR += 1
                print("  [ERR] 入库: %s" % e)
        
        time.sleep(0.3)
    
    conn.commit()
    conn.close()
    
    print("\n===== %s 完成 =====" % SITE_NAME)
    print("新增: %d" % COUNT_NEW)
    print("跳过: %d" % COUNT_SKIP)
    print("错误: %d" % COUNT_ERR)

if __name__ == '__main__':
    main()
