#!/usr/bin/env python3
"""Crawler for 万山网-通知公告 (wsxw.gov.cn)
   https://www.wsxw.gov.cn/tzgg/
   CMS: 多彩博虹, pagination: index_{N}.shtml
   Detail: div#Zoom
"""
import urllib.request, ssl, re, sqlite3, sys, time, hashlib
from datetime import datetime

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

BASE = "https://www.wsxw.gov.cn"
LIST_DIR = "/tzgg/"
DB = "/root/search.db"

SITE_NAME = "wsxw_tzgg"
MAX_PAGES = 39
INCREMENTAL = "--incremental" in sys.argv


def log(msg):
    print(f"[{SITE_NAME}] {msg}")


def fetch(url):
    req = urllib.request.Request(url, headers={
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
        "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    })
    r = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
    return r.read().decode("utf-8", errors="replace")


def parse_list(html, page_num):
    """解析 ul.NewsList > li"""
    items = []
    ul_match = re.search(r'<ul[^>]*class="NewsList"[^>]*>(.*?)</ul>', html, re.DOTALL)
    if not ul_match:
        log(f"第{page_num}页未找到 ul.NewsList")
        return items
    ul_html = ul_match.group(1)
    li_pattern = re.compile(r'<li>(.*?)</li>', re.DOTALL)
    for li_match in li_pattern.finditer(ul_html):
        li = li_match.group(1)
        a_href = re.search(r'href="([^"]+)"', li)
        a_title = re.search(r'title="([^"]+)"', li)
        if not (a_href and a_title):
            continue
        href = a_href.group(1).strip()
        title = a_title.group(1).strip()
        date_match = re.search(r'<span>([^<]*)</span>', li)
        date_str = date_match.group(1).strip() if date_match else ""
        if href.startswith("http") and "wsxw.gov.cn" not in href:
            log(f"跳过外部: {title}")
            continue
        if not href.startswith("/tzgg/"):
            continue
        items.append({
            "title": title,
            "url": BASE + href,
            "date": date_str,
        })
    return items


def get_page_url(page_num):
    if page_num == 1:
        return f"{BASE}{LIST_DIR}index.shtml"
    return f"{BASE}{LIST_DIR}index_{page_num}.shtml"


def fetch_detail(url):
    """获取详情页正文 div#Zoom"""
    html = fetch(url)
    zoom_match = re.search(r'<div[^>]*id="Zoom"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if not zoom_match:
        zoom_match = re.search(r'<div[^>]*id="Zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not zoom_match:
        log(f"未找到 div#Zoom: {url}")
        return None
    zoom_html = zoom_match.group(1)
    # Extract <p> tags, preserve as newline-separated text
    p_pattern = re.compile(r'<p[^>]*>(.*?)</p>', re.DOTALL)
    parts = []
    for p_match in p_pattern.finditer(zoom_html):
        txt = re.sub(r'<[^>]+>', '', p_match.group(1)).strip()
        if txt:
            parts.append(txt)
    if not parts:
        txt = re.sub(r'<[^>]+>', '', zoom_html).strip()
        if txt:
            parts = [txt]
    if not parts:
        log(f"正文为空: {url}")
        return None
    content = "\n".join(parts)
    return {"content": content}


def main():
    conn = sqlite3.connect(DB, timeout=60)
    cur = conn.cursor()
    
    total_new = 0
    page_end = min(MAX_PAGES, 5) if INCREMENTAL else MAX_PAGES
    
    for page_num in range(1, page_end + 1):
        url = get_page_url(page_num)
        log(f"抓取列表页 {page_num}/{page_end}: {url}")
        try:
            html = fetch(url)
        except Exception as e:
            log(f"列表页 {page_num} 失败: {e}")
            continue
        items = parse_list(html, page_num)
        log(f"  解析到 {len(items)} 条")
        
        for item in items:
            # Check duplicate by page_url
            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if cur.fetchone():
                if INCREMENTAL:
                    log(f"  已有记录，增量结束")
                    conn.close()
                    log(f"增量完成，共 {total_new} 条新增")
                    return
                continue
            
            detail = fetch_detail(item["url"])
            if detail is None:
                continue
            
            content = detail["content"]
            now = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw 
                (site_name, source_url, page_url, title, publish_date, summary, content, status, category)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                (
                    SITE_NAME,
                    item["url"],
                    item["url"],
                    item["title"],
                    item["date"],
                    content[:200],
                    content,
                    "published",
                    "通知公告",
                )
            )
            if cur.rowcount > 0:
                total_new += 1
                if total_new % 10 == 0:
                    conn.commit()
                    log(f"  已入库 {total_new} 条...")
        
        conn.commit()
        log(f"第{page_num}页完成，累计 {total_new} 条")
        time.sleep(0.3)
    
    conn.close()
    log(f"全量爬取完成，共新增 {total_new} 条")


if __name__ == "__main__":
    main()
