#!/usr/bin/env python3
"""Crawler for 彭泽县-通知公告 (pengze.gov.cn /zx/03/)
   CMS: TRS WCM, pagination: index_{N}.html (25 pages)
   Detail: div.xl-text or div.trs_editor_view
"""
import urllib.request, ssl, re, sqlite3, sys, time
from datetime import datetime

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

BASE = "https://www.pengze.gov.cn"
LIST_DIR = "/zx/03/"
DB = "/root/search.db"

SITE_NAME = "pengze_tzgg"
MAX_PAGES = 25  # createPage(25, 0, ...)
INCREMENTAL = "--incremental" in sys.argv


def log(msg):
    print(f"[{SITE_NAME}] {msg}")


def fetch(url):
    req = urllib.request.Request(url, headers={
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
        "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    })
    r = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
    return r.read().decode("utf-8", errors="replace")


def parse_list(html, page_num):
    """解析 ul.thlist > li > a[href][title] + span 日期"""
    items = []
    # 找到所有 ul.thlist
    for ul_match in re.finditer(r'<ul[^>]*class="thlist"[^>]*>(.*?)</ul>', html, re.DOTALL):
        ul_html = ul_match.group(1)
        # 提取链接和标题（title可能在href前）
        a_href = re.search(r'href="([^"]+)"', ul_html)
        a_title = re.search(r'title="([^"]*)"', ul_html)
        if not (a_href and a_title):
            continue
        href = a_href.group(1).strip()
        title = a_title.group(1).strip()
        # 日期
        span_match = re.search(r'<span>([^<]*)</span>', ul_html)
        date_str = span_match.group(1).strip() if span_match else ""
        # 标准化日期格式 YYYY.MM.DD -> YYYY-MM-DD
        date_str = date_str.replace(".", "-")
        if not href.startswith("http"):
            href = BASE + href
        items.append({
            "title": title,
            "url": href,
            "date": date_str,
        })
    return items


def get_page_url(page_num):
    """第0页就是首页 index.html"""
    if page_num == 0:
        return f"{BASE}{LIST_DIR}index.html"
    return f"{BASE}{LIST_DIR}index_{page_num}.html"


def fetch_detail(url):
    """获取正文：div.xl-text 或 div.trs_editor_view"""
    html = fetch(url)
    # Try xl-text first
    m = re.search(r'<div[^>]*class="xl-text"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not m:
        # Try trs_editor_view
        m = re.search(r'<div[^>]*class="trs_editor_view[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not m:
        # Try div.content with actual content (not breadcrumb)
        m = re.search(r'<div[^>]*class="content"[^>]*>(.*?)</div>\s*</div>\s*<!--', html, re.DOTALL)
    if not m:
        log(f"未找到正文: {url}")
        return None
    inner = m.group(1)
    # Extract text from paragraphs
    parts = []
    for p in re.findall(r'<p[^>]*>(.*?)</p>', inner, re.DOTALL):
        txt = re.sub(r'<[^>]+>', '', p).strip()
        if txt:
            parts.append(txt)
    if not parts:
        txt = re.sub(r'<[^>]+>', '', inner).strip()
        if txt:
            parts = [txt]
    if not parts:
        log(f"正文为空: {url}")
        return None
    return {"content": "\n".join(parts)}


def main():
    conn = sqlite3.connect(DB, timeout=60)
    cur = conn.cursor()
    total_new = 0
    
    page_end = min(MAX_PAGES - 1, 4) if INCREMENTAL else MAX_PAGES - 1
    # Page 0 = index.html, page 1..24 = index_N.html
    
    for page_num in range(0, page_end + 1):
        url = get_page_url(page_num)
        log(f"抓取列表页 {page_num + 1}/{page_end + 1}: {url}")
        try:
            html = fetch(url)
        except Exception as e:
            log(f"列表页 {page_num} 失败: {e}")
            continue
        items = parse_list(html, page_num)
        log(f"  解析到 {len(items)} 条")
        
        for item in items:
            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if cur.fetchone():
                if INCREMENTAL:
                    log(f"  已有记录，增量结束")
                    conn.close()
                    log(f"增量完成，共 {total_new} 条新增")
                    return
                continue
            
            detail = fetch_detail(item["url"])
            if detail is None:
                continue
            
            content = detail["content"]
            now = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw 
                (site_name, source_url, page_url, title, publish_date, summary, content, status, category)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                (
                    SITE_NAME,
                    item["url"],
                    item["url"],
                    item["title"],
                    item["date"],
                    content[:200],
                    content,
                    "published",
                    "通知公告",
                )
            )
            if cur.rowcount > 0:
                total_new += 1
                if total_new % 10 == 0:
                    conn.commit()
                    log(f"  已入库 {total_new} 条...")
        
        conn.commit()
        log(f"第{page_num + 1}页完成，累计 {total_new} 条")
        time.sleep(0.3)
    
    conn.close()
    log(f"全量爬取完成，共新增 {total_new} 条")


if __name__ == "__main__":
    main()
