#!/usr/bin/env python3
"""
xunhua.gov.cn 循化县政府 - 部门公开爬虫
=========================================
TRS CMS，纯HTML列表，requests即可。
栏目: 10568（部门公开）

用法:
  python3 crawl_xunhua.py                # 增量爬（第1页）
  python3 crawl_xunhua.py --pages=5      # 爬5页
  python3 crawl_xunhua.py --full         # 全量所有页
"""

import os, re, sys, time, json, sqlite3
import requests
from bs4 import BeautifulSoup

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")

SITE_NAME = "循化县人民政府"
DOMAIN = "www.xunhua.gov.cn"
CATEGORY_ID = 10568
CATEGORY_NAME = "部门公开"
GROUP = "青海-海东"
INDUSTRY = "政府公告"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"}

BASE_URL = "http://www.xunhua.gov.cn"
LIST_BASE = BASE_URL + "/html/%d/List" % CATEGORY_ID
DEFAULT_PAGES = 1
MAX_PAGES = 10
stats = {"new": 0, "skip": 0, "errors": 0}


def get_total_pages():
    """从首页获取总页数"""
    r = requests.get(LIST_BASE + ".html", headers=HEADERS, timeout=15)
    r.encoding = "utf-8"
    nums = re.findall(r"List-(\d+)\.html", r.text)
    if nums:
        return max(int(n) for n in nums) + 1
    return MAX_PAGES


def fetch_list_page(page_num):
    """获取列表页"""
    if page_num == 0:
        url = LIST_BASE + ".html"
    else:
        url = LIST_BASE + "-%d.html" % page_num
    r = requests.get(url, headers=HEADERS, timeout=15)
    r.encoding = "utf-8"
    return r.text


def parse_items(html):
    """解析列表页，返回 [(title, url, date_text), ...]"""
    items = []
    soup = BeautifulSoup(html, "lxml")
    # TRS CMS 列表链接模式：/html/{cat_id}/{id}.html
    for a in soup.find_all("a"):
        href = a.get("href", "")
        txt = a.get_text(strip=True)
        if len(txt) > 10 and re.search(r"/html/%d/\d+\.html$" % CATEGORY_ID, href):
            full_url = href
            # 日期可能在链接后的文本节点或li结构中
            items.append((txt, full_url, ""))
    return items


def fetch_detail(url):
    """获取详情页内容"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")
        
        # 标题
        title = ""
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()
        if not title:
            td = soup.select_one("td.heicu26")
            if td:
                title = td.get_text(strip=True)
        
        # 发布日期 - 从PubDate meta
        date = ""
        meta_d = soup.find("meta", attrs={"name": "PubDate"})
        if meta_d and meta_d.get("content"):
            raw = meta_d["content"].strip()
            m = re.search(r"(\d{4})[/-](\d{2})[/-](\d{2})", raw)
            if m:
                date = "%s-%s-%s" % (m.group(1), m.group(2), m.group(3))
        if not date:
            # 从页面文本："时间：XXXX年XX月XX日"
            m = re.search(r"时间.*?(\d{4})年(\d{2})月(\d{2})日", r.text)
            if m:
                date = "%s-%s-%s" % (m.group(1), m.group(2), m.group(3))
        
        # 正文提取
        body_html = ""
        summary = ""
        zoom = soup.select_one("#zoom")
        if zoom:
            for tag in zoom.find_all(["script", "style"]):
                tag.decompose()
            body_html = str(zoom)
            summary = zoom.get_text(strip=True)[:200]
        else:
            # TRS CMS: 正文在嵌套table中。找不含breadcrumb的最小table
            candidates = []
            for table in soup.find_all("table"):
                txt = table.get_text(strip=True)
                if len(txt) > 100 and "。﻿" in txt + "。":
                    # 排除含breadcrumb的（首页> 政府信息...）
                    if "首页" not in txt[:30]:
                        candidates.append((len(txt), str(table), txt))
            if candidates:
                # 取最长文本的table（最内层通常文本最完整）
                candidates.sort(key=lambda x: -x[0])
                body_html = candidates[0][1]
                summary = candidates[0][2][:200]
            if not body_html:
                # 终极fallback: body全文（剥离script/style）
                body = soup.find("body")
                if body:
                    for s in body.find_all(["script", "style"]):
                        s.decompose()
                    txt = body.get_text(strip=True)
                    # 去掉前面breadcrumb
                    idx = txt.find("部门公开")
                    if idx >= 0:
                        txt = txt[idx+4:]
                    if txt:
                        body_html = txt
                        summary = txt[:200]
        
        return title, date, body_html, summary
    except Exception as e:
        print("    detail error: %s" % e, file=sys.stderr)
        return "", "", "", ""


def store_record(title, page_url, publish_date, body_html="", summary=""):
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(title, page_url, source_url, site_name, publish_date, category, industry, group_name, content, summary) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (title.strip(), page_url, DOMAIN, SITE_NAME, publish_date,
             CATEGORY_NAME, INDUSTRY, GROUP, body_html, summary),
        )
        conn.commit()
        is_new = conn.total_changes > 0
        
        if not is_new and body_html:
            row = conn.execute("SELECT id, content FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row and (not row[1] or row[1].strip() == ""):
                conn.execute("UPDATE gov_raw SET content=?, summary=? WHERE id=?", (body_html, summary, row[0]))
                conn.commit()
                is_new = True
        
        if is_new:
            row = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row:
                _c2 = sqlite3.connect(SEARCH_DB, timeout=60)
                try:
                    _c2.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name) VALUES (?,?,?)",
                                (row[0], title.strip(), SITE_NAME))
                    _c2.commit()
                except:
                    pass
                finally:
                    _c2.close()
            stats["new"] += 1
        else:
            stats["skip"] += 1
    except Exception as e:
        stats["errors"] += 1
        print("  DB error: %s" % e, file=sys.stderr)
    finally:
        conn.close()


if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser(description="xunhua 部门公开爬虫")
    parser.add_argument("--full", action="store_true", help="全量所有页")
    parser.add_argument("--pages", type=int, default=None, help="爬取页数")
    parser.add_argument("n", nargs="?", type=int, default=None, help="兼容旧配置")
    args = parser.parse_args()

    os.chdir(BASE_DIR)

    total_pages = get_total_pages()
    print("总页数: %d" % total_pages)

    if args.full:
        pages_to_crawl = total_pages
    elif args.pages is not None:
        pages_to_crawl = args.pages
    elif args.n is not None:
        pages_to_crawl = args.n
    else:
        pages_to_crawl = DEFAULT_PAGES

    pages_to_crawl = min(pages_to_crawl, total_pages)
    print("将爬取 %d 页..." % pages_to_crawl)

    for p in range(pages_to_crawl):
        page_num = p  # page 0 = List.html
        try:
            html = fetch_list_page(page_num)
            items = parse_items(html)
            if not items:
                print("  第%d页: 空" % page_num)
                break
            print("  第%d页: %d条" % (page_num, len(items)))
            for title, url, _ in items:
                dt_title, dt_date, body, summary = fetch_detail(url)
                final_title = dt_title or title
                final_date = dt_date
                store_record(final_title, url, final_date, body, summary)
                time.sleep(0.5)
        except Exception as e:
            print("  第%d页错误: %s" % (page_num, e))

    print("\n完成! 新%d, 跳过%d, 错误%d" % (stats["new"], stats["skip"], stats["errors"]))
