#!/usr/bin/env python3
"""
xjxinhe.gov.cn 新和县政府 - 公示公告爬虫
============================================
数据以JSON嵌入HTML，详情页requests直达。
栏目: channelId=729（公示公告）

用法:
  python3 crawl_xjxinhe.py                # 增量爬第1页
  python3 crawl_xjxinhe.py --pages 3      # 爬3页
  python3 crawl_xjxinhe.py --full         # 全量（JSON已含所有页）
"""

import os, re, sys, time, json, sqlite3
from datetime import datetime
import requests
from bs4 import BeautifulSoup

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")

SITE_NAME = "新和县人民政府"
DOMAIN = "www.xjxinhe.gov.cn"
CATEGORY_NAME = "公示公告"
GROUP = "新疆-阿克苏"
INDUSTRY = "政府公告"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"}

LIST_URL = "https://www.xjxinhe.gov.cn/xwdt/gsgg/index.html"
DETAIL_BASE = "https://www.xjxinhe.gov.cn"
stats = {"new": 0, "skip": 0, "errors": 0}


def fetch_all_items(max_pages=1):
    """从首页JSON提取条目，只取前max_pages页"""
    r = requests.get(LIST_URL, headers=HEADERS, timeout=15)
    r.encoding = "utf-8"
    html = r.text
    
    m = re.search(r"var dataList=\[(.*?)\];", html, re.DOTALL)
    if not m:
        print("No dataList found", file=sys.stderr)
        return []
    
    raw = "[" + m.group(1) + "]"
    data = json.loads(raw)
    
    total_pages = len(data)
    pages_to_fetch = min(max_pages, total_pages)
    print("JSON共 %d 页, 爬取 %d 页" % (total_pages, pages_to_fetch))
    
    items = []
    for page_idx in range(pages_to_fetch):
        page = data[page_idx]
        for article in page.get("infolist", []):
            title = article.get("title", "")
            url = article.get("url", "")
            ts = article.get("releaseTime", 0)
            date = datetime.fromtimestamp(ts / 1000).strftime("%Y-%m-%d") if ts else ""
            # 补齐URL
            if url and not url.startswith("http"):
                url = DETAIL_BASE + url
            items.append((title, url, date))
    return items


def fetch_detail(url):
    """获取详情页#zoom正文"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")
        
        # 标题
        title = ""
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()
        
        # 日期
        date = ""
        meta_d = soup.find("meta", attrs={"name": "PubDate"})
        if meta_d and meta_d.get("content"):
            raw = meta_d["content"].strip()
            m = re.search(r"(\d{4})-(\d{2})-(\d{2})", raw)
            if m:
                date = "%s-%s-%s" % (m.group(1), m.group(2), m.group(3))
        
        # 正文
        body_html = ""
        summary = ""
        zoom = soup.select_one("#zoom")
        if zoom:
            for tag in zoom.find_all(["script", "style"]):
                tag.decompose()
            body_html = str(zoom)
            summary = zoom.get_text(strip=True)[:200]
        
        return title, date, body_html, summary
    except Exception as e:
        print("    detail error: %s" % e, file=sys.stderr)
        return "", "", "", ""


def store_record(title, page_url, publish_date, body_html="", summary=""):
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(title, page_url, source_url, site_name, publish_date, category, industry, group_name, content, summary) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (title.strip(), page_url, DOMAIN, SITE_NAME, publish_date,
             CATEGORY_NAME, INDUSTRY, GROUP, body_html, summary),
        )
        conn.commit()
        is_new = conn.total_changes > 0

        if not is_new and body_html:
            row = conn.execute("SELECT id, content FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row and (not row[1] or row[1].strip() == ""):
                conn.execute("UPDATE gov_raw SET content=?, summary=? WHERE id=?", (body_html, summary, row[0]))
                conn.commit()
                is_new = True

        if is_new:
            row = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row:
                _c2 = sqlite3.connect(SEARCH_DB, timeout=60)
                try:
                    _c2.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name) VALUES (?,?,?)",
                                (row[0], title.strip(), SITE_NAME))
                    _c2.commit()
                except:
                    pass
                finally:
                    _c2.close()
            stats["new"] += 1
        else:
            stats["skip"] += 1
    except Exception as e:
        stats["errors"] += 1
        print("  DB error: %s" % e, file=sys.stderr)
    finally:
        conn.close()


if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser(description="xjxinhe 公示公告爬虫")
    parser.add_argument("--full", action="store_true", help="全量")
    parser.add_argument("--pages", type=int, default=1, help="页数(默认1)")
    args = parser.parse_args()

    os.chdir(BASE_DIR)

    if args.full:
        pages = 999  # 全量
    else:
        pages = args.pages

    items = fetch_all_items(max_pages=pages)
    if not items:
        print("无可采集条目")
        sys.exit(1)
    
    print("共 %d 条" % len(items))

    for i, (title, url, date_from_list) in enumerate(items):
        if i % 10 == 0 and i > 0:
            print("  [%d/%d]..." % (i, len(items)))
        dt_title, dt_date, body, summary = fetch_detail(url)
        final_title = dt_title or title
        final_date = dt_date or date_from_list
        store_record(final_title, url, final_date, body, summary)
        time.sleep(0.3)

    print("\n完成! 新%d, 跳过%d, 错误%d" % (stats["new"], stats["skip"], stats["errors"]))
