#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""爬取武穴市 - 通知公告（黄冈信息公开平台）"""
import requests
import sqlite3
import re
import sys
import time
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

BASE = "http://www.wuxue.gov.cn"
DB = "/root/search.db"
CUTOFF = datetime.now() - timedelta(days=365 * 3)
SITE_NAME = "武穴市-通知公告"
CATEGORY = "县区"
MAX_PAGES = 5  # only 4 exist

ua = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
s = requests.Session()
s.headers.update({
    "User-Agent": ua,
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
})


def init_session():
    """获取session cookie"""
    s.get(BASE + "/zwgk/public/column/6636607?type=4&catId=7033930&action=list", timeout=30)


def fetch_list(page):
    """通过API获取列表页HTML"""
    data = {
        "labelName": "publicInfoList", "siteId": "6792345",
        "pageSize": "20", "pageIndex": str(page), "action": "list",
        "isDate": "true", "dateFormat": "yyyy-MM-dd", "length": "80",
        "organId": "6636607", "type": "4", "catId": "7033930",
        "cId": "", "result": "", "keyWords": "",
        "file": "/c2/xxgk/publicInfoList_newest_hg"
    }
    r = s.post(BASE + "/zwgk/site/label/8888", data=data, timeout=30,
               headers={"X-Requested-With": "XMLHttpRequest"})
    return BeautifulSoup(r.text, "html.parser")


def parse_list(soup, cutoff_date):
    items = []
    for a in soup.select("a.title"):
        href = a.get("href", "").strip()
        title = a.get("title", "") or a.get_text(strip=True)
        if not title or not href:
            continue
        # Find date from nearby span.date
        parent_li = a.find_parent("li")
        if not parent_li:
            continue
        date_span = parent_li.find("span", class_="date")
        if not date_span:
            continue
        date_str = date_span.get_text(strip=True)
        try:
            pub_date = datetime.strptime(date_str, "%Y-%m-%d")
        except:
            continue
        if pub_date < cutoff_date:
            return items, True
        url = href if href.startswith("http") else BASE + href
        items.append((title, url, date_str))
    return items, False


def fetch_detail(url):
    r = s.get(url, timeout=30)
    r.encoding = "utf-8"
    soup = BeautifulSoup(r.text, "html.parser")
    # Content: .gkwz_contnet.clearfix or .wzcon.j-fontContent
    con = soup.select_one("div.gkwz_contnet.clearfix") or soup.select_one("div.wzcon.j-fontContent")
    if not con:
        return ""
    for tag in con.find_all(["script", "style"]):
        tag.decompose()
    return re.sub(r"\s+", " ", str(con)).strip()


def save_to_db(items_data):
    conn = sqlite3.connect(DB)
    c = conn.cursor()
    new_count = 0
    for title, url, date_str, content in items_data:
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
        if c.fetchone():
            continue
        summary = content[:300] if content else ""
        c.execute(
            "INSERT INTO gov_raw (site_name, category, title, page_url, content, summary, publish_date) VALUES (?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, CATEGORY, title, url, content, summary, date_str),
        )
        new_count += 1
    conn.commit()
    conn.close()
    return new_count


def main():
    print("[%s] 开始爬取" % SITE_NAME)
    print("  初始化session...")
    init_session()

    all_data = []
    for page in range(1, MAX_PAGES + 1):
        print("  第%d页..." % page)
        soup = fetch_list(page)
        items, stopped = parse_list(soup, CUTOFF)
        print("    找到 %d 条" % len(items))
        if not items:
            print("    空页，停止")
            break
        for title, item_url, date_str in items:
            print("    详情: %s..." % title[:35])
            content = fetch_detail(item_url)
            if not content:
                print("      [SKIP] 无正文")
                continue
            print("      正文长度: %d" % len(content))
            all_data.append((title, item_url, date_str, content))
            time.sleep(0.5)
        if stopped:
            print("  遇到超3年旧数据，停止")
            break
        time.sleep(1)

    if not all_data:
        print("无新数据")
        return

    print("\n入库 %d 条..." % len(all_data))
    n = save_to_db(all_data)
    print("新增入库: %d 条" % n)

    if n > 0:
        print("重建FTS索引...")
        conn = sqlite3.connect(DB)
        conn.execute("INSERT INTO gov_raw_fts(gov_raw_fts) VALUES('rebuild')")
        conn.commit()
        conn.close()

    print("完成!")


if __name__ == "__main__":
    main()
