#!/usr/bin/env python3
"""黄冈高新区-通知公告 爬虫 (龙讯CMS - AJAX分页)"""
import re, sys, os, time, hashlib, json
import requests
from bs4 import BeautifulSoup
from datetime import datetime

BASE_URL = "https://gxq.hg.gov.cn"
LIST_URL = "https://gxq.hg.gov.cn/zwgk/site/label/8888"
DB_PATH = "/root/search.db"
SITE_NAME = "黄冈高新区-通知公告"
TABLE_NAME = "gov_raw"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "X-Requested-With": "XMLHttpRequest",
    "Referer": "https://gxq.hg.gov.cn/zwgk/public/column/6636249?type=4&catId=7026845&action=list",
}

LIST_DATA_TPL = {
    "labelName": "publicInfoList",
    "siteId": "6792345",
    "pageSize": "20",
    "pageIndex": "1",
    "action": "list",
    "isDate": "true",
    "dateFormat": "yyyy-MM-dd",
    "length": "80",
    "organId": "6636249",
    "type": "4",
    "catId": "7026845",
    "cId": "",
    "result": "暂无相关信息",
    "keyWords": "",
    "file": "/c2/xxgk/publicInfoList_newest_hg",
}


def get_list_page(page):
    """获取列表页HTML"""
    data = dict(LIST_DATA_TPL)
    data["pageIndex"] = str(page)
    try:
        r = requests.post(LIST_URL, data=data, headers=HEADERS, timeout=30, verify=False)
        if r.status_code == 200 and len(r.text) > 100:
            return r.text
    except Exception as e:
        print(f"  [!] 请求列表第{page}页失败: {e}", flush=True)
    return None


def parse_list(html):
    """解析列表，返回 [(title, url, date_str), ...]"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.select("li.clearfix"):
        a = li.select_one("a.title")
        if not a:
            continue
        href = a.get("href", "").strip()
        title = a.get("title", "").strip() or a.get_text(strip=True)
        date_span = li.select_one("span.date")
        date_str = date_span.get_text(strip=True) if date_span else ""
        if href and title:
            if href.startswith("http"):
                full_url = href
            else:
                full_url = BASE_URL + href
            items.append((title, full_url, date_str))
    return items


def fetch_detail(url):
    """获取详情页，返回 (title, pub_date, content_html)"""
    try:
        r = requests.get(url, headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
            "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
            "Accept-Language": "zh-CN,zh;q=0.9",
        }, timeout=30, verify=False)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None, None, None
    except:
        return None, None, None

    soup = BeautifulSoup(r.text, "html.parser")

    # 标题
    title_el = soup.select_one("h1.wztit")
    title = title_el.get_text(strip=True) if title_el else ""

    # 日期 - 从 meta 优先
    pub_date = ""
    meta_date = soup.select_one("meta[name='PubDate']")
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()

    # 内容
    content_el = soup.select_one("div.gkwz_contnet div.wzcon.j-fontContent")
    content = ""
    if content_el:
        content = str(content_el)

    return title, pub_date, content


def main():
    requests.packages.urllib3.disable_warnings()

    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()

    # 已存在的URL
    cur.execute(f"SELECT page_url FROM {TABLE_NAME}")
    existing_urls = set(row[0] for row in cur.fetchall())

    all_items = []
    for page in range(1, MAX_PAGES + 1):
        print(f"[*] 正在获取第{page}页列表...", flush=True)
        html = get_list_page(page)
        if not html:
            print(f"  [!] 第{page}页列表为空，跳过", flush=True)
            continue
        items = parse_list(html)
        print(f"  [→] 解析到 {len(items)} 条", flush=True)
        all_items.extend(items)
        time.sleep(0.5)

    print(f"\n[*] 总计获取列表 {len(all_items)} 条", flush=True)

    # 去重
    new_items = [(t, u, d) for t, u, d in all_items if u not in existing_urls]
    print(f"[*] 排除已存在后，新增 {len(new_items)} 条", flush=True)

    if not new_items:
        print("[✓] 无新数据需要爬取", flush=True)
        conn.close()
        return

    inserted = 0
    for i, (title, url, date_str) in enumerate(new_items, 1):
        print(f"  [{i}/{len(new_items)}] {title[:40]}...", flush=True)
        det_title, det_date, content = fetch_detail(url)

        final_title = det_title or title
        final_date = det_date or date_str

        # 验证内容
        if not content or len(content.strip()) < 50:
            print(f"    [!] 内容为空或过短，跳过", flush=True)
            continue

        # 摘要
        text_only = BeautifulSoup(content, "html.parser").get_text(strip=True)
        summary = text_only[:200] if text_only else ""

        try:
            cur.execute(
                f"INSERT OR IGNORE INTO {TABLE_NAME} (title, page_url, content, summary, publish_date, site_name) VALUES (?, ?, ?, ?, ?, ?)",
                (final_title, url, content, summary, final_date, SITE_NAME)
            )
            if cur.rowcount > 0:
                inserted += 1
                conn.commit()  # 每页提交一次，避免遗漏
        except Exception as e:
            print(f"    [!] 入库失败: {e}", flush=True)

        time.sleep(0.3)

    conn.close()
    print(f"\n[✓] 完成！入库 {inserted} 条新数据", flush=True)


if __name__ == "__main__":
    main()
