#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
黄冈市人民政府 - 政务公开栏目 (www.hg.gov.cn/zwgk/public/column/6636765?type=4&catId=7025318&action=list)
CMS: 龙讯科技 lonsun 政务公开平台
列表: AJAX POST /zwgk/site/label/8888
      data: labelName=publicInfoList, siteId=6792345, organId=6636765, pageSize=15,
            pageIndex=N, catId=7025318, type=4, action=list, length=50,
            dateFormat=yyyy年MM月dd日, file=/c2/xxgk/publicInfoList_newest_zc01
      响应 HTML: tr.xxgk_nav_con > td.info > a.title (href+title), td.fbrq 日期
      pageCount:234 (从响应 JSON 提取)
详情: https://www.hg.gov.cn/zwgk/public/6636765/{contentId}.html
      正文 div.wzcon.j-fontContent (保留 <p>/<table>), meta ArticleTitle/PubDate
样本: https://www.hg.gov.cn/zwgk/public/6636765/1766719.html
用法: python3 crawl_hg_zf_hgzq.py [--pages=N | N] [--sync]
"""
import requests
import re
import sys
import os
import json
import html as html_mod
from bs4 import BeautifulSoup

SITE_NAME = "黄冈市-回应关切"
BASE_URL = "https://www.hg.gov.cn"
AJAX_URL = "https://www.hg.gov.cn/zwgk/site/label/8888"
CAT_ID = "7025318"
ORGAN_ID = "6636765"
SITE_ID = "6792345"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "X-Requested-With": "XMLHttpRequest",
    "Referer": f"https://www.hg.gov.cn/zwgk/public/column/{ORGAN_ID}?type=4&catId={CAT_ID}&action=list",
}
CUTOFF = "2023-01-01"

QUALITY_DB = os.path.join(os.path.dirname(os.path.abspath(__file__)), "quality_results.db")
SEARCH_DB = "/root/search.db"
DOMAIN = "www.hg.gov.cn"


def push_to_searchdb(records, batch_label):
    import sqlite3
    conn = sqlite3.connect(QUALITY_DB, timeout=60)
    conn.execute("PRAGMA busy_timeout=300000")
    cur = conn.cursor()
    cur.execute("""CREATE TABLE IF NOT EXISTS crawl_results (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        site_id INTEGER DEFAULT 0,
        title TEXT, url TEXT, content TEXT,
        publish_date TEXT, summary TEXT,
        domain TEXT, category TEXT DEFAULT '',
        content_hash TEXT, crawl_time TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
        industry TEXT DEFAULT '', group_name TEXT DEFAULT '',
        script_name TEXT DEFAULT '', attachments TEXT DEFAULT '[]',
        UNIQUE(domain, url)
    )""")
    n_new = 0
    for r in records:
        try:
            cur.execute(
                "INSERT OR IGNORE INTO crawl_results (domain, title, url, content, publish_date, summary, industry, group_name, script_name, attachments) VALUES (?,?,?,?,?,?,?,?,?,?)",
                (r.get("domain", ""), r.get("title", ""), r.get("url", ""),
                 r.get("content", ""), r.get("pub_date", ""), r.get("summary", ""),
                 r.get("industry", ""), r.get("group_name", ""),
                 r.get("script_name", ""), r.get("attachments", "[]")),
            )
            n_new += cur.rowcount
        except Exception as e:
            print(f"  [push] ERR {r.get('url','')}: {e}")
    conn.commit()
    conn.close()
    print(f"[push_to_searchdb] new={n_new} skip={len(records)-n_new}")
    return n_new


def sync_to_server():
    import sqlite3
    try:
        conn = sqlite3.connect(QUALITY_DB, timeout=60)
        conn.execute("PRAGMA busy_timeout=300000")
        rows = conn.execute(
            "SELECT title, url, content, publish_date, summary, industry, group_name, script_name, attachments FROM crawl_results WHERE domain=? ORDER BY id",
            (DOMAIN,),
        ).fetchall()
        conn.close()
    except Exception as e:
        print(f"[sync] local read ERR: {e}")
        return
    if not rows:
        print("[sync] 本地没有数据")
        return
    dst = sqlite3.connect(SEARCH_DB, timeout=300)
    dst.execute("PRAGMA journal_mode=WAL")
    dst.execute("PRAGMA busy_timeout=300000")
    upd = ins = 0
    try:
        for r in rows:
            title, url, content, pub_date, summary, ind, grp, sn, att = r
            cur = dst.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            row = cur.fetchone()
            try:
                if row:
                    dst.execute(
                        "UPDATE gov_raw SET title=?, content=?, publish_date=?, summary=?, industry=?, group_name=?, script_name=?, attachments=? WHERE id=?",
                        (title, (content or "")[:500000], pub_date or "", (summary or "")[:300],
                         ind or "", grp or "", sn or "", att or "[]", row[0]))
                    upd += 1
                else:
                    dst.execute(
                        "INSERT INTO gov_raw (title, page_url, content, publish_date, summary, industry, group_name, script_name, attachments, site_name, source_url) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                        (title, url, (content or "")[:500000], pub_date or "", (summary or "")[:300],
                         ind or "", grp or "", sn or "", att or "[]", SITE_NAME, url))
                    ins += 1
            except Exception as e:
                print(f"  [sync] ERR {url}: {e}")
        dst.commit()
        dst.execute(
            "INSERT OR REPLACE INTO gov_search(rowid,title,site_name,summary) "
            "SELECT r.id,r.title,r.site_name,r.summary FROM gov_raw r "
            "WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
            (SITE_NAME,))
        dst.commit()
        total = dst.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
    finally:
        dst.close()
    print(f"[sync] OK updated={upd} inserted={ins} DB共{total}条")


def clean_title(t):
    t = re.sub(r"&middot;|&nbsp;|&#160;|\u200b|\ufeff", "", t or "")
    t = re.sub(r"^[•·]\s*", "", t)
    return re.sub(r"\s+", " ", t).strip()


def normalize_date(s):
    m = re.search(r"(\d{4})[年\-/.](\d{1,2})[月\-/.](\d{1,2})", s or "")
    if m:
        y, mo, d = int(m.group(1)), int(m.group(2)), int(m.group(3))
        return f"{y:04d}-{mo:02d}-{d:02d}"
    m = re.search(r"(\d{4})-(\d{2})-(\d{2})", s or "")
    if m:
        return m.group(0)
    return ""


def fetch_list(page):
    """POST AJAX 接口拿列表页"""
    data = {
        "labelName": "publicInfoList",
        "siteId": SITE_ID,
        "organId": ORGAN_ID,
        "pageSize": "15",
        "pageIndex": str(page),
        "catId": CAT_ID,
        "type": "4",
        "action": "list",
        "length": "50",
        "dateFormat": "yyyy年MM月dd日",
        "file": "/c2/xxgk/publicInfoList_newest_zc01",
    }
    try:
        r = requests.post(AJAX_URL, data=data, headers=HEADERS, timeout=30)
        r.raise_for_status()
        r.encoding = "utf-8"
        text = r.text
    except Exception as e:
        print(f"  [list] P{page} ERR: {e}")
        return [], False, 0
    items = []
    page_count = 0
    m = re.search(r"pageCount:(\d+)", text)
    if m:
        page_count = int(m.group(1))
    soup = BeautifulSoup(text, "html.parser")
    for tr in soup.find_all("tr", class_="xxgk_nav_con"):
        a = tr.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        m2 = re.search(r"/public/\d+/(\d+)\.html$", href)
        if not m2:
            continue
        # 只抓本机构 organId 详情链接
        if ORGAN_ID not in href and f"/public/{ORGAN_ID}/" not in href:
            continue
        title = clean_title(a.get_text(strip=True))
        if not title:
            continue
        pub_date = ""
        tds = tr.find_all("td")
        for td in tds:
            if "fbrq" in (td.get("class") or []):
                pub_date = normalize_date(td.get_text())
                break
        if not pub_date:
            txt = tr.get_text()
            m3 = re.search(r"(\d{4})年(\d{1,2})月(\d{1,2})日", txt)
            if m3:
                pub_date = f"{int(m3.group(1)):04d}-{int(m3.group(2)):02d}-{int(m3.group(3)):02d}"
        full_url = href if href.startswith("http") else BASE_URL + (href if href.startswith("/") else "/" + href)
        items.append({"title": title, "url": full_url, "pub_date": pub_date})
    # 去重
    seen = set()
    uniq = []
    for it in items:
        if it["url"] in seen:
            continue
        seen.add(it["url"])
        uniq.append(it)
    return uniq, True, page_count


def parse_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.raise_for_status()
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [detail] {url} ERR: {e}")
        return None

    title = ""
    m = soup.find("meta", attrs={"name": "ArticleTitle"})
    if m and m.get("content"):
        title = clean_title(m["content"])
    if not title:
        h = soup.find("h1")
        if h:
            title = clean_title(h.get_text())
    if not title:
        m = re.search(r"<title>(.*?)</title>", r.text, re.S)
        if m:
            title = clean_title(re.sub(r"_黄冈市人民政府$", "", m.group(1).strip()))

    pub_date = ""
    m = soup.find("meta", attrs={"name": "PubDate"})
    if m and m.get("content"):
        pub_date = m["content"].strip()[:10]
    if not pub_date:
        m = re.search(r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", r.text)
        if m:
            pub_date = m.group(1)

    body_el = soup.select_one("div.wzcon") or soup.select_one("div.j-fontContent") or soup.select_one("div.ls-article-info")
    if body_el is None:
        for el in soup.find_all(class_=re.compile("wzcon|j-fontContent|article-content|TRS_UEDITOR")):
            body_el = el
            break
    if body_el is None:
        return {"title": title, "content": "", "pub_date": pub_date, "attachments": []}

    # 附件
    atts = []
    for a in body_el.find_all("a", href=True):
        href = a["href"].strip()
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|7z)$", href, re.I) or "/attachment/" in href or "/fj/" in href or "download" in href.lower():
            abs_url = href if href.startswith("http") else ("https:" + href if href.startswith("//") else BASE_URL + (href if href.startswith("/") else "/" + href))
            name = clean_title(a.get_text(strip=True)) or href.split("/")[-1]
            if not any(x["url"] == abs_url for x in atts):
                atts.append({"name": name, "url": abs_url})
    page_att = soup.select_one("div.attachment") or soup.select_one("div.attachments") or soup.select_one("div.fj")
    if page_att:
        for a in page_att.find_all("a", href=True):
            href = a["href"].strip()
            if href:
                abs_url = href if href.startswith("http") else ("https:" + href if href.startswith("//") else BASE_URL + (href if href.startswith("/") else "/" + href))
                name = clean_title(a.get_text(strip=True)) or href.split("/")[-1]
                if not any(x["url"] == abs_url for x in atts):
                    atts.append({"name": name, "url": abs_url})

    # 附件段落从正文移除（保留表格）
    for a in body_el.find_all("a", href=True):
        href = a["href"].strip()
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|7z)$", href, re.I) or "/attachment/" in href or "/fj/" in href:
            p = a.find_parent("p")
            if p:
                p.decompose()
    # 绝对化链接
    for a in body_el.find_all("a", href=True):
        href = a["href"].strip()
        if href.startswith("/") or href.startswith("./") or (not href.startswith("http") and not href.startswith("javascript")):
            a["href"] = ("https:" + href if href.startswith("//") else BASE_URL + (href if href.startswith("/") else "/" + href))
    for img in body_el.find_all("img"):
        src = img.get("src", "")
        if src and not src.startswith("http"):
            img["src"] = ("https:" + src if src.startswith("//") else BASE_URL + (src if src.startswith("/") else "/" + src))
    # 二维码/分享/打印清理
    for el in body_el.select("div.share, div.ewm, div.qrcode, div.jzgov-qrcode-div, div.app-share, div.print, div.wzy_position"):
        el.decompose()
    for el in body_el.find_all(string=re.compile("扫一扫|分享到|打印本页")):
        p = el.find_parent("p") or el.find_parent("div")
        if p:
            p.decompose()

    body_html = str(body_el.decode_contents())
    body_html = re.sub(r"<script[^>]*>.*?</script>", "", body_html, flags=re.S | re.I)
    body_html = re.sub(r"<style[^>]*>.*?</style>", "", body_html, flags=re.S | re.I)
    body_html = re.sub(r"<p[^>]*>", "<p>", body_html)
    body_html = re.sub(r"<p>\s*</p>", "", body_html)
    body_html = re.sub(r"\s*<div id=\"docAppendix\".*$", "", body_html, flags=re.S)
    body_html = re.sub(r"\s*<div class=\"video\".*$", "", body_html, flags=re.S)

    if atts:
        has_att = any(a["url"] in body_html for a in atts)
        if not has_att:
            att_html = "".join(f'<p><a href="{a["url"]}">{html_mod.escape(a["name"])}</a></p>' for a in atts)
            body_html = body_html.rstrip() + "\n" + att_html

    text_len = len(re.sub(r"<[^>]+>", "", body_html).strip())
    has_img = "<img" in body_html
    if text_len < 5 and not has_img and not atts:
        return {"title": title, "content": "", "pub_date": pub_date, "attachments": atts}
    return {"title": title, "content": body_html, "pub_date": pub_date, "attachments": atts}


def main():
    # --sync-only: 只同步 quality_results -> search.db，不触发爬取
    sync_only = "--sync-only" in sys.argv
    max_pages = None
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            try:
                max_pages = int(a.split("=", 1)[1])
            except ValueError:
                pass
        elif a.isdigit():
            max_pages = int(a)
    print(f"[AutoPg] max_pages={max_pages} sync_only={sync_only}")
    if sync_only:
        sync_to_server()
        return

    valid = []
    total_pages = 0
    for pg in range(1, 1000):
        if max_pages and pg > max_pages:
            break
        items, has_next, page_count = fetch_list(pg)
        if page_count:
            total_pages = page_count
        print(f"[list] P{pg}: items={len(items)} (pageCount={page_count})")
        if not items:
            break
        for it in items:
            if it["pub_date"] and it["pub_date"] < CUTOFF:
                print(f"  [cutoff] {it['pub_date']} {it['title'][:40]}")
                continue
            det = parse_detail(it["url"])
            if det is None:
                print(f"  [fail] {it['title'][:50]}")
                det = {"title": it["title"], "content": "", "pub_date": it["pub_date"], "attachments": []}
            rec = {
                "domain": DOMAIN,
                "site_name": SITE_NAME,
                "title": det["title"] or it["title"],
                "url": it["url"],
                "content": det["content"],
                "pub_date": det["pub_date"] or it["pub_date"],
                "summary": re.sub(r"<[^>]+>", "", det["content"])[:200],
                "industry": "",
                "group_name": "湖北",
                "script_name": os.path.basename(sys.argv[0]),
                "attachments": json.dumps(det["attachments"], ensure_ascii=False) if det["attachments"] else "[]",
            }
            valid.append(rec)
        if not has_next:
            break
        # 提前结束：到最后一页
        if total_pages and pg >= total_pages:
            break
    print(f"[total] valid={len(valid)} (total_pages={total_pages})")
    if valid:
        push_to_searchdb(valid, os.path.basename(sys.argv[0]))
        if "--sync" in sys.argv:
            sync_to_server()


if __name__ == "__main__":
    main()
