#!/usr/bin/env python3
"""镶黄旗 (nmxhq.gov.cn) — 通知公告 Aisite/Easysite CMS"""
import re, sys, time
from datetime import datetime, timedelta
from urllib.parse import urljoin
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests
from bs4 import BeautifulSoup

BASE = "https://www.nmxhq.gov.cn"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
CUTOFF = datetime.now() - timedelta(days=3*365)
MAX_PAGES = 210
MAX_WORKERS = 8
SITE_NAME = "镶黄旗通知公告"

# Pagination: pages 1-5 static, pages 6+ via API
API_URL = ("https://www.nmxhq.gov.cn/eportal/ui"
           "?pageId=6caa3dad4a0f45cfa08e4a57386dbbea"
           "&moduleId=43ae64e404ff498cb1e5e325fc25ec21"
           "&staticRequest=yes&currentPage={}")

LIST_NUMS = {
    1: "https://www.nmxhq.gov.cn/nmxhq/zwgk/xxgk/zfxxgkml/tzgg/index.html",
    2: "https://www.nmxhq.gov.cn/nmxhq/zwgk/xxgk/zfxxgkml/tzgg/43ae64e4-2.html",
    3: "https://www.nmxhq.gov.cn/nmxhq/zwgk/xxgk/zfxxgkml/tzgg/43ae64e4-3.html",
    4: "https://www.nmxhq.gov.cn/nmxhq/zwgk/xxgk/zfxxgkml/tzgg/43ae64e4-4.html",
    5: "https://www.nmxhq.gov.cn/nmxhq/zwgk/xxgk/zfxxgkml/tzgg/43ae64e4-5.html",
}

def get_list_url(page):
    if page in LIST_NUMS:
        return LIST_NUMS[page]
    return API_URL.format(page)

def fetch(url, session):
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"  [WARN] fetch: {e}", file=sys.stderr)
        return None

def parse_list_page(html, base_url):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.select("ul#table1 li.hang1, li.hang1"):
        a = li.find("a")
        if not a or not a.get("href"):
            continue
        title = a.get("title", "").strip() or a.get_text(strip=True)
        href = a["href"].strip()
        full_url = urljoin(base_url, href)
        divs = li.find_all("div")
        date_str = divs[2].get_text(strip=True) if len(divs) >= 3 else ""
        if title and full_url:
            items.append((title, full_url, date_str))
    return items

def parse_detail(url, html):
    soup = BeautifulSoup(html, "html.parser")
    # Title
    h1 = soup.find("h1")
    title = h1.get_text(strip=True) if h1 else ""
    if not title:
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = meta["content"].strip()
    # Date
    date_str = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        date_str = meta_date["content"].strip()
    if not date_str:
        # Look for visible date text like "发布时间："
        txt = soup.get_text()
        m = re.search(r'(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}', txt)
        if m:
            date_str = m.group(1)
    # Content - try multiple containers
    content_div = (soup.select_one("div.dy_xx3")
                   or soup.select_one("#fontSize")
                   or soup.select_one("div.TRS_Editor div.Custom_UnionStyle")
                   or soup.select_one("div.TRS_Editor"))
    content_html = str(content_div) if content_div else ""
    # Strip trailing JS if present (prev/next links, scripts)
    if content_html:
        # Find last script or <br> sequence that is clearly UI
        last_p = content_html.rfind("<!--")
        if last_p > len(content_html) // 2:  # only if in latter half
            content_html = content_html[:last_p]
    return title, date_str, content_html

def date_from_str(s):
    if not s: return None
    for fmt in ["%Y-%m-%d", "%Y-%m-%d %H:%M:%S", "%Y/%m/%d", "%Y年%m月%d日"]:
        try:
            return datetime.strptime(s.strip()[:19], fmt)
        except: continue
    return None

def main():
    import sqlite3, urllib3
    urllib3.disable_warnings()
    session = requests.Session()
    session.headers.update(HEADERS)

    # Phase 1: collect all list items
    all_items, empty_pages = [], 0
    tot_pages = 0
    for page_idx in range(1, MAX_PAGES + 1):
        url = get_list_url(page_idx)
        print(f"  Page {page_idx}: {url[:90]}...", file=sys.stderr)
        html = fetch(url, session)
        if not html:
            if page_idx <= 5:
                print(f"  [ERR] static page {page_idx} failed!", file=sys.stderr)
                break
            # API might fail after last page
            print(f"  API page {page_idx} failed, stopping", file=sys.stderr)
            break
        items = parse_list_page(html, url)
        if not items:
            empty_pages += 1
            if empty_pages >= 3:
                print(f"  {empty_pages} empty pages at {page_idx}, stopping", file=sys.stderr)
                break
        else:
            empty_pages = 0
            tot_pages += 1
            # Check date of last item - if beyond 3 years, we can keep going but will filter later
        all_items.extend(items)
        print(f"  {len(items)} items (total: {len(all_items)})", file=sys.stderr)
        time.sleep(0.5 if page_idx > 5 else 0.3)

    print(f"\nTotal items collected: {len(all_items)} across {tot_pages} pages", file=sys.stderr)

    # Phase 2: filter existing
    conn = sqlite3.connect("/root/search.db", timeout=60)
    existing = set()
    try:
        for row in conn.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
            existing.add(row[0])
    except: pass

    to_fetch = [item for item in all_items if item[1] not in existing]
    skipped_dup = len(all_items) - len(to_fetch)
    print(f"Dup: {skipped_dup}, to fetch: {len(to_fetch)}", file=sys.stderr)

    # Phase 3: fetch details
    results, skipped_date = [], 0
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as pool:
        def process(item):
            title, url, list_date_str = item
            html = fetch(url, session)
            if not html: return None
            try:
                dt, dds, ch = parse_detail(url, html)
                ft = dt or title
                fds = dds or list_date_str
                fd = date_from_str(fds)
                if fd and fd < CUTOFF: return {"skip": "date"}
                return {
                    "page_url": url, "title": ft, "content": ch,
                    "publish_date": fd.strftime("%Y-%m-%d") if fd else "",
                    "site_name": SITE_NAME, "source_url": url,
                }
            except Exception as e:
                print(f"  [ERR] {url}: {e}", file=sys.stderr)
                return None
        fut_map = {pool.submit(process, item): item for item in to_fetch}
        for fut in as_completed(fut_map):
            r = fut.result()
            if r is None: continue
            if r.get("skip") == "date": skipped_date += 1; continue
            results.append(r)
    print(f"New: {len(results)}, Dup: {skipped_dup}, Date-skip: {skipped_date}", file=sys.stderr)

    # Phase 4: insert
    if results:
        conn.execute("BEGIN")
        now = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
        for r in results:
            try:
                conn.execute("""INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date, content, date_rank)
                    VALUES (?,?,?,?,?,?,?)""",
                    (r["site_name"], r["source_url"], r["page_url"], r["title"],
                     r["publish_date"], r["content"],
                     0 if not r["publish_date"] else int(r["publish_date"].replace("-", ""))))
            except Exception as e:
                print(f"  [ERR] insert: {e}", file=sys.stderr)
        conn.commit()
        # Rebuild FTS
        # QC20260926 去掉手写 gov_search 整站删除(抢锁源; FTS 由 gov_raw 触发器维护) 
        # conn.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
        conn.execute("""INSERT OR REPLACE INTO gov_search(rowid, site_name, title, page_url, publish_date, source_url, summary)
            SELECT rowid, site_name, title, page_url, publish_date, source_url, content, summary
            FROM gov_raw WHERE site_name=?""", (SITE_NAME,))
        conn.commit()
        print("FTS rebuilt", file=sys.stderr)
    conn.close()
    print(f"Done. +{len(results)}", file=sys.stderr)

if __name__ == "__main__":
    main()
