#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
内蒙古自治区生态环境厅 - 部门文件(bmwj) 爬虫
URL: https://sthjt.nmg.gov.cn/xxgk/zfxxgk/fdzdgknr/bmwj/
CMS: 内蒙古政务公开平台 静态分页
列表: index.html / index_N.html (每页10条, 共2089条≈209页)
列表表格: 序号 | 标题 | 标题(移动版) | 文号 | 成文日期 | 发布日期
详情: tYYYYMM_xxxx.html, 正文容器 trs_editor_view (docContent 不可靠)

用法:
  python3 crawl_sthjt_nmg_zfxxgk.py --pages=1     # 增量: 前1页(10条)
  python3 crawl_sthjt_nmg_zfxxgk.py --pages=5     # 首批前5页
  python3 crawl_sthjt_nmg_zfxxgk.py --pages=50    # 全量(2089条≈209页)
"""
import os, re, sys, sqlite3, time, json
from datetime import datetime, timedelta
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

def parse_pages(argv):
    if '--full' in argv:
        return 9999
    for i, a in enumerate(argv):
        if a == '--pages' and i + 1 < len(argv) and argv[i+1].isdigit():
            return int(argv[i+1])
        if a.startswith('--pages='):
            return int(a.split('=')[1])
        if a.isdigit():
            return int(a)
    return 1

MAX_PAGES = parse_pages(sys.argv[1:])
print(f"[sthjt-zfxxgk] max_pages={MAX_PAGES}")

DB_PATH = os.getenv("SEARCH_DB", "/mnt/data/search.db")
BASE = "https://sthjt.nmg.gov.cn"
LIST_URL = "https://sthjt.nmg.gov.cn/xxgk/zfxxgk/fdzdgknr/bmwj/index.html"
SITE_NAME = "内蒙古生态环境厅-部门文件"
CATEGORY = "部门文件"
GROUP_NAME = "内蒙古"
PAGE_SIZE = 10
DATE_CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Referer": "https://sthjt.nmg.gov.cn/xxgk/zfxxgk/fdzdgknr/bmwj/",
}

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    conn.row_factory = sqlite3.Row
    return conn

def clean_title(t):
    if not t:
        return t
    t = t.replace("&middot;", "").replace("&nbsp;", "").replace("&ensp;", "").replace("&emsp;", "")
    t = re.sub(r"[\u200b\u200c\u200d\ufeff]", "", t)
    return t.strip()

def is_dup(conn, page_url):
    return conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone() is not None

def insert_item(conn, title, page_url, publish_date, content):
    date_rank = int(publish_date.replace("-", "")) if publish_date and "-" in publish_date else 0
    plain = ""
    if content:
        plain = BeautifulSoup(content, "html.parser").get_text(strip=True)[:200]
    try:
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary, script_name, group_name, has_table) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, title, page_url, publish_date, content, date_rank, CATEGORY, plain,
             "crawl_sthjt_nmg_zfxxgk.py", GROUP_NAME, 1 if "<table" in (content or "") else 0),
        )
        if cur.rowcount > 0:
            try:
                conn.execute(
                    "INSERT INTO gov_search (title, site_name, summary) VALUES (?, ?, ?)",
                    (title, SITE_NAME, plain),
                )
            except Exception as e:
                print(f"    [FTS WARN] {e}")
            return True
        return False
    except Exception as e:
        print(f"    [ERR] insert: {e}")
        return False

def fetch_list(session, page_num):
    """静态分页: index.html=第1页, index_1.html=第2页, ..."""
    if page_num <= 1:
        url = LIST_URL
    else:
        url = f"https://sthjt.nmg.gov.cn/xxgk/zfxxgk/fdzdgknr/bmwj/index_{page_num-1}.html"
    resp = session.get(url, headers=HEADERS, timeout=30)
    resp.encoding = "utf-8"
    soup = BeautifulSoup(resp.text, "html.parser")
    items = []
    seen = set()
    for tr in soup.find_all("tr"):
        tds = tr.find_all("td")
        if len(tds) < 5:
            continue
        a = None
        for td in tds:
            a = td.find("a", href=True)
            if a:
                break
        if not a:
            continue
        href = a.get("href", "")
        if "t20" not in href or "sthjt.nmg.gov.cn" not in href:
            continue
        title = clean_title(a.get_text())
        # 日期: 最后一个 td (发布日期)
        date_str = clean_title(tds[-1].get_text())
        m = re.match(r"(\d{4}-\d{2}-\d{2})", date_str)
        date_str = m.group(1) if m else ""
        if href in seen:
            continue
        seen.add(href)
        items.append({"title": title, "url": href, "date": date_str})
    return items

def extract_content(node):
    soup = BeautifulSoup(str(node), "html.parser")
    for tag in soup.find_all(["script", "style", "noscript"]):
        tag.decompose()
    for tag in soup.find_all(attrs={"style": re.compile(r"display\s*:\s*none", re.I)}):
        tag.decompose()
    for a in soup.find_all("a"):
        href = a.get("href")
        if href:
            a["href"] = urljoin(BASE, href)
        img = a.find("img")
        if img:
            img.decompose()
    parts = []
    for el in soup.find_all(["p", "table", "div", "li", "h1", "h2", "h3", "h4"]):
        tag = el.name
        if tag == "tr":
            continue
        if tag == "p" and el.find_parent("table"):
            continue
        text = el.get_text(strip=True)
        if not text:
            continue
        if tag == "table":
            parts.append(str(el))
        elif tag in ("p", "div", "li", "h1", "h2", "h3", "h4"):
            if tag == "div" and el.find(["p", "table"]):
                continue
            parts.append(str(el))
    content = "\n\n".join(parts)
    segs = content.split("\n\n")
    deduped = []
    for s in segs:
        if not deduped or s != deduped[-1]:
            deduped.append(s)
    return "\n\n".join(deduped).strip()

def parse_detail(html_text, fallback_title):
    soup = BeautifulSoup(html_text, "html.parser")
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = clean_title(meta["content"])
    pub_date = ""
    meta_d = soup.find("meta", attrs={"name": "PubDate"})
    if meta_d and meta_d.get("content"):
        m2 = re.match(r"(\d{4}-\d{1,2}-\d{1,2})", meta_d["content"].strip())
        if m2:
            pub_date = m2.group(1)
    content = ""
    # 优先: trs_editor_view (部门文件正文区, 已验证)
    div = soup.find("div", class_=re.compile(r"trs_editor_view"))
    if div:
        content = extract_content(div)
    if not content:
        div = soup.find("div", class_=re.compile(r"docContent"))
        if div:
            content = extract_content(div)
    if not content:
        for cid in ["zoom", "content", "article", "text", "mainText", "zoomcon"]:
            div = soup.find("div", id=cid)
            if div:
                content = extract_content(div)
                break
    if not content:
        # 回退: 找正文区域 class 含 content/article/zw (排除 share-box 等导航)
        for cls in ["content", "article", "TRS_Editor", "zwgk_content", "main-content"]:
            div = soup.find("div", class_=re.compile(cls))
            if div:
                content = extract_content(div)
                break
    # 清理: 若正文以分享工具栏开头, 截断到 trs_editor_view 或 docContent
    if content and ("share-box" in content or "分享到" in content):
        for cls in ["trs_editor_view", "docContent"]:
            div = soup.find("div", class_=re.compile(cls))
            if div:
                content = extract_content(div)
                break
    if not title:
        title = fallback_title
    return title, content, pub_date

def main():
    conn = get_conn()
    session = requests.Session()

    # 先访问列表页建立会话
    try:
        session.get(LIST_URL, headers=HEADERS, timeout=30)
    except Exception as e:
        print(f"  [WARN] session init: {e}")

    pages = MAX_PAGES
    print(f"[sthjt-zfxxgk] scanning {pages} pages of 部门文件(bmwj)")

    all_items = []
    for pg in range(1, pages + 1):
        time.sleep(0.4)
        try:
            items = fetch_list(session, pg)
            all_items.extend(items)
            print(f"  page {pg}: +{len(items)} items")
            if len(items) == 0:
                break
        except Exception as e:
            print(f"  [WARN] page {pg}: {e}")
            time.sleep(3)

    print(f"[sthjt-zfxxgk] collected {len(all_items)} list items")

    new_count = dup_count = err_count = 0
    for idx, item in enumerate(all_items, 1):
        title, url, date_str = item["title"], item["url"], item["date"]
        try:
            if date_str and date_str < DATE_CUTOFF:
                continue
            if not url or "sthjt.nmg.gov.cn" not in url:
                dup_count += 1
                continue
            if is_dup(conn, url):
                dup_count += 1
                continue
            resp = session.get(url, headers=HEADERS, timeout=30)
            resp.encoding = "utf-8"
            d_title, content, pub_date = parse_detail(resp.text, title)
            if not pub_date:
                pub_date = date_str
            if insert_item(conn, d_title, url, pub_date, content):
                new_count += 1
            else:
                dup_count += 1
        except Exception as e:
            err_count += 1
            print(f"  [ERR] {title[:40]}: {e}")
        conn.commit()
        time.sleep(0.3)
        if idx % 20 == 0:
            print(f"  [{idx}/{len(all_items)}] new={new_count} dup={dup_count} err={err_count}")

    conn.close()
    print(f"[sthjt-zfxxgk] DONE: +{new_count} new, {dup_count} dup, {err_count} err")
    print(f"新增: {new_count}")

if __name__ == "__main__":
    main()
