#!/usr/bin/env python3
"""托克托县人民政府 — 部门文件 (服务器正式版)
栏目: http://www.tuoxian.gov.cn/zfxxgkzl/fdzdgknr_22198/bmwj/
CMS: TRS，静态分页 index_N.html (共 334 页)
详情: meta ArticleTitle + meta PubDate + div.TRS_UEDITOR
⚠️ 2026-08-10: 服务器 IP 被奇安信云 WAF 随机拦截(502, WZWS-RAY 头)，
   本地 Mac 可稳定访问 → 当前走「本地抓取→JSONL→scp→import_tuoxian_bmwj.py 导入」降级闭环。
   本脚本为 WAF 解锁后恢复用，配置注册 enabled=false。
"""
import requests, re, sys, os, time, sqlite3
import urllib3
urllib3.disable_warnings()

BASE_URL = "http://www.tuoxian.gov.cn"
LIST_BASE = "http://www.tuoxian.gov.cn/zfxxgkzl/fdzdgknr_22198/bmwj/"
SITE_NAME = "托克托县人民政府-部门文件"
GROUP = "内蒙古"
INDUSTRY = "政府公告"
SCRIPT_NAME = "crawl_tuoxian_bmwj.py"
MAX_PAGES = 5
CUTOFF = "2023-08-10"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "http://www.tuoxian.gov.cn/",
}
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

session = requests.Session()
session.headers.update(HEADERS)
session.verify = False


def fetch(url):
    for i in range(5):
        try:
            r = session.get(url, timeout=20, allow_redirects=True)
            if r.status_code == 200:
                return r.text
            print(f"  retry {i+1}: {r.status_code}", flush=True)
        except Exception as e:
            print(f"  retry {i+1}: {str(e)[:60]}", flush=True)
        time.sleep(3 + i * 3)
    return None


def parse_list(html):
    items = []
    seen = set()
    rows = re.findall(r"<tr[^>]*>.*?</tr>", html, re.S | re.I)
    for row in rows:
        m = re.search(r'<a[^>]*href="([^"]+)"[^>]*>([^<]*)</a>', row)
        if not m:
            continue
        href, title = m.group(1), m.group(2).strip()
        title = re.sub(r"\s+", " ", title).strip()
        if len(title) < 4 or not re.search(r"\.html?$", href, re.I):
            continue
        if href.startswith("./"):
            full = LIST_BASE + href[2:]
        elif href.startswith("/"):
            full = BASE_URL + href
        elif href.startswith("http"):
            full = href
        else:
            full = LIST_BASE + href
        if not full.startswith(BASE_URL):
            continue
        if full in seen:
            continue
        seen.add(full)
        date_str = ""
        dm = re.search(r"(\d{4}-\d{2}-\d{2})", row)
        if dm:
            date_str = dm.group(1)
        items.append((title, full, date_str))
    return items


def parse_detail(html):
    title = ""
    m = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]+)"', html)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r"<title>([^<]+)</title>", html)
        if m:
            title = m.group(1).replace("托克托县人民政府_", "").strip()
    date_str = ""
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="([^"]+)"', html)
    if m:
        dm = re.search(r"(\d{4}-\d{2}-\d{2})", m.group(1))
        if dm:
            date_str = dm.group(1)
    content = ""
    m = re.search(r'<div[^>]*class="[^"]*TRS_UEDITOR[^"]*"[^>]*>', html)
    if m:
        start = m.end()
        depth = 1
        i = start
        while i < len(html) and depth > 0:
            if html.startswith("<div", i):
                depth += 1
                i += 4
            elif html.startswith("</div>", i):
                depth -= 1
                i += 6
            else:
                i += 1
        content = html[start:i]
    if content:
        content = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", content, flags=re.S | re.I)
        content = re.sub(r'\s*<div id="docAppendix".*$', "", content, flags=re.S)
        content = re.sub(r'\s*<div class="video" style="display: none;">.*$', "", content, flags=re.S)
        content = re.sub(r"</div>\s*(?:</div>\s*)+$", "</div>", content)
        content = content.strip()
    return title, date_str, content


def crawl():
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cursor = conn.cursor()
    all_items = []
    for pg in range(1, MAX_PAGES + 1):
        url = LIST_BASE + (f"index_{pg}.html" if pg > 1 else "")
        html = fetch(url)
        if not html:
            print(f"[{SCRIPT_NAME}] Page {pg}: fetch fail", flush=True)
            break
        items = parse_list(html)
        if not items:
            print(f"[{SCRIPT_NAME}] Page {pg}: 0 items (end)", flush=True)
            break
        all_items.extend(items)
        print(f"[{SCRIPT_NAME}] Page {pg}: {len(items)} items", flush=True)
        time.sleep(0.5)
    print(f"[{SCRIPT_NAME}] Total: {len(all_items)} items")
    new_count = skip_count = 0
    for title, item_url, date_str in all_items:
        existing = cursor.execute(
            "SELECT id FROM gov_raw WHERE page_url = ?", (item_url,)
        ).fetchone()
        if existing:
            skip_count += 1
            continue
        detail_html = fetch(item_url)
        if not detail_html:
            skip_count += 1
            continue
        real_title, real_date, content = parse_detail(detail_html)
        if not real_title:
            real_title = title
        if not real_date:
            real_date = date_str
        if not content or len(content.strip()) < 10:
            content = "正文为空"
        try:
            cursor.execute(
                """INSERT INTO gov_raw (source_url, page_url, title, publish_date, site_name, content, group_name, industry, script_name)
                   VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                (item_url, item_url, real_title, real_date, SITE_NAME, content, GROUP, INDUSTRY, SCRIPT_NAME),
            )
            conn.commit()
            new_count += 1
            print(f"[{SCRIPT_NAME}] +{new_count}: {real_title[:40]} ({real_date})", flush=True)
        except Exception as e:
            conn.rollback()
            skip_count += 1
        time.sleep(0.8)
    # FTS 增量同步
    cursor.execute(
        """INSERT INTO gov_search(rowid,title,site_name,summary)
           SELECT r.id, r.title, r.site_name, substr(r.content,1,500)
           FROM gov_raw r
           WHERE r.id NOT IN (SELECT rowid FROM gov_search)
             AND r.site_name = ?""",
        (SITE_NAME,),
    )
    conn.commit()
    conn.close()
    print(f"\n[{SCRIPT_NAME}] Done. New: {new_count}, Skipped: {skip_count}")


if __name__ == "__main__":
    crawl()
