#!/usr/bin/env python3
"""宁晋县-乡镇动态 爬虫 (静态分页 list.jsp?classId=52&pn=N)
"""
import os, sqlite3, re, time, requests
from datetime import datetime, date
from concurrent.futures import ThreadPoolExecutor, as_completed

BASE = "https://www.ningjin.gov.cn"
LIST_URL = "https://www.ningjin.gov.cn/channel/list/52.html"
PAGE_URL = "https://www.ningjin.gov.cn/xxgk/list/list.jsp?classId=52&pn={}"
CUTOFF_DATE = date(2023, 6, 17)
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "宁晋县-乡镇动态"
MAX_WORKERS = 8
HEADERS = {"User-Agent": "Mozilla/5.0"}
MAX_PAGES = 123


def collect_urls():
    """收集所有文章URL，超3年自动停"""
    all_urls = []
    for pn in range(1, MAX_PAGES + 1):
        url = LIST_URL if pn == 1 else PAGE_URL.format(pn)
        for att in range(3):
            try:
                r = requests.get(url, timeout=30, headers=HEADERS)
                r.encoding = "utf-8"
                if r.status_code != 200:
                    time.sleep(2)
                    continue
                # 提取链接和日期
                items = re.findall(r'href="(/single/52/(\d+)\.html)"[^>]*>[^<]*</a>\s*<span class="time">([^<]+)</span>', r.text)
                all_before_cutoff = True
                for href, id_str, date_str in items:
                    date_clean = date_str.strip()
                    full_url = BASE + href if href.startswith('/') else href
                    try:
                        d = date.fromisoformat(date_clean[:10])
                        if d >= CUTOFF_DATE:
                            all_before_cutoff = False
                            all_urls.append((full_url, date_clean))
                    except:
                        all_urls.append((full_url, date_clean))
                if all_before_cutoff and pn > 1:
                    print(f"  第{pn}页全部超3年，停止扫描", flush=True)
                    return all_urls
                break
            except:
                if att < 2:
                    time.sleep(2)
        if pn % 10 == 0:
            print(f"  扫描页 {pn}/{MAX_PAGES}... ({len(all_urls)}条)", flush=True)
    return all_urls


def fetch_detail(url):
    for att in range(3):
        try:
            r = requests.get(url, timeout=30, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code != 200:
                time.sleep(2)
                continue
            html = r.text

            # 标题: 优先 h1.c-h1
            title = ""
            m = re.search(r'<h1 class="c-h1">(.*?)</h1>', html)
            if m:
                title = m.group(1).strip()
            if not title:
                m = re.search(r'<title>(.*?)</title>', html)
                if m:
                    title = m.group(1).replace("宁晋县人民政府 - ", "").strip()

            # 正文: c-detail 深度计数
            content = ""
            m = re.search(r'<div class="c-detail">', html)
            if m:
                start = m.end()
                depth = 1
                i = start
                while i < len(html) - 5 and depth > 0:
                    if html[i:i+4] == '<div' and html[i+4] in (' ', '>', '\n', '\t', '\r'):
                        depth += 1
                    elif html[i:i+6] == '</div>':
                        depth -= 1
                    i += 1
                content = html[start:i-6].strip()

            if not content:
                content = title

            return (url, title, content)
        except:
            if att < 2:
                time.sleep(2)
    return (url, None, None)


def main():
    mode = os.environ.get("MODE", "full")
    print(f"=== {SITE_NAME} (mode={mode}) ===", flush=True)

    # 收集URL（含日期）
    items = collect_urls()
    print(f"  总URL: {len(items)}条", flush=True)
    if not items:
        print("  无数据", flush=True)
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    cursor = conn.cursor()
    total_new = 0
    done = 0

    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
        fut_map = {}
        for url, date_str in items:
            # 日期过滤（先粗筛）
            if date_str:
                try:
                    d = date.fromisoformat(date_str[:10])
                    if d < CUTOFF_DATE:
                        continue
                except:
                    pass
            fut = executor.submit(fetch_detail, url)
            fut_map[fut] = url

        for fut in as_completed(fut_map):
            url, title, content = fut.result()
            done += 1
            if done % 50 == 0:
                print(f"  详情 {done}/{len(fut_map)}...", flush=True)
            if title is None:
                continue
            if not content:
                content = title

            # 从URL取日期作为备用
            pub_date = ""
            # Look up from items
            for u, d in items:
                if u == url:
                    pub_date = d[:10] if d else ""
                    break

            try:
                cursor.execute("""INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?)""",
                    (SITE_NAME, title, url, content, pub_date, content, 0))
                if cursor.rowcount > 0:
                    total_new += 1
            except Exception as e:
                print(f"  [DB_ERROR] {e}", flush=True)
            if done % 50 == 0:
                conn.commit()
    conn.commit()

    cursor.execute("SELECT COUNT(*), MIN(publish_date), MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    cnt, min_d, max_d = cursor.fetchone()
    conn.close()
    print(f"\n=== 完成 ===", flush=True)
    print(f"  新增: {total_new}条", flush=True)
    print(f"  累计: {cnt}条 ({min_d} ~ {max_d})", flush=True)


if __name__ == "__main__":
    main()
