#!/usr/bin/env python3
"""宁晋县人民政府 - 政务公开/栏目24 爬虫 (静态分页 /channel/list/24_N.html)

Same CMS structure as channel 52:
  - List:    /channel/list/24.html (p1), /channel/list/24_N.html (p2+)
  - Detail:  /single/24/{ID}.html
  - Title:   <h1 class="c-h1">
  - Content: <div class="c-detail"> (depth-counting)
  - Date:    <span class="time">YYYY-MM-DD</span> in list page
"""

import os, sqlite3, re, time, socket, requests
from datetime import datetime, date
from concurrent.futures import ThreadPoolExecutor, as_completed

# ── Custom DNS resolution ──────────────────────────────────
# The server's default DNS can't resolve some .gov.cn domains.
# We force-resolve www.ningjin.gov.cn to 183.196.191.98.
_FORCED_IP = "183.196.191.98"
_FORCED_HOST = "www.ningjin.gov.cn"
_original_getaddrinfo = socket.getaddrinfo

def _patched_getaddrinfo(host, port, *args, **kwargs):
    if host == _FORCED_HOST:
        return [(socket.AF_INET, socket.SOCK_STREAM, 6, "", (_FORCED_IP, port))]
    return _original_getaddrinfo(host, port, *args, **kwargs)

# Apply the patch
socket.getaddrinfo = _patched_getaddrinfo

BASE = "https://www.ningjin.gov.cn"
LIST_URL = "https://www.ningjin.gov.cn/channel/list/24.html"
PAGE_URL = "https://www.ningjin.gov.cn/channel/list/24_{}.html"
CUTOFF_DATE = date(2023, 6, 17)  # 3年截止
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "宁晋县-政务公开"
MAX_WORKERS = 8
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml",
}
MAX_PAGES = 15  # 实际只有10页有数据


def collect_urls():
    """收集所有文章URL，超3年自动停"""
    all_urls = []
    for pn in range(1, MAX_PAGES + 1):
        url = LIST_URL if pn == 1 else PAGE_URL.format(pn)
        for att in range(3):
            try:
                r = requests.get(url, timeout=30, headers=HEADERS)
                r.encoding = "utf-8"
                if r.status_code != 200:
                    time.sleep(2)
                    continue
                # 提取 /single/24/ID.html 链接 + 日期
                items = re.findall(
                    r'href="(/single/24/(\d+)\.html)"[^>]*>(.*?)</a>\s*<span[^>]*class="[^"]*time[^"]*"[^>]*>([^<]+)</span>',
                    r.text,
                    re.S,
                )
                all_before_cutoff = True if pn > 1 else False
                new_count = 0
                for href, id_str, title_html, date_str in items:
                    title = re.sub(r"<[^>]+>", "", title_html).strip()
                    if not title or title == "登录":
                        continue
                    date_clean = date_str.strip()[:10]
                    full_url = BASE + href
                    try:
                        d = date.fromisoformat(date_clean)
                        if d >= CUTOFF_DATE:
                            all_before_cutoff = False
                            all_urls.append((full_url, date_clean))
                            new_count += 1
                            if new_count <= 3:
                                print(f"    [{date_clean}] {title[:50]}", flush=True)
                    except:
                        all_urls.append((full_url, date_clean))
                        new_count += 1
                if new_count == 0 and pn > 1:
                    print(f"  第{pn}页无有效数据，停止扫描", flush=True)
                    return all_urls
                if all_before_cutoff and pn > 1:
                    print(f"  第{pn}页全部超3年({items[-1][3] if items else 'N/A'}), 停止扫描", flush=True)
                    return all_urls
                break
            except Exception as e:
                if att < 2:
                    time.sleep(2)
        if pn % 5 == 0:
            print(f"  扫描页 {pn}/{MAX_PAGES}... ({len(all_urls)}条)", flush=True)
    return all_urls


def fetch_detail(url):
    """抓取详情页: 标题(h1.c-h1) + 正文(div.c-detail)"""
    for att in range(3):
        try:
            r = requests.get(url, timeout=30, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code != 200:
                time.sleep(2)
                continue
            html = r.text

            # 标题: h1.c-h1
            title = ""
            m = re.search(r'<h1[^>]*class="[^"]*c-h1[^"]*"[^>]*>(.*?)</h1>', html, re.S)
            if m:
                title = re.sub(r"<[^>]+>", "", m.group(1)).strip()
            if not title:
                m = re.search(r"<title>(.*?)</title>", html)
                if m:
                    title = m.group(1).replace("宁晋县人民政府", "").strip(" -")

            # 正文: c-detail 深度计数
            content = ""
            m = re.search(r'<div[^>]*class="[^"]*c-detail[^"]*"[^>]*>', html)
            if m:
                start = m.end()
                depth = 1
                i = start
                while i < len(html) - 6 and depth > 0:
                    # <div opener
                    if html[i : i + 4] == "<div" and html[i + 4] in (" ", ">", "\n", "\t", "\r", "/"):
                        if html[i : i + 5] != "<div/":
                            depth += 1
                    # </div> closer
                    if html[i : i + 6] == "</div>":
                        depth -= 1
                    i += 1
                content = html[start : i - 6].strip()
            if not content:
                content = title

            return (url, title, content)
        except:
            if att < 2:
                time.sleep(2)
    return (url, None, None)


def main():
    mode = os.environ.get("MODE", "full")
    print(f"=== {SITE_NAME} (mode={mode}) ===", flush=True)

    # 收集URL（含日期）
    items = collect_urls()
    print(f"  总URL: {len(items)}条", flush=True)
    if not items:
        print("  无数据", flush=True)
        return

    if mode == "test":
        print("TEST模式: 只抓前2条验证", flush=True)
        items = items[:2]
        conn = None
    else:
        conn = sqlite3.connect(DB_PATH)
        cursor = conn.cursor()
    total_new = 0
    done = 0

    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
        fut_map = {}
        for url, date_str in items:
            if date_str:
                try:
                    d = date.fromisoformat(date_str[:10])
                    if d < CUTOFF_DATE:
                        continue
                except:
                    pass
            fut = executor.submit(fetch_detail, url)
            fut_map[fut] = (url, date_str)

        for fut in as_completed(fut_map):
            url, title, content = fut.result()
            date_str = fut_map[fut][1]
            done += 1
            if done % 50 == 0:
                print(f"  详情 {done}/{len(fut_map)}...", flush=True)
            if title is None:
                print(f"  [FAIL] {url}", flush=True)
                continue
            if not content:
                content = title

            pub_date = date_str[:10] if date_str else ""

            if mode == "test":
                print(f"\n  === 测试: [{pub_date}] {title[:50]} ===")
                print(f"  URL: {url}")
                print(f"  正文长度: {len(content)} 字符")
                print(f"  正文预览: {content[:200]}")
                continue

            try:
                cursor.execute(
                    """INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?)""",
                    (SITE_NAME, title, url, content, pub_date, content, 0),
                )
                if cursor.rowcount > 0:
                    total_new += 1
            except Exception as e:
                print(f"  [DB_ERROR] {e}", flush=True)
            if done % 50 == 0:
                conn.commit()

    if mode == "test":
        print(f"\n=== 测试完成 ({len(fut_map)}条) ===")
        return

    conn.commit()
    cursor.execute(
        "SELECT COUNT(*), MIN(publish_date), MAX(publish_date) FROM gov_raw WHERE site_name=?",
        (SITE_NAME,),
    )
    cnt, min_d, max_d = cursor.fetchone()
    conn.close()
    print(f"\n=== 完成 ===", flush=True)
    print(f"  新增: {total_new}条", flush=True)
    print(f"  累计: {cnt}条 ({min_d} ~ {max_d})", flush=True)


if __name__ == "__main__":
    main()
