#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
茂名市自然资源局 - 新闻动态/批前公示（pqgs）
http://zrzyj.maoming.gov.cn/xwdt/pqgs/

站点特征（2026-08-11 探测）:
- 政府门户网站群模板（nfw-cms），列表 6条/页，index_N.html 分页，共 200 页（P201 起 404）
- 列表标题完整（a 文本即完整标题，无截断）
- 详情页: h1 标题 + meta ArticleTitle/PubDate + <article class="info"> 正文（<p> 段落）
- 附件: <a class="nfw-cms-attachment" href="绝对URL">（attachment/0/x/x/ID.ext）
- 日期范围 2026-08-07 ~ 2024-05-21，全量在 CUTOFF 窗口内（无截断）

用法:
  python3 crawl_maoming_pqgs.py --pages=5    # 首批前5页
  python3 crawl_maoming_pqgs.py --pages=200  # 全量
  python3 crawl_maoming_pqgs.py --pages=1    # 测试
"""
import re
import sys
import json
import html
import urllib3
import requests
from urllib.parse import urljoin

urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

from crawler_lib import push_to_searchdb

SITE_NAME = "茂名市自然资源局-批前公示"
BASE_URL = "http://zrzyj.maoming.gov.cn"
LIST_URL = BASE_URL + "/xwdt/pqgs/index.html"
CUTOFF = "2023-08-10"   # 3 年窗口（本站实际最老 2024-05-21，全量无截断）
CATEGORY = "自然资源"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Referer": LIST_URL,
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}

def clean_title(t):
    t = html.unescape(t or "")
    t = re.sub(r"^(?:&middot;|&nbsp;|\s|•|·)+", "", t)
    t = t.replace("\u200b", "").replace("\ufeff", "")
    return t.strip()

def fetch_list_page(page):
    """返回列表 items"""
    if page == 1:
        url = LIST_URL
    else:
        url = f"{BASE_URL}/xwdt/pqgs/index_{page}.html"
    r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
    if r.status_code != 200:
        return []
    r.encoding = r.apparent_encoding or "utf-8"
    raw = r.text
    items = []
    for m in re.finditer(
        r"<a[^>]+href=\"([^\"]*post_\d+\.html)\"[^>]*>(.*?)</a>",
        raw, re.S):
        href, title = m.group(1), m.group(2)
        title = clean_title(title)
        if not title:
            continue
        detail_url = urljoin(url, href)
        items.append({"title": title, "url": detail_url})
    # 去重（保持顺序）
    seen = set()
    out = []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            out.append(it)
    return out

def fetch_detail(detail_url, list_title):
    """返回 (content_html, attachments, pub_date)"""
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30, verify=False)
        r.raise_for_status()
        r.encoding = r.apparent_encoding or "utf-8"
        raw = r.text
    except Exception:
        return None, [], None
    # 标题: meta ArticleTitle
    title = list_title
    tm = re.search(r'<meta name="ArticleTitle" content="([^"]+)"', raw)
    if tm:
        title = clean_title(tm.group(1)) or title
    # 日期: meta PubDate
    pub_date = None
    dm = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})"', raw)
    if dm:
        pub_date = dm.group(1)
    # 附件: nfw-cms-attachment 绝对 URL
    atts = []
    for am in re.finditer(r'<a[^>]*class="nfw-cms-attachment"[^>]*href="([^"]+)"[^>]*>(.*?)</a>', raw, re.S):
        href, name = am.group(1), am.group(2)
        abs_url = href if href.startswith("http") else urljoin(detail_url, href)
        name = clean_title(re.sub(r"<[^>]+>", "", name)) or abs_url.split("/")[-1]
        atts.append({"name": name, "url": abs_url})
    # 正文: <article class="info"> 内 <p> 段落，保留 HTML
    body = ""
    m = re.search(r'<article[^>]*class="info"[^>]*>(.*?)</article>', raw, re.S)
    if m:
        body = m.group(1)
    else:
        # 回退: nfw-cms-content 或 TRS 容器
        for pat in [r'class="nfw-cms-content"[^>]*>(.*?)</div>', r'class="TRS_Editor"[^>]*>(.*?)</div>']:
            mm = re.search(pat, raw, re.S)
            if mm:
                body = mm.group(1)
                break
    if body:
        # 只删含 nfw-cms-attachment 链接的段落（防附件段重复），保留正文段落
        body = re.sub(
            r'<p[^>]*>(?:(?!</p>).)*nfw-cms-attachment(?:(?!</p>).)*</p>',
            '', body, flags=re.S)
        # 保留 <p> 段落（规范化），剥 span 保留文本
        body = re.sub(r'<span[^>]*>', '', body)
        body = re.sub(r'</span>', '', body)
        body = re.sub(r'<p[^>]*>', '<p>', body)
        body = re.sub(r'<br\s*/?>', '', body)
        body = re.sub(r'<p>\s*</p>', '', body)  # 删空段落
        # 附件链接若已在正文保留（绝对 URL）
        for a in atts:
            if a["url"] not in body:
                body += f'<p><a href="{a["url"]}">{a["name"]}</a></p>'
    else:
        # 无正文容器 → 标题段 + 附件段
        parts = [f"<p>{title}</p>"]
        for a in atts:
            parts.append(f'<p><a href="{a["url"]}">{a["name"]}</a></p>')
        body = "".join(parts)
    return body, atts, pub_date

def main():
    max_pages = None
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            max_pages = int(a.split("=")[1])
        elif a.isdigit():
            max_pages = int(a)
    all_items = []
    seen = set()
    page = 1
    while True:
        if max_pages and page > max_pages:
            break
        items = fetch_list_page(page)
        if not items:
            print(f"[P{page}] empty/404, stop")
            break
        new = 0
        for it in items:
            if it["url"] in seen:
                continue
            seen.add(it["url"])
            all_items.append(it)
            new += 1
        print(f"[P{page}] got {len(items)} items, new {new}, total {len(all_items)}")
        page += 1
    print(f"[AutoPg] max_pages={max_pages}, collected {len(all_items)}")

    valid = []
    for i, it in enumerate(all_items):
        content, atts, pub_date = fetch_detail(it["url"], it["title"])
        if content is None:
            content = f"<p>{it['title']}</p>"
        valid.append({
            "site_name": SITE_NAME,
            "title": it["title"],
            "pub_date": pub_date or "",
            "content": content,
            "source_url": it["url"],
            "url": it["url"],
            "attachments": json.dumps(atts, ensure_ascii=False) if atts else "[]",
        })
        if (i + 1) % 50 == 0:
            print(f"[Detail] {i+1}/{len(all_items)}")
    print(f"[Detail] done {len(valid)}")
    push_to_searchdb(valid, CATEGORY)

if __name__ == "__main__":
    main()
