#!/usr/bin/env python3
"""
crawl_rushan_gggs.py - 乳山市政府-通知公告 (col68544)
站点: www.rushan.gov.cn (TRS jpage 系, 无 WAF, 全 curl 直连)
列表: POST /module/web/jpage/dataproxy.jsp?startrecord=N&endrecord=M&perpage=15
  body: col=1&appid=1&webid=122&path=/&columnid=68544&sourceContentType=1&unitid=498890
        &webname=乳山市政府&permissiontype=0
  返回 XML: <totalrecord>301</totalrecord> + <record><![CDATA[<li><a href title>标题</a><span>[日期]</span>...
  组缓存: 每组返回 46 条(3页), startrecord 步进 15
详情: /art/YYYY/M/D/art_68544_XXXX.html, meta ArticleTitle/PubDate, 正文 ContentStart->ContentEnd
"""
import sys, os, re, json, urllib.request, urllib.parse
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE = "http://www.rushan.gov.cn"
SITE_NAME = "乳山市-通知公告"
PROXY = f"{BASE}/module/web/jpage/dataproxy.jsp"
PARAM = {
    "col": "1", "appid": "1", "webid": "122", "path": "/",
    "columnid": "68544", "sourceContentType": "1", "unitid": "498890",
    "webname": "乳山市政府", "permissiontype": "0",
}
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"


def http_get(url):
    req = urllib.request.Request(url, headers={"User-Agent": UA})
    with urllib.request.urlopen(req, timeout=40) as r:
        return r.read().decode('utf-8', errors='ignore')


def http_post(url, data):
    body = urllib.parse.urlencode(data).encode()
    req = urllib.request.Request(url, data=body, headers={"User-Agent": UA, "Content-Type": "application/x-www-form-urlencoded"})
    with urllib.request.urlopen(req, timeout=40) as r:
        return r.read().decode('utf-8', errors='ignore')


def fetch_group(start):
    url = f"{PROXY}?startrecord={start}&endrecord={start + 29}&perpage=15"
    return http_post(url, PARAM)


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=5)
    parser.add_argument("--all", action="store_true")
    args = parser.parse_args()
    pages = 21 if args.all else args.pages
    print(f"  {SITE_NAME} pages={pages}")

    # 1. 列表: 每组 startrecord 步进 15, 每组 46 条(3页缓存)
    items = []
    seen = set()
    for g in range(pages):
        start = g * 15 + 1
        try:
            txt = fetch_group(start)
        except Exception as e:
            print(f"  [LIST ERR g{g+1}] {str(e)[:100]}")
            break
        records = re.findall(r'<record><!\[CDATA\[([\s\S]*?)\]\]></record>', txt)
        total_m = re.search(r'<totalrecord>(\d+)</totalrecord>', txt)
        total = int(total_m.group(1)) if total_m else 0
        for rec in records:
            m = re.search(r'<a href=\\?\'([^\'"]*art_[^\'"]*)\'?\\?[^>]*title=\\?\'([^\'"]*)\'?', rec)
            if not m:
                m = re.search(r'<a href="([^"]*art_[^"]*)"[^>]*title="([^"]*)"', rec)
            if not m:
                continue
            url, title = m.group(1), m.group(2)
            dm = re.search(r'\[(\d{4}-\d{2}-\d{2})\]', rec)
            date = dm.group(1) if dm else ''
            full = url if url.startswith('http') else BASE + url
            if full not in seen:
                seen.add(full)
                items.append({"url": full, "title": title.strip(), "date": date})
        print(f"  组{g+1} (start={start}): {len(records)} 条, 累计 {len(items)} (total={total})")
        if start + 29 >= total:
            break

    # 2. 详情页
    records = []
    for idx, it in enumerate(items):
        print(f"  [{idx+1}/{len(items)}] {it['title'][:38]}...")
        try:
            html = http_get(it['url'])
        except Exception as e:
            print(f"    [DETAIL ERR] {str(e)[:100]}")
            continue
        title = it['title']
        m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
        if m and m.group(1).strip():
            title = m.group(1).strip()
        pub_date = it['date']
        m = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', html)
        if m:
            dm = re.search(r'\d{4}-\d{2}-\d{2}', m.group(1))
            if dm:
                pub_date = dm.group()
        content = ""
        si = html.find('<meta name="ContentStart">')
        ei = html.find('<meta name="ContentEnd">')
        if si > 0 and ei > si:
            content = html[si + len('<meta name="ContentStart">'):ei].strip()
        if not content:
            m = re.search(r'<!--<\$\[信息内容\]>begin-->([\s\S]*?)<!--<\$\[信息内容\]>end-->', html)
            if m:
                content = m.group(1).strip()
        if not content.strip():
            print("    [SKIP] 正文空")
            continue
        content = re.sub(r'href="/', 'href="' + BASE + '/', content)
        content = re.sub(r'src="/', 'src="' + BASE + '/', content)
        records.append({
            "title": title,
            "url": it['url'],
            "pub_date": pub_date,
            "site_name": SITE_NAME,
            "content": content,
            "summary": "",
        })

    print(f"  共 {len(records)} 条有效")
    if records:
        push_to_searchdb(records, "rushan_gggs")
    return len(records)


if __name__ == "__main__":
    main()
    sys.exit(0)
