#!/usr/bin/env python3
"""
crawl_yichang_fgw.py - 宜昌市发改委-行政服务中心窗口公告 (fgw.yichang.gov.cn)
站点: 无防护纯静态 (Bootstrap 页面)
列表: list-40625-{N}.html (N=1..6), 每页 20 条
  列表项: <div class="list-views"><div class="listview-bt"><a href="content-40625-{id}-1.html">标题</a></div><div class="listview-date">日期</div></div>
详情: content-40625-{id}-1.html
  meta ArticleTitle/PubDate/ContentSource, 正文 div.txtcontent-div (截断 <!--main-->)
  内容: 项目审批核准和备案公告 (月度, 审批类+备案类, 含项目代码)
"""
import sys, os, re, time, urllib.request
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE = "http://fgw.yichang.gov.cn"
LIST_ID = "40625"
SITE_NAME = "宜昌市发改委-行政服务中心窗口公告"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"


def fetch(url, timeout=30):
    req = urllib.request.Request(url, headers={"User-Agent": UA})
    with urllib.request.urlopen(req, timeout=timeout) as resp:
        return resp.read().decode("utf-8", errors="replace")


def run(pages=5):
    # 列表
    items_all = []
    seen = set()
    for pg in range(1, pages + 1):
        url = f"{BASE}/list-{LIST_ID}-{pg}.html"
        try:
            html = fetch(url)
        except Exception as e:
            print(f"  [LIST ERR p{pg}] {str(e)[:100]}")
            break
        rows = re.findall(
            r'<div class="listview-bt"><a href="([^"]*content-\d+-\d+-1\.html)"[^>]*>(?:<!--.*?-->)?([^<]{5,200})</a></div>\s*<div class="listview-date">([^<]*)</div>',
            html
        )
        new_rows = 0
        for u, t, d in rows:
            u2 = u if u.startswith('http') else BASE + u
            title = re.sub(r'<!--.*?-->', '', t).strip()
            if title and u2 not in seen:
                seen.add(u2)
                items_all.append({"url": u2, "title": title, "date": d.strip()[:10]})
                new_rows += 1
        print(f"  第{pg}页: 解析 {len(rows)} 条, 新增 {new_rows} (累计 {len(items_all)})")
        if len(rows) < 10:
            print(f"  [p{pg}] 列表结束")
            break

    print(f"  [LIST] 去重后 {len(items_all)} 条")

    # 详情
    records = []
    for idx, it in enumerate(items_all):
        print(f"  [{idx+1}/{len(items_all)}] {it['title'][:40]}...")
        try:
            dh = fetch(it['url'])
        except Exception as e:
            print(f"    [DETAIL ERR] {str(e)[:100]}")
            continue
        title = it['title']
        m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', dh)
        if m and m.group(1).strip():
            title = m.group(1).strip()
        pub_date = it['date']
        m = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', dh)
        if m:
            dm = re.search(r'\d{4}-\d{2}-\d{2}', m.group(1))
            if dm:
                pub_date = dm.group()
        content = ""
        m = re.search(r'<div class="txtcontent-div">([\s\S]*?)<!--main-->', dh)
        if m:
            content = m.group(1).strip()
        if not content:
            m = re.search(r'<div class="txtcontent-div">([\s\S]*?)</div>\s*</div>\s*</div>', dh)
            if m:
                content = m.group(1).strip()
        if not content.strip():
            print("    [SKIP] 正文空")
            continue
        content = re.sub(r'href="/', 'href="' + BASE + '/', content)
        content = re.sub(r'src="/', 'src="' + BASE + '/', content)
        records.append({
            "title": title,
            "url": it['url'],
            "pub_date": pub_date,
            "site_name": SITE_NAME,
            "content": content,
            "summary": "",
        })
        time.sleep(0.3)

    print(f"  共 {len(records)} 条有效")
    if records:
        push_to_searchdb(records, "yichang_fgw")
    return len(records)


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=5)
    args = parser.parse_args()
    n = run(pages=args.pages)
    print(f"Done: {n} records")
    return 0


if __name__ == "__main__":
    sys.exit(main())
