#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
石家庄高新技术产业开发区 - 公示公告 爬虫
列表: http://www.shidz.gov.cn/columns/fc6b4309-d198-4db4-9c9a-8358ef6ba98c/index.html
  首页静态渲染 15 条; 分页走 blocks AJAX 接口 (关键: 必须带 fix=0, 否则返回 0B):
    /columns/{colid}/templates/{tid}/blocks/{blockid}?page=N&fix=0
    colid=fc6b4309-d198-4db4-9c9a-8358ef6ba98c
    tid=146f96b5-fcde-4c1c-8672-7399dfeb4a44
    blockid=cd2eb42d-5993-49fc-a805-b3e2aaded14a
  总数: 13385 条 / 893 页 (每页 15 条)
详情: /columns/UUID/YYYYMM/DD/UUID.html
  标题: #biaoti 内 div; 日期: 发布时间：YYYY-MM-DD; 正文: #conN
用法: python3 crawl_shidz_gsgg.py [--pages=N] [--limit=N]
"""
import os, sys, re, time, json, html as html_mod
import requests
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE_URL = "http://www.shidz.gov.cn"
SITE_NAME = "石家庄高新技术产业开发区-公示公告"
CATEGORY = "公示公告"
COLID = "fc6b4309-d198-4db4-9c9a-8358ef6ba98c"
TID = "146f96b5-fcde-4c1c-8672-7399dfeb4a44"
BLOCKID = "cd2eb42d-5993-49fc-a805-b3e2aaded14a"
BLOCK_URL = f"{BASE_URL}/columns/{COLID}/templates/{TID}/blocks/{BLOCKID}"
PAGE_SIZE = 15
TOTAL_PAGES = 893
CUTOFF = "2020-01-01"  # 早于 2020 年过滤

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
HEADERS = {"User-Agent": UA, "Accept": "*/*", "Referer": f"{BASE_URL}/columns/{COLID}/index.html"}


def parse_args():
    pages = 0
    limit = 0
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            pages = int(a.split("=")[1])
        elif a.startswith("--limit="):
            limit = int(a.split("=")[1])
    return pages, limit


def clean_title(t):
    t = html_mod.unescape(t or "")
    t = re.sub(r"[\u200b\u200e\u200f\ufeff\xa0]", "", t)
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"^[•·\-—]\s*", "", t)
    return t.strip()


def fetch_page(session, page):
    """请求 blocks 分页接口, 返回 [{title, href, date}]"""
    try:
        r = session.get(BLOCK_URL, params={"page": page, "fix": 0}, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        html = r.text
        if not html or len(html) < 100:
            print(f"    [WARN] page {page} 空响应 ({len(html)}B)", flush=True)
            return []
        items = []
        seen = set()
        # 匹配: <span>YYYY-MM-DD</span> <a href=".../YYYYMM/DD/UUID.html" title="标题">
        for m in re.finditer(r"<span>(\d{4}-\d{2}-\d{2})</span>\s*<a href=\"([^\"]+)\"[^>]*title=\"([^\"]+)\"", html):
            date, href, title = m.group(1), m.group(2), m.group(3)
            if href in seen:
                continue
            seen.add(href)
            href = urljoin(BASE_URL, href) if not href.startswith("http") else href
            t = clean_title(title)
            if len(t) < 4:
                continue
            items.append({"title": t, "href": href, "date": date})
        return items
    except Exception as e:
        print(f"    [ERR] page {page}: {e}", flush=True)
        return []


def fetch_detail(session, url):
    """详情: #biaoti 标题 + 发布时间 + #conN 正文 (与 crawl_shidz_tzgg.py 同款)"""
    try:
        r = session.get(url, headers={"User-Agent": UA, "Referer": BASE_URL + "/"}, timeout=30)
        r.encoding = "utf-8"
        html = r.text
        title = ""
        i = html.find('id="biaoti"')
        if i >= 0:
            m = re.search(r"<div[^>]*>\s*([^<]{4,100})\s*</div>", html[i:i+800])
            if m:
                title = clean_title(m.group(1))
        if not title:
            m = re.search(r"<title>([^<]+)</title>", html)
            if m:
                title = clean_title(m.group(1).split("_")[0])
        pub_date = ""
        m = re.search(r"发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})", html)
        if m:
            pub_date = m.group(1)
        if not pub_date:
            m = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", html)
            if m:
                pub_date = "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))
        # 正文: #conN
        body = ""
        i = html.find('id="conN"')
        if i >= 0:
            start = html.find(">", i) + 1
            depth = 1
            j = start
            while j < len(html) and depth > 0:
                if html[j:j+4] == "<div":
                    depth += 1
                    j += 4
                elif html[j:j+6] == "</div>":
                    depth -= 1
                    j += 6
                else:
                    j += 1
            raw = html[start:j-6]
            raw = re.sub(r"<script[\s\S]*?</script>", "", raw)
            raw = re.sub(r"<style[\s\S]*?</style>", "", raw)
            raw = re.sub(r'href="([^"]*)"', lambda m: 'href="%s"' % (urljoin(url, m.group(1)) if not m.group(1).startswith(("http", "#", "javascript")) else m.group(1)), raw)
            raw = re.sub(r'src="([^"]*)"', lambda m: 'src="%s"' % (urljoin(url, m.group(1)) if not m.group(1).startswith(("http", "data:", "javascript")) else m.group(1)), raw)
            raw = re.sub(r'\sstyle="[^"]*"', "", raw)
            raw = re.sub(r"<span[^>]*>|</span>|<strong[^>]*>|</strong>|<b[^>]*>|</b>|<font[^>]*>|</font>", "", raw)
            body = raw.strip()
        if not body or len(body) < 20:
            body = ""
        return title, pub_date, body
    except Exception as e:
        print(f"    [ERR] detail {url}: {e}", flush=True)
        return "", "", ""


def main():
    pages, limit = parse_args()
    print(f"[SHIDZ-GSGG] SITE={SITE_NAME} pages={pages} limit={limit}", flush=True)

    session = requests.Session()
    session.headers.update({"User-Agent": UA})

    all_items = []
    seen = set()
    if pages <= 0:
        # 默认只抓首页 15 条
        pages = 1
    for pg in range(1, pages + 1):
        items = fetch_page(session, pg)
        new_count = 0
        for it in items:
            if it["href"] in seen:
                continue
            seen.add(it["href"])
            all_items.append(it)
            new_count += 1
        print(f"  [page {pg}] {len(items)}条(新{new_count}), 累计{len(all_items)}", flush=True)
        if not items:
            print(f"  [page {pg}] 空页, 停止翻页", flush=True)
            break
        time.sleep(0.4)

    # CUTOFF 过滤
    before = len(all_items)
    all_items = [it for it in all_items if (it.get("date") or "9999") >= CUTOFF]
    if len(all_items) < before:
        print(f"  CUTOFF {CUTOFF} 过滤 {before - len(all_items)} 条", flush=True)

    print(f"列表完成: {len(all_items)} 条, 抓详情...", flush=True)
    if limit and len(all_items) > limit:
        all_items = all_items[:limit]

    results = []
    for i, it in enumerate(all_items):
        d_title, d_date, body = fetch_detail(session, it["href"])
        title = d_title or it["title"]
        pub_date = d_date or it.get("date", "")
        summary = re.sub(r"<[^>]+>", " ", body) if body else title
        summary = re.sub(r"\s+", " ", summary).strip()[:300]
        atts = []
        if body:
            for m in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd|wps))"[^>]*>([^<]*)</a>', body, re.I):
                atts.append({"name": m.group(2).strip(), "url": m.group(1)})
        results.append({
            "site_name": SITE_NAME, "source_url": it["href"], "url": it["href"],
            "title": title, "pub_date": pub_date, "content": body,
            "summary": summary, "category": CATEGORY,
            "attachments": json.dumps(atts, ensure_ascii=False) if atts else "",
        })
        if (i + 1) % 10 == 0:
            print(f"  detail {i+1}/{len(all_items)}", flush=True)
        time.sleep(0.3)

    print(f"  pushing {len(results)} items to searchdb...", flush=True)
    push_to_searchdb(results, batch_label=SITE_NAME)
    empty = sum(1 for r in results if not (r.get("content") or "").strip())
    print(f"  DONE. pushed={len(results)} 空正文={empty}", flush=True)


if __name__ == "__main__":
    main()
