#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
抚州市生态环境局 - 公告公示 (col4363) 爬虫
列表: https://hbj.jxfz.gov.cn/col/col4363/index.html
  jpage 装饰分页: POST /module/web/jpage/dataproxy.jsp
  params: page/appid/webid=34/path/columnid=4363/unitid=77467/webname(URL编码)
  返回 CDATA: <record><li><a href='http://hbj.jxfz.gov.cn/art/...' title='标题'>标题</a><span class=bt-data-time>日期</span></li></record>
  total 211 条 / 3 页
详情: /art/YYYY/M/D/art_4363_ID.html
  标题: <title> 或 h1
  正文: TRS_UEDITOR 等容器 (平衡 div)
用法: python3 crawl_jxfz_hbj_gggs.py [--pages=N]
"""
import os, sys, re, time, json, html as html_mod
import requests
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE_URL = "https://hbj.jxfz.gov.cn"
SITE_NAME = "抚州市生态环境局-公告公示"
CATEGORY = "公告公示"
MAX_PAGES = 3
CUTOFF = "2023-08-10"

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
LIST_REFERER = BASE_URL + "/col/col4363/index.html"


def parse_args():
    pages = 0
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            pages = int(a.split("=")[1])
        elif a.isdigit():
            pages = int(a)
    return pages


def clean_title(t):
    t = html_mod.unescape(t or "")
    t = re.sub(r"[\u200b\u200e\u200f\ufeff\xa0]", "", t)
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"^[•·\-—]\s*", "", t)
    return t.strip()


def fetch_list(session, page):
    url = BASE_URL + "/module/web/jpage/dataproxy.jsp"
    params = {
        "page": str(page),
        "appid": "1",
        "webid": "34",
        "path": "/",
        "columnid": "4363",
        "unitid": "77467",
        # webname 双重编码原样传（requests 会再编码 % -> %25，服务端要求如此）
        "webname": "%25E6%258A%259A%25E5%25B7%259E%25E5%25B8%2582%25E7%2594%259F%25E6%2580%2581%25E7%258E%25AF%25E5%25A2%2583%25E5%25B1%2580",
    }
    try:
        r = session.post(url, data=params, headers={
            "User-Agent": UA, "Referer": LIST_REFERER,
            "X-Requested-With": "XMLHttpRequest",
        }, timeout=30, allow_redirects=False)
        if r.status_code != 200:
            return []
        text = r.text
        items = []
        # 提取 CDATA record
        recs = re.findall(r"<record><!\[CDATA\[(.*?)\]\]></record>", text, re.S)
        if not recs:
            # 或直接 <li> 块
            recs = re.findall(r"<li.*?</li>", text, re.S)
        for rec in recs:
            m = re.search(r"<a[^>]*href='([^']+)'[^>]*title='([^']*)'[^>]*>(?:[^<]*)</a>\s*<span[^>]*>([^<]*)</span>", rec)
            if not m:
                m = re.search(r"<a[^>]*href='([^']+)'[^>]*title='([^']*)'", rec)
            if not m:
                continue
            href, title = m.group(1), clean_title(m.group(2))
            date = ""
            if len(m.groups()) >= 3:
                date = m.group(3).strip()
            if not date:
                dm = re.search(r"(\d{4}-\d{2}-\d{2})", rec)
                if dm:
                    date = dm.group(1)
            if len(title) < 4:
                continue
            if href.startswith("/"):
                href = BASE_URL + href
            elif not href.startswith("http"):
                href = urljoin(LIST_REFERER, href)
            # 域名匹配（响应 href 为 http:// 而 BASE_URL 为 https://，只比对域名）
            if "hbj.jxfz.gov.cn" not in href:
                continue
            items.append({"title": title, "href": href, "date": date})
        return items
    except Exception as e:
        print(f"    [ERR] list page {page}: {e}", flush=True)
        return []


def fetch_detail(session, url):
    try:
        r = session.get(url, headers={"User-Agent": UA, "Referer": LIST_REFERER}, timeout=30)
        html = r.text
        title = ""
        m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
        if m:
            title = clean_title(m.group(1))
        if not title:
            m = re.search(r"<h1[^>]*>([^<]+)</h1>", html)
            if m:
                title = clean_title(m.group(1))
        if not title:
            m = re.search(r"<title>([^<]+)</title>", html)
            if m:
                title = clean_title(m.group(1).split("_")[0])
        pub_date = ""
        m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
        if m:
            pub_date = m.group(1).strip()[:10]
        if not pub_date:
            m = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", html)
            if m:
                pub_date = "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))
        body = ""
        for cid in ['TRS_Editor', 'TRS_UEDITOR', 'zoom', 'article-content', 'content', 'wzcon', 'article_content_01']:
            m = re.search(r'(?:class|id)\s*=\s*["\']?[^"\'>]*\b' + re.escape(cid) + r'\b[^"\'>]*["\']?', html)
            if not m:
                continue
            i = m.start()
            start = html.find(">", i) + 1
            depth = 1
            j = start
            while j < len(html):
                if html[j:j+4] == "<div":
                    depth += 1
                    j += 4
                elif html[j:j+6] == "</div>":
                    depth -= 1
                    j += 6
                    if depth == 0:
                        break
                else:
                    j += 1
            raw = html[start:j-6]
            raw = re.sub(r"<script[\s\S]*?</script>", "", raw)
            raw = re.sub(r"<style[\s\S]*?</style>", "", raw)
            # 清理 TRS 注释/标记残留
            raw = re.sub(r"<!--[\s\S]*?-->", "", raw)
            raw = re.sub(r"<meta[^>]*ContentStart[^>]*>", "", raw)
            raw = re.sub(r"<meta[^>]*ContentEnd[^>]*>", "", raw)
            raw = re.sub(r"<\$\[[^\]]*\]>", "", raw)
            raw = re.sub(r'href="([^"]*)"', lambda m2: 'href="%s"' % (urljoin(url, m2.group(1)) if not m2.group(1).startswith(("http", "#", "javascript")) else m2.group(1)), raw)
            raw = re.sub(r'src="([^"]*)"', lambda m2: 'src="%s"' % (urljoin(url, m2.group(1)) if not m2.group(1).startswith(("http", "data:", "javascript")) else m2.group(1)), raw)
            raw = re.sub(r'\sstyle="[^"]*"', "", raw)
            raw = re.sub(r"<span[^>]*>|</span>|<strong[^>]*>|</strong>|<b[^>]*>|</b>|<font[^>]*>|</font>", "", raw)
            raw = re.sub(r'^\s*<div[^>]*>\s*', '', raw)
            raw = re.sub(r'<p[^>]*>', '<p>', raw)
            raw = re.sub(r'<p>\s*&nbsp;\s*</p>', '', raw)
            raw = re.sub(r'<p>\s*</p>', '', raw)
            if '<p>' not in raw and '<table' not in raw:
                raw = '<p>' + raw + '</p>'
            body = raw.strip()
            if len(body) >= 20:
                break
        if not body or len(body) < 20:
            body = ""
        return title, pub_date, body
    except Exception as e:
        print(f"    [ERR] detail {url}: {e}", flush=True)
        return "", "", ""


def main():
    max_pages = parse_args() or MAX_PAGES
    print(f"[JXFZ-HBJ-GGGS] SITE={SITE_NAME} max_pages={max_pages}", flush=True)
    session = requests.Session()
    session.headers.update({"User-Agent": UA})

    all_items = []
    seen = set()
    for pg in range(1, max_pages + 1):
        items = fetch_list(session, pg)
        if not items:
            print(f"  [END] page {pg} empty", flush=True)
            break
        new_count = 0
        for it in items:
            if it["href"] in seen:
                continue
            seen.add(it["href"])
            all_items.append(it)
            new_count += 1
        print(f"  Page {pg}: {len(items)}条(新{new_count}) 累计{len(all_items)}", flush=True)
        time.sleep(0.4)

    print(f"列表完成: {len(all_items)} 条, 抓详情...", flush=True)
    results = []
    for i, it in enumerate(all_items):
        d_title, d_date, body = fetch_detail(session, it["href"])
        title = d_title or it["title"]
        pub_date = d_date or it.get("date", "")
        summary = re.sub(r"<[^>]+>", " ", body) if body else title
        summary = re.sub(r"\s+", " ", summary).strip()[:300]
        atts = []
        if body:
            for m in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd|wps))"[^>]*>([^<]*)</a>', body, re.I):
                atts.append({"name": m.group(2).strip(), "url": m.group(1)})
        results.append({
            "site_name": SITE_NAME, "source_url": it["href"], "url": it["href"],
            "title": title, "pub_date": pub_date, "content": body,
            "summary": summary, "category": CATEGORY,
            "attachments": json.dumps(atts, ensure_ascii=False) if atts else "",
        })
        if (i + 1) % 10 == 0:
            print(f"  detail {i+1}/{len(all_items)}", flush=True)
        time.sleep(0.3)

    print(f"  pushing {len(results)} items...", flush=True)
    push_to_searchdb(results, batch_label=SITE_NAME)
    empty = sum(1 for r in results if not (r.get("content") or "").strip())
    print(f"  DONE. pushed={len(results)} 空正文={empty}", flush=True)


if __name__ == "__main__":
    main()
