#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
泰安高新技术产业开发区 - 通知公告 (col47833) 爬虫
列表: http://gxq.taian.gov.cn/col/col47833/index.html
  大汉网络 (hanweb) 系统, jquery.jpage.js 范围式分页
  API: POST /module/web/jpage/dataproxy.jsp
    URL参数: startrecord/endrecord/perpage/unitid/webid/path/webname/col/columnid/sourceContentType/permissiontype
    POST body: 同 ajaxParam (必须, 否则空响应)
  totalRecord=2143, perPage=22 => 98 页
详情: /art/YYYY/M/D/art_47833_ID.html
  标题: <meta name="ArticleTitle"> (完整标题)
  正文: <div class="bt_content">
用法: python3 crawl_taian_gxq_tzgg.py [--pages=N] [--limit=N]
"""
import os, sys, re, time, json, html as html_mod
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin, quote

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE_URL = "http://gxq.taian.gov.cn"
LIST_URL = BASE_URL + "/col/col47833/index.html"
SITE_NAME = "泰安高新技术产业开发区-通知公告"
CATEGORY = "通知公告"
MAX_PAGES = 98   # totalRecord=2143 / perPage=22 ≈ 98 页
PER_PAGE = 22

WEBNAME = "泰安高新技术产业开发区"
PATH_ENC = quote(BASE_URL + "/", safe="")
WEBNAME_ENC = quote(quote(WEBNAME, safe=""), safe="")

API_URL = "/module/web/jpage/dataproxy.jsp"
AJAX_PARAM = {
    "col": "1", "webid": "334", "path": BASE_URL + "/",
    "columnid": "47833", "sourceContentType": "1", "unitid": "144798",
    "webname": WEBNAME, "permissiontype": "0",
}

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
HEADERS = {"User-Agent": UA, "Referer": LIST_URL, "X-Requested-With": "XMLHttpRequest",
           "Accept": "application/xml, text/xml, */*"}


def parse_args():
    pages, limit = 0, 0
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            pages = int(a.split("=")[1])
        elif a.startswith("--limit="):
            limit = int(a.split("=")[1])
    return pages, limit


def clean_title(t):
    t = html_mod.unescape(t or "")
    t = re.sub(r"[\u200b\u200e\u200f\ufeff\xa0]", "", t)
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"^[•·\-—]\s*", "", t)
    return t.strip()


def fetch_list_range(session, start, end):
    """POST dataproxy.jsp 范围分页 (startrecord~endrecord)"""
    params = {
        "startrecord": start, "endrecord": end, "perpage": PER_PAGE,
        "unitid": AJAX_PARAM["unitid"], "webid": AJAX_PARAM["webid"],
        "path": PATH_ENC, "webname": WEBNAME_ENC,
        "col": AJAX_PARAM["col"], "columnid": AJAX_PARAM["columnid"],
        "sourceContentType": AJAX_PARAM["sourceContentType"], "permissiontype": AJAX_PARAM["permissiontype"],
    }
    for attempt in range(3):
        try:
            r = session.post(BASE_URL + API_URL, params=params, data=AJAX_PARAM, headers=HEADERS, timeout=30)
            if r.status_code != 200:
                print(f"    [WARN] list {start}-{end} status={r.status_code}, retry", flush=True)
                time.sleep(3)
                continue
            text = r.text
            total_m = re.search(r"<totalrecord>(\d+)<", text)
            total = int(total_m.group(1)) if total_m else 0
            items = []
            for rec in re.findall(r"<record><!\[CDATA\[(.*?)\]\]></record>", text, re.S):
                am = re.search(r'<a href="([^"]+)"[^>]*title="([^"]*)"', rec)
                if not am:
                    am = re.search(r'<a href="([^"]+)"[^>]*>([^<]+)</a>', rec)
                if not am:
                    continue
                href, title = am.group(1), clean_title(am.group(2))
                if len(title) < 4:
                    continue
                date = ""
                dm = re.search(r"(\d{4})/(\d{1,2})/(\d{1,2})", rec)
                if dm:
                    date = "%s-%02d-%02d" % (dm.group(1), int(dm.group(2)), int(dm.group(3)))
                items.append({"title": title, "href": href, "date": date})
            return total, items
        except Exception as e:
            print(f"    [WARN] list {start}-{end} err {e}, retry {attempt}", flush=True)
            time.sleep(3)
    return 0, []


def fetch_detail(session, url):
    """详情页: meta ArticleTitle + div.bt_content"""
    try:
        r = session.get(url, headers={"User-Agent": UA, "Referer": LIST_URL}, timeout=30)
        r.encoding = "utf-8"
        html = r.text
        title = ""
        m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
        if m:
            title = clean_title(m.group(1))
        if not title:
            m = re.search(r"<h1[^>]*>([^<]+)</h1>", html)
            if m:
                title = clean_title(m.group(1))
        pub_date = ""
        m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
        if m:
            pub_date = m.group(1).strip()[:10]
        if not pub_date:
            m = re.search(r"(\d{4})[-/](\d{1,2})[-/](\d{1,2})", html)
            if m:
                pub_date = "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))
        body = ""
        i = html.find('class="bt_content"')
        if i >= 0:
            start = html.find(">", i) + 1
            depth = 0
            j = start
            while j < len(html):
                if html[j:j+4] == "<div":
                    depth += 1
                    j += 4
                elif html[j:j+6] == "</div>":
                    depth -= 1
                    j += 6
                    if depth == 0:
                        break
                else:
                    j += 1
            raw = html[start:j-6]
            raw = re.sub(r"<script[\s\S]*?</script>", "", raw)
            raw = re.sub(r"<style[\s\S]*?</style>", "", raw)
            raw = re.sub(r'href="(/[^"]*)"', lambda m: 'href="%s%s"' % (BASE_URL, m.group(1)), raw)
            raw = re.sub(r'src="(/[^"]*)"', lambda m: 'src="%s%s"' % (BASE_URL, m.group(1)), raw)
            raw = re.sub(r'\sstyle="[^"]*"', "", raw)
            raw = re.sub(r"<span[^>]*>|</span>|<strong[^>]*>|</strong>|<b[^>]*>|</b>|<font[^>]*>|</font>", "", raw)
            body = raw.strip()
        if not body or len(body) < 20:
            body = ""
        return title, pub_date, body
    except Exception as e:
        print(f"    [ERR] detail {url}: {e}", flush=True)
        return "", "", ""


def main():
    max_pages, limit = parse_args()
    if not max_pages:
        max_pages = MAX_PAGES
    print(f"[Taian-GXQ] SITE={SITE_NAME} max_pages={max_pages}", flush=True)

    session = requests.Session()
    session.headers.update({"User-Agent": UA})

    # 访问首页拿 cookie
    try:
        session.get(LIST_URL, timeout=30)
    except Exception:
        pass

    all_items = []
    seen = set()
    total = 0
    for pg in range(1, max_pages + 1):
        start = (pg - 1) * PER_PAGE + 1
        end = pg * PER_PAGE
        total, items = fetch_list_range(session, start, end)
        new_count = 0
        for it in items:
            href = it["href"]
            if not href.startswith("http"):
                href = urljoin(BASE_URL, href)
            # 外链文章（微信/第三方平台）跳过，不入库
            if "taian.gov.cn" not in href:
                continue
            if href in seen:
                continue
            seen.add(href)
            all_items.append(it)
            new_count += 1
        print(f"  Page {pg}: {len(items)}条(新{new_count}) 累计{len(all_items)}", flush=True)
        if not items:
            print(f"  [END] page {pg} empty", flush=True)
            break
        if len(all_items) >= total or len(all_items) >= 2200:
            print(f"  [DONE] reached {len(all_items)} / total {total}", flush=True)
            break
        time.sleep(0.4)

    print(f"列表完成: {len(all_items)} 条, 抓详情...", flush=True)
    if limit and len(all_items) > limit:
        all_items = all_items[:limit]

    results = []
    for i, it in enumerate(all_items):
        detail_url = it["href"]
        if not detail_url.startswith("http"):
            detail_url = urljoin(BASE_URL, detail_url)
        d_title, d_date, body = fetch_detail(session, detail_url)
        title = d_title or it["title"]
        pub_date = d_date or it.get("date", "")
        summary = re.sub(r"<[^>]+>", " ", body) if body else title
        summary = re.sub(r"\s+", " ", summary).strip()[:300]
        atts = []
        if body:
            for m in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|zip|rar|ofd|wps))"[^>]*>([^<]*)</a>', body, re.I):
                atts.append({"name": m.group(2).strip(), "url": m.group(1)})
        results.append({
            "site_name": SITE_NAME, "source_url": detail_url, "url": detail_url,
            "title": title, "pub_date": pub_date, "content": body,
            "summary": summary, "category": CATEGORY,
            "attachments": json.dumps(atts, ensure_ascii=False) if atts else "",
        })
        if (i + 1) % 10 == 0:
            print(f"  detail {i+1}/{len(all_items)}", flush=True)
        time.sleep(0.3)

    print(f"  pushing {len(results)} items to searchdb...", flush=True)
    push_to_searchdb(results, batch_label=SITE_NAME)
    empty = sum(1 for r in results if not (r.get("content") or "").strip())
    print(f"  DONE. pushed={len(results)} 空正文={empty}", flush=True)


if __name__ == "__main__":
    main()
