#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
浙江 JPAAS 站群通用爬虫 (同扬州kfq模板)
用法: python3 crawl_zjjpaas.py --site=nanhu [--pages=N]
站点表:
  nanhu  = 嘉兴南湖区-建设项目环评审批  (www.nanhu.gov.cn, webId=3164)
  jxgq   = 嘉兴港区-公告公示           (jxgq.jiaxing.gov.cn, webId=3177)
  jiande = 建德市-行政许可             (www.jiande.gov.cn, webId=2210)
  linan  = 临安区-建设项目环评信息公示  (www.linan.gov.cn, webId=2242)
API: /api-gateway/jpaas-publish-server/front/page/build/unit
列表: 返回 html 片段 <li><a href="//host/art/Y/M/D/art_ID.html">标题</a><span>YYYY-MM-DD</span></li>
详情: meta ArticleTitle/PubDate + id=zoom 正文(平衡div)
"""
import requests, re, sqlite3, time, sys, json, os, warnings
warnings.filterwarnings('ignore')
from datetime import datetime, timedelta
from urllib.parse import urljoin

SITES = {
    "nanhu": {
        "site_name": "嘉兴南湖区-建设项目环评审批",
        "domain": "www.nanhu.gov.cn",
        "webId": "3164",
        "tplSetId": "zsRKKTw8HJcEkqdXqPoUm",
        "pageId": "1229658638",
        "tagId": "新闻列表",
    },
    "jxgq": {
        "site_name": "嘉兴港区-公告公示",
        "domain": "jxgq.jiaxing.gov.cn",
        "webId": "3177",
        "tplSetId": "juDpibDTuQ7ELC7U6h4Sb",
        "pageId": "1229398092",
        "tagId": "第一部分_list",
        "content_pat": "class=\"box_wzy_ys\"",  # ⚠️ 本页 id="zoom" 是标题容器, 正文在 box_wzy_ys
    },
    "jiande": {
        "site_name": "建德市-行政许可",
        "domain": "www.jiande.gov.cn",
        "webId": "2210",
        "tplSetId": "ELWAAQQXOD87oUzmcMS2i",
        "pageId": "1229416058",
        "tagId": "标题列表",
    },
    "linan": {
        "site_name": "临安区-建设项目环评信息公示",
        "domain": "www.linan.gov.cn",
        "webId": "2242",
        "tplSetId": "VuAIYKXLw7THdhTjKO2yP",
        "pageId": "1229882622",
        "tagId": "信息列表1",
    },
    "sthjj": {
        "site_name": "嘉兴市生态环境局-建设项目环评信息公示",
        "domain": "sthjj.jiaxing.gov.cn",
        "webId": "3188",
        "tplSetId": "BGV1JgNlVJkbjXcPT4vlP",
        "pageId": "1229856885",
        "tagId": "列表",
    },
}

SITE = None
_MAX_PG = None
for _a in sys.argv[1:]:
    if _a.startswith("--site="):
        SITE = _a.split("=", 1)[1]
    elif _a.startswith("--pages="):
        try:
            _MAX_PG = int(_a.split("=", 1)[1])
        except Exception:
            pass
    elif _a.isdigit():
        _MAX_PG = int(_a)

if not SITE or SITE not in SITES:
    print("用法: python3 crawl_zjjpaas.py --site=nanhu|jxgq|jiande|linan [--pages=N]")
    sys.exit(1)

CFG = SITES[SITE]
SITE_NAME = CFG["site_name"]
DOMAIN = CFG["domain"]
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36"}

API_URL = f"https://{DOMAIN}/api-gateway/jpaas-publish-server/front/page/build/unit"
API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": CFG["webId"],
    "tplSetId": CFG["tplSetId"],
    "pageType": "column",
    "tagId": CFG["tagId"],
    "editType": "null",
    "pageId": CFG["pageId"],
}


def fetch_list(page, page_size=15):
    params = dict(API_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page, "pageSize": page_size})
    for retry in range(3):
        try:
            r = requests.get(API_URL, params=params, headers=HEADERS, timeout=20, verify=False)
            if r.status_code == 200:
                d = r.json()
                html = d.get("data", {}).get("html", "")
                items = []
                # 格式1: <a href="..." title="..."><li><span>●</span>标题<time>日期</time></li></a>
                for m in re.finditer(
                        r'<a[^>]+href="([^"]+)"[^>]*title="([^"]*)"[^>]*>\s*<li[^>]*>(.*?)</li>',
                        html, re.DOTALL):
                    href, title, lic = m.group(1), m.group(2).strip(), m.group(3)
                    dm = re.search(r'(?:<span[^>]*>|<time[^>]*>)(\d{4}-\d{2}-\d{2})', lic)
                    if not dm:
                        continue
                    if href.startswith("//"):
                        href = "https:" + href
                    elif not href.startswith("http"):
                        href = urljoin(f"https://{DOMAIN}/", href)
                    items.append({"url": href, "title": title, "date": dm.group(1)})
                if not items:
                    # 格式2: <li><a href>标题</a><span>YYYY-MM-DD</span></li>
                    for li in re.findall(r'<li[^>]*>(.*?)</li>', html, re.DOTALL):
                        m = re.search(
                            r'href="([^"]+)"[^>]*>\s*(?:<[^>]+>)*([^<]+?)\s*<', li, re.DOTALL)
                        if not m:
                            continue
                        href, title = m.group(1), m.group(2).strip()
                        dm = re.search(r'<span[^>]*>(\d{4}-\d{2}-\d{2})</span>', li)
                        if not dm:
                            continue
                        if href.startswith("//"):
                            href = "https:" + href
                        elif not href.startswith("http"):
                            href = urljoin(f"https://{DOMAIN}/", href)
                        items.append({"url": href, "title": title, "date": dm.group(1)})
                if not items:
                    # 格式3: <li class="public-li"><a href><div class="public-text">标题</div><div class="public-time">日期</div></a></li>
                    for m in re.finditer(
                            r'<li[^>]*class="[^"]*public-li[^"]*"[^>]*>\s*<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>',
                            html, re.DOTALL):
                        href, lic = m.group(1), m.group(2)
                        tm = re.search(r'class="public-text[^"]*"[^>]*>(.*?)</div>', lic, re.DOTALL)
                        dm = re.search(r'class="public-time[^"]*"[^>]*>(\d{4}-\d{2}-\d{2})', lic)
                        if not tm or not dm:
                            continue
                        title = re.sub(r'<[^>]+>', '', tm.group(1)).strip()
                        if href.startswith("//"):
                            href = "https:" + href
                        elif not href.startswith("http"):
                            href = urljoin(f"https://{DOMAIN}/", href)
                        items.append({"url": href, "title": title, "date": dm.group(1)})
                if not items:
                    # 格式4: <li class="cf border-line"><a class="fl" href><i></i>标题</a><span class="fr">日期</span></li>
                    for li in re.findall(r'<li[^>]*>(.*?)</li>', html, re.DOTALL):
                        m = re.search(
                            r'<a[^>]+href="([^"]+)"[^>]*>\s*<i[^>]*>\s*</i>\s*([^<]+?)\s*</a>\s*<span[^>]*>(\d{4}-\d{2}-\d{2})',
                            li, re.DOTALL)
                        if not m:
                            continue
                        href, title, date = m.group(1), m.group(2).strip(), m.group(3)
                        if href.startswith("//"):
                            href = "https:" + href
                        elif not href.startswith("http"):
                            href = urljoin(f"https://{DOMAIN}/", href)
                        items.append({"url": href, "title": title, "date": date})
                count_m = re.search(r'count="(\d+)"', html)
                total_count = int(count_m.group(1)) if count_m else 0
                return items, total_count
        except Exception as e:
            print(f"  [WARN] API page {page} err: {e}")
            if retry < 2:
                time.sleep(2)
    return [], 0


def extract_content(html):
    """提取正文 - 站点 content_pat 优先, 否则通用 fallback (平衡div)"""
    pats = []
    if CFG.get("content_pat"):
        pats.append((CFG["content_pat"], "div"))
    pats += [('id="zoom"', 'div'), ('class="content_new"', 'div'),
             ('class="contentShow"', 'div'), ('class="TRS_Editor"', 'div'),
             ('class="article"', 'div'), ('class="box_wzy_ys"', 'div')]
    for pat, tag in pats:
        i = html.find(pat)
        if i < 0:
            continue
        ds = html.rfind(f"<{tag}", 0, i)
        if ds < 0:
            continue
        s = html[ds:]
        d = 0
        for j in range(len(s)):
            if s[j:j+4] == f"<{tag}" and (j+4 >= len(s) or s[j+4] in " >\n\r\t"):
                d += 1
            elif s[j:j+3+len(tag)] == f"</{tag}>":
                d -= 1
                if d == 0:
                    gt = s.find(">", 0, j)
                    c = s[gt+1:j] if gt > 0 else s[7:j]
                    # ⚠️ JS 动态 PDF 公告: var pdfUrl/pdfurl='...pdf' + iframe 渲染 → 转附件段落
                    pm = re.search(r"var pdf[Uu]rl\s*=\s*['\"]([^'\"]+\.(?:pdf|docx?|xlsx?|wps))['\"]", c, re.I)
                    link_m = re.search(r'<a[^>]*href=["\']([^"\']+\.(?:pdf|docx?|xlsx?|wps))["\']', c, re.I)
                    # 先删 script/style 再判断文本(script 内 JS 字符串含 <p 会干扰判断)
                    c2 = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', c, flags=re.DOTALL|re.I)
                    c2 = re.sub(r'<p[^>]*>', '<p>', c2)
                    text = re.sub(r'<[^>]+>', '', c2).strip()
                    # 有效内容: 文本>10 或 带文字的链接(空<a href=...pdf></a>不算) 或 图片
                    has_named_link = bool(re.search(r'<a[^>]*>\s*[^<\s]', c2, re.DOTALL))
                    if len(text) > 10 or has_named_link or '<img' in c2:
                        return c2.strip()
                    # 正文无实质文本但检测到 PDF → 转附件段落
                    h = None
                    if pm:
                        h = pm.group(1)
                    elif link_m:
                        h = link_m.group(1)
                    if h:
                        if h.startswith("/"):
                            h = f"https://{DOMAIN}" + h
                        return f'<p><a href="{h}">PDF原文</a></p>'
                    return ""
    return ""


def extract_meta(html):
    title = ""
    tm = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]*)"', html)
    if tm:
        title = tm.group(1)
    if not title:
        ttm = re.search(r'<title>(.*?)<', html)
        if ttm:
            title = ttm.group(1).strip()
    date = ""
    for pm in re.finditer(r'PubDate[^>]*content="([^"]*)"', html):
        d = pm.group(1).strip()
        cm = re.match(r'(\d{4})-(\d{1,2})-(\d{1,2})', d)
        if cm:
            date = f"{cm.group(1)}-{cm.group(2).zfill(2)}-{cm.group(3).zfill(2)}"
            break
    if not date:
        dm = re.search(r'(\d{4}-\d{2}-\d{2})', html)
        if dm:
            date = dm.group(1)
    return title, date


def fetch_url(url):
    for retry in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
            if r.status_code == 200:
                r.encoding = "utf-8"
                return r.text
        except Exception as e:
            if retry < 2:
                time.sleep(2)
    return None


def run(max_pages=200):
    print(f"\n{'='*50}")
    print(f"🚀 {SITE_NAME} ({DOMAIN})")
    print(f"{'='*50}")

    items, total = fetch_list(1)
    if not items:
        print("⚠  API无返回")
        return
    print(f"  共{total}条, 15条/页 = {(total+14)//15}页")

    total_pages = min(max_pages, (total + 14) // 15)
    all_list_items = list(items)
    for pg in range(2, min(total_pages, _MAX_PG or total_pages) + 1):
        its, _ = fetch_list(pg)
        if not its:
            print(f"  第{pg}页: 空 -> 结束")
            break
        all_list_items.extend(its)
        dates = [it["date"] for it in its if it["date"]]
        print(f"  第{pg}页: {len(its)} 条 ({dates[-1] if dates else '?'} ~ {dates[0] if dates else '?'})")
        if dates and max(dates) < CUTOFF:
            print(f"  该页已全部早于截断日{CUTOFF}，停止翻页")
            break
        time.sleep(0.15)

    print(f"\n📊 API共 {len(all_list_items)} 条")
    out = []
    err = 0
    for i, item in enumerate(all_list_items, 1):
        if item["date"] and item["date"] < CUTOFF:
            continue
        html = fetch_url(item["url"])
        if not html:
            err += 1
            continue
        pt, pd = extract_meta(html)
        real_title = pt if pt else item["title"]
        real_date = pd if pd else item["date"]
        if real_date and real_date < CUTOFF:
            continue
        content = extract_content(html)
        text_len = len(re.sub(r'<[^>]+>', '', content).strip()) if content else 0
        # 有效: 文本≥10 或 含图片 或 含链接(PDF附件段落 text_len 可能仅5)
        if text_len < 10 and '<img' not in (content or '') and '<a href=' not in (content or ''):
            print(f"  ⚠ 空正文: {real_title[:40]} | {item["url"]}")
            err += 1
            continue
        out.append({
            "site_name": SITE_NAME,
            "title": real_title,
            "pub_date": real_date,
            "content": content,
            "source_url": item["url"],
            "url": item["url"],
        })
        if i % 10 == 0:
            print(f"  ...{i}/{len(all_list_items)}")
        time.sleep(0.2)

    print(f"\n📦 待入库: {len(out)}, 失败: {err}")
    sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
    from crawler_lib import push_to_searchdb
    push_to_searchdb(out, batch_label=SITE_NAME)


if __name__ == "__main__":
    run(max_pages=200)
