#!/usr/bin/env python3
"""crawl_pj_hpgs.py - 浦江县政务公开 (浙江 JPAAS 集约化)

站点: https://www.pj.gov.cn (无 WAF 域名, 主站 pujiang.gov.cn 被创宇盾保护)
CMS: 大汉 jcms1 (浙江政务网集约化 JPAAS), webId=3609
列表: unitbuild API GET /api-gateway/jpaas-publish-server/front/page/build/unit
      tagId=组配分类list + paramJson={pageNo, pageSize:15, search:{xxgkId,xxgkType,className}}
      条目 <a class="fl" href title>标题</a> + <span class="fr">日期</span>
分页: layui.laypage, pageSize=15
详情: div.content > div.art_tit(标题) + div.wenz(正文), meta[PubDate], 附件 /api-gateway/jpaas-web-server/front/document/download?fileUrl=...
栏目:
  --col hpgs    环评公示:  xxgkId=PJ001 className=建设项目环境影响评价信息公示 (18条)
  --col ecology 环境保护:  xxgkId=T001-27 xxgkType=xxgk_combination className=生态环境 (121条, 环境质量/公报/决策)
"""
import sys, os, re, json, argparse, time
from datetime import datetime, timedelta
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb, normalize_pub_date

import requests
import warnings
warnings.filterwarnings("ignore")

BASE = 'https://www.pj.gov.cn'
GROUP = '浙江'

# 双栏目配置
COL_MAP = {
    'hpgs': {
        'site_name': '浦江县-建设项目环境影响评价信息公示',
        'page_id': '1229856886',
        'xxgk_id': 'PJ001',
        'xxgk_type': '',
        'class_name': '建设项目环境影响评价信息公示',
        'cutoff_days': 365 * 3,
        'referer': BASE + '/col/col1229856886/index.html?isshow=fdgk',
    },
    'ecology': {
        'site_name': '浦江县-环境保护',
        'page_id': '1229196496',
        'xxgk_id': 'T001-27',
        'xxgk_type': 'xxgk_combination',
        'class_name': '生态环境',
        'cutoff_days': 365 * 10,
        'referer': BASE + '/col/col1229196496/index.html?number=B001-03',
    },
}

CUTOFF = (datetime.now() - timedelta(days=365*3)).strftime('%Y-%m-%d')
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Referer': BASE + '/col/col1229856886/index.html?isshow=fdgk',
}

API = BASE + '/api-gateway/jpaas-publish-server/front/page/build/unit'
PAGE_SIZE = 15
CLASSNAME = '建设项目环境影响评价信息公示'

session = requests.Session()
session.headers.update(HEADERS)


def fetch_list(page_no, col_cfg, max_retries=3):
    """调用 unitbuild API 获取列表页"""
    search = json.dumps({"xxgkId": col_cfg['xxgk_id'], "xxgkType": col_cfg['xxgk_type'], "className": col_cfg['class_name']}, ensure_ascii=False)
    params = {
        'parseType': 'bulidstatic', 'webId': '3609', 'tplSetId': 'yt8rdEAl5gTL95ioBNtPN',
        'pageType': 'column', 'tagId': '组配分类list', 'editType': 'null', 'pageId': col_cfg['page_id'],
        'paramJson': json.dumps({"pageNo": page_no, "pageSize": PAGE_SIZE, "search": search}, ensure_ascii=False),
    }
    for attempt in range(max_retries):
        try:
            r = session.get(API, params=params, timeout=20, verify=False)
            d = r.json()
            html = d.get('data', {}).get('html', '')
            # 解析条目: <a class="fl" href title>标题</a> ... <span class="fr">日期</span>
            items = []
            for m in re.finditer(
                r'<a class="fl" href="([^"]+)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>\s*<span class="fr">([^<]*)</span>',
                html, re.S):
                href, title, body, date = m.groups()
                t = re.sub(r'<[^>]+>', '', body).strip()
                if not t:
                    t = title.strip()
                t = re.sub(r'^[•·]\s*', '', t).strip()
                url = href if href.startswith('http') else urljoin(BASE, href)
                items.append({'url': url, 'title': t, 'pub_date': date.strip()})
            return items
        except Exception as e:
            if attempt < max_retries - 1:
                time.sleep(2 * (attempt + 1))
            else:
                print(f'    [WARN] list page {page_no} error: {e}')
                return []
    return []


def extract_balanced_div(html, open_tag_re):
    """平衡 div 匹配"""
    m = re.search(open_tag_re, html)
    if not m:
        return ''
    gt = html.find('>', m.start())
    if gt == -1:
        return ''
    i = gt + 1
    depth = 1
    for mm in re.finditer(r'<div[\s>]|</div>', html[i:]):
        if mm.group(0).startswith('<div'):
            depth += 1
        else:
            depth -= 1
            if depth == 0:
                return html[i:i + mm.start()]
    return html[i:]


def fetch_detail(url):
    """抓详情页: 标题/日期/正文/附件"""
    for _attempt in range(3):
        try:
            r = session.get(url, timeout=20, verify=False)
            r.encoding = r.apparent_encoding or 'utf-8'
            html = r.text
            break
        except Exception:
            html = ''
            time.sleep(1)
    if not html:
        return '', '', '', ''

    # 标题: meta ArticleTitle 优先, 回退 div.art_tit, 回退 <title>
    title = ''
    m = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]*)"', html, re.I)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r'<div class="art_tit">\s*([^<]+?)\s*</div>', html, re.S)
        if m:
            title = m.group(1).strip()
    if not title:
        m = re.search(r'<title>\s*([^<]+?)\s*</title>', html, re.DOTALL)
        if m:
            title = re.sub(r'\s+', '', m.group(1)).strip()
            title = re.split(r'[-–—|_]', title)[0].strip()
    title = re.sub(r'[\u200b\ufeff]', '', title).strip()
    title = re.sub(r'^[•·]\s*', '', title).strip()

    # 日期: meta[PubDate] 优先, 回退 meta[pubDate]
    pub_date = ''
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="([^"]*)"', html, re.I)
    if m:
        pub_date = m.group(1).strip()[:10]
    if not pub_date:
        m = re.search(r'<meta[^>]*name="pubDate"[^>]*content="([^"]*)"', html, re.I)
        if m:
            pub_date = m.group(1).strip()[:10]

    # 正文: div.wenz (在 div.content 内)
    content = extract_balanced_div(html, r'<div[^>]*class="wenz"[^>]*>')
    if not content.strip():
        content = extract_balanced_div(html, r'<div[^>]*class="content"[^>]*>')

    if content.strip():
        # 清理: 移除 script/style
        content = re.sub(r'<script[\s\S]*?</script>', '', content)
        content = re.sub(r'<style[\s\S]*?</style>', '', content)
        # 清理模板注释 (pdf预览等)
        content = re.sub(r'<!--\s*pdf文件预览\s*-->', '', content)
        content = re.sub(r'<!--[\s\S]*?-->', '', content)
        # 相对链接/图片绝对化
        content = re.sub(r'(href|src)="(?!/?(?:https?:|//|javascript:|#|data:))([^"]*)"',
                         lambda mm: f'{mm.group(1)}="{urljoin(BASE, mm.group(2))}"', content)
        # 附件绝对化 (download 接口相对路径)
        content = re.sub(r'href="(/api-gateway/jpaas-web-server/front/document/download[^"]*)"',
                         lambda mm: f'href="{BASE}{mm.group(1)}"', content)
        content = content.strip()

    # 附件收集 (正文内 <a> 含 download 或文件扩展名)
    attachments = []
    for m in re.finditer(r'<a[^>]+href="([^"]+)"[^>]*>([^<]*)</a>', content or ''):
        href, name = m.group(1), m.group(2).strip()
        if any(k in href for k in ['download', '.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar', '.wps']):
            attachments.append({'name': name, 'url': href})
    return title, pub_date, content, json.dumps(attachments, ensure_ascii=False)


def run(max_pages=5, col='hpgs'):
    col_cfg = COL_MAP.get(col)
    if not col_cfg:
        print(f'  Unknown col: {col}')
        return 0
    site_name = col_cfg['site_name']
    cutoff = (datetime.now() - timedelta(days=col_cfg['cutoff_days'])).strftime('%Y-%m-%d')
    print(f"  Site: {site_name} (cutoff >= {cutoff})")
    records = []
    total_new = 0
    total_skip = 0
    cutoff_hit = False
    for page in range(1, max_pages + 1):
        items = fetch_list(page, col_cfg)
        if not items:
            break
        print(f'  Page {page}: {len(items)} items')
        for it in items:
            if it['pub_date'] and it['pub_date'] < cutoff:
                cutoff_hit = True
                continue
            title, pd, content, att = fetch_detail(it['url'])
            if not title:
                title = it['title']
            if not content.strip():
                print(f'    [EMPTY] {title[:50]}')
                continue
            records.append({
                'title': title, 'url': it['url'], 'source_url': it['url'],
                'pub_date': pd or it['pub_date'],
                'site_name': site_name, 'content': content,
                'summary': '', 'attachments': att, 'group_name': GROUP,
            })
        if cutoff_hit:
            print('  CUTOFF reached, stop')
            break
    print(f'  Total records: {len(records)}')
    if records:
        push_to_searchdb(records, f"pj_{col}")
    return len(records)


if __name__ == '__main__':
    ap = argparse.ArgumentParser()
    ap.add_argument('--pages', type=int, default=5)
    ap.add_argument('--col', type=str, default='hpgs', choices=list(COL_MAP.keys()))
    ap.add_argument('--sync-only', action='store_true')
    args = ap.parse_args()
    if args.sync_only:
        sys.exit(0)
    cnt = run(max_pages=args.pages, col=args.col)
    print(f'Done: {cnt} records')
