#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""溧阳市人民政府 (www.liyang.gov.cn) 政府信息公开爬虫 — PHPCMS + 创宇盾
用法: python3 crawl_liyang_xxgk.py [--catid 42151,42170] [--pages N] [--out /tmp/liyang_xxgk.jsonl] [--start-page N]
列表: /content/xxgk/index?catid=XXXX&page=N (iframe 内容, 必须 Referer=catid=42149 父页面)
详情: /html/czly/{年}/{栏目}_{MMDD}/{ID}.html, meta ArticleTitle/PubDate/ContentSource + td#czfxcontent
过盾: 完整浏览器头 + Referer (创宇盾)
注意: 创宇盾频率敏感, 限速 1.5-2s + 403 退避重试
"""
import argparse, json, re, sys, time, random, os
from urllib.request import Request, urlopen
from urllib.error import HTTPError, URLError
from urllib.parse import urljoin

BASE = 'https://www.liyang.gov.cn'
PARENT_URL = BASE + '/index.php?m=content&c=index&a=lists&catid=42149'  # 父页面 (Referer 必需)
SITE_NAME = 'liyang_xxgk'
UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36'
HEADERS = {
    'User-Agent': UA,
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9',
}

CATS = {
    43641: '生态环境',
    42151: '规章行政规范性文件',
    40811: '政府信息公开制度',
    42170: '政策解读图解及简明问答',
    42282: '专项规划和区域规划',
    43445: '财政预决算公开',
    42157: '行政事业性收费',
    42201: '原生态环境历史信息',
    42199: '社会救助公益事业',
    42238: '扶贫信息乡村振兴',
    42323: '公共文化服务',
    42186: '应急管理通知公告',
    42269: '基层政务公开目录清单',
    42205: '政府信息公开指南',
}


def fetch(url, referer=PARENT_URL, timeout=25, retries=3):
    """带完整头 + Referer 抓取, 403/超时退避重试"""
    headers = dict(HEADERS)
    headers['Referer'] = referer
    for attempt in range(retries):
        try:
            req = Request(url, headers=headers)
            resp = urlopen(req, timeout=timeout)
            return resp.read().decode('utf-8', errors='ignore')
        except HTTPError as e:
            if e.code == 403:
                wait = 20 + attempt * 20
                print(f'  [403] 退避 {wait}s ({url[-60:]})', flush=True)
                time.sleep(wait)
                continue
            if e.code in (404, 410):
                return ''
            print(f'  [HTTP {e.code}] {url[-60:]}', flush=True)
            time.sleep(3)
        except (URLError, TimeoutError, OSError) as e:
            print(f'  [ERR {type(e).__name__}] {url[-60:]}', flush=True)
            time.sleep(5 + attempt * 5)
    return ''


def clean_html(html, base_url=None):
    """白名单清洗: 保留 <p>/<table>/<a>, unwrap 装饰标签, 清 inline style, 链接绝对化"""
    from bs4 import BeautifulSoup
    base = base_url or BASE
    soup = BeautifulSoup(html, 'html.parser')
    for v in soup.find_all('video'):
        src = v.get('src') or ''
        if not src:
            s = v.find('source')
            if s:
                src = s.get('src', '', timeout=30)
        if not src:
            p = v.find('param', attrs={'name': 'url'})
            if p:
                src = p.get('value', '')
        if src:
            abs_src = src if src.startswith('http') else urljoin(base, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看视频'
            v.replace_with(a)
        else:
            v.decompose()
    for t in soup(['script', 'style', 'iframe', 'object', 'embed']):
        t.decompose()
    for t in soup.find_all(True):
        for attr in ('style', 'class', 'align', 'valign', 'border', 'cellpadding', 'cellspacing',
                     'width', 'height', 'bgcolor', 'face', 'color', 'size', 'lang', 'dir',
                     'setedaria', 'tabindex', 'role'):
            if attr in t.attrs:
                del t[attr]
        for attr in list(t.attrs):
            if attr.startswith('aria-'):
                del t[attr]
    for img in soup.find_all('img'):
        src = img.get('src') or img.get('data-src') or ''
        if src:
            abs_src = src if src.startswith('http') else urljoin(base, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看图片'
            img.replace_with(a)
        else:
            img.decompose()
    for t in soup.find_all(['div', 'span', 'font', 'center', 'b', 'strong', 'em', 'i', 'u', 's', 'label', 'h1', 'h2', 'h3', 'td']):
        if t.name == 'td' and t.find_parent('table'):
            continue  # 表格内 td 保留结构
        t.unwrap()
    for a in soup.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith('javascript'):
            a['href'] = href if href.startswith('http') else urljoin(base, href)
        elif href:
            a.decompose()
    for p in soup.find_all('p'):
        if not p.get_text(strip=True) and not p.find('table') and not p.find('a'):
            p.decompose()
    return str(soup)


def parse_list_html(html):
    """xxgk 列表: div.xxgkList > p > a[href] + span.time"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for p in soup.select('div.xxgkList p'):
        a = p.find('a', href=True)
        if not a:
            continue
        href = a['href']
        if not re.search(r'/html/czly/\d{4}/\w+_\d{4}/\d+\.html', href):
            continue
        title = (a.get('title') or a.get_text(strip=True) or '').strip()
        span = p.find('span', class_='time')
        date = span.get_text(strip=True) if span else ''
        if title:
            items.append({'title': title, 'url': urljoin(BASE, href), 'date': date})
    seen, out = set(), []
    for it in items:
        if it['url'] not in seen:
            seen.add(it['url'])
            out.append(it)
    return out


def parse_detail_html(html, url):
    """详情: meta 字段 + td#czfxcontent + 附件"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    def meta(name):
        m = soup.find('meta', attrs={'name': name})
        return (m.get('content') or '').strip() if m else ''
    title = meta('ArticleTitle') or ''
    pub = meta('PubDate') or ''
    source = meta('ContentSource') or ''
    zoom = soup.find('td', id='czfxcontent') or soup.find('div', id='czfxcontent') \
        or soup.find('div', id='zoomcon') or soup.find('div', class_='gzk-content')
    body = ''
    att_links = []
    if zoom:
        # ⚠️ 必须"先收集、后删除": 畸形HTML里可能嵌套<a>，find_all 会同时
        # 返回外层与内层；在循环内 decompose 外层会让内层 attrs 变 None → 崩溃
        _anchors = list(zoom.find_all('a', href=True))
        for a in _anchors:
            if getattr(a, 'attrs', None) is None:
                continue
            h = (a.get('href') or '').lower()
            if not h:
                continue
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|caj|wps|et|jpg|jpeg|png|gif|mp4)$', h):
                t = a.get_text(strip=True) or '附件'
                fu = h if h.startswith('http') else urljoin(url, h)
                if (t, fu) not in att_links:
                    att_links.append((t, fu))
        for a in _anchors:
            if getattr(a, 'attrs', None) is not None:
                a.decompose()
        body = clean_html(str(zoom), url)
        if att_links:
            att_p = ''.join(f'<p><a href="{fu}">{t}</a></p>' for t, fu in att_links)
            body = (body + '\n' + att_p).strip()
    return {
        'title': title,
        'publish_date': pub[:10] if pub else '',
        'content': body,
        'source': source,
        'attachments': [{'fileName': t, 'fileUrl': fu} for t, fu in att_links],
    }


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--catid', default='43641,42151,40811,42170,42282,43445,42157,42201,42199,42238,42323,42186,42269,42205',
                    help='子栏目 catid 逗号分隔')
    ap.add_argument('--pages', type=int, default=0, help='0=全量')
    ap.add_argument('--out', default='/tmp/liyang_xxgk.jsonl')
    ap.add_argument('--max-detail-fail', type=int, default=15)
    args = ap.parse_args()

    catids = [int(x) for x in args.catid.split(',') if x.strip()]
    stats = {'items': 0, 'detail_ok': 0, 'detail_fail': 0, 'empty': 0, 'skipped': 0}
    out_fp = open(args.out, 'a', encoding='utf-8')
    done_urls = set()
    if os.path.exists(args.out):
        for line in open(args.out, encoding='utf-8'):
            try:
                done_urls.add(json.loads(line)['page_url'])
            except Exception:
                pass

    # 1. 抓各子栏目列表
    all_items = []
    for catid in catids:
        name = CATS.get(catid, str(catid))
        page_no = 1
        while True:
            url = f'{BASE}/content/xxgk/index?catid={catid}&page={page_no}' if page_no > 1 else f'{BASE}/content/xxgk/index?catid={catid}'
            html = fetch(url)
            if not html:
                print(f'[{name}] 页{page_no} 抓取失败, 停止', flush=True)
                break
            if 'xxgkList' not in html:
                print(f'[{name}] 无xxgkList (len={len(html)}), 停止', flush=True)
                break
            items = parse_list_html(html)
            all_items.extend(items)
            # 分页判断: 从 page=N 链接取最大页
            pages = re.findall(r'page=(\d+)', html)
            maxp = max(int(x) for x in pages) if pages else 1
            if page_no < maxp and (args.pages == 0 or page_no < args.pages):
                page_no += 1
                time.sleep(random.uniform(0.8, 1.5))
                continue
            # 无下一页
            print(f'[{name}] {len(items)} 条 (页{page_no})', flush=True)
            break
        print(f'[{name}] 累计 {len(all_items)} 条', flush=True)
    stats['items'] = len(all_items)
    print(f'列表完成: {len(all_items)} 条 (已抓过 {len(done_urls)})', flush=True)

    # 2. 抓详情 (增量)
    new_items = [it for it in all_items if it['url'] not in done_urls]
    print(f'待抓详情: {len(new_items)}', flush=True)
    for idx, it in enumerate(new_items, 1):
        html = fetch(it['url'])
        if not html:
            stats['detail_fail'] += 1
            print(f'详情 {idx}/{len(new_items)}: 抓取失败 {it["url"]}', flush=True)
            if stats['detail_fail'] >= args.max_detail_fail:
                print('失败过多, 中止', flush=True)
                break
            continue
        d = parse_detail_html(html, it['url'])
        if not d['title']:
            d['title'] = it['title']
        if not d['publish_date']:
            d['publish_date'] = it['date']
        content = d['content']
        if len(content) < 10:
            stats['empty'] += 1
        rec = {
            'title': d['title'],
            'publish_date': d['publish_date'],
            'content': content,
            'page_url': it['url'],
            'source_url': it['url'],
            'site_name': SITE_NAME,
            'summary': re.sub(r'<[^>]+>', ' ', content).strip()[:200],
            'date_rank': 0,
            'author': d['source'],
            'content_source': d['source'],
            'attachments': d['attachments'],
        }
        out_fp.write(json.dumps(rec, ensure_ascii=False) + '\n')
        out_fp.flush()
        stats['detail_ok'] += 1
        if idx % 20 == 0 or idx == len(new_items):
            print(f'详情 {idx}/{len(new_items)}: ok={stats["detail_ok"]} fail={stats["detail_fail"]}', flush=True)
        time.sleep(random.uniform(1.5, 2.2))
    out_fp.close()
    print(f'完成: {json.dumps(stats, ensure_ascii=False)}', flush=True)


if __name__ == '__main__':
    main()
