#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""溧阳市人民政府 (www.liyang.gov.cn) 原通知公告历史信息爬虫 — PHPCMS + 创宇盾
用法: python3 crawl_liyang_tzgg.py [--pages N] [--out /tmp/liyang_tzgg.jsonl] [--start-page N]
过盾: 完整浏览器头 + Referer (创宇盾对 curl 默认 UA 403, 完整头放行)
列表: /class/LBODQDEQ (15条/页, 共100页1498条), 分页 /class/LBODQDEQ/{N}
详情: /html/czly/{年}/LBODQDEQ_{MMDD}/{ID}.html, meta ArticleTitle/PubDate/ContentSource + td#czfxcontent
注意: 创宇盾频率敏感 (浦江案例 12s 内连续请求拉黑), 限速 1.5-2s + 403 退避重试
"""
import argparse, json, re, sys, time, random, os
from urllib.request import Request, urlopen
from urllib.error import HTTPError, URLError
from urllib.parse import urljoin

BASE = 'https://www.liyang.gov.cn'
LIST_URL = BASE + '/class/LBODQDEQ'
SITE_NAME = 'liyang_tzgg'
UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36'
HEADERS = {
    'User-Agent': UA,
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9',
}


def fetch(url, referer=LIST_URL, timeout=25, retries=3):
    """带完整头 + Referer 抓取, 403/超时退避重试"""
    headers = dict(HEADERS)
    headers['Referer'] = referer
    for attempt in range(retries):
        try:
            req = Request(url, headers=headers)
            resp = urlopen(req, timeout=timeout)
            return resp.read().decode('utf-8', errors='ignore')
        except HTTPError as e:
            if e.code == 403:
                wait = 20 + attempt * 20
                print(f'  [403] 退避 {wait}s ({url[-60:]})', flush=True)
                time.sleep(wait)
                continue
            if e.code in (404, 410):
                return ''
            print(f'  [HTTP {e.code}] {url[-60:]}', flush=True)
            time.sleep(3)
        except (URLError, TimeoutError, OSError) as e:
            print(f'  [ERR {type(e).__name__}] {url[-60:]}', flush=True)
            time.sleep(5 + attempt * 5)
    return ''


def clean_html(html, base_url=None):
    """白名单清洗: 保留 <p>/<table>/<a>, unwrap 装饰标签, 清 inline style, 链接绝对化"""
    from bs4 import BeautifulSoup
    base = base_url or BASE
    soup = BeautifulSoup(html, 'html.parser')
    for v in soup.find_all('video'):
        src = v.get('src') or ''
        if not src:
            s = v.find('source')
            if s:
                src = s.get('src', '', timeout=30)
        if not src:
            p = v.find('param', attrs={'name': 'url'})
            if p:
                src = p.get('value', '')
        if src:
            abs_src = src if src.startswith('http') else urljoin(base, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看视频'
            v.replace_with(a)
        else:
            v.decompose()
    for t in soup(['script', 'style', 'iframe', 'object', 'embed']):
        t.decompose()
    for t in soup.find_all(True):
        for attr in ('style', 'class', 'align', 'valign', 'border', 'cellpadding', 'cellspacing',
                     'width', 'height', 'bgcolor', 'face', 'color', 'size', 'lang', 'dir',
                     'setedaria', 'tabindex', 'role'):
            if attr in t.attrs:
                del t[attr]
        for attr in list(t.attrs):
            if attr.startswith('aria-'):
                del t[attr]
    for img in soup.find_all('img'):
        src = img.get('src') or img.get('data-src') or ''
        if src:
            abs_src = src if src.startswith('http') else urljoin(base, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看图片'
            img.replace_with(a)
        else:
            img.decompose()
    for t in soup.find_all(['div', 'span', 'font', 'center', 'b', 'strong', 'em', 'i', 'u', 's', 'label', 'h1', 'h2', 'h3', 'td']):
        if t.name == 'td' and t.find_parent('table'):
            continue  # 表格内 td 保留结构
        t.unwrap()
    for a in soup.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith('javascript'):
            a['href'] = href if href.startswith('http') else urljoin(base, href)
        elif href:
            a.decompose()
    for p in soup.find_all('p'):
        if not p.get_text(strip=True) and not p.find('table') and not p.find('a'):
            p.decompose()
    return str(soup)


def parse_list_html(html):
    """列表: li > a[title] + 日期文本"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for a in soup.find_all('a', href=True):
        href = a['href']
        if not re.search(r'/html/czly/\d{4}/LBODQDEQ_\d{4}/\d+\.html', href):
            continue
        title = (a.get('title') or a.get_text(strip=True) or '').strip()
        date = ''
        # 日期在 a 后面的文本节点
        nxt = a.next_sibling
        if nxt:
            dm = re.search(r'(\d{4}-\d{2}-\d{2})', str(nxt))
            if dm:
                date = dm.group(1)
        if title:
            items.append({'title': title, 'url': urljoin(BASE, href), 'date': date})
    # 去重
    seen, out = set(), []
    for it in items:
        if it['url'] not in seen:
            seen.add(it['url'])
            out.append(it)
    return out


def parse_detail_html(html, url):
    """详情: meta 字段 + td#czfxcontent + 附件"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    def meta(name):
        m = soup.find('meta', attrs={'name': name}) or soup.find('meta', attrs={'name': name.lower()})
        return (m.get('content') or '').strip() if m else ''
    title = meta('ArticleTitle') or ''
    pub = meta('PubDate') or ''
    source = meta('ContentSource') or ''
    zoom = soup.find('td', id='czfxcontent') or soup.find('div', id='czfxcontent')
    body = ''
    att_links = []
    if zoom:
        for a in zoom.find_all('a', href=True):
            h = a['href'].lower()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|caj|wps|et|jpg|jpeg|png|gif|mp4)$', h):
                t = a.get_text(strip=True) or '附件'
                fu = h if h.startswith('http') else urljoin(url, h)
                att_links.append((t, fu))
                a.decompose()
        body = clean_html(str(zoom), url)
        if att_links:
            att_p = ''.join(f'<p><a href="{fu}">{t}</a></p>' for t, fu in att_links)
            body = (body + '\n' + att_p).strip()
    return {
        'title': title,
        'publish_date': pub[:10] if pub else '',
        'content': body,
        'source': source,
        'attachments': [{'fileName': t, 'fileUrl': fu} for t, fu in att_links],
    }


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--pages', type=int, default=0, help='0=全量100页')
    ap.add_argument('--start-page', type=int, default=1)
    ap.add_argument('--out', default='/tmp/liyang_tzgg.jsonl')
    ap.add_argument('--max-detail-fail', type=int, default=15)
    args = ap.parse_args()

    stats = {'items': 0, 'detail_ok': 0, 'detail_fail': 0, 'empty': 0, 'skipped': 0}
    out_fp = open(args.out, 'a', encoding='utf-8')
    # 已写文件去重 (断点续传)
    done_urls = set()
    if os.path.exists(args.out):
        for line in open(args.out, encoding='utf-8'):
            try:
                done_urls.add(json.loads(line)['page_url'])
            except Exception:
                pass

    # 1. 首列表页探测总页数
    html = fetch(LIST_URL)
    if not html:
        print('首列表页抓取失败', flush=True)
        return
    pages_total = 100
    m = re.search(r'Fx_PageDiv2_1_3\">(\d+)</span>', html)
    if m:
        pages_total = int(m.group(1))
    if args.pages > 0:
        pages_total = min(pages_total, args.pages)
    print(f'总页数={pages_total}', flush=True)

    # 2. 抓列表
    all_items = []
    for pg in range(args.start_page, pages_total + 1):
        url = LIST_URL if pg == 1 else f'{LIST_URL}/{pg}'
        h = fetch(url)
        if not h:
            print(f'页{pg} 抓取失败, 停止翻页', flush=True)
            break
        items = parse_list_html(h)
        all_items.extend(items)
        print(f'页{pg}: {len(items)} 条, 累计 {len(all_items)}', flush=True)
        time.sleep(random.uniform(0.8, 1.5))
    stats['items'] = len(all_items)
    print(f'列表完成: {len(all_items)} 条 (已抓过 {len(done_urls)})', flush=True)

    # 3. 抓详情 (增量跳过)
    new_items = [it for it in all_items if it['url'] not in done_urls]
    print(f'待抓详情: {len(new_items)}', flush=True)
    for idx, it in enumerate(new_items, 1):
        html = fetch(it['url'])
        if not html:
            stats['detail_fail'] += 1
            print(f'详情 {idx}/{len(new_items)}: 抓取失败 {it["url"]}', flush=True)
            if stats['detail_fail'] >= args.max_detail_fail:
                print('失败过多, 中止', flush=True)
                break
            continue
        d = parse_detail_html(html, it['url'])
        if not d['title']:
            d['title'] = it['title']
        if not d['publish_date']:
            d['publish_date'] = it['date']
        content = d['content']
        if len(content) < 10:
            stats['empty'] += 1
        rec = {
            'title': d['title'],
            'publish_date': d['publish_date'],
            'content': content,
            'page_url': it['url'],
            'source_url': it['url'],
            'site_name': SITE_NAME,
            'summary': re.sub(r'<[^>]+>', ' ', content).strip()[:200],
            'date_rank': 0,
            'author': d['source'],
            'content_source': d['source'],
            'attachments': d['attachments'],
        }
        out_fp.write(json.dumps(rec, ensure_ascii=False) + '\n')
        out_fp.flush()
        stats['detail_ok'] += 1
        if idx % 20 == 0 or idx == len(new_items):
            print(f'详情 {idx}/{len(new_items)}: ok={stats["detail_ok"]} fail={stats["detail_fail"]}', flush=True)
        time.sleep(random.uniform(1.5, 2.2))
    out_fp.close()
    print(f'完成: {json.dumps(stats, ensure_ascii=False)}', flush=True)


if __name__ == '__main__':
    main()
