#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""yanshou100.com 水土保持公示网 - 增量抓取 v3 (修复表格/段落/附件格式)

v3 修复 (2026-08-19 用户反馈):
1. 正文保留 <p> 段落 + <table> 表格 HTML (不再剥平)
   - clean_html 用 bs4: 去 script/style/iframe/object, div/span/font/center unwrap,
     a 转绝对 URL, img 转查看链接, 表格结构完整保留
2. 附件转正文内嵌 <p><a href="绝对URL">附件名</a></p> 段落 (多附件独立成段)
   - attachments 字段同时保留 [{fileName,fileUrl}] 数组
3. 导入不走 import_jsonl_v2 的 strip_html 剥平逻辑 → 直接写 gov_raw (保留 HTML)
   - push_to_searchdb 保留 content 原样

用法: python3 crawl_yanshou100_inc.py --pages 5 --type yanshou|qita --out /tmp/x.jsonl [--existing /tmp/ids.txt]
"""
import argparse, asyncio, json, re, sys, random, os

UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36'
BASE = 'https://www.yanshou100.com'

async def api_post(page, url, payload, form=False):
    js = f"""
    async () => {{
        const opts = {{
            method: 'POST',
            headers: {{'X-Requested-With': 'XMLHttpRequest', 'Accept': 'application/json, text/javascript, */*; q=0.01'}}
        }};
        if ({'true' if form else 'false'}) {{
            opts.headers['Content-Type'] = 'application/x-www-form-urlencoded';
            opts.body = new URLSearchParams({json.dumps(payload)}).toString();
        }} else {{
            opts.headers['Content-Type'] = 'application/json';
            opts.body = JSON.stringify({json.dumps(payload)});
        }}
        const r = await fetch('{BASE}' + '{url}', opts);
        const ct = r.headers.get('content-type') || '';
        if (ct.includes('application/json')) {{
            return await r.json();
        }}
        return {{'code': 'notjson', 'msg': (await r.text()).slice(0, 120)}};
    }}
    """
    try:
        return await page.evaluate(js)
    except Exception as e:
        return {'code': 'err', 'msg': str(e)[:100]}


def clean_html(html):
    """白名单清洗: 保留 <p>/<table>/<a>/<img>, unwrap 装饰标签, 链接绝对化"""
    from bs4 import BeautifulSoup
    from urllib.parse import urljoin
    soup = BeautifulSoup(html, 'html.parser')
    for t in soup(['script', 'style', 'iframe', 'object']):
        t.decompose()
    # 图片转查看链接
    for img in soup.find_all('img'):
        src = img.get('src') or img.get('data-src') or ''
        if src:
            abs_src = src if src.startswith('http') else urljoin(BASE, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看图片'
            img.replace_with(a)
        else:
            img.decompose()
    # 装饰标签 unwrap (保留内部内容)
    # 去掉 hidden 属性 (JS 渲染后行可见)
    for t in soup.find_all(attrs={'hidden': True}):
        del t['hidden']
    for t in soup.find_all(['div', 'span', 'font', 'center', 'b', 'strong', 'em', 'i', 'u', 's', 'label']):
        t.unwrap()
    # a 链接绝对化
    for a in soup.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith('javascript'):
            a['href'] = href if href.startswith('http') else urljoin(BASE, href)
        elif href and href.startswith('javascript'):
            # jumpToFile 附件由 attachment 容器单独处理, 这里剥掉无意义 js 链接
            a.decompose()
    # 段落清理: 空 p 删除
    for p in soup.find_all('p'):
        if not p.get_text(strip=True) and not p.find('table') and not p.find('img'):
            p.decompose()
    return str(soup)


async def fetch_detail_page(page, item_id, stats, site_name):
    url = f'{BASE}/item_detail.html?id={item_id}'
    for attempt in range(4):
        try:
            await page.goto(url, wait_until='domcontentloaded', timeout=30000)
        except Exception:
            pass
        await page.wait_for_timeout(2200)
        c = await page.content()
        t = re.search(r'id="itemName1"[^>]*>\s*([^<]+?)\s*<', c)
        if not t or not t.group(1).strip() or t.group(1).strip() in ('&nbsp;', '&nbsp;&nbsp;'):
            t = re.search(r'id="itemName2"[^>]*>\s*([^<]+?)\s*<', c)
        title = re.sub(r'\s+', ' ', t.group(1)).strip() if t else ''
        title = title.replace('&nbsp;', '').strip()
        if title:
            break
        await asyncio.sleep(2.0 * (attempt + 1))
    if not title:
        stats['detail_fail'] += 1
        return None
    dm = re.search(r'id="createDate"[^>]*>\s*时间[：:]\s*([^<]+?)\s*<', c)
    pub = dm.group(1).strip()[:10] if dm else ''
    def field(fid):
        m = re.search(r'id="' + fid + r'"[^>]*>\s*([^<]+?)\s*<', c)
        return re.sub(r'\s+', ' ', m.group(1)).strip() if m else ''
    company = field('companyName')

    # 正文: 从 itemInfo DOM 取完整 innerHTML (含嵌套表格)
    body = ''
    try:
        body = await page.evaluate("""
        (() => {
            const info = document.getElementById('itemInfo');
            if (!info) return '';
            let tbl = info.closest('table');
            if (!tbl) return '';
            const attTr = tbl.querySelector('tr#tr_attachment');
            if (attTr) attTr.remove();
            return tbl.outerHTML;
        })()
        """)
    except Exception:
        pass
    if not body:
        im = re.search(r'id="itemInfo"[^>]*>([\s\S]*?)</div>', c)
        body = im.group(1) if im else ''
    body = clean_html(body) if body else ''

    # 附件: 从 attachment 容器提取 jumpToFile
    atts = []
    for m in re.finditer(r"jumpToFile\('([^']+)','([^']+)'\)", c):
        fa, fn = m.group(1), m.group(2)
        atts.append({'fileName': fn, 'fileUrl': f'{BASE}/item/showFile?fileId={fa}&fileName={fn}'})
    # 附件去重 (按 fileUrl)
    seen = set()
    atts2 = []
    for a in atts:
        if a['fileUrl'] not in seen:
            seen.add(a['fileUrl'])
            atts2.append(a)

    # 附件转正文内嵌 <p><a href> 段落 (用户偏好: 多附件独立成段)
    att_p = ''
    for a in atts2:
        att_p += f'<p><a href="{a["fileUrl"]}">{a["fileName"]}</a></p>'
    if att_p:
        body = (body + '\n' + att_p).strip()

    item = {
        'title': title,
        'url': url,
        'page_url': url,
        'publish_date': pub,
        'content': body,
        'author': company,
        'site_name': site_name,
        'attachments': atts2,
    }
    stats['ok'] += 1
    return item


async def main():
    global args_type
    ap = argparse.ArgumentParser()
    ap.add_argument('--pages', type=int, default=5)
    ap.add_argument('--type', default='yanshou', help='yanshou/jiance/fangan/qita')
    ap.add_argument('--out', default='/tmp/ys100_inc.jsonl')
    ap.add_argument('--existing', default='', help='已存在 itemId 集合文件')
    args = ap.parse_args()
    args_type = args.type
    site_name = 'yanshou100_' + args.type

    existing_ids = set()
    if args.existing and os.path.exists(args.existing):
        with open(args.existing, encoding='utf-8') as f:
            for line in f:
                line = line.strip()
                if line:
                    existing_ids.add(line)
    print(f'已有 itemId: {len(existing_ids)}', flush=True)

    stats = {'ok': 0, 'detail_fail': 0, 'skip_existing': 0}
    out_fp = open(args.out, 'w', encoding='utf-8')
    try:
        from playwright.async_api import async_playwright
        async with async_playwright() as p:
            browser = await p.chromium.launch(headless=True)
            ctx = await browser.new_context(user_agent=UA, ignore_https_errors=True)
            page = await ctx.new_page()
            await page.goto(BASE + '/', wait_until='domcontentloaded', timeout=30000)
            await page.wait_for_timeout(1500)

            all_ids = []
            for pg in range(1, args.pages + 1):
                payload = {"page": pg, "limit": 30, "publicType": args.type,
                           "itemName": "", "start": "", "end": "", "city": "",
                           "area": "", "parType": "", "sunType": ""}
                d = await api_post(page, '/item/search', payload)
                if not d or (str(d.get('code')) not in ('0', '200')):
                    print(f'页{pg}: API失败 {str(d)[:200]}', flush=True)
                    break
                items = d.get('data') or []
                for it in items:
                    if it.get('itemId') not in all_ids:
                        all_ids.append(it['itemId'])
                print(f'页{pg}: {len(items)} 条, 累计 {len(all_ids)}, total={d.get("count")}', flush=True)
                await asyncio.sleep(1)
            print(f'列表共 {len(all_ids)} 个 itemId', flush=True)

            new_ids = [i for i in all_ids if str(i) not in existing_ids]
            stats['skip_existing'] = len(all_ids) - len(new_ids)
            print(f'需抓 {len(new_ids)} (跳过已有 {stats["skip_existing"]})', flush=True)

            for i, iid in enumerate(new_ids):
                item = await fetch_detail_page(page, iid, stats, site_name)
                if item:
                    out_fp.write(json.dumps(item, ensure_ascii=False) + '\n')
                    out_fp.flush()
                await asyncio.sleep(random.uniform(1.0, 1.8))
                if (i + 1) % 10 == 0:
                    print(f'详情 {i+1}/{len(new_ids)}: ok={stats["ok"]} fail={stats["detail_fail"]}', flush=True)
            await browser.close()
    finally:
        out_fp.close()
    print(f"完成: ok={stats['ok']} fail={stats['detail_fail']} skip_existing={stats['skip_existing']}", flush=True)

if __name__ == '__main__':
    asyncio.run(main())
