#!/usr/bin/env python3
"""本地 Mac 抓取 yanshou100 (绕过服务器 IP 限流) → JSONL
用法: python3 crawl_yanshou100_local.py --pages 5 --type yanshou|qita --out /tmp/x.jsonl
"""
import argparse, asyncio, json, re, sys, random, os

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36'
BASE = 'https://www.yanshou100.com'

async def api_post(page, url, payload, form=False):
    js = f"""
    async () => {{
        const opts = {{
            method: 'POST',
            headers: {{'X-Requested-With': 'XMLHttpRequest', 'Accept': 'application/json, text/javascript, */*; q=0.01'}}
        }};
        if ({'true' if form else 'false'}) {{
            opts.headers['Content-Type'] = 'application/x-www-form-urlencoded';
            opts.body = new URLSearchParams({json.dumps(payload)}).toString();
        }} else {{
            opts.headers['Content-Type'] = 'application/json';
            opts.body = JSON.stringify({json.dumps(payload)});
        }}
        const r = await fetch('{BASE}' + '{url}', opts);
        const ct = r.headers.get('content-type') || '';
        if (ct.includes('application/json')) {{
            return await r.json();
        }}
        return {{'code': 'notjson', 'msg': (await r.text()).slice(0, 120)}};
    }}
    """
    try:
        return await page.evaluate(js)
    except Exception as e:
        return {'code': 'err', 'msg': str(e)[:100]}

def clean_html(html):
    """白名单清洗: 保留 <p>/<table>/<a>, unwrap 装饰标签, 链接绝对化"""
    from bs4 import BeautifulSoup
    from urllib.parse import urljoin
    soup = BeautifulSoup(html, 'html.parser')
    for t in soup(['script', 'style', 'iframe', 'object']):
        t.decompose()
    for img in soup.find_all('img'):
        src = img.get('src') or img.get('data-src') or ''
        if src:
            abs_src = src if src.startswith('http') else urljoin(BASE, src)
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看图片'
            img.replace_with(a)
        else:
            img.decompose()
    # 去掉 hidden 属性 (JS 渲染后行可见)
    for t in soup.find_all(attrs={'hidden': True}):
        del t['hidden']
    for t in soup.find_all(['div', 'span', 'font', 'center', 'b', 'strong', 'em', 'i', 'u', 's', 'label']):
        t.unwrap()
    for a in soup.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith('javascript'):
            a['href'] = href if href.startswith('http') else urljoin(BASE, href)
        elif href and href.startswith('javascript'):
            a.decompose()
    for p in soup.find_all('p'):
        if not p.get_text(strip=True) and not p.find('table') and not p.find('img'):
            p.decompose()
    return str(soup)

async def fetch_detail_page(page, item_id, stats, site_name):
    url = f'{BASE}/item_detail.html?id={item_id}'
    for attempt in range(4):
        try:
            await page.goto(url, wait_until='domcontentloaded', timeout=30000)
        except Exception:
            pass
        await page.wait_for_timeout(2200)
        c = await page.content()
        t = re.search(r'id="itemName1"[^>]*>\s*([^<]+?)\s*<', c)
        if not t or not t.group(1).strip() or t.group(1).strip() in ('&nbsp;', '&nbsp;&nbsp;'):
            t = re.search(r'id="itemName2"[^>]*>\s*([^<]+?)\s*<', c)
        title = re.sub(r'\s+', ' ', t.group(1)).strip() if t else ''
        title = title.replace('&nbsp;', '').strip()
        if title:
            break
        await asyncio.sleep(2.0 * (attempt + 1))
    if not title:
        stats['detail_fail'] += 1
        return None
    dm = re.search(r'id="createDate"[^>]*>\s*时间[：:]\s*([^<]+?)\s*<', c)
    pub = dm.group(1).strip()[:10] if dm else ''
    def field(fid):
        m = re.search(r'id="' + fid + r'"[^>]*>\s*([^<]+?)\s*<', c)
        return re.sub(r'\s+', ' ', m.group(1)).strip() if m else ''
    company = field('companyName')
    # 正文: 取 itemInfo 所在完整表格 (含 项目/项目类型/建设单位/编制单位/监测单位/地理位置/说明 字段行)
    # 表格结构: <table><tr><td><label>字段名</label></td><td><div id=itemX></div></td></tr>...</table>
    # 附件行 (tr_attachment) 单独用 jumpToFile 正则处理, 不入表格
    body = ''
    try:
        body = await page.evaluate("""
        (() => {
            const info = document.getElementById('itemInfo');
            if (!info) return '';
            let tbl = info.closest('table');
            if (!tbl) return '';
            const attTr = tbl.querySelector('tr#tr_attachment');
            if (attTr) attTr.remove();
            return tbl.outerHTML;
        })()
        """)
    except Exception:
        pass
    if not body:
        im = re.search(r'id="itemInfo"[^>]*>([\s\S]*?)</div>', c)
        body = im.group(1) if im else ''
    body = clean_html(body) if body else ''
    atts = []
    for m in re.finditer(r"jumpToFile\('([^']+)','([^']+)'\)", c):
        fa, fn = m.group(1), m.group(2)
        atts.append({'fileName': fn, 'fileUrl': f'{BASE}/item/showFile?fileId={fa}&fileName={fn}'})
    seen = set()
    atts2 = []
    for a in atts:
        if a['fileUrl'] not in seen:
            seen.add(a['fileUrl'])
            atts2.append(a)
    att_p = ''
    for a in atts2:
        att_p += f'<p><a href="{a["fileUrl"]}">{a["fileName"]}</a></p>'
    if att_p:
        body = (body + '\n' + att_p).strip()
    item = {
        'title': title,
        'url': url,
        'page_url': url,
        'publish_date': pub,
        'content': body,
        'author': company,
        'site_name': site_name,
        'attachments': atts2,
    }
    stats['ok'] += 1
    return item

async def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--pages', type=int, default=5)
    ap.add_argument('--type', default='yanshou')
    ap.add_argument('--out', default='/tmp/ys100_local.jsonl')
    args = ap.parse_args()
    site_name = 'yanshou100_' + args.type
    stats = {'ok': 0, 'detail_fail': 0}
    out_fp = open(args.out, 'w', encoding='utf-8')
    try:
        from playwright.async_api import async_playwright
        async with async_playwright() as p:
            browser = await p.chromium.launch(headless=True)
            ctx = await browser.new_context(user_agent=UA, ignore_https_errors=True)
            page = await ctx.new_page()
            await page.goto(BASE + '/', wait_until='domcontentloaded', timeout=30000)
            await page.wait_for_timeout(2000)
            all_ids = []
            for pg in range(1, args.pages + 1):
                payload = {"page": pg, "limit": 30, "publicType": args.type,
                           "itemName": "", "start": "", "end": "", "city": "",
                           "area": "", "parType": "", "sunType": ""}
                d = await api_post(page, '/item/search', payload)
                if not d or (str(d.get('code')) not in ('0', '200')):
                    print(f'页{pg}: API失败 {str(d)[:200]}', flush=True)
                    break
                items = d.get('data') or []
                for it in items:
                    if it.get('itemId') not in all_ids:
                        all_ids.append(it['itemId'])
                print(f'页{pg}: {len(items)} 条, 累计 {len(all_ids)}, total={d.get("count")}', flush=True)
                await asyncio.sleep(1)
            print(f'共 {len(all_ids)} 个 itemId', flush=True)
            for i, iid in enumerate(all_ids):
                item = await fetch_detail_page(page, iid, stats, site_name)
                if item:
                    out_fp.write(json.dumps(item, ensure_ascii=False) + '\n')
                    out_fp.flush()
                await asyncio.sleep(random.uniform(1.0, 1.8))
                if (i + 1) % 10 == 0:
                    print(f'详情 {i+1}/{len(all_ids)}: ok={stats["ok"]} fail={stats["detail_fail"]}', flush=True)
            await browser.close()
    finally:
        out_fp.close()
    print(f"完成: ok={stats['ok']} fail={stats['detail_fail']}", flush=True)

if __name__ == '__main__':
    asyncio.run(main())
