#!/usr/bin/env python3
"""crawl_sxjjb_rdjj.py — 山西经济日报 热点聚焦栏目 (sxjjb.cn/zz/rdjj/)
云锁 YunSuo WAF: playwright headless 执行 JS 自动过盾 (security_session_verify cookie)
列表: /zz/rdjj/index.htm (p1) + /newslist.aspx?page=N&mid=165 (p2+), 86页
容器: ul.wenhua_sub_list > li > a[title]+span{2026/7/31}
详情: /zz/rdjj/news{6位}.htm; 标题 h2.tfsubTitle; 正文 div.tfcontent
"""
import json, os, re, sys, time, random
from datetime import datetime
from bs4 import BeautifulSoup
from playwright.sync_api import sync_playwright

BASE = "http://www.sxjjb.cn"
LIST1 = BASE + "/zz/rdjj/index.htm"
MID = "165"
IP = "221.204.237.122"
OUT = "/tmp/sxjjb_rdjj.jsonl"
MAX_PAGES = int(sys.argv[1]) if len(sys.argv) > 1 else 90
SLEEP_MIN, SLEEP_MAX = 0.6, 1.4
MAX_FAIL = 10

KEEP = {'p', 'table', 'tr', 'td', 'th', 'thead', 'tbody', 'a', 'img', 'br', 'h1', 'h2', 'h3', 'h4', 'ul', 'ol', 'li', 'strong', 'em', 'b'}
UNWRAP = {'span', 'div', 'font', 'label', 'section', 'article'}
DROP = {'script', 'style', 'iframe', 'form', 'input', 'button', 'nav', 'header', 'footer'}

def clean_html(html):
    if not html:
        return ""
    soup = BeautifulSoup(html, 'html.parser')
    for tag in soup.find_all(True):
        if tag.name in DROP:
            tag.decompose()
            continue
        if tag.name == 'img':
            src = tag.get('src') or tag.get('data-src') or ''
            if src:
                if src.startswith('/'):
                    src = BASE + src
                a = soup.new_tag('a')
                a['href'] = src
                a.string = '查看图片'
                tag.replace_with(a)
            else:
                tag.decompose()
            continue
        if tag.name == 'a':
            href = tag.get('href') or ''
            if href.startswith('/'):
                tag['href'] = BASE + href
            elif href and not href.startswith('http'):
                tag['href'] = BASE + '/' + href.lstrip('./')
        for attr in list(tag.attrs):
            if attr.startswith(('on', 'style', 'class', 'id', 'aria-', 'data-', 'target', 'rel', 'align', 'width', 'height', 'border', 'cellpadding', 'cellspacing', 'valign', 'scope', 'rowspan', 'colspan')):
                del tag[attr]
    for tag in soup.find_all(UNWRAP):
        tag.unwrap()
    for tag in soup.find_all(['td', 'th']):
        if not tag.find_parent('table'):
            tag.unwrap()
    out = str(soup)
    out = re.sub(r'\n{3,}', '\n\n', out)
    out = re.sub(r'[ \t]{2,}', ' ', out)
    return out.strip()

def parse_list(html):
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    ul = soup.find('ul', class_='wenhua_sub_list')
    if not ul:
        return items, 0
    for li in ul.find_all('li'):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '')
        # 只保留站内详情: /zz/rdjj/news{6位}.htm
        m = re.search(r'/zz/rdjj/news(\d+)\.htm$', href)
        if not m:
            continue
        title = a.get('title') or a.get_text(strip=True)
        span = li.find('span')
        date_raw = span.get_text(strip=True) if span else ''
        dm = re.search(r'(\d{4})/(\d{1,2})/(\d{1,2})', date_raw)
        date = f"{dm.group(1)}-{int(dm.group(2)):02d}-{int(dm.group(3)):02d}" if dm else ''
        items.append({
            'id': m.group(1),
            'title': title.strip(),
            'date': date,
            'url': BASE + href,
        })
    # 总页数
    total_pages = 0
    for m in re.finditer(r'/newslist\.aspx\?page=(\d+)&mid=\d+', html):
        total_pages = max(total_pages, int(m.group(1)))
    return items, total_pages

def parse_detail(html):
    soup = BeautifulSoup(html, 'html.parser')
    h2 = soup.find('h2', class_='tfsubTitle')
    title = h2.get_text(strip=True) if h2 else ''
    box = soup.find('div', class_='tfcontent')
    body = ''
    if box:
        body = clean_html(str(box))
    # 来源/发布时间
    source, author = '', ''
    pm = re.search(r'来源[:：]\s*([^<\s][^<\n]{0,50})', html)
    if pm:
        source = pm.group(1).strip()
    am = re.search(r'作者[:：]\s*([^<\s][^<\n]{0,30})', html)
    if am:
        author = am.group(1).strip()
    dm = re.search(r'发布时间[:：]\s*(\d{4})/(\d{1,2})/(\d{1,2})', html)
    date = ''
    if dm:
        date = f"{dm.group(1)}-{int(dm.group(2)):02d}-{int(dm.group(3)):02d}"
    # 附件: 正文内链接已绝对化, 收集 doc/pdf/xls 类
    attachments = []
    if box:
        for a in box.find_all('a'):
            href = a.get('href', '')
            if re.search(r'\.(docx?|pdf|xlsx?|zip|rar|wps)$', href, re.I):
                attachments.append({'name': a.get_text(strip=True), 'url': href})
    return {'title': title, 'body': body, 'date': date, 'source': source, 'author': author, 'attachments': attachments}

def main():
    done_ids = set()
    if os.path.exists(OUT):
        for line in open(OUT, encoding='utf-8'):
            try:
                done_ids.add(json.loads(line)['id'])
            except Exception:
                pass
    print(f"[*] 已有 {len(done_ids)} 条, 断点续传", flush=True)

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=[
            "--disable-blink-features=AutomationControlled", "--no-sandbox",
            f"--host-resolver-rules=MAP www.sxjjb.cn {IP}, MAP sxjjb.cn {IP}",
        ])
        ctx = browser.new_context(
            user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
            locale="zh-CN", viewport={"width": 1920, "height": 1080},
        )
        page = ctx.new_page()
        # 过云锁 (首页触发 JS)
        try:
            page.goto(LIST1, timeout=30000, wait_until="domcontentloaded")
        except Exception as e:
            print(f"[!] goto exc: {e}", flush=True)
        passed = False
        for i in range(10):
            time.sleep(2)
            try:
                html = page.content()
                if 'wenhua_sub_list' in html:
                    passed = True
                    break
            except Exception:
                pass
        if not passed:
            print("[X] 云锁未过", flush=True)
            browser.close()
            sys.exit(1)
        print("[✓] 云锁已过", flush=True)

        # 抓列表
        all_items = []
        fail = 0
        for pno in range(1, MAX_PAGES + 1):
            if pno == 1:
                url = LIST1
            else:
                url = f"{BASE}/newslist.aspx?page={pno}&mid={MID}"
            try:
                page.goto(url, timeout=30000, wait_until="domcontentloaded")
                time.sleep(1.0)
                html = page.content()
            except Exception as e:
                print(f"[!] 列表页 {pno} 失败: {e}", flush=True)
                fail += 1
                if fail >= MAX_FAIL:
                    break
                continue
            items, total_pages = parse_list(html)
            has_container = 'wenhua_sub_list' in html
            if not has_container:
                fail += 1
                if fail >= MAX_FAIL:
                    print(f"[!] 连续 {MAX_FAIL} 页无列表容器, 中止", flush=True)
                    break
                continue
            fail = 0
            if not items:
                print(f"[list] p{pno}/{total_pages or '?'}: 0 站内 (全外链), 累计 {len(all_items)}", flush=True)
            else:
                seen_ids = {it['id'] for it in all_items}
                new_items = [it for it in items if it['id'] not in seen_ids]
                all_items.extend(new_items)
                print(f"[list] p{pno}/{total_pages or '?'}: {len(items)} 站内(新{len(new_items)}), 累计 {len(all_items)}", flush=True)
            if pno >= total_pages and total_pages > 0:
                break
            time.sleep(random.uniform(SLEEP_MIN, SLEEP_MAX))

        print(f"[*] 列表共 {len(all_items)} 条, 开始抓详情", flush=True)

        f = open(OUT, 'a', encoding='utf-8')
        ok = 0
        fail = 0
        for it in all_items:
            if it['id'] in done_ids:
                continue
            try:
                page.goto(it['url'], timeout=30000, wait_until="domcontentloaded")
                time.sleep(0.8)
                html = page.content()
            except Exception as e:
                print(f"[!] 详情 {it['id']} 失败: {e}", flush=True)
                fail += 1
                if fail >= MAX_FAIL:
                    print("[!] 失败过多, 中止 (下次断点续传)", flush=True)
                    break
                continue
            det = parse_detail(html)
            if not det['body']:
                fail += 1
                if fail >= MAX_FAIL:
                    break
                print(f"[warn] 空正文 {it['id']} {it['title'][:30]}", flush=True)
            else:
                fail = 0
            rec = {
                'id': it['id'],
                'title': det['title'] or it['title'],
                'date': det['date'] or it['date'],
                'url': it['url'],
                'source': 'sxjjb',
                'site': 'sxjjb_rdjj',
                'src_name': det['source'],
                'author': det['author'],
                'body': det['body'],
                'attachments': det['attachments'],
                'crawl_time': datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
            }
            f.write(json.dumps(rec, ensure_ascii=False) + '\n')
            f.flush()
            done_ids.add(it['id'])
            ok += 1
            if ok % 30 == 0:
                print(f"[detail] +{ok} 条 (累计 {len(done_ids)})", flush=True)
            time.sleep(random.uniform(SLEEP_MIN, SLEEP_MAX))
        f.close()
        browser.close()
        print(f"[✓] 完成: 本次新增 {ok} 条, 总计 {len(done_ids)} 条 -> {OUT}", flush=True)
        # 自动导入 DB
        if os.path.exists(OUT):
            import_into_db(OUT)

def import_into_db(jsonl_path):
    """导入 JSONL 到 gov_raw (自动增量入库)"""
    import sqlite3
    from datetime import datetime as _dt
    dbp = os.environ.get('SEARCH_DB', '/root/search.db')
    db = sqlite3.connect(dbp, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA busy_timeout=300000")
    db.execute("PRAGMA synchronous=NORMAL")
    db.execute("""CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(
        title, site_name, summary, tokenize=trigram
    )""")
    db.execute("""CREATE TRIGGER IF NOT EXISTS trg_gov_raw_fts_ins AFTER INSERT ON gov_raw BEGIN
      INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
      VALUES (new.id, new.title, new.site_name, new.summary);
    END""")
    db.execute("""CREATE TRIGGER IF NOT EXISTS trg_gov_raw_fts_del AFTER DELETE ON gov_raw BEGIN
      DELETE FROM gov_search WHERE rowid = old.id;
    END""")
    db.execute("""CREATE TRIGGER IF NOT EXISTS trg_gov_raw_fts_upd AFTER UPDATE ON gov_raw BEGIN
      DELETE FROM gov_search WHERE rowid = old.id;
      INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
      VALUES (new.id, new.title, new.site_name, new.summary);
    END""")
    db.commit()
    added = 0
    errors = 0
    with open(jsonl_path, encoding='utf-8') as f:
        for line in f:
            line = line.strip()
            if not line:
                continue
            try:
                item = json.loads(line)
                title = (item.get("title") or "")[:500]
                page_url = (item.get("page_url") or item.get("url") or "")[:1000]
                content = (item.get("content") or item.get("body") or "").strip()
                publish_date = (item.get("publish_date") or item.get("date") or "")[:20]
                site_name = (item.get("site_name") or "sxjjb_rdjj")[:100]
                att = item.get("attachments") or []
                attachments_str = json.dumps(att, ensure_ascii=False) if att else ""
                if not page_url or not title:
                    errors += 1
                    continue
                plain = re.sub(r'<[^>]+>', ' ', content)
                plain = re.sub(r'\s+', ' ', plain).strip()
                summary = plain[:200]
                dr = 0
                dm = re.search(r'(\d{4})-(\d{2})-(\d{2})', publish_date)
                if dm:
                    try:
                        dr = int(_dt(int(dm.group(1)), int(dm.group(2)), int(dm.group(3))).timestamp())
                    except Exception:
                        dr = 0
                has_table = 1 if '<table' in content else 0
                db.execute("DELETE FROM gov_raw WHERE page_url = ?", (page_url,))
                db.execute(
                    "INSERT INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments, script_name, summary, date_rank, has_table) VALUES (?, ?, ?, ?, ?, ?, 'synced', ?, 'crawl_sxjjb_rdjj.py', ?, ?, ?)",
                    (title, page_url, content, publish_date, site_name, page_url, attachments_str, summary, dr, has_table)
                )
                added += 1
            except Exception as e:
                errors += 1
    db.commit()
    db.close()
    print(f"[DB] 导入完成: added={added} errors={errors}", flush=True)

if __name__ == '__main__':
    main()
