#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""补抓大邑 c138361 详情失败的 7 条"""
import json, os, re, sys, time, sqlite3
from playwright.sync_api import sync_playwright

SITE_NAME = "大邑县-通知公告"
BASE = "https://www.day.gov.cn"
LMCODE = "761e9373602340e792d0145a6a6a6478"
LIST_API = f"/es-search/search/{LMCODE}?_template=zhaofa/day_public&_isAgg=1&_pageSize=20&page="
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
DB_PATH = "/root/search.db"
CATEGORY = "通知公告"
GROUP_NAME = "四川"

FAIL_KEYWORDS = [
    "公开选调工作人员笔试成绩及进入",
    "公开考试招聘中小学教师考试总成绩及进入体检人",
    "上半年公开考试招聘中小学教师拟聘用人员",
    "下半年公开考核招聘急需紧缺教育人才公告",
    "军队文职人员考试培训专项整治",
    "2024年公开招聘教师公告",
]

def clean_title(t):
    t = re.sub(r"^[•··\s]+", "", t or "").strip()
    t = t.replace("\u200b", "").replace("\ufeff", "")
    return t

def clean_content(html):
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    for tag in soup.find_all(lambda t: t.name and ":" in t.name):
        tag.decompose()
    for tag in soup.find_all(["span", "font"]):
        tag.unwrap()
    while True:
        nested = [p for p in soup.find_all("p") if p.find("p")]
        if not nested:
            break
        nested[0].unwrap()
    for p in soup.find_all("p"):
        if p.get("style"):
            p.attrs = {"align": p.get("align")} if p.get("align") else {}
    return str(soup)

def extract_attachments(detail_html, detail_url):
    atts = []
    for m in re.finditer(r'<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>', detail_html, re.I | re.S):
        href, name_html = m.group(1).strip(), m.group(2)
        if not href or href.startswith("javascript"):
            continue
        if re.search(r"\.(pdf|docx?|xlsx?|zip|rar|7z)$", href, re.I):
            name = re.sub(r"<[^>]+>", "", name_html).strip()[:100]
            if href.startswith("/"):
                href = BASE + href
            elif not href.startswith("http"):
                href = detail_url.rsplit("/", 1)[0] + "/" + href
            atts.append({"name": name or href.split("/")[-1], "url": href})
    seen = set()
    uniq = []
    for a in atts:
        if a["url"] not in seen:
            seen.add(a["url"])
            uniq.append(a)
    return uniq

def fetch_ok(page, url, retries=4):
    for i in range(retries):
        try:
            return page.evaluate("""async (u) => { const r = await fetch(u); return await r.text(); }""", url)
        except Exception:
            time.sleep(3 + i * 2)
    return None

def main():
    with sync_playwright() as p:
        browser = p.chromium.launch(
            headless=True,
            args=["--no-sandbox", "--disable-blink-features=AutomationControlled", "--disable-dev-shm-usage"])
        ctx = browser.new_context(user_agent=UA, locale="zh-CN", viewport={"width": 1366, "height": 900})
        page = ctx.new_page()
        try:
            page.goto(BASE + "/day/c138361/list.shtml", timeout=60000, wait_until="commit")
        except Exception:
            pass
        passed = False
        for i in range(20):
            page.wait_for_timeout(3000)
            try:
                html = page.content()
            except Exception:
                continue
            if "$_ts" not in html and len(html) > 5000:
                print(f"瑞数挑战通过 {i*3}s", flush=True)
                passed = True
                break
        if not passed:
            print("挑战失败!", flush=True)
            browser.close()
            return

        # 1. 重扫列表，找失败条目的 URL
        target_urls = {}
        for pn in range(1, 67):
            html = fetch_ok(page, BASE + LIST_API + str(pn))
            if not html:
                continue
            for m in re.finditer(r'<li>\s*<h6><a href="([^"]+)"[^>]*?title=[\'"]([^\'"]+)[\'"]', html, re.S):
                href, title = m.group(1), clean_title(m.group(2))
                for kw in FAIL_KEYWORDS:
                    if kw in title:
                        target_urls[title] = href if href.startswith("http") else BASE + href
                        break
            time.sleep(0.3)
        print(f"找到 {len(target_urls)} 个目标URL:", flush=True)
        for t, u in target_urls.items():
            print(f"  {t[:40]} -> {u}", flush=True)

        # 2. 逐个抓详情入库
        conn = sqlite3.connect(DB_PATH, timeout=290)
        conn.execute("PRAGMA busy_timeout=290000")
        cur = conn.cursor()
        added = 0
        for title, url in target_urls.items():
            detail = fetch_ok(page, url, retries=5)
            if not detail:
                print(f"  [FAIL] 仍失败: {title[:40]}", flush=True)
                continue
            m = re.search(r'<div[^>]*class="details_text"[^>]*>(.*?)</div>\s*</div>', detail, re.S)
            if m:
                content_html = m.group(1)
            else:
                m2 = re.search(r'class="details_con[^"]*"[^>]*>(.*?)$', detail, re.S)
                content_html = m2.group(1)[:8000] if m2 else ""
            content = clean_content(content_html) if content_html else ""
            text = re.sub(r"<[^>]+>", "", content).strip()
            if len(text) < 10 and "<img" not in content:
                print(f"  [FAIL] 空正文: {title[:40]}", flush=True)
                continue
            if "<img" in content:
                content = re.sub(
                    r'(<img[^>]+src=")([^"]+)(")',
                    lambda mm: mm.group(1) + (mm.group(2) if mm.group(2).startswith("http") else (BASE + mm.group(2) if mm.group(2).startswith("/") else url.rsplit("/", 1)[0] + "/" + mm.group(2))) + mm.group(3),
                    content)
            # 日期: meta PubDate 或 URL
            dm = re.search(r'<meta[^>]+name="PubDate"[^>]+content="([^"]+)"', detail)
            pub_date = dm.group(1)[:10] if dm else ""
            if not pub_date:
                um = re.search(r"/(\d{4}-\d{2}/\d{2})/", url)
                pub_date = um.group(1).replace("/", "-") if um else ""
            atts = extract_attachments(detail, url)
            try:
                cur.execute(
                    """INSERT OR IGNORE INTO gov_raw
                       (site_name, source_url, page_url, title, publish_date, content, summary, category, script_name, attachments, group_name, industry)
                       VALUES (?,?,?,?,?,?,?,?,?,?,?,?)""",
                    (SITE_NAME, url, url, title, pub_date, content,
                     content[:500], CATEGORY, "crawl_dayi_tzgg_c138361.py",
                     json.dumps(atts, ensure_ascii=False) if atts else "", GROUP_NAME, "other"))
                if cur.rowcount > 0:
                    added += 1
                    print(f"  [OK] {title[:40]} | {pub_date}", flush=True)
                else:
                    print(f"  [SKIP] 已存在: {title[:40]}", flush=True)
            except sqlite3.Error as e:
                print(f"  DB err: {e}", flush=True)
            conn.commit()
            time.sleep(1.5)
        conn.close()
        browser.close()
        print(f"=== 补抓完成: 新增 {added} ===", flush=True)

if __name__ == "__main__":
    main()
