#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""补抓大邑 c138361 外链详情 (cdpta iframe / mod.gov.cn)"""
import json, os, re, sys, time, sqlite3, urllib.request, ssl

SITE_NAME = "大邑县-通知公告"
CATEGORY = "通知公告"
GROUP_NAME = "四川"
DB_PATH = "/root/search.db"

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"

TARGETS = [
    ("2026年成都市大邑县事业单位公开选调工作人员笔试成绩及进入资格复审人员名单公示",
     "https://cdpta.cdrsigc.com/netpage/noticecontent.jsp?typeid=60&contentid=/frt/frtuploadfile/uploadfile/bulletin/2026/4bced77b4ea3917446c99cd20cc556fe.html"),
    ("大邑县2026年公开考试招聘中小学教师考试总成绩及进入体检人员名单公示",
     "https://cdpta.cdrsigc.com/netpage/noticecontent.jsp?typeid=60&contentid=/frt/frtuploadfile/uploadfile/bulletin/2026/7f16019ccc57ee693a0c3324e49e6af8.html"),
    ("成都市大邑县2025年上半年公开考试招聘中小学教师拟聘用人员名单公示（一）",
     "https://cdpta.cdrsigc.com/netpage/noticecontent.jsp?typeid=60&contentid=/frt/frtuploadfile/uploadfile/bulletin/2025/250805161535190250805161535190ope2.html"),
    ("成都市大邑县2025年下半年公开考核招聘急需紧缺教育人才公告",
     "https://cdpta.cdrsigc.com/netpage/noticecontent.jsp?typeid=60&contentid=/frt/frtuploadfile/uploadfile/bulletin/2025/250711113231371250711113231371pvto.html"),
    ("关于开展军队文职人员考试培训专项整治的通告",
     "http://www.mod.gov.cn/gfbw/qwfb/yw_214049/16389499.html"),
    ("大邑县2024年公开招聘教师公告",
     "https://cdpta.cdrsigc.com/frt/frtuploadfile/uploadfile/bulletin/2024/2403151528369362403151528369365f98.html"),
]

def fetch(url):
    try:
        req = urllib.request.Request(url, headers={"User-Agent": UA})
        r = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
        return r.read().decode("utf-8", errors="replace")
    except Exception as e:
        print(f"  fetch失败 {url[:60]}: {e}")
        return ""

def extract_from_cdpta(html, title):
    """cdpta: iframe src → 抓内层 → 纯 p 段落"""
    m = re.search(r'<iframe[^>]*src="([^"]+)"', html)
    if m:
        src = m.group(1)
        if not src.startswith("http"):
            src = "https://cdpta.cdrsigc.com" + src
        inner = fetch(src)
        if inner:
            return inner
    # 无 iframe 但直接是内容 (2024教师公告案例)
    if "<p" in html:
        return html
    return ""

def extract_from_mod(html):
    m = re.search(r'<div[^>]*class="article-content[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.S)
    if not m:
        m = re.search(r'<div[^>]*class="article-content[^"]*"[^>]*>(.*)', html, re.S)
    return m.group(1) if m else ""

def clean_content(html):
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    for tag in soup.find_all(lambda t: t.name and ":" in t.name):
        tag.decompose()
    for tag in soup.find_all(["span", "font"]):
        tag.unwrap()
    while True:
        nested = [p for p in soup.find_all("p") if p.find("p")]
        if not nested:
            break
        nested[0].unwrap()
    for p in soup.find_all("p"):
        if p.get("style"):
            p.attrs = {"align": p.get("align")} if p.get("align") else {}
    return str(soup)

def main():
    conn = sqlite3.connect(DB_PATH, timeout=290)
    conn.execute("PRAGMA busy_timeout=290000")
    cur = conn.cursor()
    added = 0
    for title, url in TARGETS:
        html = fetch(url)
        if not html:
            print(f"  [FAIL] 抓取失败: {title[:40]}")
            continue
        if "cdpta" in url or "cdrsigc" in url:
            content_html = extract_from_cdpta(html, title)
        elif "mod.gov.cn" in url:
            content_html = extract_from_mod(html)
        else:
            content_html = html
        if not content_html:
            print(f"  [FAIL] 无正文: {title[:40]}")
            continue
        content = clean_content(content_html)
        text = re.sub(r"<[^>]+>", "", content).strip()
        if len(text) < 10 and "<img" not in content:
            print(f"  [FAIL] 空正文: {title[:40]}")
            continue
        # img 绝对化 (cdpta 相对路径 → cdrsigc 域)
        if "<img" in content:
            base_domain = "https://cdpta.cdrsigc.com" if "cdpta" in url or "cdrsigc" in url else "https://www.mod.gov.cn"
            content = re.sub(
                r'(<img[^>]+src=")([^"]+)(")',
                lambda mm: mm.group(1) + (mm.group(2) if mm.group(2).startswith("http") else (base_domain + mm.group(2) if mm.group(2).startswith("/") else url.rsplit("/", 1)[0] + "/" + mm.group(2))) + mm.group(3),
                content)
        # 日期: URL 中提取 /2026/ 或 /2025/ 年
        dm = re.search(r"/(20\d{2})/", url)
        pub_date = dm.group(1) + "-01-01" if dm else ""
        try:
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (site_name, source_url, page_url, title, publish_date, content, summary, category, script_name, attachments, group_name, industry)
                   VALUES (?,?,?,?,?,?,?,?,?,?,?,?)""",
                (SITE_NAME, url, url, title, pub_date, content,
                 content[:500], CATEGORY, "crawl_dayi_tzgg_c138361.py",
                 "", GROUP_NAME, "other"))
            if cur.rowcount > 0:
                added += 1
                print(f"  [OK] {title[:40]} | {pub_date} | 正文{len(text)}字符")
            else:
                print(f"  [SKIP] 已存在: {title[:40]}")
        except sqlite3.Error as e:
            print(f"  DB err: {e}")
        conn.commit()
        time.sleep(1)
    conn.close()
    print(f"=== 补抓完成: 新增 {added} ===")

if __name__ == "__main__":
    main()
