#!/usr/bin/env python3
"""Crawler for 科尔沁区-生态环境 (keerqin.gov.cn /zwgk/zfxxgk/fdzdgknr/zdlyxx/sthj/)
   CMS: TRS 信息公开, pagination: index_{N}.html (50 pages, ~509 records)
   Detail: div.lis_list_part2_content or div.trs_editor_view
"""
import urllib.request, urllib.parse, ssl, re, sqlite3, sys, time
from bs4 import BeautifulSoup
from datetime import datetime

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

BASE = "http://www.keerqin.gov.cn"
LIST_DIR = "/zwgk/zfxxgk/fdzdgknr/zdlyxx/sthj/"
DB = "/root/search.db"

SITE_NAME = "keerqin_sthj"
MAX_PAGES = 51  # index.html (0) + index_1..index_50
INCREMENTAL = "--incremental" in sys.argv


def log(msg):
    print(f"[{SITE_NAME}] {msg}")


def fetch(url):
    req = urllib.request.Request(url, headers={
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
        "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    })
    r = urllib.request.urlopen(req, context=ssl_ctx, timeout=30)
    return r.read().decode("utf-8", errors="replace")


def parse_list(html, page_num):
    """解析 tbody > tr > td > a[href] + td date"""
    items = []
    tbody_match = re.search(r"<tbody>(.*?)</tbody>", html, re.DOTALL)
    if not tbody_match:
        log(f"第{page_num}页未找到 tbody")
        return items
    tbody_html = tbody_match.group(1)
    for tr_match in re.finditer(r"<tr>(.*?)</tr>", tbody_html, re.DOTALL):
        tr = tr_match.group(1)
        a_match = re.search(r'href="([^"]+)"[^>]*>([^<]+)', tr)
        if not a_match:
            continue
        href = a_match.group(1).strip()
        title = a_match.group(2).strip()
        # Date is in the last <td> before </tr>
        tds = re.findall(r"<td[^>]*>([^<]*)</td>", tr)
        date_str = tds[-1].strip() if tds else ""
        if not href.startswith("http"):
            href = BASE + href
        items.append({
            "title": title,
            "url": href,
            "date": date_str,
        })
    return items


def get_page_url(page_num):
    """page 0 = index.html, page 1..50 = index_N.html"""
    if page_num == 0:
        return f"{BASE}{LIST_DIR}index.html"
    return f"{BASE}{LIST_DIR}index_{page_num}.html"


def fetch_detail(url):
    """正文：div.lis_list_part2_content 或 div.trs_editor_view，HTML 输出（表格保留+附件内嵌）"""
    html = fetch(url)
    soup = BeautifulSoup(html, "html.parser")
    content_div = (
        soup.select_one("div.lis_list_part2_content")
        or soup.select_one("div.trs_editor_view")
        or soup.select_one("div.cont_cont")
    )
    if not content_div:
        log(f"未找到正文: {url}")
        return None

    def to_abs(u):
        u = (u or "").strip()
        if not u:
            return ""
        if u.startswith("//"):
            return "http:" + u
        if u.startswith("http"):
            return u
        return urllib.parse.urljoin(url, u)

    def clean(fragment):
        """unwrap 样式包装 + 清内联 style + 绝对化链接 + 删空锚点/style 标签"""
        fsoup = BeautifulSoup(fragment, "html.parser")
        # 删除 Word 残留 style 标签
        for st in fsoup.find_all("style"):
            st.decompose()
        # OLE_LINK / ms 锚点（带文本的）unwrap 保留文本（兼容 id 和 name 属性）
        for a in fsoup.find_all("a"):
            aid = a.get("id") or a.get("name") or ""
            if "OLE_LINK" in aid or aid.startswith("ms"):
                a.unwrap()
        # 删除空锚点（无文本/无 img）
        for a in fsoup.find_all("a"):
            if not a.get_text(strip=True) and not a.find("img"):
                a.decompose()
        # 删除空 p（仅含 br 或无内容）
        for p in fsoup.find_all("p"):
            if not p.get_text(strip=True) and not p.find("img") and not p.find("a"):
                p.decompose()
        for tag in fsoup.find_all(["span", "font", "div"]):
            tag.unwrap()
        for tag in fsoup.find_all(["p", "td", "tr", "th", "tbody", "table", "b", "strong", "u", "i", "em", "h1", "h2", "h3", "a", "br", "sup", "sub", "img"]):
            if tag.has_attr("style"):
                del tag["style"]
        for a in fsoup.find_all("a", href=True):
            a["href"] = to_abs(a["href"])
        for img in fsoup.find_all("img", src=True):
            src = img.get("src", "")
            if "icon_" in src or "/sysimage/" in src or "/filetypeimages/" in src:
                img.decompose()
                continue
            img["src"] = to_abs(img["src"])
        return str(fsoup)

    parts = []
    for el in content_div.find_all(recursive=False):
        if el.name in ("p", "div"):
            inner = clean(el.decode_contents())
            txt = re.sub(r"<[^>]+>", "", inner).strip()
            # 去嵌套 p（内容已是 <p> 时不包外层）
            if inner.strip().startswith("<p>") and inner.strip().endswith("</p>"):
                parts.append(inner.strip())
            elif txt or "<a " in inner:
                parts.append(f"<p>{inner}</p>")
        elif el.name == "table":
            parts.append(f"<table>{clean(el.decode_contents())}</table>")
        else:
            txt = el.get_text(" ", strip=True)
            if txt:
                parts.append(f"<p>{txt}</p>")

    content = "\n".join(parts)
    if not content:
        txt = content_div.get_text(" ", strip=True)
        if txt:
            content = f"<p>{txt}</p>"
    if not content:
        log(f"正文为空: {url}")
        return None
    return {"content": content}


def main():
    conn = sqlite3.connect(DB, timeout=60)
    cur = conn.cursor()
    total_new = 0
    
    page_end = min(MAX_PAGES - 1, 4) if INCREMENTAL else MAX_PAGES - 1
    
    for page_num in range(0, page_end + 1):
        url = get_page_url(page_num)
        log(f"抓取列表页 {page_num + 1}/{page_end + 1}: {url}")
        try:
            html = fetch(url)
        except Exception as e:
            log(f"列表页 {page_num} 失败: {e}")
            continue
        items = parse_list(html, page_num)
        log(f"  解析到 {len(items)} 条")
        
        for item in items:
            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if cur.fetchone():
                if INCREMENTAL:
                    log(f"  已有记录，增量结束")
                    conn.close()
                    log(f"增量完成，共 {total_new} 条新增")
                    return
                continue
            
            detail = fetch_detail(item["url"])
            if detail is None:
                continue
            
            content = detail["content"]
            now = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw 
                (site_name, source_url, page_url, title, publish_date, summary, content, status, category)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                (
                    SITE_NAME,
                    item["url"],
                    item["url"],
                    item["title"],
                    item["date"],
                    content[:200],
                    content,
                    "published",
                    "生态环境",
                )
            )
            if cur.rowcount > 0:
                total_new += 1
                if total_new % 10 == 0:
                    conn.commit()
                    log(f"  已入库 {total_new} 条...")
        
        conn.commit()
        log(f"第{page_num + 1}页完成，累计 {total_new} 条")
        time.sleep(0.3)
    
    conn.close()
    log(f"全量爬取完成，共新增 {total_new} 条")


if __name__ == "__main__":
    main()
