#!/usr/bin/env python3
"""
crawl_xjss.py — 鄯善县人民政府网-生态环保
CMS: UCAP (数融平台)
列表: /ssx/c106105/list.shtml (第1页), /ssx/c106105/list_{N}.shtml (第2页起)
详情: div.content > div.contentbox > ucapcontent
分页: 15条/页, JS参数 createPageHTML('page_div',5,1,'list','shtml',69)
"""

import re, sys, json, time, sqlite3, os
import requests
from bs4 import BeautifulSoup

# === _MAX_PG preamble ===
import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES

BASE_URL = "http://www.xjss.gov.cn"
LIST_P1 = "/ssx/c106105/list.shtml"
LIST_PN = "/ssx/c106105/list_{}.shtml"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "鄯善县人民政府网-生态环保"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

TOTAL_PAGES = 69

stats = {"new": 0, "skip": 0, "empty": 0, "errors": 0}


def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code == 200:
            return r.text
    except:
        pass
    return None


def parse_list(html):
    """解析列表页，返回 [(url, title, date), ...]"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.find("div", class_="list_rbox")
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        title = a.get("title", "") or a.get_text(strip=True)
        span = li.find("span")
        date = span.get_text(strip=True) if span else ""
        if href.startswith("/"):
            href = BASE_URL + href
        elif not href.startswith("http"):
            href = BASE_URL + "/" + href.lstrip("/")
        items.append((href, title, date))
    return items


def parse_detail(html, page_url):
    """解析详情页，返回 (title, date, content_html, attachments_json)"""
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    date = ""
    content = ""
    attachments = []

    # Title from ucaptitle
    ucap_title = soup.find("ucaptitle")
    if ucap_title:
        title = ucap_title.get_text(strip=True)
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True).replace("鄯善县人民政府网-", "").strip()

    # Date
    pub = soup.find("publishtime")
    if pub:
        date = pub.get_text(strip=True)
    if not date:
        meta = soup.find("meta", attrs={"name": "PubDate"})
        if meta:
            date = meta.get("content", "").strip()[:10]

    # Content from ucapcontent
    content_div = soup.find("ucapcontent")
    if content_div:
        # Extract HTML content - convert tables to markdown, keep formatting
        parts = []
        for child in content_div.children:
            if hasattr(child, "name"):
                tag = child.name
                if tag == "table":
                    parts.append("\n" + str(child) + "\n")
                elif tag == "img":
                    src = child.get("src", "")
                    alt = child.get("alt", "")
                    if src.startswith("/"):
                        src = BASE_URL + src
                    parts.append(f"\n![{alt}]({src})\n")
                elif tag in ("p", "div", "span"):
                    txt = child.get_text(" ", strip=True).strip()
                    if txt:
                        parts.append(txt)
                else:
                    txt = child.get_text(strip=True)
                    if txt:
                        parts.append(txt)
            elif isinstance(child, str):
                txt = child.strip()
                if txt:
                    parts.append(txt)
        content = "\n\n".join(parts)
        content = re.sub(r"\n{3,}", "\n\n", content).strip()
    else:
        # Fallback to div.contentbox > div.ewebeditor_doc
        cb = soup.find("div", class_="contentbox")
        if cb:
            content = cb.get_text(" ", strip=True).strip()

    # Attachments
    appendix = soup.find("ul", id="tblAppendix")
    if appendix:
        for li in appendix.find_all("li"):
            a = li.find("a")
            if a:
                href = a.get("href", "")
                text = a.get_text(strip=True)
                if not href.startswith("http"):
                    href = BASE_URL + "/" + href.lstrip("/")
                attachments.append({"title": text, "url": href})

    att_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
    return title, date, content, att_json


def insert_article(url, title, date, content, attachments):
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    try:
        summary = re.sub(r"<[^>]+>", "", content)[:200] if content else title[:200]
        db.execute(
            "INSERT OR IGNORE INTO gov_raw(site_name,source_url,page_url,title,publish_date,content,summary,status,category,attachments) VALUES(?,?,?,?,?,?,?,?,?,?)",
            (SITE_NAME, url, url, title, date, content, summary, "active", "生态环保", attachments)
        )
        affected = db.total_changes
        db.commit()
        if affected:
            db.execute(
                "INSERT INTO gov_search(rowid,title,site_name,summary) SELECT r.id,r.title,r.site_name,r.summary FROM gov_raw r WHERE r.page_url=? AND r.id NOT IN (SELECT rowid FROM gov_search)",
                (url,)
            )
            db.commit()
        db.close()
        return affected
    except Exception as e:
        db.close()
        print(f"  DB error: {e}")
        return 0


def run():
    print(f"\n{'='*50}")
    print(f"🚀 {SITE_NAME}")
    print(f"{'='*50}")

    max_pages = _MAX_PG if _MAX_PG else TOTAL_PAGES
    all_items = []

    for page in range(1, max_pages + 1):
        if page == 1:
            url = BASE_URL + LIST_P1
        else:
            url = BASE_URL + LIST_PN.format(page)
        html = fetch(url)
        if not html:
            print(f"  ❌ 第{page}页获取失败")
            continue
        items = parse_list(html)
        if not items:
            print(f"  📄 第{page}页: 空")
            break
        all_items.extend(items)
        print(f"  📄 第{page}页: {len(items)}条 → 累计{len(all_items)}条")
        time.sleep(0.3)

    print(f"\n📊 列表共 {len(all_items)} 条")

    # DB dedup
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    existing = set()
    try:
        cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        for row in cur.fetchall():
            existing.add(row[0])
    except:
        pass
    db.close()
    print(f"  DB已有: {len(existing)} 条")

    all_items = [it for it in all_items if it[0] not in existing]
    print(f"  待抓: {len(all_items)} 条")

    if not all_items:
        print("  ✅ 全部已存在")
        return

    new = skip = empty = errors = 0
    for i, (url, list_title, list_date) in enumerate(all_items, 1):
        html = fetch(url)
        if not html:
            errors += 1
            continue

        title, date, content, attachments = parse_detail(html, url)
        real_title = title or list_title
        real_date = date or list_date

        text_content = re.sub(r"<[^>]+>", "", content).strip()
        if not text_content:
            empty += 1
            continue

        if insert_article(url, real_title, real_date, content, attachments):
            new += 1
        else:
            skip += 1

        if i % 10 == 0:
            print(f"  ...{i}/{len(all_items)}")
        time.sleep(0.3)

    # FTS sync
    print(f"\n📊 FTS同步...")
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        db.execute("PRAGMA busy_timeout=30000")
        db.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
        db.execute("INSERT INTO gov_search(rowid,title,site_name,summary) SELECT rowid,title,site_name,summary FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        db.commit()
        cnt = db.execute("SELECT COUNT(*) FROM gov_search WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
        print(f"  FTS: {cnt} 条")
    except Exception as e:
        print(f"  FTS error: {e}")
    finally:
        db.close()

    print(f"\n✅ {SITE_NAME}: 新增{new}, 跳过{skip}, 空正文{empty}, 错误{errors}")


if __name__ == "__main__":
    run()
