#!/usr/bin/env python3
"""苏银产业园-通知公告 脚本（表格修复版）
https://sycyy.yinchuan.gov.cn/xwzx/tzgg/
"""
import json, re, sys, time, requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "苏银产业园-通知公告"
CATEGORY = "宁夏"
BASE_URL = "https://sycyy.yinchuan.gov.cn/xwzx/tzgg/"
MAX_PAGES = 5
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "https://sycyy.yinchuan.gov.cn/",
}
INSERT_SQL = """INSERT OR IGNORE INTO gov_raw 
    (site_name, page_url, title, publish_date, summary, content, category, attachments)
    VALUES (?, ?, ?, ?, '', ?, ?, ?)"""

def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn

def fetch(url, timeout=30):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = "utf-8"
    return r

# ── Table rendering ──────────────────────────────────────────
def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def extract_content_from_div(div):
    """Extract text preserving tables and paragraphs.
    Iterates direct children in order to maintain document flow.
    """
    parts = []
    for child in div.children:
        if child.name is None:
            text = child.strip()
            if text:
                parts.append(text)
            continue
        tag_name = child.name.lower()
        if tag_name == "table":
            md = table_to_markdown(child)
            if md:
                parts.append(md)
        elif tag_name == "p":
            text = child.get_text(" ", strip=True)
            if text:
                parts.append(text)
        elif tag_name in ("br",):
            continue
        elif tag_name == "div":
            # Recurse
            sub = extract_content_from_div(child)
            if sub:
                parts.append(sub)
        elif tag_name in ("center", "section", "article"):
            sub = extract_content_from_div(child)
            if sub:
                parts.append(sub)
        elif tag_name in ("ul", "ol"):
            for li in child.find_all("li", recursive=False):
                text = li.get_text(" ", strip=True)
                if text:
                    parts.append("- " + text)
        elif tag_name in ("script", "style"):
            continue
        else:
            text = child.get_text(" ", strip=True)
            if text:
                parts.append(text)
    return "\n\n".join(parts)

# ── List page ────────────────────────────────────────────────
def extract_list_page(page_num):
    if page_num == 1:
        url = "https://sycyy.yinchuan.gov.cn/xwzx/tzgg/index.html"
    else:
        url = f"https://sycyy.yinchuan.gov.cn/xwzx/tzgg/index_{page_num - 1}.html"
    try:
        r = fetch(url)
    except Exception as e:
        print(f"  [ERROR] Page {page_num}: {e}", flush=True)
        return []
    if r.status_code != 200:
        print(f"  [ERROR] HTTP {r.status_code} page {page_num}", flush=True)
        return []
    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    list_div = soup.select_one("div.wblist") or soup
    for li in list_div.find_all("li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "").strip()
        if not href or href.startswith("javascript"):
            continue
        if not re.search(r'20\d{6}', href) and not re.search(r'\.html$', href):
            continue
        title = a.get("title", "").strip() or a.get_text(strip=True)
        full_url = urljoin("https://sycyy.yinchuan.gov.cn/xwzx/tzgg/", href)
        li_text = li.get_text(strip=True)
        dm = re.search(r'(20\d{2}[-/]\d{2}[-/]\d{2})', li_text)
        date_str = dm.group(1) if dm else ""
        items.append({"title": title, "url": full_url, "date": date_str})
    print(f"  Page {page_num}: {len(items)} items", flush=True)
    return items

# ── Detail page ──────────────────────────────────────────────
def extract_detail(url):
    try:
        r = fetch(url)
    except Exception as e:
        print(f"  [ERROR] Fetch {url}: {e}", flush=True)
        return None
    if r.status_code != 200:
        print(f"  [ERROR] HTTP {r.status_code} for {url}", flush=True)
        return None
    soup = BeautifulSoup(r.text, "html.parser")

    # Title from ArticleTitle meta
    title = ""
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()

    # Date from PubDate meta
    pub_date = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        pub_date = md["content"].strip()

    # Content divs to try, in order
    content_div = None
    for selector in ["div.content.clearfix", "div.ue_table", "div.view", "div.TRS_Editor"]:
        content_div = soup.select_one(selector)
        if content_div:
            break

    content_text = ""
    attachments_list = []

    if content_div:
        content_text = extract_content_from_div(content_div)

        # Attachments
        for a in content_div.find_all("a", href=True):
            h = a["href"]
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|rar|zip)$', h, re.IGNORECASE):
                at = a.get_text(strip=True) or h.split("/")[-1]
                full_url = urljoin(url, h)
                attachments_list.append({"name": at, "url": full_url})

    if not content_text.strip():
        content_text = f"[{title or url.split('/')[-1]}]({url})\n（本文为PDF附件）"

    return {
        "title": title,
        "content": content_text,
        "pub_date": pub_date,
        "attachments": json.dumps(attachments_list, ensure_ascii=False) if attachments_list else "",
    }

# ── Main ─────────────────────────────────────────────────────
def crawl(test_mode=False, max_pages=MAX_PAGES):
    print(f"[{SITE_NAME}] Starting, max_pages={max_pages}", flush=True)
    conn = init_db()
    cur = conn.cursor()
    total_inserted = total_skipped = pages_crawled = 0
    for pn in range(1, max_pages + 1):
        items = extract_list_page(pn)
        if not items:
            print(f"  Page {pn} empty, stop", flush=True)
            break
        pages_crawled += 1
        for item in items:
            title, url, date = item["title"], item["url"], item["date"]
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if cur.fetchone():
                total_skipped += 1
                continue
            detail = extract_detail(url)
            if detail is None:
                total_skipped += 1
                continue
            if detail["title"] and len(detail["title"]) > len(title):
                title = detail["title"]
            content = detail["content"]
            if len(content.strip()) < 20:
                content = f"[{title}]({url})\n（本文为PDF附件）"
            attachments = detail["attachments"]
            cur.execute(INSERT_SQL, (SITE_NAME, url, title, date or detail["pub_date"],
                                      content, CATEGORY, attachments))
            total_inserted += 1
            if test_mode and total_inserted >= 10:
                break
        conn.commit()
        if test_mode and total_inserted >= 10:
            print(f"  Test mode: stop after {total_inserted}", flush=True)
            break
        time.sleep(0.5)
    conn.close()
    print(f"[{SITE_NAME}] Done. Inserted={total_inserted}, Skipped={total_skipped}, Pages={pages_crawled}", flush=True)
    return total_inserted

if __name__ == "__main__":
    test_mode = "--test" in sys.argv
    mp = 3 if test_mode else MAX_PAGES
    if "--max-pages" in sys.argv:
        idx = sys.argv.index("--max-pages")
        if idx + 1 < len(sys.argv):
            mp = int(sys.argv[idx + 1])
    crawl(test_mode=test_mode, max_pages=mp)
