#!/usr/bin/env python3
"""乌达区-措施和实施情况爬虫
站点: https://www.wuda.gov.cn
栏目: 重点领域信息 > 生态环境 > 措施和实施情况 (cshssqk)
分页: index_N.shtml (N=2..15), 10条/页, 共15页约146条
CMS: ZCMS
"""

import os, re, sys, time, subprocess
from bs4 import BeautifulSoup
import requests

DB_PATH = os.getenv("SEARCH_DB", "/mnt/data/search.db")
BASE_URL = "https://www.wuda.gov.cn"
LIST_PATH = "/zfxxgk/fdzdgknr/zdlyxx/sthj/cshssqk/"
MAX_PAGES = 15
SITE_NAME = "乌达区-措施和实施情况"
INDUSTRY = "政府公告"
GROUP = "内蒙古"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def fetch(url, retries=3):
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
            else:
                print(f"  [ERROR] {url}: {e}", file=sys.stderr)
                return None


def parse_list(html):
    """解析列表页HTML，返回 [{page_url, title, date}]"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for tr in soup.select("table#table1 tbody tr"):
        a = tr.select_one("td:nth-child(2) a")
        date_span = tr.select_one("td:nth-child(4) span") or tr.select_one("td span")
        if not a:
            continue
        href = a.get("href", "")
        if not href.startswith("http"):
            href = BASE_URL + href
        title = a.get("title", "") or a.get_text(strip=True)
        if not title:
            continue
        pub_date = date_span.get_text(strip=True) if date_span else ""
        dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", pub_date)
        if dm:
            pub_date = dm.group(1)
        items.append({
            "page_url": href,
            "title": title.strip(),
            "date": pub_date,
        })
    return items


def parse_detail(html, url):
    """解析详情页，返回 (title, publish_date, content)"""
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    publish_date = ""
    content = ""

    # Title
    title_tag = soup.find("title")
    if title_tag:
        t = title_tag.get_text(strip=True)
        t = t.split("_措施和实施情况")[0].split("_乌达区")[0].strip()
        if t:
            title = t

    # Publish date
    body_text = soup.get_text()
    dm = re.search(r"发布日期[：:]?\s*(\d{4}-\d{1,2}-\d{1,2})", body_text)
    if dm:
        publish_date = dm.group(1)

    # Content
    content_div = soup.select_one("div.Detail_text")
    if content_div:
        content = content_div.get_text(strip=True)
        content = re.sub(r"\s{3,}", "\n\n", content)

    return title, publish_date, content


def save_to_db(page_url, title, publish_date, content):
    """插入到search.db"""
    # Check dup
    check_sql = f"SELECT rowid FROM gov_raw WHERE page_url = '{page_url.replace(chr(39), chr(39)+chr(39))}'"
    result = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, check_sql], capture_output=True, text=True, timeout=10)
    if result.stdout.strip():
        return "dup"

    def esc(s):
        return s.replace("'", "''") if s else ""

    summary = content[:200] if content else ""

    sql = f"""INSERT INTO gov_raw (page_url, title, publish_date, content, site_name, industry, summary)
VALUES ('{esc(page_url)}', '{esc(title)}', '{esc(publish_date)}', '{esc(content)}', '{esc(SITE_NAME)}', '{INDUSTRY}', '{esc(summary)}')"""

    result = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, text=True, timeout=10)
    if result.returncode != 0 and "UNIQUE" not in result.stderr:
        print(f"  [DB ERROR] {result.stderr}", file=sys.stderr)
        return "error"

    # FTS sync
    sync_sql = f"""INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
SELECT rowid, title, site_name, summary FROM gov_raw
WHERE page_url = '{esc(page_url)}' AND rowid NOT IN (SELECT rowid FROM gov_search)"""
    subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sync_sql], capture_output=True, text=True, timeout=10)
    return "new"


def crawl():
    max_pages_str = sys.argv[1] if len(sys.argv) > 1 else str(MAX_PAGES)
    try:
        max_pages = int(max_pages_str)
    except ValueError:
        max_pages = MAX_PAGES

    print(f"[{SITE_NAME}] Starting crawl, max_pages={max_pages}", flush=True)
    all_items = []

    for page in range(1, max_pages + 1):
        if page == 1:
            url = BASE_URL + LIST_PATH
        else:
            url = BASE_URL + LIST_PATH + f"index_{page}.shtml"
        html = fetch(url)
        if not html:
            break
        items = parse_list(html)
        if not items:
            break
        print(f"  Page {page}: {len(items)} items", flush=True)
        all_items.extend(items)
        if len(items) < 10:
            break

    print(f"Total list items: {len(all_items)}", flush=True)

    new_count = dup_count = error_count = 0
    for item in all_items:
        print(f"  Fetching: {item['title'][:40]}...", flush=True)
        detail_html = fetch(item["page_url"])
        if not detail_html:
            error_count += 1
            continue
        title, publish_date, content = parse_detail(detail_html, item["page_url"])
        if not title:
            title = item.get("title", "")
        if not publish_date:
            publish_date = item.get("date", "")
        if not title:
            error_count += 1
            continue
        try:
            result = save_to_db(item["page_url"], title, publish_date, content)
            if result == "new":
                new_count += 1
            elif result == "dup":
                dup_count += 1
            else:
                error_count += 1
        except Exception as e:
            print(f"  [ERROR] DB: {e}", file=sys.stderr)
            error_count += 1

    print(f"\n=== {SITE_NAME} Done ===", flush=True)
    print(f"New: {new_count}, Dup: {dup_count}, Error: {error_count}", flush=True)


if __name__ == "__main__":
    crawl()
