#!/usr/bin/env python3
"""威县人民政府 - 公告公示"""
import sys, os, re, json, time, sqlite3, html
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

BASE = "https://www.weixian.gov.cn"
LIST_URL = BASE + "/channel/list/269.html"
SITE = "威县人民政府"
COLUMN = "公告公示"
PROVINCE = "河北"
TOTAL_PAGES = 5
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

session = requests.Session()
session.headers.update({"User-Agent": "Mozilla/5.0"})
session.verify = False


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def strip_html(text):
    if not text:
        return ""
    text = html.unescape(text)
    text = re.sub(r'</?(?:p|div|h[1-6]|li|tr|blockquote|section|article|br\s*/?)[^>]*>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'<[^>]+>', '', text)
    text = re.sub(r'[ \t]+', ' ', text)
    text = re.sub(r'\n{3,}', '\n\n', text)
    return text.strip()


def import_to_db(record):
    try:
        db = sqlite3.connect(DB_PATH, timeout=10)
        title = (record.get("title") or "")[:500]
        page_url = (record.get("page_url") or "")[:1000]
        content = record.get("content") or ""
        content_text = strip_html(content)[:500]
        publish_date = (record.get("publish_date") or "")[:20]
        site_name = (record.get("site_name") or "unknown")[:100]
        script_name = (record.get("script_name") or "")[:200]
        attachments_str = json.dumps(record.get("attachments") or [], ensure_ascii=False)

        old = db.execute("SELECT rowid FROM gov_raw WHERE page_url = ?", (page_url,)).fetchone()
        if old:
            db.execute("DELETE FROM gov_search WHERE rowid = ?", (old[0],))
            db.execute("DELETE FROM gov_raw WHERE page_url = ?", (page_url,))

        db.execute(
            "INSERT INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments, script_name) VALUES (?,?,?,?,?,?,'synced',?,?)",
            (title, page_url, content, publish_date, site_name, page_url, attachments_str, script_name)
        )
        new_rowid = db.execute("SELECT last_insert_rowid()").fetchone()[0]
        db.execute(
            "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
            (new_rowid, title, site_name, content_text)
        )
        db.commit()
        db.close()
        log(f"  ✅ {title[:30]}")
        return True
    except Exception as e:
        log(f"  ❌ DB: {e}")
        return False


def fetch_list(page=1):
    url = LIST_URL if page == 1 else f"{BASE}/channel/list/269_{page}.html"
    log(f"  列表[{page}]: {url}")
    try:
        r = session.get(url, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        log(f"  ❌ {e}")
        return []

    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    for a in soup.find_all("a", href=re.compile(r"^/single/269/")):
        title = a.get_text(strip=True)
        href = urljoin(BASE, a["href"])
        date_span = a.find_next_sibling("span", class_="time")
        pub_date = date_span.get_text(strip=True) if date_span else ""
        title = re.sub(r'^[\s\u00b7\u00a0·]+', '', title).strip()
        if title:
            items.append((title, href, pub_date))
    log(f"  → {len(items)} 条")
    return items


def fetch_detail(url):
    log(f"    详情: {url.split('/')[-1]}")
    try:
        r = session.get(url, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        log(f"    ❌ {e}")
        return "", ""

    soup = BeautifulSoup(r.text, "html.parser")
    h1 = soup.find("h1", class_="c-h1")
    title = h1.get_text(strip=True) if h1 else ""
    title = re.sub(r'^[\s\u00b7\u00a0·]+', '', title).strip()

    detail = soup.find("div", class_="c-detail")
    if not detail:
        return title, ""

    parts = []
    for child in detail.children:
        if child.name == "p":
            t = child.get_text("", strip=True)
            if t:
                parts.append(t)
        elif child.name == "table":
            parts.append(str(child))
        elif child.name == "div":
            for sub in child.find_all(["p", "table"], recursive=False):
                if sub.name == "p":
                    t = sub.get_text("", strip=True)
                    if t:
                        parts.append(t)
                elif sub.name == "table":
                    parts.append(str(sub))
    return title, "\n\n".join(parts)


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=TOTAL_PAGES)
    args = parser.parse_args()

    count = 0
    script_name = os.path.basename(__file__)

    for page in range(1, args.pages + 1):
        items = fetch_list(page)
        if not items:
            break
        for title, url, pub_date in items:
            dt, content = fetch_detail(url)
            if not content:
                continue
            if import_to_db({
                "title": dt or title,
                "page_url": url,
                "content": content,
                "publish_date": pub_date,
                "site_name": SITE,
                "script_name": script_name,
            }):
                count += 1
            time.sleep(0.5)
        log(f"  第{page}页完成，累计{count}条")
        time.sleep(1)

    log(f"\n✅ 威县爬取完成，共入库 {count} 条")

if __name__ == "__main__":
    main()
