#!/usr/bin/env python3
"""
襄汾县人民政府 - 通知公告
URL: http://www.xiangfen.gov.cn/channels/4251.html
CMS: TRS WCM (UCAP-CONTENT)
List: channels/4251.html, channels/4251_2.html, channels/4251_3.html ...
Detail: div.pages_content#UCAP-CONTENT > p
"""
import sys, os, re, json, time, sqlite3
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

BASE = "http://www.xiangfen.gov.cn"
LIST_PREFIX = BASE + "/channels/4251"
SITE = "襄汾县人民政府-通知公告"
COLUMN = "通知公告"
PROVINCE = "山西"
PER_PAGE = 20  # ~20 per page based on 431/22

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
session = requests.Session()
session.headers.update(HEADERS)
session.verify = False


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def fetch(url, retries=3):
    for attempt in range(retries):
        try:
            r = session.get(url, timeout=30)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            log(f"  [WARN] 请求失败 (尝试 {attempt+1}/{retries}): {e}")
            time.sleep(2)
    return ""


def parse_list_page(html):
    """从列表页解析标题、日期、URL"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    # 每项在 div.con > ul 中
    con = soup.find("div", class_="con")
    if not con:
        # fallback: look for the list area
        for ul in soup.find_all("ul"):
            li_title = ul.find("li", class_="title")
            if li_title and li_title.find("a"):
                a = li_title.find("a")
                li_time = ul.find("li", class_="time")
                title = a.get("title") or a.get_text(strip=True)
                href = a.get("href", "")
                date_str = li_time.get_text(strip=True) if li_time else ""
                if title and href:
                    items.append({
                        "url": urljoin(BASE, href),
                        "title": title.strip(),
                        "date": date_str,
                    })
        return items

    for ul in con.find_all("ul"):
        li_title = ul.find("li", class_="title")
        if not li_title or not li_title.find("a"):
            continue
        a = li_title.find("a")
        li_time = ul.find("li", class_="time")

        title = a.get("title") or a.get_text(strip=True)
        href = a.get("href", "")
        date_str = li_time.get_text(strip=True) if li_time else ""

        if title and href:
            items.append({
                "url": urljoin(BASE, href),
                "title": title.strip(),
                "date": date_str,
            })
    return items


def fetch_detail(detail_url, list_title):
    """提取详情页正文"""
    html = fetch(detail_url)
    if not html:
        return list_title, "", "", []
    soup = BeautifulSoup(html, "html.parser")

    # 标题: h1 in .cont.w
    full_title = list_title
    h1 = soup.find("h1")
    if h1:
        t = h1.get_text(strip=True)
        if t:
            full_title = t

    # 日期: .pages-date（不能以 list_title 兜底，否则标题污染日期字段）
    date_str = ""
    pd = soup.find("div", class_="pages-date")
    if pd:
        d = pd.get_text(strip=True)[:10]
        if d:
            date_str = d
    if not date_str:
        meta = soup.find("meta", attrs={"name": "PubDate"})
        if meta and meta.get("content"):
            date_str = meta["content"][:10]
    if not date_str:
        m = re.search(r"20\d{2}[-/年]\d{1,2}[-/月]\d{1,2}", html[:8000])
        if m:
            date_str = m.group(0).replace("年", "-").replace("月", "-").replace("/", "-")[:10]

    # 正文: div.pages_content#UCAP-CONTENT
    content_div = soup.find("div", class_="pages_content")
    content = ""
    attachments = []
    if content_div:
        # 只取直接子元素<p>，避免嵌套重复
        parts = []
        for child in content_div.find_all(["p", "div", "table"], recursive=False):
            text = child.get_text("", strip=True)
            if text and len(text) > 3:
                parts.append(text)
            # 图片附件
            for img in child.find_all("img"):
                src = img.get("src", "")
                if src:
                    attachments.append({"name": "正文附图", "url": urljoin(BASE, src)})
        if not parts:
            # fallback: 全部内容
            parts = [content_div.get_text("", strip=True)]
        content = "\n\n".join(parts) if parts else ""

    if not content and attachments:
        content = f"[本公告为图片格式，附件{len(attachments)}张]"
    elif attachments:
        content += f"\n\n[公告附图{len(attachments)}张]"

    return full_title, content, date_str, attachments


def import_to_db(record):
    try:
        db = sqlite3.connect(DB_PATH, timeout=10)
        title = (record.get("title") or "")[:500]
        page_url = (record.get("page_url") or "")[:1000]
        content = record.get("content") or ""
        publish_date = (record.get("publish_date") or "")[:20]
        site_name = (record.get("site_name") or "unknown")[:100]
        attachments_str = json.dumps(record.get("attachments") or [], ensure_ascii=False)

        old_rowids = db.execute("SELECT rowid FROM gov_raw WHERE page_url = ?", (page_url,)).fetchall()
        for (rid,) in old_rowids:
            db.execute("DELETE FROM gov_search WHERE rowid = ?", (rid,))

        db.execute(
            "INSERT OR REPLACE INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?, ?, ?, ?, ?, ?, 'synced', ?)",
            (title, page_url, content, publish_date, site_name, page_url, attachments_str)
        )
        new_rowid = db.execute("SELECT last_insert_rowid()").fetchone()[0]
        summary = content[:500] if content else title[:500]
        db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                   (new_rowid, title, site_name, summary))
        db.commit()
        db.close()
        log(f"  ✅ {title[:30]}")
        return True
    except Exception as e:
        log(f"  [ERR] DB import failed: {e}")
        return False


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=5,
                        help="爬取前N页（每页~20条，默认5页=~100条）")
    args = parser.parse_args()

    total_pages = min(args.pages, 22)  # max 22 pages available
    count = 0

    for page in range(1, total_pages + 1):
        page_url = f"{LIST_PREFIX}.html" if page == 1 else f"{LIST_PREFIX}_{page}.html"
        log(f"📄 列表[page={page}]: .../{page_url.split('/')[-1]}")
        html = fetch(page_url)
        if not html:
            break
        items = parse_list_page(html)
        if not items:
            log(f"  → 无更多数据")
            break
        log(f"  → {len(items)} 条")

        for idx, item in enumerate(items, 1):
            url = item["url"]
            list_title = item["title"]
            list_date = item["date"]
            # 外链文章（微信公众号/外部平台）跳过不入库
            if "xiangfen.gov.cn" not in url:
                log(f"  ({idx}/{len(items)}) 外链跳过: {url[:60]}")
                continue
            log(f"  ({idx}/{len(items)}) 详情: {url.split('/')[-1]}")
            full_title, content, detail_date, attachments = fetch_detail(url, list_title)

            record = {
                "title": full_title,
                "page_url": url,
                "publish_date": detail_date or list_date,
                "content": content,
                "attachments": attachments,
                "site_name": SITE,
                "column": COLUMN,
                "province": PROVINCE,
            }
            ok = import_to_db(record)
            if ok:
                count += 1
            time.sleep(0.5)

    log(f"\n✅ {SITE} 爬取完成，共入库 {count} 条")


if __name__ == "__main__":
    main()
