#!/usr/bin/env python3
"""
抚顺市人民政府 - 通知公告 爬虫
URL: https://www.fushun.gov.cn/ywdt/001007/moreinfo.html
CMS: Webbuilder 5.0, 20条/页, 共496条(25页)
仅爬第1页SSR（API不可用）
"""
import requests, json, re, time, os, sys
from bs4 import BeautifulSoup

BASE_URL = "https://www.fushun.gov.cn"
LIST_URL = f"{BASE_URL}/ywdt/001007/moreinfo.html"
MAX_PAGES = 5
SITE_NAME = "抚顺市人民政府"
DB_PATH = "/root/search.db"
GROUP = "公示公告"

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
session = requests.Session()
session.headers.update(HEADERS)
requests.packages.urllib3.disable_warnings()


def extract_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")
    article_div = soup.find("div", class_="ewb-article")
    
    title = ""
    if article_div:
        h3 = article_div.find("h3")
        if h3:
            title = h3.get_text(strip=True)
    if not title:
        mt = soup.find("meta", attrs={"name": "ArticleTitle"})
        title = mt.get("content", "") if mt else ""
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)

    pub_date = ""
    dt = soup.find("meta", attrs={"name": "PubDate"})
    if dt:
        pub_date = dt.get("content", "")[:10]
    if not pub_date:
        dp = re.search(r'发布日期[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})', html)
        if dp:
            pub_date = dp.group(1)
    if not pub_date:
        dp = re.search(r'发布时间[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})', html)
        if dp:
            pub_date = dp.group(1)

    content_parts = []
    attachments = []
    info_div = soup.find("div", class_="ewb-article-info")
    if not info_div and article_div:
        info_div = article_div

    if info_div:
        for table in info_div.find_all("table"):
            content_parts.append(str(table))
        for p in info_div.find_all(["p"]):
            txt = p.get_text(strip=True)
            if txt and len(txt) > 5:
                if not any(kw in txt for kw in ["责任编辑", "初审", "复审", "终审", "[纠错]"]):
                    content_parts.append(txt)
        for a in soup.find_all("a", href=True):
            h = a["href"]
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', h, re.I):
                if not h.startswith("http"):
                    h = BASE_URL + h
                an = a.get_text(strip=True) or h.split("/")[-1]
                attachments.append({"name": an, "url": h})
                content_parts.append(f"[{an}]({h})")

    content = "\n\n".join(content_parts) if content_parts else ""
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>'
    return title, pub_date, content, attachments


def main():
    print(f"=== {SITE_NAME} - 全量爬取 ===")
    print(f"Fetching page 1: {LIST_URL}")
    r = session.get(LIST_URL, timeout=15, verify=False)
    r.encoding = "utf-8"
    soup = BeautifulSoup(r.text, "html.parser")

    items = []
    for li in soup.select("li.sec-right-item"):
        a = li.find("a", class_="sec-right-name")
        t = li.find("div", class_="sec-right-time")
        if a and a.get("href"):
            title = a.get("title", "") or a.get_text(strip=True)
            href = a["href"]
            if not href.startswith("http"):
                href = BASE_URL + href
            date = t.get_text(strip=True) if t else ""
            items.append({"title": title, "url": href, "date": date})
    print(f"  -> {len(items)} items")

    import sqlite3
    existing = set()
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        existing = set(r[0] for r in conn.execute("SELECT page_url FROM gov_raw WHERE page_url IS NOT NULL").fetchall())
        conn.close()
    except:
        pass
    print(f"  Existing in DB: {len(existing)}")

    new_items = [it for it in items if it["url"] not in existing]
    print(f"  New to insert: {len(new_items)}")

    # In incremental mode, only process new items
    process_items = items
    if "--incremental" in sys.argv:
        process_items = new_items
        print(f"  Incremental mode: processing {len(process_items)} new items")
        if not process_items:
            print("  Nothing new to crawl")
            print("Done!")
            return

    results = []
    for i, item in enumerate(process_items):
        url = item["url"]
        print(f"  [{i+1}/{len(process_items)}] {item['title'][:40]}... ", end="", flush=True)
        try:
            r = session.get(url, timeout=15, verify=False)
            r.encoding = "utf-8"
            title, pub_date, content, attachments = extract_detail(r.text, url)
            results.append({
                "title": title or item["title"],
                "url": url,
                "date": pub_date or item["date"],
                "content": content,
                "site_name": SITE_NAME,
                "attachments": attachments,
                "summary": title or item["title"],
            })
            print("+1")
        except Exception as e:
            print(f"ERROR: {e}")
        time.sleep(0.3)

    if results:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        imported = 0
        for item in results:
            try:
                conn.execute(
                    """INSERT OR IGNORE INTO gov_raw 
                       (page_url, title, publish_date, content, site_name, summary, attachments, category)
                       VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
                    (item["url"], item["title"], item["date"], item["content"],
                     SITE_NAME, item["title"],
                     json.dumps(item["attachments"], ensure_ascii=False), "通知公告")
                )
                imported += 1
            except Exception as e:
                print(f"  Error: {e}")
        conn.commit()
        # Count actual new records for this site
        count = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=? AND page_url IN ({})".format(
            ",".join("?" for _ in results)), [SITE_NAME] + [it["url"] for it in results]).fetchone()[0]
        conn.close()
        print(f"\nInserted/confirmed {imported} records ({count} total for this site)")
    print("Done!")


if __name__ == "__main__":
    main()
