#!/usr/bin/env python3
"""Crawler for 国家级遂宁经济技术开发区 - 公示公告
   https://snjkq.suining.gov.cn/xinwen/list/0508cc5f620b47cfad06549fa5fe3a95.html
   CMS: Custom (Bootstrap), pagination: ?page=N
"""

import requests
import sqlite3
import re
import os
import sys
from bs4 import BeautifulSoup
from datetime import datetime

# ---------- CONFIG ----------
BASE_LIST = "/xinwen/list/0508cc5f620b47cfad06549fa5fe3a95.html"
DOMAIN = "https://snjkq.suining.gov.cn"
MAX_PAGES = 5  # Top 5 pages for daily runs
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": f"{DOMAIN}/",
}
DB_PATH = "/root/search.db"
CRAWLER_NAME = "snjkq"

# --- Check if incremental mode ---
incremental = "--incremental" in sys.argv

s = requests.Session()
s.headers.update(HEADERS)
s.verify = False


def get_list_page(page_num):
    url = f"{DOMAIN}{BASE_LIST}?page={page_num}"
    r = s.get(url, timeout=30)
    r.encoding = "utf-8"
    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    lis = soup.select("ul.news-list li")
    for li in lis:
        a = li.find("a")
        date_span = li.find("span")
        if not a or not date_span:
            continue
        title = a.get("title", "").strip()
        href = a.get("href", "").strip()
        date_str = date_span.get_text(strip=True)
        if href and not href.startswith("http"):
            href = DOMAIN + href
        items.append({"title": title, "url": href, "date": date_str})
    return items


def get_detail(url):
    try:
        r = s.get(url, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] Fetch detail failed: {e}")
        return "", "", [], ""
    soup = BeautifulSoup(r.text, "html.parser")

    # Title
    title_el = soup.select_one("div.h2.fw-bold")
    title = title_el.get_text(strip=True) if title_el else ""

    # Publish date from article-info area
    date_str = ""
    info_el = soup.select_one(".article-info")
    if info_el:
        spans = info_el.find_all("span")
        for span in spans:
            txt = span.get_text(strip=True)
            if "发布时间" in txt or re.search(r"\d{4}-\d{2}-\d{2}", txt):
                m = re.search(r"(\d{4}-\d{2}-\d{2})", txt)
                if m:
                    date_str = m.group(1)
                    break

    # Content
    article = soup.select_one("article.article-content")
    if not article:
        article = soup.select_one("div.xqing-web-box")
    content_html = ""
    attachments = []
    if article:
        # Remove scripts/styles
        for tag in article.find_all(["script", "style"]):
            tag.decompose()
        # Get content as HTML, preserving <p>, <table>, etc.
        content_html = str(article)
        # Clean up whitespace issues
        content_html = re.sub(r' *<table', '<table', content_html)
        content_html = re.sub(r'>\s+<', '>\n<', content_html)

        # Extract attachments
        file_div = article.select_one("div.xqing-web-file")
        if file_div:
            for a_tag in file_div.find_all("a"):
                link = a_tag.get("href", "").strip()
                fname = a_tag.get_text(strip=True)
                if link and not link.startswith("http"):
                    link = DOMAIN + link
                attachments.append({"name": fname, "url": link})

    return title, content_html, attachments, date_str


def crawl():
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    c.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT, url TEXT UNIQUE, content TEXT,
        summary TEXT, publish_date TEXT,
        crawl_time TEXT, source TEXT, site_name TEXT,
        category TEXT, error TEXT
    )""")
    conn.commit()

    count_new = 0
    count_skip = 0
    count_error = 0

    # Determine how many pages to crawl
    max_pages = MAX_PAGES if incremental else 999

    for page in range(1, max_pages + 1):
        print(f"--- Page {page} ---")
        items = get_list_page(page)
        if not items:
            print(f"  No more items, stopping.")
            break
        print(f"  Found {len(items)} items")
        for item in items:
            page_url = item["url"]
            list_title = item["title"]
            list_date = item["date"]

            # Check if exists by page_url
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if c.fetchone():
                count_skip += 1
                continue

            print(f"  Fetching: {list_title[:40]}...")
            title, content_html, attachments, detail_date = get_detail(page_url)

            if not content_html or len(content_html.strip()) < 50:
                print(f"  [WARN] Content too short, using list data")
                content_html = f"<p>{list_title}</p>"

            # Use detail date if available, else list date
            pub_date = detail_date or list_date

            # Build summary
            clean_text = re.sub(r'<[^>]+>', ' ', content_html)
            clean_text = re.sub(r'\s+', ' ', clean_text).strip()
            summary = clean_text[:300]

            # Build content with attachments
            full_content = content_html
            if attachments:
                full_content += '\n\n<p><strong>附件：</strong></p>\n'
                for att in attachments:
                    full_content += f'<p><a href="{att["url"]}">{att["name"]}</a></p>\n'

            try:
                c.execute("""INSERT OR IGNORE INTO gov_raw
                    (title, page_url, source_url, content, summary, publish_date, site_name, category, status)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                    (title or list_title, page_url, DOMAIN, full_content, summary,
                     pub_date, "遂宁经济技术开发区", "公示公告", "published"))
                if c.rowcount > 0:
                    count_new += 1
                    conn.commit()  # Commit after each item
                    print(f"    [+] Imported")
                else:
                    count_skip += 1
            except Exception as e:
                print(f"    [ERROR] DB insert: {e}")
                count_error += 1
                conn.rollback()

    conn.close()
    print(f"\n===== Done =====")
    print(f"New: {count_new}, Skip: {count_skip}, Error: {count_error}")


if __name__ == "__main__":
    crawl()
