#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
灌南县人民政府 - 政府信息公开
http://xxgk.guannan.gov.cn/list.html
Custom CMS
"""
import json, re, sys, time, requests, sqlite3, os
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime

DB_PATH = "/root/search.db"
SITE_NAME = "灌南县人民政府-政府信息公开"
CATEGORY = "政府"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
SESSION = requests.Session()
SESSION.headers.update(HEADERS)

INSERT_SQL = """INSERT OR IGNORE INTO gov_raw 
    (site_name, page_url, title, publish_date, summary, content, category, attachments)
    VALUES (?, ?, ?, ?, '', ?, ?, ?)"""


def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=5000")
    conn.execute("PRAGMA journal_mode=WAL")
    return conn


def fetch_list(page=1):
    """Fetch list page, return list of (title, url)."""
    if page == 1:
        url = "http://xxgk.guannan.gov.cn/list.html"
    else:
        url = f"http://xxgk.guannan.gov.cn/list/0-0-0-0-{page}.html"
    
    resp = SESSION.get(url, timeout=30)
    resp.encoding = "utf-8"
    soup = BeautifulSoup(resp.text, "html.parser")
    
    items = []
    for a in soup.find_all("a", href=re.compile(r"/news/show-\d+\.html")):
        href = a.get("href", "")
        title = a.get_text(strip=True)
        if not title or not href:
            continue
        full_url = urljoin("http://xxgk.guannan.gov.cn", href)
        items.append((title, full_url))
    
    return items


def fetch_detail(url):
    """Fetch detail page, return (title, date, content, attachments_json)."""
    resp = SESSION.get(url, timeout=30)
    resp.encoding = "utf-8"
    soup = BeautifulSoup(resp.text, "html.parser")
    
    # Title from h3.title
    h3 = soup.find("h3", class_="title")
    title = h3.get_text(strip=True) if h3 else ""
    if not title:
        mt = soup.find("meta", {"name": "ArticleTitle"})
        if mt and mt.get("content"):
            title = mt.get("content").strip()
    
    # Date from meta
    date_str = ""
    m = soup.find("meta", {"name": "PubDate"})
    if m and m.get("content"):
        raw = m.get("content").strip()
        try:
            dt = datetime.strptime(raw, "%Y/%m/%d %H:%M:%S")
            date_str = dt.strftime("%Y-%m-%d")
        except:
            date_str = raw
    
    # Content from div#zoom
    content_parts = []
    attachments = []
    
    zoom = soup.find("div", id="zoom")
    if not zoom:
        zoom = soup.find("div", class_="entry")
    
    if zoom:
        # Collect <p> text preserving order
        for p in zoom.find_all("p"):
            txt = p.get_text(strip=True)
            if txt:
                content_parts.append(txt)
        
        # Tables rendered as Markdown
        for table in zoom.find_all("table"):
            rows = []
            for tr in table.find_all("tr"):
                cells = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
                if cells:
                    rows.append("| " + " | ".join(cells) + " |")
            if rows:
                content_parts.append("\n".join(rows))
        
        # Attachments
        for a in zoom.find_all("a"):
            href = a.get("href", "")
            if any(href.lower().endswith(ext) for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar"]):
                fname = a.get_text(strip=True) or os.path.basename(href)
                furl = urljoin("http://xxgk.guannan.gov.cn", href)
                attachments.append({"name": fname, "url": furl})
    
    content = "\n\n".join(content_parts)
    
    # Empty content fallback
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>'
        if attachments:
            for att in attachments:
                content += f"\n📎 [{att['name']}]({att['url']})"
    
    return title, date_str, content, json.dumps(attachments, ensure_ascii=False) if attachments else ""


def crawl(max_pages=5):
    all_items = []
    for page in range(1, max_pages + 1):
        print(f"Fetching list page {page}...")
        items = fetch_list(page)
        if not items:
            print(f"  No items on page {page}")
            break
        print(f"  Found {len(items)} items")
        all_items.extend(items)
        time.sleep(1)
    
    print(f"\nTotal items: {len(all_items)}")
    
    conn = init_db()
    inserted = 0
    skipped = 0
    errors = 0
    
    for title, url in all_items:
        try:
            cur = conn.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if cur.fetchone():
                skipped += 1
                continue
            
            d_title, d_date, content, attachments = fetch_detail(url)
            
            conn.execute(INSERT_SQL, (
                SITE_NAME, url, d_title or title, d_date or "",
                content, CATEGORY, attachments
            ))
            conn.commit()
            inserted += 1
            print(f"  [{inserted}] {d_title or title}")
            time.sleep(0.5)
            
        except Exception as e:
            errors += 1
            print(f"  ERROR: {url} - {e}")
            time.sleep(2)
    
    conn.close()
    print(f"\nDone. Inserted: {inserted}, Skipped: {skipped}, Errors: {errors}")
    return inserted


if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--test", action="store_true")
    parser.add_argument("--max-pages", type=int, default=5)
    args = parser.parse_args()
    
    if args.test:
        print("=== TEST MODE ===")
        items = fetch_list(1)
        for title, url in items[:3]:
            print(f"\nTITLE: {title}")
            print(f"URL: {url}")
            d_title, d_date, content, attachments = fetch_detail(url)
            print(f"DETAIL: {d_title}")
            print(f"DATE: {d_date}")
            print(f"CONTENT: {content[:300]}")
            print(f"ATTACH: {attachments}")
        print(f"\nTotal on page 1: {len(items)}")
    else:
        crawl(max_pages=args.max_pages)
