#!/usr/bin/env python3
"""
龙南经济技术开发区 - 行政审批公示公告
URL: https://lnjkq.ganzhou.gov.cn/lnjkq/xzspgsgg/list.shtml
CMS: TRS with createPageHTML, detail: div.TRS_Editor
"""

import requests, re, sys, os, json
from bs4 import BeautifulSoup
from datetime import datetime

DB_PATH = "/root/search.db"
SITE_NAME = "龙南经开区-行政审批公示公告"
BASE_URL = "https://lnjkq.ganzhou.gov.cn"
LIST_URL = BASE_URL + "/lnjkq/xzspgsgg/list.shtml"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

import argparse
parser = argparse.ArgumentParser()
parser.add_argument("--pages", type=int, default=0, help="Pages to crawl (0 = all)")
parser.add_argument("--verbose", "-v", action="store_true")
args = parser.parse_args()
MAX_PAGES = args.pages if args.pages > 0 else 99

def log(msg):
    if args.verbose:
        print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}")

# ─── Fetch list page ───
def fetch_page(page_num):
    if page_num == 1:
        url = LIST_URL
    else:
        url = f"{BASE_URL}/lnjkq/xzspgsgg/list_{page_num}.shtml"
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            log(f"Page {page_num} returned {r.status_code}")
            return None
        return r.text
    except Exception as e:
        log(f"Error page {page_num}: {e}")
        return None

def parse_list(html):
    """Parse list page, return list of {title_from_detail will be filled, href, date}"""
    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_="list_newsrl")
    if not ul:
        return [], 0
    items = []
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if not href or "xzspgsgg" not in href:
            continue
        if href.startswith("/"):
            href = BASE_URL + href
        span = li.find("span")
        date = span.get_text(strip=True) if span else ""
        items.append({"href": href, "date": date})
    return items, 0

# ─── Fetch detail ───
def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None, "", ""
    except:
        return None, "", ""

    soup = BeautifulSoup(r.text, "html.parser")

    # Title from ArticleTitle meta
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    title = meta["content"].strip() if meta else ""

    # Content from div.TRS_Editor
    content = ""
    editor = soup.find("div", class_="TRS_Editor")
    if editor:
        parts = []
        for child in editor.children:
            if not hasattr(child, "name"):
                continue
            if child.name == "table":
                parts.append(str(child))
            elif child.name in ("p", "div"):
                txt = child.get_text(strip=True)
                if txt and len(txt) > 2:
                    parts.append(txt)
            elif child.name == "img":
                src = child.get("src", "")
                if src:
                    parts.append(f"![image]({src})")
        content = "\n\n".join(parts)
    else:
        # Fallback to div.content
        cdiv = soup.find("div", class_="content")
        if cdiv:
            content = cdiv.get_text(separator="\n", strip=True)

    content = re.sub(r"[ \t]+", " ", content).strip()

    # Attachments
    attachments = []
    for a_tag in soup.find_all("a", href=True):
        h = a_tag["href"]
        ext = os.path.splitext(h)[1].lower()
        if ext in (".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar"):
            if h.startswith("/"):
                h = BASE_URL + h
            elif not h.startswith("http"):
                h = BASE_URL + "/" + h.lstrip("/")
            name = a_tag.get_text(strip=True) or os.path.basename(h)
            attachments.append({"name": name, "url": h})
            content += f"\n[附件] {name}: {h}"

    return title, content, json.dumps(attachments, ensure_ascii=False)

# ─── Main ───
all_items = []
seen_hrefs = set()

log("Fetching list pages...")
for page_num in range(1, MAX_PAGES + 1):
    html = fetch_page(page_num)
    if html is None:
        break
    items, _ = parse_list(html)
    if not items:
        if page_num > 1:
            break  # No more pages
        log(f"  Page {page_num}: empty, skipping")
        continue
    new_count = 0
    for item in items:
        if item["href"] in seen_hrefs:
            continue
        seen_hrefs.add(item["href"])
        all_items.append(item)
        new_count += 1
    log(f"  Page {page_num}: +{new_count} (total: {len(all_items)})")
    if new_count == 0:
        break

print(f"\n列表合计: {len(all_items)} 条")

if not all_items:
    print("No items to process.")
    sys.exit(0)

# Insert DB
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
try:
    from crawler_lib import push_to_searchdb, get_connection
except ImportError:
    import sqlite3
    def push_to_searchdb(items, site_name):
        conn = sqlite3.connect(DB_PATH, timeout=30)
        conn.execute("PRAGMA journal_mode=WAL")
        conn.execute("PRAGMA busy_timeout=30000")
        c = conn.cursor()
        new = skip = 0
        for item in items:
            try:
                c.execute(
                    """INSERT OR IGNORE INTO gov_raw 
                    (site_name, source_url, page_url, title, publish_date, date_rank, summary, content, attachments)
                    VALUES (?,?,?,?,?,?,?,?,?)""",
                    (site_name, item.get("url",""), item.get("url",""),
                     item.get("title",""), item.get("pub_date",""),
                     item.get("date_rank",0), item.get("summary",""),
                     item.get("content",""), item.get("attachments","[]"))
                )
                if c.lastrowid and c.lastrowid > 0:
                    new += 1
                else:
                    skip += 1
            except:
                skip += 1
        conn.commit()
        conn.close()
        return new, skip
    def get_connection():
        conn = sqlite3.connect(DB_PATH, timeout=30)
        conn.execute("PRAGMA journal_mode=WAL")
        conn.execute("PRAGMA busy_timeout=30000")
        return conn

print(f"共 {len(all_items)} 条，开始抓取详情...")
items_for_db = []

for idx, item in enumerate(all_items):
    log(f"  [{idx+1}/{len(all_items)}] {item['href'][-40:]}...")
    title, content, attachments = fetch_detail(item["href"])

    pub_date = item["date"]
    date_rank = 0
    if pub_date:
        try:
            dt = datetime.strptime(pub_date[:10], "%Y-%m-%d")
            date_rank = int(dt.strftime("%Y%m%d"))
        except:
            pass

    summary = (BeautifulSoup(content, "html.parser").get_text(strip=True) if "<" in content else content)[:500]

    items_for_db.append({
        "title": title,
        "url": item["href"],
        "content": content,
        "pub_date": pub_date,
        "date_rank": date_rank,
        "summary": summary,
        "source_url": item["href"],
        "attachments": attachments,
        "page_url": item["href"],
    })

print(f"\n入库中 ({len(items_for_db)} 条)...")
new, skip = push_to_searchdb(items_for_db, SITE_NAME)
print(f"入库完成: new={new}, skip={skip}")

conn = get_connection()
count = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
conn.close()
print(f"DB中 {SITE_NAME} 共 {count} 条")
