#!/usr/bin/env python3
"""
鹤山市人民政府 - 多个子站统一爬虫
CMS: 汉林Hanweb (静态HTML分页)
详情: 标题<meta ArticleTitle> / 正文div#zoom .TRS_UEDITOR
用法: python3 crawl_heshan.py [site_key]
      不传参数默认爬环评信息，传具体key爬对应子站
"""

import os
import re
import sys
import time
import requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

BASE_URL = "https://www.heshan.gov.cn"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

# Site configuration: key => (site_name, list_path, category, max_pages_or_0_for_auto)
SITE_MAP = {
    "环评信息": ("heshan.gov.cn-环评信息", "/zwgk/zdlyxxgk/hjbhxxgk/jsxmhjyxpjxx", "hjxx", 20),
    "古劳镇-通知公告": ("古劳镇-通知公告", "/zwgk/xxgk/hssglz/qt/tzgg", "通知公告", 0),
    "古劳镇-其他": ("古劳镇-其他信息", "/zwgk/xxgk/hssglz/qt", "其他", 0),
    "龙口镇-部门文件": ("龙口镇-部门文件", "/zwgk/xxgk/hsslkz/bmwj", "部门文件", 0),
    "龙口镇-工作动态通知公告": ("龙口镇-工作动态通知公告", "/zwgk/xxgk/hsslkz/gzdt/tzgg", "通知公告", 0),
}

session = requests.Session()
session.headers.update(HEADERS)

def fetch(url):
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [ERR] fetch failed: {url} - {e}")
        return None

def get_total_pages(list_path):
    """Auto-detect total pages from first page"""
    url = f"{BASE_URL}{list_path}/index.html"
    html = fetch(url)
    if not html:
        return 1
    
    soup = BeautifulSoup(html, 'html.parser')
    max_page = 1
    
    # Find "最后一页" link
    for a in soup.find_all('a', href=True):
        txt = a.get_text(strip=True)
        href = a['href']
        if '最后一页' in txt or '末页' in txt:
            m = re.search(r'index_(\d+)\.html', href)
            if m:
                max_page = int(m.group(1))
                print(f"  Last page link: {href} => {max_page} pages")
                return max_page
    
    # Find largest numbered link
    for a in soup.find_all('a', href=True):
        href = a['href']
        m = re.search(r'index_(\d+)\.html', href)
        if m:
            p = int(m.group(1))
            if p > max_page:
                max_page = p
    
    if max_page > 1:
        print(f"  Max page link: index_{max_page}.html")
    
    return max_page

def parse_list(html, list_path):
    """Parse list page, return list of (url, title)"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    
    # Match content/post_ links within this list path
    for a_tag in soup.find_all("a", href=True):
        href = a_tag['href'].strip()
        title = a_tag.get("title", "").strip()
        if not title:
            title = a_tag.get_text(strip=True)
        if not title or len(title) < 5:
            continue
        if 'content/post_' not in href:
            continue
        
        if not href.startswith("http"):
            if href.startswith("/"):
                href = BASE_URL + href
            else:
                href = BASE_URL + list_path + "/" + href.lstrip('./')
        
        items.append((href, title))
    
    # Deduplicate
    seen = set()
    unique = []
    for url, title in items:
        if url not in seen:
            seen.add(url)
            unique.append((url, title))
    return unique

def parse_detail(html, url):
    """Parse detail page, return (title, publish_date, content_html)"""
    soup = BeautifulSoup(html, 'html.parser')
    
    title = None
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    
    if not title:
        title_el = soup.select_one("div.title span")
        if title_el:
            title = title_el.get_text(strip=True)
    
    if not title:
        print(f"  [ERR] no title for {url}")
        return None, None, None
    
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_date["content"])
        if m:
            pub_date = m.group(1)
    
    content_div = soup.select_one("div#zoom")
    if content_div:
        for tag in content_div.find_all(['script', 'style']):
            tag.decompose()
        content_html = str(content_div).strip()
    else:
        print(f"  [WARN] no content div for {url}")
        content_html = ""
    
    return title, pub_date, content_html

def main():
    import sqlite3
    
    # Determine which site to crawl
    if len(sys.argv) > 1:
        site_key = sys.argv[1]
    else:
        site_key = "环评信息"  # Default for backward compatibility
    
    if site_key not in SITE_MAP:
        print(f"Error: unknown site key '{site_key}'")
        print(f"Available: {', '.join(SITE_MAP.keys())}")
        sys.exit(1)
    
    SITE_NAME, LIST_PATH, CATEGORY, MAX_PAGES = SITE_MAP[site_key]
    
    SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
    
    print(f"Site: {SITE_NAME}")
    print(f"List: {LIST_PATH}")
    print(f"Cutoff: {CUTOFF}")
    
    # Auto-detect page count
    total_pages = get_total_pages(LIST_PATH)
    if MAX_PAGES > 0 and total_pages < MAX_PAGES:
        total_pages = MAX_PAGES  # Use hardcoded max if auto-detect found fewer
    print(f"Total pages: {total_pages}")
    
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    
    # Check existing
    existing = set()
    for row in cur.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
        existing.add(row[0])
    print(f"Existing items: {len(existing)}")
    
    total_new = 0
    total_skip = 0
    
    for page_num in range(1, total_pages + 1):
        if page_num == 1:
            page_url = f"{BASE_URL}{LIST_PATH}/index.html"
        else:
            page_url = f"{BASE_URL}{LIST_PATH}/index_{page_num}.html"
        
        print(f"\n--- Page {page_num}: {page_url} ---")
        html = fetch(page_url)
        if not html:
            print(f"  [SKIP] fetch failed")
            continue
        
        items = parse_list(html, LIST_PATH)
        if not items:
            print(f"  [SKIP] no items (stopping)")
            if page_num == 1:
                print("[FATAL] Page 1 empty, aborting")
                return
            break
        
        print(f"  Found {len(items)} items")
        
        page_new = 0
        for url, list_title in items:
            if url in existing:
                continue
            
            detail_html = fetch(url)
            if not detail_html:
                print(f"  [SKIP] detail failed: {list_title[:40]}")
                continue
            
            title, date_str, content = parse_detail(detail_html, url)
            if not title:
                continue
            
            if date_str and date_str < CUTOFF:
                total_skip += 1
                continue
            
            date_rank = 0
            if date_str:
                try:
                    dt = datetime.strptime(date_str, "%Y-%m-%d")
                    date_rank = int(dt.timestamp())
                except:
                    date_rank = int(date_str.replace("-", ""))
            
            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(site_name, title, page_url, publish_date, content, date_rank, category) "
                    "VALUES (?, ?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, title, url, date_str, content, date_rank, CATEGORY)
                )
                if cur.rowcount > 0:
                    page_new += 1
                    total_new += 1
                    existing.add(url)
            except Exception as e:
                print(f"  [ERR] insert: {e}")
            
            if page_new % 10 == 0:
                conn.commit()
        
        conn.commit()
        print(f"  Page {page_num}: +{page_new} new (total {total_new}, skip {total_skip})")
        time.sleep(0.3)
    
    print(f"\n{'='*50}")
    print(f"Final: {total_new} new, {total_skip} skipped (超3年) for {SITE_NAME}")
    
    if total_new > 0:
        print("FTS rebuild...")
        try:
            cur.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
            conn.commit()
            print("  FTS rebuilt OK")
        except Exception as e:
            print(f"  [WARN] FTS rebuild: {e}")
    
    conn.close()
    print("Done.")

if __name__ == "__main__":
    main()
