"""Crawl rhs.np.gov.cn - 南平荣华山产业组团通知公告"""
import requests
import sqlite3
import re
import warnings
from bs4 import BeautifulSoup
warnings.filterwarnings('ignore')

BASE = "https://rhs.np.gov.cn"
DB_PATH = "/root/search.db"
SITE_NAME = "南平荣华山产业组团-通知公告"
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def extract_clean_text(soup):
    cont = soup.select_one('div.content')
    if not cont:
        return ""
    parts = []
    for el in cont.descendants:
        if el.name in ('p', 'div', 'span'):
            txt = el.get_text(strip=True)
            if txt:
                parts.append(txt)
        elif el.name == 'table':
            parts.append(str(el))
    return '\n\n'.join(parts) if parts else cont.get_text(separator='\n\n', strip=True)

def crawl():
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA busy_timeout=30000")
    
    total_new = 0
    total_pages = 7
    
    for page in range(1, total_pages + 1):
        if page == 1:
            list_url = f"{BASE}/cms/html/nprhscyzt/tzgg/index.html"
        else:
            list_url = f"{BASE}/cms/sitemanage/index.shtml?siteId=100390700393700000&page={page}"
        
        try:
            resp = requests.get(list_url, headers=headers, timeout=30, verify=False)
            resp.encoding = 'utf-8'
        except Exception as e:
            print(f"Page {page} ERROR: {e}")
            continue
        
        soup = BeautifulSoup(resp.text, 'html.parser')
        items = soup.select('ul#resources li')
        
        if not items:
            print(f"Page {page}: no items found")
            continue
        
        for li in items:
            a = li.find('a')
            if not a:
                continue
            href = a.get('href', '')
            title = a.get_text(strip=True)
            if not href or not title:
                continue
            
            if href.startswith('/'):
                page_url = f"{BASE}{href}"
            else:
                page_url = f"{BASE}/{href}" if not href.startswith('http') else href
            page_url = page_url.split('?')[0]
            
            # Date from list
            date_span = li.find('span', class_='list-time')
            list_date = date_span.get_text(strip=True) if date_span else ""
            
            # Check exists
            exists = db.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (page_url,)).fetchone()
            if exists:
                print(f"  SKIP: {title[:30]}")
                continue
            
            # Fetch detail
            try:
                dresp = requests.get(page_url, headers=headers, timeout=30, verify=False)
                dresp.encoding = 'utf-8'
                dsoup = BeautifulSoup(dresp.text, 'html.parser')
            except Exception as e:
                print(f"  FETCH ERR: {title[:30]} - {e}")
                continue
            
            # Title from detail
            title_div = dsoup.select_one('div.main div.title')
            page_title = title_div.get_text(strip=True) if title_div else title
            
            # Date from detail
            date = list_date
            bt = dsoup.select_one('div.bt-a')
            if bt:
                m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', bt.get_text())
                if m:
                    date = m.group(1).replace('/', '-')
            
            # Content
            content = extract_clean_text(dsoup)
            if not content:
                content_div = dsoup.select_one('div.content')
                if content_div:
                    content = content_div.get_text(separator='\n\n', strip=True)
            if not content:
                content = "(无内容)"
            
            summary = content[:500]
            
            try:
                db.execute(
                    "INSERT INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, page_url, page_title, content, date, summary)
                )
                db.commit()
                total_new += 1
                print(f"  [{page}] {page_title[:35]:.35s} | {date}")
            except sqlite3.IntegrityError:
                print(f"  DUP: {page_title[:30]}")
                db.rollback()
            except Exception as e:
                db.rollback()
                print(f"  DB ERR: {page_title[:30]} - {e}")
    
    db.close()
    print(f"\n✅ Done. Total new: {total_new}")

if __name__ == "__main__":
    crawl()
