"""Crawl jzsfyq.luan.gov.cn - Lonsun CMS"""
import requests
import sqlite3
import re
from bs4 import BeautifulSoup
import warnings
warnings.filterwarnings('ignore')

BASE_URL = "https://jzsfyq.luan.gov.cn"
LIST_URL = f"{BASE_URL}/zwzx/tzgg/index.html"
DB_PATH = "/root/search.db"
SITE_NAME = "安徽六安金安经济开发区-公示公告"

headers = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
}

def extract_clean_text(soup):
    """Extract content from article page preserving HTML structure"""
    cont = soup.select_one("div#J_content.j-fontContent.newscontnet")
    if not cont:
        cont = soup.select_one("div.newscontnet")
    if not cont:
        return ""
    
    result_parts = []
    for el in cont.descendants:
        if el.name == 'p' or el.name == 'div':
            text = el.get_text(strip=True)
            if text:
                result_parts.append(text)
        elif el.name == 'table':
            result_parts.append(str(el))
    
    if not result_parts:
        return cont.get_text(separator='\n\n', strip=True)
    
    return '\n\n'.join(result_parts)


def crawl():
    resp = requests.get(LIST_URL, headers=headers, timeout=30, verify=False)
    resp.encoding = 'utf-8'
    soup = BeautifulSoup(resp.text, 'html.parser')
    
    items = soup.select('ul.doc_list li a.left')
    print(f"Found {len(items)} items on page")
    
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA busy_timeout=30000")
    
    new_count = 0
    for a in items:
        href = a.get('href', '')
        title = a.get('title', '') or a.get_text(strip=True)
        if not href or not title:
            continue
        if not href.startswith('http'):
            href = BASE_URL + href if href.startswith('/') else f"{BASE_URL}/zwzx/tzgg/{href}"
        
        page_url = href.split('?')[0]
        exists = db.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (page_url,)).fetchone()
        if exists:
            print(f"  SKIP (exists): {title[:30]}")
            continue
        
        try:
            dresp = requests.get(page_url, headers=headers, timeout=30, verify=False)
            dresp.encoding = 'utf-8'
            dsoup = BeautifulSoup(dresp.text, 'html.parser')
        except Exception as e:
            print(f"  ERROR fetching {page_url}: {e}")
            continue
        
        # Title
        h1 = dsoup.select_one('h1.newstitle')
        page_title = h1.get_text(strip=True) if h1 else title
        
        # Date
        date = ""
        info_div = dsoup.select_one('div.newsinfo')
        if info_div:
            txt = info_div.get_text()
            m = re.search(r'发布时间[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})', txt)
            if m:
                date = m.group(1).replace('/', '-')
        
        # Content
        content = extract_clean_text(dsoup)
        if not content:
            content_div = dsoup.select_one("div#J_content.j-fontContent.newscontnet")
            if content_div:
                content = content_div.get_text(separator='\n\n', strip=True)
        
        if not content:
            print(f"  WARN: empty content for {page_title}")
            content = "(无内容)"
        
        summary = content[:500] if content else ""
        
        try:
            db.execute(
                "INSERT INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?, ?, ?, ?, ?, ?)",
                (SITE_NAME, page_url, page_title, content, date, summary)
            )
            db.commit()
            new_count += 1
            print(f"  OK: {page_title[:30]} | {date}")
        except sqlite3.IntegrityError:
            print(f"  DUPLICATE: {page_title[:30]}")
            db.rollback()
        except Exception as e:
            db.rollback()
            print(f"  DB ERROR: {e}")
    
    db.close()
    print(f"\nDone. New articles: {new_count}")

if __name__ == "__main__":
    crawl()
