#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# 南充经开区 - 通知公告 爬虫
# CMS: TRS WCM
# List: index.html, index_1.html (2 pages, 10 items/page)
# Detail content: trs_editor_view (new) or xl-article tables (old)

import requests
import re
import sqlite3
from bs4 import BeautifulSoup
from datetime import datetime

BASE_URL = "https://www.nanchong.gov.cn/jjkfq/xwdt/tzgg"
BASE_DOMAIN = "https://www.nanchong.gov.cn"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}

DB_PATH = "/root/search.db"

def get_pages():
    """Get all article URLs from list pages"""
    articles = []
    pages = ["index.html", "index_1.html"]
    
    for page in pages:
        url = f"{BASE_URL}/{page}"
        try:
            r = requests.get(url, headers=HEADERS, verify=False, timeout=30)
            r.encoding = "utf-8"
            if r.status_code != 200:
                continue
            soup = BeautifulSoup(r.text, "html.parser")
            
            for ul in soup.find_all("ul", class_="list-ul"):
                for li in ul.find_all("li"):
                    a = li.find("a")
                    span = li.find("span")
                    if a and span:
                        href = a.get("href", "")
                        title = a.get_text(strip=True)
                        date_str = span.get_text(strip=True)
                        
                        if href and title and date_str:
                            if href.startswith("./"):
                                full_url = BASE_URL + href[1:]
                            elif href.startswith("/"):
                                full_url = BASE_DOMAIN + href
                            else:
                                full_url = BASE_URL + "/" + href
                            
                            articles.append({
                                "url": full_url,
                                "title": title,
                                "date": date_str
                            })
        except Exception as e:
            print(f"  [ERR] {page}: {e}")
    
    return articles

def extract_inner_html(tag):
    """Extract inner HTML from a BeautifulSoup tag, treating tables correctly"""
    if tag is None:
        return None
    content = "".join(str(child) for child in tag.contents)
    return content.strip()

def get_detail(url):
    """Fetch detail page and extract content and date"""
    try:
        r = requests.get(url, headers=HEADERS, verify=False, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200 or len(r.text) < 1000:
            return None, None
        
        soup = BeautifulSoup(r.text, "html.parser")
        
        # Try trs_editor_view first (new articles with rich text content)
        div = soup.find("div", class_=re.compile(r"trs_editor_view"))
        if div:
            content = extract_inner_html(div)
            text_only = re.sub(r'<[^>]+>', '', content).strip()
            if len(text_only) >= 50:
                # Get date
                pub_date = ""
                meta = soup.find("meta", attrs={"name": "PubDate"})
                if meta and meta.get("content"):
                    pub_date = meta["content"].strip()[:10]
                return content, pub_date
        
        # Try xl-article (old articles - tables with PDF links)
        xl_div = soup.find("div", class_="xl-article")
        if xl_div:
            # Remove style blocks
            for style_tag in xl_div.find_all("style"):
                style_tag.decompose()
            for script_tag in xl_div.find_all("script"):
                script_tag.decompose()
            
            # Look for tables (old articles)
            tables = xl_div.find_all("table")
            if tables:
                content_parts = []
                for table in tables:
                    rows = table.find_all("tr")
                    for row in rows:
                        cells = row.find_all("td")
                        if len(cells) >= 2:
                            # First cell: icon/link, second cell: file name + link
                            link = cells[1].find("a")
                            if link:
                                text = link.get_text(strip=True)
                                file_url = link.get("href", "")
                                content_parts.append(f'<p><a href="{file_url}">{text}</a></p>')
                            else:
                                text = cells[1].get_text(strip=True)
                                if text:
                                    content_parts.append(f'<p>{text}</p>')
                if content_parts:
                    content = "\n".join(content_parts)
                    text_only = " ".join(re.sub(r'<[^>]+>', ' ', content).split())
                    if len(text_only) >= 20:
                        pub_date = ""
                        meta = soup.find("meta", attrs={"name": "PubDate"})
                        if meta and meta.get("content"):
                            pub_date = meta["content"].strip()[:10]
                        return content, pub_date
            
            # Look for any remaining text content
            text_only = xl_div.get_text(strip=True)
            if len(text_only) >= 50:
                content = extract_inner_html(xl_div)
                pub_date = ""
                meta = soup.find("meta", attrs={"name": "PubDate"})
                if meta and meta.get("content"):
                    pub_date = meta["content"].strip()[:10]
                return content, pub_date
        
        return None, None
    
    except Exception as e:
        print(f"  [ERR] {url[:80]}: {e}")
        return None, None

def main():
    print("=" * 60)
    print("南充经开区 - 通知公告 爬虫")
    print(f"启动时间: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
    print("=" * 60)
    
    # Step 1: Get list
    print("\n[1/3] 抓取列表页...")
    articles = get_pages()
    print(f"  共找到 {len(articles)} 条记录")
    
    if not articles:
        print("  未找到任何记录，退出")
        return
    
    # Step 2: Check DB for existing URLs
    print("\n[2/3] 检查数据库去重...")
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    
    c.execute("SELECT page_url FROM gov_raw")
    existing = set(row[0] for row in c.fetchall())
    
    new_articles = [a for a in articles if a["url"] not in existing]
    print(f"  新增: {len(new_articles)}/{len(articles)} (已存在: {len(existing)})")
    
    if not new_articles:
        print("  没有新记录，跳过详情抓取")
        conn.close()
        return
    
    # Step 3: Fetch details
    print("\n[3/3] 抓取详情页...")
    imported = 0
    skipped = 0
    
    for i, article in enumerate(new_articles, 1):
        title_short = article["title"][:45].replace("\n", " ")
        print(f"  [{i}/{len(new_articles)}] {title_short}...", end=" ", flush=True)
        
        content, pub_date = get_detail(article["url"])
        
        if not content:
            print("✗ 无正文")
            skipped += 1
            continue
        
        date = pub_date or article["date"]
        
        text_only = re.sub(r'<[^>]+>', '', content).strip()
        summary = text_only[:200]
        
        try:
            c.execute("""INSERT OR IGNORE INTO gov_raw 
                (title, content, page_url, summary, site_name, publish_date, category)
                VALUES (?, ?, ?, ?, ?, ?, ?)""",
                (article["title"], content, article["url"], summary,
                 "南充经开区", date, "环评公示"))
            if c.rowcount > 0:
                conn.commit()
                imported += 1
                print(f"✓ ({len(text_only)}字)")
            else:
                print(f"• 已存在")
        except Exception as e:
            print(f"✗ DB错误: {e}")
            conn.rollback()
    
    conn.close()
    print(f"\n{'=' * 60}")
    print(f"导入完成: {imported} 条 (跳过 {skipped} 条)")
    print(f"{'=' * 60}")

if __name__ == "__main__":
    main()
