#!/usr/bin/env python3
"""
南康区-公告公示 爬虫
URL: http://www.nkjx.gov.cn/nkqxxgk/nk1404/list.shtml
CMS: UCAP CMS
列表: ul.newsList > li > a[href] + span.time
分页: list.shtml → list_2.shtml → list_3.shtml ... (198页, 20条/页)
详情: div#zoomcon > UCAPCONTENT > p/table
日期: <meta name="PubDate" content="..."/>
附件: a[href] 含 .pdf/.doc/.xls/.zip
"""

import re, sys, time, os
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime

BASE_URL = "http://www.nkjx.gov.cn/nkqxxgk/nk1404/list.shtml"
SITE_NAME = "南康区-公告公示"
GROUP = "江西"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}
TIMEOUT = 15
DELAY = 1.0

DB_PATH = "/root/search.db"

def debug(msg):
    print(f"  [DEBUG] {msg}", file=sys.stderr)

def parse_date(text):
    m = re.search(r'(\d{4})[-/年](\d{1,2})[-/月](\d{1,2})', text)
    if m:
        return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return ""

def get_soup(url):
    for retry in range(3):
        try:
            resp = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            resp.encoding = 'utf-8'
            if resp.status_code == 200:
                return BeautifulSoup(resp.text, 'html.parser')
        except Exception as e:
            debug(f"  请求失败 (重试 {retry+1}/3): {e}")
            time.sleep(2)
    return None

def fetch_list_page(page_num):
    if page_num == 1:
        url = BASE_URL
    else:
        url = f"http://www.nkjx.gov.cn/nkqxxgk/nk1404/list_{page_num}.shtml"
    soup = get_soup(url)
    if not soup:
        return []
    
    items = []
    ul = soup.find('ul', class_='newsList')
    if not ul:
        # try pageList which also has newsList class
        ul = soup.find('ul', class_='pageList')
    if not ul:
        debug(f"  未找到列表 ul")
        return []
    
    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        href = urljoin(BASE_URL, a['href'])
        title = a.get('title', '') or a.text.strip()
        
        date_span = li.find('span', class_='time')
        date_str = date_span.text.strip() if date_span else ""
        items.append({'title': title, 'url': href, 'date': date_str})
    
    return items

def get_total_pages(soup):
    """从 createPageHTML('page_div',10, 1,'list','shtml',198) 提取总页数"""
    scripts = soup.find_all('script')
    for s in scripts:
        text = s.string or ''
        m = re.search(r'createPageHTML\([^,]+,\s*(\d+),\s*\d+,\s*[^,]+,\s*[^,]+,\s*(\d+)\)', text)
        if m:
            return int(m.group(2))
    return 1

def parse_ucap_content(con_txt):
    """解析 UCAPCONTENT 内的内容，提取文本段落、表格、图片、附件"""
    parts = []
    attachments = []
    
    for child in con_txt.children:
        if not child.name:
            text = str(child).strip()
            if text and not all(c in ' \n\r\t\u3000' for c in text):
                parts.append(text)
            continue
        
        tag = child.name.lower()
        
        if tag == 'p':
            # Check for block-level children
            has_block = any(c.name in ('table', 'div', 'ul', 'ol') for c in child.find_all(recursive=False))
            if has_block:
                for sub in child.children:
                    if not sub.name:
                        t = str(sub).strip()
                        if t and not all(c in ' \n\r\t\u3000' for c in t):
                            parts.append(t)
                        continue
                    if sub.name == 'table':
                        parts.append(str(sub))
                    elif sub.name in ('p', 'div'):
                        t = sub.get_text(strip=True)
                        if t:
                            parts.append(t)
                    else:
                        t = sub.get_text(strip=True)
                        if t:
                            parts.append(t)
            else:
                txt = child.get_text(strip=True)
                if txt and not all(c in ' \n\r\t\u3000 \xa0' for c in txt):
                    parts.append(txt)
        
        elif tag == 'table':
            parts.append(str(child))
        elif tag == 'img':
            src = child.get('src', '')
            alt = child.get('alt', '')
            if src:
                parts.append(f"![{alt}]({urljoin('http://www.nkjx.gov.cn', src)})")
        elif tag in ('ul', 'ol'):
            txt = child.get_text(strip=True)
            if txt:
                parts.append(txt)
        elif tag == 'br':
            pass
        elif tag == 'div':
            txt = child.get_text(strip=True)
            if txt:
                parts.append(txt)
        else:
            txt = child.get_text(strip=True)
            if txt:
                parts.append(txt)
    
    # Extract attachments from UCAPCONTENT
    for a in con_txt.find_all('a', href=True):
        href = a['href']
        text = a.text.strip()
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ceb|ofd|ppt|pptx)$', href.lower()):
            full_url = urljoin('http://www.nkjx.gov.cn', href)
            attachments.append(f"[{text}]({full_url})")
    
    # Dedup
    seen = set()
    unique_parts = []
    for p in parts:
        key = p[:100]
        if key not in seen:
            seen.add(key)
            unique_parts.append(p)
    
    content = '\n\n'.join(unique_parts)
    return content, attachments

def extract_content(soup):
    parts = []
    attachments = []
    
    # Title from <UCAPTITLE>
    ucap_title = soup.find('ucaptitle')
    title = ucap_title.text.strip() if ucap_title else ""
    
    # Title fallback from h1
    if not title:
        h1 = soup.find('h1', class_='article-title')
        if h1:
            title = h1.get_text(strip=True)
    
    # Title fallback from meta
    if not title:
        meta = soup.find('meta', attrs={'name': 'ArticleTitle'})
        if meta and meta.get('content'):
            title = meta['content'].strip()
    
    # Title fallback from <title>
    if not title:
        t = soup.find('title')
        if t:
            title = t.text.strip()
            title = re.sub(r'\s*\|\s*南康区信息公开.*$', '', title).strip()
    
    # PubDate from 生成日期 dd.scrq (real article date)
    pub_date = ""
    dd = soup.find('dd', class_='scrq')
    if dd:
        div = dd.find('div', class_='display-block')
        if div:
            pub_date = parse_date(div.text)
    
    # PubDate fallback from list date
    if not pub_date:
        meta_date = soup.find('meta', attrs={'name': 'PubDate'})
        if meta_date and meta_date.get('content'):
            pub_date = parse_date(meta_date['content'])
    
    zoom = soup.find('div', id='zoomcon')
    if zoom:
        ucap = zoom.find('ucapcontent')
        if ucap:
            content, attach = parse_ucap_content(ucap)
            parts.append(content)
            attachments.extend(attach)
        else:
            # Try direct content
            content, attach = parse_ucap_content(zoom)
            parts.append(content)
            attachments.extend(attach)
    
    content = '\n\n'.join(p for p in parts if p)
    return title, content, pub_date, attachments

def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        page_url TEXT,
        title TEXT,
        content TEXT,
        publish_date TEXT,
        site_name TEXT,
        summary TEXT,
        attachments TEXT,
        date_rank TEXT,
        created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
    )''')
    c.execute('''CREATE INDEX IF NOT EXISTS idx_gov_raw_url 
        ON gov_raw(page_url, site_name)''')
    conn.commit()
    conn.close()

def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    imported = 0
    for item in items:
        summary = item['content'][:200] if item['content'] else ''
        try:
            c.execute('''INSERT OR REPLACE INTO gov_raw 
                (page_url, title, content, publish_date, site_name, summary, attachments, date_rank)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)''', (
                item['url'],
                item['title'],
                item['content'],
                item['pub_date'],
                SITE_NAME,
                summary,
                '\n'.join(item['attachments']) if item['attachments'] else '',
                item['pub_date'] or '0000-00-00',
            ))
            imported += 1
        except Exception as e:
            debug(f"  入库失败: {item['url']} - {e}")
    conn.commit()
    conn.close()
    return imported

def main():
    import argparse
    parser = argparse.ArgumentParser(description='南康区-公告公示 爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数, 0=全部')
    parser.add_argument('--skip-db', action='store_true', help='跳过入库')
    args = parser.parse_args()
    
    init_db()
    
    print(f"===== {SITE_NAME} 爬取 =====")
    soup = get_soup(BASE_URL)
    if not soup:
        print("❌ 无法访问首页")
        return
    
    total_pages = get_total_pages(soup)
    print(f"总页数: {total_pages}")
    
    pages_to_fetch = total_pages if args.pages == 0 else min(args.pages, total_pages)
    print(f"将爬取: {pages_to_fetch} 页")
    
    all_items = []
    page_items_count = []
    
    for page in range(1, pages_to_fetch + 1):
        print(f"\n--- 第 {page}/{pages_to_fetch} 页 ---")
        items = fetch_list_page(page)
        if not items:
            print(f"  第 {page} 页无数据")
            continue
        
        page_items_count.append(len(items))
        print(f"  列表项: {len(items)} 条")
        
        for idx, item in enumerate(items):
            print(f"  [{len(all_items)+1}] {item['title'][:40]}...")
            
            detail_soup = get_soup(item['url'])
            if not detail_soup:
                print(f"    ⚠️ 详情页请求失败")
                all_items.append({
                    'title': item['title'],
                    'url': item['url'],
                    'content': '',
                    'pub_date': item['date'],
                    'attachments': [],
                })
                continue
            
            title, content, pub_date, attachments = extract_content(detail_soup)
            final_title = title or item['title']
            final_date = pub_date or item['date']
            
            all_items.append({
                'title': final_title,
                'url': item['url'],
                'content': content,
                'pub_date': final_date,
                'attachments': attachments,
            })
            
            if content:
                seg_count = len(content.split('\n\n'))
                print(f"    seg={seg_count} | attach={'YES' if attachments else 'no'} | date={final_date}")
            else:
                print(f"    ⚠️ 空正文 | attach={'YES' if attachments else 'no'}")
            
            time.sleep(DELAY)
    
    total = len(all_items)
    with_body = sum(1 for i in all_items if i['content'])
    total_attach = sum(len(i['attachments']) for i in all_items)
    avg_seg = sum(len(i['content'].split('\n\n')) for i in all_items if i['content']) / max(with_body, 1)
    
    print(f"\n===== {SITE_NAME} 爬取完成 =====")
    print(f"共爬取: {total} 条")
    print(f"有正文: {with_body} 条 ({with_body/total*100:.1f}%)" if total else "有正文: 0 条")
    print(f"平均段落数: {avg_seg:.1f}")
    print(f"附件数: {total_attach}")
    if page_items_count:
        print(f"页均条数: {sum(page_items_count)/len(page_items_count):.1f}")
    
    if not args.skip_db:
        imported = save_to_db(all_items)
        print(f"入库: {imported} 条")
    else:
        print("入库: 已跳过")

if __name__ == '__main__':
    main()
