#!/usr/bin/env python3
"""
宁国市人民政府-通知公告 爬虫
URL: http://www.ningguo.gov.cn/News/showList/833/page_1.html
CMS: 自定义ASP.NET MVC
列表: div.m-listrg > li > a[href] + span(日期)
分页: /News/showList/833/page_N.html (15条/页)
详情: div.m-dttexts.j-fontContent 正文
日期: div.m-pgpdbox1 "发布时间：YYYY-MM-DD"
附件: a[href] 含 .pdf/.doc/.xls/.zip
"""

import re, sys, time, os
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://www.ningguo.gov.cn/News/showList/833/page_1.html"
SITE_NAME = "宁国市-通知公告"
GROUP = "安徽"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}
TIMEOUT = 15
DELAY = 1.0

DB_PATH = "/root/search.db"

def debug(msg):
    print(f"  [DEBUG] {msg}", file=sys.stderr)

def parse_date(text):
    m = re.search(r'(\d{4})[-/年](\d{1,2})[-/月](\d{1,2})', text)
    if m:
        return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return ""

def get_soup(url):
    for retry in range(3):
        try:
            resp = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            resp.encoding = 'utf-8'
            if resp.status_code == 200:
                return BeautifulSoup(resp.text, 'lxml')
        except Exception as e:
            debug(f"  请求失败 (重试 {retry+1}/3): {e}")
            time.sleep(2)
    return None


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def parse_content(soup):
    """Parse content from m-dttexts div, handling tables and br separators.
    Uses .contents (direct children in order) to maintain sequence."""
    parts = []
    attachments = []
    
    content_div = soup.find('div', class_='m-dttexts')
    if not content_div:
        return '', []
    
    # Walk all direct children in document order
    all_elements = list(content_div.children)
    # Filter to only named elements, keeping order
    named_elements = [ch for ch in all_elements if ch.name]
    # Also include direct text children
    text_parts = []
    for child in content_div.children:
        if not child.name:
            t = str(child).strip()
            if t:
                text_parts.append(t)
    
    if text_parts:
        parts.extend(text_parts)
    
    for child in named_elements:
        tag = child.name.lower()
        
        if tag == 'p':
            # Check if this p contains tables
            inner_tables = child.find_all('table', recursive=False)
            
            if inner_tables:
                for sub in child.children:
                    if not sub.name:
                        t = str(sub).strip()
                        if t and not all(c in ' \n\r\t\u3000' for c in t):
                            parts.append(t)
                        continue
                    sn = sub.name.lower()
                    if sn == 'table':
                        parts.append(table_to_markdown(sub))
                    elif sn == 'br':
                        parts.append('')
                    else:
                        t = sub.get_text(strip=True)
                        if t and not all(c in ' \n\r\t\u3000 \xa0' for c in t):
                            parts.append(t)
            else:
                txt = child.get_text(strip=True)
                if txt and not all(c in ' \n\r\t\u3000 \xa0' for c in txt):
                    parts.append(txt)
        
        elif tag == 'table':
            parts.append(table_to_markdown(child))
        elif tag == 'br':
            parts.append('')
        elif tag == 'hr':
            parts.append('---')
        elif tag == 'img':
            src = child.get('src', '')
            alt = child.get('alt', '')
            if src:
                parts.append(f"![{alt}]({urljoin('http://www.ningguo.gov.cn', src)})")
        elif tag in ('ul', 'ol'):
            txt = child.get_text(strip=True)
            if txt:
                parts.append(txt)
        elif tag == 'div':
            txt = child.get_text(strip=True)
            if txt and not all(c in ' \n\r\t\u3000 \xa0' for c in txt):
                parts.append(txt)
        elif tag in ('span', 'strong', 'em', 'b', 'i', 'u', 'a'):
            txt = child.get_text(strip=True)
            if txt and not all(c in ' \n\r\t\u3000 \xa0' for c in txt):
                parts.append(txt)
        elif tag == 'br':
            parts.append('')
        elif tag == 'hr':
            parts.append('---')
    
    # Attachments from the content div
    for a in content_div.find_all('a', href=True):
        href = a['href']
        text = a.text.strip()
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ceb|ofd|ppt|pptx)$', href.lower()):
            full_url = urljoin('http://www.ningguo.gov.cn', href)
            attachments.append(f"[{text}]({full_url})")
    
    # Clean up
    cleaned = []
    for p in parts:
        if p == '' and cleaned and cleaned[-1] == '':
            continue
        cleaned.append(p)
    
    content = '\n\n'.join(cleaned)
    content = re.sub(r'\n{3,}', '\n\n', content)
    return content.strip(), attachments

def fetch_list_page(page_num):
    if page_num == 1:
        url = BASE_URL
    else:
        url = f"http://www.ningguo.gov.cn/News/showList/833/page_{page_num}.html"
    soup = get_soup(url)
    if not soup:
        return []
    
    items = []
    container = soup.find('div', class_='m-listrg')
    if not container:
        return []
    
    # The li items are inside m_cglist > ul or directly
    list_div = container.find('div', class_='m_cglist')
    if list_div:
        ul = list_div.find('ul')
        if ul:
            list_container = ul
        else:
            list_container = list_div
    else:
        list_container = container
    
    for li in list_container.find_all('li', recursive=False):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        href = urljoin(BASE_URL, a['href'])
        title = a.get('title', '') or a.text.strip()
        date_span = li.find('span')
        date_str = date_span.text.strip() if date_span else ""
        items.append({'title': title, 'url': href, 'date': date_str})
    
    return items

def extract_content(soup):
    title = ""
    pub_date = ""
    
    # Title from h1
    h1 = soup.find('h1')
    if h1:
        title = h1.text.strip()
    
    # Title fallback from <title>
    if not title:
        t = soup.find('title')
        if t:
            title = t.text.strip()
            title = re.sub(r'\s*[-–—|].*$', '', title).strip()
    
    # Date from header div
    header = soup.find('div', class_='m-pgpdbox1')
    if header:
        pub_date = parse_date(header.text)
    
    # Content from m-dttexts
    content, attachments = parse_content(soup)
    return title, content, pub_date, attachments

def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        page_url TEXT,
        title TEXT,
        content TEXT,
        publish_date TEXT,
        site_name TEXT,
        summary TEXT,
        attachments TEXT,
        date_rank TEXT,
        created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
    )''')
    c.execute('''CREATE INDEX IF NOT EXISTS idx_gov_raw_url 
        ON gov_raw(page_url, site_name)''')
    conn.commit()
    conn.close()

def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    imported = 0
    for item in items:
        summary = item['content'][:200] if item['content'] else ''
        try:
            c.execute('''INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, \'crawl_ningguo_tzgg.py\')''', (
                item['url'],
                item['title'],
                item['content'],
                item['pub_date'],
                SITE_NAME,
                summary,
                '\n'.join(item['attachments']) if item['attachments'] else '',
                item['pub_date'] or '0000-00-00',
            ))
            imported += 1
        except Exception as e:
            debug(f"  入库失败: {item['url']} - {e}")
    conn.commit()
    conn.close()
    return imported

def main():
    import argparse
    parser = argparse.ArgumentParser(description='宁国市-通知公告 爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数, 0=全部')
    parser.add_argument('--skip-db', action='store_true', help='跳过入库')
    args = parser.parse_args()
    
    init_db()
    
    print(f"===== {SITE_NAME} 爬取 =====")
    pages_to_fetch = args.pages
    print(f"将爬取: {pages_to_fetch} 页")
    
    all_items = []
    page_items_count = []
    
    for page in range(1, pages_to_fetch + 1):
        print(f"\n--- 第 {page}/{pages_to_fetch} 页 ---")
        items = fetch_list_page(page)
        if not items:
            print(f"  第 {page} 页无数据")
            continue
        
        page_items_count.append(len(items))
        print(f"  列表项: {len(items)} 条")
        
        for idx, item in enumerate(items):
            print(f"  [{len(all_items)+1}] {item['title'][:40]}...")
            
            detail_soup = get_soup(item['url'])
            if not detail_soup:
                print(f"    ⚠️ 详情页请求失败")
                all_items.append({
                    'title': item['title'],
                    'url': item['url'],
                    'content': '',
                    'pub_date': item['date'],
                    'attachments': [],
                })
                continue
            
            title, content, pub_date, attachments = extract_content(detail_soup)
            final_title = title or item['title']
            final_date = pub_date or item['date']
            
            all_items.append({
                'title': final_title,
                'url': item['url'],
                'content': content,
                'pub_date': final_date,
                'attachments': attachments,
            })
            
            if content:
                seg_count = len(content.split('\n\n'))
                print(f"    seg={seg_count} | attach={'YES' if attachments else 'no'} | date={final_date}")
            else:
                print(f"    ⚠️ 空正文 | attach={'YES' if attachments else 'no'}")
            
            time.sleep(DELAY)
    
    total = len(all_items)
    with_body = sum(1 for i in all_items if i['content'])
    total_attach = sum(len(i['attachments']) for i in all_items)
    avg_seg = sum(len(i['content'].split('\n\n')) for i in all_items if i['content']) / max(with_body, 1)
    
    print(f"\n===== {SITE_NAME} 爬取完成 =====")
    print(f"共爬取: {total} 条")
    print(f"有正文: {with_body} 条 ({with_body/total*100:.1f}%)" if total else "有正文: 0 条")
    print(f"平均段落数: {avg_seg:.1f}")
    print(f"附件数: {total_attach}")
    if page_items_count:
        print(f"页均条数: {sum(page_items_count)/len(page_items_count):.1f}")
    
    if not args.skip_db:
        # Clean old data
        import sqlite3
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
        c.execute("DELETE FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        conn.commit()
        conn.close()
        
        imported = save_to_db(all_items)
        print(f"入库: {imported} 条")
    else:
        print("入库: 已跳过")

if __name__ == '__main__':
    main()
