#!/usr/bin/env python3
"""
第一师阿拉尔市 通知公告 爬虫
URL: http://www.nys.gov.cn:8090/xwzx/tzgg
CMS: 自定义ASP.NET MVC CMS
列表: ul.newsList > li > a + span.date
分页: /xwzx/tzgg_2  → /xwzx/tzgg_136  (共2703条, 20条/页)
详情: /xwzx/tzgg/content_XXXXX
正文: div.printArea > div.conTxt (第一个)
日期: div.property 中 "发布时间：..."
附件: a[href] 含 .pdf/.doc/.xls/.zip
"""

import re, sys, time, os
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime

BASE_URL = "http://www.nys.gov.cn:8090/xwzx/tzgg"
SITE_NAME = "第一师阿拉尔市"
GROUP = "新疆兵团"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}
TIMEOUT = 15
DELAY = 1.0

DB_PATH = "/root/search.db"

def debug(msg):
    print(f"  [DEBUG] {msg}", file=sys.stderr)

def parse_date(text):
    m = re.search(r'(\d{4})[-/年](\d{1,2})[-/月](\d{1,2})', text)
    if m:
        return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return ""

def get_soup(url):
    for retry in range(3):
        try:
            resp = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            resp.encoding = 'utf-8'
            if resp.status_code == 200:
                return BeautifulSoup(resp.text, 'html.parser')
        except Exception as e:
            debug(f" 请求失败 (重试 {retry+1}/3): {e}")
            time.sleep(2)
    return None

def fetch_list_page(page_num):
    if page_num == 1:
        url = BASE_URL
    else:
        url = f"{BASE_URL}_{page_num}"
    soup = get_soup(url)
    if not soup:
        return []
    
    items = []
    ul = soup.find('ul', class_='newsList')
    if not ul:
        return items
    
    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        href = urljoin(BASE_URL, a['href'])
        title_attr = a.get('title', '')
        title_text = a.text.strip()
        
        title = title_text
        if title_attr and '\u6807\u9898\uff1a' in title_attr:
            m = re.search(r'\u6807\u9898\uff1a(.+)', title_attr)
            if m:
                t = m.group(1).strip()
                if t:
                    title = t
        
        date_span = li.find('span', class_='date')
        date_str = date_span.text.strip() if date_span else ""
        items.append({'title': title, 'url': href, 'date': date_str})
    
    return items

def get_total_pages(soup):
    page_div = soup.find('div', class_='page')
    if not page_div:
        return 1
    total_span = page_div.find('span', class_='total')
    if total_span:
        m = re.search(r'\u5171(\d+)\u9875', total_span.text)
        if m:
            return int(m.group(1))
    return 1

def extract_content(soup):
    parts = []
    attachments = []
    
    pa = soup.find('div', class_='printArea')
    if not pa:
        return "", [], "", []
    
    con_txt = pa.find('div', class_='conTxt')
    if not con_txt:
        return "", [], "", []
    
    title = ""
    h2 = pa.find('h2', class_='title')
    if h2:
        title = h2.text.strip()
    if not title:
        t = soup.find('title')
        if t:
            title = t.text.strip()
            title = re.sub(r'_\u901a\u77e5\u516c\u544a.*$', '', title).strip()
    
    pub_date = ""
    prop = pa.find('div', class_='property')
    if prop:
        pub_date = parse_date(prop.text)
    
    for child in con_txt.children:
        if not child.name:
            text = str(child).strip()
            if text and not all(c in ' \n\r\t\u3000' for c in text):
                parts.append(text)
            continue
        
        tag = child.name.lower()
        
        if tag in ('p', 'div'):
            has_block = any(c.name in ('table', 'div', 'ul', 'ol') for c in child.find_all(recursive=False))
            if has_block:
                for sub in child.children:
                    if not sub.name:
                        t = str(sub).strip()
                        if t and not all(c in ' \n\r\t\u3000' for c in t):
                            parts.append(t)
                        continue
                    if sub.name == 'table':
                        parts.append(str(sub))
                    elif sub.name in ('p', 'div'):
                        sub_txt = sub.get_text(strip=True)
                        if sub_txt and not all(c in ' \n\r\t\u3000' for c in sub_txt):
                            parts.append(sub_txt)
                    else:
                        t = sub.get_text(strip=True)
                        if t:
                            parts.append(t)
            else:
                txt = child.get_text(strip=True)
                if txt and not all(c in ' \n\r\t\u3000 \xa0' for c in txt):
                    parts.append(txt)
        
        elif tag == 'table':
            parts.append(str(child))
        elif tag == 'img':
            src = child.get('src', '')
            alt = child.get('alt', '')
            if src:
                parts.append(f"![{alt}]({urljoin(BASE_URL, src)})")
        elif tag in ('ul', 'ol'):
            txt = child.get_text(strip=True)
            if txt:
                parts.append(txt)
        elif tag == 'br':
            pass
        else:
            txt = child.get_text(strip=True)
            if txt:
                parts.append(txt)
    
    # attachments
    for a in con_txt.find_all('a', href=True):
        href = a['href']
        text = a.text.strip()
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ceb|ofd|ppt|pptx)$', href.lower()):
            full_url = urljoin(BASE_URL, href)
            attachments.append(f"[{text}]({full_url})")
    
    for a in pa.find_all('a', href=True):
        href = a['href']
        text = a.text.strip()
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ceb|ofd|ppt|pptx)$', href.lower()):
            full_url = urljoin(BASE_URL, href)
            entry = f"[{text}]({full_url})"
            if entry not in attachments:
                attachments.append(entry)
    
    seen = set()
    unique_parts = []
    for p in parts:
        key = p[:100]
        if key not in seen:
            seen.add(key)
            unique_parts.append(p)
    
    content = '\n\n'.join(unique_parts)
    return title, content, pub_date, attachments

def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        page_url TEXT,
        title TEXT,
        content TEXT,
        publish_date TEXT,
        site_name TEXT,
        summary TEXT,
        attachments TEXT,
        date_rank TEXT,
        created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
    )''')
    c.execute('''CREATE INDEX IF NOT EXISTS idx_gov_raw_url 
        ON gov_raw(page_url, site_name)''')
    conn.commit()
    conn.close()

def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    imported = 0
    for item in items:
        summary = item['content'][:200] if item['content'] else ''
        try:
            c.execute('''INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, \'crawl_nys_tzgg.py\')''', (
                item['url'],
                item['title'],
                item['content'],
                item['pub_date'],
                SITE_NAME,
                summary,
                '\n'.join(item['attachments']) if item['attachments'] else '',
                item['pub_date'] or '0000-00-00',
            ))
            imported += 1
        except Exception as e:
            debug(f"  入库失败: {item['url']} - {e}")
    conn.commit()
    conn.close()
    return imported

def main():
    import argparse
    parser = argparse.ArgumentParser(description='第一师阿拉尔市 通知公告 爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数, 0=全部')
    parser.add_argument('--skip-db', action='store_true', help='跳过入库')
    args = parser.parse_args()
    
    init_db()
    
    print(f"===== {SITE_NAME} 爬取 =====")
    soup = get_soup(BASE_URL)
    if not soup:
        print("❌ 无法访问首页")
        return
    
    total_pages = get_total_pages(soup)
    print(f"总页数: {total_pages}")
    
    pages_to_fetch = total_pages if args.pages == 0 else min(args.pages, total_pages)
    print(f"将爬取: {pages_to_fetch} 页")
    
    all_items = []
    page_items_count = []
    
    for page in range(1, pages_to_fetch + 1):
        print(f"\n--- 第 {page}/{pages_to_fetch} 页 ---")
        items = fetch_list_page(page)
        if not items:
            print(f"  第 {page} 页无数据")
            continue
        
        page_items_count.append(len(items))
        print(f"  列表项: {len(items)} 条")
        
        for idx, item in enumerate(items):
            print(f"  [{len(all_items)+1}] {item['title'][:40]}...")
            
            detail_soup = get_soup(item['url'])
            if not detail_soup:
                print(f"    ⚠️ 详情页请求失败")
                all_items.append({
                    'title': item['title'],
                    'url': item['url'],
                    'content': '',
                    'pub_date': item['date'],
                    'attachments': [],
                })
                continue
            
            title, content, pub_date, attachments = extract_content(detail_soup)
            final_title = title or item['title']
            final_date = pub_date or item['date']
            
            all_items.append({
                'title': final_title,
                'url': item['url'],
                'content': content,
                'pub_date': final_date,
                'attachments': attachments,
            })
            
            if content:
                seg_count = len(content.split('\n\n'))
                print(f"    seg={seg_count} | attach={'YES' if attachments else 'no'} | date={final_date}")
            else:
                print(f"    ⚠️ 空正文 | attach={'YES' if attachments else 'no'}")
            
            time.sleep(DELAY)
    
    total = len(all_items)
    with_body = sum(1 for i in all_items if i['content'])
    total_attach = sum(len(i['attachments']) for i in all_items)
    avg_seg = sum(len(i['content'].split('\n\n')) for i in all_items if i['content']) / max(with_body, 1)
    
    print(f"\n===== {SITE_NAME} 爬取完成 =====")
    print(f"共爬取: {total} 条")
    print(f"有正文: {with_body} 条 ({with_body/total*100:.1f}%)" if total else "有正文: 0 条")
    print(f"平均段落数: {avg_seg:.1f}")
    print(f"附件数: {total_attach}")
    if page_items_count:
        print(f"页均条数: {sum(page_items_count)/len(page_items_count):.1f}")
    
    if not args.skip_db:
        imported = save_to_db(all_items)
        print(f"入库: {imported} 条")
    else:
        print("入库: 已跳过")

if __name__ == '__main__':
    main()
