#!/usr/bin/env python3
"""
万荣县人民政府 - 公示公告 爬虫
http://www.wanrong.gov.cn/zfxxgk/gsgg/
"""
import requests
import re
import sys
import json
import time
import sqlite3
import os
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE_URL = "http://www.wanrong.gov.cn"
SITE_NAME = "万荣县人民政府"
COLUMN = "公示公告"
MAX_PAGES = 5
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 30

def get_page(url, session):
    for attempt in range(3):
        try:
            r = session.get(url, headers=HEADERS, timeout=TIMEOUT)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if attempt < 2:
                time.sleep(3)
    return None

def parse_list(html):
    """Parse list page from ul.pd20 li items"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='pd20')
    if not ul:
        return items
    for li in ul.find_all('li'):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a['href']
        if not href.startswith('http'):
            href = BASE_URL + href
        title = a.get('title', '') or a.get_text(strip=True)
        date_span = li.find('span')
        date = date_span.get_text(strip=True) if date_span else ''
        if href and title:
            items.append({'url': href, 'title': title, 'date': date})
    return items

def parse_detail(html, url):
    """Parse detail page"""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from meta
    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()
    
    # Date from meta
    date = ''
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date and meta_date.get('content'):
        date = meta_date['content'].strip()
    
    # Source
    source = ''
    meta_source = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta_source and meta_source.get('content'):
        source = meta_source['content'].strip()
    
    # Content from div.ConBox_nr
    content_div = soup.find('div', class_='ConBox_nr')
    content_parts = []
    attachments = []
    has_text = False
    total_text_len = 0
    found_images = False
    
    if content_div:
        # Skip the first header div and footer
        for elem in content_div.find_all(['p', 'table', 'img']):
            tag = elem.name
            # Skip the header paragraph with date/source info
            if tag == 'p':
                text = elem.get_text(strip=True)
                if not text or '发布日期' in text[:4] or text in ['打印本页', '【字体：大中小】']:
                    continue
                content_parts.append(text)
                has_text = True
                total_text_len += len(text)
            elif tag == 'table':
                content_parts.append(str(elem))
                total_text_len += len(elem.get_text(strip=True))
            elif tag == 'img':
                src = elem.get('src', '')
                if src and not src.startswith('http'):
                    src = BASE_URL + src
                alt = elem.get('alt', '')
                found_images = True
                content_parts.append(f'![{alt}]({src})')
        
        # Attachments
        for a in content_div.find_all('a', href=True):
            href = a['href']
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                if not href.startswith('http'):
                    href = BASE_URL + href
                name = a.get_text(strip=True) or href.split('/')[-1]
                attachments.append({'name': name, 'url': href})
    
    # Image-only / low-text fallback
    if total_text_len < 20:
        fallback = f"[{title}]({url})"
        if found_images:
            content_parts = [fallback] + [p for p in content_parts if p.startswith('![')]
        else:
            content_parts = [fallback]
            pdf_links = [a for a in attachments if re.search(r'\.pdf$', a['url'], re.I)]
            if pdf_links:
                for pdf in pdf_links:
                    content_parts.append(f"[PDF附件: {pdf['name']}]({pdf['url']})")
    
    content = '\n\n'.join(content_parts)
    summary = re.sub(r'\s+', ' ', content[:200]).strip() if content else title
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    
    return title, date, content, summary, attachments_json

def save_to_db(items, conn):
    cursor = conn.cursor()
    count = 0
    for item in items:
        publish_date = item.get('date', '')
        if publish_date:
            try:
                dt = datetime.strptime(publish_date[:10], '%Y-%m-%d')
                if dt > datetime.now().replace(hour=0, minute=0, second=0, microsecond=0) + __import__('datetime').timedelta(days=30):
                    publish_date = ''
            except:
                publish_date = ''
        
        cursor.execute("""
            INSERT OR REPLACE INTO gov_raw (page_url, site_name, title, publish_date, summary, content, attachments, status, group_name, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, 'active', ?, 'crawl_wanrong.py')
        """, (
            item['url'], SITE_NAME, item['title'],
            publish_date, item.get('summary', ''),
            item.get('content', ''), item.get('attachments', '[]'),
            SITE_NAME
        ))
        count += 1
    conn.commit()
    return count

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES)
    args = parser.parse_args()
    max_pages = args.max_pages
    
    session = requests.Session()
    all_items = []
    
    for page_num in range(1, max_pages + 1):
        if page_num == 1:
            url = f"{BASE_URL}/zfxxgk/gsgg/index.shtml"
        else:
            url = f"{BASE_URL}/zfxxgk/gsgg/index_{page_num}.shtml"
        
        print(f"[列表页] 第{page_num}页: {url}", flush=True)
        html = get_page(url, session)
        if not html:
            print(f"  -> 获取失败，停止翻页", flush=True)
            break
        
        items = parse_list(html)
        if not items:
            print(f"  -> 无数据，停止翻页", flush=True)
            break
        
        print(f"  -> {len(items)}条", flush=True)
        
        # Detect total pages
        if page_num == 1:
            m = re.search(r'"pageCount":"(\d+)"', html)
            if m:
                total_pages = int(m.group(1))
                actual_max = min(total_pages, max_pages)
                if actual_max < max_pages:
                    max_pages = actual_max
                    print(f"  (实际共{total_pages}页，取{max_pages}页)", flush=True)
        
        all_items.extend(items)
        if page_num >= max_pages:
            break
    
    print(f"\n列表共 {len(all_items)} 条，开始获取详情...", flush=True)
    
    detailed = []
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:30]}...", flush=True)
        html = get_page(item['url'], session)
        if not html:
            print(f"    -> 详情获取失败，跳过", flush=True)
            continue
        
        title, date, content, summary, attachments = parse_detail(html, item['url'])
        item['title'] = title or item['title']
        item['date'] = date or item['date']
        item['content'] = content
        item['summary'] = summary
        item['attachments'] = attachments
        detailed.append(item)
        
        if (i + 1) % 10 == 0:
            time.sleep(1)
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    count = save_to_db(detailed, conn)
    conn.close()
    
    errors = len(all_items) - len(detailed)
    print(f"\n完成: {count}条, 异常: {errors}")

if __name__ == '__main__':
    main()
