#!/usr/bin/env python3
"""
定远县人民政府 - 通知公告 爬虫
http://www.dingyuan.gov.cn/zwdt/tzgg/
"""
import requests
import re
import sys
import json
import time
import sqlite3
import os
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE_URL = "https://www.dingyuan.gov.cn"
SITE_NAME = "定远县人民政府"
COLUMN = "通知公告"
MAX_PAGES = 5
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 30

def get_page(url, session):
    for attempt in range(3):
        try:
            r = session.get(url, headers=HEADERS, timeout=TIMEOUT)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if attempt < 2:
                time.sleep(3)
    return None

def parse_list(html):
    """Parse list page, return items"""
    items = re.findall(
        r'<span class="right date">([^<]+)</span>\s*<a href="([^"]+)"[^>]*title="([^"]+)"',
        html
    )
    result = []
    for date, href, title in items:
        result.append({
            'url': href,
            'title': title.strip(),
            'date': date.strip(),
        })
    return result

def parse_detail(html, url):
    """Parse detail page"""
    soup = BeautifulSoup(html, 'html.parser')
    
    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()
    
    date = ''
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date and meta_date.get('content'):
        date = meta_date['content'].strip()
    
    source = ''
    meta_source = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta_source and meta_source.get('content'):
        source = meta_source['content'].strip()
    
    # Content from div.lmcontent
    content_div = soup.find('div', class_='lmcontent') or soup.find('div', id='J_content')
    content_parts = []
    attachments = []
    total_text_len = 0
    found_images = False
    
    if content_div:
        for img in content_div.find_all('img'):
            src = img.get('src', '')
            if src and not src.startswith('http'):
                # Make URL absolute
                src = 'https://www.dingyuan.gov.cn' + src
            alt = img.get('alt', '')
            found_images = True
            content_parts.append(f'![{alt}]({src})')
        
        for elem in content_div.find_all(['p', 'table']):
            tag = elem.name
            if tag == 'table':
                content_parts.append(str(elem))
                total_text_len += len(elem.get_text(strip=True))
            elif tag == 'p':
                text = elem.get_text(strip=True)
                if text:
                    content_parts.append(text)
                    total_text_len += len(text)
        
        for a in content_div.find_all('a', href=True):
            href = a['href']
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                if not href.startswith('http'):
                    href = 'https://www.dingyuan.gov.cn' + href
                name = a.get_text(strip=True) or href.split('/')[-1]
                attachments.append({'name': name, 'url': href})
    
    if total_text_len < 20:
        fallback = f"[{title}]({url})"
        if found_images:
            content_parts = [fallback] + [p for p in content_parts if p.startswith('![')]
        else:
            content_parts = [fallback]
            pdf_links = [a for a in attachments if re.search(r'\.pdf$', a['url'], re.I)]
            if pdf_links:
                for pdf in pdf_links:
                    content_parts.append(f"[PDF附件: {pdf['name']}]({pdf['url']})")
    
    content = '\n\n'.join(content_parts)
    summary = re.sub(r'\s+', ' ', content[:200]).strip() if content else title
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    
    return title, date, content, summary, attachments_json

def save_to_db(items, conn):
    cursor = conn.cursor()
    count = 0
    for item in items:
        publish_date = item.get('date', '')
        if publish_date:
            try:
                dt = datetime.strptime(publish_date[:10], '%Y-%m-%d')
                if dt > datetime.now().replace(hour=0, minute=0, second=0, microsecond=0) + __import__('datetime').timedelta(days=30):
                    publish_date = ''
            except:
                publish_date = ''
        
        cursor.execute("""
            INSERT OR REPLACE INTO gov_raw (page_url, site_name, title, publish_date, summary, content, attachments, status, group_name, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, 'active', ?, 'crawl_dingyuan.py')
        """, (
            item['url'], SITE_NAME, item['title'],
            publish_date, item.get('summary', ''),
            item.get('content', ''), item.get('attachments', '[]'),
            SITE_NAME
        ))
        count += 1
    conn.commit()
    return count

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES)
    args = parser.parse_args()
    max_pages = args.max_pages
    
    session = requests.Session()
    all_items = []
    
    # Page 1: use index.html (pre-rendered)
    print(f"[列表页] 第1页: {BASE_URL}/zwdt/tzgg/index.html", flush=True)
    html = get_page(f"{BASE_URL}/zwdt/tzgg/index.html", session)
    if html:
        items = parse_list(html)
        print(f"  -> {len(items)}条", flush=True)
        all_items.extend(items)
        m = re.search(r'pageCount:(\d+)', html)
        if m:
            print(f"  (共{m.group(1)}页)", flush=True)
    
    # Pages 2+: use API endpoint
    for page_num in range(2, max_pages + 1):
        url = f'https://www.dingyuan.gov.cn/content/column/160686244?pageIndex={page_num-1}&pageSize=20'
        print(f"[列表页] 第{page_num}页: {url}", flush=True)
        html = get_page(url, session)
        if not html:
            print(f"  -> 获取失败", flush=True)
            break
        
        items = parse_list(html)
        if not items:
            print(f"  -> 无数据", flush=True)
            break
        
        print(f"  -> {len(items)}条", flush=True)
        all_items.extend(items)
        
        if page_num == 1:
            m = re.search(r'pageCount:(\d+)', html)
            if m:
                page_count = int(m.group(1))
                print(f"  (共{page_count}页)", flush=True)
    
    print(f"\n列表共 {len(all_items)} 条，开始获取详情...", flush=True)
    
    detailed = []
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:30]}...", flush=True)
        html = get_page(item['url'], session)
        if not html:
            print(f"    -> 详情获取失败，跳过", flush=True)
            continue
        
        title, date, content, summary, attachments = parse_detail(html, item['url'])
        item['title'] = title or item['title']
        item['date'] = date or item['date']
        item['content'] = content
        item['summary'] = summary
        item['attachments'] = attachments
        detailed.append(item)
        
        if (i + 1) % 10 == 0:
            time.sleep(1)
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    count = save_to_db(detailed, conn)
    conn.close()
    
    errors = len(all_items) - len(detailed)
    print(f"\n完成: {count}条, 异常: {errors}")

if __name__ == '__main__':
    main()
