#!/usr/bin/env python3
"""
普定县人民政府 - 通知公告 爬虫
https://www.aspd.gov.cn/xwzx/tzgg/
"""
import re
import sys
import json
import time
import argparse
import requests
from bs4 import BeautifulSoup

SITE_NAME = "普定县-通知公告"
GROUP = "贵州"
BASE_URL = "https://www.aspd.gov.cn/xwzx/tzgg/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
DB_PATH = "/root/search.db"

def fetch(url, max_retries=3):
    for attempt in range(max_retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
            print(f"  HTTP {r.status_code} for {url}, retry {attempt+1}")
        except Exception as e:
            print(f"  Error: {e}, retry {attempt+1}")
        time.sleep(2)
    return None

def parse_list_page(html):
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    news_list = soup.find(class_="NewsList")
    if not news_list:
        print("  WARNING: NewsList not found")
        return items
    for a in news_list.find_all('a', href=True):
        title = a.get_text(strip=True)
        href = a['href']
        if title and href and '/tzgg/' in href:
            if not href.startswith('http'):
                href = requests.compat.urljoin(BASE_URL, href)
            items.append((title, href))
    return items

def parse_detail(html, url):
    soup = BeautifulSoup(html, 'html.parser')
    
    title_el = soup.find(class_="ArticleTitle")
    title = title_el.get_text(strip=True) if title_el else ""
    if not title:
        t = soup.find('title')
        if t:
            title = t.get_text(strip=True)
    
    date_match = re.search(r"var pubdata='(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})", html)
    publish_date = date_match.group(1) if date_match else ""
    
    content_el = soup.find(class_=lambda c: c and 'trs_editor_view' in c)
    if not content_el:
        content_el = soup.find(class_="nry")
    
    content = ""
    attachments = []
    if content_el:
        for a in content_el.find_all('a', href=True):
            href = a['href']
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|xlsm)$', href, re.I):
                if not href.startswith('http'):
                    href = requests.compat.urljoin(url, href)
                attach_title = a.get_text(strip=True) or href.split('/')[-1]
                attachments.append({"title": attach_title, "url": href})
        
        for tag in content_el.find_all(['script', 'style']):
            tag.decompose()
        # Build content paragraph-by-paragraph: each <p> = one paragraph
        # Use get_text() with no separator within each block element
        parts = []
        for child in content_el.children:
            child_text = child.get_text(strip=True)
            if child_text:
                if child.name == 'p':
                    parts.append(child_text)
                elif child.name in ('div', 'section'):
                    # Recursively handle inner <p> tags
                    for inner_p in child.find_all('p', recursive=False):
                        inner_text = inner_p.get_text(strip=True)
                        if inner_text:
                            parts.append(inner_text)
                    if not child.find_all('p', recursive=False):
                        parts.append(child_text)
                elif child.name == 'table':
                    # Extract table as text
                    rows = child.find_all('tr')
                    table_lines = []
                    for row in rows:
                        cells = [cell.get_text(strip=True) for cell in row.find_all(['td', 'th'])]
                        if cells:
                            table_lines.append(' | '.join(cells))
                    if table_lines:
                        parts.append('\n'.join(table_lines))
                elif child.name is None:
                    # Direct text node
                    parts.append(child_text)
                else:
                    parts.append(child_text)
        content = '\n\n'.join(parts)
    
    # PDF empty content fallback
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>'
        if attachments:
            content += "\n\n附件：\n" + "\n".join(f"[{a['title']}]({a['url']})" for a in attachments)
    
    return {
        "title": title,
        "publish_date": publish_date,
        "content": content,
        "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        "page_url": url,
    }

def crawl_pages(start_page, end_page):
    all_items = []
    for page in range(start_page, end_page + 1):
        if page == 1:
            url = BASE_URL
        else:
            url = f"{BASE_URL}index_{page-1}.html"
        
        print(f"  Page {page}/{end_page}: {url}")
        html = fetch(url)
        if not html:
            print(f"  FAILED to fetch page {page}")
            continue
        
        items = parse_list_page(html)
        if not items:
            print(f"  No items on page {page}, stopping")
            break
        
        print(f"  Found {len(items)} items")
        
        for i, (title, item_url) in enumerate(items):
            print(f"    [{i+1}/{len(items)}] {title[:50]}...")
            detail_html = fetch(item_url)
            if not detail_html:
                print(f"      FAILED detail")
                continue
            
            detail = parse_detail(detail_html, item_url)
            detail['site_name'] = SITE_NAME
            detail['group'] = GROUP
            
            all_items.append(detail)
            time.sleep(1)
        
        time.sleep(1)
    
    return all_items

def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=5000")
    c = conn.cursor()
    
    inserted = 0
    skipped = 0
    for item in items:
        try:
            c.execute("""
                INSERT OR IGNORE INTO gov_raw 
                (site_name, page_url, title, publish_date, summary, content, category, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)
            """, (
                item['site_name'],
                item['page_url'],
                item['title'],
                item.get('publish_date', ''),
                '',
                item.get('content', ''),
                item.get('group', ''),
                item.get('attachments', ''),
            ))
            if c.rowcount > 0:
                inserted += 1
            else:
                skipped += 1
        except Exception as e:
            print(f"    DB Error: {e}")
    
    conn.commit()
    conn.close()
    return inserted, skipped

def main():
    parser = argparse.ArgumentParser(description=f"爬取{SITE_NAME}")
    parser.add_argument('--test', action='store_true', help='测试模式：爬第1页')
    parser.add_argument('--incremental', action='store_true', help='增量模式：只爬第1页')
    parser.add_argument('--max-pages', type=int, default=5, help='最大页数')
    args = parser.parse_args()
    
    if args.test or args.incremental:
        end = 1
        mode = "测试" if args.test else "增量"
    else:
        end = args.max_pages
        mode = "全量(前%d页)" % end
    
    print(f"=== {mode}模式: {SITE_NAME} ===")
    items = crawl_pages(1, end)
    
    if not items:
        print("No items collected")
        return
    
    print(f"\n共获取 {len(items)} 条数据")
    inserted, skipped = save_to_db(items)
    print(f"入库: 新增 {inserted}, 跳过 {skipped}")
    
    print(f"\n=== 前5条预览 ===")
    for item in items[:5]:
        print(f"  [{item['publish_date']}] {item['title'][:60]}")
        if item['attachments']:
            atts = json.loads(item['attachments'])
            if atts:
                print(f"    附件: {', '.join(a['title'] for a in atts[:3])}")

if __name__ == '__main__':
    main()
