#!/usr/bin/env python3
"""颍东区人民政府 - 重大行政决策预公开 爬虫
URL: https://www.yd.gov.cn/OpennessContent/showList/1650/192939/page_1.html
CMS: 政府信息公开平台 (Hanweb OpennessTarget)
AJAX列表: /OpennessTarget/1650/192939/page_N.html (> table tr > td.bt + td.cwrq)
详情: /OpennessContent/show/{id}.html (> div.g-detailbox#zoom 正文)
分页: page_2.html, page_3.html (共3页, 35条)
"""

import sys, os, re, time, json, logging, argparse, sqlite3
from datetime import datetime, timedelta
from urllib.parse import urljoin

import requests
from bs4 import BeautifulSoup

BASE_URL = "https://www.yd.gov.cn"
LIST_API = "/OpennessTarget/1650/192939"
SITE_NAME = "颍东区-重大行政决策预公开"
MAX_PAGES = 5
DATE_CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
REQUEST_DELAY = 0.3
DB_PATH = "/root/search.db"

logging.basicConfig(level=logging.INFO, format='[%(asctime)s] %(levelname)s %(message)s', datefmt='%H:%M:%S')
log = logging.getLogger(__name__)

HEADERS = {'User-Agent': 'Mozilla/5.0'}

def init_db():
    db = sqlite3.connect(DB_PATH)
    db.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        site_name TEXT NOT NULL,
        title TEXT NOT NULL,
        page_url TEXT NOT NULL UNIQUE,
        publish_date TEXT,
        source_url TEXT,
        content TEXT,
        crawl_time TEXT DEFAULT (datetime('now','localtime'))
    )""")
    db.commit()
    return db

def record_exists(db, page_url):
    return db.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (page_url,)).fetchone() is not None

def save_record(db, item):
    db.execute("INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, source_url, content) VALUES (?,?,?,?,?,?)",
               (item['site_name'], item['title'], item['page_url'], item.get('publish_date',''), item.get('source_url',''), item.get('content','')))
    return db.cursor().rowcount

def fetch(url):
    try:
        r = requests.get(url, timeout=20, headers=HEADERS)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        log.error(f"请求失败 {url[:60]}: {e}")
        return None

def parse_list_items(html):
    """从AJAX列表HTML提取文章"""
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for tr in soup.select('tr'):
        bt = tr.select_one('td.bt')
        cwrq = tr.select_one('td.cwrq')
        if not bt or not cwrq:
            continue
        a = bt.find('a', href=True)
        if not a:
            continue
        title = a.get_text(strip=True)
        href = a['href']
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)
        date = cwrq.get_text(strip=True)
        items.append({'title': title, 'page_url': href, 'publish_date': date})
    return items

def parse_detail(html, page_url):
    """解析详情页"""
    soup = BeautifulSoup(html, 'html.parser')
    h1 = soup.find('h1')
    title = h1.get_text(strip=True) if h1 else ''
    m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    pub_date = m.group(1) if m else ''
    content_div = soup.select_one('div.g-detailbox')
    content_html = ''
    if content_div:
        for s in content_div.select('script,style'):
            s.decompose()
        content_html = str(content_div).strip()
    return title, pub_date, content_html

def crawl(full=True):
    db = init_db()
    total_saved = 0
    total_skipped = 0
    total_count = 0

    # 第1页
    html1 = fetch(f"{BASE_URL}{LIST_API}/page_1.html")
    if not html1:
        return {"total": 0, "saved": 0, "skipped": 0}

    all_items = parse_list_items(html1)
    log.info(f"第1页: {len(all_items)} 条")

    # 后续页
    pages = MAX_PAGES if full else 1
    for page in range(2, pages + 1):
        time.sleep(REQUEST_DELAY)
        html = fetch(f"{BASE_URL}{LIST_API}/page_{page}.html")
        if not html:
            break
        items = parse_list_items(html)
        if not items:
            break
        log.info(f"第{page}页: {len(items)} 条")
        all_items.extend(items)

    log.info(f"共 {len(all_items)} 条")

    # 详情
    for i, item in enumerate(all_items):
        if item['publish_date'] and item['publish_date'] < DATE_CUTOFF:
            log.info(f"  跳过(日期过旧): {item['title'][:40]} ({item['publish_date']})")
            continue

        if record_exists(db, item['page_url']):
            total_skipped += 1
            continue

        total_count += 1
        log.info(f"  详情 [{total_count}/{len(all_items)}]: {item['title'][:50]}...")
        html = fetch(item['page_url'])
        if html:
            title, pub_date, content = parse_detail(html, item['page_url'])
            item['title'] = title or item['title']
            item['publish_date'] = pub_date or item['publish_date']
            item['content'] = content
            item['source_url'] = "颍东区人民政府"
            item['site_name'] = SITE_NAME

            if save_record(db, item):
                total_saved += 1
            else:
                total_skipped += 1
            db.commit()
        time.sleep(REQUEST_DELAY)

    log.info(f"完成: total={total_count}, saved={total_saved}, skipped={total_skipped}")
    return {"total": total_count, "saved": total_saved, "skipped": total_skipped}

if __name__ == '__main__':
    parser = argparse.ArgumentParser()
    parser.add_argument('--full', action='store_true')
    args = parser.parse_args()
    result = crawl(full=args.full)
    print(json.dumps(result, ensure_ascii=False))
