#!/usr/bin/env python3
"""获嘉县人民政府 - 行政许可和行政处罚 爬虫
列表: <li><span>2026年06月10日</span><a href="...">title</a>
分页: list-1.html → list-8.html
详情: <h1>标题, <div class="mt15 details-box">正文
日期: 从正文文本提取 YYYY-MM-DD
"""

import sys, os, re, time, json, logging, argparse, sqlite3
from datetime import datetime, timedelta
from urllib.parse import urljoin

import requests
from bs4 import BeautifulSoup

BASE_URL = "https://www.huojia.gov.cn/htmls/xingzhengxukehexingzhengchufa"
SITE_NAME = "获嘉县-行政许可和行政处罚"
MAX_PAGES = 20
DATE_CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
REQUEST_DELAY = 0.3
DB_PATH = "/root/search.db"

logging.basicConfig(level=logging.INFO, format='[%(asctime)s] %(levelname)s %(message)s', datefmt='%H:%M:%S')
log = logging.getLogger(__name__)
HEADERS = {'User-Agent': 'Mozilla/5.0'}

def init_db():
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        site_name TEXT NOT NULL,
        title TEXT NOT NULL,
        page_url TEXT NOT NULL UNIQUE,
        publish_date TEXT,
        source_url TEXT,
        content TEXT,
        crawl_time TEXT DEFAULT (datetime('now','localtime'))
    )""")
    db.commit()
    return db

def record_exists(db, page_url):
    return db.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (page_url,)).fetchone() is not None

def save_record(db, item):
    db.execute("INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, source_url, content) VALUES (?,?,?,?,?,?)",
               (item['site_name'], item['title'], item['page_url'], item.get('publish_date',''), item.get('source_url',''), item.get('content','')))
    return db.cursor().rowcount

def fetch(url):
    try:
        r = requests.get(url, timeout=20, headers=HEADERS)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        log.error(f"请求失败 {url[:60]}: {e}")
        return None

def parse_list_items(html):
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for li in soup.find_all('li'):
        span = li.find('span')
        a = li.find('a', href=True)
        if not (span and a and 'xingzhengxukehexingzhengchufa' in a.get('href','')):
            continue
        date_str = span.get_text(strip=True)
        # 转换 2026年06月10日 → 2026-06-10
        m = re.match(r'(\d{4})年(\d{1,2})月(\d{1,2})日', date_str)
        if not m:
            continue
        pub_date = f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
        href = a['href']
        if not href.startswith('http'):
            href = urljoin("https://www.huojia.gov.cn", href)
        title = a.get_text(strip=True)
        items.append({'title': title, 'page_url': href, 'publish_date': pub_date})
    return items

def parse_detail(html, page_url):
    soup = BeautifulSoup(html, 'html.parser')
    h1 = soup.find('h1')
    title = h1.get_text(strip=True) if h1 else ''
    # 日期从文本提取
    m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    pub_date = m.group(1) if m else ''
    content_div = soup.select_one('.details-box')
    content_html = ''
    if content_div:
        for s in content_div.select('script,style'):
            s.decompose()
        content_html = str(content_div).strip()
    return title, pub_date, content_html

def crawl(full=True):
    db = init_db()
    total_saved = total_skipped = total_count = 0
    pages = MAX_PAGES if full else 1

    all_items = []
    for page in range(1, pages + 1):
        url = f"{BASE_URL}/list-{page}.html"
        html = fetch(url)
        if not html: break
        items = parse_list_items(html)
        if not items: break
        dates = [i['publish_date'] for i in items if i['publish_date']]
        if dates and max(dates) < DATE_CUTOFF:
            log.info(f"第{page}页 → 日期过旧，终止")
            break
        log.info(f"第{page}页: {len(items)} 条 [{items[0]['publish_date']}→{items[-1]['publish_date']}]")
        all_items.extend(items)
        time.sleep(REQUEST_DELAY)

    # 去重
    seen = set()
    unique = []
    for item in all_items:
        if item['page_url'] not in seen:
            seen.add(item['page_url'])
            unique.append(item)

    log.info(f"共 {len(unique)} 条")

    for i, item in enumerate(unique):
        if item['publish_date'] and item['publish_date'] < DATE_CUTOFF:
            log.info(f"  跳过(日期过旧): {item['title'][:40]} ({item['publish_date']})")
            continue
        if record_exists(db, item['page_url']):
            total_skipped += 1; continue

        total_count += 1
        log.info(f"  详情 [{total_count}/{len(unique)}]: {item['title'][:50]}...")
        html = fetch(item['page_url'])
        if html:
            title, pub_date, content = parse_detail(html, item['page_url'])
            item['title'] = title or item['title']
            item['publish_date'] = pub_date or item['publish_date']
            item['content'] = content
            item['source_url'] = "获嘉县人民政府"
            item['site_name'] = SITE_NAME
            if save_record(db, item): total_saved += 1
            else: total_skipped += 1
            db.commit()
        time.sleep(REQUEST_DELAY)

    log.info(f"完成: total={total_count}, saved={total_saved}, skipped={total_skipped}")
    return {"total": total_count, "saved": total_saved, "skipped": total_skipped}

if __name__ == '__main__':
    parser = argparse.ArgumentParser()
    parser.add_argument('--full', action='store_true')
    args = parser.parse_args()
    result = crawl(full=args.full)
    print(json.dumps(result, ensure_ascii=False))
