#!/usr/bin/env python3
"""
芮城县人民政府 - 公示公告 爬虫
http://www.rcx.gov.cn/zfxxgk/gsgg/index.shtml
自定义CMS，4页，20条/页
"""

import re, sys, time, os
import json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime

BASE_URL = "http://www.rcx.gov.cn/zfxxgk/gsgg"
SITE_NAME = "芮城县"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}
TIMEOUT = 30
RETRIES = 3
DB_PATH = "/mnt/data/search.db"


def ensure_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            page_url TEXT UNIQUE,
            title TEXT,
            content TEXT,
            publish_date TEXT,
            site_name TEXT,
            summary TEXT,
            attachments TEXT,
            date_rank TEXT
        )
    """)
    conn.commit()
    conn.close()


def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            page_url TEXT UNIQUE,
            title TEXT,
            content TEXT,
            publish_date TEXT,
            site_name TEXT,
            summary TEXT,
            attachments TEXT,
            date_rank TEXT
        )
    """)
    imported = 0
    for item in items:
        try:
            conn.execute("""
                INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_rcx_gsgg.py')
            """, (
                item["page_url"], item["title"], item["content"], item["publish_date"],
                item["site_name"], item["summary"], item["attachments"], item["date_rank"]
            ))
            imported += 1
        except Exception as e:
            print(f"  入库失败: {item['page_url'][:50]} - {e}", file=sys.stderr)
    conn.commit()
    conn.close()
    return imported


def fetch_url(url, stream=False):
    for attempt in range(RETRIES):
        try:
            r = requests.get(url, headers=HEADERS, timeout=TIMEOUT, stream=stream)
            if r.status_code == 200:
                return r
            elif r.status_code == 404:
                return None
        except requests.RequestException:
            if attempt < RETRIES - 1:
                time.sleep(2)
    return None


def clean_title(title):
    if not title:
        return ""
    title = re.sub(r'[-–—]\s*芮城县人民政府[门门户网站]*\s*$', '', title).strip()
    return title


def extract_date_from_text(text):
    m = re.search(r'(\d{4}[-/.]\d{1,2}[-/.]\d{1,2})', str(text))
    if m:
        return m.group(1).replace('/', '-').replace('.', '-')
    return ""


def fetch_detail(url):
    r = fetch_url(url)
    if not r:
        return None
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')

    # 标题
    title = ""
    title_tag = soup.find('title')
    if title_tag:
        title = clean_title(title_tag.get_text(strip=True))
    if not title:
        meta = soup.find('meta', attrs={"name": "ArticleTitle"})
        if meta and meta.get('content'):
            title = meta['content'].strip()
    if not title:
        h3 = soup.find('h3', class_=lambda c: c and 'f36' in c)
        if h3:
            title = h3.get_text(strip=True)

    # 日期
    pub_date = ""
    meta_date = soup.find('meta', attrs={"name": "PubDate"})
    if meta_date and meta_date.get('content'):
        pub_date = meta_date['content'].strip()[:10]
    if not pub_date:
        p_date = soup.find('p', class_=lambda c: c and 'date' in c)
        if p_date:
            pub_date = extract_date_from_text(p_date.get_text())
    if not pub_date:
        pub_date = extract_date_from_text(url)

    # 正文
    content_html = ""
    zoom = soup.find('div', id='Zoom')
    if zoom:
        parts = []
        for child in zoom.children:
            if child.name == 'p':
                inner = child.get_text(strip=True)
                if not inner or inner.strip() in ('&nbsp;', ''):
                    continue
                parts.append(inner)
            elif child.name == 'table':
                parts.append(str(child))
        content_html = '\n\n'.join(parts)

    # 附件
    attachments = []
    if zoom:
        for a in zoom.find_all('a'):
            href = a.get('href', '')
            if href and href.endswith(('.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar')):
                full_url = urljoin(url, href)
                attachments.append(full_url)

    # 摘要
    summary = ""
    if content_html:
        clean_text = re.sub(r'<[^>]+>', '', content_html)
        summary = clean_text[:200]

    return {
        "title": title,
        "pub_date": pub_date,
        "content": content_html,
        "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        "summary": summary,
    }


def fetch_list_page(page_num):
    if page_num == 1:
        url = f"{BASE_URL}/index.shtml"
    else:
        url = f"{BASE_URL}/index_{page_num}.shtml"

    r = fetch_url(url)
    if not r:
        return []
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')

    items = []
    ul = soup.select_one('div.rightCon div.box_list ul.pd20')
    if not ul:
        return []

    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '')
        if not href:
            continue
        detail_url = urljoin(BASE_URL + "/", href)
        title = a.get('title', '') or a.get_text(strip=True)
        span = li.find('span')
        date_str = span.get_text(strip=True) if span else ""
        items.append({
            "url": detail_url,
            "title": title,
            "list_date": date_str,
        })
    return items


def fetch_all_items(max_pages=0):
    all_items = []

    r = fetch_url(f"{BASE_URL}/index.shtml")
    if not r:
        print("无法访问首页")
        return all_items

    r.encoding = 'utf-8'
    page_match = re.search(r'pageCount["\']\s*:\s*["\'](\d+)["\']', r.text)
    total_pages = int(page_match.group(1)) if page_match else 1
    print(f"总页数: {total_pages}")

    pages_to_fetch = total_pages if max_pages == 0 else min(max_pages, total_pages)
    print(f"将爬取: {pages_to_fetch} 页")

    fetched = 0
    empty_body = 0
    total_seg = 0
    total_attach = 0

    for page in range(1, pages_to_fetch + 1):
        print(f"\n--- 第 {page}/{pages_to_fetch} 页 ---")
        list_items = fetch_list_page(page)
        if not list_items:
            print("  (空)")
            continue
        print(f"  列表项: {len(list_items)} 条")

        for item in list_items:
            fetched += 1
            detail = fetch_detail(item["url"])
            if not detail or not detail["content"]:
                print(f"  [{fetched}] {item['title'][:40]}...")
                print(f"    ⚠️ 空正文")
                empty_body += 1
                detail = detail or {"title": item["title"], "pub_date": item["list_date"],
                                    "content": "", "attachments": "", "summary": ""}
            else:
                seg_count = len(re.findall(r'\n\n', detail["content"])) + 1
                total_seg += seg_count
                attach_count = len(json.loads(detail["attachments"])) if detail.get("attachments") else 0
                if attach_count > 0:
                    total_attach += 1
                date_str = detail.get("pub_date") or item.get("list_date") or ""
                print(f"  [{fetched}] {item['title'][:40]}...")
                print(f"    seg={seg_count} | attach={'YES' if attach_count > 0 else 'no'} | date={date_str}")

            all_items.append({
                "page_url": item["url"],
                "title": detail.get("title", item["title"]),
                "content": detail.get("content", ""),
                "publish_date": detail.get("pub_date", item["list_date"]),
                "site_name": SITE_NAME,
                "summary": detail.get("summary", ""),
                "attachments": detail.get("attachments", ""),
                "date_rank": detail.get("pub_date", item["list_date"]),
            })

    perc = 100 - (empty_body / fetched * 100) if fetched else 0
    print(f"\n===== DONE =====")
    print(f"Total: {fetched}")
    print(f"With body: {fetched - empty_body} ({perc:.1f}%)")
    if fetched - empty_body > 0:
        print(f"Avg seg: {total_seg/(fetched-empty_body):.1f}")
    print(f"Attachments: {total_attach}")

    return all_items


def main():
    import argparse
    parser = argparse.ArgumentParser(description='芮城县公示公告')
    parser.add_argument('--pages', type=int, default=0, help='爬取页数, 0=全部')
    parser.add_argument('--skip-db', action='store_true', help='跳过入库')
    args = parser.parse_args()

    print(f"===== {SITE_NAME} 爬取 =====")

    all_items = fetch_all_items(max_pages=args.pages if args.pages > 0 else 0)
    if not all_items:
        print("无数据")
        return

    if not args.skip_db:
        ensure_db()
        imported = save_to_db(all_items)
        print(f"入库: {imported} 条")
    else:
        print("跳过入库")


if __name__ == '__main__':
    main()
