#!/usr/bin/env python3
"""
平定县人民政府 - 通知公告 爬虫
http://www.pd.gov.cn/ywdt/ggtz/
TRS CMS, 16页, 18条/页
仅爬前5页 + 日跑增量
"""

import re, sys, time, os
import json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SITE_NAME = "平定县"
BASE_URL = "http://www.pd.gov.cn/ywdt/ggtz"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Host": "www.pd.gov.cn",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
# 该站点需要指定DNS
DNS_SERVER = "114.114.114.114"
TIMEOUT = 30
RETRIES = 3
DB_PATH = "/mnt/data/search.db"


def ensure_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            page_url TEXT UNIQUE,
            title TEXT,
            content TEXT,
            publish_date TEXT,
            site_name TEXT,
            summary TEXT,
            attachments TEXT,
            date_rank TEXT
        )
    """)
    conn.commit()
    conn.close()


def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            page_url TEXT UNIQUE,
            title TEXT,
            content TEXT,
            publish_date TEXT,
            site_name TEXT,
            summary TEXT,
            attachments TEXT,
            date_rank TEXT
        )
    """)
    imported = 0
    for item in items:
        try:
            conn.execute("""
                INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_pd_ggtz.py')
            """, (
                item["page_url"], item["title"], item["content"], item["publish_date"],
                item["site_name"], item["summary"], item["attachments"], item["date_rank"]
            ))
            imported += 1
        except Exception as e:
            print(f"  入库失败: {item['page_url'][:50]} - {e}", file=sys.stderr)
    conn.commit()
    conn.close()
    return imported


def fetch_url(url, stream=False):
    """用指定DNS解析域名"""
    for attempt in range(RETRIES):
        try:
            # pd.gov.cn 解析到 124.166.241.60
            resolved_url = url.replace("www.pd.gov.cn", "124.166.241.60")
            r = requests.get(resolved_url, headers=HEADERS, timeout=TIMEOUT, stream=stream)
            if r.status_code == 200:
                return r
            elif r.status_code == 404:
                return None
        except requests.RequestException:
            if attempt < RETRIES - 1:
                time.sleep(2)
    return None


def clean_title(title):
    if not title:
        return ""
    title = re.sub(r'[-–—]\s*平定县[人民政政府]*\s*', '', title).strip()
    title = re.sub(r'\s+', ' ', title)
    return title


def extract_date_from_text(text):
    m = re.search(r'(\d{4}[-/.]\d{1,2}[-/.]\d{1,2})', str(text))
    if m:
        return m.group(1).replace('/', '-').replace('.', '-')
    return ""


def fetch_detail(url):
    r = fetch_url(url)
    if not r:
        return None
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')

    # 标题：.leader-name
    title = ""
    h3 = soup.find('h3', class_='leader-name')
    if h3:
        title = h3.get_text(strip=True)
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            title = clean_title(title_tag.get_text(strip=True))

    # 日期：.property span
    pub_date = ""
    prop = soup.find('div', class_='property')
    if prop:
        spans = prop.find_all('span')
        for s in spans:
            txt = s.get_text(strip=True)
            m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', txt)
            if m:
                pub_date = m.group(1).replace('/', '-')
                break
    if not pub_date:
        pub_date = extract_date_from_text(url)

    # 正文：.view.TRS_UEDITOR
    content_html = ""
    view = soup.select_one('.view.TRS_UEDITOR, .TRS_UEDITOR, .leader-text .view, div.view')
    if not view:
        view = soup.select_one('.leader-text .view')
    if view:
        parts = []
        for child in view.children:
            if child.name == 'p':
                inner = child.get_text(strip=True)
                if not inner or inner.strip() in ('&nbsp;', ''):
                    continue
                parts.append(inner)
            elif child.name == 'table':
                parts.append(str(child))
        content_html = '\n\n'.join(parts)

    # 附件
    attachments = []
    if view:
        for a in view.find_all('a'):
            href = a.get('href', '')
            if href and href.endswith(('.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar')):
                full_url = urljoin(url, href)
                attachments.append(full_url)

    # 摘要
    summary = ""
    if content_html:
        clean_text = re.sub(r'<[^>]+>', '', content_html)
        summary = clean_text[:200]

    return {
        "title": title,
        "pub_date": pub_date,
        "content": content_html,
        "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        "summary": summary,
    }


def fetch_list_page(page_num):
    """获取列表页"""
    if page_num == 1:
        url = f"{BASE_URL}/"
    else:
        url = f"{BASE_URL}/index_{page_num - 1}.html"

    r = fetch_url(url)
    if not r:
        return []
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')

    items = []
    list_div = soup.find('div', class_='common-list')
    if not list_div:
        return []

    for ul in list_div.find_all('ul'):
        for li in ul.find_all('li', recursive=False):
            a = li.find('a')
            if not a:
                continue
            href = a.get('href', '')
            if not href:
                continue
            detail_url = urljoin(BASE_URL + "/", href)
            title = a.get('title', '') or a.get_text(strip=True)
            span = li.find('span')
            date_str = span.get_text(strip=True) if span else ""
            items.append({
                "url": detail_url,
                "title": title,
                "list_date": date_str,
            })
    return items


def fetch_all_items(max_pages=0):
    all_items = []

    # 先获取总页数
    r = fetch_url(f"{BASE_URL}/")
    if not r:
        print("无法访问首页")
        return all_items

    r.encoding = 'utf-8'
    # createPageHTML(16,0,"index","html")
    page_match = re.search(r'createPageHTML\((\d+)', r.text)
    total_pages = int(page_match.group(1)) if page_match else 1
    print(f"总页数: {total_pages}")

    pages_to_fetch = total_pages if max_pages == 0 else min(max_pages, total_pages)
    print(f"将爬取: {pages_to_fetch} 页")

    fetched = 0
    empty_body = 0
    total_seg = 0
    total_attach = 0

    for page in range(1, pages_to_fetch + 1):
        print(f"\n--- 第 {page}/{pages_to_fetch} 页 ---")
        list_items = fetch_list_page(page)
        if not list_items:
            print("  (空)")
            continue
        print(f"  列表项: {len(list_items)} 条")

        for item in list_items:
            fetched += 1
            detail = fetch_detail(item["url"])
            if not detail or not detail["content"]:
                print(f"  [{fetched}] {item['title'][:40]}...")
                print(f"    ⚠️ 空正文")
                empty_body += 1
                detail = detail or {"title": item["title"], "pub_date": item["list_date"],
                                    "content": "", "attachments": "", "summary": ""}
            else:
                seg_count = len(re.findall(r'\n\n', detail["content"])) + 1
                total_seg += seg_count
                attach_count = len(json.loads(detail["attachments"])) if detail.get("attachments") else 0
                if attach_count > 0:
                    total_attach += 1
                date_str = detail.get("pub_date") or item.get("list_date") or ""
                print(f"  [{fetched}] {item['title'][:40]}...")
                print(f"    seg={seg_count} | attach={'YES' if attach_count > 0 else 'no'} | date={date_str}")

            all_items.append({
                "page_url": item["url"],
                "title": detail.get("title", item["title"]),
                "content": detail.get("content", ""),
                "publish_date": detail.get("pub_date", item["list_date"]),
                "site_name": SITE_NAME,
                "summary": detail.get("summary", ""),
                "attachments": detail.get("attachments", ""),
                "date_rank": detail.get("pub_date", item["list_date"]),
            })

    perc = 100 - (empty_body / fetched * 100) if fetched else 0
    print(f"\n===== DONE =====")
    print(f"Total: {fetched}")
    print(f"With body: {fetched - empty_body} ({perc:.1f}%)")
    if fetched - empty_body > 0:
        print(f"Avg seg: {total_seg/(fetched-empty_body):.1f}")
    print(f"Attachments: {total_attach}")

    return all_items


def main():
    import argparse
    parser = argparse.ArgumentParser(description='平定县通知公告')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数, 0=全部')
    parser.add_argument('--skip-db', action='store_true', help='跳过入库')
    args = parser.parse_args()

    print(f"===== {SITE_NAME} 爬取 =====")

    all_items = fetch_all_items(max_pages=args.pages if args.pages > 0 else 0)
    if not all_items:
        print("无数据")
        return

    if not args.skip_db:
        ensure_db()
        imported = save_to_db(all_items)
        print(f"入库: {imported} 条")
    else:
        print("跳过入库")


if __name__ == '__main__':
    main()
