#!/usr/bin/env python3
"""爬虫：交城县人民政府 — 通知公告
www.sx-jc.gov.cn/xxgk/tzgg/
"""

import os
import sys
import re
import time
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
MAX_PAGES_DEFAULT = 56  # 共56页约1104条

# 支持 --pages 参数
import argparse as _AP
_AP_PARSER = _AP.ArgumentParser()
_AP_PARSER.add_argument("--pages", type=int, default=0, help="限制页数")
_AP_ARGS, _ = _AP_PARSER.parse_known_args()
MAX_PAGES = _AP_ARGS.pages if _AP_ARGS.pages > 0 else MAX_PAGES_DEFAULT

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
BASE_URL = "http://www.sx-jc.gov.cn/xxgk/tzgg/"
SITE_NAME = "交城县通知公告"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def get_page_url(page):
    """第1页=index.html, 第N页(N>=2)=index_{N-1}.shtml"""
    if page == 1:
        return BASE_URL
    return f"{BASE_URL}index_{page - 1}.shtml"

def extract_detail(detail_url):
    """提取详情页：标题、日期、来源、正文、附件"""
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        return None, None, None, None, None

    soup = BeautifulSoup(r.text, "html.parser")

    # 标题
    h2 = soup.select_one(".con1 .t h2")
    title = h2.get_text(strip=True) if h2 else ""

    # 日期 + 来源
    pub_date = ""
    source_text = ""
    info_p = soup.select_one(".con1 .t > p")
    if info_p:
        text = info_p.get_text(strip=True)
        # 时间：2026-07-08
        m = re.search(r"时间[：:]\s*(\d{4}-\d{2}-\d{2})", text)
        if m:
            pub_date = m.group(1)
        # 来源： XXX
        m2 = re.search(r"来源[：:]\s*(.+)", text)
        if m2:
            source_text = m2.group(1).strip()

    # 正文
    article = soup.select_one(".con1 .article-con")
    if not article:
        return title, pub_date, source_text, "", ""
    
    # 附件
    attachments = []
    for a_tag in article.find_all("a"):
        href = a_tag.get("href", "")
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href.lower()):
            full_url = urljoin(detail_url, href)
            text = a_tag.get_text(strip=True)
            attachments.append(f"[{text}]({full_url})")
    
    # 提取正文段落和表格
    parts = []
    has_table = bool(article.find("table"))
    
    for tag in article.find_all(["p", "table"]):
        # 跳过table内部的p（防止重复）
        if tag.name == "p" and tag.find_parent("table"):
            continue
        
        if tag.name == "table":
            if has_table:
                parts.append(str(tag))
            continue
        
        # p标签
        text = tag.get_text(strip=True)
        # 折叠内部换行
        text = re.sub(r"\n+", "", text)
        if text:
            parts.append(text)
    
    if not parts:
        # 回退：取所有div文本
        div = article.find("div")
        if div:
            text = re.sub(r"\n+", "", div.get_text(strip=True))
            if text:
                parts.append(text)
    
    content = "\n\n".join(parts)
    attachments_str = "\n".join(attachments) if attachments else ""
    
    # 如果正文为空但有附件
    if not content.strip() and attachments:
        content = f"[{title}]({detail_url})\n\n{attachments_str}"
    
    return title, pub_date, source_text, content, attachments_str

def crawl():
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    cur = conn.cursor()
    
    total = 0
    for page in range(1, MAX_PAGES + 1):
        url = get_page_url(page)
        print(f"  第{page}页: {url}")
        
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERROR] {e}")
            time.sleep(2)
            continue
        
        soup = BeautifulSoup(r.text, "html.parser")
        table = soup.select_one("table.right_cont_table")
        if not table:
            print(f"  [WARN] 未找到列表table")
            # 检查是否有数据
            items = []
        else:
            rows = table.select("tbody tr")
            items = []
            for row in rows:
                a_tag = row.select_one("td.text-left a")
                time_tag = row.select_one("td span.pubtime")
                if not a_tag or not time_tag:
                    continue
                href = a_tag.get("href", "")
                title = a_tag.get("title", "") or a_tag.get_text(strip=True)
                pub_date = time_tag.get_text(strip=True)
                if not href:
                    continue
                detail_url = urljoin(url, href)
                items.append((title, pub_date, detail_url))
        
        if not items:
            print(f"  第{page}页无数据，终止")
            break
        
        for title, list_date, detail_url in items:
            print(f"    提取: {title[:40]}...")
            # 从详情页获取完整信息
            full_title, pub_date, source_text, content, attachments_str = extract_detail(detail_url)
            
            if not full_title:
                full_title = title
            if not pub_date:
                pub_date = list_date
            
            try:
                cur.execute("""
                    INSERT OR REPLACE INTO gov_raw
                    (page_url, site_name, title, publish_date, content, summary, attachments, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                """, (
                    detail_url,
                    SITE_NAME,
                    full_title,
                    pub_date,
                    content,
                    source_text,
                    attachments_str,
                    0 - int(pub_date.replace("-", "") + "0000") if pub_date else 0
                ))
                total += 1
            except Exception as e:
                print(f"    [DB ERROR] {e}")
            
            time.sleep(0.3)  # 礼貌延时
        
        conn.commit()
        print(f"  第{page}页完成，累计{total}条")
        time.sleep(0.5)
    
    conn.close()
    print(f"\n总计: {total}条")

if __name__ == "__main__":
    crawl()
