#!/usr/bin/env python3
"""水城区人民政府网 — 通知公告爬虫
http://www.shuicheng.gov.cn/newsite/zwdt/tzgg/
Custom CMS, index_N.html 分页, 8条/页, 37页 ~291条
"""
import os, sys, re, json, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
from urllib.parse import urljoin

import argparse as _AP
_AP_PARSER = _AP.ArgumentParser()
_AP_PARSER.add_argument("--pages", type=int, default=0, help="限制页数(0=不限,用MAX_PAGES)")
_AP_ARGS = _AP_PARSER.parse_args()
_MAX_PAGES_ARG = _AP_ARGS.pages

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.shuicheng.gov.cn/newsite/zwdt/tzgg/"
SITE_NAME = "水城区人民政府网-通知公告"
THRESHOLD = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
MAX_PAGES = 40

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Referer": "http://www.shuicheng.gov.cn/",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

session = requests.Session()
session.headers.update(HEADERS)
import sqlite3

def store_item(title, date_str, content, page_url):
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw 
            (title, publish_date, content, page_url, site_name)
            VALUES (?, ?, ?, ?, ?)
        """, (title.strip(), date_str, content.strip(), page_url.strip(), SITE_NAME))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected
    except Exception as e:
        print(f"[ERROR] 写入失败 {page_url}: {e}")
        return 0

def parse_list_page(html, page_num):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    news_list = soup.find(class_="news-list")
    if not news_list:
        return items, False
    li_tags = news_list.find_all("li")
    for li in li_tags:
        a_tag = li.find("a")
        if not a_tag:
            continue
        href = a_tag.get("href", "")
        if "tzgg" not in href:
            continue
        title = a_tag.get_text(strip=True)
        if not title or len(title) < 5:
            continue
        date_str = ""
        date_span = li.find("span", class_="fl")
        if date_span:
            date_match = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", date_span.get_text())
            if date_match:
                date_str = date_match.group(1)
        if href.startswith("http"):
            page_url = href
        elif href.startswith("/"):
            page_url = "http://www.shuicheng.gov.cn" + href
        else:
            page_url = urljoin(BASE_URL, href)
        items.append((title, date_str, page_url))
    return items, True

def parse_detail_page(html, page_url):
    soup = BeautifulSoup(html, "html.parser")
    
    # Title
    title_tag = soup.select_one("div.detailsHead h1.title")
    if not title_tag:
        title_tag = soup.find("title")
        title = re.sub(r'^水城区人民政府网-\s*\S+\s+', '', title_tag.get_text(strip=True)) if title_tag else "N/A"
    else:
        title = title_tag.get_text(strip=True)
    
    # Date
    date_str = ""
    date_span = soup.select_one("div.detailsHead p.attribute span")
    if date_span:
        date_match = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", date_span.get_text())
        if date_match:
            date_str = date_match.group(1)
    
    # Content
    content_div = soup.find(class_="TRS_UEDITOR") or soup.find(class_="trs_editor_view")
    if not content_div:
        content_div = soup.find(class_="detailsMain")
    
    content = ""
    if content_div:
        # Fix relative URLs in the content
        soup_c = BeautifulSoup(str(content_div), "html.parser")
        for a in soup_c.find_all("a", href=True):
            href = a["href"]
            if href.startswith("./"):
                a["href"] = urljoin(page_url, href)
            elif href.startswith("/"):
                a["href"] = "http://www.shuicheng.gov.cn" + href
        for img in soup_c.find_all("img", src=True):
            src = img["src"]
            if src.startswith("/"):
                img["src"] = "http://www.shuicheng.gov.cn" + src
            elif src.startswith("./"):
                img["src"] = urljoin(page_url, src)
        
        paras = []
        for tag in soup_c.find_all(["p", "div", "td"]):
            text = tag.get_text(strip=True)
            if text:
                paras.append(text)
        content = "\n\n".join(paras) if paras else soup_c.get_text(separator="\n", strip=True)
    
    # Attachments
    attachments = []
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|caj)$', href, re.I):
            text = a.get_text(strip=True)
            if href.startswith("./") or href.startswith("/"):
                full_url = urljoin(page_url, href) if href.startswith("./") else "http://www.shuicheng.gov.cn" + href
            elif href.startswith("http"):
                full_url = href
            else:
                full_url = urljoin(page_url, href)
            attachments.append(f"[{text}]({full_url})")
    if attachments:
        content += "\n\n---\n附件：\n" + "\n".join(attachments)
    
    return title, date_str, content

def main():
    print(f"[INFO] 开始爬取: {SITE_NAME}")
    print(f"[INFO] 阈值: {THRESHOLD}, 最大页数: {MAX_PAGES}")
    
    total_new = 0
    stop_crawl = False

    try:
        _cx = sqlite3.connect(DB_PATH, timeout=60)
        existing = set(r[0] for r in _cx.execute(
            "SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall())
        _cx.close()
    except Exception as e:
        print(f"[WARN] 读取已入库URL失败: {e}")
        existing = set()
    print(f"[INFO] 已入库 {len(existing)} 条URL，重复的直接跳过详情")

    _page_cap = min(MAX_PAGES, _MAX_PAGES_ARG) if _MAX_PAGES_ARG > 0 else MAX_PAGES
    for page_num in range(_page_cap):
        if stop_crawl:
            break
        
        page_url = BASE_URL if page_num == 0 else f"http://www.shuicheng.gov.cn/newsite/zwdt/tzgg/index_{page_num}.html"
        print(f"\n[PAGE] 第{page_num+1}页: {page_url}")
        
        try:
            r = session.get(page_url, timeout=15)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"[ERROR] 请求失败: {e}")
            time.sleep(1)
            continue
        
        if r.status_code != 200:
            print(f"[WARN] HTTP {r.status_code}, 停止")
            break
        
        items, has_data = parse_list_page(r.text, page_num + 1)
        if not has_data:
            print("[INFO] 无news-list, 停止")
            break
        
        print(f"[INFO] {len(items)} 条")
        
        for title, date_str, detail_url in items:
            if date_str and date_str < THRESHOLD:
                print(f"[STOP] {date_str} 超出3年阈值")
                stop_crawl = True
                break
            
            if detail_url in existing:
                print(f"  [-] 已存在: {title[:30]}...")
                continue
            time.sleep(0.3)
            try:
                rd = session.get(detail_url, timeout=15)
                rd.encoding = "utf-8"
            except Exception as e:
                print(f"[ERROR] 详情页请求失败: {e}")
                continue
            
            if rd.status_code != 200:
                continue
            
            detail_title, detail_date, content = parse_detail_page(rd.text, detail_url)
            final_title = detail_title if detail_title and detail_title != "N/A" else title
            final_date = detail_date or date_str
            
            affected = store_item(final_title, final_date, content, detail_url)
            if affected > 0:
                total_new += 1
                print(f"  [+] {final_title[:40]}... ({final_date})")
            else:
                print(f"  [-] 已存在: {final_title[:30]}...")
    
    print(f"\n=== 完成 ===")
    print(f"新增: {total_new}")

if __name__ == "__main__":
    main()
