#!/usr/bin/env python3
"""柳州市柳江区人民政府 — 通知公告爬虫
http://www.liujiang.gov.cn/xwzx/ljztgg/
开普云CMS, createPageHTML分页
"""
import os, sys, re, json, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.liujiang.gov.cn/xwzx/ljztgg/"
SITE_NAME = "柳州市柳江区人民政府"
THRESHOLD = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
PAGE_SIZE = 15  # items per page
MAX_PAGES = 200

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Referer": "http://www.liujiang.gov.cn/",
}

session = requests.Session()
session.headers.update(HEADERS)

import sqlite3

def get_existing_urls():
    """获取数据库中已有的URL"""
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        existing = {row[0] for row in c.fetchall()}
        conn.close()
        return existing
    except Exception as e:
        print(f"[WARN] 无法查询已有URL: {e}")
        return set()

def store_item(title, date_str, content, page_url):
    """写入数据库"""
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw 
            (title, publish_date, content, page_url, site_name)
            VALUES (?, ?, ?, ?, ?)
        """, (title.strip(), date_str, content.strip(), page_url.strip(), SITE_NAME))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected
    except Exception as e:
        print(f"[ERROR] 写入失败 {page_url}: {e}")
        return 0

def parse_list_page(html, page_num):
    """解析列表页"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    # PC版本列表项
    li_tags = soup.select("ul.list-group.list-Special > li.list-group-item.PC-SHOW")
    for li in li_tags:
        a_tag = li.find("a")
        date_div = li.find("div", class_="layout-fixed")
        if not a_tag or not date_div:
            continue
        href = a_tag.get("href", "")
        title = a_tag.get_text(strip=True)
        date_str = date_div.get_text(strip=True)
        
        # 构建完整URL
        if href.startswith("./"):
            # 相对路径: ./202601/t20260113_3712702.shtml
            page_url = BASE_URL.rstrip("/") + "/" + href[2:]
        elif href.startswith("http"):
            page_url = href
        else:
            page_url = BASE_URL.rstrip("/") + "/" + href.lstrip("/")
        
        items.append((title, date_str, page_url))
    
    return items

def fetch_detail(url):
    """获取详情页内容"""
    try:
        resp = session.get(url, timeout=15)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            return None, None, None
        soup = BeautifulSoup(resp.text, "html.parser")
        
        # 标题取meta ArticleTitle
        title = ""
        meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta_title:
            title = meta_title.get("content", "")
        if not title:
            h1 = soup.find("h1")
            if h1:
                title = h1.get_text(strip=True)
        
        # 日期取meta PubDate
        date_str = ""
        meta_date = soup.find("meta", attrs={"name": "PubDate"})
        if meta_date:
            date_str = meta_date.get("content", "").strip()[:10]
        
        # 来源取meta ContentSource
        source = ""
        meta_source = soup.find("meta", attrs={"name": "ContentSource"})
        if meta_source:
            source = meta_source.get("content", "")
        
        # 正文取 div.trs_editor_view.TRS_UEDITOR
        content_div = soup.select_one("div.trs_editor_view.TRS_UEDITOR")
        if not content_div:
            content_div = soup.select_one("div.TRS_UEDITOR")
        
        content = ""
        if content_div:
            # 保留HTML结构
            content = str(content_div)
        else:
            # fallback
            content_div = soup.select_one("div.contentTextBox")
            if content_div:
                content = str(content_div)
        
        return title, date_str, source, content
        
    except Exception as e:
        print(f"[ERROR] 获取详情失败 {url}: {e}")
        return None, None, None, None

def crawl():
    print(f"[INFO] 柳州市柳江区 — 通知公告 爬虫启动")
    print(f"[INFO] 时间阈值: {THRESHOLD}")
    
    existing_urls = get_existing_urls()
    print(f"[INFO] 已有URL数: {len(existing_urls)}")
    
    new_count = 0
    skip_count = 0
    stop_crawl = False
    
    for page_num in range(MAX_PAGES):
        if stop_crawl:
            break
        
        # 构建分页URL
        if page_num == 0:
            page_url = BASE_URL
        else:
            page_url = f"http://www.liujiang.gov.cn/xwzx/ljztgg/index_{page_num}.shtml"
        
        print(f"[PAGE] 第{page_num+1}页: {page_url}")
        
        try:
            resp = session.get(page_url, timeout=15)
            resp.encoding = "utf-8"
            if resp.status_code != 200:
                print(f"[WARN] 第{page_num+1}页返回 {resp.status_code}, 停止翻页")
                break
        except Exception as e:
            print(f"[ERROR] 第{page_num+1}页请求失败: {e}")
            break
        
        items = parse_list_page(resp.text, page_num)
        if not items:
            print(f"[WARN] 第{page_num+1}页无数据, 停止翻页")
            break
        
        print(f"[PAGE] 第{page_num+1}页获取 {len(items)} 条")
        
        for title, date_str, detail_url in items:
            # 日期过滤
            if date_str < THRESHOLD:
                print(f"[SKIP] 日期超限 {date_str}: {title[:30]}")
                skip_count += 1
                stop_crawl = True
                continue
            
            # 去重
            if detail_url in existing_urls:
                skip_count += 1
                continue
            
            # 获取详情
            dt_title, dt_date, source, content = fetch_detail(detail_url)
            if not dt_title:
                dt_title = title
            
            final_title = dt_title or title
            final_date = dt_date or date_str
            
            if not content:
                print(f"[WARN] 正文为空: {final_title[:30]}")
                # 仍然写入但标记
                content = ""
            
            affected = store_item(final_title, final_date, content, detail_url)
            if affected > 0:
                new_count += 1
                existing_urls.add(detail_url)
            
            # 短暂延时
            time.sleep(0.3)
        
        if stop_crawl:
            print(f"[INFO] 遇到超限日期，停止翻页")
            break
    
    print(f"\n[SUMMARY] 完成!")
    print(f"[SUMMARY] 新增: {new_count}")
    print(f"[SUMMARY] 跳过: {skip_count}")
    
    # 输出JSON结果
    result = {
        "new": new_count,
        "skipped": skip_count,
        "site_name": SITE_NAME
    }
    print(f"\n[RESULT]{json.dumps(result, ensure_ascii=False)}")

if __name__ == "__main__":
    crawl()
