#!/usr/bin/env python3
"""湛江市生态环境局 — 行政公示爬虫
https://www.zhanjiang.gov.cn/zjsfw/bmdh/sthjj/zwgk/xzgs/
Custom CMS, index_N.html分页
"""
import os, sys, re, json, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "https://www.zhanjiang.gov.cn/zjsfw/bmdh/sthjj/zwgk/xzgs/"
SITE_NAME = "湛江市生态环境局-行政公示"
THRESHOLD = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
MAX_PAGES = 100

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Referer": "https://www.zhanjiang.gov.cn/",
}

session = requests.Session()
session.headers.update(HEADERS)

import sqlite3

def get_existing_urls():
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        existing = {row[0] for row in c.fetchall()}
        conn.close()
        return existing
    except Exception as e:
        print(f"[WARN] 无法查询已有URL: {e}")
        return set()

def store_item(title, date_str, content, page_url):
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw 
            (title, publish_date, content, page_url, site_name)
            VALUES (?, ?, ?, ?, ?)
        """, (title.strip(), date_str, content.strip(), page_url.strip(), SITE_NAME))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected
    except Exception as e:
        print(f"[ERROR] 写入失败 {page_url}: {e}")
        return 0

def parse_list_page(html, page_num):
    """解析列表页"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    li_tags = soup.select("ul.list li")
    for li in li_tags:
        a_tag = li.find("a")
        time_span = li.find("span", class_="time")
        if not a_tag or not time_span:
            continue
        href = a_tag.get("href", "")
        title = a_tag.get_text(strip=True)
        date_str = time_span.get_text(strip=True)
        
        # 构建完整URL
        if href.startswith("http"):
            page_url = href
        elif href.startswith("/"):
            page_url = f"https://www.zhanjiang.gov.cn{href}"
        else:
            page_url = BASE_URL.rstrip("/") + "/" + href.lstrip("./")
        
        items.append((title, date_str, page_url))
    
    return items

def fetch_detail(url):
    """获取详情页内容"""
    try:
        resp = session.get(url, timeout=15)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            return None, None, None
        soup = BeautifulSoup(resp.text, "html.parser")
        
        # 标题
        title = ""
        meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta_title:
            title = meta_title.get("content", "")
        if not title:
            h3 = soup.find("h3")
            if h3:
                title = h3.get_text(strip=True)
        
        # 日期
        date_str = ""
        meta_date = soup.find("meta", attrs={"name": "PubDate"})
        if meta_date:
            date_str = meta_date.get("content", "").strip()[:10]
        
        # 来源
        source = ""
        ly_span = soup.find("span", class_="ly")
        if ly_span:
            src = ly_span.get_text(strip=True)
            src = re.sub(r'^来源[：:]', '', src).strip()
            source = src
        
        # 正文
        content = ""
        article_div = soup.select_one("div.article")
        if article_div:
            content = str(article_div)
        
        return title or "", date_str, content
        
    except Exception as e:
        print(f"[ERROR] 获取详情失败 {url}: {e}")
        return None, None, None

def crawl():
    print(f"[INFO] 湛江市生态环境局 — 行政公示 爬虫启动")
    print(f"[INFO] 时间阈值: {THRESHOLD}")
    
    existing_urls = get_existing_urls()
    print(f"[INFO] 已有URL数: {len(existing_urls)}")
    
    new_count = 0
    skip_count = 0
    stop_crawl = False
    
    for page_num in range(MAX_PAGES):
        if stop_crawl:
            break
        
        if page_num == 0:
            page_url = BASE_URL
        else:
            page_url = f"{BASE_URL}index_{page_num + 1}.html"
        
        print(f"[PAGE] 第{page_num+1}页: {page_url}")
        
        try:
            resp = session.get(page_url, timeout=15)
            resp.encoding = "utf-8"
            if resp.status_code != 200:
                print(f"[INFO] 第{page_num+1}页返回 {resp.status_code}, 停止")
                break
        except Exception as e:
            print(f"[ERROR] 第{page_num+1}页请求失败: {e}")
            break
        
        items = parse_list_page(resp.text, page_num)
        if not items:
            print(f"[INFO] 第{page_num+1}页无数据, 停止")
            break
        
        print(f"[PAGE] 第{page_num+1}页获取 {len(items)} 条")
        
        for title, date_str, detail_url in items:
            # 日期过滤
            if date_str < THRESHOLD:
                print(f"[SKIP] 日期超限 {date_str}: {title[:30]}")
                skip_count += 1
                stop_crawl = True
                continue
            
            # 去重
            if detail_url in existing_urls:
                skip_count += 1
                continue
            
            # 获取详情
            dt_title, dt_date, content = fetch_detail(detail_url)
            final_title = dt_title or title
            final_date = dt_date or date_str
            
            if not content:
                print(f"[WARN] 正文为空: {final_title[:30]}")
                content = ""
            
            affected = store_item(final_title, final_date, content, detail_url)
            if affected > 0:
                new_count += 1
                existing_urls.add(detail_url)
            
            time.sleep(0.2)
        
        if stop_crawl:
            print(f"[INFO] 遇到超限日期，停止翻页")
            break
    
    print(f"\n[SUMMARY] 完成!")
    print(f"[SUMMARY] 新增: {new_count}")
    print(f"[SUMMARY] 跳过: {skip_count}")
    
    result = {"new": new_count, "skipped": skip_count, "site_name": SITE_NAME}
    print(f"\n[RESULT]{json.dumps(result, ensure_ascii=False)}")

if __name__ == "__main__":
    crawl()
