#!/usr/bin/env python3
"""
常德市人民政府 - 公示公告
https://www.changde.gov.cn/cdzx/gsgg
WebFuture CMS v15.2.8, 静态分页 /cdzx/gsgg_{n}
列表: ul.newsList > li (span.date + a[title])
详情: h2.title + div.conTxt + 附件

注意: 该站有 WAF CC攻击防御，请求间隔必须 ≥1秒
"""
import os, sys, re, time, json, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "常德市-公示公告"
BASE_URL = "https://www.changde.gov.cn"
LIST_URL_TPL = BASE_URL + "/cdzx/gsgg"        # page 1
LIST_URL_TPL_N = BASE_URL + "/cdzx/gsgg_{}"    # pages 2+
THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
TOTAL_PAGES = 10

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "https://www.changde.gov.cn/",
}

def fetch(url, retries=5):
    """Fetch URL with rate limiting and retry on WAF blocks."""
    import requests
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            if r.status_code == 403:
                wait = 10 * (i + 1)
                print(f"    ⚠️ WAF 403, 等待 {wait}s 后重试 ({i+1}/{retries})")
                time.sleep(wait)
                continue
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
        except requests.RequestException as e:
            if i < retries - 1:
                wait = 5 * (i + 1)
                print(f"    ⚠️ 请求异常: {e}, {wait}s后重试")
                time.sleep(wait)
    return None

def parse_list(html):
    """Parse list page, return list of (title, url, date)."""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='newsList')
    if not ul:
        return items
    
    for li in ul.find_all('li'):
        a = li.find('a')
        if not a:
            continue
        title = a.get('title', '').strip()
        if not title:
            title = a.get_text(strip=True)
        href = a.get('href', '').strip()
        if not href or href.startswith('javascript'):
            continue
        url = urljoin(BASE_URL, href)
        
        span_date = li.find('span', class_='date')
        date = span_date.get_text(strip=True) if span_date else ''
        
        if title and url and date:
            items.append((title, url, date))
    
    return items

def parse_detail(html, url):
    """Parse detail page data."""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title
    title_el = soup.find('h2', class_='title')
    title = title_el.get_text(strip=True) if title_el else ''
    
    # Date and source
    date = ''
    source = ''
    prop = soup.find('div', class_='property')
    if prop:
        for sp in prop.find_all('span'):
            txt = sp.get_text(strip=True)
            dm = re.search(r'(\d{4}-\d{2}-\d{2}\s*\d{2}:\d{2})', txt)
            if dm:
                date = dm.group(1)
            if '来源' in txt:
                source = re.sub(r'^来源[：:]?\s*', '', txt).strip()
    
    # Content body
    content_parts = []
    attachments = []
    
    conTxt = soup.find('div', class_='conTxt')
    if conTxt:
        for el in conTxt.children:
            if isinstance(el, str):
                txt = el.strip()
                if txt:
                    content_parts.append(txt)
            elif el.name in ('p', 'div', 'section'):
                txt = el.get_text('\n', strip=True)
                if txt:
                    content_parts.append(txt)
            elif el.name == 'table':
                txt = el.get_text('\n', strip=True)
                if txt:
                    content_parts.append(txt)
                    content_parts.append('')
            elif el.name == 'br':
                content_parts.append('')
        
        for a_tag in conTxt.find_all('a', href=re.compile(r'\.(doc|docx|pdf|xls|xlsx)(\?|$)', re.I)):
            att_url = urljoin(BASE_URL, a_tag.get('href', ''))
            att_name = a_tag.get_text(strip=True) or os.path.basename(att_url).split('?')[0]
            if att_url and att_name:
                attachments.append({'url': att_url, 'name': att_name})
    
    content_text = '\n'.join(content_parts).strip()
    return title, content_text, date, source, attachments

def push_to_db(items):
    """Push items to gov_raw."""
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA busy_timeout=5000")
    c = conn.cursor()
    
    new_count = 0
    skip_count = 0
    update_count = 0
    
    for title, url, date, content, source, atts_json in items:
        if not title:
            continue
        
        summary = content[:200] if content else title
        c.execute("SELECT id, site_name FROM gov_raw WHERE page_url=?", (url,))
        row = c.fetchone()
        if row:
            # Update site_name if wrong
            if row[1] != SITE_NAME:
                c.execute("UPDATE gov_raw SET site_name=? WHERE id=?", (SITE_NAME, row[0]))
                update_count += 1
            skip_count += 1
            continue
        
        c.execute(
            "INSERT INTO gov_raw (title, page_url, publish_date, site_name, content, source_url, attachments, category, summary) VALUES (?,?,?,?,?,?,?,?,?)",
            (title, url, date, SITE_NAME, content, source, atts_json, '政府公告', summary)
        )
        new_count += 1
    
    conn.commit()
    conn.close()
    return new_count, skip_count, update_count

def main():
    print(f"[{SITE_NAME}] 开始爬取，阈值: {THRESHOLD}")
    
    all_items = []
    
    # Step 1: Parse list pages (sequential, slow due to WAF)
    for page in range(1, TOTAL_PAGES + 1):
        list_url = LIST_URL_TPL if page == 1 else LIST_URL_TPL_N.format(page)
        print(f"  列表第{page}页: {list_url}")
        html = fetch(list_url)
        if not html:
            print(f"    ⚠️ 获取失败，跳过")
            time.sleep(3)
            continue
        
        items = parse_list(html)
        print(f"    → {len(items)} 条")
        
        for title, url, date in items:
            if date >= THRESHOLD:
                all_items.append((title, url, date))
            else:
                print(f"    ⏹️ 遇到超阈值日期 {date}，停止翻页")
                break
        else:
            time.sleep(1.5)  # Rate limit between list pages
            continue
        break
    
    print(f"共 {len(all_items)} 条待处理（近3年）")
    
    # Step 2: Fetch details (sequential, rate-limited)
    detail_results = []
    for i, (title, url, date) in enumerate(all_items):
        if (i+1) % 10 == 0:
            print(f"  详情: {i+1}/{len(all_items)}")
        
        html = fetch(url)
        if html:
            d_title, content, d_date, source, atts = parse_detail(html, url)
            if not d_title:
                d_title = title
            if not d_date:
                d_date = date
            atts_json = json.dumps(atts, ensure_ascii=False) if atts else '[]'
            detail_results.append((d_title, url, d_date, content, source, atts_json))
        
        time.sleep(1.2)  # CRITICAL: 1.2s between detail requests to avoid WAF
    
    # Step 3: Push to DB
    detail_results.sort(key=lambda x: x[2], reverse=True)
    
    push_items = [(dt, url, dd, cont, src, atts) for dt, url, dd, cont, src, atts in detail_results]
    new_count, skip_count, update_count = push_to_db(push_items)
    
    print(f"\n[{SITE_NAME}] 完成")
    print(f"  总处理: {len(detail_results)}")
    print(f"  新增: {new_count}")
    print(f"  跳过(已存在): {skip_count}")
    print(f"  更新站点名: {update_count}")

def incremental():
    """Incremental crawl - only page 1."""
    print(f"[{SITE_NAME}] 增量爬取")
    html = fetch(LIST_URL_TPL)
    if not html:
        print("  ⚠️ 获取列表失败")
        return
    
    items = parse_list(html)
    print(f"  列表: {len(items)} 条")
    
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    new = 0
    skip = 0
    
    for title, url, date in items:
        if date < THRESHOLD:
            continue
        
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
        if c.fetchone():
            skip += 1
            continue
        
        html = fetch(url)
        if not html:
            print(f"  ⚠️ 详情获取失败: {url.split('/')[-1]}")
            continue
        
        d_title, content, d_date, source, atts = parse_detail(html, url)
        if not d_title: d_title = title
        if not d_date: d_date = date
        atts_json = json.dumps(atts, ensure_ascii=False) if atts else '[]'
        summary = content[:200] if content else d_title
        
        c.execute(
            "INSERT INTO gov_raw (title, page_url, publish_date, site_name, content, source_url, attachments, category, summary) VALUES (?,?,?,?,?,?,?,?,?)",
            (d_title, url, d_date, SITE_NAME, content, source, atts_json, '政府公告', summary)
        )
        new += 1
        time.sleep(1.2)
    
    conn.commit()
    conn.close()
    print(f"  新增: {new}, 跳过: {skip}")

if __name__ == '__main__':
    if len(sys.argv) > 1 and sys.argv[1] == '--incremental':
        incremental()
    else:
        main()
