#!/usr/bin/env python3
import os
"""纳雍县人民政府 - 环境保护爬虫
静态HTML分页 (index_N.html) + requests详情页
"""
import requests, re, sqlite3, sys, time, os
from datetime import datetime, timedelta

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "纳雍县-环境保护"
BASE = "https://www.gznayong.gov.cn"
LIST_BASE = BASE + "/gtny/xq/hjbh_5232852"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
MAX_PAGES = 5
H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch(url):
    for retry in range(2):
        try:
            r = requests.get(url, headers=H, timeout=30)
            if r.status_code == 200:
                r.encoding = "utf-8"
                return r.text
        except:
            time.sleep(2)
    return None

def parse_list(html):
    items = []
    # Find all article links with t20 pattern
    for m in re.finditer(r'<a[^>]*title="([^"]+)"[^>]*href="([^"]+)"[^>]*>', html):
        title = m.group(1).strip()
        href = m.group(2)
        if not href.startswith("http"):
            href = BASE + href
        
        # Extract date from URL: t20260602 → 2026-06-02
        date = ""
        dm = re.search(r'/t(\d{8})_', href)
        if dm:
            try:
                d = dm.group(1)
                date = f"{d[:4]}-{d[4:6]}-{d[6:8]}"
            except:
                pass
        
        if title:
            items.append({"url": href, "title": title, "date": date})
    
    return items

def extract_detail(html):
    title = ""
    m = re.search(r'<title>(.*?)<', html)
    if m:
        title = m.group(1).strip()
    
    publish_date = ""
    m = re.search(r'(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}', html)
    if m:
        publish_date = m.group(1)
    if not publish_date:
        m = re.findall(r'(\d{4}-\d{2}-\d{2})', html)
        if m:
            publish_date = m[0]
    
    content = ""
    m = re.search(r'TRS_UEDITOR[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
        content = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r'\s*(style|class|align|lang|dir|width|height|border|cellpadding|cellspacing)="[^"]*"', '', content)
    
    return title, publish_date, content

def run(max_pages=None):
    if max_pages is None:
        max_pages = MAX_PAGES
    print(f"\n{'='*50}\n🚀 {SITE_NAME}\n{'='*50}")
    
    all_items = []
    for pg in range(1, max_pages + 1):
        url = LIST_BASE + "/" if pg == 1 else f"{LIST_BASE}/index_{pg-1}.html"
        html = fetch(url)
        if not html:
            print(f"  ⚠️ 第{pg}页取不到")
            break
        items = parse_list(html)
        # Dedup
        existing = {i["url"] for i in all_items}
        new_items = [i for i in items if i["url"] not in existing]
        if not new_items:
            print(f"  → 第{pg}页无新条目")
            break
        print(f"📋 第{pg}页: {len(new_items)} 条")
        all_items.extend(new_items)
        time.sleep(0.5)
    
    if not all_items:
        print("  ⚠️ 无数据")
        return
    
    print(f"\n📊 共 {len(all_items)} 条 (去重后)")
    
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    new = skip = no_body = 0
    
    for i, item in enumerate(all_items, 1):
        # Filter by date
        if item["date"] and item["date"] < CUTOFF:
            skip += 1
            continue
        
        html = fetch(item["url"])
        if not html:
            print(f"  ⚠️ 详情取不到: {item['title'][:30]}")
            skip += 1
            continue
        
        title, date, content = extract_detail(html)
        if not title:
            title = item["title"]
        if not date:
            date = item["date"]
        
        text_len = len(re.sub(r'<[^>]+>', '', content).strip()) if content else 0
        if text_len < 10:
            no_body += 1
            continue
        
        try:
            dr = 0
            try:
                dr = int(datetime.strptime(date, "%Y-%m-%d").timestamp())
            except:
                dr = int(time.time())
            summary = re.sub(r'<[^>]+>', '', content).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_raw(site_name,source_url,page_url,title,publish_date,content,summary,date_rank,category) VALUES(?,?,?,?,?,?,?,?,?)",
                (SITE_NAME, item["url"], item["url"], title, date, content, summary, dr, "环境保护")
            )
            if conn.total_changes:
                new += 1
            else:
                skip += 1
        except Exception as e:
            print(f"  ❌ DB: {e}")
            skip += 1
        
        if i % 10 == 0:
            conn.commit()
            print(f"  ...{i}/{len(all_items)}")
        time.sleep(0.3)
    
    conn.commit()
    try:
        # QC20260926 去掉手写 gov_search 整站删除(抢锁源; FTS 由 gov_raw 触发器维护) 
        # conn.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
        conn.execute("INSERT OR REPLACE INTO gov_search(rowid,title,site_name,summary) SELECT rowid,title,site_name,summary FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        conn.commit()
    except Exception as e:
        print(f"⚠️ FTS: {e}")
    conn.close()
    
    print(f"\n✅ {SITE_NAME}: 新增{new}, 空正文{no_body}, 跳过{skip}")

if __name__ == "__main__":
    mp = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    run(mp)
