#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""磁县-漳河经济开发区 (catid=1553)"""
import re, os, sys, urllib.request
from datetime import datetime, timedelta

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "磁县-漳河经济开发区"
BASE_URL = "http://www.cixian.gov.cn"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch(url):
    req = urllib.request.Request(url, headers={"User-Agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"})
    resp = urllib.request.urlopen(req, timeout=30)
    raw = resp.read()
    return raw.decode("gb2312", errors="ignore")

def get_total_pages(catid):
    url = f"{BASE_URL}/index.php?m=content&c=index&a=lists&catid={catid}"
    html = fetch(url)
    m = re.search(r'共(\d+)页', html)
    if m: return int(m.group(1))
    pages = re.findall(r'<a[^>]*href="[^"]*page=(\d+)"', html)
    if pages: return max(int(p) for p in pages)
    return 1

def main():
    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    
    catid = 1553
    total_pages = get_total_pages(catid)
    print(f"Total pages: {total_pages}")
    
    total_new = 0
    total_skipped = 0
    
    for page in range(1, total_pages + 1):
        url = f"{BASE_URL}/index.php?m=content&c=index&a=lists&catid={catid}&page={page}"
        html = fetch(url)
        
        items = re.findall(
            r'<li><span class="rt">(\d{4}-\d{2}-\d{2})</span><a href="([^"]+)"[^>]*>([^<]+)</a></li>',
            html
        )
        
        for date, page_url, title in items:
            title = title.strip()
            if not title: continue
            
            if date < CUTOFF:
                total_skipped += 1
                continue
            
            full_url = page_url if page_url.startswith("http") else BASE_URL + page_url
            
            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (full_url,))
            if cur.fetchone():
                total_skipped += 1
                continue
            
            # Fetch detail
            try:
                detail_html = fetch(page_url if page_url.startswith("http") else BASE_URL + page_url)
                m = re.search(r'<title>([^<]+)</title>', detail_html)
                detail_title = m.group(1).strip() if m else title
                detail_title = re.sub(r'\s*-\s*漳河经济开发区\s*-\s*磁县人民政府\s*$', '', detail_title)
                
                m = re.search(r'class="text"[^>]*>(.*?)</div>', detail_html, re.DOTALL)
                content = m.group(1).strip() if m else "正文为空"
                if content:
                    content = re.sub(r'<div[^>]*style="[^"]*display:\s*none[^"]*"[^>]*>.*?</div>', '', content, flags=re.DOTALL)
                    content = re.sub(r'<iframe[^>]*>.*?</iframe>', '', content, flags=re.DOTALL)
                    content = content.strip()
                
                m = re.search(r'时间[：:]\s*(\d{4}-\d{2}-\d{2})', detail_html)
                pub_date = m.group(1) if m else date
                
                if content == "正文为空" or not content:
                    total_skipped += 1
                    continue
                
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                    (SITE_NAME, full_url, detail_title, pub_date, content, int(date.replace("-","")), "漳河经济开发区")
                )
                if cur.rowcount > 0:
                    total_new += 1
                    print(f"  [{total_new}] {detail_title[:40]} | {pub_date}")
                else:
                    total_skipped += 1
            except Exception as e:
                print(f"  [ERR] {title[:30]}: {e}")
                total_skipped += 1
    
    conn.commit()
    conn.close()
    print(f"\n结果: {total_new} 新增, {total_skipped} 跳过")

if __name__ == "__main__":
    main()
