#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""黔西市-生态环境 (TRS分页, 自定义CMS)"""
import re, os, urllib.request, ssl, sqlite3
from datetime import datetime, timedelta

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "黔西市-生态环境"
BASE = "https://www.gzqianxi.gov.cn"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

def fetch(url):
    req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"})
    resp = urllib.request.urlopen(req, timeout=30, context=ctx)
    return resp.read().decode("utf-8", errors="ignore")

def fetch_list(page=0):
    if page == 0:
        url = BASE + "/zwgk2022/zdly/hjbh/index.html"
    else:
        url = BASE + "/zwgk2022/zdly/hjbh/index_{}.html".format(page)
    html = fetch(url)
    # Parse items
    items = []
    # Find all list items by matching <li> blocks
    for m in re.finditer(
        r'<a[^>]*href="(https?://[^"]+)"[^>]*title="([^"]*)"[^>]*>.*?<div class="date-span">\s*(\d{4}-\d{2}-\d{2})\s*</div>',
        html, re.DOTALL
    ):
        items.append((m.group(2).strip(), m.group(1), m.group(3)))
    return items

def fetch_detail(url):
    html = fetch(url)
    
    # Title
    title = ""
    m = re.search(r'<h1[^>]*style="[^"]*"[^>]*>([^<]+)</h1>', html)
    if m: title = m.group(1).strip()
    if not title:
        m = re.search(r'<title>([^<]+)</title>', html)
        if m: title = m.group(1).strip()
    
    # Content from id="content" (contains TRS_UEDITOR)
    content = "正文为空"
    m = re.search(r'<div id="content"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if m:
        c = m.group(1).strip()
        if len(c) > 50:
            content = c
    
    if content != "正文为空":
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
        content = re.sub(r'<iframe[^>]*>.*?</iframe>', '', content, flags=re.DOTALL)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
        content = content.strip()
        if not content:
            content = "正文为空"
    
    # Date
    date = ""
    m = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    if m: date = m.group(1)
    if not date:
        m = re.search(r'(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}', html)
        if m: date = m.group(1)
    
    title = re.sub(r'<[^>]+>', '', title).strip()
    return title, date, content

def main():
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    
    # Check how many pages
    raw_html = fetch(BASE + "/zwgk2022/zdly/hjbh/index.html")
    pages_m = re.search(r'createPageHTML\((\d+)', raw_html)
    total_pages = int(pages_m.group(1)) if pages_m else 10
    print("Total pages: {}".format(total_pages))
    
    total_new = 0
    total_skipped = 0
    
    for page in range(total_pages):
        items = fetch_list(page)
        if not items:
            print("  Page {}: no items".format(page))
            continue
        
        page_new = 0
        for title, url, date in items:
            if date < CUTOFF:
                total_skipped += 1
                continue
            
            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if cur.fetchone():
                total_skipped += 1
                continue
            
            try:
                detail_title, pub_date, content = fetch_detail(url)
                if content == "正文为空":
                    total_skipped += 1
                    continue
                
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                    (SITE_NAME, url, detail_title or title, pub_date or date, content, int(date.replace("-", "")), "生态环境")
                )
                if cur.rowcount > 0:
                    total_new += 1
                    page_new += 1
            except Exception as e:
                print("  ERR: {} - {}".format(title[:30], str(e)[:60]))
                total_skipped += 1
        
        conn.commit()
        print("  Page {}: {} new (total {})".format(page, page_new, total_new))
    
    conn.close()
    print("\n结果: {} 新增, {} 跳过".format(total_new, total_skipped))

if __name__ == "__main__":
    main()
