#!/usr/bin/env python3
"""
crawl_ningguo_tzgg.py - 宁国经开区管委会通知公告爬虫
Site: www.ningguo.gov.cn
Branch: 455 (宁国经开区管委会)
Channel: 23136 (通知公告)
WAF: DBAppWAF - requires token_verified=true cookie
"""

import sys, re, time, os, json
import sqlite3
from datetime import datetime, timezone, timedelta
from urllib.request import Request, urlopen

BASE_URL = "https://www.ningguo.gov.cn"
LIST_TPL = f"{BASE_URL}/XxgkContent/showList/455/23136/page_{{}}.html"
SITE_NAME = "ningguo_jkq_tzgg"
SOURCE = "宁国经开区管委会通知公告"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Cookie": "token_verified=true"
}
TIMEOUT = 20
MAX_RETRIES = 3
SLEEP = 0.2

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

def get_conn():
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn

def fetch(url, retries=MAX_RETRIES):
    for attempt in range(retries):
        try:
            req = Request(url, headers=HEADERS)
            with urlopen(req, timeout=TIMEOUT) as resp:
                data = resp.read()
            try:
                return data.decode("utf-8")
            except:
                return data.decode("gbk", errors="replace")
        except Exception as e:
            if attempt < retries - 1:
                time.sleep(SLEEP * (attempt + 1))
            else:
                print(f"  [WARN] Failed {url}: {e}", file=sys.stderr)
                return None

def parse_list_page(html):
    """Extract (title, url, date) from list page"""
    items = re.findall(
        r'<li>.*?<a[^>]*href="(/OpennessContent/show/\d+\.html)"[^>]*title="([^"]*)"[^>]*>.*?</a>',
        html, re.DOTALL
    )
    result = []
    seen_urls = set()
    for href, title_attr in items:
        full_url = BASE_URL + href
        if full_url in seen_urls:
            continue
        seen_urls.add(full_url)
        # Extract date from <span> after the </a>
        rest = html[html.index(href)+len(href):html.index(href)+len(href)+300]
        date_m = re.search(r'<span>([^<]+)</span>', rest)
        date = date_m.group(1).strip() if date_m else ""
        # Clean title
        title = title_attr.strip()
        if not title:
            continue
        result.append((title, full_url, date))
    return result

def parse_detail(html, url):
    """Extract (title, publish_date, content, source) from detail page"""
    # Title
    title_m = re.search(r'class="text-center u-title">([^<]+)', html)
    title = title_m.group(1).strip() if title_m else ""
    
    # Date
    date_m = re.search(r'发布时间[：:]\s*([\d-]+\s*[\d:]*)\s*<', html)
    publish_date = date_m.group(1).strip() if date_m else ""
    if publish_date:
        publish_date = publish_date[:10]
    
    # Content
    content_m = re.search(r'class="g-detailbox[^"]*"[^>]* id="zoom"[^>]*>(.*?)</div>\s*<div[^>]*class="share-main', html, re.DOTALL)
    if not content_m:
        content_m = re.search(r'id="zoom"[^>]*>(.*?)</div>\s*<div[^>]*class="share-main', html, re.DOTALL)
    content = content_m.group(1).strip() if content_m else ""
    
    # Source
    source_m = re.search(r'来源[：:]\s*([^<]+)', html)
    source = source_m.group(1).strip() if source_m else ""
    
    return title, publish_date, content, source

def clean_date(date_str):
    """Normalize date to YYYY-MM-DD"""
    if not date_str:
        return ""
    date_str = date_str.strip()
    m = re.match(r'(\d{4})[-/](\d{1,2})[-/](\d{1,2})', date_str)
    if m:
        return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return date_str

def get_cutoff_date():
    """3 years ago date"""
    return (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")

def main():
    full_mode = "--full" in sys.argv
    
    cutoff = get_cutoff_date()
    print(f"[{SITE_NAME}] Cutoff date: {cutoff}", flush=True)
    
    new_count = 0
    skip_count = 0
    
    conn = get_conn()
    
    # Dynamically discover max pages
    max_page = 0
    for pg in range(1, 31):
        test_url = LIST_TPL.format(pg)
        test_html = fetch(test_url)
        if not test_html:
            break
        test_items = parse_list_page(test_html)
        if not test_items:
            break
        max_page = pg
        if pg == 1:
            print(f"  First page has {len(test_items)} items", flush=True)
    
    if max_page == 0:
        print("  No valid pages found!", flush=True)
        return
    
    print(f"  Total pages: {max_page}", flush=True)
    
    for page in range(1, max_page + 1):
        print(f"\n--- Page {page}/{max_page} ---", flush=True)
        list_url = LIST_TPL.format(page)
        html = fetch(list_url)
        
        if not html:
            print(f"  Failed to fetch page {page}!", flush=True)
            continue
        
        items = parse_list_page(html)
        print(f"  Found {len(items)} items", flush=True)
        
        for title, detail_url, list_date in items:
            # Check date
            item_date = clean_date(list_date)
            if item_date and item_date < cutoff:
                print(f"  [SKIP] {item_date} {title[:50]}... (too old)", flush=True)
                skip_count += 1
                continue
            
            # Check if already exists
            existing = conn.execute(
                "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
                (detail_url, SITE_NAME)
            ).fetchone()
            if existing:
                print(f"  [SKIP] {item_date} {title[:50]}... (already exists)", flush=True)
                skip_count += 1
                continue
            
            # Fetch detail
            time.sleep(SLEEP)
            detail_html = fetch(detail_url)
            if not detail_html:
                print(f"  [WARN] Failed to fetch detail: {title[:40]}", flush=True)
                skip_count += 1
                continue
            
            detail_title, pub_date, content, source = parse_detail(detail_html, detail_url)
            
            if not detail_title:
                detail_title = title
            if not pub_date:
                pub_date = item_date
            
            # Check detail date against cutoff
            if pub_date and pub_date < cutoff:
                print(f"  [SKIP] {pub_date} {title[:50]}... (too old)", flush=True)
                skip_count += 1
                continue
            
            # Generate summary
            summary = ""
            if content:
                text = re.sub(r'<[^>]+>', ' ', content)
                text = re.sub(r'\s+', ' ', text).strip()
                summary = text[:200]
            
            # Insert into DB
            try:
                existing_id = conn.execute(
                    "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
                    (detail_url, SITE_NAME)
                ).fetchone()
                if existing_id:
                    print(f"  [SKIP] {pub_date} {title[:50]}... (duplicate)", flush=True)
                    skip_count += 1
                    continue
                
                conn.execute(
                    """INSERT INTO gov_raw (site_name, title, content, publish_date, source_url, page_url, summary)
                       VALUES (?, ?, ?, ?, ?, ?, ?)""",
                    (SITE_NAME, detail_title, content, pub_date, detail_url, detail_url, summary)
                )
                record_id = conn.execute("SELECT last_insert_rowid()").fetchone()[0]
                
                if record_id:
                    conn.execute(
                        "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                        (record_id, detail_title, SITE_NAME, summary)
                    )
                
                conn.commit()
                new_count += 1
                print(f"  [NEW] {pub_date} {title[:60]}... ✅", flush=True)
            except Exception as e:
                print(f"  [ERROR] DB insert failed: {e}", flush=True)
                conn.rollback()
    
    conn.close()
    
    print(f"\n=== Done! New: {new_count}, Skipped: {skip_count}, Pages: {max_page} ===", flush=True)
    print(f"Site: {SITE_NAME} - 宁国经开区管委会通知公告", flush=True)

if __name__ == "__main__":
    main()
