#!/usr/bin/env python3
import os
"""
怀宁先锋网 - 通知公告 爬虫
http://zzb.ahhn.gov.cn/index/index/lst/id/10048.html
"""
import re
import sys
import time
import sqlite3
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timezone, timedelta
from urllib.parse import urljoin

BASE_URL = "http://zzb.ahhn.gov.cn/index/index/lst/id/10048.html"
DOMAIN = "zzb.ahhn.gov.cn"
SITE_NAME = "怀宁先锋网"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES = 5
INCREMENTAL = "--incremental" in sys.argv
YEAR_LIMIT = 3

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

conn = sqlite3.connect(DB_PATH, timeout=60)
c = conn.cursor()

cutoff_date = datetime.now(timezone(timedelta(hours=8))).replace(tzinfo=None) - timedelta(days=365 * YEAR_LIMIT)

def get_last_page():
    url = f"{BASE_URL}?id=10048&page=100"
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        pages = re.findall(r'page=(\d+)', r.text)
        if pages:
            return max(int(p) for p in pages)
    except:
        pass
    return 1

def parse_list_page(html):
    items = []
    parts = re.split(r'<div class="newslist yenews">', html)
    for part in parts[1:]:
        m = re.search(r'<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>', part, re.DOTALL)
        if not m:
            continue
        url = m.group(1).strip()
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        dm = re.search(r'<span>\s*(\d{4}-\d{2}-\d{2})\s*</span>', part)
        date_str = dm.group(1) if dm else ""
        items.append((title, url, date_str))
    return items

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print(f"  [ERROR] fetch {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    title_tag = soup.find("title")
    title = ""
    if title_tag:
        t = title_tag.get_text(strip=True)
        for suffix in ["-怀宁先锋网", "—怀宁先锋网", "-怀宁先锋", "—怀宁先锋"]:
            if t.endswith(suffix):
                t = t[: -len(suffix)]
                break
        title = t.strip()

    content_div = soup.find("article", class_=re.compile(r"content-wz"))
    if not content_div:
        content_div = soup.find("article", class_=re.compile(r"content-sj"))
    if not content_div:
        content_div = soup.find("div", class_=re.compile(r"content-wz"))
    if not content_div:
        content_div = soup.find("div", class_=re.compile(r"content-sj"))

    content_html = ""
    if content_div:
        content_html = str(content_div)

    date_str = ""
    date_patterns = [
        r'发布：(\d{4}-\d{2}-\d{2})',
        r'发布时间：(\d{4}-\d{2}-\d{2})',
        r'日期：(\d{4}-\d{2}-\d{2})',
    ]
    for pat in date_patterns:
        dm = re.search(pat, html)
        if dm:
            date_str = dm.group(1)
            break

    return title, content_html, date_str

def article_exists(page_url):
    c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url = ?", (page_url,))
    return c.fetchone()[0] > 0

def sync_fts(row_id, title, site_name):
    """Manually sync FTS5 after REPLACE (trigger only fires on INSERT)."""
    try:
        c.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                  (row_id, title, site_name, ""))
    except:
        pass

def insert_article(title, page_url, content_html, date_str):
    full_url = urljoin(BASE_URL, page_url)
    display_url = full_url.replace("http://", "").replace("https://", "")
    summary = ""
    
    # Check if exists
    c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (full_url,))
    row = c.fetchone()
    existing_id = row[0] if row else None
    
    try:
        c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url, summary, category, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_ahhn.py')""",
            (full_url, title, content_html, date_str, SITE_NAME, display_url, summary, ""))
        
        if existing_id:
            row_id = existing_id
        else:
            row_id = c.lastrowid
        
        # Sync FTS for both INSERT and REPLACE
        if row_id:
            sync_fts(row_id, title, SITE_NAME)
        return True
    except Exception as e:
        print(f"  [DB ERROR] {e}")
        return False

def main():
    total_new = 0
    total_updated = 0
    total_skipped = 0

    last_page = get_last_page()
    if last_page > MAX_PAGES:
        last_page = MAX_PAGES
        print(f"  Pages limited to {MAX_PAGES}")

    print(f"  Pages to crawl: 1-{last_page}")

    for page in range(1, last_page + 1):
        if page == 1:
            url = BASE_URL
        else:
            url = f"{BASE_URL}?id=10048&page={page}"

        print(f"  List page {page}/{last_page}: {url}")

        # retry up to 3 times for page 1 (known timeout issue)
        r = None
        for attempt in range(3):
            try:
                r = requests.get(url, headers=HEADERS, timeout=20)
                r.encoding = "utf-8"
                break
            except Exception as e:
                if attempt < 2:
                    print(f"    Retry {attempt+1}...")
                    time.sleep(3)
                else:
                    print(f"  [ERROR] list page {page}: {e}")
                    r = None
        
        if r is None:
            continue

        items = parse_list_page(r.text)
        if not items:
            print(f"  [WARN] No items found on page {page}")
            continue

        print(f"    Found {len(items)} items")

        for title, item_url, list_date in items:
            full_url = urljoin(BASE_URL, item_url)

            if list_date:
                try:
                    item_date = datetime.strptime(list_date, "%Y-%m-%d")
                    if item_date < cutoff_date.replace(tzinfo=None):
                        total_skipped += 1
                        continue
                except:
                    pass

            exists = article_exists(full_url)
            if INCREMENTAL and exists:
                total_skipped += 1
                continue

            print(f"    Fetching: {title[:30]}...", end=" ")
            d_title, content_html, detail_date = fetch_detail(full_url)

            if not d_title:
                d_title = title
            if not content_html:
                content_html = "<p>内容加载失败</p>"
                print("no content", end=" ")
            if not detail_date and list_date:
                detail_date = list_date

            if insert_article(d_title, item_url, content_html, detail_date):
                if exists:
                    total_updated += 1
                    print("updated")
                else:
                    total_new += 1
                    print("inserted")
            else:
                print("FAILED")

            time.sleep(0.3)

    conn.commit()
    conn.close()

    print(f"\n  ✅ 完成: 新增 {total_new}, 更新 {total_updated}, 跳过 {total_skipped}")

if __name__ == "__main__":
    main()
