#!/usr/bin/env python3
import os
"""
大冶市人民政府 - 生态环境领域 爬虫
http://www.hbdaye.gov.cn/zfxxgk/fdgknr/sjjczwgk/sjsthjly/
"""
import re
import sys
import time
import sqlite3
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timezone, timedelta
from urllib.parse import urljoin

BASE_URL = "http://www.hbdaye.gov.cn/zfxxgk/fdgknr/sjjczwgk/sjsthjly/"
DOMAIN = "www.hbdaye.gov.cn"
SITE_NAME = "大冶市生态环境"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES = 12
INCREMENTAL = "--incremental" in sys.argv
YEAR_LIMIT = 3

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

conn = sqlite3.connect(DB_PATH, timeout=60)

conn.execute("PRAGMA busy_timeout=60000")
c = conn.cursor()

cutoff_date = datetime.now(timezone(timedelta(hours=8))).replace(tzinfo=None) - timedelta(days=365 * YEAR_LIMIT)

def parse_list_page(html):
    """Parse listing page. Each article has 2 <li>s: title+date, then summary+pagination."""
    items = []
    # Pattern: <li><h2><a href="./202606/..." target="_blank">TITLE</a></h2><span>发布时间：2026年06月08日</span></li>
    pattern = r'<li>\s*<h2>\s*<a href="([^"]+)"[^>]*>([^<]+)</a>\s*</h2>\s*<span>发布时间：(\d{4})年(\d{2})月(\d{2})日</span>\s*</li>'
    for m in re.finditer(pattern, html):
        url = m.group(1).strip()
        title = m.group(2).strip()
        date_str = f"{m.group(3)}-{m.group(4)}-{m.group(5)}"
        items.append((title, url, date_str))
    return items

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print(f"  [ERROR] fetch {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    # Title
    title = ""
    h1 = soup.find("h1")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True)

    # Date
    date_str = ""
    date_text = re.search(r'发表日期：(\d{4})-(\d{2})-(\d{2})', html)
    if date_text:
        date_str = f"{date_text.group(1)}-{date_text.group(2)}-{date_text.group(3)}"

    # Content: display_text > view.TRS_UEDITOR
    content_html = ""
    display_text = soup.find("div", class_="display_text", id="fontzoom")
    if display_text:
        trs_div = display_text.find("div", class_=lambda c: c and "TRS_UEDITOR" in c if c else False)
        if trs_div:
            content_html = str(trs_div)
        else:
            content_html = str(display_text)

    if not content_html:
        content_html = "<p>内容加载失败</p>"

    return title, content_html, date_str

def article_exists(page_url):
    c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url = ?", (page_url,))
    return c.fetchone()[0] > 0

def sync_fts(row_id, title, site_name):
    try:
        # 2026-09-22: 先提交 gov_raw —— 下面手动写 FTS 会因触发器已写过同一
        #   rowid 而 IntegrityError，若不先 commit，这条记录会被一并回滚（静默丢数据）
        conn.commit()
        c.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                  (row_id, title, site_name, ""))
    except:
        pass

def insert_article(title, page_url, content_html, date_str):
    display_url = page_url.replace("http://", "").replace("https://", "")
    summary = ""

    c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
    row = c.fetchone()
    existing_id = row[0] if row else None

    try:
        c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url, summary, category, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_daye.py')""",
            (page_url, title, content_html, date_str, SITE_NAME, display_url, summary, ""))

        row_id = existing_id or c.lastrowid
        if row_id:
            sync_fts(row_id, title, SITE_NAME)
        return True
    except Exception as e:
        print(f"  [DB ERROR] {e}")
        return False

def get_page_url(page_num):
    if page_num == 0:
        return BASE_URL + "index.shtml"
    return f"{BASE_URL}index_{page_num}.shtml"

def main():
    total_new = 0
    total_updated = 0
    total_skipped = 0

# QC20260925 去掉整站清空再重灌(抢锁+中途死掉会清空整站; page_url 有 UNIQUE 索引，插入本就幂等)     c.execute("DELETE FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
    conn.commit()

    print(f"  Pages to crawl: 0-{MAX_PAGES-1}")

    for page in range(0, MAX_PAGES):
        url = get_page_url(page)
        print(f"  List page {page}/{MAX_PAGES-1}")

        r = None
        for attempt in range(3):
            try:
                r = requests.get(url, headers=HEADERS, timeout=20)
                r.encoding = "utf-8"
                break
            except Exception as e:
                if attempt < 2:
                    print(f"    Retry {attempt+1}...")
                    time.sleep(2)

        if r is None:
            print(f"  [SKIP] page {page} failed")
            continue

        items = parse_list_page(r.text)
        if not items:
            print(f"    No items found")
            continue

        print(f"    Found {len(items)} items")

        for title, item_url, list_date in items:
            full_url = urljoin(BASE_URL, item_url)

            if list_date:
                try:
                    item_date = datetime.strptime(list_date, "%Y-%m-%d")
                    if item_date < cutoff_date.replace(tzinfo=None):
                        total_skipped += 1
                        continue
                except:
                    pass

            exists = article_exists(full_url)
            if INCREMENTAL and exists:
                total_skipped += 1
                continue

            print(f"    Fetching: {title[:35]}...", end=" ")
            d_title, content_html, detail_date = fetch_detail(full_url)

            if not d_title:
                d_title = title
            if not detail_date and list_date:
                detail_date = list_date

            if insert_article(d_title, full_url, content_html, detail_date):
                if exists:
                    total_updated += 1
                    print("updated")
                else:
                    total_new += 1
                    print("inserted")
            else:
                print("FAILED")

            time.sleep(0.3)

        conn.commit()
        print(f"  [COMMIT] page {page} done")

    conn.close()
    print(f"\n  ✅ 完成: 新增 {total_new}, 更新 {total_updated}, 跳过 {total_skipped}")

if __name__ == "__main__":
    main()
