#!/usr/bin/env python3
"""
镇远县人民政府 - 通知公告 爬虫
CMS: TRS（拓尔思）
分页: createPageHTML(92, 0, "index", "html", 1367)
"""

import os, sys, re, time, hashlib
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "镇远县-通知公告"
BASE_URL = "https://www.zygov.gov.cn/xwzx/tzgg"

CUTOFF_DATE = datetime.strptime("2023-06-18", "%Y-%m-%d")
MAX_WORKERS = 8
PAGE_DELAY = 0.5
MAX_PAGES = 92

session = requests.Session()
session.headers.update({
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
})

def get_conn():
    import sqlite3
    return sqlite3.connect(DB_PATH, timeout=60)

def exists(conn, url):
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
    return cur.fetchone() is not None

def insert(conn, title, content, pub_date, page_url):
    now = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
    date_rank = 0
    try:
        d = datetime.strptime(pub_date[:10], "%Y-%m-%d")
        date_rank = int(d.strftime("%Y%m%d"))
    except:
        pass
    try:
        cur = conn.execute("""INSERT OR IGNORE INTO gov_raw 
            (site_name, title, page_url, source_url, publish_date, date_rank, 
             summary, status, category, content, visits, tags)
            VALUES (?,?,?,?,?,?,?,?,?,?,0,'')""",
            (SITE_NAME, title, page_url, "", pub_date[:10], date_rank,
             "", "published", "通知公告", content))
        return cur.rowcount > 0
    except Exception as e:
        print(f"  [DB] {e}", file=sys.stderr)
        return False

def fetch_list_page(page_num):
    """Fetch list page. page_num=0 -> index.html, page_num=1 -> index_1.html"""
    if page_num == 0:
        url = f"{BASE_URL}/index.html"
    else:
        url = f"{BASE_URL}/index_{page_num}.html"
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f"  [错误] {url}: {e}", file=sys.stderr)
        return []

    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='NewsList')
    if not ul:
        return []
    for li in ul.find_all('li'):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a['href'].strip()
        if not href:
            continue
        title = a.get('title', '') or a.get_text(strip=True)
        date_span = li.find('span')
        pub_date = date_span.get_text(strip=True) if date_span else ''
        full_url = href if href.startswith('http') else f"https://www.zygov.gov.cn{href}" if href.startswith('/') else href
        items.append({'url': full_url, 'title': title, 'pub_date': pub_date})
    return items

def fetch_detail(item):
    """Fetch detail page content."""
    url = item['url']
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except:
        return {**item, 'content': ''}

    soup = BeautifulSoup(html, 'html.parser')

    # Title
    title = item['title']
    meta_t = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_t and meta_t.get('content'):
        title = meta_t['content'].strip()

    # Date
    pub_date = item['pub_date']
    meta_d = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_d and meta_d.get('content'):
        pub_date = meta_d['content'].strip()

    # Content
    content_div = soup.find('div', class_='trs_editor_view')
    if not content_div:
        content_div = soup.find('div', class_='TRS_UEDITOR')
    if not content_div:
        content_div = soup.find('div', class_='Article_zw')
    if not content_div:
        content_div = soup.find('div', id='c')

    content = str(content_div) if content_div else ''
    return {'url': url, 'title': title, 'pub_date': pub_date, 'content': content}

def main():
    print(f"[信息] 镇远县-通知公告, TRS分页 {MAX_PAGES}页")

    # Collect list
    all_items = []
    for pi in range(MAX_PAGES):
        items = fetch_list_page(pi)
        all_items.extend(items)
        print(f"  第{pi+1}/{MAX_PAGES}页 -> {len(items)}条 (累计{len(all_items)}条)")
        time.sleep(PAGE_DELAY)

    print(f"[列表] 共 {len(all_items)} 条")

    # Filter 3yr
    filtered = []
    skipped_old = 0
    for item in all_items:
        try:
            d = datetime.strptime(item['pub_date'], '%Y-%m-%d')
            if d < CUTOFF_DATE:
                skipped_old += 1
                continue
        except:
            pass
        filtered.append(item)
    print(f"[过滤] 超3年: {skipped_old}, 有效: {len(filtered)}")

    if not filtered:
        print("[完成] 无数据")
        return

    # Check dupes
    conn = get_conn()
    new_items = []
    skipped_dup = 0
    for item in filtered:
        if exists(conn, item['url']):
            skipped_dup += 1
        else:
            new_items.append(item)
    conn.close()
    print(f"[去重] 已存在: {skipped_dup}, 新增: {len(new_items)}")

    if not new_items:
        print("[完成] 无新增")
        return

    # Fetch details
    print(f"[详情] 获取{len(new_items)}条详情 (并发{MAX_WORKERS})...")
    ok = 0
    conn = get_conn()
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as ex:
        futures = [ex.submit(fetch_detail, item) for item in new_items]
        for i, f in enumerate(as_completed(futures), 1):
            r = f.result()
            if r.get('content') and insert(conn, r['title'], r['content'], r['pub_date'], r['url']):
                ok += 1
            if i % 20 == 0:
                print(f"  进度: {i}/{len(new_items)} (入库{ok})")
    conn.commit()
    conn.close()

    # Summary
    print(f"\n{'='*60}")
    print(f"站点: {SITE_NAME}")
    print(f"新增: {ok} 条")
    print(f"跳过(超3年): {skipped_old}")
    print(f"跳过(重复): {skipped_dup}")
    if ok > 0:
        for r_ in new_items[-3:]:
            print(f"  [{r_['pub_date']}] {r_['title'][:50]}")
    print(f"{'='*60}")

if __name__ == '__main__':
    main()
