#!/usr/bin/env python3
"""
明光市人民政府 - 行政审批 - 服务器端爬虫
用于每日增量爬取（仅检查前10页最新数据）
"""

import os, sys, re, json, time
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "明光市-行政审批"
ORGAN_ID = "161056936"
CAT_ID = "170077440"
API_URL = "https://www.mingguang.gov.cn/czxxgk/site/label/8888"
HOST = "www.mingguang.gov.cn"

CUTOFF_DAYS = 7
MAX_WORKERS = 5
LIST_DELAY = 2.0
RETRY_MAX = 3

session = requests.Session()
session.headers.update({
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Referer': 'https://www.mingguang.gov.cn/site/tpl/162748858',
})

def safe_post(url, data):
    for attempt in range(RETRY_MAX):
        try:
            r = session.post(url, data=data, timeout=30)
            r.encoding = 'utf-8'
            if r.text and len(r.text) > 100:
                return r.text
        except Exception as e:
            print(f"  [重试{attempt+1}] {e}", file=sys.stderr)
        time.sleep(3 * (attempt + 1))
    return ""

def safe_get(url):
    for attempt in range(RETRY_MAX):
        try:
            r = session.get(url, timeout=30)
            r.encoding = 'utf-8'
            if r.text and len(r.text) > 200:
                return r.text
        except Exception as e:
            print(f"  [重试{attempt+1}] {e}", file=sys.stderr)
        time.sleep(2 * (attempt + 1))
    return ""

def get_conn():
    import sqlite3
    return sqlite3.connect(DB_PATH)

def exists(conn, url):
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
    return cur.fetchone() is not None

def insert(conn, title, content, pub_date, page_url):
    now = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
    date_rank = 0
    try:
        d = datetime.strptime(pub_date[:10], "%Y-%m-%d")
        date_rank = int(d.strftime("%Y%m%d"))
    except:
        pass
    try:
        cur = conn.execute("""INSERT OR IGNORE INTO gov_raw 
            (site_name, title, page_url, source_url, publish_date, date_rank, 
             summary, status, category, content, visits, tags)
            VALUES (?,?,?,?,?,?,?,?,?,?,0,'')""",
            (SITE_NAME, title, page_url, "", pub_date[:10], date_rank,
             "", "published", "环评审批", content))
        return cur.rowcount > 0
    except Exception as e:
        print(f"  [DB] {e}", file=sys.stderr)
        return False

def fetch_list_page(page_index):
    data = {
        'labelName': 'publicInfoList', 'siteId': '2653861', 'pageSize': '20',
        'dateFormat': 'yyyy-MM-dd', 'length': '50', 'isDate': 'true',
        'result': '暂无相关信息', 'organId': ORGAN_ID, 'action': 'list',
        'type': '4', 'catId': CAT_ID, 'pcatId': '', 'typeNode': '8',
        'file': '/c1/chuzhou/publicInfoList_sdzt_new', 'pageIndex': str(page_index),
    }
    html = safe_post(API_URL, data)
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    for li in soup.select('li'):
        a = li.find('a', href=True)
        if not a: continue
        href = a['href'].strip()
        if not href: continue
        title = a.get('title', '') or a.get_text(strip=True)
        date_span = li.find('span', class_='date')
        pub_date = date_span.get_text(strip=True) if date_span else ''
        full_url = href if href.startswith('http') else f'https://{HOST}{href}' if href.startswith('/') else href
        items.append({'url': full_url, 'title': title, 'pub_date': pub_date})
    return items

def get_total_pages():
    data = {
        'labelName': 'publicInfoList', 'siteId': '2653861', 'pageSize': '20',
        'dateFormat': 'yyyy-MM-dd', 'length': '50', 'isDate': 'true',
        'result': '暂无相关信息', 'organId': ORGAN_ID, 'action': 'list',
        'type': '4', 'catId': CAT_ID, 'pcatId': '', 'typeNode': '8',
        'file': '/c1/chuzhou/publicInfoList_sdzt_new', 'pageIndex': '1',
    }
    html = safe_post(API_URL, data)
    m = re.search(r'pageCount\s*:\s*(\d+)', html)
    return int(m.group(1)) if m else 1

def fetch_detail(item):
    for attempt in range(RETRY_MAX):
        html = safe_get(item['url'])
        if html and len(html) > 200:
            break
        time.sleep(2 * (attempt + 1))
    else:
        return {**item, 'content': ''}
    soup = BeautifulSoup(html, 'html.parser')
    title_el = soup.find('h1', class_='newstitle')
    title = title_el.get_text(strip=True) if title_el else item['title']
    pub_date = ''
    meta = soup.find('meta', attrs={'name': 'PubDate'})
    if meta and meta.get('content'):
        pub_date = meta['content'].strip()
    if not pub_date: pub_date = item['pub_date']
    content_div = (soup.find('div', class_='gkwz_contnet') or soup.find('div', class_='j-fontContent')
                   or soup.find('div', class_='xxgkcontent') or soup.find('div', class_='xxgk_contnet'))
    content = str(content_div) if content_div else ''
    return {'url': item['url'], 'title': title, 'pub_date': pub_date, 'content': content}

def main():
    print(f"[信息] 获取总页数...")
    total_pages = get_total_pages()
    max_pages = min(total_pages, 10)
    print(f"[信息] 共{total_pages}页, 增量扫描前{max_pages}页")

    all_items = []
    for pi in range(1, max_pages + 1):
        items = fetch_list_page(pi)
        all_items.extend(items)
        print(f"  第{pi}/{max_pages}页 -> {len(items)}条 (累计{len(all_items)}条)")
        time.sleep(LIST_DELAY)

    cutoff = datetime.now() - timedelta(days=CUTOFF_DAYS)
    filtered = []
    for item in all_items:
        try:
            d = datetime.strptime(item['pub_date'], '%Y-%m-%d')
            if d < cutoff:
                continue
        except: pass
        filtered.append(item)
    print(f"[过滤] 7日内: {len(filtered)} 条")

    conn = get_conn()
    new_items = []
    for item in filtered:
        if not exists(conn, item['url']):
            new_items.append(item)
    conn.close()
    print(f"[去重] 待获取: {len(new_items)} 条")

    if not new_items:
        print("✅ 明光市-行政审批 增量: 无新增")
        return

    ok = 0
    conn = get_conn()
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as ex:
        futures = [ex.submit(fetch_detail, item) for item in new_items]
        for i, f in enumerate(as_completed(futures), 1):
            r = f.result()
            if r.get('content') and insert(conn, r['title'], r['content'], r['pub_date'], r['url']):
                ok += 1
            if i % 10 == 0:
                print(f"  进度: {i}/{len(new_items)} (入库{ok})")
    conn.commit()
    conn.close()
    print(f"✅ 明光市-行政审批 增量完成: 新增{ok}条")

if __name__ == '__main__':
    main()
