#!/usr/bin/env python3
"""临泉县人民政府-信息公开 通用爬虫 (列表AJAX+详情SSR, 全requests)
用法: python3 crawl_linquan_hj.py [site_name] [branch_id] [code] [max_pages]
默认: 水利局-重大行政决策预公开 (799/64847)
"""
import re, requests, sqlite3, os, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DEFAULT_SITE = '临泉县水利局-重大行政决策预公开'
DEFAULT_BRANCH = '799'
DEFAULT_CODE = '64847'
BASE = 'https://www.linquan.gov.cn'
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

def fetch_html(url):
    """获取页面原始字节，强制UTF-8解码"""
    r = requests.get(url, headers=HEADERS, timeout=30)
    return r.content.decode('utf-8', errors='replace')

def get_total_pages(branch_id, code):
    """逐页检查直到返回无数据的页"""
    total = 1
    for pg in range(2, 101):  # 最多100页
        try:
            r = requests.head(BASE + f'/OpennessTarget/{branch_id}/{code}/page_{pg}.html', headers=HEADERS, timeout=10)
            if r.status_code != 200:
                return pg - 1
            # 也检查内容是否非空（有些页返回200但为空）
            r2 = requests.get(BASE + f'/OpennessTarget/{branch_id}/{code}/page_{pg}.html', headers=HEADERS, timeout=15)
            html = r2.content.decode('utf-8', errors='replace')
            if '<td' not in html:
                return pg - 1
            total = pg
        except:
            return pg - 1
    return total

def get_list_page(branch_id, code, page=1):
    """获取列表页数据（AJAX端点）"""
    url = BASE + f'/OpennessTarget/{branch_id}/{code}/page_{page}.html'
    try:
        html = fetch_html(url)
        soup = BeautifulSoup(html, 'lxml')
        items = []
        for tr in soup.select('table tr'):
            tds = tr.find_all('td')
            if len(tds) >= 3:
                a_primary = tds[1].find('a') if tds[1] else None
                if not a_primary:
                    continue
                href = a_primary.get('href', '')
                if not href.startswith('http'):
                    href = urljoin(url, href)
                items.append({
                    'title': a_primary.get_text(strip=True),
                    'url': href,
                    'date': tds[2].get_text(strip=True),
                })
        return items
    except Exception as e:
        print(f'  ⚠️ 第{page}页加载失败: {e}')
        return []

def parse_detail(url):
    """解析详情页正文（SSR渲染）"""
    try:
        html = fetch_html(url)
        soup = BeautifulSoup(html, 'lxml')
        div = soup.find('div', class_='g-detailbox')
        if div:
            for tag in div(['script', 'style']):
                tag.decompose()
            html_content = str(div)
            html_content = re.sub(r'<!--富文本-->', '', html_content)
            return html_content
    except Exception as e:
        print(f'  ⚠️ 详情解析失败 {url}: {e}')
    return ''

def run(site_name, branch_id, code, max_pages=None):
    total_pages = get_total_pages(branch_id, code)
    if max_pages and max_pages < total_pages:
        total_pages = max_pages
    print(f'📋 {site_name}: 共 {total_pages} 页')

    conn = sqlite3.connect(DB_PATH)
    new, skip = 0, 0

    for pg in range(1, total_pages + 1):
        items = get_list_page(branch_id, code, pg)
        if not items:
            print(f'  ⚠️ 第{pg}页无数据')
            continue
        print(f'📄 第{pg}/{total_pages}页 ({len(items)}条)')
        for item in items:
            body = parse_detail(item['url'])
            try:
                conn.execute("""
                    INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date, content, summary)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (site_name, item['url'], item['url'], item['title'],
                      item['date'], body, item['title']))
                if conn.total_changes > 0:
                    new += 1
                else:
                    skip += 1
            except Exception as e:
                print(f'  ❌ - {e}')
            time.sleep(0.3)
        conn.commit()

    conn.close()
    print(f'\n✅ {site_name}: 新增{new}, 跳过{skip}, 共{total_pages}页')

if __name__ == '__main__':
    args = sys.argv[1:]
    site_name = args[0] if len(args) >= 1 else DEFAULT_SITE
    branch_id = args[1] if len(args) >= 2 else DEFAULT_BRANCH
    code = args[2] if len(args) >= 3 else DEFAULT_CODE
    max_p = int(args[3]) if len(args) >= 4 and args[3].isdigit() else None
    run(site_name, branch_id, code, max_p)
