#!/usr/bin/env python3
"""临沭县人民政府-公告/环境保护 通用爬虫 (VSB9静态站,反向分页)
用法: python3 crawl_linshu_vsb.py <site_name> <base_path> [max_pages]
示例: python3 crawl_linshu_vsb.py '临沭县-公告' 'yw/gsgg/gg' 1
      python3 crawl_linshu_vsb.py '临沭县-环境保护' 'gk/zfxxgk/fdzdgknr/zdly/shgysyjsly/hjbh'
"""
import re, requests, sqlite3, os, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE = 'http://www.linshu.gov.cn'
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

def fetch_html(url):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    return r.text

def get_total_pages_and_base(base_path):
    """获取总页数，同时返回page 1的完整HTML"""
    url = f'{BASE}/{base_path}.htm'
    html = fetch_html(url)
    soup = BeautifulSoup(html, 'lxml')
    # 从分页组件获取总页数
    pb = soup.find('div', class_='pb_sys_common')
    total_pages = 1
    if pb:
        # 找尾页链接的文字
        last_link = pb.find('a', class_='p_last') or pb.find('span', class_='p_last')
        if last_link:
            a = last_link.find('a')
            if a:
                href = a.get('href', '')
                m = re.search(r'/(\d+)\.htm', href)
                if m:
                    total_pages = int(m.group(1))
        if not last_link:
            # 从分页链接找到最大的数字
            nums = []
            for a in pb.find_all('a'):
                href = a.get('href', '')
                m = re.search(r'/(\d+)\.htm', href)
                if m:
                    nums.append(int(m.group(1)))
            if nums:
                total_pages = max(nums)
        # 如果找到了共X条, 每页15条推算
        total_text = pb.get_text()
        m = re.search(r'共(\d+)条', total_text)
        if m:
            total_count = int(m.group(1))
            items_per_page = len(soup.select('tr[id^="line"], li[id^="line"]'))
            if items_per_page > 0:
                total_pages_from_count = (total_count + items_per_page - 1) // items_per_page
                if total_pages_from_count > total_pages:
                    total_pages = total_pages_from_count
    return total_pages, soup

def get_list_page(base_path, total_pages, page=1):
    """获取列表页数据  page N 对应URL {base}/{total-N+1}.htm"""
    if page == 1:
        url = f'{BASE}/{base_path}.htm'
    else:
        # 反向编号: page N → {total_pages - N + 1}.htm
        rev_num = total_pages - page + 1
        url = f'{BASE}/{base_path}/{rev_num}.htm'
    try:
        html = fetch_html(url)
        soup = BeautifulSoup(html, 'lxml')
        items = []
        # 支持两种格式: tr表格 和 li列表
        rows = soup.select('tr[id^="line"]')
        is_table = len(rows) > 0
        if not is_table:
            rows = soup.select('li[id^="line"]')
        for row in rows:
            if is_table:
                tds = row.find_all('td')
                if len(tds) < 4:
                    continue
                a = tds[1].find('a')
                date_str = tds[3].get_text(strip=True)
            else:
                a = row.find('a')
                span = row.find('span')
                date_str = span.get_text(strip=True) if span else ''
            if a:
                href = a.get('href', '')
                if not href.startswith('http'):
                    href = urljoin(url, href)
                items.append({
                    'title': a.get_text(strip=True),
                    'url': href,
                    'date': date_str,
                })
        return items
    except Exception as e:
        print(f'  ⚠️ 第{page}页加载失败: {e}')
        return []

def parse_detail(url):
    """解析详情页正文 (VSB9)"""
    try:
        html = fetch_html(url)
        soup = BeautifulSoup(html, 'lxml')
        # 正文容器: div.newscontent_s 或 div#vsb_content_XXX
        div = soup.find('div', class_='newscontent_s')
        if not div:
            div = soup.find('div', id=re.compile(r'vsb_content_\d+'))
        if div:
            for tag in div(['script', 'style']):
                tag.decompose()
            return str(div)
    except Exception as e:
        print(f'  ⚠️ 详情解析失败 {url}: {e}')
    return ''

def run(site_name, base_path, max_pages=None):
    print(f'🔍 获取总页数...')
    total_pages, _ = get_total_pages_and_base(base_path)
    if max_pages and max_pages < total_pages:
        total_pages = max_pages
    print(f'📋 {site_name}: 共 {total_pages} 页')

    conn = sqlite3.connect(DB_PATH, timeout=60)
    new, skip = 0, 0

    for pg in range(1, total_pages + 1):
        items = get_list_page(base_path, total_pages, pg)
        if not items:
            print(f'  ⚠️ 第{pg}页无数据')
            continue
        print(f'📄 第{pg}/{total_pages}页 ({len(items)}条)')
        for item in items:
            body = parse_detail(item['url'])
            try:
                conn.execute("""
                    INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date, content, summary)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (site_name, item['url'], item['url'], item['title'],
                      item['date'], body, item['title']))
                if conn.total_changes > 0:
                    new += 1
                else:
                    skip += 1
            except Exception as e:
                print(f'  ❌ - {e}')
            time.sleep(0.3)
        conn.commit()

    conn.close()
    print(f'\n✅ {site_name}: 新增{new}, 跳过{skip}, 共{total_pages}页')

if __name__ == '__main__':
    if len(sys.argv) < 3:
        print(f'用法: python3 {sys.argv[0]} <site_name> <base_path> [max_pages]')
        sys.exit(1)
    site_name = sys.argv[1]
    base_path = sys.argv[2]
    max_p = int(sys.argv[3]) if len(sys.argv) >= 4 and sys.argv[3].isdigit() else None
    run(site_name, base_path, max_p)
