#!/usr/bin/env python3
"""洛宁县人民政府-环境保护领域 爬虫 (静态HTML分页)"""
import re, requests, sqlite3, os, sys, time
from bs4 import BeautifulSoup

SITE_NAME = '洛宁县-环境保护领域'
BASE = 'https://www.luoning.gov.cn'
LIST = '/zwgk/jczwgkgfhbzh/hjbhly/index.html'
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

def get_total_pages():
    """从分页信息获取总页数"""
    r = requests.get(BASE + LIST, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'lxml')
    page_dec = soup.find('div', id='pageDec')
    if page_dec:
        pagecount = page_dec.get('pagecount', '0')
        pagesize = page_dec.get('pagesize', '24')
        try:
            return (int(pagecount) + int(pagesize) - 1) // int(pagesize)
        except:
            pass
    return 70  # fallback

def parse_list(page=1):
    url = BASE + LIST if page == 1 else BASE + f'/zwgk/jczwgkgfhbzh/hjbhly/index_{page}.html'
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
    except:
        return []
    soup = BeautifulSoup(r.text, 'lxml')
    items = []
    for li in soup.select('ul li'):
        span = li.find('span', class_='date')
        a = li.find('a')
        if a and span:
            href = a.get('href', '')
            if not href.startswith('http'):
                href = BASE + href
            items.append({
                'title': a.get_text(strip=True),
                'url': href,
                'date': span.get_text(strip=True).strip(),
            })
    return items

def parse_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        div = soup.find('div', class_='content')
        if div:
            for tag in div(['script', 'style']):
                tag.decompose()
            return str(div)
    except:
        pass
    return ''

def run(max_pages=None):
    total_pages = get_total_pages()
    if max_pages and max_pages < total_pages:
        total_pages = max_pages
    
    print(f'📋 共 {total_pages} 页')
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    new, skip = 0, 0
    
    for pg in range(1, total_pages + 1):
        items = parse_list(pg)
        if not items:
            break
        print(f'📄 第{pg}/{total_pages}页 ({len(items)}条)')
        for item in items:
            body = parse_detail(item['url'])
            try:
                conn.execute("""
                    INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date, content, summary)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (SITE_NAME, item['url'], item['url'], item['title'],
                      item['date'], body, item['title']))
                if conn.total_changes > 0:
                    new += 1
                else:
                    skip += 1
            except Exception as e:
                print(f'  ❌ - {e}')
            time.sleep(0.3)
        conn.commit()
    
    conn.close()
    print(f'\n✅ {SITE_NAME}: 新增{new}, 跳过{skip}, 共{total_pages}页')

if __name__ == '__main__':
    max_p = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    run(max_p)
