#!/usr/bin/env python3
"""凌源市人民政府-生态环境 爬虫 (glist 静态页)"""
import re, requests, sqlite3, os, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SITE_NAME = '凌源市-生态环境'
BASE = 'https://www.lingyuan.gov.cn'
LIST = '/lyszf/zwgk/fdzdgknr/wrfz/glist.html'
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

def get_total_pages():
    """获取总页数（从分页组件提取）"""
    r = requests.get(BASE + LIST, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    m = re.search(r'当前第\s*\d+/(\d+)\s*页', r.text)
    if m:
        return int(m.group(1))
    return 1

def parse_list(page=1):
    """解析列表页"""
    if page == 1:
        url = BASE + LIST
    else:
        url = BASE + f'/lyszf/zwgk/fdzdgknr/wrfz/glist_{page}.html'
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        print(f'  ⚠️ 请求失败 {url}: {e}')
        return []
    soup = BeautifulSoup(r.text, 'lxml')
    items = []
    for a in soup.find_all('a', objectid=True):
        href = a.get('href', '')
        if not href or '/html/LYSZF/' not in href:
            continue
        b = a.find('b')
        span = a.find('span')
        if not b or not span:
            continue
        title = b.get_text(strip=True)
        date = span.get_text(strip=True)
        if not href.startswith('http'):
            href = urljoin(url, href)
        items.append({
            'title': title,
            'url': href,
            'date': date,
        })
    return items

def parse_detail(url):
    """解析详情页正文"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        div = soup.find('div', class_='center-info')
        if not div:
            # 备用: 找包含最多p标签的div
            divs = soup.find_all('div')
            for d in divs:
                ps = d.find_all('p')
                if len(ps) >= 3:
                    div = d
                    break
        if div:
            for tag in div(['script', 'style', 'iframe']):
                tag.decompose()
            # 清理内联样式，保留HTML结构
            html = str(div)
            html = re.sub(r'\s*style="[^"]*"', '', html)
            html = re.sub(r'\s*class="[^"]*"', '', html)
            return html
    except Exception as e:
        print(f'  ⚠️ 详情解析失败 {url}: {e}')
    return ''

def run(max_pages=None):
    total_pages = get_total_pages()
    if max_pages and max_pages < total_pages:
        total_pages = max_pages

    print(f'📋 共 {total_pages} 页')

    conn = sqlite3.connect(DB_PATH, timeout=60)
    new, skip = 0, 0

    for pg in range(1, total_pages + 1):
        items = parse_list(pg)
        if not items:
            print(f'  ⚠️ 第{pg}页无数据')
            continue
        print(f'📄 第{pg}/{total_pages}页 ({len(items)}条)')
        for item in items:
            body = parse_detail(item['url'])
            try:
                conn.execute("""
                    INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date, content, summary)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (SITE_NAME, item['url'], item['url'], item['title'],
                      item['date'], body, item['title']))
                if conn.total_changes > 0:
                    new += 1
                else:
                    skip += 1
            except Exception as e:
                print(f'  ❌ - {e}')
            time.sleep(0.3)
        conn.commit()

    conn.close()
    print(f'\n✅ {SITE_NAME}: 新增{new}, 跳过{skip}, 共{total_pages}页')

if __name__ == '__main__':
    max_p = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    run(max_p)
