#!/usr/bin/env python3
"""宁陵县-建设项目环境影响评价信息 爬虫 (PowerEasy CMS)
URL: https://www.ningling.gov.cn/zwgk/fdzdgknr/zdlygk/hjbh/jsxmhjyxpjxx
分页: /jsxmhjyxpjxx (第1页), /jsxmhjyxpjxx_2 (第2页), ..., /jsxmhjyxpjxx_29 (第29页)
"""
import requests, re, sqlite3, sys
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from concurrent.futures import ThreadPoolExecutor, as_completed

DB_PATH = '/root/search.db'
BASE_URL = 'https://www.ningling.gov.cn'
LIST_URL = '/zwgk/fdzdgknr/zdlygk/hjbh/jsxmhjyxpjxx'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
SITE_NAME = '宁陵县-建设项目环境影响评价信息'
TOTAL_PAGES = 29
MAX_WORKERS = 8

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=60000")
    return conn

def crawl_page(page_url):
    articles = []
    try:
        resp = requests.get(page_url, headers=HEADERS, timeout=15, verify=False)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return articles
        soup = BeautifulSoup(resp.text, 'html.parser')
        for a in soup.select('a[href*="content_"][target="_blank"]'):
            href = a.get('href', '')
            if not href.startswith('http'):
                href = BASE_URL + href
            title_attr = a.get('title', '')
            title = ''
            pub_date = ''
            if title_attr:
                m_t = re.search(r'标题[：:]\s*(.*?)(?:&#|\\r|\\n|\n|$)', title_attr)
                if m_t: title = m_t.group(1).strip()
                m_d = re.search(r'发表时间[：:]\s*(\d{4}-\d{2}-\d{2})', title_attr)
                if m_d: pub_date = m_d.group(1)
            if not title:
                title = a.get_text(strip=True)
            articles.append((href, title, pub_date))
    except Exception as e:
        print(f'[ERROR] 列表页: {page_url} - {e}', file=sys.stderr)
    return articles

def get_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return '', '', ''
        soup = BeautifulSoup(resp.text, 'html.parser')
        h2 = soup.select_one('h2.title')
        title = h2.get_text(strip=True) if h2 else ''
        prop = soup.select_one('.property')
        pub_date = ''
        if prop:
            m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', prop.get_text())
            if m: pub_date = m.group(1)
        content_el = soup.select_one('.conTxt')
        content = str(content_el) if content_el else ''
        return title, content, pub_date
    except Exception as e:
        print(f'[ERROR] 详情页: {url} - {e}', file=sys.stderr)
        return '', '', ''

def main():
    daily_mode = '--daily' in sys.argv
    conn = get_conn(); cur = conn.cursor()
    pages = [BASE_URL + LIST_URL] if daily_mode else [BASE_URL + LIST_URL] + [f'{BASE_URL}{LIST_URL}_{i}' for i in range(2, TOTAL_PAGES + 1)]
    all_articles = []
    for pu in pages:
        arts = crawl_page(pu)
        print(f'[列表页] {pu.split("/")[-1]} -> {len(arts)} 条')
        all_articles.extend(arts)
    to_fetch = []
    skip_date = skip_dup = 0
    for href, title, ld in all_articles:
        if ld < THREE_YEARS_AGO: skip_date += 1; continue
        cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
        if cur.fetchone(): skip_dup += 1; continue
        to_fetch.append((href, title, ld))
    print(f'\n需抓取详情页: {len(to_fetch)} 条')
    added = 0
    if to_fetch:
        with ThreadPoolExecutor(max_workers=MAX_WORKERS) as ex:
            fm = {ex.submit(get_detail, u): (u, t, d) for u, t, d in to_fetch}
            for i, ft in enumerate(as_completed(fm), 1):
                u, t, d = fm[ft]
                dt, c, dd = ft.result()
                ftitle = dt or t; fdate = dd or d
                sm = BeautifulSoup(c or '', 'html.parser').get_text(strip=True)[:500] if c else ''
                cur.execute("INSERT OR IGNORE INTO gov_raw (site_name,page_url,title,content,publish_date,summary) VALUES (?,?,?,?,?,?)",
                           (SITE_NAME, u, ftitle, c, fdate, sm))
                added += 1
                lb = f'[{added}/{len(to_fetch)}]' if not daily_mode else '  [ADD]'
                print(f'  {lb} {fdate} {ftitle[:40]}')
    conn.commit()
    print(f'\n同步 FTS ({added} 条新增)...')
    cur.execute("INSERT INTO gov_search(rowid, title, site_name, summary) SELECT r.id,r.title,r.site_name,r.summary FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?", (SITE_NAME,))
    conn.commit(); conn.close()
    print(f'\n=== 完成 ===\n新增: {added}, 跳过: {skip_dup}, 超3年: {skip_date}')

if __name__ == '__main__':
    main()
