#!/usr/bin/env python3
"""福建省冶金工业设计院 - 环境评价公示 爬虫"""
import re, requests, sqlite3, os, sys, time, json
from bs4 import BeautifulSoup

SITE_NAME = '福建省冶金工业设计院-环境评价'
BASE = 'https://www.fjyjy.cn'
LIST = '/Home/Main?m=content&c=index&a=lists&catid=28'
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

def fetch_list():
    """列表页所有数据已内嵌在 JSON 中"""
    r = requests.get(BASE + LIST, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    m = re.search(r'<div id="jsonData" class="hide">(.*?)</div>', r.text, re.DOTALL)
    if not m:
        print('❌ 未找到 jsonData')
        return []
    items = json.loads(m.group(1).strip())
    result = []
    for item in items:
        title = item.get('porrname', '').strip()
        date = item.get('date', '').strip()
        id_val = item.get('id', '')
        if title and id_val:
            url = f'{BASE}/Home/Main?m=content&c=index&a=show&catid=28&id={id_val}'
            result.append({'title': title, 'url': url, 'date': date, 'id': id_val})
    return result

def parse_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        div = soup.find('div', class_='showCon') or soup.find('div', class_='boxCon')
        if div:
            for tag in div(['script', 'style']):
                tag.decompose()
            return str(div)
    except Exception as e:
        print(f'    ⚠️ {url[:60]} - {e}')
    return ''

def run(max_items=None):
    items = fetch_list()
    print(f'📋 列表页共 {len(items)} 条记录')
    
    if not items:
        return

    conn = sqlite3.connect(DB_PATH)
    new, skip = 0, 0
    
    limit = max_items if max_items else len(items)

    for idx, item in enumerate(items[:limit]):
        body = parse_detail(item['url'])
        try:
            conn.execute("""
                INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date, content, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (SITE_NAME, item['url'], item['url'], item['title'],
                  item['date'], body, item['title']))
            if conn.total_changes > 0:
                new += 1
                print(f'  ✅ [{idx+1}/{limit}] {item["title"][:50]}')
            else:
                skip += 1
        except Exception as e:
            print(f'  ❌ {item["title"][:30]} - {e}')
        time.sleep(0.5)
        
        if (idx + 1) % 20 == 0:
            conn.commit()

    conn.commit()
    conn.close()
    print(f'\n✅ {SITE_NAME}: 新增{new}, 跳过{skip}, 已处理{limit}条')
    if new > 0:
        conn2 = sqlite3.connect(DB_PATH)
        total = conn2.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
        conn2.close()
        print(f'   累计 {total} 条')

if __name__ == '__main__':
    max_n = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    run(max_n)
