#!/usr/bin/env python3
"""
石嘴山市惠农区人民政府 - 生态环境保护 爬虫
"""

import requests, re, time, sqlite3
from datetime import datetime, date
from bs4 import BeautifulSoup
import os

SITE_NAME = '惠农区人民政府-生态环境保护'
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = date(2023, 6, 1)
BASE = 'https://www.huinong.gov.cn'
HEADERS = {'User-Agent': 'Mozilla/5.0'}
COL_PATH = '/zwgk/fdzdgknr/shgysyjsly/sthjbh'

def get_list(page):
    if page == 1:
        url = f'{BASE}{COL_PATH}/'
    else:
        url = f'{BASE}{COL_PATH}/index_{page}.html'
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        hn = soup.find('div', id='hn-content')
        if not hn:
            return []
        ul = hn.find('ul')
        if not ul:
            return []
        items = []
        for li in ul.find_all('li'):
            a = li.find('a')
            span = li.find('span')
            if not a or not span:
                continue
            href = a.get('href', '')
            title = (a.get("title") or a.get_text(strip=True) or "").strip()
            pub_date = span.get_text(strip=True)
            if href and title and pub_date:
                # Handle relative URLs
                if href.startswith('./'):
                    href = f'{BASE}{COL_PATH}/{href[2:]}'
                elif href.startswith('/'):
                    href = f'{BASE}{href}'
                elif not href.startswith('http'):
                    href = f'{BASE}{COL_PATH}/{href}'
                items.append({'url': href, 'title': title, 'pub_date': pub_date})
        return items
    except Exception as e:
        print(f'  [ERR] page {page}: {e}')
        return []

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        # Try #zoom first
        zoom = soup.find(id='zoom')
        if zoom:
            return str(zoom)
        # Then g-detailbox
        gd = soup.find('div', class_='g-detailbox')
        if gd:
            return str(gd)
        return ''
    except:
        return ''

def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_added = 0
    no_content = 0

    for page in range(1, 7):
        print(f'Page {page}/6...', end=' ', flush=True)
        items = get_list(page)
        if not items:
            print('empty')
            continue

        page_added = 0
        for item in items:
            try:
                d = datetime.strptime(item['pub_date'], '%Y-%m-%d').date()
            except:
                d = date(1900, 1, 1)
            if d < CUTOFF_DATE:
                continue

            c.execute('SELECT 1 FROM gov_raw WHERE page_url = ?', (item['url'],))
            if c.fetchone():
                continue

            content = fetch_detail(item['url'])
            if not content:
                no_content += 1

            c.execute('''INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date, content, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?)''',
                (SITE_NAME, item['url'], item['url'], item['title'],
                 item['pub_date'], content, item['title']))
            page_added += 1
            total_added += 1
            if total_added % 10 == 0:
                time.sleep(0.3)

        print(f'+{page_added}')
        conn.commit()
        time.sleep(0.5)

    conn.close()
    print(f'\nDone! Added: {total_added}, NoContent: {no_content}')

if __name__ == '__main__':
    main()
