#!/usr/bin/env python3
"""
新绛县人民政府 - 新绛环保宣传专栏 爬虫
"""

import requests, re, time, sqlite3
from datetime import datetime, date
from bs4 import BeautifulSoup
import os

SITE_NAME = '新绛县人民政府-新绛环保宣传专栏'
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = date(2023, 6, 1)
BASE = 'http://www.jiangzhou.gov.cn'

def get_list(page):
    if page == 1:
        url = f'{BASE}/ztzl/xjhbxczl/index.shtml'
    else:
        url = f'{BASE}/ztzl/xjhbxczl/index_{page}.shtml'
    try:
        r = requests.get(url, headers={'User-Agent': 'Mozilla/5.0'}, timeout=15)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        ul = soup.find('ul', class_='List_list')
        if not ul:
            return []
        items = []
        for li in ul.find_all('li'):
            a = li.find('a')
            if not a:
                continue
            href = a.get('href', '')
            title = a.get_text(strip=True)
            if not href or not title:
                continue
            if href.startswith('/'):
                href = BASE + href
            # Extract date - look for span or text after link
            txt = li.get_text(strip=True)
            # Remove the title text
            date_part = txt.replace(title, '').strip().strip('[]()').strip()
            # Parse date format YYYY-MM-DD
            dates = re.findall(r'(\d{4}-\d{2}-\d{2})', date_part)
            pub_date = dates[0] if dates else ''
            if pub_date and title:
                items.append({'url': href, 'title': title, 'pub_date': pub_date})
        return items
    except Exception as e:
        print(f'  [ERR] page {page}: {e}')
        return []

def fetch_detail(url):
    try:
        r = requests.get(url, headers={'User-Agent': 'Mozilla/5.0'}, timeout=20)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'lxml')
        # Try #Zoom first
        zoom = soup.find(id='Zoom')
        if zoom:
            return re.sub(r'\n\s*\n', '\n', str(zoom))
        # Then ConBox_nr
        cnr = soup.find('div', class_='ConBox_nr')
        if cnr:
            return re.sub(r'\n\s*\n', '\n', str(cnr))
        return ''
    except:
        return ''

def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_added = 0
    no_content = 0

    for page in range(1, 21):
        print(f'Page {page}/20...', end=' ', flush=True)
        items = get_list(page)
        if not items:
            print('empty')
            continue
        
        page_added = 0
        all_before = True
        for item in items:
            try:
                d = datetime.strptime(item['pub_date'], '%Y-%m-%d').date()
            except:
                d = date(1900, 1, 1)
            if d < CUTOFF_DATE:
                continue
            all_before = False
            
            c.execute('SELECT 1 FROM gov_raw WHERE page_url = ?', (item['url'],))
            if c.fetchone():
                continue
            
            content = fetch_detail(item['url'])
            if not content:
                no_content += 1
            
            c.execute('''INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date, content, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?)''',
                (SITE_NAME, item['url'], item['url'], item['title'],
                 item['pub_date'], content, item['title']))
            page_added += 1
            total_added += 1
            if total_added % 5 == 0:
                time.sleep(0.3)
        
        print(f'+{page_added}')
        conn.commit()
        if all_before:
            print('  All before cutoff, stop')
            break
        time.sleep(0.5)

    conn.close()
    print(f'\nDone! Added: {total_added}, NoContent: {no_content}')

if __name__ == '__main__':
    main()
