#!/usr/bin/env python3
"""环县-重大项目 爬虫 (Hanweb CMS)"""
import requests, re, sqlite3, sys, os, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

VERIFY = False
requests.packages.urllib3.disable_warnings()
BASE = 'https://www.huanxian.gov.cn'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36'}
DB = '/root/search.db'
SITE_NAME = '环县_重大项目'

def get_text_safe(soup):
    if not soup: return ''
    return soup.get_text(separator=' ', strip=True)

def extract_content(html):
    soup = BeautifulSoup(html, 'html.parser')
    con = soup.find('div', class_='conTxt')
    if not con:
        return '', []
    attachments = []
    for a in con.find_all('a', href=True):
        h = a['href']
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', h, re.I):
            attachments.append(urljoin(BASE, h))
    paragraphs = []
    for p in con.find_all('p'):
        txt = p.get_text(strip=True)
        if txt:
            paragraphs.append(txt)
    return '\n\n'.join(paragraphs), attachments

def parse_list_page(url):
    try:
        r = requests.get(url, headers=HEADERS, verify=VERIFY, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        print(f'[ERROR] Fetch {url}: {e}', flush=True)
        return []
    soup = BeautifulSoup(r.text, 'html.parser')
    ul = soup.find('ul', class_='infoList')
    if not ul:
        return []
    items = []
    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        title = get_text_safe(a)
        if not title:
            continue
        href = urljoin(BASE, a['href'])
        date_span = li.find('span', class_='date')
        date = get_text_safe(date_span) if date_span else ''
        items.append((title, href, date))
    return items

def get_existing_urls():
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    urls = set()
    for row in c.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
        urls.add(row[0])
    conn.close()
    return urls

def save_batch(items):
    if not items:
        return 0
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    saved = 0
    for item in items:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, page_url, content, summary, publish_date, site_name) VALUES (?,?,?,?,?,?)",
                (item['title'], item['url'], item['content'], item['content'][:500], item['pub_date'], SITE_NAME)
            )
            if c.rowcount > 0:
                conn.commit()
                saved += 1
        except Exception as e:
            print(f'[DB ERROR] {e}', flush=True)
            conn.rollback()
    conn.close()
    return saved

def main():
    existing_urls = get_existing_urls()
    print(f'[START] Already have {len(existing_urls)} items in DB', flush=True)
    
    total_pages = 59
    batch = []
    total_new = 0
    
    for page in range(1, total_pages + 1):
        if page == 1:
            url = f'{BASE}/zwgk/fdzdgknr/zdxm'
        else:
            url = f'{BASE}/zwgk/fdzdgknr/zdxm_{page}'
        
        print(f'[PAGE {page}/{total_pages}]', flush=True)
        items = parse_list_page(url)
        if not items:
            print(f'  Empty page {page}, stopping', flush=True)
            break
        
        print(f'  {len(items)} items on page', flush=True)
        
        for i, (title, detail_url, date) in enumerate(items, 1):
            if detail_url in existing_urls:
                continue
            
            try:
                r = requests.get(detail_url, headers=HEADERS, verify=VERIFY, timeout=30)
                r.encoding = 'utf-8'
            except Exception as e:
                print(f'  [SKIP] {title[:40]} - {e}', flush=True)
                continue
            
            content, attachments = extract_content(r.text)
            if not content:
                print(f'  [SKIP] No content: {title[:40]}', flush=True)
                continue
            
            batch.append({
                'title': title,
                'url': detail_url,
                'pub_date': date,
                'content': content,
                'attachments': '\n'.join(attachments) if attachments else ''
            })
            
            if len(batch) >= 30:
                saved = save_batch(batch)
                total_new += saved
                print(f'  >> Saved {saved} (cumulative: {total_new})', flush=True)
                batch = []
            
            time.sleep(0.2)
        
        # Save per page
        if batch:
            saved = save_batch(batch)
            total_new += saved
            batch = []
        
        print(f'  [PAGE DONE] Cumulative new: {total_new}', flush=True)
        time.sleep(0.5)
    
    if batch:
        saved = save_batch(batch)
        total_new += saved
    
    final_count = len(get_existing_urls())
    print(f'[DONE] New this run: {total_new}, Total in DB: {final_count}', flush=True)

if __name__ == '__main__':
    main()
