#!/usr/bin/env python3
"""犍为县-公示公告 爬虫 (自定义CMS)"""
import requests, re, sqlite3, sys, os, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

VERIFY = False
requests.packages.urllib3.disable_warnings()
BASE = 'http://www.qianwei.gov.cn/qwx/gsgg'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
DB = '/root/search.db'
SITE_NAME = '犍为县_公示公告'
TOTAL_PAGES = 34
MAX_PAGES = 34

def get_text(soup):
    return soup.get_text(separator=' ', strip=True) if soup else ''

def extract_content(html):
    soup = BeautifulSoup(html, 'html.parser')
    detail = soup.find('div', class_='detail')
    if not detail:
        return '', []
    
    # Attachments
    attachments = []
    for a in detail.find_all('a', href=True):
        h = a['href']
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', h, re.I):
            attachments.append(urljoin(BASE, h))
    
    # Get all paragraphs - content is in &nbsp; separated text
    text_parts = []
    for p in detail.find_all('p'):
        txt = p.get_text(strip=True)
        if txt and txt != '&nbsp;':
            text_parts.append(txt)
    
    # Also get text directly from detail div for non-p content
    for br in detail.find_all('br'):
        br.replace_with('\n')
    
    all_text = detail.get_text(separator='\n', strip=True)
    # Split by newlines and filter
    lines = [l.strip() for l in all_text.split('\n') if l.strip()]
    
    # Remove header/footer lines
    skip_patterns = ['文章来源', '发布时间', '浏览量', '打印', '分享']
    clean_lines = []
    for l in lines:
        if any(p in l for p in skip_patterns):
            continue
        if l in ['公示公告', '&nbsp;']:
            continue
        clean_lines.append(l)
    
    content = '\n\n'.join(clean_lines) if clean_lines else '\n\n'.join(text_parts)
    return content, attachments

def parse_list_page(url):
    try:
        r = requests.get(url, headers=HEADERS, verify=VERIFY, timeout=30)
        r.encoding = 'utf-8'
        if len(r.text) < 500:
            return []
    except Exception as e:
        print('[ERROR] {}: {}'.format(url, e), flush=True)
        return []
    
    soup = BeautifulSoup(r.text, 'html.parser')
    items = []
    
    for li in soup.find_all('li', id='content_li'):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        href = urljoin(url, a['href'])
        title = a.get_text(strip=True)
        if not title:
            title = a.get('title', '')
        if not title:
            continue
        span = li.find('span')
        date = span.get_text(strip=True) if span else ''
        items.append((title, href, date))
    
    return items

def get_existing_urls():
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    urls = set()
    for row in c.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
        urls.add(row[0])
    conn.close()
    return urls

def save_batch(items):
    if not items:
        return 0
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    saved = 0
    for item in items:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, page_url, content, summary, publish_date, site_name) VALUES (?,?,?,?,?,?)",
                (item['title'], item['url'], item['content'], item['content'][:500], item['pub_date'], SITE_NAME)
            )
            if c.rowcount > 0:
                conn.commit()
                saved += 1
        except Exception as e:
            print('[DB ERROR] {}'.format(e), flush=True)
            conn.rollback()
    conn.close()
    return saved

def get_page_url(page):
    if page == 1:
        return BASE + '/glist.shtml'
    else:
        return BASE + '/glist_{}.shtml'.format(page)

def main():
    existing_urls = get_existing_urls()
    print('[START] Already in DB: {}'.format(len(existing_urls)), flush=True)
    
    # Collect all list items from all pages
    all_items = []
    for page in range(1, TOTAL_PAGES + 1):
        url = get_page_url(page)
        print('[PAGE {}/{}]'.format(page, TOTAL_PAGES), flush=True)
        items = parse_list_page(url)
        if not items:
            print('  Empty page {}, stopping'.format(page), flush=True)
            break
        all_items.extend(items)
        print('  {} items'.format(len(items)), flush=True)
        time.sleep(0.5)
    
    print('[TOTAL] {} items from list'.format(len(all_items)), flush=True)
    
    # Crawl details
    batch = []
    total_new = 0
    
    for i, (title, detail_url, date) in enumerate(all_items, 1):
        if detail_url in existing_urls:
            continue
        
        try:
            r = requests.get(detail_url, headers=HEADERS, verify=VERIFY, timeout=30)
            r.encoding = 'utf-8'
        except Exception as e:
            print('  [SKIP] {} - {}'.format(title[:40], e), flush=True)
            continue
        
        content, attachments = extract_content(r.text)
        if not content or len(content) < 20:
            print('  [SKIP] No/short content: {}'.format(title[:40]), flush=True)
            continue
        
        batch.append({
            'title': title,
            'url': detail_url,
            'pub_date': date,
            'content': content,
            'attachments': '\n'.join(attachments) if attachments else ''
        })
        
        if len(batch) >= 20:
            saved = save_batch(batch)
            total_new += saved
            print('  [{}/{}] Batch saved: {}, cumulative: {}'.format(i, len(all_items), saved, total_new), flush=True)
            batch = []
        
        time.sleep(0.3)
    
    if batch:
        saved = save_batch(batch)
        total_new += saved
    
    print('[DONE] New: {}, Total in DB: {}'.format(total_new, len(existing_urls) + total_new), flush=True)

if __name__ == '__main__':
    main()
