#!/usr/bin/env python3
"""
柳河县人民政府 - 环境审批 爬虫
URL: http://www.jllh.gov.cn/xxgk/hjbh/hjsp/
CMS: 自定义 (7页)
"""
import os, re, json, time, sqlite3
from datetime import datetime, date
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE = "http://www.jllh.gov.cn/xxgk/hjbh/hjsp/"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
SITE_NAME = "柳河县环境审批"
CATEGORY = "hjsp"
THRESHOLD = date(2023, 6, 18)

COUNT_NEW = COUNT_SKIP = COUNT_ERR = 0

def get_soup(url, timeout=20):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = 'utf-8'
    return BeautifulSoup(r.text, 'html.parser')

def parse_date(s):
    s = s.strip().replace('.', '-')
    try:
        return datetime.strptime(s[:10], '%Y-%m-%d').date()
    except:
        return None

def date_rank(d):
    return int(d.strftime('%Y%m%d'))

def fetch_detail(url):
    try:
        soup = get_soup(url, timeout=20)
    except Exception as e:
        return None, None, None, str(e)
    
    # Title: <h2> in div.lhx_news
    title = ''
    news_div = soup.find('div', class_='lhx_news')
    if news_div:
        h2 = news_div.find('h2')
        if h2:
            title = h2.get_text(strip=True)
    
    # Date: <h4>发布时间：YYYY-MM-DD
    pubdate = None
    if news_div:
        h4 = news_div.find('h4')
        if h4:
            m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', h4.get_text())
            if m:
                pubdate = parse_date(m.group(1))
    
    # Content: div.TRS_Editor inside div.lhx_news
    content_html = ''
    if news_div:
        editor = news_div.find('div', class_='TRS_Editor')
        if editor:
            content_html = str(editor)
        else:
            # fallback: everything after <h4>
            for child in news_div.children:
                if hasattr(child, 'name') and child.name == 'p':
                    content_html = str(child)
                    break
    
    if not title:
        t = soup.find('title')
        if t:
            title = t.get_text(strip=True).split('|')[0].strip()
    
    return title, pubdate, content_html, None

def safe_insert(c, conn, values, max_retries=10):
    for attempt in range(max_retries):
        try:
            c.execute("""INSERT OR IGNORE INTO gov_raw 
                (site_name, title, page_url, source_url, publish_date, date_rank, summary, status, category, content, visits, tags)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 0, '')""", values)
            return c.rowcount > 0, None
        except sqlite3.OperationalError as e:
            if 'locked' in str(e):
                time.sleep(1)
                continue
            return False, str(e)
    return False, "DB locked after %d retries" % max_retries

def main():
    global COUNT_NEW, COUNT_SKIP, COUNT_ERR
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    # Get total pages from page 0
    soup = get_soup(BASE)
    total_pages = 0
    for s in soup.find_all('script'):
        if s.string and 'countPage' in s.string:
            m = re.search(r'var countPage\s*=\s*(\d+)', s.string)
            if m:
                total_pages = int(m.group(1))
                break
    
    print("[%s] 共 %d 页" % (SITE_NAME, total_pages))
    
    for page in range(total_pages):
        if page == 0:
            url = BASE
        else:
            url = "%sindex_%d.html" % (BASE, page)
        
        try:
            soup = get_soup(url, timeout=20)
        except Exception as e:
            COUNT_ERR += 1
            print("  [ERR] 第%d页: %s" % (page+1, e))
            continue
        
        items = soup.select('ul.ul_list > li')
        if not items:
            continue
        
        for li in items:
            a = li.find('a')
            span = li.find('span')
            if not a or not span:
                continue
            
            href = a.get('href', '').strip()
            title = a.get('title', '') or a.get_text(strip=True)
            date_str = span.get_text(strip=True)
            item_date = parse_date(date_str)
            
            if not href:
                continue
            href = urljoin(BASE, href) if not href.startswith('http') else href
            
            if item_date and item_date < THRESHOLD:
                COUNT_SKIP += 1
                continue
            
            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
            if c.fetchone():
                COUNT_SKIP += 1
                continue
            
            det_title, det_date, content_html, err = fetch_detail(href)
            if err:
                COUNT_ERR += 1
                print("  [ERR] %s: %s" % (href[:60], err))
                continue
            
            final_title = det_title or title
            final_date = det_date or item_date
            if not final_date:
                COUNT_SKIP += 1
                continue
            
            if not content_html or len(content_html.strip()) < 50:
                COUNT_SKIP += 1
                continue
            
            date_str_final = final_date.strftime('%Y-%m-%d')
            dr = date_rank(final_date)
            summary = BeautifulSoup(content_html, 'html.parser').get_text(strip=True)[:200]
            
            values = (SITE_NAME, final_title, href, href, date_str_final, dr, summary, 'published', CATEGORY, content_html)
            ok, e = safe_insert(c, conn, values)
            if ok:
                COUNT_NEW += 1
                if COUNT_NEW <= 3:
                    print("  + %s | %s" % (final_title[:50], date_str_final))
            elif e:
                COUNT_ERR += 1
                print("  [ERR] 入库: %s" % e)
        
        time.sleep(0.3)
    
    conn.commit()
    conn.close()
    
    print("\n===== %s 完成 =====" % SITE_NAME)
    print("新增: %d" % COUNT_NEW)
    print("跳过: %d" % COUNT_SKIP)
    print("错误: %d" % COUNT_ERR)

if __name__ == '__main__':
    main()
