#!/usr/bin/env python3
"""
新疆生态环境厅 - 项目拟审批公示 爬虫
URL: https://sthjt.xinjiang.gov.cn/xjepd/hjyxpjnsp/common_list1.shtml
CMS: TRS - createPageHTML('page-div',10,1,'common_list1','shtml',100)
"""
import os, re, sys, json, time, requests
from bs4 import BeautifulSoup
from datetime import datetime, date

DB_PATH = os.environ.get("SEARCH_DB", "/root/search.db")
SITE_NAME = "新疆生态环境厅项目拟审批公示"
CATEGORY = "hjyxpjnsp"
THRESHOLD = date(2023, 6, 18)

BASE = "https://sthjt.xinjiang.gov.cn"
LIST_URL = BASE + "/xjepd/hjyxpjnsp/common_list1.shtml"
MAX_PAGES = 11
PAGE_SIZE = 10

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

COUNT_NEW = COUNT_SKIP = COUNT_ERR = 0

def safe_insert(cursor, conn, values):
    """Insert with retry for DB lock"""
    for attempt in range(10):
        try:
            cursor.execute("""INSERT OR IGNORE INTO gov_raw 
                (site_name, title, page_url, source_url, publish_date, date_rank, summary, status, category, content, visits, tags)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 0, '')""", values)
            return cursor.rowcount > 0, None
        except Exception as e:
            if 'locked' in str(e):
                time.sleep(1)
                continue
            return False, str(e)
    return False, "DB locked"

def date_rank(d):
    base = date(2020, 1, 1)
    return (d - base).days

def fetch_list_page(page_num):
    """Fetch list page and extract items"""
    if page_num == 1:
        url = LIST_URL
    else:
        url = LIST_URL.replace(".shtml", "_%d.shtml" % page_num)
    
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            print("  HTTP %d for page %d" % (r.status_code, page_num))
            return []
        
        soup = BeautifulSoup(r.text, 'html.parser')
        items = []
        
        # Find all list items with article links
        for li in soup.find_all('li'):
            a = li.find('a', href=re.compile(r'/xjepd/hjyxpjnsp/\d{6}/'))
            if not a:
                continue
            
            href = a.get('href', '')
            if not href:
                continue
            
            # Full URL
            if href.startswith('/'):
                full_url = BASE + href
            elif href.startswith('http'):
                full_url = href
            else:
                full_url = BASE + '/' + href
            
            title = a.text.strip()
            if not title:
                title = a.get('title', '')
            
            # Extract date
            spans = li.find_all('span')
            date_str = ''
            for span in spans:
                txt = span.text.strip()
                if re.match(r'\d{4}-\d{2}-\d{2}', txt):
                    date_str = txt
                    break
            
            if not date_str:
                # Try from URL path: /202606/...
                m = re.search(r'/(\d{6})/', href)
                if m:
                    yyyymm = m.group(1)
                    date_str = yyyymm[:4] + '-' + yyyymm[4:6] + '-01'
            
            items.append({
                'title': title,
                'url': full_url,
                'date': date_str
            })
        
        return items
    except Exception as e:
        print("  Error page %d: %s" % (page_num, e))
        return []

def fetch_detail(url):
    """Fetch detail page and extract HTML content"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return None, "HTTP %d" % r.status_code
        
        soup = BeautifulSoup(r.text, 'html.parser')
        
        # Find content - prefer the actual article body
        content_div = soup.find('div', id='NewsContent')
        if not content_div:
            content_div = soup.find('div', class_=lambda x: x and 'detail2-detail' in x if x else False)
        if not content_div:
            content_div = soup.find('div', class_=lambda x: x and 'TRS_UEDITOR' in x if x else False)
        if not content_div:
            # Try UCAPCONTENT
            ucap = soup.find('UCAPCONTENT')
            if ucap:
                return str(ucap), None
        if not content_div:
            # Try detail2-content (broader)
            content_div = soup.find('div', class_=lambda x: x and 'detail2-content' in x if x else False)
        if not content_div:
            # Try any div with significant text content
            for div in soup.find_all('div'):
                txt = div.text.strip()
                if len(txt) > 200 and '名称' in txt:
                    content_div = div
                    break
        
        if content_div:
            return str(content_div), None
        return None, "Content div not found"
    except Exception as e:
        return None, str(e)

def main():
    global COUNT_NEW, COUNT_SKIP, COUNT_ERR
    
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    print("[%s] 开始爬取..." % SITE_NAME)
    
    all_items = []
    
    for page in range(1, MAX_PAGES + 1):
        items = fetch_list_page(page)
        if not items:
            print("  第%d页: 无数据，停止翻页" % page)
            break
        
        all_items.extend(items)
        print("  第%d页: %d条，累计%d条" % (page, len(items), len(all_items)))
        
        # If page has fewer items than page_size, it's the last page
        if len(items) < PAGE_SIZE:
            print("  末页，停止")
            break
        
        time.sleep(0.5)
    
    print("\n共获取 %d 条列表数据" % len(all_items))
    
    # Process each item
    for i, item in enumerate(all_items):
        title = item['title']
        url = item['url']
        date_str = item['date']
        
        # Parse date
        try:
            item_date = datetime.strptime(date_str, '%Y-%m-%d').date()
        except:
            try:
                item_date = datetime.strptime(date_str[:7], '%Y-%m').date()
            except:
                item_date = None
        
        if not item_date:
            COUNT_ERR += 1
            print("  [%d/%d] %s - 日期无效" % (i+1, len(all_items), title[:40]))
            continue
        
        if item_date < THRESHOLD:
            COUNT_SKIP += 1
            continue
        
        # Check if exists
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
        if c.fetchone():
            COUNT_SKIP += 1
            continue
        
        # Fetch detail
        content_html, error = fetch_detail(url)
        if error:
            print("  [%d/%d] %s - 详情失败: %s" % (i+1, len(all_items), title[:40], error))
            COUNT_ERR += 1
            continue
        
        dr = date_rank(item_date)
        summary = title[:200]
        
        values = (SITE_NAME, title, url, url, date_str, dr, summary, 'published', CATEGORY, content_html)
        ok, e = safe_insert(c, conn, values)
        if ok:
            COUNT_NEW += 1
            print("  [%d/%d] + %s | %s" % (i+1, len(all_items), title[:50], date_str))
        elif e:
            COUNT_ERR += 1
        
        time.sleep(0.3)
    
    conn.commit()
    conn.close()
    
    print("\n===== %s 完成 =====" % SITE_NAME)
    print("新增: %d" % COUNT_NEW)
    print("跳过: %d" % COUNT_SKIP)
    print("错误: %d" % COUNT_ERR)

if __name__ == '__main__':
    main()
