#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
分宜县人民政府 - 建设项目环境影响评价文件审批 爬虫
UCAP CMS
List: http://www.fenyi.gov.cn/fenyi/jsxmhj1/bmxxgk_list.shtml (+ _2.shtml ~ _7.shtml)
Detail: http://www.fenyi.gov.cn/fenyi/jsxmhj1/YYYY-MM/DD/content_xxx.shtml
"""
import requests, re, json, sqlite3, time, os, sys
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = '/root/search.db'
SCRIPT_NAME = 'crawl_fenyi_hpj.py'
SITE_NAME = '分宜县-建设项目环评审批'
PROVINCE = '江西'
BASE_URL = 'http://www.fenyi.gov.cn'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

conn = sqlite3.connect(DB_PATH, timeout=60)
cur = conn.cursor()
total_new = 0

def save_article(title, page_url, publish_date, content_text, attachments):
    global total_new
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    try:
        cur.execute("""
            INSERT OR IGNORE INTO gov_raw (title, page_url, publish_date, content, site_name, category, attachments)
            VALUES (?, ?, ?, ?, ?, ?, ?)
        """, (title.strip(), page_url, publish_date, content_text.strip(), SITE_NAME, '建设项目环境影响评价文件审批', attachments_json))
        if cur.rowcount > 0:
            total_new += 1
            return True
    except Exception as e:
        pass
    return False

def table_to_md(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def fetch_detail(url):
    """Fetch detail page with UCAP CMS content"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        html = r.text
    except:
        return None, []
    
    soup = BeautifulSoup(html, 'html.parser')
    
    # Find content div: UCAP uses wzcon or content div
    content_div = soup.find('div', class_='wzcon')
    if not content_div:
        content_div = soup.find('ucapcontent')
    if not content_div:
        content_div = soup.find('div', id=lambda i: i and 'content' in i.lower() if i else None)
    if not content_div:
        return None, []
    
    # Remove scripts and styles
    for s in content_div.find_all('script'):
        s.decompose()
    for s in content_div.find_all('style'):
        s.decompose()
    
    # Process tables
    table_tags = content_div.find_all('table')
    table_mds = []
    for t in table_tags:
        all_text = t.get_text(strip=True)
        if len(all_text) < 15:
            t.decompose()
            continue
        md = table_to_md(t)
        if md:
            table_mds.append(md)
        t.decompose()
    
    # Get text
    raw = content_div.get_text()
    lines = [l.strip() for l in raw.split('\n') if l.strip() and len(l.strip()) > 3]
    
    # Filter noise
    skip_kw = ['【打印本页】', '【关闭窗口】', '打印本页', '关闭窗口', '扫一扫在手机打开当前页']
    filtered = [l for l in lines if not any(s in l for s in skip_kw)]
    
    result = []
    for line in filtered:
        if any(k in line for k in ['来源：', '发布时间：']):
            continue
        result.append(line)
    
    # Insert table(s)
    if table_mds:
        insert_pos = -1
        for i, p in enumerate(result):
            if any(k in p for k in ['联系电话', '通讯地址']):
                insert_pos = i
                break
        if insert_pos > 0:
            # Insert after the paragraph before 通讯地址
            result.insert(insert_pos, table_mds[0])
        else:
            result.append(table_mds[0])
    
    content_text = '\n\n'.join(result)
    
    # Attachments
    attachments = []
    for a in content_div.find_all('a', href=True):
        href = a['href']
        name = a.get_text(strip=True) or href.split('/')[-1].split('?')[0]
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            if href.startswith('/'):
                href = BASE_URL + href
            elif not href.startswith('http'):
                href = BASE_URL + '/' + href.lstrip('/')
            attachments.append({'name': name, 'url': href})
    
    seen = set()
    unique = []
    for a in attachments:
        if a['url'] not in seen:
            seen.add(a['url'])
            unique.append(a)
    
    return content_text, unique


def extract_list(html):
    """Extract articles from list page"""
    soup = BeautifulSoup(html, 'html.parser')
    articles = []
    
    ul = soup.find('ul', class_=lambda c: c and 'pageList' in ' '.join(c).lower() if c else False)
    if not ul:
        for u in soup.find_all('ul'):
            lis = u.find_all('li')
            if len(lis) > 5:
                ul = u
                break
    
    if not ul:
        return articles
    
    for li in ul.find_all('li'):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a['href']
        title = a.get('title', '') or a.get_text(strip=True)
        if len(title) < 10:
            continue
        
        if href.startswith('/'):
            full_url = BASE_URL + href
        elif not href.startswith('http'):
            full_url = BASE_URL + '/' + href.lstrip('/')
        else:
            full_url = href
        
        # Find date from span.time
        time_span = li.find('span', class_='time')
        if time_span:
            date = time_span.get_text(strip=True)
        else:
            txt = li.get_text()
            dm = re.search(r'\d{4}-\d{2}-\d{2}', txt)
            date = dm.group(1) if dm else ''
        
        articles.append((title, full_url, date))
    
    return articles


# Main
print('Scraping 分宜县-建设项目环评审批...')
new_items = 0
total_pages = 7

for page in range(1, total_pages + 1):
    if page == 1:
        url = f'{BASE_URL}/fenyi/jsxmhj1/bmxxgk_list.shtml'
    else:
        url = f'{BASE_URL}/fenyi/jsxmhj1/bmxxgk_list_{page}.shtml'
    
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        print(f'  Page {page} error: {e}')
        break
    
    articles = extract_list(html)
    if not articles:
        print(f'  Page {page}: no articles, stopping')
        break
    
    for title, page_url, date in articles:
        if not date:
            continue
        if date < '2020-01-01':
            continue
        
        content_text, attachments = fetch_detail(page_url)
        if not content_text:
            content_text = f'<p><a href="{page_url}">{title}</a></p>'
        
        save_article(title, page_url, date, content_text, attachments)
    
    print(f'  Page {page}/{total_pages}: {len(articles)} articles, new so far: {total_new}')

conn.commit()
conn.close()
print(f'\nDone! New: {total_new}')
