#!/usr/bin/env python3
"""


潜江市人民政府 (hbqj.gov.cn) - 建设项目环境影响评价公示爬虫
站点：https://www.hbqj.gov.cn/xxgk/xxgkml/szfxxgkml/gysyjs/hjbh/jsxmhjyxpjgs/
CMS: TRS
列表：index.html + index_{n}.html 分页(25页×18条=450条)
"""
import requests
import re
import json
import time
import os
import sys
from bs4 import BeautifulSoup

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
BASE_URL = 'https://www.hbqj.gov.cn'
LIST_PATH = '/xxgk/xxgkml/szfxxgkml/gysyjs/hjbh/jsxmhjyxpjgs/'
LIST_URL = BASE_URL + LIST_PATH
SITE_NAME = '潜江市人民政府-建设项目环境影响评价公示'
GROUP = '潜江市'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

TOTAL_PAGES = 25

def fetch(url, encoding='utf-8'):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = encoding
    return r.text

def extract_list_items(html):
    """解析列表页"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    
    # Find all TRS article links (pattern: tYYYYMM_ID.html)
    for a_tag in soup.find_all('a', href=re.compile(r't\d+_\d+\.html')):
        href = a_tag.get('href', '')
        if not href.startswith('http'):
            if href.startswith('/'):
                href = BASE_URL + href
            else:
                href = BASE_URL + '/' + href
        
        # Title from title attribute (complete)
        title = a_tag.get('title', '')
        if not title:
            title = a_tag.get_text(strip=True)
        if not title or len(title) < 8:
            continue
        
        # Skip navigation links
        if title in ('市场准入负面清单', '首页', '上一页', '下一页', '末页', '尾页'):
            continue
        
        items.append({'title': title, 'link': href})
    
    return items

def extract_detail(html, url):
    """解析详情页"""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from meta
    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()
    
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            t = title_tag.get_text(strip=True)
            # Clean up: remove site name suffix
            t = re.sub(r'[_\s]*潜江市人民政府\s*$', '', t).strip()
            if t:
                title = t
    
    # Date from meta
    date_str = ''
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date and meta_date.get('content'):
        date_match = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', meta_date['content'])
        if date_match:
            date_str = date_match.group(1)
    
    # Content - TRS_UEDITOR
    content = ''
    content_div = soup.find('div', class_=lambda c: c and 'TRS_UEDITOR' in c)
    if content_div:
        content = str(content_div)
    
    if not content:
        # Fallback: look for view class
        for cls_val in ['view', 'detail', 'xxgk_article', 'con_text']:
            for div in soup.find_all('div', class_=lambda c, cv=cls_val: c and cv in c):
                text = div.get_text(strip=True)
                if len(text) > 100:
                    content = str(div)
                    break
            if content:
                break
    
    # Attachments
    attachments = []
    for a_tag in soup.find_all('a', href=re.compile(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', re.I)):
        href = a_tag.get('href', '')
        if href.startswith('/'):
            href = BASE_URL + href
        elif href.startswith('//'):
            href = 'https:' + href
        elif not href.startswith('http'):
            href = BASE_URL + '/' + href
        
        name = a_tag.get_text(strip=True) or href.split('/')[-1]
        attachments.append({'url': href, 'name': name})
    
    return {
        'title': title,
        'date': date_str,
        'content': content.strip(),
        'attachments': attachments
    }

def crawl_all(incremental=False):
    """爬取全部文章"""
    if incremental:
        page_range = [1]
    else:
        page_range = list(range(1, min(TOTAL_PAGES, _MAX_PG or TOTAL_PAGES)+1))
    
    results = []
    
    for page_num in page_range:
        if page_num == 1:
            url = LIST_URL
        else:
            url = BASE_URL + LIST_PATH + 'index_%d.html' % page_num
        
        print('  Fetching list page %d/%d...' % (page_num, TOTAL_PAGES))
        html = fetch(url)
        items = extract_list_items(html)
        
        if not items:
            print('    No more items, stopping')
            break
        
        print('    Found %d items' % len(items))
        
        for item in items:
            print('  Fetching: %s' % item['title'][:50])
            try:
                detail_html = fetch(item['link'])
                detail = extract_detail(detail_html, item['link'])
                
                result = {
                    'title': detail['title'] or item['title'],
                    'link': item['link'],
                    'date': detail['date'],
                    'content': detail['content'],
                    'attachments': detail['attachments'],
                    'site_name': SITE_NAME,
                    'group': GROUP
                }
                results.append(result)
                time.sleep(0.3)
            except Exception as e:
                print('    ERROR: %s' % e)
                results.append({
                    'title': item['title'],
                    'link': item['link'],
                    'date': '',
                    'content': '',
                    'attachments': [],
                    'site_name': SITE_NAME,
                    'group': GROUP
                })
    
    return results

def export_jsonl(results, output_path):
    with open(output_path, 'w', encoding='utf-8') as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + '\n')
    print('Exported %d records to %s' % (len(results), output_path))

if __name__ == '__main__':
    output_dir = '/root/gov_crawler/output'
    os.makedirs(output_dir, exist_ok=True)
    
    incremental = '--incremental' in sys.argv or '1' in sys.argv
    
    print('Crawling %s (incremental=%s)' % (SITE_NAME, incremental))
    results = crawl_all(incremental=incremental)
    
    if results:
        jsonl_path = os.path.join(output_dir, 'hbqj_eia.jsonl')
        export_jsonl(results, jsonl_path)
        print('\nDone! %d articles total' % len(results))
    else:
        print('No results found!')
