#!/usr/bin/env python3
"""
和县人民政府 - 和县经济开发区管委会 意见反馈爬虫
URL: https://www.hx.gov.cn/xxgk/opennessContent/?branch_id=57a3df762c262ea9a00aad3b&column_code=30200
CMS: 和县人民政府信息公开系统
列表: AJAX加载 (/xxgk/opennessTarget/), 需token_verified Cookie绕过
详情: div.m-zw > h1标题 + div.u-funs信息行 + div.m-contnet正文
"""

import re
import sys
import json
import time
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'https://www.hx.gov.cn'
LIST_URL = BASE_URL + '/xxgk/opennessContent/?branch_id=57a3df762c262ea9a00aad3b&column_code=30200'
AJAX_URL = BASE_URL + '/xxgk/opennessTarget/?branch_id=57a3df762c262ea9a00aad3b&column_code=30200'
DB_PATH = '/root/search.db'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

site_name = '和县经济开发区管委会-意见反馈'
branch_id = '57a3df762c262ea9a00aad3b'
column_code = '30200'


def get_ajax_session():
    """获取带有token_verified Cookie的Session以访问AJAX接口"""
    session = requests.Session()
    session.headers.update(HEADERS)
    
    # Step 1: 访问主页面获取PHPSESSID
    resp = session.get(LIST_URL, timeout=30)
    
    # Step 2: 访问AJAX接口，触发token设置
    resp = session.get(AJAX_URL, headers={
        'X-Requested-With': 'XMLHttpRequest',
        'Referer': LIST_URL,
    }, timeout=30)
    
    # Step 3: 手动设置token_verified Cookie
    session.cookies.set('token_verified', 'true', domain='www.hx.gov.cn', path='/')
    
    return session


def fetch_list(session):
    """获取列表页文章列表"""
    resp = session.get(AJAX_URL, headers={
        'X-Requested-With': 'XMLHttpRequest',
        'Referer': LIST_URL,
    }, timeout=30)
    resp.encoding = 'utf-8'
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    items = []
    
    rows = soup.select('table tr')
    for row in rows:
        tds = row.find_all('td')
        if len(tds) < 3:
            continue
        
        a = tds[1].find('a')
        if not a or not a.get('href'):
            continue
        
        href = a['href'].strip()
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)
        
        title = a.get_text(strip=True)
        date = tds[2].get_text(strip=True)
        
        if title:
            items.append({'title': title, 'url': href, 'date': date})
    
    return items


def extract_detail(soup, url, session=None):
    """提取详情页内容"""
    # 标题 - h1
    title = ''
    h1 = soup.find('h1')
    if h1:
        title = h1.get_text(strip=True)
    
    # 发布日期
    publish_date = ''
    info_div = soup.find('div', class_='u-funs')
    if info_div:
        txt = info_div.get_text()
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', txt)
        if m:
            publish_date = m.group(1)
    
    # 来源
    content_source = ''
    if info_div:
        m = re.search(r'信息来源[：:]\s*([^ ]+?)(?:\s|$)', info_div.get_text())
        if m:
            content_source = m.group(1).strip()
    
    # 正文 - div.m-contnet
    content_parts = []
    images = []
    
    content_div = soup.find('div', class_='m-contnet')
    if not content_div:
        content_div = soup.find('div', class_='j-fontContent')
    
    if content_div:
        _extract_content(content_div, content_parts, images)
    
    content = '\n\n'.join(content_parts)
    
    # 检查正文是否太短或"无反馈意见"类，尝试从意见征集链接获取真实正文
    if (len(content.strip()) < 100 or '无反馈意见' in content) and session:
        right_div = soup.find('div', class_='m-right') or soup.find('div', class_='m-xgxx')
        if right_div:
            opinion_link = right_div.find('a')
            if opinion_link and opinion_link.get('href'):
                target_url = opinion_link['href'].strip()
                if not target_url.startswith('http'):
                    target_url = urljoin(BASE_URL, target_url)
                try:
                    resp = session.get(target_url, timeout=30)
                    resp.encoding = 'utf-8'
                    soup2 = BeautifulSoup(resp.text, 'html.parser')
                    target_div = soup2.find('div', class_='m-contnet')
                    if target_div:
                        target_parts = []
                        _extract_content(target_div, target_parts, images)
                        if target_parts:
                            content = '\n\n'.join(target_parts)
                            # 从原文取标题
                            h1_target = soup2.find('h1')
                            if h1_target:
                                content = '[原文标题: {0}]\n\n{1}'.format(h1_target.get_text(strip=True), content)
                            # 从原文取附件
                            for xgxx in soup2.find_all('div', class_='m-xgxx'):
                                for a in xgxx.find_all('a'):
                                    href = a.get('href', '')
                                    if href.endswith('.pdf') or href.endswith('.doc') or href.endswith('.docx'):
                                        if not href.startswith('http'):
                                            href = urljoin(BASE_URL, href)
                                        fname = a.get_text(strip=True) or href.split('/')[-1]
                                        content += '\n\n附件：<p><a href="{1}">{0}</a></p>'.format(fname, href)
                except Exception as e:
                    print('  WARNING: 获取意见征集正文失败: %s' % e)
    
    # 正文降级
    if len(content.strip()) < 20:
        content = '<p><a href="{1}">{0}</a></p>'.format(title, url)
    
    summary = content[:300] if len(content) > 300 else content
    
    return {
        'title': title,
        'content': content,
        'summary': summary,
        'publish_date': publish_date,
        'source_url': content_source,
        'attachments': '',
    }


def _extract_content(content_div, content_parts, images):
    """从内容div中提取段落、表格、图片、附件"""
    for child in content_div.find_all(['p', 'table', 'img', 'a'], recursive=True):
        if child.name == 'p':
            txt = child.get_text(strip=True)
            # 跳过"扫一扫"等无关内容
            if txt and len(txt) > 2 and '扫一扫' not in txt:
                content_parts.append(txt)
        elif child.name == 'table':
            content_parts.append(str(child))
        elif child.name == 'img':
            src = child.get('src', '') or child.get('filepath', '')
            alt = child.get('alt', '')
            if src and 'icon' not in src:
                if not src.startswith('http'):
                    src = urljoin(BASE_URL, src)
                images.append({'src': src, 'alt': alt})
                content_parts.append('![{0}]({1})'.format(alt, src))
        elif child.name == 'a':
            href = child.get('href', '')
            if href.endswith('.pdf') or href.endswith('.doc') or href.endswith('.docx'):
                if not href.startswith('http'):
                    href = urljoin(BASE_URL, href)
                fname = child.get_text(strip=True) or href.split('/')[-1]
                content_parts.append('[附件: {0}]({1})'.format(fname, href))


def main():
    import argparse
    parser = argparse.ArgumentParser(description='和县经济开发区管委会-意见反馈爬虫')
    parser.add_argument('--max-pages', type=int, default=1, help='最大页数(仅1页)')
    parser.add_argument('--full', action='store_true', help='全量(仅1页)')
    args = parser.parse_args()
    
    print('获取列表页...')
    
    try:
        session = get_ajax_session()
    except Exception as e:
        print('ERROR: 获取token失败: %s' % e)
        return
    
    items = fetch_list(session)
    print('获取到%d篇文章' % len(items))
    
    if not items:
        print('列表为空')
        return
    
    print('\n开始抓取详情...')
    conn = sqlite3.connect(DB_PATH, timeout=60)
    inserted = 0
    updated = 0
    errors = 0
    
    for idx, item in enumerate(items):
        try:
            title_short = item['title'][:40] if len(item['title']) > 40 else item['title']
            print('  [%d/%d] %s...' % (idx+1, len(items), title_short))
            
            resp = session.get(item['url'], timeout=30)
            resp.encoding = 'utf-8'
            soup = BeautifulSoup(resp.text, 'html.parser')
            
            detail = extract_detail(soup, item['url'], session)
            if not detail['title']:
                detail['title'] = item['title']
            
            page_url = item['url']
            publish_date = detail['publish_date'] or item['date'][:10]
            content = detail['content']
            summary = detail['summary']
            title = detail['title']
            source_url = detail['source_url']
            attachments = detail['attachments']
            
            # 查重
            cur = conn.execute(
                'SELECT id FROM gov_raw WHERE page_url=? AND site_name=?',
                (page_url, site_name)
            )
            row = cur.fetchone()
            
            if row:
                conn.execute('''
                    UPDATE gov_raw SET title=?, content=?, summary=?, publish_date=?,
                    source_url=?, attachments=? WHERE id=?
                ''', (title, content, summary, publish_date, source_url, attachments, row[0]))
                updated += 1
            else:
                conn.execute('''
                    INSERT INTO gov_raw (title, content, summary, site_name,
                    page_url, publish_date, source_url, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                ''', (title, content, summary, site_name,
                      page_url, publish_date, source_url, attachments))
                inserted += 1
            
            time.sleep(0.3)
            
        except Exception as e:
            print('  ERROR: %s' % e)
            errors += 1
    
    conn.commit()
    conn.close()
    print('\n完成！新增%d条，更新%d条，错误%d条' % (inserted, updated, errors))


if __name__ == '__main__':
    import sqlite3
    main()
