#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
霍邱县人民政府 - 行政审批信息公开 爬虫
CMS: 六安政府网站群 (luan_govc)，信息公开目录系统
列表API: GET /site/label/8888 (返回HTML片段)
        params: labelName=publicInfoList, siteId=6786971, pageSize=15,
                pageIndex=N, organId=6600621, type=4, catId=7365565,
                action=list, file=/c3/huoqiu/publicInfoList_newest
详情页: /public/6600621/{id}.html
       标题: meta[ArticleTitle]
       日期: meta[PubDate]
       正文: div.gkwz_contnet.j-fontContent (段落/表格/附件链接)
日跑: --max-pages 1 (增量取最新页)
"""

import sys, os, re, time, sqlite3, argparse, json, subprocess
from datetime import datetime
from urllib.parse import urljoin, urlencode
import requests
from bs4 import BeautifulSoup, NavigableString
import warnings
warnings.filterwarnings("ignore")

BASE_URL = "https://www.huoqiu.gov.cn"
SITE_NAME = "霍邱县人民政府"
COLUMN_NAME = "行政审批"
GROUP = "安徽省"
DB_PATH = "/mnt/data/search.db"

API_URL = "https://www.huoqiu.gov.cn/site/label/8888"
API_PARAMS = {
    "labelName": "publicInfoList",
    "siteId": "6786971",
    "pageSize": "15",
    "action": "list",
    "isDate": "true",
    "dateFormat": "yyyy-MM-dd",
    "length": "50",
    "organId": "6600621",
    "type": "4",
    "catId": "7365565",
    "cId": "",
    "result": "暂无相关信息",
    "file": "/c3/huoqiu/publicInfoList_newest",
}
PAGE_SIZE = 15
TOTAL_PAGES = 32  # 32 pages × 15 = ~480 articles

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def log(msg):
    print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}")


def get_page(url, params=None, session=None, retries=3):
    """获取页面内容"""
    for i in range(retries):
        try:
            s = session or requests.Session()
            headers = dict(HEADERS)
            headers["Referer"] = f"{BASE_URL}/public/column/6600621?type=4&catId=7365565&action=list&nav=3"
            resp = s.get(url, params=params, headers=headers, timeout=30)
            if resp.status_code == 200:
                resp.encoding = 'utf-8'
                return resp.text
            log(f"  HTTP {resp.status_code}，重试 ({i+1}/{retries})")
            time.sleep(2)
        except Exception as e:
            log(f"  请求异常: {e}，重试 ({i+1}/{retries})")
            time.sleep(3)
    return None


def get_list_page(page_num, session=None):
    """获取列表页"""
    params = dict(API_PARAMS)
    params["pageIndex"] = str(page_num)
    url = f"{API_URL}?{urlencode(params)}"
    return get_page(url, session=session)


def parse_list(html):
    """解析列表返回 [{url, title, date}]"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='xxgk_nav_list')
    if not ul:
        log("  ⚠️ 未找到 xxgk_nav_list")
        return items

    for li in ul.find_all('li', recursive=False):
        a = li.find('a', class_='title')
        if not a or not a.get('href'):
            continue

        href = a['href'].strip()
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)

        # 标题在a标签的直接文本中（去除img等子标签的干扰）
        title = ''
        for child in a.children:
            if child.name is None:  # text node
                t = str(child).strip()
                if t:
                    title += t
        title = re.sub(r'\s+', ' ', title).strip()
        if not title:
            title = a.get_text(strip=True)

        date_span = li.find('span', class_='date')
        pub_date = date_span.get_text(strip=True) if date_span else ''

        if title and href:
            items.append({'url': href, 'title': title, 'pub_date': pub_date})

    return items


def fetch_detail(url, session=None):
    """获取详情页"""
    html = get_page(url, session=session)
    if not html:
        return None

    soup = BeautifulSoup(html, 'html.parser')

    # 标题
    title = ''
    mt = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if mt and mt.get('content'):
        title = mt['content'].strip()
    if not title:
        h1 = soup.find('h1')
        if h1:
            title = h1.get_text(strip=True)

    # 日期
    pub_date = ''
    md = soup.find('meta', attrs={'name': 'PubDate'})
    if md and md.get('content'):
        pub_date = md['content'].strip()

    # 来源
    source = ''
    ms = soup.find('meta', attrs={'name': 'ContentSource'})
    if ms and ms.get('content'):
        source = ms['content'].strip()

    # 正文
    content_div = soup.select_one('div.gkwz_contnet.j-fontContent')

    attachments = []
    content_text = ''
    has_table = False

    if content_div:
        # 提取附件链接
        for a in content_div.find_all('a'):
            href = a.get('href', '')
            text = a.get_text(strip=True) or '附件'
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|7z)(\?|#|$)', href, re.I):
                if not href.startswith('http'):
                    href = urljoin(BASE_URL, href)
                attachments.append({'url': href, 'text': text})

        has_table = bool(content_div.find('table'))

        # 生成文本
        parts = []

        def extract_paragraphs(element, is_root=False):
            """递归提取段落，<br>视为换行分隔"""
            if isinstance(element, NavigableString):
                txt = str(element).strip()
                return [txt] if txt else []
            
            paragraphs = []
            current = []
            for child in element.children:
                if isinstance(child, NavigableString):
                    txt = str(child).strip()
                    if txt:
                        for line in txt.split('\n'):
                            line = line.strip()
                            if line:
                                current.append(line)
                elif child.name == 'br':
                    if current:
                        paragraphs.append(' '.join(current))
                        current = []
                elif child.name == 'table':
                    if current:
                        paragraphs.append(' '.join(current))
                        current = []
                    rows = []
                    for tr in child.find_all('tr'):
                        cells = [td.get_text(strip=True) for td in tr.find_all(['td', 'th'])]
                        rows.append(' | '.join(cells))
                    if rows:
                        paragraphs.append('\n'.join(rows))
                elif child.name in ('img',):
                    src = child.get('src', '') or child.get('data-src', '')
                    if src:
                        if not src.startswith('http'):
                            src = urljoin(BASE_URL, src)
                        current.append(f'[图片]({src})')
                elif child.name in ('div', 'p'):
                    sub_paras = extract_paragraphs(child)
                    if current:
                        paragraphs.append(' '.join(current))
                        current = []
                    paragraphs.extend(sub_paras)
                else:
                    # 内联元素（span, font, a, strong等）— 递归进去处理br
                    sub_paras = extract_paragraphs(child)
                    if len(sub_paras) > 1:
                        # 子元素返回了多个段落(由br分隔)，把当前积累的先flush
                        if current:
                            paragraphs.append(' '.join(current))
                            current = []
                        paragraphs.extend(sub_paras)
                    elif sub_paras:
                        # 单段落文本，追加到当前段落积累
                        current.append(sub_paras[0])

            if current:
                paragraphs.append(' '.join(current))
            return paragraphs

        for elem in content_div.children:
            name = getattr(elem, 'name', None)
            if name in ('script', 'style'):
                continue
            if name == 'div' and 'clear' in (elem.get('class') or []):
                continue  # 跳过clear div
            if name == 'div' and 'bqxx' in (elem.get('class') or []):
                continue  # 跳过标签div
            
            paras = extract_paragraphs(elem)
            parts.extend(paras)

        content_text = '\n\n'.join(p for p in parts if p)
        content_text = re.sub(r'\n{3,}', '\n\n', content_text).strip()

    # 纯附件页
    if not content_text and attachments:
        links = '\n'.join(f"[{a['text']}]({a['url']})" for a in attachments)
        content_text = f"附件下载：\n{links}"

    return {
        'title': title,
        'pub_date': pub_date,
        'source': source,
        'content_text': content_text,
        'has_table': has_table,
        'attachments': attachments,
    }


def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT NOT NULL,
        page_url TEXT NOT NULL UNIQUE,
        source_url TEXT,
        publish_date TEXT,
        content TEXT,
        site_name TEXT,
        group_name TEXT,
        industry TEXT DEFAULT 'other',
        attachments TEXT DEFAULT '',
        has_table INTEGER DEFAULT 0,
        summary TEXT DEFAULT ''
    )''')
    c.execute('''CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(
        title, site_name, summary, tokenize=trigram
    )''')
    conn.commit()
    return conn, c


def insert_article(conn, cur, data):
    """INSERT OR IGNORE 入库"""
    try:
        content = data.get('content', '') or ''
        summary = content[:200] if content else ''
        attachments_str = json.dumps(data.get('attachments', []), ensure_ascii=False)
        has_table = 1 if data.get('has_table') else 0

        cur.execute('''INSERT OR IGNORE INTO gov_raw
            (page_url, title, publish_date, content, site_name, source_url,
             group_name, industry, attachments, has_table, summary)
            VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''', (
            data['url'],
            data.get('title', ''),
            data.get('pub_date', ''),
            content,
            data.get('site_name', SITE_NAME),
            data['url'],
            data.get('group_name', GROUP),
            'other',
            attachments_str,
            has_table,
            summary,
        ))
        if cur.rowcount == 0:
            return False

        rowid = cur.lastrowid
        try:
            # 2026-09-22: 先提交 gov_raw —— 下面手动写 FTS 会因触发器已写过同一
            #   rowid 而 IntegrityError，若不先 commit，这条记录会被一并回滚（静默丢数据）
            conn.commit()
            cur.execute('INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)',
                        (rowid, data.get('title', ''), data.get('site_name', SITE_NAME), summary))
        except sqlite3.OperationalError:
            conn.commit()
            _fts_fallback(rowid, data.get('title', ''), data.get('site_name', SITE_NAME), summary)
        conn.commit()
        return True
    except Exception as e:
        log(f"  入库异常: {e}")
        return False


def _fts_fallback(rowid, title, site_name, summary):
    try:
        subprocess.run([
            'sqlite3', "-cmd", ".timeout 60000", DB_PATH,
            f"INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES ({rowid}, '{title.replace(chr(39), chr(39)+chr(39))}', '{site_name.replace(chr(39), chr(39)+chr(39))}', '{summary.replace(chr(39), chr(39)+chr(39))}')"
        ], capture_output=True, timeout=10)
    except Exception as e:
        log(f"  FTS fallback失败: {e}")


def crawl(max_pages=5, local=False):
    """主爬虫"""
    global DB_PATH
    if local:
        DB_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'huoqiu_test.db')

    log(f"开始爬取 {SITE_NAME} - {COLUMN_NAME}，{max_pages}页")
    if local:
        log(f"本地测试模式，DB: {DB_PATH}")

    conn, cur = init_db()
    session = requests.Session()
    session.headers.update(HEADERS)

    # 第1阶段：收集所有列表页
    all_articles = []
    for page in range(1, max_pages + 1):
        log(f"列表页 {page}")
        html = get_list_page(page, session)
        if not html:
            log(f"  ❌ 第{page}页获取失败，停止")
            break
        items = parse_list(html)
        if not items:
            log(f"  ⚠️ 第{page}页无文章，停止")
            break
        log(f"  ✅ {len(items)} 条")
        all_articles.extend(items)
        time.sleep(1)

    log(f"\n总计 {len(all_articles)} 条待爬详情")

    # 第2阶段：详情入库
    new_count = 0
    skip_count = 0
    for idx, art in enumerate(all_articles, 1):
        log(f"[{idx}/{len(all_articles)}] {art['title'][:50]}...")

        detail = fetch_detail(art['url'], session)
        if not detail:
            log(f"  ❌ 详情获取失败")
            skip_count += 1
            time.sleep(0.8)
            continue

        # 列表标题完整，优先使用
        final_title = art['title'] or detail['title']
        final_date = art['pub_date'] or detail['pub_date']

        ok = insert_article(conn, cur, {
            'title': final_title,
            'url': art['url'],
            'pub_date': final_date,
            'content': detail['content_text'],
            'site_name': SITE_NAME,
            'group_name': GROUP,
            'attachments': detail.get('attachments', []),
            'has_table': detail.get('has_table', False),
        })
        if ok:
            new_count += 1
            log(f"  ✅ 入库")
        else:
            skip_count += 1
            log(f"  ⏭️ 已存在")
        time.sleep(0.5)

    conn.close()
    log(f"\n{'='*50}")
    log(f"完成! 新增: {new_count}, 跳过: {skip_count}")
    return new_count


if __name__ == '__main__':
    parser = argparse.ArgumentParser(description='霍邱县行政审批信息公开爬虫')
    parser.add_argument('--max-pages', type=int, default=5, help='最大爬取页数')
    parser.add_argument('--local', action='store_true', help='本地测试模式')
    args = parser.parse_args()
    crawl(max_pages=args.max_pages, local=args.local)
