#!/usr/bin/env python3
"""心连心化学工业集团-新闻中心爬虫

列表: /News/index/cid/63.html (page 1), /News/index/cid/63/p/N.html (pages 2-71)
详情: /News/content/cid/63/id/ID.html
内容: div.edit-content > p
日期: 发布时间：YYYY-MM-DD (span.s1内文本)
附件: PDF链接在div.bont中

Enterprise ThinkPHP CMS, ~566条 (2006~2026)
日跑默认5页
"""

import requests
import re
import sys
import json
import time
import os
import argparse
from bs4 import BeautifulSoup
from urllib.parse import urljoin

sys.path.insert(0, '/root/gov_crawler')
from crawler_lib import push_to_searchdb
import sqlite3

DB_PATH = '/root/search.db'


def get_db_connection():
    try:
        conn = sqlite3.connect(DB_PATH, timeout=30)
        conn.execute("PRAGMA busy_timeout=30000")
        conn.execute("PRAGMA journal_mode=WAL")
        return conn
    except Exception as e:
        print(f'  WARN DB connect: {e}')
        return None

BASE_URL = 'https://www.hnxlx.com.cn'
LIST_URL = '/News/index/cid/63.html'
LIST_PAGE_URL = '/News/index/cid/63/p/{}.html'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

SITE_NAME = '心连心化学-新闻中心'

sess = requests.Session()
sess.headers.update(HEADERS)


def fetch_list(url):
    """抓取列表页 HTML"""
    resp = sess.get(url, timeout=30)
    resp.encoding = 'utf-8'
    return resp.text


def parse_list(html, base_url):
    """解析列表页，返回 items list"""
    soup = BeautifulSoup(html, 'html.parser')
    items = []

    # 列表项：li.wow.fadeInUp
    lis = soup.find_all('li', class_='wow fadeInUp')
    for li in lis:
        a_tag = li.find('a')
        if not a_tag:
            continue

        href = a_tag.get('href', '')
        if not href:
            continue

        # 标题
        h4 = li.find('h4')
        title = h4.get_text(strip=True) if h4 else ''
        if not title:
            continue

        # 完整URL
        if href.startswith('http'):
            detail_url = href
        else:
            detail_url = urljoin(base_url, href)

        # 日期
        em = li.find('em')
        span = li.find('span')
        pub_date = ''
        if em and span:
            day = em.get_text(strip=True)
            year_month = span.get_text(strip=True)
            pub_date = f'{year_month}-{day}'
        if pub_date:
            pub_date = re.sub(r'[年月]', '-', pub_date).replace('日', '')
            pub_date = pub_date.replace('/', '-').replace('.', '-')

        items.append({
            'title': title,
            'url': detail_url,
            'pub_date': pub_date,
        })

    return items


def fetch_detail(item):
    """抓取详情页，提取标题、日期、正文"""
    url = item['url']

    # 外部链接（微信文章、PDF等）跳过详情提取
    if 'hnxlx.com.cn' not in url:
        title = item['title']
        pub_date = item.get('pub_date', '')
        content = f'[外部链接] {url}'
        attachments = []
        return title, pub_date, content, attachments

    # 跳过非HTML资源（PDF、图片等）
    ext_skip = ('.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar',
                '.jpg', '.jpeg', '.png', '.gif', '.bmp', '.mp4', '.mp3')
    if url.lower().endswith(ext_skip):
        title = item['title']
        pub_date = item.get('pub_date', '')
        content = f'[文件下载] {url}'
        attachments = [{'name': title, 'url': url}]
        return title, pub_date, content, attachments

    try:
        resp = sess.get(url, timeout=15)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f'  WARN skip: {url} - {e}')
        return item['title'], item.get('pub_date', ''), '', []

    soup = BeautifulSoup(html, 'html.parser')

    # 标题
    title_div = soup.find('div', class_='news-title')
    title = title_div.get_text(strip=True) if title_div else item['title']

    # 日期
    date_div = soup.find('span', class_='s1')
    pub_date = ''
    if date_div:
        txt = date_div.get_text(strip=True)
        m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
        if m:
            pub_date = m.group(1)

    if not pub_date:
        pub_date = item.get('pub_date', '')

    # 正文
    content_div = soup.find('div', class_='edit-content')
    content = ''
    if content_div:
        parts = []
        for elem in content_div.find_all(['p', 'table', 'img'], recursive=True):
            if elem.name == 'p':
                # 检查 p 内是否还有 p（嵌套结构）
                if elem.find('p'):
                    continue
                # 2026-08-13 修复: 跳过表格内部的 p (否则表格行内容被重复提取, 正文表格重复)
                if elem.find_parent('table'):
                    continue
                txt = ''
                for child in elem.children:
                    if child.name == 'img':
                        src = child.get('src', '')
                        alt = child.get('alt', '')
                        if src:
                            full_src = urljoin(url, src)
                            txt += f'![{alt}]({full_src}) '
                    elif child.name == 'br':
                        txt += '\n'
                    elif child.name == 'a':
                        a_href = child.get('href', '')
                        a_text = child.get_text(strip=True)
                        if a_href and a_text:
                            full_href = urljoin(url, a_href)
                            txt += f'[{a_text}]({full_href}) '
                        elif a_text:
                            txt += a_text + ' '
                    elif isinstance(child, str):
                        txt += child.strip() + ' '
                    elif child.name in ('span', 'strong', 'b', 'em'):
                        txt += child.get_text(' ', strip=True) + ' '
                txt = re.sub(r'[ \t]+', ' ', txt).strip()
                if txt:
                    parts.append(txt)
            elif elem.name == 'table':
                table_html = str(elem)
                parts.append(table_html)
            elif elem.name == 'img':
                src = elem.get('src', '')
                alt = elem.get('alt', '')
                if src:
                    full_src = urljoin(url, src)
                    parts.append(f'![{alt}]({full_src})')
        content = '\n\n'.join(parts)

    # 附件提取
    attachments = []
    bont_divs = soup.find_all('div', class_='bont')
    for bont in bont_divs:
        pdf_links = bont.find_all('a', href=True)
        for a in pdf_links:
            href = a.get('href', '')
            text = a.get_text(strip=True)
            if href and text and not href.startswith('javascript:'):
                if not href.startswith('http'):
                    href = urljoin(url, href)
                attachments.append({'name': text, 'url': href})
                # 同时嵌入到正文末尾
                content += f'\n\n<p><a href="{href}">{text}</a></p>'

    return title, pub_date, content, attachments


def main():
    parser = argparse.ArgumentParser(description='心连心化学-新闻中心爬虫')
    parser.add_argument('--pages', type=int, default=71, help='抓取页数，默认全量(71页)')
    args = parser.parse_args()

    all_items = []
    max_pages = args.pages

    # 第一步：主页
    print(f'Page 1...')
    html = fetch_list(BASE_URL + LIST_URL)
    items = parse_list(html, BASE_URL)
    print(f'  found {len(items)} items')
    all_items.extend(items)

    # 第二步：后续页
    for page in range(2, max_pages + 1):
        url = BASE_URL + LIST_PAGE_URL.format(page)
        print(f'Page {page}...')
        try:
            html = fetch_list(url)
            items = parse_list(html, BASE_URL)
            if not items:
                print(f'  empty page, stop')
                break
            print(f'  found {len(items)} items')
            all_items.extend(items)
        except Exception as e:
            print(f'  WARN: {e}')
            break
        time.sleep(0.5)

    print(f'\nTotal list: {len(all_items)} items')

    # 第三步：抓取详情
    records = []
    for i, item in enumerate(all_items):
        print(f'  [{i+1}/{len(all_items)}] {item["title"][:40]}...')
        title, pub_date, content, attachments = fetch_detail(item)

        if not title:
            title = item['title']

        title = re.sub(r'\s+', ' ', title).strip()

        try:
            summary = content[:500] if content else ''
        except:
            summary = ''

        record = {
            'url': item['url'],
            'source_url': item['url'],
            'title': title,
            'content': content,
            'summary': summary,
            'pub_date': pub_date,
            'site_name': SITE_NAME,
            'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
        }
        records.append(record)
        time.sleep(0.3)

    # 入库
    print(f'\nInserting {len(records)} records...')
    push_to_searchdb(records, SITE_NAME)

    # FTS验证
    conn = get_db_connection()
    raw_count = 0
    fts_count = 0
    if conn:
        try:
            cursor = conn.execute(
                "SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)
            )
            raw_count = cursor.fetchone()[0]
            cursor = conn.execute(
                "SELECT COUNT(*) FROM gov_search WHERE site_name=?", (SITE_NAME,)
            )
            fts_count = cursor.fetchone()[0]
            print(f'\nDB: {raw_count} rows (FTS: {fts_count} rows)')
        except Exception as e:
            print(f'  WARN DB query: {e}')
        finally:
            conn.close()

    print('DONE')


if __name__ == '__main__':
    main()
