#!/usr/bin/env python3
"""中汇发化学集团 新闻动态 爬虫
List: news.html (内嵌文章列表，WebsiteOnline平台)
Detail: page195.html?article_id=XX
"""
import re, requests
from bs4 import BeautifulSoup
from datetime import datetime
from urllib.parse import urljoin
import sqlite3, os

# 使用英文域名避免IDN编码问题
BASE_URL = "http://www.zhonghuifa.com"
SITE_NAME = "zhonghuifa.com-新闻动态"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CATEGORY = "eia"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
CUTOFF_DATE = "2023-06-16"


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.row_factory = sqlite3.Row
    return conn


def clean_text(text):
    if not text:
        return ""
    return re.sub(r'\s+', ' ', text.strip())


def extract_detail(session, article_id):
    """获取详情页"""
    url = f"{BASE_URL}/page195.html?article_id={article_id}"
    try:
        resp = session.get(url, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [WARN] 获取详情页失败 {url}: {e}", flush=True)
        return None, None, None, None, None

    html = resp.text
    soup = BeautifulSoup(html, 'html.parser')

    # 标题
    title_el = soup.select_one('div.artdetail_title')
    title = title_el.get_text(strip=True) if title_el else ''
    if not title:
        return None, None, None, None, None

    # 日期 - from .artview_info text "发布时间:2026-06-12"
    info_el = soup.select_one('.artview_info')
    pub_date = ''
    if info_el:
        info_text = info_el.get_text(strip=True)
        date_m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', info_text)
        if date_m:
            pub_date = date_m.group(1)
    if not pub_date:
        date_m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
        pub_date = date_m.group(1) if date_m else ''
    if not pub_date:
        return None, None, None, None, None

    # 来源/作者
    source = ''
    source_m = re.search(r'来源[：:]\s*([^\s|]+)', html)
    if source_m:
        source = source_m.group(1).strip()

    # 正文
    content = ''
    content_el = soup.select_one('div.artview_content div.artview_detail')
    if content_el:
        content = str(content_el)

    summary = ''
    if content:
        text_obj = BeautifulSoup(content, 'html.parser')
        text = text_obj.get_text(separator=' ', strip=True)
        summary = clean_text(text[:200])

    return title, pub_date, content, summary, url


def main():
    print(f"=== {SITE_NAME} 爬虫 ===", flush=True)
    print(f"数据库: {DB_PATH}", flush=True)

    conn = get_conn()
    cur = conn.cursor()
    cutoff_dt = datetime.strptime(CUTOFF_DATE, "%Y-%m-%d").date()

    session = requests.Session()
    session.get(f"{BASE_URL}/news.html", headers=HEADERS, timeout=15)

    # 分页获取所有文章 (共8页,每页8条,64篇)
    articles = []
    for page in range(1, 9):
        resp = session.get(f"{BASE_URL}/news.html?page={page}", headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, 'html.parser')

        for li in soup.select('li.wpart-border-line'):
            a_tag = li.select_one('a.articleid')
            time_el = li.select_one('span.wp-new-ar-pro-time')
            if a_tag and time_el:
                article_id = a_tag.get('articleid', '')
                title = a_tag.get('title', '') or a_tag.get_text(strip=True)
                pub_date = time_el.get_text(strip=True)
                # 去重
                if not any(a['id'] == article_id for a in articles):
                    articles.append({
                        'id': article_id,
                        'title': clean_text(title),
                        'pub_date': pub_date,
                    })

    print(f"列表页共 {len(articles)} 篇文章 ({len(articles)//8}页)", flush=True)

    total_new = 0
    total_skip = 0
    total_error = 0

    for art in articles:
        article_id = art['id']
        url = f"{BASE_URL}/page195.html?article_id={article_id}"

        # 查重
        cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
        if cur.fetchone():
            total_skip += 1
            continue

        # 日期过滤
        pub_date = art['pub_date']
        pub_dt = datetime.strptime(pub_date, "%Y-%m-%d").date()
        if pub_dt < cutoff_dt:
            total_skip += 1
            continue

        # 获取详情
        print(f"  [{total_new+total_skip+total_error+1}/{len(articles)}] article_id={article_id}", flush=True)
        title, detail_date, content, summary, detail_url = extract_detail(session, article_id)

        if not title:
            print(f"  [SKIP] 无详情", flush=True)
            total_error += 1
            continue

        date_rank = int(pub_date.replace('-', ''))
        source_url = f"{BASE_URL}/news.html"

        try:
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category, date_rank, source_url) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
                (detail_url, title, content, SITE_NAME, pub_date, summary, CATEGORY, date_rank, source_url)
            )
            conn.commit()
            total_new += 1
            print(f"  [OK] {title[:40]}... | {pub_date}", flush=True)
        except Exception as e:
            print(f"  [ERROR] 入库失败: {title[:30]}... - {e}", flush=True)
            conn.rollback()
            total_error += 1

    conn.close()
    print(f"\n=== 完成: 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error} ===", flush=True)


if __name__ == '__main__':
    main()
