#!/usr/bin/env python3
"""
安徽昊源化工集团 (www.chinahaoyuan.com) 爬虫
环境保护栏目，19页共约190条环评/环境信息公示
"""
import re
import requests
import sqlite3
import os
import sys
import time

SITE_NAME = '安徽昊源化工'
SITE_DOMAIN = 'www.chinahaoyuan.com'
BASE_URL = 'https://www.chinahaoyuan.com'
LIST_URL = 'https://www.chinahaoyuan.com/lingdaoweiwen.html'

DB_PATH = os.environ.get('DB_PATH', os.path.join(os.path.dirname(os.path.abspath(__file__)), 'search.db'))

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}


def fetch_list_page(page=1):
    """获取分页列表"""
    url = f'{BASE_URL}/lingdaoweiwen/{page}.html' if page > 1 else LIST_URL
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f"  [ERROR] 列表页第{page}页请求失败: {e}")
        return [], 0

    items = []
    # 每条记录: <li><a href="/lingdaoweiwen/detail/N.html" title="标题"><span>日期</span><h3>标题</h3><p></p></a></li>
    pattern = r'<li><a href="(/lingdaoweiwen/detail/\d+\.html)"[^>]*title="([^"]*)"[^>]*><span>([^<]+)</span><h3>([^<]+)</h3>'
    matches = re.findall(pattern, html)

    for href, title_attr, date_str, h3_title in matches:
        title = h3_title.strip() or title_attr.strip()
        date_str = date_str.strip()
        if not title:
            continue
        items.append({
            'title': title,
            'url': f'{BASE_URL}{href}',
            'publish_date': date_str[:10],
        })

    # 解析总页数
    total_pages = 1
    # 从尾页链接获取
    last_page = re.search(r'/lingdaoweiwen/(\d+)\.html[^>]*>[尾|末]', html)
    if last_page:
        total_pages = int(last_page.group(1))

    print(f"  第{page}页: 找到 {len(items)} 条 (共 {total_pages} 页)")
    return items, total_pages


def fetch_detail(url):
    """获取详情页完整标题和正文"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f"    [ERROR] 详情页请求失败 {url}: {e}")
        return None

    result = {}

    # 提取完整标题: <title>标题</title>
    title_match = re.search(r'<title>([^<]+)</title>', html)
    if title_match:
        result['title'] = title_match.group(1).strip()

    # 提取发布日期: 发布时间：2026-05-07
    date_match = re.search(r'发布时间[：:]?\s*([\d-]+)', html)
    if date_match:
        result['publish_date'] = date_match.group(1).strip()

    # 提取正文: <div class="n_content_c">...</div>
    content_match = re.search(r'class="n_content_c"[^>]*>(.*?)</div>', html, re.DOTALL)
    if content_match:
        content_html = content_match.group(1).strip()
        # 补全文件链接
        content_html = re.sub(r'href="(?!https?://)', f'href="{BASE_URL}', content_html)
        content_html = re.sub(r'src="(?!https?://)', f'src="{BASE_URL}', content_html)
        result['content'] = content_html
    else:
        result['content'] = ''

    return result


def get_db():
    """获取数据库连接"""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT,
        content TEXT,
        publish_date TEXT,
        url TEXT UNIQUE,
        site_name TEXT,
        source TEXT,
        attachments TEXT,
        created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
    )''')
    conn.commit()
    return conn


def save_item(conn, item):
    """保存单条数据"""
    c = conn.cursor()
    try:
        c.execute('''
            INSERT OR IGNORE INTO gov_raw (title, content, publish_date, url, site_name, source)
            VALUES (?, ?, ?, ?, ?, ?)
        ''', (
            item['title'],
            item.get('content', ''),
            item.get('publish_date', ''),
            item['url'],
            SITE_NAME,
            f'{SITE_NAME} - 环境保护',
        ))
        if c.rowcount > 0:
            print(f"    ✅ 新增: {item['title'][:50]}...")
        else:
            print(f"    ⏭️  已存在: {item['title'][:50]}...")
        conn.commit()
    except Exception as e:
        print(f"    ❌ 写入失败: {e}")


def run(max_pages=None):
    """主运行函数"""
    print(f"={'='*50}")
    print(f"  🌐 {SITE_NAME} ({SITE_DOMAIN}) 爬虫")
    print(f"={'='*50}")

    conn = get_db()

    # 先获取第一页，确认总页数
    items, total_pages = fetch_list_page(1)
    if not items:
        print("⚠️ 无法获取列表数据，退出")
        conn.close()
        return

    print(f"\n📊 共 {total_pages} 页")

    if max_pages and max_pages < total_pages:
        print(f"   ⚠️  限制只爬 {max_pages} 页")
        total_pages = max_pages

    # 处理每一页
    for page in range(1, total_pages + 1):
        if page > 1:
            items, _ = fetch_list_page(page)
            if not items:
                break

        print(f"\n📄 第{page}页 ({len(items)} 条)")
        for idx, item in enumerate(items):
            print(f"  ▶️  [{idx+1}/{len(items)}] {item['title'][:50]}...")
            detail = fetch_detail(item['url'])
            if detail:
                if detail.get('title'):
                    item['title'] = detail['title']
                if detail.get('publish_date'):
                    item['publish_date'] = detail['publish_date']
                item['content'] = detail.get('content', '')
            save_item(conn, item)
            time.sleep(0.5)

    conn.close()

    # 统计
    conn2 = sqlite3.connect(DB_PATH, timeout=60)
    count = conn2.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
    conn2.close()
    print(f"\n={'='*50}")
    print(f"  ✅ 完成! {SITE_NAME} 共导入 {count} 条数据到 {DB_PATH}")
    print(f"={'='*50}")


if __name__ == '__main__':
    max_pages = None
    if len(sys.argv) > 1:
        max_pages = int(sys.argv[1])
    run(max_pages)
