#!/usr/bin/env python3
"""
临沂市泰尔化工科技有限公司 (www.chinataier.cn) 爬虫
中企动力平台，新闻资讯栏目，单页无分页
"""
import re
import requests
import sqlite3
import os
import sys
import time

SITE_NAME = '临沂泰尔化工'
SITE_DOMAIN = 'www.chinataier.cn'
BASE_URL = 'https://www.chinataier.cn'
LIST_URL = 'https://www.chinataier.cn/news/1626467264910741504.html'

DB_PATH = os.environ.get('DB_PATH', os.path.join(os.path.dirname(os.path.abspath(__file__)), 'search.db'))

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}


def fetch_list():
    """获取列表页所有条目"""
    try:
        resp = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f"  [ERROR] 列表页请求失败: {e}")
        return []

    items = []
    # 每条记录:
    # <P class="e_timeFormat-4 s_title">2022.03.01</P>
    # <p class="e_text-5 s_title"><a href="/news/34.html" target="_self">标题</a></p>
    # <p class="e_text-7 s_title">摘要</p>
    pattern = r'<P\s+class="e_timeFormat-4[^"]*s_title[^"]*"[^>]*>([^<]+)</P>.*?<p\s+class="e_text-5[^"]*s_title[^"]*"[^>]*>.*?<a\s+href="(/news/\d+\.html)"[^>]*>([^<]+)</a>'
    matches = re.findall(pattern, html, re.DOTALL)

    seen_urls = set()
    for date_str, href, title in matches:
        title = title.strip()
        if not title or href in seen_urls:
            continue
        seen_urls.add(href)
        # 格式化日期: 2022.03.01 → 2022-03-01
        date_clean = date_str.strip().replace('.', '-')
        # 只保留日期部分 (YYYY-MM-DD)
        date_clean = date_clean[:10] if len(date_clean) >= 10 else date_clean
        items.append({
            'title': title,
            'url': f'{BASE_URL}{href}',
            'publish_date': date_clean,
        })

    print(f"  列表页: 找到 {len(items)} 条")
    return items


def fetch_detail(url):
    """获取详情页正文"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f"    [ERROR] 详情页请求失败 {url}: {e}")
        return None

    result = {}

    # 提取完整标题: <title>标题_临沂市泰尔化工科技有限公司</title>
    title_match = re.search(r'<title>([^<]+?)(?:_[\u4e00-\u9fff]+)?</title>', html)
    if title_match:
        result['title'] = title_match.group(1).strip()

    # 提取正文: <div class="e_richText-XX">...</div>
    content_match = re.search(r'<div[^>]*class="[^"]*e_richText[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
    if content_match:
        content_html = content_match.group(1).strip()
        # 补全图片src
        content_html = re.sub(r'src="(?!https?://)', f'src="{BASE_URL}/', content_html)
        content_html = re.sub(r'href="(?!https?://)', f'href="{BASE_URL}/', content_html)
        result['content'] = content_html
    else:
        result['content'] = ''

    return result


def get_db():
    """获取数据库连接"""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT,
        content TEXT,
        publish_date TEXT,
        url TEXT UNIQUE,
        site_name TEXT,
        source TEXT,
        attachments TEXT,
        created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
    )''')
    conn.commit()
    return conn


def save_item(conn, item):
    """保存单条数据"""
    c = conn.cursor()
    try:
        c.execute('''
            INSERT OR IGNORE INTO gov_raw (title, content, publish_date, url, site_name, source)
            VALUES (?, ?, ?, ?, ?, ?)
        ''', (
            item['title'],
            item.get('content', ''),
            item.get('publish_date', ''),
            item['url'],
            SITE_NAME,
            f'{SITE_NAME} - 新闻资讯',
        ))
        if c.rowcount > 0:
            print(f"    ✅ 新增: {item['title'][:50]}...")
        else:
            print(f"    ⏭️  已存在: {item['title'][:50]}...")
        conn.commit()
    except Exception as e:
        print(f"    ❌ 写入失败: {e}")


def run(incremental=False):
    """主运行函数"""
    print(f"={'='*50}")
    print(f"  🌐 {SITE_NAME} ({SITE_DOMAIN}) 爬虫")
    print(f"={'='*50}")

    conn = get_db()

    items = fetch_list()
    if not items:
        print("⚠️ 无法获取列表数据，退出")
        conn.close()
        return

    print(f"\n📊 共 {len(items)} 条")
    print(f"\n📄 开始抓取详情页...")

    for idx, item in enumerate(items):
        print(f"  ▶️  [{idx+1}/{len(items)}] {item['title'][:50]}...")
        detail = fetch_detail(item['url'])
        if detail:
            item['title'] = detail.get('title', item['title'])
            item['content'] = detail.get('content', '')
        save_item(conn, item)
        time.sleep(0.5)

    conn.close()

    # 统计
    conn2 = sqlite3.connect(DB_PATH, timeout=60)
    count = conn2.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
    conn2.close()
    print(f"\n={'='*50}")
    print(f"  ✅ 完成! {SITE_NAME} 共导入 {count} 条数据到 {DB_PATH}")
    print(f"={'='*50}")


if __name__ == '__main__':
    run(incremental=len(sys.argv) > 1 and sys.argv[1] in ("1", "--incremental"))