#!/usr/bin/env python3
"""
陕煤集团 - 通知公告 (shccig.com)
列表: /Lists/index.html?id=97 (page 1) /index/Lists/index.html?id=97&page=N (pages 2+)
详情: /index/Article/index.html?id=N&cid=97
正文: div.article-con
"""
import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "陕煤集团-通知公告"
BASE_URL = "https://www.shccig.com"
LIST_URL_P1 = "https://www.shccig.com/Lists/index.html?id=97"
LIST_URL_PN = "https://www.shccig.com/index/Lists/index.html?id=97&page=%d"
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except:
        return None

def parse_list(html):
    """从列表页提取 (date, title, href) 三元组"""
    items = []
    # 匹配 <li><span class="fr">日期</span><a href="/index/Article/index.html?id=N&cid=97">标题</a></li>
    pattern = (
        r'<li[^>]*>.*?'
        r'<span[^>]*class="fr"[^>]*>(.*?)</span>.*?'
        r'<a[^>]*href="/index/Article/index.html\?id=(\d+)(?:&amp;|&)cid=97"[^>]*>(.*?)</a>'
        r'.*?</li>'
    )
    for m in re.finditer(pattern, html, re.DOTALL):
        date_str = m.group(1).strip()
        article_id = m.group(2)
        title = re.sub(r'<[^>]+>', '', m.group(3)).strip()
        title = re.sub(r'\s+', ' ', title).strip()
        if not title or len(title) < 5:
            continue
        href = f"/index/Article/index.html?id={article_id}&cid=97"
        full_url = BASE_URL + href
        items.append((date_str, title, full_url))
    return items

def get_page_url(page):
    if page <= 1:
        return LIST_URL_P1
    return LIST_URL_PN % page

def extract_nested_div(html, start_pos):
    """用深度计数提取嵌套 <div> 的内容"""
    depth = 0
    i = start_pos
    n = len(html)
    while i < n:
        # 跳过注释
        if html[i:i+4] == '<!--':
            end = html.find('-->', i+4)
            if end > i:
                i = end + 3
            else:
                i += 1
            continue
        # 闭合 div
        if html[i:i+6] == '</div>':
            if depth == 0:
                return html[start_pos:i]
            depth -= 1
            i += 6
            continue
        # 开放 div
        if html[i:i+4] == '<div' and html[i+4] in (' ', '>', '\n', '\t', '\r'):
            depth += 1
            i += 4
            continue
        i += 1
    return html[start_pos:]

def fetch_detail(url):
    """获取详情页，返回 (title, content, publish_date)"""
    html = fetch(url)
    if not html:
        return None, None, None

    result = {}

    # 标题: 从 <title> 标签提取（去掉站点名后缀）
    m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
    if m:
        title = m.group(1).strip()
        # 去掉尾部 " - 通知公告 陕西煤业化工... "
        title = re.sub(r'[\s\-–—]+\S.*$', '', title).strip()
        result["title"] = title

    # 如果 <title> 提取不理想，用最后一个 h1
    if not result.get("title") or len(result["title"]) < 5:
        h1s = re.findall(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
        if h1s:
            last_h1 = re.sub(r'<[^>]+>', '', h1s[-1]).strip()
            if len(last_h1) > 5:
                result["title"] = last_h1

    # 正文: div.article-con
    con_start = html.find('class="article-con"')
    if con_start > 0:
        # 找到所属 <div> 的开始位置
        div_start = html.rfind('<div', 0, con_start)
        if div_start >= 0:
            content = extract_nested_div(html, div_start)
            # 去除开头的 div 标签属性
            content = re.sub(r'^<div[^>]*>', '', content).strip()
            # 移除 script / style
            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)
            if len(content) > 100:
                result["content"] = content

    # 兜底: 常见正文容器
    if not result.get("content"):
        for cls in ['content', 'article', 'text', 'main', 'con', 'detail']:
            m = re.search(
                r'<(div|section)[^>]*class="' + cls + r'[^"]*"[^>]*>(.*?)</\1>',
                html, re.DOTALL
            )
            if m and len(m.group(2)) > 100:
                content = m.group(2).strip()
                content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
                content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)
                result["content"] = content
                break

    return result.get("title"), result.get("content"), None  # date comes from list page

def main():
    print(f"\n{'='*50}")
    print(f"🏠 {SITE_NAME}")
    print(f"{'='*50}")

    # Step 1: 遍历列表页，收集 (date, title, url)
    all_list = []  # (date, title, url)
    seen_urls = set()
    for page in range(1, MAX_PAGES + 1):
        url = get_page_url(page)
        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch(url)
        if not html:
            print("❌ 请求失败")
            break
        items = parse_list(html)
        # 去重
        new_items = []
        for date, title, item_url in items:
            if item_url not in seen_urls:
                seen_urls.add(item_url)
                new_items.append((date, title, item_url))
        print(f"✅ {len(new_items)} 条（跳过 {len(items) - len(new_items)} 条重复）")
        if not new_items:
            print("  无更多新条目")
            break
        all_list.extend(new_items)

    print(f"\n📊 列表总计: {len(all_list)} 条")

    # Step 2: 逐一获取详情
    all_items = []
    for i, (date_str, list_title, detail_url) in enumerate(all_list):
        print(f"  [{i+1}/{len(all_list)}] {list_title[:50]}...", end=" ", flush=True)
        dt, content, _ = fetch_detail(detail_url)
        time.sleep(0.3)

        title = dt or list_title
        summary = re.sub(r"<[^>]+>", " ", content or "").strip()[:300]
        summary = re.sub(r"\s+", " ", summary)

        all_items.append({
            "site_name": SITE_NAME,
            "title": title,
            "url": detail_url,
            "content": content or "",
            "pub_date": date_str,
            "summary": summary,
            "tags": SITE_NAME,
        })
        print("✅")

    # Step 3: 推送至服务器 search.db
    if all_items:
        push_to_searchdb(all_items, "shccig_tzgg")
    print(f"\n✅ 完成! 共 {len(all_items)} 条")


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time() - t0:.1f}s")
