#!/usr/bin/env python3
"""山东金诚石化集团-节能环保信息公示 爬虫 (静态页)
"""
import os, sqlite3, re, time, requests
from datetime import datetime, date
from concurrent.futures import ThreadPoolExecutor, as_completed

BASE = "https://www.sdjcsh.com"
LIST_URL = "https://www.sdjcsh.com/Environmental-acceptance-publicity.html"
CUTOFF_DATE = date(2023, 6, 17)
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "山东金诚石化-节能环保信息公示"
MAX_WORKERS = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}


def get_article_urls():
    """从列表页提取所有文章链接"""
    r = requests.get(LIST_URL, timeout=30, headers=HEADERS)
    r.encoding = "utf-8"
    urls = set()
    # 找所有 /news/ 开头的 .html 链接
    for m in re.finditer(r'href="(/news/[^"]+\.html)"', r.text):
        urls.add(BASE + m.group(1))
    return list(urls)


def extract_content(html):
    """深度计数法提取 enlarge_detailed_info 内部HTML"""
    m = re.search(r'class="[^"]*enlarge_detailed_info[^"]*"[^>]*>', html)
    if not m:
        return ""
    start = m.end()
    depth = 1
    i = start
    while i < len(html) - 5 and depth > 0:
        if html[i:i+4] == '<div' and html[i+4] in (' ', '>', '\n', '\t', '\r'):
            depth += 1
        elif html[i:i+6] == '</div>':
            depth -= 1
        i += 1
    inner = html[start:i-6].strip()
    # 去掉外层的 textLineP div
    inner = re.sub(r'^<div class="textLineP[^"]*"[^>]*>', '', inner)
    inner = re.sub(r'</div>\s*$', '', inner)
    return inner.strip()


def fetch_detail(url):
    for att in range(3):
        try:
            r = requests.get(url, timeout=30, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code != 200:
                time.sleep(2)
                continue
            html = r.text

            # 标题
            title = ""
            m = re.search(r'<title>(.*?)</title>', html)
            if m:
                title = m.group(1)
                for suffix in ['-山东金诚石化集团', '_山东金诚石化集团']:
                    if title.endswith(suffix):
                        title = title[:-len(suffix)].strip()

            # 日期
            pub_date = ""
            m = re.search(r'data-source="publish_time"[^>]*>.*?(\d{4})[/-](\d{2})[/-](\d{2})', html)
            if m:
                pub_date = f"{m.group(1)}-{m.group(2)}-{m.group(3)}"

            # 正文
            content = extract_content(html)
            if not content:
                content = title

            return (url, title, content, pub_date)
        except:
            if att < 2:
                time.sleep(2)
    return (url, None, None, None)


def main():
    mode = os.environ.get("MODE", "full")
    print(f"=== {SITE_NAME} (mode={mode}) ===", flush=True)

    urls = get_article_urls()
    print(f"  文章链接: {len(urls)}条", flush=True)
    for u in urls:
        print(f"    {u}", flush=True)
    if not urls:
        print("  无数据", flush=True)
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    cursor = conn.cursor()
    total_new = 0
    done = 0

    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
        fut_map = {executor.submit(fetch_detail, url): url for url in urls}
        for fut in as_completed(fut_map):
            url, title, content, pub_date = fut.result()
            done += 1
            if title is None:
                continue
            if not content:
                content = title

            # 过滤近3年
            try:
                d = date.fromisoformat(pub_date[:10]) if pub_date else date.today()
                if d < CUTOFF_DATE:
                    print(f"  [跳过] {pub_date} {title[:30]}", flush=True)
                    continue
            except:
                pass

            try:
                cursor.execute("""INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, date_rank) VALUES (?, ?, ?, ?, ?, ?, ?)""",
                    (SITE_NAME, title, url, content, pub_date[:10] if pub_date else "", content, 0))
                if cursor.rowcount > 0:
                    total_new += 1
                    print(f"  + {title[:40]} ({pub_date})", flush=True)
            except Exception as e:
                print(f"  [DB_ERROR] {e}", flush=True)
    conn.commit()

    cursor.execute("SELECT COUNT(*), MIN(publish_date), MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    cnt, min_d, max_d = cursor.fetchone()
    conn.close()
    print(f"\n=== 完成 ===", flush=True)
    print(f"  新增: {total_new}条", flush=True)
    print(f"  累计: {cnt}条 ({min_d} ~ {max_d})", flush=True)


if __name__ == "__main__":
    main()
