#!/usr/bin/env python3
"""
内蒙古自治区生态环境厅 - 项目受理情况 爬虫
https://sthjt.nmg.gov.cn/wrfz2021/hpsp/xmslqk_8091/
TRS-like custom CMS, static pagination index_N.html
"""
import os, sys, re, json, time, requests, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from concurrent.futures import ThreadPoolExecutor, as_completed

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "内蒙古自治区生态环境厅-项目受理情况"
BASE_URL = "https://sthjt.nmg.gov.cn/wrfz2021/hpsp/xmslqk_8091/"
THREADS = 10
THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
TOTAL_PAGES = 16  # countPage = 16

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

def get_list_urls():
    """生成所有列表页URL"""
    urls = []
    for i in range(TOTAL_PAGES):
        if i == 0:
            urls.append(BASE_URL + "index.html")
        else:
            urls.append(BASE_URL + f"index_{i}.html")
    return urls

def parse_list(html, list_url):
    """解析列表页，提取条目信息"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.select("div.n_tab_box ul li"):
        a_tag = li.find("a")
        if not a_tag:
            continue
        href = a_tag.get("href", "")
        title = a_tag.get("title", "") or a_tag.get_text(strip=True)
        if not href or not title:
            continue

        # 日期: <span><span>2026-06-18</span></span>
        date_text = ""
        spans = li.find_all("span", recursive=True)
        for sp in spans:
            txt = sp.get_text(strip=True)
            if re.match(r"\d{4}-\d{2}-\d{2}", txt):
                date_text = txt
                break

        full_url = urljoin(list_url, href.strip())

        items.append({
            "title": title.strip(),
            "date": date_text.strip(),
            "url": full_url,
        })
    return items

def fetch_detail(url):
    """抓取详情页"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        soup = BeautifulSoup(r.text, "html.parser")

        # 标题
        title = ""
        meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta_title and meta_title.get("content"):
            title = meta_title["content"].strip()
        if not title:
            h3 = soup.find("h3")
            if h3:
                title = h3.get_text(strip=True)

        # 日期
        publish_date = ""
        meta_date = soup.find("meta", attrs={"name": "PubDate"})
        if meta_date and meta_date.get("content"):
            pub = meta_date["content"].strip()
            if pub:
                publish_date = pub.split(" ")[0] if " " in pub else pub[:10]

        # 正文
        content_div = soup.select_one("div.trs_editor_view.TRS_UEDITOR")
        content = ""
        if content_div:
            content = str(content_div)
        else:
            zoom = soup.select_one("#zoomfont")
            if zoom:
                content = str(zoom)

        if not content or len(content.strip()) < 50:
            return None

        return {
            "title": title,
            "date": publish_date,
            "content": content,
        }
    except Exception as e:
        print(f"  [!] 详情失败: {os.path.basename(url)} - {e}", file=sys.stderr)
        return None

def store_item(item):
    """写入search.db"""
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw (title, publish_date, site_name, page_url, content, summary)
            VALUES (?, ?, ?, ?, ?, ?)
        """, (
            item["title"], item["date"], SITE_NAME, item["url"],
            item.get("content", ""), item.get("summary", "")
        ))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected > 0
    except Exception as e:
        print(f"  [!] DB写入失败: {e}", file=sys.stderr)
        return False

def main():
    import urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

    print(f"[*] 站点: {SITE_NAME}")
    print(f"[*] 时间阈值: {THRESHOLD}")
    print(f"[*] 总页数: {TOTAL_PAGES}")

    # 1. 抓取所有列表页
    all_items = []
    list_urls = get_list_urls()

    for page_idx, list_url in enumerate(list_urls):
        try:
            r = requests.get(list_url, headers=HEADERS, timeout=15, verify=False)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print(f"  [!] 列表页 {page_idx+1}/{TOTAL_PAGES} 失败: HTTP {r.status_code}")
                continue
            items = parse_list(r.text, list_url)
            print(f"  [✓] 列表页 {page_idx+1}/{TOTAL_PAGES}: {len(items)} 条")
            all_items.extend(items)
        except Exception as e:
            print(f"  [!] 列表页 {page_idx+1}/{TOTAL_PAGES} 异常: {e}")

    print(f"\n[*] 共获取 {len(all_items)} 条")

    # 2. 过滤近3年
    items_to_fetch = [it for it in all_items if it["date"] >= THRESHOLD]
    print(f"[*] 3年内({THRESHOLD}~): {len(items_to_fetch)} 条")
    print(f"[*] 3年外: {len(all_items) - len(items_to_fetch)} 条")

    # 3. 去重
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        existing_urls = set()
        for it in items_to_fetch:
            c = conn.cursor()
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (it["url"], SITE_NAME))
            if c.fetchone():
                existing_urls.add(it["url"])
        conn.close()
    except Exception as e:
        print(f"  [!] 去重查询失败: {e}")
        existing_urls = set()

    to_crawl = [it for it in items_to_fetch if it["url"] not in existing_urls]
    print(f"[*] 需爬详情: {len(to_crawl)} 条 (已有 {len(items_to_fetch)-len(to_crawl)} 条跳过)")

    if not to_crawl:
        print("[*] 无需新增")
        return

    # 4. 并发爬详情
    new_count = 0
    skip_count = 0

    with ThreadPoolExecutor(max_workers=THREADS) as executor:
        fut_map = {executor.submit(fetch_detail, it["url"]): it for it in to_crawl}
        for fut in as_completed(fut_map):
            it = fut_map[fut]
            try:
                detail = fut.result()
                if detail:
                    it["content"] = detail["content"]
                    it["title"] = detail.get("title", it["title"])
                    it["date"] = detail.get("date", it["date"])
                    if store_item(it):
                        new_count += 1
                        print(f"  [✓] #{new_count} {it['title'][:40]}")
                    else:
                        skip_count += 1
                else:
                    skip_count += 1
            except Exception as e:
                print(f"  [!] 处理异常: {e}")
                skip_count += 1

    print(f"\n{'='*50}")
    print(f"  新增: {new_count}")
    print(f"  跳过(已存在/无正文): {skip_count}")
    print(f"  总计: {len(to_crawl)}")
    print(f"{'='*50}")

if __name__ == "__main__":
    main()
