#!/usr/bin/env python3
"""
crawl_ahsthj.py — 安徽省生态环境厅-建设项目环评审批
列表: requests + API (无WAF, JSON)
详情: Playwright (WAF绕过)
"""

import re, time, os, sys
from datetime import datetime, timedelta
import requests
from playwright.sync_api import sync_playwright

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "安徽省生态环境厅-建设项目环评审批"
THREE_YEARS_AGO = datetime.now() - timedelta(days=3 * 365)
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Referer": "https://sthjt.ah.gov.cn/",
}
BROWSER_ARGS = ["--no-sandbox", "--disable-blink-features=AutomationControlled", "--disable-dev-shm-usage"]

API_URL = "https://sthjt.ah.gov.cn/site/label/8888"
BASE_DETAIL = "https://sthjt.ah.gov.cn/public/21691"
import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES


def get_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn

def insert_article(conn, art):
    date_rank = 0
    if art["publish_date"] and len(art["publish_date"]) >= 10:
        try:
            date_rank = int(art["publish_date"][:10].replace("-", ""))
        except:
            pass
    conn.execute(
        """INSERT OR IGNORE INTO gov_raw
           (site_name, page_url, title, publish_date, content, source_url, date_rank)
           VALUES (?, ?, ?, ?, ?, ?, ?)""",
        (SITE_NAME, art["url"], art["title"], art["publish_date"],
         art["content"], art.get("source", SITE_NAME), date_rank)
    )

def get_list_api(page_index):
    """获取列表页JSON"""
    params = {
        "labelName": "publicInfoList", "siteId": 6788031, "organId": 21691,
        "pageSize": 20, "pageIndex": page_index, "isDate": "true",
        "dateFormat": "yyyy-MM-dd", "type": 4, "isJson": "true", "catId": 32709971,
    }
    try:
        resp = requests.get(API_URL, params=params, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            return None, 0, []
        data = resp.json()
        return data, data.get("total", 0), data.get("data", [])
    except Exception as e:
        return None, 0, []

def parse_items(raw_items):
    """解析列表项"""
    articles = []
    for item in raw_items:
        cid = item.get("contentId")
        title = item.get("title", "").strip()
        pub_date = (item.get("publishDate") or "")[:10]
        if not cid or not title:
            continue
        if pub_date:
            try:
                dt = datetime.strptime(pub_date, "%Y-%m-%d")
                if dt < THREE_YEARS_AGO:
                    continue
            except:
                pass
        articles.append({
            "url": f"{BASE_DETAIL}/{cid}.html",
            "title": title,
            "date": pub_date,
            "source": item.get("author", SITE_NAME) or SITE_NAME,
            "content_id": cid,
        })
    return articles

def crawl_detail_playwright(page, url):
    """Playwright爬取详情"""
    try:
        try:
            page.goto(url, wait_until="domcontentloaded", timeout=20000)
            time.sleep(1.5)
        except:
            try:
                page.goto(url, timeout=15000)
                time.sleep(2)
            except:
                return "", ""
    except:
        return "", ""
    
    # 内容 
    content_el = page.query_selector(".j-fontContent.newscontnet.minh500")
    content = ""
    if content_el:
        content = content_el.inner_html().strip()
    
    # 发布日期
    pub_date = ""
    info_el = page.query_selector(".newsinfo")
    if info_el:
        m = re.search(r"(\d{4}-\d{2}-\d{2})", info_el.inner_text())
        if m:
            pub_date = m.group(1)
    
    return content, pub_date

def main():
    print(f"\n{'='*50}")
    print(f"[{SITE_NAME}]")

    all_items = []
    total_pages = 0

    # ========== 阶段1: 列表 ==========
    print("阶段1: 爬取列表 (requests + API)...")
    for pi in range(1, min(200, _MAX_PG or 200)):
        data, total, raw = get_list_api(pi)
        if data is None or not raw:
            if pi == 1:
                print("  [ERR] 列表API失败")
                return
            break
        if pi == 1:
            total_pages = data.get("pageCount", 0)
            print(f"  总记录: {total}, 总页数: {total_pages}")
        
        arts = parse_items(raw)
        if not arts:
            break
        all_items.extend(arts)
        print(f"  page {pi}: {len(arts)} 条 (累计 {len(all_items)})")
        time.sleep(0.3)

    print(f"\n  列表合计: {len(all_items)} 条 (近3年)")
    if not all_items:
        return

    # ========== 阶段2: 详情 (Playwright) ==========
    total_inserted = 0
    conn = get_db()
    print(f"阶段2: 爬取详情 (Playwright, {len(all_items)} 条)...")

    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=BROWSER_ARGS)
        page = browser.new_page()

        for i, item in enumerate(all_items):
            content, pub_date = crawl_detail_playwright(page, item["url"])
            if not content or len(content) < 50:
                print(f"  [{i+1}/{len(all_items)}] ❌ 内容空: {item['title'][:40]}")
                continue

            art = {
                "title": item["title"],
                "url": item["url"],
                "publish_date": pub_date or item["date"],
                "content": content,
                "source": item["source"],
            }
            insert_article(conn, art)
            total_inserted += 1

            if i % 10 == 0:
                conn.commit()
            if i % 5 == 0:
                print(f"  [{i+1}/{len(all_items)}] ✅ {total_inserted} 条 -- {item['title'][:40]}")

            time.sleep(0.5)

        conn.commit()
        browser.close()

    conn.close()
    print(f"\n{'='*50}")
    print(f"✅ {SITE_NAME} 完成!")
    print(f"  列表: {len(all_items)} 条, 入库: {total_inserted} 条")
    print(f"{'='*50}")

if __name__ == "__main__":
    main()
