#!/usr/bin/env python3
import os
"""
合肥市生态环境局 - 建设项目环评公示
URL: https://sthjj.hefei.gov.cn/hbyw/hpsp/jsxmhpgs/index.html
365cyd.cn JS 挑战 + 分页API: /content/column/6800451?pageIndex=N
"""
import sys, re, json, subprocess, time, sqlite3
from datetime import datetime, timedelta
from urllib.parse import urljoin
import requests

BASE_URL = "https://sthjj.hefei.gov.cn"
LIST_PATH = "/hbyw/hpsp/jsxmhpgs/index.html"
COLUMN_API = "/content/column/6800451"
SITE_NAME = "合肥市-建设项目环评公示"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

# Solve JS challenge via Node.js helper
JSL_SCRIPT = "/tmp/solve_hefei_jsl.js"

def solve_jsl(url_path: str) -> dict:
    """Run Node.js script to solve JS challenge, returns {jsluid, clearance}"""
    try:
        result = subprocess.run(
            ["node", JSL_SCRIPT],
            capture_output=True, text=True, timeout=30
        )
        out = result.stdout.strip()
        for line in out.split("\n"):
            line = line.strip()
            if line.startswith("{"):
                try:
                    data = json.loads(line)
                    if "jsluid" in data and "clearance" in data:
                        return data
                except:
                    continue
        print(f"  [JSL ERROR] {out[:200]}", flush=True)
        if result.stderr:
            print(f"  [JSL STDERR] {result.stderr[:200]}", flush=True)
    except subprocess.TimeoutExpired:
        print("  [JSL ERROR] Timeout", flush=True)
    except Exception as e:
        print(f"  [JSL ERROR] {e}", flush=True)
    return {}


def fetch_with_jsl(url_path: str, jsluid: str, clearance: str) -> str | None:
    """使用JSL cookies请求页面，自动检测编码"""
    try:
        resp = requests.get(
            f"{BASE_URL}{url_path}",
            headers={
                "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
                "Cookie": f"__jsluid_s={jsluid}; __jsl_clearance_s={clearance}"
            },
            timeout=30
        )
        if resp.status_code != 200:
            return None
        raw = resp.content
        # 1. Try resp.encoding (from Content-Type header)
        try:
            result = raw.decode(resp.encoding or "utf-8", errors="strict")
            if "<title>" in result:
                return result
        except:
            pass
        # 2. Detect charset from HTML meta
        charset = None
        import re as _re
        # Match: charset=utf-8 or charset="utf-8" or charset='utf-8'
        m = _re.search(rb'''<meta[^>]*charset[\s]*=[\s]*["']?([a-zA-Z0-9_-]+)''', raw, _re.I)
        if m:
            charset = m.group(1).decode().lower()
        if not charset:
            m = _re.search(rb'''Content-Type[^>]*charset=([a-zA-Z0-9_-]+)''', raw, _re.I)
            if m:
                charset = m.group(1).decode().lower()
        # 3. Try detected charset, common Chinese encodings, utf-8
        for enc in ([charset] if charset else []) + ["utf-8", "gbk", "gb2312", "gb18030"]:
            if not enc:
                continue
            try:
                result = raw.decode(enc, errors="strict")
                if "<title>" in result:
                    return result
            except:
                continue
        # 4. Last resort
        return raw.decode("utf-8", errors="replace")
    except Exception as e:
        print(f"    [ERROR] 请求失败: {e}", flush=True)
        return None

def parse_list_items(html: str) -> list:
    """解析列表页的文章条目"""
    items = []
    # 查找 doc_list 容器
    m = re.search(r'class="doc_list[^"]*"[^>]*>(.*?)</ul>', html, re.DOTALL)
    if not m:
        return items
    ul_html = m.group(1)
    # 提取每个 li
    lis = re.findall(r'<li[^>]*>(.*?)</li>', ul_html, re.DOTALL)
    for li in lis:
        # 日期
        date_m = re.search(r'(\d{4}-\d{2}-\d{2})', li)
        date = date_m.group(1) if date_m else ""
        # URL
        url_m = re.search(r'href="([^"]+)"', li)
        url = url_m.group(1) if url_m else ""
        # 标题 (从 title 属性)
        title_m = re.search(r'title="([^"]*)"', li)
        title = title_m.group(1).strip() if title_m else ""
        if not title:
            # 从 span 提取
            span_m = re.search(r'<span[^>]*>(.*?)</span>', li)
            if span_m:
                title = re.sub(r'<[^>]+>', '', span_m.group(1)).strip()
        if url and title:
            items.append({"title": title, "url": url, "date": date})
    return items


def extract_detail_content(html: str) -> tuple:
    """提取详情页的正文HTML和标题"""
    title = ""
    title_m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
    if title_m:
        title = title_m.group(1).strip()
        title = re.sub(r'[_-].*?(合肥|生态环境|政府).*$', '', title).strip()

    content = ""
    # 合肥政府网站常用容器 - ID优先
    for cid in ['zoom', 'Zoom', 'article', 'content', 'news-content']:
        m = re.search(rf'id\s*=\s*["\']?{cid}["\']?[^>]*>(.*?)</div>', html, re.DOTALL | re.IGNORECASE)
        if m:
            content = m.group(1).strip()
            break
    if not content:
        for cls in ['TRS_Editor', 'conTxt', 'article-content', 'Custom_UnionStyle',
                    'ewb-article', 'newscontnet', 'minh500']:
            m = re.search(rf'class="[^"]*{cls}[^"]*">(.*?)</div>', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
                break
    if not content:
        for cid in ['Zoom', 'article', 'content', 'news-content']:
            m = re.search(rf'id="[^"]*{cid}[^"]*">(.*?)</div>', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
                break
    # 补全相对链接
    if content:
        content = re.sub(r'href="/', f'href="{BASE_URL}/', content)
        content = re.sub(r'src="/', f'src="{BASE_URL}/', content)
    return title, content


def get_summary(html_content: str) -> str:
    if not html_content:
        return ""
    text = re.sub(r'<[^>]+>', '', html_content)
    text = re.sub(r'\s+', ' ', text).strip()
    return text[:200]


def crawl():
    print(f"[INFO] 合肥市环评审批爬虫启动", flush=True)
    print(f"[INFO] 截止日期: {CUTOFF_DATE}", flush=True)

    # Step 1: 解决JS挑战获取cookies
    print(f"  [JSL] 正在破解JS挑战...", flush=True)
    cookies = solve_jsl(LIST_PATH)
    if not cookies:
        print("[ERROR] JS挑战破解失败", flush=True)
        return
    print(f"  [JSL] 通过! jsluid={cookies['jsluid'][:8]}...", flush=True)

    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()

    total_new = 0
    total_old = 0
    total_skip = 0
    page = 0
    max_pages = 50

    while page < max_pages:
        page += 1
        # 第一页和后续页使用不同的URL
        if page == 1:
            url_path = LIST_PATH
        else:
            url_path = f"{COLUMN_API}?pageIndex={page}"

        print(f"  [PAGE {page}] {url_path}", flush=True)
        html = fetch_with_jsl(url_path, cookies["jsluid"], cookies["clearance"])
        if not html:
            print(f"    [ERROR] 页面获取失败", flush=True)
            break

        items = parse_list_items(html)
        if not items:
            print(f"    [WARN] 未找到文章列表，结束", flush=True)
            break

        print(f"    {len(items)} 条", flush=True)
        for item in items:
            date = item["date"]
            if date and date < CUTOFF_DATE:
                total_skip += 1
                continue

            page_url = item["url"]
            title = item["title"]

            # 获取详情
            detail_html = fetch_with_jsl(page_url.replace(BASE_URL, ""),
                                         cookies["jsluid"], cookies["clearance"])
            detail_title = title
            content_html = ""
            if detail_html:
                dt, content_html = extract_detail_content(detail_html)
                if dt:
                    detail_title = dt
            else:
                print(f"    [WARN] 详情获取失败: {page_url[-40:]}", flush=True)
            summary = get_summary(content_html) or detail_title

            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw (title, page_url, site_name, summary, content, publish_date) "
                    "VALUES (?, ?, ?, ?, ?, ?)",
                    (detail_title, page_url, SITE_NAME, summary, content_html, date)
                )
                if cur.rowcount > 0:
                    total_new += 1
                else:
                    total_old += 1
            except Exception as e:
                print(f"    [ERROR] 入库失败: {detail_title[:30]} - {e}", flush=True)

            time.sleep(0.3)  # 礼貌间隔

        # 检查是否有下一页
        if page == 1:
            # 获取总页数
            total_m = re.search(r'pageCount:\s*(\d+)', html)
            if total_m:
                max_pages = int(total_m.group(1))
                print(f"    总页数: {max_pages}", flush=True)
            else:
                # 如果没有更多，退出
                if not html:
                    break
                elif not items:
                    break

        time.sleep(1)

    conn.commit()
    conn.close()
    print(f"\n[DONE] 合肥市爬取完成", flush=True)
    print(f"  爬取页数: {page}", flush=True)
    print(f"  新增: {total_new} 条", flush=True)
    print(f"  已存在: {total_old} 条", flush=True)
    print(f"  跳过(日期超限): {total_skip} 条", flush=True)


def incremental():
    """增量: 仅爬第一页"""
    print(f"[增量] 合肥市增量爬取", flush=True)
    cookies = solve_jsl(LIST_PATH)
    if not cookies:
        print("[ERROR] JS破解失败", flush=True)
        return

    html = fetch_with_jsl(LIST_PATH, cookies["jsluid"], cookies["clearance"])
    if not html:
        print("[ERROR] 页面获取失败", flush=True)
        return

    items = parse_list_items(html)
    print(f"  最新 {len(items)} 条", flush=True)

    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()
    new_count = 0
    for item in items:
        page_url = item["url"]
        detail_html = fetch_with_jsl(page_url.replace(BASE_URL, ""),
                                     cookies["jsluid"], cookies["clearance"])
        content_html = ""
        if detail_html:
            _, content_html = extract_detail_content(detail_html)
        summary = get_summary(content_html) or item["title"]
        try:
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (title, page_url, site_name, summary, content, publish_date) "
                "VALUES (?, ?, ?, ?, ?, ?)",
                (item["title"], page_url, SITE_NAME, summary, content_html, item["date"])
            )
            if cur.rowcount > 0:
                new_count += 1
        except Exception as e:
            print(f"    [ERROR] {item['title'][:20]} - {e}", flush=True)
    conn.commit()
    conn.close()
    print(f"  新增入库: {new_count} 条", flush=True)


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "full"
    if mode == "incremental":
        incremental()
    else:
        crawl()
