#!/usr/bin/env python3
"""濉溪县-环评意见征集爬虫 (GET分页版)"""
import json, re, os, time, sys, requests, sqlite3
from datetime import datetime, timedelta

SITE_NAME = "濉溪县-环评意见征集"
BASE_URL = "https://www.sxx.gov.cn"
LIST_URL = "https://www.sxx.gov.cn/zwgk/public/column/1981?type=4&catId=4741481&action=list"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
PAGE_DELAY = 1.0
CUTOFF_DAYS = 9999

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": "https://www.sxx.gov.cn/",
}

# 环评关键词过滤
EIA_KEYWORDS = ["环评", "环境影响评价", "环境", "环保", "公示", "公众参与", "建设项目"]

def fetch_list_page(page):
    url = f"{LIST_URL}&page={page}"
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  第{page}页请求失败: {e}")
        return None

def parse_list(html):
    items = []
    for m in re.finditer(r'<li class="clearfix">(.*?)</li>', html, re.DOTALL):
        li = m.group(1)
        a = re.search(r'href="([^"]+)"[^>]*class="title"[^>]*>(.*?)</a>', li, re.DOTALL)
        span = re.search(r'<span class="date">([^<]+)</span>', li)
        if a:
            title = re.sub(r'<[^>]+>', '', a.group(2)).strip()
            url = a.group(1).strip()
            date = span.group(1).strip() if span else ""
            items.append({"title": title, "url": url, "date": date})
    # Get pageCount
    pc_m = re.search(r'pageCount[^:]*:(\d+)', html)
    page_count = int(pc_m.group(1)) if pc_m else 0
    return items, page_count

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        return {"content": "", "error": str(e)}

    title = ""
    m = re.search(r'ArticleTitle["\' ]*content=["\']([^"\']+)', html)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r'<title>([^<]+)</title>', html)
        if m:
            title = re.sub(r'[-_—]\s*濉溪.*$', '', m.group(1)).strip()

    content = ""
    m = re.search(r'class="wzcon[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    if not content or len(content) < 100:
        m = re.search(r'gkwz_container[^>]*>(.*?)</div>', html, re.DOTALL)
        if m:
            content = m.group(1).strip()

    return {"content": content, "title": title, "error": "" if content else "empty"}

def is_eia(title):
    for kw in EIA_KEYWORDS:
        if kw in title:
            return True
    return False

def main():
    is_incremental = len(sys.argv) >= 2
    cutoff_days = int(sys.argv[1]) if is_incremental else 9999
    print(f"=== {SITE_NAME} === {'增量' if is_incremental else '全量'}")

    now = datetime.now()
    all_items = []

    # 先获取第1页（包含pageCount）
    html = fetch_list_page(1)
    if not html:
        print("无法获取列表页")
        return
    items, total_pages = parse_list(html)
    print(f"总页数: {total_pages}")
    
    # 处理所有页
    for page in range(1, total_pages + 1):
        if page > 1:
            time.sleep(PAGE_DELAY)
            html = fetch_list_page(page)
            if not html:
                continue
            items, _ = parse_list(html)
        
        for item in items:
            if not is_eia(item["title"]):
                continue
            if item["date"]:
                try:
                    dt = datetime.strptime(item["date"], "%Y-%m-%d")
                    days_diff = (now - dt).days
                    if days_diff > cutoff_days:
                        continue
                except:
                    pass
            all_items.append(item)

    print(f"筛选出 {len(all_items)} 条环评相关记录")
    if not all_items:
        print("无新数据")
        return

    # 去重
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    existing = set()
    for item in all_items:
        c.execute("SELECT page_url FROM gov_raw WHERE page_url=?", (item["url"],))
        if c.fetchone():
            existing.add(item["url"])
    conn.close()

    to_fetch = [it for it in all_items if it["url"] not in existing]
    print(f"待抓详情: {len(to_fetch)} 条")

    # 抓详情
    success = []
    for i, item in enumerate(to_fetch):
        time.sleep(0.3)
        detail = fetch_detail(item["url"])
        if detail.get("error") and "empty" not in detail["error"]:
            print(f"  [{i+1}] 失败: {detail['error']}")
            continue
        success.append({
            "title": detail.get("title") or item["title"],
            "url": item["url"],
            "date": item["date"],
            "content": detail.get("content", ""),
            "content_plain": re.sub(r'<[^>]+>', '', detail.get("content", "")).strip(),
        })
        if (i + 1) % 10 == 0:
            print(f"  [{i+1}/{len(to_fetch)}] 已抓取...")

    # 入库
    inserted = 0
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    for item in success:
        try:
            c.execute("""INSERT OR IGNORE INTO gov_raw
                (title, page_url, site_name, summary, content, publish_date)
                VALUES (?, ?, ?, ?, ?, ?)""",
                (item["title"], item["url"], SITE_NAME,
                 item["content_plain"][:500], item["content"], item["date"]))
            if c.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"  入库失败: {e}")
    conn.commit()
    conn.close()
    print(f"入库: {inserted} 条")

    # FTS更新
    if inserted > 0:
        conn = sqlite3.connect(DB_PATH)
        c = conn.cursor()
        # 更新fts_search（standalone表，需INSERT新记录）
        for item in success:
            rid = None
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            row = c.fetchone()
            if row:
                rid = row[0]
                summary = re.sub(r'<[^>]+>', '', item["content"] or "")[:1000]
                c.execute("INSERT OR IGNORE INTO fts_search(rowid, site_name, title, summary) VALUES (?, ?, ?, ?)",
                         (rid, SITE_NAME, item["title"], summary))
        conn.commit()
        conn.close()
        print("FTS更新完成")

    print(f"=== 完成, 新增 {inserted} 条 ===")

if __name__ == "__main__":
    main()
