#!/usr/bin/env python3
"""
颍上县生态环境分局 - 建设项目环境影响评价审批
https://www.ahys.gov.cn/OpennessContent/showList/654/50806/page_1.html
Same CMS as 颍州区马寨镇 (OpennessTarget AJAX)
"""
import os, sys, re, requests, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from concurrent.futures import ThreadPoolExecutor, as_completed

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "颍上县生态环境分局-建设项目环境影响评价审批"
BASE_URL = "https://www.ahys.gov.cn"
THREADS = 10
THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
TOTAL_PAGES = 32  # pagecount="32"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "X-Requested-With": "XMLHttpRequest",
}

import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

def get_list_url(page):
    return f"{BASE_URL}/OpennessTarget/654/50806/page_{page}.html"

def parse_list(html):
    """解析AJAX列表页"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.select("ul li"):
        a = li.find("a")
        span = li.find("span")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get("title", "") or a.get_text(strip=True)
        date = span.get_text(strip=True)
        if not href or not title or not date:
            continue
        if not re.match(r"\d{4}-\d{2}-\d{2}", date):
            continue
        full_url = BASE_URL + href if href.startswith("/") else href
        items.append({"title": title.strip(), "url": full_url, "date": date})
    return items

def fetch_detail(url):
    """抓取详情页"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=15, verify=False)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        soup = BeautifulSoup(r.text, "html.parser")

        # 标题
        title = ""
        h1 = soup.select_one("div.u-cont h1")
        if h1:
            title = h1.get_text(strip=True)
        if not title:
            meta = soup.find("meta", attrs={"name": "ArticleTitle"})
            if meta and meta.get("content"):
                title = meta["content"].strip()
        if not title:
            h1 = soup.find("h1")
            if h1:
                title = h1.get_text(strip=True)

        # 日期
        publish_date = ""
        meta_d = soup.find("meta", attrs={"name": "PubDate"})
        if meta_d and meta_d.get("content"):
            pub = meta_d["content"].strip()
            publish_date = pub.split(" ")[0] if " " in pub else pub[:10]

        # 正文
        content = ""
        zoom = soup.select_one("div#zoom")
        if zoom:
            content = str(zoom)
        else:
            cont = soup.select_one("div.cont-con") or soup.select_one("div.content")
            if cont:
                content = str(cont)

        if not content or len(content.strip()) < 50:
            return None

        return {"title": title, "date": publish_date, "content": content}
    except Exception as e:
        print(f"  [!] 详情失败: {os.path.basename(url)} - {e}", file=sys.stderr)
        return None

def store_item(item):
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw (title, publish_date, site_name, page_url, content, summary)
            VALUES (?, ?, ?, ?, ?, ?)
        """, (
            item["title"], item["date"], SITE_NAME, item["url"],
            item.get("content", ""), item.get("summary", "")
        ))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected > 0
    except Exception as e:
        print(f"  [!] DB写入失败: {e}", file=sys.stderr)
        return False

def main():
    print(f"[*] 站点: {SITE_NAME}")
    print(f"[*] 时间阈值: {THRESHOLD}")
    print(f"[*] 总页数: {TOTAL_PAGES}")

    # 1. 列表页
    all_items = []
    for page in range(1, TOTAL_PAGES + 1):
        url = get_list_url(page)
        try:
            r = requests.get(url, headers=HEADERS, timeout=15, verify=False)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print(f"  [!] 列表页 {page} 失败: HTTP {r.status_code}")
                continue
            items = parse_list(r.text)
            all_items.extend(items)
            if page % 5 == 0:
                print(f"  [→] 列表页 {page}/{TOTAL_PAGES}: 累计 {len(all_items)} 条")
        except Exception as e:
            print(f"  [!] 列表页 {page} 异常: {e}")

    print(f"\n[*] 共获取 {len(all_items)} 条")

    # 2. 过滤
    items_to_fetch = [it for it in all_items if it["date"] >= THRESHOLD]
    print(f"[*] 3年内({THRESHOLD}~): {len(items_to_fetch)} 条")
    print(f"[*] 3年外: {len(all_items) - len(items_to_fetch)} 条")

    if not items_to_fetch:
        print("[*] 无需新增")
        return

    # 3. 去重
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        existing = set()
        for it in items_to_fetch:
            c = conn.cursor()
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (it["url"], SITE_NAME))
            if c.fetchone():
                existing.add(it["url"])
        conn.close()
    except:
        existing = set()

    to_crawl = [it for it in items_to_fetch if it["url"] not in existing]
    print(f"[*] 需爬详情: {len(to_crawl)} 条 (已有 {len(items_to_fetch)-len(to_crawl)} 条跳过)")

    if not to_crawl:
        print("[*] 无需新增")
        return

    # 4. 并发详情
    new_count = 0
    skip_count = 0

    with ThreadPoolExecutor(max_workers=THREADS) as executor:
        fut_map = {executor.submit(fetch_detail, it["url"]): it for it in to_crawl}
        for fut in as_completed(fut_map):
            it = fut_map[fut]
            try:
                detail = fut.result()
                if detail:
                    it["content"] = detail["content"]
                    it["title"] = detail.get("title", it["title"])
                    it["date"] = detail.get("date", it["date"])
                    if store_item(it):
                        new_count += 1
                        if new_count % 30 == 0:
                            print(f"  [✓] #{new_count} {it['title'][:35]}")
                else:
                    skip_count += 1
            except Exception as e:
                print(f"  [!] 异常: {e}")
                skip_count += 1

    print(f"\n{'='*50}")
    print(f"  新增: {new_count}")
    print(f"  跳过: {skip_count}")
    print(f"  总计: {len(to_crawl)}")
    print(f"{'='*50}")

if __name__ == "__main__":
    main()
