#!/usr/bin/env python3
"""安陆市生态环境局 - 生态环境栏目 (concurrent)"""
import os, sys, re, requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed
import sqlite3

BASE_URL = "http://www.anlu.gov.cn"
LIST_URL = BASE_URL + "/c/alssthjj/sthj.jhtml?LMCL=PWI88i"
LIST_PAGE_TPL = BASE_URL + "/c/alssthjj/sthj_{}.jhtml?LMCL=PWI88i"
SITE_NAME = "安陆市生态环境局-生态环境"
DATE_THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}

import urllib3
urllib3.disable_warnings()

def get_total_pages():
    try:
        r = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        total_m = re.search(r"共(\d+)条", r.text)
        if total_m:
            return (int(total_m.group(1)) + 14) // 15
        pages = re.findall(r"sthj_(\d+)\.jhtml", r.text)
        return max(int(p) for p in pages) if pages else 1
    except Exception as e:
        print(f"[ERROR] get pages: {e}", file=sys.stderr)
        return 1

def fetch_page_items(page):
    url = LIST_URL if page == 1 else LIST_PAGE_TPL.format(page)
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    for li in soup.select("ul.news-list li"):
        a = li.find("a", class_="news-title")
        span = li.find("span", class_="news-time")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get("title", "") or a.get_text(strip=True)
        date = span.get_text(strip=True)
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append({"title": title, "url": href, "date": date})
    return items

def fetch_detail(item):
    """获取详情并存储"""
    try:
        r = requests.get(item["url"], headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        # 正文
        content = ""
        article = soup.select_one("article.htmledit_views")
        if article:
            content = str(article)
        # 来源
        source = ""
        ms = soup.find("meta", attrs={"name": "ContentSource"})
        if ms:
            source = ms.get("content", "")
        # 存储
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw (title, page_url, publish_date, content, site_name)
            VALUES (?, ?, ?, ?, ?)
        """, (item["title"].strip(), item["url"].strip(), item["date"], content, SITE_NAME))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected > 0, item["date"], item["title"][:50]
    except Exception as e:
        return False, item["date"], f"ERROR: {e}"

def main():
    total_pages = get_total_pages()
    print(f"总页数: {total_pages}")

    # 先收集所有在3年内的条目
    all_items = []
    for page in range(1, total_pages + 1):
        print(f"\n--- 读取第{page}页列表 ---")
        try:
            items = fetch_page_items(page)
        except Exception as e:
            print(f"  [ERROR] page {page}: {e}")
            continue

        hit_old = False
        for item in items:
            if item["date"] < DATE_THRESHOLD:
                hit_old = True
                continue
            all_items.append(item)

        if hit_old:
            remaining = total_pages - page
            print(f"  检测到超期数据，剩余{remaining}页跳过")
            break

    print(f"\n共{len(all_items)}条待抓取详情")

    # 并发抓取详情
    new_count = 0
    skip_count = 0
    done = 0

    with ThreadPoolExecutor(max_workers=10) as executor:
        futures = {executor.submit(fetch_detail, item): item for item in all_items}
        for f in as_completed(futures):
            done += 1
            is_new, date, title = f.result()
            if is_new:
                new_count += 1
                status = "✓ 新增"
            else:
                skip_count += 1
                status = "- 已存在"
            if done % 10 == 0 or done == len(all_items):
                print(f"  [{done}/{len(all_items)}] {date} {title[:40]}... {status}")
            else:
                print(f"  {date} {title[:40]}... {status}")

    print(f"\n\n=== 完成 ===")
    print(f"新增: {new_count}, 跳过: {skip_count}")

if __name__ == "__main__":
    main()
