#!/usr/bin/env python3
"""第四师·可克达拉市 - 环保公示"""
import os, sys, re, requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

BASE_URL = "http://www.cocodala.gov.cn"
LIST_URL = BASE_URL + "/html/1647/"
LIST_PAGE_TPL = BASE_URL + "/html/1647/list-{}.html"
SITE_NAME = "第四师·可克达拉市-环保公示"
DATE_THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

import urllib3
urllib3.disable_warnings()

def get_total_pages():
    """获取总页数 - 从第一页的翻页链接提取"""
    try:
        r = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        # 找最大页码
        pages = re.findall(r'list-(\d+)\.html', r.text)
        if pages:
            return max(int(p) for p in pages)
        return 1
    except Exception as e:
        print(f"[ERROR] get pages: {e}", file=sys.stderr)
        return 1

def parse_list(html):
    """解析列表页"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for li in soup.select("ul.xxgk-list li"):
        a = li.find("a")
        span = li.find("span", class_="datatime")
        if not a or not span:
            continue
        href = a.get("href", "")
        title = a.get("title", "") or a.get_text(strip=True)
        date = span.get_text(strip=True)
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append({"title": title, "url": href, "date": date})
    return items

def fetch_detail(url):
    """获取详情"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        # 正文
        content = ""
        zoom = soup.select_one("div.article-content#zoom")
        if zoom:
            content = str(zoom)
        # 来源
        source = ""
        ms = soup.find("meta", attrs={"name": "ContentSource"})
        if ms:
            source = ms.get("content", "")
        return content, source
    except Exception as e:
        print(f"  [ERROR] detail {url}: {e}", file=sys.stderr)
        return "", ""

def store_item(title, url, date, content, source):
    """写入search.db"""
    import sqlite3
    try:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw (title, page_url, publish_date, content, site_name)
            VALUES (?, ?, ?, ?, ?)
        """, (title.strip(), url.strip(), date, content, SITE_NAME))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected > 0
    except Exception as e:
        print(f"  [ERROR] store: {e}", file=sys.stderr)
        return False

def main():
    # 页数限制: 支持裸数字 / --pages=N / --pages N (默认0=全量)
    max_pages = 0
    for i, a in enumerate(sys.argv):
        if a.isdigit():
            max_pages = int(a)
        elif a.startswith("--pages="):
            try:
                max_pages = int(a.split("=")[1])
            except ValueError:
                pass
        elif a == "--pages" and i + 1 < len(sys.argv) and sys.argv[i+1].isdigit():
            max_pages = int(sys.argv[i+1])

    total_pages = get_total_pages()
    if max_pages > 0:
        total_pages = min(total_pages, max_pages)
    print(f"总页数: {total_pages} (limit={max_pages})")

    new_count = 0
    skip_count = 0

    for page in range(1, total_pages + 1):
        url = LIST_URL if page == 1 else LIST_PAGE_TPL.format(page)
        print(f"\n--- 第{page}页 ---")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            items = parse_list(r.text)
        except Exception as e:
            print(f"  [ERROR] page {page}: {e}")
            continue

        hit_old = False
        for item in items:
            if item["date"] < DATE_THRESHOLD:
                skip_count += 1
                hit_old = True
                continue
            print(f"  {item['date']} {item['title'][:50]}...")
            content, source = fetch_detail(item["url"])
            if store_item(item["title"], item["url"], item["date"], content, source):
                new_count += 1
                print(f"    ✓ 新增")
            else:
                skip_count += 1
                print(f"    - 已存在")

        # 如果这一页已经有超3年的数据，后面的页更旧，直接结束
        if hit_old:
            remaining = total_pages - page
            print(f"  检测到超期数据，剩余{remaining}页跳过")
            skip_count += remaining * 30  # 估算
            break

    print(f"\n\n=== 完成 ===")
    print(f"新增: {new_count}, 跳过: {skip_count}")

if __name__ == "__main__":
    main()
