#!/usr/bin/env python3
"""爬取阜阳经济开发区 - 通知公告 (column 4218)

列表: /Content/showList/4218/page_{n}.html
  44页, 每页15条, 共655条
详情: /Content/show/{id}.html
  正文: <div class="m-dttexts f-clearfix j-fontContent" id="zoom">
  标题: <h1 class="u-lgtit">
  日期: <span>发布时间：YYYY-MM-DD HH:MM</span>
  来源: <span>来源：XXX</span>
"""

import sys, re, os, time
from datetime import datetime, timedelta
import sqlite3
import requests
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "阜阳经济开发区-通知公告"
BASE_URL = "https://invest.fy.gov.cn"
LIST_URL = BASE_URL + "/Content/showList/4218/page_{n}.html"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}


def fetch_list(page):
    """获取一页列表, 返回 [(url, title, date), ...]"""
    url = LIST_URL.format(n=page)
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print("  [ERROR] list page %d: %s" % (page, e), flush=True)
        return []

    soup = BeautifulSoup(html, "html.parser")
    ul = soup.select_one("div.m-cglist ul")
    if not ul:
        return []

    items = []
    for li in ul.find_all("li"):
        a = li.find("a")
        span = li.find("span")
        if a and span:
            href = a.get("href", "")
            title = (a.get("title") or a.get_text(strip=True)).strip()
            date = span.get_text(strip=True)
            if href and not href.startswith("http"):
                href = BASE_URL + href
            items.append((href, title, date))

    return items


def fetch_detail(url):
    """获取详情页, 返回 {title, date, source, content}"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print("  [ERROR] detail: %s - %s" % (url, e), flush=True)
        return {"title": "", "date": "", "source": "", "content": ""}

    # 标题
    title = ""
    m = re.search(r'<h1[^>]*class="u-lgtit[^"]*"[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m:
        title = m.group(1).strip()

    # 日期
    date = ""
    m = re.search(r"发布时间[：:]\s*([\d-]+)", html)
    if m:
        date = m.group(1).strip()

    # 来源
    source = ""
    m = re.search(r"来源[：:]\s*([^<]+?)(?:</span>|<)", html)
    if m:
        source = m.group(1).strip()

    # 正文 - depth count for <div class="m-dttexts f-clearfix j-fontContent" id="zoom">
    content = ""
    # Try with id="zoom" first (more specific)
    idx = html.find('id="zoom"')
    if idx >= 0:
        # Find the enclosing div start
        div_start = html.rfind("<div", 0, idx)
        if div_start >= 0:
            # Find the > after <div ...>
            gt = html.find(">", div_start)
            if gt >= 0:
                start = gt + 1
                depth = 1
                pos = start
                while pos < len(html) and depth > 0:
                    if html[pos:pos+4] == "<div":
                        depth += 1
                        pos += 4
                    elif html[pos:pos+6] == "</div>":
                        depth -= 1
                        pos += 6
                    else:
                        pos += 1
                content = html[start:pos-6]

    # Clean script/style
    if content:
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)
        content = content.strip()

    return {"title": title, "date": date, "source": source, "content": content}


def store(conn, item):
    """插入一条记录"""
    tags = item.get("source", "") or "通知公告"
    try:
        conn.execute(
            """INSERT OR IGNORE INTO gov_raw
               (site_name, source_url, page_url, title, publish_date,
                summary, content, status, category, tags)
               VALUES (?, ?, ?, ?, ?, ?, ?, 'active', ?, ?)""",
            (
                SITE_NAME,
                BASE_URL,
                item["url"],
                item["title"],
                item["date"],
                (item["content"] or "")[:500],
                item["content"] or "",
                "通知公告",
                tags,
            ),
        )
        conn.commit()
        return conn.total_changes > 0
    except Exception as e:
        print("  [ERROR] store: %s" % e, flush=True)
        return False


def main():
    is_inc = "incremental" in sys.argv
    mode = "增量" if is_inc else "全量"
    print("=== %s模式: %s ===\n" % (mode, SITE_NAME))

    # 获取第一页数据, 同时看总页数
    items = fetch_list(1)
    if not items:
        print("❌ 无法获取列表数据")
        return

    # 从页面获取总页数
    total_pages = 44  # 已知固定值
    print("总页数: %d, 每页15条, 预计~655条" % total_pages)

    max_pages = 1 if is_inc else total_pages

    conn = sqlite3.connect(DB_PATH)
    total_new = 0
    total_skip = 0

    for pi in range(1, max_pages + 1):
        print("\n--- 第 %d/%d 页 ---" % (pi, max_pages), flush=True)

        items = fetch_list(pi)
        if not items:
            print("  (空页, 停止)")
            break

        first_date = items[0][2]
        last_date = items[-1][2]
        print("  条目: %d, 日期: %s ~ %s" % (len(items), last_date, first_date), flush=True)

        if last_date < CUTOFF:
            print("  ⏹️  已过3年期限 (%s < %s), 停止" % (last_date, CUTOFF))
            break

        for url, title, date in items:
            if date < CUTOFF:
                continue

            detail = fetch_detail(url)
            if not detail["content"]:
                print("  ⚠️  正文为空: %s" % title[:50], flush=True)

            item_data = {
                "url": url,
                "title": detail["title"] or title,
                "date": detail["date"] or date,
                "content": detail["content"],
                "source": detail["source"],
            }

            if store(conn, item_data):
                total_new += 1
                print("  ✅ %s | %s" % (item_data["date"], item_data["title"][:60]), flush=True)
            else:
                total_skip += 1

            time.sleep(0.3)

    conn.close()
    print("\n=== 完成: 新增 %d, 跳过 %d ===" % (total_new, total_skip))


if __name__ == "__main__":
    main()
