#!/usr/bin/env python3
"""爬取平乡县人民政府 - 公告公示 (channel 14)



列表:
  第1页:  /channel/list/14.html
  2~10页: /channel/list/14_{n}.html
  11~76页: /xxgk/list/list.jsp?classId=14&pn={n}
  #  每页20条, 共76页~1520条

详情: /single/14/{id}.html
  正文: <div class="content">...深度计数取闭合
  日期: <meta PubDate>
  标题: <title> 或 <meta ArticleTitle>
  来源: <meta ContentSource>
"""

import sys, re, os, json, time
from datetime import datetime, timedelta
import sqlite3
import requests
from bs4 import BeautifulSoup

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "平乡县人民政府-公告公示"
BASE_URL = "https://www.pingxiangxian.gov.cn"
LIST_STATIC = BASE_URL + "/channel/list/14{n}.html"
LIST_DYNAMIC = BASE_URL + "/xxgk/list/list.jsp?classId=14&pn={n}"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
MAX_CONCURRENT = 10

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": BASE_URL + "/",
}


def get_list_url(page):
    """获取指定页码的列表URL"""
    if page == 1:
        return LIST_STATIC.format(n="")
    elif page <= 10:
        return LIST_STATIC.format(n="_" + str(page))
    else:
        return LIST_DYNAMIC.format(n=page)


def fetch_list(page):
    """获取一页列表数据, 返回 [(url, title, date), ...], total"""
    url = get_list_url(page)
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print("  [ERROR] fetch list page %d: %s" % (page, e), flush=True)
        return [], 0

    # 解析列表条目
    items = []
    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_="info-list-xxgk")
    if not ul:
        return [], 0

    for li in ul.find_all("li"):
        a = li.find("a")
        span = li.find("span", class_="time")
        if a and span:
            href = a.get("href", "")
            title = a.get("title", "") or a.get_text(strip=True)
            date = span.get_text(strip=True)
            if href:
                if href.startswith("/"):
                    href = BASE_URL + href
                items.append((href, title, date))

    # 从页面获取总页数
    total_pages = 0
    m = re.search(r"var pageCount = parseInt\('(\d+)'\)", html)
    if m:
        total_pages = int(m.group(1))

    return items, total_pages


def extract_content(html):
    """提取<div class=\"content\">中的正文内容, 深度计数法"""
    idx = html.find('<div class="content">')
    if idx < 0:
        return ""
    start = idx + len('<div class="content">')
    depth = 1
    pos = start
    while pos < len(html) and depth > 0:
        if html[pos:pos+4] == '<div':
            depth += 1
            pos += 4
        elif html[pos:pos+6] == '</div>':
            depth -= 1
            pos += 6
        else:
            pos += 1
    content = html[start:pos-6]
    # 清理 script/style
    content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
    content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)
    return content.strip()


def fetch_detail(url):
    """获取详情页: 返回 {title, date, content, source}"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print("  [ERROR] fetch detail: %s - %s" % (url, e), flush=True)
        return {"title": "", "date": "", "content": "", "source": ""}

    # 标题
    title = ""
    m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
    if m:
        title = m.group(1)
    if not title:
        m = re.search(r'<title>(.*?)</title>', html)
        if m:
            t = m.group(1)
            title = t.replace(" - 平乡县人民政府", "").strip()

    # 日期
    date = ""
    m = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', html)
    if m:
        date = m.group(1)[:10]

    # 来源
    source = ""
    m = re.search(r'<meta\s+name="ContentSource"\s+content="([^"]*)"', html)
    if m:
        source = m.group(1)

    # 正文
    content = extract_content(html)

    return {"title": title, "date": date, "content": content, "source": source}


def store(conn, item):
    """插入一条记录到 gov_raw, 返回 True/False"""
    url = item["url"]
    site_url = url  # page_url
    source_url = BASE_URL  # source_url
    source = item.get("source", "")
    tags = source if source else "公告公示"

    try:
        conn.execute(
            """INSERT OR IGNORE INTO gov_raw
               (site_name, source_url, page_url, title, publish_date,
                summary, content, status, category, tags)
               VALUES (?, ?, ?, ?, ?, ?, ?, 'active', ?, ?)""",
            (
                SITE_NAME,
                source_url,
                url,
                item["title"],
                item["date"],
                (item["content"] or "")[:500],
                item["content"] or "",
                "公告公示",
                tags,
            ),
        )
        conn.commit()
        return conn.total_changes > 0
    except Exception as e:
        print("  [ERROR] store: %s" % e, flush=True)
        return False


def main():
    is_inc = "incremental" in sys.argv
    mode = "增量" if is_inc else "全量"
    print("=== %s模式: %s ===\n" % (mode, SITE_NAME))

    # 先取首页获得总页数
    items, total_pages = fetch_list(1)
    if not items:
        print("❌ 无法获取列表数据")
        return
    print("总页数: %d, 每页20条" % total_pages)

    if total_pages == 0:
        total_pages = 76  # 兜底

    conn = sqlite3.connect(DB_PATH, timeout=60)
    total_new = 0
    total_skip = 0

    for pi in range(1, min(total_pages, _MAX_PG or total_pages)+1):
        print("\n--- 第 %d/%d 页 ---" % (pi, total_pages), flush=True)

        items, _ = fetch_list(pi)
        if not items:
            print("  (空页, 停止)")
            break

        print("  条目: %d" % len(items), flush=True)

        # 检查第一篇文章的日期是否超过3年
        first_date = items[0][2]
        last_date = items[-1][2]
        print("  日期范围: %s ~ %s" % (last_date, first_date), flush=True)

        if last_date < CUTOFF:
            print("  ⏹️  已过3年期限 (%s < %s), 停止" % (last_date, CUTOFF))
            break

        for url, title, date in items:
            # 日期过滤
            if date < CUTOFF:
                continue

            # 获取详情
            detail = fetch_detail(url)
            if not detail["content"]:
                print("  ⚠️  正文为空: %s" % title[:50], flush=True)

            item_data = {
                "url": url,
                "title": detail["title"] or title,
                "date": detail["date"] or date,
                "content": detail["content"],
                "source": detail["source"],
            }

            if store(conn, item_data):
                total_new += 1
                print("  ✅ %s | %s" % (item_data["date"], item_data["title"][:60]), flush=True)
            else:
                total_skip += 1

            time.sleep(0.3)  # 礼貌间隔

    conn.close()
    print("\n=== 完成: 新增 %d, 跳过 %d ===" % (total_new, total_skip))


if __name__ == "__main__":
    main()
