#!/usr/bin/env python3
import os
"""新泰市—生态治理栏目爬虫（环评审批公告）
URL: http://www.xintai.gov.cn/col/col92333/index.html
JPage AJAX分页 + 详情页 sp_content
"""

import requests, re, sqlite3, sys
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36"}
BASE = "http://www.xintai.gov.cn"
LIST_URL = BASE + "/col/col92333/index.html"
PROXY_URL = BASE + "/module/web/jpage/dataproxy.jsp"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
SITE = "新泰市-生态治理"

PROXY_PARAMS = {
    "col": "1", "webid": "341",
    "path": "http://www.xintai.gov.cn/",
    "columnid": "92333", "unitid": "146655",
    "webname": "新泰市人民政府",
    "sourceContentType": "1", "permissiontype": "0"
}

def fetch_initial_records():
    """从首页HTML的XML datastore提取前45条"""
    r = requests.get(LIST_URL, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    items = []
    # 提取 <li> 记录
    for li in re.findall(r'<li class="clearfix">(.*?)</li>', r.text, re.DOTALL):
        a = re.search(r'<a[^>]*href="([^"]+)"[^>]*>([^<]+)</a>', li)
        date = re.search(r'<span>(\d{4}-\d{2}-\d{2})</span>', li)
        if a and date:
            items.append({"title": a.group(2).strip(), "date": date.group(1), "url": a.group(1)})
    return items

def fetch_proxy_records(start, end):
    """通过JPage AJAX代理获取记录"""
    url = "%s?startrecord=%d&endrecord=%d&perpage=%d&unitid=%s&webid=%s&path=%s&webname=%s&col=%s&columnid=%s&sourceContentType=%s&permissiontype=%s" % (
        PROXY_URL, start, end, 15,
        PROXY_PARAMS["unitid"], PROXY_PARAMS["webid"], PROXY_PARAMS["path"],
        PROXY_PARAMS["webname"], PROXY_PARAMS["col"], PROXY_PARAMS["columnid"],
        PROXY_PARAMS["sourceContentType"], PROXY_PARAMS["permissiontype"]
    )
    r = requests.post(url, headers={**HEADERS, "Referer": LIST_URL, "X-Requested-With": "XMLHttpRequest"},
                      data=PROXY_PARAMS, timeout=30)
    items = []
    for rec in re.findall(r'<record><!\[CDATA\[(.*?)\]\]></record>', r.text, re.DOTALL):
        a = re.search(r'<a[^>]*href="([^"]+)"[^>]*>([^<]+)</a>', rec)
        date = re.search(r'<span>(\d{4}-\d{2}-\d{2})</span>', rec)
        if a and date:
            items.append({"title": a.group(2).strip(), "date": date.group(1), "url": a.group(1)})
    return items

def get_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except:
        return ""
    soup = BeautifulSoup(r.text, "html.parser")
    div = soup.select_one("div.sp_content")
    if div:
        return str(div)
    return ""

def main():
    today = datetime.now().strftime("%Y-%m-%d")
    count = 0
    conn = sqlite3.connect(os.getenv("SEARCH_DB", "/root/search.db"), timeout=60)
    c = conn.cursor()

    # Step 1: initial 45 records from HTML
    print("抓取首页数据...")
    items = fetch_initial_records()
    print("  首页获取 %d 条" % len(items))

    # Step 2: proxy records (要检测是否超3年)
    # 大致估算需要多少页。记录46起从2025-11开始，要到2023-06约需80~100条
    # 每次取15条，取到超3年为止
    for start in range(46, 76, 15):  # limited: 2pg
        end = min(start + 14, 419)
        print("  AJAX %d-%d..." % (start, end), end="", flush=True)
        batch = fetch_proxy_records(start, end)
        if not batch:
            print("空，结束")
            break
        items.extend(batch)
        print("%d条" % len(batch))
        # 检查最后一条日期，如果已超3年停止
        if batch[-1]["date"] < CUTOFF:
            break

    print("\n共获取 %d 条记录，开始爬详情..." % len(items))

    for it in items:
        if it["date"] < CUTOFF:
            continue
        sys.stdout.write("  [%s] %s... " % (it["date"], it["title"][:40]))
        sys.stdout.flush()
        body = get_detail(it["url"])
        if not body:
            print("无正文")
            continue
        try:
            c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url, summary, category, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_xintai.py')""",
                (it["url"], it["title"], body, it["date"], SITE, it["url"], "", ""))
            conn.commit()
            count += 1
            print("OK")
        except Exception as e:
            print("DBERR: %s" % e)

    conn.close()
    print("\n✅ 新泰市：共入库 %d 条" % count)

if __name__ == "__main__":
    main()
