#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""新泰市-通知公告 (TRS/大汉版通 jPage AJAX, col47996)"""

import requests, re, sqlite3, os
from datetime import datetime, timedelta

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36"
}
BASE = "http://www.xintai.gov.cn"
LIST_URL = BASE + "/col/col47996/index.html"
PROXY_URL = BASE + "/module/web/jpage/dataproxy.jsp"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
SITE = "新泰市-通知公告"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
PER_PAGE = 15

PROXY_PARAMS = {
    "col": "1", "webid": "341",
    "path": "http://www.xintai.gov.cn/",
    "columnid": "47996", "unitid": "146655",
    "webname": "新泰市人民政府",
    "sourceContentType": "1", "permissiontype": "0"
}

def fetch_proxy_records(start, end):
    url = "%s?startrecord=%d&endrecord=%d&perpage=%d&unitid=%s&webid=%s&path=%s&webname=%s&col=%s&columnid=%s&sourceContentType=%s&permissiontype=%s" % (
        PROXY_URL, start, end, PER_PAGE,
        PROXY_PARAMS["unitid"], PROXY_PARAMS["webid"], PROXY_PARAMS["path"],
        PROXY_PARAMS["webname"], PROXY_PARAMS["col"], PROXY_PARAMS["columnid"],
        PROXY_PARAMS["sourceContentType"], PROXY_PARAMS["permissiontype"]
    )
    r = requests.post(url, headers={**HEADERS, "Referer": LIST_URL, "X-Requested-With": "XMLHttpRequest"},
                      data=PROXY_PARAMS, timeout=30)
    items = []
    for rec in re.findall(r"<record><!\[CDATA\[(.*?)\]\]></record>", r.text, re.DOTALL):
        a = re.search(r'<a[^>]*href="([^"]+)"[^>]*>([^<]+)</a>', rec)
        date = re.search(r"<span>(\d{4}-\d{2}-\d{2})</span>", rec)
        if a and date:
            items.append({"title": a.group(2).strip(), "date": date.group(1), "url": a.group(1)})
    return items

def get_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except:
        return "", "正文为空"

    # Title from h1
    title = ""
    m = re.search(r"<h1[^>]*>\s*([^<]+?)\s*</h1>", r.text)
    if m: title = m.group(1).strip()
    if not title:
        m = re.search(r"<title>([^<]+)</title>", r.text)
        if m: title = m.group(1).strip()

    # Content from div.sp_content
    content = "正文为空"
    m = re.search(r'<div class="sp_content"[^>]*>(.*?)</div>', r.text, re.DOTALL)
    if m:
        c = m.group(1).strip()
        if len(c) > 50:
            content = c

    if content != "正文为空":
        content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL)
        content = re.sub(r"<iframe[^>]*>.*?</iframe>", "", content, flags=re.DOTALL)
        content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL)
        # Remove content markers
        content = re.sub(r"<!--.*?-->", "", content)
        content = re.sub(r"<meta[^>]*>", "", content)
        content = content.strip()
        if not content:
            content = "正文为空"

    title = re.sub(r"<[^>]+>", "", title).strip()
    return title, content

def main():
    pages = int(os.environ.get("PAGES", "5"))

    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    cur = conn.cursor()
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")

    total_new = 0
    total_skipped = 0

    for page in range(1, pages + 1):
        start = (page - 1) * PER_PAGE + 1
        end = page * PER_PAGE
        items = fetch_proxy_records(start, end)
        print(f"Page {page}: {len(items)} items (records {start}-{end})")

        if not items:
            break

        for item in items:
            title = item["title"]
            url = item["url"]
            daytime = item["date"]

            if not title or not url:
                continue
            if daytime < CUTOFF:
                total_skipped += 1
                continue

            if not url.startswith("http"):
                url = BASE + url if url.startswith("/") else BASE + "/" + url

            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if cur.fetchone():
                total_skipped += 1
                continue

            try:
                detail_title, content = get_detail(url)
                if content == "正文为空":
                    total_skipped += 1
                    continue

                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content, date_rank, category) VALUES (?,?,?,?,?,?,?)",
                    (SITE, url, detail_title or title, daytime, content, int(daytime.replace("-", "")), "政府公告")
                )
                if cur.rowcount > 0:
                    total_new += 1
            except Exception as e:
                print(f"  ERR: {title[:30]} - {str(e)[:60]}")
                total_skipped += 1

        conn.commit()

    conn.close()
    print(f"\n结果: {total_new} 新增, {total_skipped} 跳过")

if __name__ == "__main__":
    main()
