#!/usr/bin/env python3
"""
贵溪市人民政府-环境审批 (www.guixi.gov.cn)
========================================
CMS: 大汉版通 (Hanweb) - JPage AJAX
列表API: /module/web/jpage/dataproxy.jsp (POST)
详情: /art/YYYY/M/D/art_COLUMNID_ARTICLEID.html, <div id=zoom> 含正文
"""
import re, sys, os, time
from datetime import datetime, timedelta, timezone
from urllib.parse import quote
import requests, urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME  = "贵溪市-环境审批"
PROXY_URL  = "http://www.guixi.gov.cn/module/web/jpage/dataproxy.jsp"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF     = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS    = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

UNITID = "28132"
COLUMNID = "6947"


def fetch_list_page(page, per_page=20):
    start = (page - 1) * per_page + 1
    end = page * per_page
    url_params = "startrecord={}&endrecord={}&perpage={}".format(start, end, per_page)
    url_params += "&unitid={}&webid=56&path=http://www.guixi.gov.cn/".format(UNITID)
    url_params += "&webname=" + quote("贵溪市人民政府")
    url_params += "&col=1&columnid={}&sourceContentType=3&permissiontype=0".format(COLUMNID)

    post_data = {
        "col": "1", "webid": "56", "path": "http://www.guixi.gov.cn/",
        "columnid": COLUMNID, "sourceContentType": "3", "unitid": UNITID,
        "webname": "贵溪市人民政府", "permissiontype": "0",
    }
    try:
        r = requests.post(PROXY_URL + "?" + url_params, data=post_data, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("  ! list page {} fail: {}".format(page, str(e)[:60]))
        return None


def parse_list(html):
    items = []
    # Record format: <span>2025-04-01</span>  <a href="..." title="...">title</a>
    for m in re.finditer(
        r'<span>(\d{4}-\d{2}-\d{2})</span>\s*<a[^>]*href="([^"]+)"[^>]*title="([^"]*)"',
        html
    ):
        pub_date = m.group(1)
        url = m.group(2)
        title = m.group(3).strip()
        if not url.startswith("http"):
            url = "http://www.guixi.gov.cn" + url
        items.append({"title": title, "url": url, "pub_date": pub_date})
    return items


def get_total():
    html = fetch_list_page(1, 1)
    if not html:
        return 0
    m = re.search(r"totalrecord>(\d+)", html)
    return int(m.group(1)) if m else 0


def fetch_detail(url):
    """Extract article body from <div id=zoom> using depth counting"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        html = r.text
        idx = html.find('id=zoom')
        if idx < 0:
            idx = html.find('id="zoom"')
        if idx < 0:
            return ""
        div_start = html.rfind("<div", 0, idx)
        if div_start < 0:
            return ""
        section = html[div_start:]
        depth = 0
        content = ""
        for i in range(len(section)):
            if section[i:i+4] == "<div" and (i+4 >= len(section) or section[i+4] in " >\n\r\t"):
                depth += 1
            elif section[i:i+6] == "</div>":
                depth -= 1
                if depth == 0:
                    content = section[7:i]
                    break
        content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL|re.I)
        content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL|re.I)
        return content.strip()
    except Exception as e:
        print("  ! detail fail: {}".format(str(e)[:60]))
        return ""


def to_db(items):
    if not items:
        return 0, 0
    import sqlite3
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, fail = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, title, page_url, content, publish_date, summary, tags) "
                "VALUES (?,?,?,?,?,?,?)",
                (
                    SITE_NAME,
                    (it.get("title") or "")[:500],
                    it.get("url", ""),
                    it.get("content", ""),
                    (it.get("pub_date") or "")[:10],
                    "",
                    "环境审批",
                )
            )
            if db.total_changes > 0:
                ok += 1
            else:
                fail += 1
        except Exception as e:
            fail += 1
    db.commit()
    db.close()
    return ok, fail


def main():
    print("=== {} ===".format(SITE_NAME))
    print("日期阈值: {}".format(CUTOFF))

    total = get_total()
    print("总记录数: {}".format(total))

    if total == 0:
        print("无记录")
        return

    per_page = 20
    total_pages = (total + per_page - 1) // per_page

    all_items = []
    for page in range(1, min(total_pages, 5) + 1):
        html = fetch_list_page(page)
        if not html:
            break
        items = parse_list(html)
        if not items:
            print("  [Page {}] empty".format(page))
            break
        new = 0
        for item in items:
            if item["pub_date"] and item["pub_date"] < CUTOFF:
                continue
            all_items.append(item)
            new += 1
        f = items[0]["pub_date"] if items else "?"
        l = items[-1]["pub_date"] if items else "?"
        print("  [Page {}] {} items ({} ~ {}), within 3yr: {}".format(page, len(items), f, l, new))
        if new == 0:
            break

    print("\n共{}条待抓取详情".format(len(all_items)))

    if not all_items:
        print("无需抓取")
        return

    for i, item in enumerate(all_items):
        print("  [{}/{}] {}... ".format(i+1, len(all_items), item['title'][:40]), end="", flush=True)
        content = fetch_detail(item["url"])
        item["content"] = content
        print("{}B".format(len(content)))

    ok, fail = to_db(all_items)
    print("\n=== 完成 ===")
    print("新增: {}, 跳过: {}".format(ok, fail))


if __name__ == "__main__":
    main()
