#!/usr/bin/env python3
"""
贵溪市人民政府-通知 (www.guixi.gov.cn)
==================================
CMS: 大汉版通 (Hanweb) - JPage AJAX (无CSRF)
列表API: /module/web/jpage/dataproxy.jsp (POST, startrecord/endrecord)
详情: /art/2026/6/8/art_6229_1594503.html, <div id=zoom> 含正文

用法:
    python3 crawl_guixi_tz.py          # 全量(最多78页~3588条)
    python3 crawl_guixi_tz.py --test   # 测试 5 条
"""

import re, sys, os, time
from datetime import datetime, timedelta, timezone
from urllib.parse import quote
import requests, urllib3
import os
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME  = "贵溪市-通知"
PROXY_URL  = "http://www.guixi.gov.cn/module/web/jpage/dataproxy.jsp"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES = 5  # 3yr cutoff at page 78
CUTOFF     = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS    = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch_list(page):
    """Fetch JPage via POST"""
    start = (page - 1) * 15 + 1
    end = page * 15
    url_params = f"startrecord={start}&endrecord={end}&perpage=15"
    url_params += "&unitid=28132&webid=56&path=http://www.guixi.gov.cn/"
    url_params += "&webname=" + quote("贵溪市人民政府")
    url_params += "&col=1&columnid=6229&sourceContentType=1&permissiontype=0"
    
    post_data = {
        "col": "1", "webid": "56", "path": "http://www.guixi.gov.cn/",
        "columnid": "6229", "sourceContentType": "1", "unitid": "28132",
        "webname": "贵溪市人民政府", "permissiontype": "0",
    }
    try:
        r = requests.post(PROXY_URL + "?" + url_params, data=post_data, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  ! list page {page} fail: {str(e)[:60]}")
        return None

def parse_list(html):
    items = []
    for m in re.finditer(
        r'<li><span>(\d{4}-\d{2}-\d{2})</span>\s*<a[^>]*href="([^"]+)"[^>]*title="([^"]*)"',
        html, re.DOTALL
    ):
        pub_date = m.group(1)
        url = m.group(2)
        title = m.group(3).strip()
        items.append({"title": title, "url": url, "pub_date": pub_date})
    return items

def fetch_detail(url):
    """Extract body from <div id=zoom> — preserves attach links + tables"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        html = r.text
        idx = html.find('id=zoom')
        if idx < 0:
            idx = html.find('id="zoom"')
        if idx < 0:
            return ""
        div_start = html.rfind("<div", 0, idx)
        if div_start < 0:
            return ""
        section = html[div_start:]
        depth = 0
        content = ""
        for i in range(len(section)):
            if section[i:i+4] == "<div" and (i+4 >= len(section) or section[i+4] in " >\n\r\t"):
                depth += 1
            elif section[i:i+6] == "</div>":
                depth -= 1
                if depth == 0:
                    gt = section.find(">")
                    content = section[gt+1:i]
                    break
        from bs4 import BeautifulSoup
        MARKER = "\x00P\x00"
        BASE = "http://www.guixi.gov.cn"
        soup = BeautifulSoup(content, "html.parser")
        for tag in soup.find_all(["script", "style"]):
            tag.decompose()
        # Save attachment links as Markdown
        attach_links = []
        for a in soup.find_all("a"):
            href = a.get("href", "")
            txt = a.get_text(strip=True)
            if href and not href.startswith("javascript") and not href.startswith("#"):
                if href.startswith("/"):
                    href = BASE + href
                elif not href.startswith("http"):
                    href = url.rstrip(url.split("/")[-1]) + href
                attach_links.append(f"[{txt}]({href})")
            a.decompose()
        # Save tables
        tables_html = []
        for table in soup.find_all("table"):
            tables_html.append(str(table))
            table.decompose()
        # Paragraph markers
        for tag in soup.find_all(["p", "h1", "h2", "h3", "h4", "h5", "h6"]):
            tag.insert(0, MARKER)
            tag.append(MARKER)
        for br in soup.find_all("br"):
            br.replace_with(MARKER)
        # Unwrap inline
        for tag in soup.find_all(["span", "b", "strong", "font", "em", "i", "u", "s"]):
            tag.unwrap()
        text = soup.get_text(separator="", strip=True)
        text = re.sub(r"\x00P\x00(\s*\x00P\x00)+", "\x00P\x00", text)
        text = text.replace("\x00P\x00", "\n\n")
        text = re.sub(r"\n{3,}", "\n\n", text)
        text = text.strip()
        for tbl in tables_html:
            text += f"\n\n{tbl}"
        for link in attach_links:
            text += f"\n\n{link}"
        if not text or len(text) < 20:
            text = f"[{url.split('/')[-1]}]({url})"
            if attach_links:
                text += "\n\n" + "\n".join(attach_links)
        return text
    except Exception as e:
        print(f"  ! detail fail: {str(e)[:60]}")
        return ""

def to_db(items):
    if not items:
        return 0, 0
    import sqlite3
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, fail = 0, 0
    for it in items:
        try:
            db.execute(
                "INSERT OR REPLACE INTO gov_raw "
                "(site_name, title, page_url, content, publish_date, summary, tags) "
                "VALUES (?,?,?,?,?,?,?)",
                (
                    SITE_NAME,
                    (it.get("title") or "")[:500],
                    it.get("url", ""),
                    it.get("content", ""),
                    (it.get("pub_date") or "")[:10],
                    "",
                    "通知",
                )
            )
            if db.total_changes > 0:
                ok += 1
            else:
                fail += 1
        except Exception as e:
            fail += 1
    db.execute(
        "INSERT INTO gov_search(rowid, title, site_name, summary) "
        "SELECT r.id, r.title, r.site_name, r.summary "
        "FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,)
    )
    db.commit()
    db.close()
    return ok, fail

def crawl(test=False):
    all_items = []
    seen_urls = set()
    max_p = 5 if test else MAX_PAGES
    for page in range(1, max_p + 1):
        html = fetch_list(page)
        if not html:
            break
        items = parse_list(html)
        if not items:
            print(f"  [Page {page}] empty, reached end")
            break
        new = 0
        for item in items:
            if item["url"] in seen_urls:
                continue
            seen_urls.add(item["url"])
            if item["pub_date"] and item["pub_date"] < CUTOFF:
                continue
            all_items.append(item)
            new += 1
        f = items[0]["pub_date"] if items else "?"
        l = items[-1]["pub_date"] if items else "?"
        print(f"  [Page {page}] {len(items)} items ({f} ~ {l}), new: {new}")
        if new == 0:
            break
    print(f"  Total: {len(all_items)} items within 3yr")
    if test:
        all_items = all_items[:5]
        print(f"  TEST mode: {len(all_items)} items")
    if not all_items:
        return 0, 0
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:40]}... ", end="", flush=True)
        content = fetch_detail(item["url"])
        item["content"] = content
        print(f"{len(content)}B")
    ok, fail = to_db(all_items)
    return ok, fail

if __name__ == "__main__":
    test = "--test" in sys.argv
    mode = "TEST" if test else "FULL"
    print()
    print(f"[{SITE_NAME}] {mode}")
    t0 = time.time()
    ok, fail = crawl(test=test)
    print(f"  Time: {round(time.time()-t0, 1)}s")
    print(f"  New: {ok} Skip: {fail}")