#!/usr/bin/env python3
import os
# -*- coding: utf-8 -*-
"""
开平市沙塘镇人民政府信息公开 - 通知公告(3555) 部门文件(6334)
广东标准GKMLPT系统 Vue SPA + JSON API
Usage:
  python3 crawl_kpstz_gkml.py --id 3555 --name kpstz_tzgg   # 通知公告
  python3 crawl_kpstz_gkml.py --id 6334 --name kpstz_bmwj   # 部门文件
"""
import sys, os, re, time, json, requests
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DOMAIN = "https://www.kaiping.gov.cn"
BASE_PATH = "/jmkpsstz"
API_LIST = DOMAIN + BASE_PATH + "/gkmlpt/api/all/%s"
MAX_PAGES = 5


def fetch(url, encoding="utf-8"):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = encoding
        return r.text
    except Exception as e:
        print("  FETCH ERROR %s: %s" % (url, e), file=sys.stderr)
        return None


def fetch_json(url):
    html = fetch(url)
    if not html:
        return None
    try:
        return json.loads(html)
    except:
        return None


def parse_ts(ts):
    import datetime
    dt = datetime.datetime.fromtimestamp(ts)
    return dt.strftime("%Y-%m-%d")


def get_detail(url, title_from_list):
    html = fetch(url)
    if not html:
        return "", "", [], title_from_list

    soup = BeautifulSoup(html, "lxml")

    full_title = title_from_list
    pt = soup.select_one("title")
    if pt:
        t = pt.get_text(strip=True)
        if t:
            full_title = t

    pub_date = ""
    m = re.search(r"(\d{4}[-/]\d{2}[-/]\d{2})", html)
    if m:
        pub_date = m.group(1).replace("/", "-")

    body = soup.select_one("div.article-content")
    content_html = ""
    attachments = []

    if body:
        for a_tag in body.find_all("a", href=True):
            h = a_tag["href"].strip()
            txt = a_tag.get_text(strip=True)
            if re.search(r"\.(docx?|pdf|pptx?|xlsx?|jpg|png|gif)($|\?)", h, re.IGNORECASE) or "/download/" in h:
                fname = txt or h.split("/")[-1]
                attach_url = h if h.startswith("http") else DOMAIN + h
                attachments.append({"url": attach_url, "name": fname})
                a_tag.replace_with("[%s](%s)" % (fname, attach_url))

        tbl_soup = BeautifulSoup(str(body), "html.parser")
        tables = tbl_soup.find_all("table")
        table_markers = []
        for i, tb in enumerate(tables):
            key = "___TBL_%d___" % i
            table_markers.append((key, str(tb)))
            tb.replace_with(key)

        body_html = str(tbl_soup)
        body_soup = BeautifulSoup(body_html, "html.parser")

        for tag in body_soup.find_all(["script", "style"]):
            tag.decompose()
        for tag in body_soup.find_all(["span", "font", "b", "strong", "i", "em", "u"]):
            tag.unwrap()
        for p in body_soup.find_all("p"):
            p.insert(0, "\n\n")
            p.unwrap()
        for d in body_soup.find_all("div"):
            d.insert(0, "\n")
            d.unwrap()
        for br in body_soup.find_all("br"):
            br.replace_with("\n")

        content_html = body_soup.get_text(separator="", strip=False)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html).strip()

        for key, tbl_html in table_markers:
            content_html = content_html.replace(key, "\n\n" + tbl_html + "\n\n")

    if not content_html or len(content_html.strip()) < 20:
        content_html = '<p><a href="%s">%s</a></p>' % (url, full_title)

    return content_html, pub_date, attachments, full_title


def main():
    import argparse
    parser = argparse.ArgumentParser(description="开平市沙塘镇 GKMLPT 爬虫")
    parser.add_argument("--id", default="3555", help="栏目ID (3555=通知公告, 6334=部门文件)")
    parser.add_argument("--name", default="kpstz_gkml", help="站点名称")
    parser.add_argument("--pages", type=int, default=5, help="爬取页数 (默认5)")
    args = parser.parse_args()

    col_id = args.id
    site_name = args.name

    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()

    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (site_name,))
    existing = c.fetchone()[0]

    print("栏目ID=%s 站点=%s: 已有 %d 条" % (col_id, site_name, existing), file=sys.stderr)

    all_new = 0
    for pg in range(1, args.pages + 1):
        api_url = API_LIST % col_id + "?page=%d&pageSize=20" % pg
        data = fetch_json(api_url)
        if not data:
            print("  第%d页 API失败" % pg, file=sys.stderr)
            continue

        articles = data.get("articles", [])
        if not articles:
            print("  第%d页 无条目" % pg, file=sys.stderr)
            break

        total = data.get("total", 0)
        print("  第%d页: %d 条 (共%d)" % (pg, len(articles), total), file=sys.stderr)

        for art in articles:
            detail_url = art.get("url", "")
            if not detail_url:
                continue
            detail_url = detail_url.replace("http://", "https://")

            title = art.get("title", "")
            ts = art.get("display_publish_time", 0) or art.get("date", 0)
            date = parse_ts(ts) if ts else ""
            attachment_list = art.get("attachment", [])

            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, site_name))
            if c.fetchone():
                continue

            content, pub_date, attachments, full_title = get_detail(detail_url, title)
            if not date and pub_date:
                date = pub_date

            for att in attachment_list:
                att_url = att.get("url", "")
                att_name = att.get("name", "")
                if att_url:
                    att_url = att_url.replace("http://", "https://")
                    if not any(a["url"] == att_url for a in attachments):
                        attachments.append({"url": att_url, "name": att_name or att_url.split("/")[-1]})

            try:
                c.execute(
                    """INSERT INTO gov_raw (site_name, page_url, title, content, publish_date, source_url, status, attachments)
                       VALUES (?,?,?,?,?,?,?,?)""",
                    (site_name, detail_url, full_title[:500], content, date[:20],
                     detail_url, "synced", json.dumps(attachments, ensure_ascii=False))
                )
                all_new += 1
                print("    + %s (%s) body:%dB attach:%d" % (full_title[:50], date, len(content), len(attachments)), file=sys.stderr)
            except Exception as e:
                print("    ! 入库失败: %s" % e, file=sys.stderr)

        time.sleep(0.3)

    conn.commit()
    print("\n完成! 新增 %d 条" % all_new, file=sys.stderr)



    conn.close()


if __name__ == "__main__":
    main()
