#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""开平市城市管理和综合执法局 - 公众参与 (3365) 爬虫
CMS: 广东标准 GKMLPT (Vue SPA + JSON API)
列表: GET https://www.kaiping.gov.cn/jmkpscgj/gkmlpt/api/all/3365
     ⚠️ API 忽略分页参数恒返回全部 87 条 (total=87, 2020-06 ~ 2026-07)
详情: /jmkpscgj/gkmlpt/content/{class}/{id//1000}/post_{id}.html
     meta ArticleTitle (无 PubDate meta, 日期用 API date 字段) + div.article-content 正文
     ⚠️ 页面有两个 article-content div, 第二个 policy-article-content 为空, 取第一个
"""
import sys, re, time, json, ssl, sqlite3, urllib.request, datetime
from urllib.parse import urljoin
from bs4 import BeautifulSoup

SITE_NAME = "开平市城市管理和综合执法局-公众参与"
SCRIPT_NAME = "crawl_kpscgj_gkml.py"
GROUP_NAME = "广东"
API_URL = "https://www.kaiping.gov.cn/jmkpscgj/gkmlpt/api/all/3365"
DB_PATH = "/mnt/data/search.db"

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "application/json,text/html,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def http_get(url, timeout=30, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [WARN] 请求失败 {url}: {e}", file=sys.stderr)
                return ""
            time.sleep(1.2)
    return ""


def clean_title(title):
    """strip &middot;&nbsp; 实体前缀 和省略号截断后缀"""
    title = title.replace("&middot;", "").replace("&nbsp;", "")
    title = title.replace("\u200b", "").replace("\ufeff", "")
    title = re.sub(r"^[\s\xa0·\u00b7]+", "", title)
    title = re.sub(r"\s*\.{3,}\s*$", "", title)
    return title.strip()


def fetch_page(page):
    """第page页列表条目。API 单次返回全部, 仅 page==1 有效。"""
    if page != 1:
        return []
    text = http_get(API_URL)
    if not text:
        return []
    try:
        d = json.loads(text)
    except Exception:
        return []
    items = []
    for a in d.get("articles", []):
        url = a.get("url", "")
        title = clean_title(a.get("title", ""))
        if not url or not title:
            continue
        date = ""
        ts = a.get("date") or a.get("first_publish_time") or 0
        if ts:
            try:
                date = datetime.datetime.fromtimestamp(ts).strftime("%Y-%m-%d")
            except Exception:
                date = ""
        items.append((url, title, date))
    return items


def extract_detail(html, page_url, list_title):
    """提取 (title, date, content_html)"""
    title = ""
    mt = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"', html)
    if mt:
        title = clean_title(mt.group(1))
    if not title:
        title = list_title

    soup = BeautifulSoup(html, "html.parser")
    cm = soup.find("div", class_="article-content")
    if not cm:
        return title, "", ""

    # 附件收集 (article-content 内)
    attachments = []
    for a in cm.find_all("a", href=True):
        href = a["href"]
        txt = a.get_text(strip=True)
        if (re.search(r"(?i)\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|ofd)(\?|$)", href)
                or "download" in href.lower() or "attach" in href.lower() or txt.startswith("附件")):
            abs_url = urljoin(page_url, href)
            if not abs_url.startswith(("http://", "https://")):
                abs_url = "http:" + abs_url if abs_url.startswith("//") else abs_url
            name = txt if txt else abs_url.split("/")[-1]
            attachments.append((a, name, abs_url))
    for a, name, abs_url in attachments:
        new_a = soup.new_tag("a", href=abs_url, target="_blank")
        new_a.string = name
        if a.parent is not None:
            p = a.parent
            idx_a = None
            for ci, child in enumerate(p.contents):
                if child is a:
                    idx_a = ci
                    break
            if idx_a is not None:
                for child in list(p.contents[:idx_a]):
                    if getattr(child, "name", None) == "img":
                        child.decompose()
        a.replace_with(new_a)
    attached_hrefs = {abs_url for _, _, abs_url in attachments}

    # 正文段落提取
    parts = []
    for el in cm.find_all(recursive=False):
        if el.name == "table":
            parts.append(str(el))
            continue
        if el.name in ("p", "div", "h1", "h2", "h3", "ul", "ol", "li", "blockquote"):
            txt = el.get_text("", strip=True)
            if not txt and not el.find("img"):
                continue
            inner_tables = el.find_all("table")
            el_has_attach = any(
                urljoin(page_url, x["href"]) in attached_hrefs or x["href"] in attached_hrefs
                for x in el.find_all("a", href=True))
            if inner_tables or el_has_attach:
                parts.append(str(el))
                continue
            parts.append(txt)
        else:
            txt = el.get_text("", strip=True)
            if txt:
                parts.append(txt)

    # 首段若与标题重复则去掉 (gkmlpt 正文第一段常是标题)
    if parts:
        first_key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", parts[0]))
        title_key = re.sub(r"\s+", "", title)
        if first_key and first_key == title_key:
            parts = parts[1:]

    # 段落去重
    seen = set()
    final_parts = []
    for p in parts:
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if key and key not in seen:
            seen.add(key)
            final_parts.append(p)

    content = "\n\n".join(final_parts)
    content_unescaped = content.replace("&amp;", "&")
    for _, name, abs_url in attachments:
        if abs_url not in content and abs_url not in content_unescaped:
            content += f'\n\n<p><a href="{abs_url}" target="_blank">{name}</a></p>'

    content = re.sub(r"打印本页[^\n]*", "", content)
    content = re.sub(r"\n{3,}", "\n\n", content).strip()
    return title, "", content


def main():
    max_pages = 1
    args = sys.argv[1:]
    i = 0
    while i < len(args):
        a = args[i]
        if a.startswith("--pages="):
            max_pages = int(a.split("=", 1)[1])
        elif a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
            i += 1
        i += 1

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()
    total_new = 0
    total_dup = 0
    total_skip = 0

    for page in range(1, max_pages + 1):
        items = fetch_page(page)
        if not items:
            if page == 1:
                print("  [WARN] 第1页无数据", file=sys.stderr)
            break
        print(f"  第{page}页: 找到 {len(items)} 条")
        for url, title, date in items:
            try:
                c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
                if c.fetchone():
                    total_dup += 1
                    continue
            except sqlite3.OperationalError:
                time.sleep(3)
                continue
            dhtml = http_get(url)
            if not dhtml:
                total_skip += 1
                continue
            d_title, _, content = extract_detail(dhtml, url, title)
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                total_skip += 1
                continue
            if not d_title:
                d_title = title
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if "<table" in content else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (d_title, summary, content, url, url, date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                rid = cur.lastrowid
                c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, d_title, SITE_NAME, summary))
                conn.commit()
                total_new += 1
                print(f"    [{date}] {d_title[:45]}")
            except sqlite3.OperationalError as e:
                if "locked" in str(e):
                    conn.rollback()
                    time.sleep(8)
                    try:
                        cur = c.execute(
                            "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                            (d_title, summary, content, url, url, date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                        rid = cur.lastrowid
                        c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                                  (rid, d_title, SITE_NAME, summary))
                        conn.commit()
                        total_new += 1
                        print(f"    [{date}] {d_title[:45]} (重试成功)")
                    except Exception:
                        conn.rollback()
                        total_skip += 1
                else:
                    total_skip += 1
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.3)

    conn.close()
    print(f"\n新增: {total_new}  重复: {total_dup}  过短/404: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
