#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_huidong_zrzyj.py - 惠东县自然资源局-业务工作
https://www.huidong.gov.cn/hzhdzrzyj/gkmlpt/index (广东标准 gkmlpt Vue SPA + JSON API)
列表: GET https://www.huidong.gov.cn/hzhdzrzyj/gkmlpt/api/all/2831?page={N}&sid=752157
     ⚠️ page 参数是 5 的倍数(每次取5页, 100条); total=6535 (不动产/规划/环评公告)
     分类2831=业务工作, 子类含不动产登记2833/国土空间规划10955等
详情: API url 字段 + ".html" 后缀 (content/{a}/{b}/post_{id}.html)
     meta ArticleTitle (无 PubDate meta, 日期用 API date 时间戳) + div.article-content 正文
"""
import sys
import re
import time
import json
import ssl
import sqlite3
import datetime
import urllib.request
from urllib.parse import urljoin
from bs4 import BeautifulSoup

SITE_NAME = "惠东县自然资源局-业务工作"
SCRIPT_NAME = "crawl_huidong_zrzyj.py"
GROUP_NAME = "广东"
API_TPL = "https://www.huidong.gov.cn/hzhdzrzyj/gkmlpt/api/all/{classify}?page={page}&sid={sid}"
SID = "752157"
CLASSIFY = "2831"
DB_PATH = "/mnt/data/search.db"
JSONL_PATH = ""

# 日期过滤: 只抓 2026 年以来的(业务工作 6535 条历史, 全量太多)
MIN_DATE = datetime.date(2026, 1, 1)

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "application/json,text/html,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "https://www.huidong.gov.cn/hzhdzrzyj/gkmlpt/index",
}

# --pages 参数: 每个 API 调用取 5 页(100条), 默认 5 → 最多 500 条
_MAX_PAGES = 5
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

# JSONL 测试模式
for i, a in enumerate(sys.argv):
    if a == "--jsonl" and i + 1 < len(sys.argv):
        JSONL_PATH = sys.argv[i + 1]


def http_get(url, timeout=40, retries=5):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [WARN] 请求失败 {url}: {e}", file=sys.stderr)
                return ""
            time.sleep(2.5)
    return ""


def clean_title(title):
    """strip &middot;&nbsp; 实体前缀 和省略号截断后缀"""
    if not title:
        return ""
    title = title.replace("&middot;", "").replace("&nbsp;", "")
    title = title.replace("\u200b", "").replace("\ufeff", "")
    title = re.sub(r"^[\s\xa0·\u00b7]+", "", title)
    title = re.sub(r"\s*\.{3,}\s*$", "", title)
    return title.strip()


def fetch_batch(batch_idx):
    """第 batch_idx 批(每批=API一次调用, 取5页100条)"""
    page_param = (batch_idx + 1) * 5
    url = API_TPL.format(classify=CLASSIFY, page=page_param, sid=SID)
    text = http_get(url)
    if not text:
        return []
    try:
        d = json.loads(text)
    except Exception:
        return []
    items = []
    for a in d.get("articles", []):
        url = a.get("url", "")
        title = clean_title(a.get("title", ""))
        if not url or not title:
            continue
        # 详情 URL 补 .html 后缀
        if not url.endswith(".html"):
            url = url + ".html"
        ts = a.get("date") or a.get("first_publish_time") or 0
        date = ""
        if ts:
            try:
                date = datetime.datetime.fromtimestamp(ts).strftime("%Y-%m-%d")
            except Exception:
                date = ""
        items.append((url, title, date))
    return items


def extract_detail(html, page_url, list_title):
    title = ""
    mt = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"', html)
    if mt:
        title = clean_title(mt.group(1))
    if not title:
        title = list_title

    soup = BeautifulSoup(html, "html.parser")
    cm = soup.find("div", class_="article-content")
    if not cm:
        return title, "", ""
    content_html = "".join(str(c) for c in cm.contents)
    # 附件收集 (article-content 内)
    attachments = []
    for a in cm.find_all("a", href=True):
        href = a["href"]
        txt = a.get_text(strip=True)
        if (re.search(r"(?i)\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|ofd)(\?|$)", href)
                or "download" in href.lower() or "attach" in href.lower() or txt.startswith("附件")):
            abs_url = urljoin(page_url, href)
            if not abs_url.startswith(("http://", "https://")):
                abs_url = "http:" + abs_url if abs_url.startswith("//") else abs_url
            name = txt if txt else abs_url.split("/")[-1]
            attachments.append((a, name, abs_url))
    for a, name, abs_url in attachments:
        new_a = soup.new_tag("a", href=abs_url, target="_blank")
        new_a.string = name
        if a.parent is not None:
            p = a.parent
            idx_a = None
            for ci, child in enumerate(p.contents):
                if child is a:
                    idx_a = ci
                    break
            if idx_a is not None:
                for child in list(p.contents[:idx_a]):
                    if child.name is None and not child.strip():
                        child.extract()
                for child in list(p.contents[idx_a + 1:]):
                    if child.name is None and not child.strip():
                        child.extract()
            a.replace_with(new_a)
    return title, "", "".join(str(c) for c in cm.contents)


def html_to_text(content, page_url):
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<meta[^>]*>", "", content, flags=re.I)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)

    attachments = []
    link_protect = {}

    content = re.sub(r"【字号：[^】]*】", "", content)
    content = re.sub(r"发布(?:时间|日期)[:：][^<]{0,30}", "", content)

    soup = BeautifulSoup(content, "html.parser")
    # 附件已在 extract_detail 转为绝对 URL <a>, 这里收集并保护
    for a in soup.find_all("a", href=True):
        href = a["href"]
        txt = a.get_text(strip=True)
        if re.search(r"(?i)\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|ofd)(\?|$)", href) or "download" in href.lower():
            abs_url = href
            key = f"__ATTACH__{len(link_protect)}__"
            link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt or "附件"}</a>'
            a.replace_with(key)
    # 纯图片段落
    for p in soup.find_all("p"):
        imgs = p.find_all("img")
        if imgs and not p.get_text(strip=True):
            srcs = []
            for img in imgs:
                src = img.get("src") or img.get("oldsrc") or ""
                if src:
                    abs_url = urljoin(page_url, src)
                    srcs.append(f'<a href="{abs_url}" target="_blank">{src.split("/")[-1].split("?")[0] or "图片"}</a>')
            if srcs:
                key = f"__IMG__{len(link_protect)}__"
                link_protect[key] = "<p>" + "<br/>".join(srcs) + "</p>"
                p.replace_with(key)
    content = str(soup)

    def _link_repl(m):
        href = m.group(1)
        if href.startswith("javascript:"):
            return ""
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = txt.strip()
        if not txt:
            txt = "附件"
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<a href="{href}" target="_blank">{txt}</a>'
        return key
    content = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content, flags=re.S | re.I)

    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0
    content = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content, flags=re.S | re.I)

    content = re.sub(r"</p>", "</p>\n\n", content, flags=re.I)
    content = re.sub(r"<br\s*/?>", "\n", content, flags=re.I)
    content = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+|ucapcontent)\b[^>]*>", "", content, flags=re.I)

    parts = []
    for block in content.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        b = re.sub(r"(?<=>)\s*[\r\n\t]+\s*", "", b)
        b = re.sub(r"\s*[\r\n\t]+\s*(?=<)", "", b)
        b = b.replace("\r", "").replace("\t", " ")
        b = re.sub(r"[ \t]{2,}", " ", b)
        plain = re.sub(r"<[^>]+>", "", b)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)

    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)
    out = re.sub(r"__IMG__\d+__", "", out)
    out = re.sub(r"__ATTACH__\d+__", "", out)

    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    return out, has_table, attachments


def store_record(conn, url, title, content, date, has_table):
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (url, SITE_NAME))
    if cur.fetchone():
        return False
    try:
        industry = "other"
        try:
            from crawler_lib import classify_industry
            industry = classify_industry(title)
        except Exception:
            pass
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, industry),
        )
        if cur.rowcount > 0:
            summary = re.sub(r"<[^>]+>", "", content)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, SITE_NAME, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, has_table)
        raise


def main():
    conn = None
    if not JSONL_PATH:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute("PRAGMA busy_timeout=60000")
        conn.execute("PRAGMA journal_mode=WAL")

    new_count = 0
    skip_count = 0
    seen_urls = set()
    jsonl_rows = []

    for batch in range(_PAGES):
        items = fetch_batch(batch)
        print(f"Batch {batch+1}: got {len(items)} items", flush=True)
        if not items:
            print("Empty batch, stop", flush=True)
            break
        for abs_url, title, date in items:
            if abs_url in seen_urls:
                continue
            seen_urls.add(abs_url)
            # 日期过滤
            if date:
                try:
                    if datetime.date.fromisoformat(date) < MIN_DATE:
                        continue
                except Exception:
                    pass
            html = http_get(abs_url)
            if not html:
                skip_count += 1
                continue
            dtitle, _, content_html = extract_detail(html, abs_url, title)
            content, has_table, atts = html_to_text(content_html, abs_url)
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            if plain_len < 10:
                print(f"  skip empty: {dtitle[:40]}", flush=True)
                skip_count += 1
                continue
            if JSONL_PATH:
                jsonl_rows.append({
                    "page_url": abs_url, "title": dtitle, "content": content,
                    "publish_date": date, "has_table": has_table,
                    "site_name": SITE_NAME, "group_name": GROUP_NAME,
                    "script_name": SCRIPT_NAME,
                })
                new_count += 1
                continue
            ok = store_record(conn, abs_url, dtitle, content, date, has_table)
            if ok:
                new_count += 1
                print(f"  + {dtitle[:40]} ({date})", flush=True)
            else:
                skip_count += 1
            time.sleep(0.2)

    if JSONL_PATH:
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(json.dumps(row, ensure_ascii=False) + "\n")
        print(f"JSONL written: {JSONL_PATH} ({len(jsonl_rows)} rows)", flush=True)

    if conn is not None:
        conn.commit()
        conn.close()
    print(f"新增: {new_count}", flush=True)
    print(f"跳过: {skip_count}", flush=True)
    print(f"=== {SITE_NAME} done ===", flush=True)


if __name__ == "__main__":
    main()
