#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
郓城县人民政府 — 通知公告 (www.cnyc.gov.cn/ycx_list/?ch=通知公告)
列表: POST /els-service/article/{page}/15  JSON {"dq":"2c90808483d171790183e4d0219b0002","catas":["yc1590299564160716800"]}
      返回 data.contents[]: xxid/subject/fwdate/dwid/dwname; elementsTotal=2381, 159页 (2016-03~)
详情: http://www.cnyc.gov.cn/2c90808483d171790183e4d0219b0002/{dwid}/{xxid}.html
      正文: 内嵌 <script> var memo = "..." (unicode转义 HTML, 含附件 <a href="/upload-service/...">)
      标题: <title> | 日期: 列表 fwdate 最可靠
入库: crawler_lib.push_to_searchdb → /root/search.db (FTS触发器自动维护)
注意: 纯 curl 可解 (正文在初始 HTML 的 JS 字符串里, 无需 playwright)
"""
import sys, os, re, json, time, urllib.parse, random, urllib.request, ssl
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "郓城县人民政府-通知公告"
BASE = "http://www.cnyc.gov.cn"
DQ = "2c90808483d171790183e4d0219b0002"
CATA = "yc1590299564160716800"  # 通知公告
CUTOFF = "2016-01-01"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
CTX = ssl._create_unverified_context()

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1


def http_get(url, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers={"User-Agent": UA})
            resp = urllib.request.urlopen(req, timeout=30, context=CTX)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [HTTP-ERR] {url[:70]} : {str(e)[:50]}")
                return None
            time.sleep(2)
    return None


def http_post_json(url, payload, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, data=json.dumps(payload).encode("utf-8"),
                                         headers={"User-Agent": UA, "Content-Type": "application/json;charset=utf-8"})
            resp = urllib.request.urlopen(req, timeout=30, context=CTX)
            return json.loads(resp.read().decode("utf-8", errors="replace"))
        except Exception as e:
            if i == retries - 1:
                print(f"  [API-ERR] {url[:70]} : {str(e)[:50]}")
                return None
            time.sleep(2)
    return None


def list_items(page):
    """POST 栏目 API 拿一页列表"""
    url = f"{BASE}/els-service/article/{page}/15"
    data = http_post_json(url, {"dq": DQ, "catas": [CATA]})
    if not data or not data.get("success"):
        print(f"  [API-FAIL] page {page}")
        return []
    contents = data.get("data", {}).get("contents", []) or []
    items = []
    for c in contents:
        subject = re.sub(r"<[^>]+>", "", c.get("subject") or "").strip()
        xxid = c.get("xxid") or ""
        dwid = c.get("dwid") or ""
        fwdate = (c.get("fwdate") or "")[:10]
        if not xxid or not subject:
            continue
        url = c.get("url") or ""
        if not url:
            url = f"{BASE}/2c90808483d171790183e4d0219b0002/{dwid}/{xxid}.html"
        else:
            url = urllib.parse.urljoin(BASE, url)
        items.append({"url": url, "title": subject, "date": fwdate})
    return items


def extract_js_string(html_text, var_name):
    """从 JS 里提取 var xxx = "..." (unicode 转义 JSON 字符串)"""
    i = html_text.find(var_name)
    if i == -1:
        return None
    j = html_text.find('"', i)
    if j == -1:
        return None
    j += 1
    buf = []
    while j < len(html_text):
        c = html_text[j]
        if c == "\\":
            buf.append(html_text[j:j + 2])
            j += 2
            continue
        if c == '"':
            rest = html_text[j + 1:j + 6].lstrip()
            if rest.startswith(";") or rest.startswith(",") or rest.startswith(")") or rest.startswith("}") or rest == "" or rest[0] == "\n":
                break
            buf.append(c)
            j += 1
            continue
        buf.append(c)
        j += 1
    raw = "".join(buf)
    try:
        return json.loads('"' + raw + '"')
    except Exception:
        return raw


def parse_detail(html_text):
    """从详情页提取 标题/日期/正文(memo JS 字符串)"""
    title = ""
    m = re.search(r"<title>(.*?)</title>", html_text, re.S)
    if m:
        title = re.sub(r"\s+", " ", m.group(1)).strip()
    pub_date = ""
    m = re.search(r"name=['\"]PubDate['\"]\s+content=['\"]([^'\"]+)['\"]", html_text)
    if not m:
        m = re.search(r"content=['\"]([^'\"]+)['\"]\s+name=['\"]PubDate['\"]", html_text)
    if m:
        pub_date = m.group(1).strip()[:10]
    # 正文: var memo = "..." (unicode 转义 HTML)
    body = extract_js_string(html_text, "var memo")
    if body is None:
        # 兜底: ozoom 静态内容
        m = re.search(r'<div id="ozoom"[^>]*>(.*?)</div>\s*</div>\s*</div>\s*</div>', html_text, re.S)
        if m:
            body = m.group(1)
    if not body or not body.strip():
        return None
    return {"title": title, "pub_date": pub_date, "body": body}


def html_to_text(content, page_url):
    """正文 HTML → 存储格式: 附件绝对化 + 表格保留 + 段落化"""
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)
    content2 = content

    attachments = []
    link_protect = {}

    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = txt.strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        is_attach = re.search(r"\.(pdf|docx?|xlsx?|pptx?|zip|rar|wps|et|dps|txt|ofd)(\?|$)", abs_url, re.I)
        if is_attach:
            attachments.append((abs_url, txt))
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<p><a href="{abs_url}" target="_blank">{txt}</a></p>'
        return key

    content2 = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content2, flags=re.S | re.I)
    content2 = re.sub(r"<a\s[^>]*href='([^']+)'[^>]*>(.*?)</a>", _link_repl, content2, flags=re.S | re.I)

    def _img_repl(m):
        src = m.group(1)
        abs_src = urllib.parse.urljoin(page_url, src)
        if re.search(r"ueditor|editor|icon|logo|qrcode|ewm|btn", abs_src, re.I):
            return ""
        return f'<p><a href="{abs_src}" target="_blank">查看图片</a></p>'

    content2 = re.sub(r'<img[^>]*src="([^"]+)"[^>]*>', _img_repl, content2, flags=re.S | re.I)
    content2 = re.sub(r"<img[^>]*src='([^']+)'[^>]*>", _img_repl, content2, flags=re.S | re.I)

    table_protect = []

    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"

    content2 = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content2, flags=re.S | re.I)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0

    content2 = re.sub(r"</p>", "</p>\n\n", content2, flags=re.I)
    content2 = re.sub(r"<br\s*/?>", "\n", content2, flags=re.I)
    content2 = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content2, flags=re.I)

    parts = []
    for block in content2.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)
    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key or key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()
    return out, has_table, attachments


def run():
    print(f"🚀 {SITE_NAME} — 爬取 {_PAGES} 页")
    items_all = []
    for pg in range(1, _PAGES + 1):
        items = list_items(pg)
        print(f"  第{pg}页: {len(items)} 条")
        if not items:
            if pg > 1:
                break
            continue
        # 截断检测 (前5页内不豁免)
        cutoff_hit = False
        for it in items:
            if it["date"] and it["date"] < CUTOFF:
                if pg <= 5:
                    print(f"  [p{pg}] 有早于 {CUTOFF} 的记录(前5页不豁免, 仍抓)")
                else:
                    cutoff_hit = True
                    break
            items_all.append(it)
        if cutoff_hit:
            print(f"  [CUTOFF] 第{pg}页出现 < {CUTOFF}")
            break
        time.sleep(random.uniform(0.5, 1.2))

    print(f"  [LIST] 共 {len(items_all)} 条")
    if not items_all:
        return

    records = []
    for i, it in enumerate(items_all, 1):
        html_text = http_get(it["url"])
        if not html_text:
            print(f"  [{i}/{len(items_all)}] [ERR] {it['title'][:40]}")
            continue
        d = parse_detail(html_text)
        if not d:
            print(f"  [{i}/{len(items_all)}] [NO-BODY] {it['title'][:40]}")
            continue
        title = d["title"] or it["title"]
        pub_date = d["pub_date"] or it["date"]
        content, has_table, attachments = html_to_text(d["body"], it["url"])
        plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
        has_links = "<a href" in content
        if plain_len < 10 and not has_links:
            print(f"  [{i}/{len(items_all)}] [EMPTY] {title[:40]} 源站空正文, 跳过")
            continue
        records.append({
            "site_name": SITE_NAME, "title": title, "pub_date": pub_date,
            "url": it["url"], "content": content, "attachments": json.dumps(attachments, ensure_ascii=False),
        })
        print(f"  [{i}/{len(items_all)}] 最新: {title[:55]}...")
        time.sleep(random.uniform(0.3, 0.8))

    new_c = push_to_searchdb(records, batch_label="cnyc_tzgg")
    print(f"Done: {new_c} records")


if __name__ == "__main__":
    run()
