#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_jcx_tzgg.py - 江城哈尼族彝族自治县人民政府 通知公告

列表: https://www.jcx.gov.cn/xwzx/tzgg.htm
CMS: TRS vsb 静态列表（同族：望谟 line_u6_N；本站是 line_u9_N）
  分页: ⚠️ **倒序静态链接** —— 页脚 `<a href="tzgg/32.htm" class="Next">下页</a>`
        第 1 页 = /xwzx/tzgg.htm ；第 N 页(N≥2) = /xwzx/tzgg/{33-N+1}.htm（编号递减）
        实测：tzgg/1.htm=最老(2024-02, 第33页) / tzgg/32.htm=第2页 / tzgg/28.htm=第6页
        页脚另有「尾页」也带 class="Next" → **必须按链接文字「下页」取**，不能按 class 取第一个
        本脚本**跟随「下页」链接**逐页（不按数字猜，符合该 CMS 既有教训）
        页脚「共492条 1/33」可校验总页数
  条目: <li id="line_u9_N"><a href="../info/15974/526231.htm" title="完整标题">…</a>
        <b class="date">2026-09-24</b></li>   15 条/页
        ⚠️ 链接文字可能被站点截断（带 ...），**优先取 title 属性**
  详情: meta ArticleTitle / PubDate / ColumnName / ContentSource
        正文: div.article_detail（页面唯一）
        附件: 正文**内**的 ../../virtual_attach_file.vsb?afc=... （vsb 虚拟附件，无扩展名）
              ⚠️ 链接文字里夹 CMS 的「已下载 次」计数器 → 需清洗
        噪声: div.nextList（上一条/下一条）在正文容器外，天然不混入
"""
import sys, os, re, time, json, html as html_lib, urllib.parse
from datetime import datetime, timedelta

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

JSONL_PATH = os.environ.get("JSONL_PATH", "")
ENC = "utf-8"

SITE_NAME = "江城县人民政府-通知公告"
GROUP_NAME = "云南"
SCRIPT_NAME = "crawl_jcx_tzgg.py"
CATEGORY = "通知公告"
BASE_URL = "https://www.jcx.gov.cn"
LIST_URL = "https://www.jcx.gov.cn/xwzx/tzgg.htm"
COL_PATH = "/info/"
CUTOFF = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

_MAX_PAGES = 5
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1


def clean_title(t):
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"^\s*(?:&middot;|·|\u00b7)?\s*(?:&nbsp;|\u00a0)?\s*", "", t)
    t = re.sub(r"^[•·]\s*", "", t)
    t = re.sub(r"<[^>]+>", "", t)
    t = re.sub(r"\s+", " ", t)
    t = t.replace("\u201c", "“").replace("\u201d", "”")
    return t.strip()


def fetch(url, timeout=30):
    h = dict(HEADERS)
    last_err = None
    for attempt in range(3):
        try:
            import requests, urllib3
            urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
            r = requests.get(url, headers=h, timeout=timeout, verify=False)
            if r.status_code == 200:
                r.encoding = ENC
                return r.text
            last_err = "HTTP %s" % r.status_code
        except Exception as e:
            last_err = str(e)[:80]
        time.sleep(2)
    print("  [fetch fail] %s -> %s" % (url[:90], last_err), flush=True)
    return None


def parse_list(html_text):
    """li#line_u9_N → a[title] + b.date；优先 title 属性（链接文字可能被截断）"""
    items = []
    for m in re.finditer(r'<li[^>]*id="line_u9_\d+"[^>]*>(.*?)</li>', html_text, re.S | re.I):
        li = m.group(1)
        am = re.search(r'<a\b[^>]*href="([^"]+)"[^>]*>(.*?)</a>', li, re.S | re.I)
        if not am:
            continue
        href = html_lib.unescape(am.group(1)).strip()
        if COL_PATH not in href or not href.lower().endswith(".htm"):
            continue
        tm = re.search(r'title="([^"]*)"', am.group(0), re.I)
        title = clean_title(tm.group(1) if tm else am.group(2))
        dm = re.search(r'<b class="date">\s*(\d{4})-(\d{1,2})-(\d{1,2})', li, re.I)
        if not dm or not title or len(title) < 2:
            continue
        date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
        items.append((urllib.parse.urljoin(BASE_URL, href), title, date))
    return items


def next_page_url(html_text, cur_url):
    """跟随「下页」链接（⚠️「尾页」也是 class=Next，必须按链接文字判断）"""
    for m in re.finditer(r'<a\b[^>]*href="([^"]+)"[^>]*>\s*下页\s*</a>', html_text, re.I):
        return urllib.parse.urljoin(cur_url, html_lib.unescape(m.group(1)).strip())
    return None


def html_to_text(content, page_url):
    """正文 HTML 清洗：保留 <p>/<table>，附件转 <p><a>绝对URL</a></p> 段落"""
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<meta[^>]*>", "", content, flags=re.I)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)

    attachments = []
    link_protect = {}

    content = re.sub(r"【字号：[^】]*】", "", content)

    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for a in _soup.find_all("a", href=True):
                href = a["href"]
                if (a.get("appendix") or a.get("data-appendix")
                        or re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps|et|ofd)", href, re.I)
                        or "virtual_attach_file" in href or "/module/download/" in href):
                    _href = (href or "").strip()
                    _oldsrc = (a.get("oldsrc") or a.get("OLDSRC") or "").strip()
                    if _href and not re.match(r"^(?:javascript:|mailto:|tel:|#)", _href, re.I):
                        real_href = _href
                    else:
                        real_href = _oldsrc or _href
                    txt = (a.get_text(strip=True) or a.get("_title") or a.get("title")
                           or os.path.basename(real_href.split("?")[0]) or "附件")
                    abs_url = urllib.parse.urljoin(page_url, real_href)
                    attachments.append((abs_url, txt))
                    key = "__ATTACH__%d__" % len(link_protect)
                    link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
                    a.replace_with(key)
            for img in _soup.find_all("img"):
                src = (img.get("src") or img.get("oldsrc") or "")
                if re.search(r"(icons?/|fileTypeImages/icon_)", src, re.I):
                    img.decompose()
            for p in _soup.find_all("p"):
                imgs = p.find_all("img")
                if imgs and not p.get_text(strip=True):
                    srcs = []
                    for img in imgs:
                        src = img.get("src") or img.get("oldsrc") or ""
                        if not src:
                            continue
                        abs_url = urllib.parse.urljoin(page_url, src)
                        txt = (img.get("alt") or img.get("title") or "").strip() \
                            or os.path.basename(src.split("?")[0]) or "图片"
                        srcs.append('<a href="%s" target="_blank">%s</a>' % (abs_url, txt))
                    if srcs:
                        key = "__IMG__%d__" % len(link_protect)
                        link_protect[key] = "<p>" + "<br/>".join(srcs) + "</p>"
                        p.replace_with(key)
            for img in _soup.find_all("img"):
                src = img.get("src") or img.get("oldsrc") or ""
                if not src:
                    continue
                abs_url = urllib.parse.urljoin(page_url, src)
                txt = (img.get("alt") or img.get("title") or "").strip() \
                    or os.path.basename(src.split("?")[0]) or "图片"
                key = "__IMG__%d__" % len(link_protect)
                link_protect[key] = '<p><a href="%s" target="_blank">%s</a></p>' % (abs_url, txt)
                img.replace_with(key)
            content = str(_soup)
        except Exception:
            pass

    def _link_repl(m):
        href = m.group(1)
        if re.match(r"^(?:javascript:|mailto:|tel:|s?ftp:|#)", href, re.I):
            return "" if href.lower().startswith("javascript:") else m.group(0)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        attachments.append((abs_url, txt))
        key = "__LINK__%d__" % len(link_protect)
        link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
        return key
    content = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl,
                     content, flags=re.S | re.I)

    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for tag in _soup.find_all(True):
                for attr in ("style", "class", "lang", "dir", "align", "valign", "width",
                             "height", "border", "cellpadding", "cellspacing"):
                    tag.attrs.pop(attr, None)
            for d in _soup.find_all("div"):
                # 叶子 div（不含 div/table 子块）= 一个段落（安徽 showList 系等用 <div> 分段）
                #   → 改名 <p> 保住分段；否则 unwrap 会丢段落边界，正文糊成一坨
                if d.find("div") is None and d.find("table") is None:
                    if not d.get_text(strip=True) and not d.find("img") and not d.find("a"):
                        d.decompose()
                    else:
                        d.name = "p"
                elif not d.get_text(strip=True) and not d.find("table") \
                        and not d.find("img") and not d.find("a"):
                    d.decompose()
                else:
                    d.unwrap()
            content = str(_soup)
        except Exception:
            pass

    table_protect = []

    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return "__TBL__%d__" % (len(table_protect) - 1)

    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0
    prev = None
    while prev != content:
        prev = content
        content = re.sub(r"<table(?:(?!<table).)*?</table>", _tbl_repl,
                         content, flags=re.S | re.I)

    # 附件列表 <ul><li>…</li></ul> → 每项独立段落
    #   （用户规范：多附件独立成段；且 <ul> 塞进 <p> 是非法嵌套）
    content = re.sub(r"<li\b[^>]*>", "<p>", content, flags=re.I)
    content = re.sub(r"</li\s*>", "</p>\n\n", content, flags=re.I)
    content = re.sub(r"</?(?:ul|ol)\b[^>]*>", "\n\n", content, flags=re.I)
    # 去重复/嵌套标签（内层 <p> 常带 style 等属性，须匹配 <p ...>）
    for _ in range(3):
        content = re.sub(r"<p\b[^>]*>\s*<p\b[^>]*>", "<p>", content, flags=re.I)
        content = re.sub(r"</p\s*>\s*</p\s*>", "</p>", content, flags=re.I)
    content = re.sub(r"<p\b[^>]*>\s*(</p\s*>)", r"\1", content, flags=re.I)
    content = re.sub(r"<p\b[^>]*>\s*$", "", content, flags=re.I)

    content = re.sub(r"</p>", "</p>\n\n", content, flags=re.I)
    content = re.sub(r"<br\s*/?>", "\n", content, flags=re.I)
    content = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+|strong|em|b|i|u)\b[^>]*>", "",
                     content, flags=re.I)

    parts = []
    for block in content.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        b = re.sub(r"(?<=>)\s*[\r\n\t]+\s*", "", b)
        b = re.sub(r"\s*[\r\n\t]+\s*(?=<)", "", b)
        b = b.replace("\r", "").replace("\t", " ")
        b = re.sub(r"[ \t]{2,}", " ", b)
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        if re.match(r"^[【\[（(]?\s*(?:返回顶部|回到顶部|返回列表|打印(?:本页)?|关闭(?:窗口)?|分享|扫一扫|上一篇|下一篇|顶部)\s*[】\]）)]?$", plain):
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace("__TBL__%d__" % i, tbl)
        # 未被 <p> 包裹、且不是块级内容 → 一律补 <p>（否则裸文本会被浏览器折叠掉换行）
        if (not re.search(r"<p[ >]", b, re.I)
                and not re.search(r"<(?:table|ul|ol|h[1-6])\b", b, re.I)):
            b = "<p>%s</p>" % b
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace("__TBL__%d__" % i, tbl)

    out = re.sub(r"__(?:LINK|TBL|IMG|ATTACH)__\d+__", "", out)
    out = html_lib.unescape(out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    # 站点特有清洗：vsb 附件区的「已下载 次」计数器 + 「附件【…】」包裹
    out = re.sub(r"已下载\s*\d*\s*次", "", out)
    out = re.sub(r"附件【\s*(<a\b.*?</a>)\s*】", r"<p>\1</p>", out, flags=re.S | re.I)
    out = re.sub(r"附件【\s*】", "", out)
    out = re.sub(r"<p>\s*</p>", "", out)
    out = re.sub(r"[ \t]{2,}", " ", out)
    out = re.sub(r"\n{3,}", "\n\n", out)

    # 最终兜底：源站存在 <p><ul><li> 非法嵌套，BeautifulSoup 解析后会残留空 <p></p>
    #   紧贴内容 → 在这里统一归零（只删冗余/嵌套/空标签，不动内容）
    for _ in range(3):
        out = re.sub(r"<p\b[^>]*>\s*<p\b[^>]*>", "<p>", out, flags=re.I)
        out = re.sub(r"</p\s*>\s*</p\s*>", "</p>", out, flags=re.I)
    out = re.sub(r"<p\b[^>]*>\s*</p\s*>", "", out, flags=re.I)
    out = re.sub(r"<p>\s*\n", "<p>", out, flags=re.I)

    # 最终兜底：去嵌套/空的 <p>（源站 <div><div> 常见的产物）
    for _ in range(3):
        out = re.sub(r"<p\b[^>]*>\s*<p\b[^>]*>", "<p>", out, flags=re.I)
        out = re.sub(r"</p\s*>\s*</p\s*>", "</p>", out, flags=re.I)
    out = re.sub(r"<p\b[^>]*>\s*</p\s*>", "", out, flags=re.I)

    seen, final = set(), []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key or key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    dedup, seen_u = [], set()
    for u, t in attachments:
        if u in seen_u:
            continue
        seen_u.add(u)
        dedup.append((u, t))

    return out, has_table, dedup


def parse_detail(html_text, page_url):
    title = date = content_html = ""
    soup = None
    if BeautifulSoup:
        try:
            soup = BeautifulSoup(html_text, "html.parser")
            mt = soup.find("meta", attrs={"name": re.compile(r"ArticleTitle", re.I)})
            if mt and mt.get("content"):
                title = clean_title(mt["content"])
            md = soup.find("meta", attrs={"name": re.compile(r"PubDate", re.I)})
            if md and md.get("content"):
                dm = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", md["content"])
                if dm:
                    date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
            if not title:
                h1 = soup.find("h1")
                if h1:
                    title = clean_title(h1.get_text(" ", strip=True))
            if not title:
                tt = soup.find("title")
                if tt:
                    t = clean_title(tt.get_text(strip=True))
                    title = re.sub(r"[-—|]\s*江城哈尼族彝族自治县人民政府\s*$", "", t).strip() or t
            el = soup.select_one("div.article_detail")
            if el is None:
                el = soup.select_one("div.vsbcontent") or soup.select_one("#vsb_content")
            if el is not None:
                content_html = "".join(str(c) for c in el.contents)
        except Exception:
            soup = None
    if not date:
        dm = re.search(r"(\d{4})[-年/](\d{1,2})[-月/](\d{1,2})", html_text)
        if dm:
            date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
    return title, date, content_html


def main():
    sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
    from crawler_lib import push_to_searchdb

    print("=== %s (%s) pages=%d cutoff=%s ===" % (SITE_NAME, SCRIPT_NAME, _PAGES, CUTOFF),
          flush=True)

    new_count = skip_count = cutoff_count = 0
    seen_urls = set()
    jsonl_rows = []
    batch = []

    url = LIST_URL
    for page in range(1, _PAGES + 1):
        html_text = fetch(url)
        if not html_text:
            print("Page %d fetch error, stop" % page, flush=True)
            break
        items = parse_list(html_text)
        if not items:
            print("Page %d: empty, stop" % page, flush=True)
            break
        nav = re.search(r"共(\d+)条\s+(\d+)/(\d+)", html_text)
        print("Page %d: found %d items%s [%s]" % (
            page, len(items),
            (" 页码%s/%s 共%s条" % (nav.group(2), nav.group(3), nav.group(1))) if nav else "",
            url.replace(BASE_URL, "")), flush=True)

        for abs_url, title, date in items:
            if abs_url in seen_urls:
                continue
            seen_urls.add(abs_url)
            if date and date < CUTOFF:
                cutoff_count += 1
                continue
            dhtml = fetch(abs_url)
            if not dhtml:
                skip_count += 1
                continue
            dtitle, ddate, content_html = parse_detail(dhtml, abs_url)
            if not dtitle:
                dtitle = title
            if not ddate:
                ddate = date
            content, has_table, atts = html_to_text(content_html, abs_url)
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            no_para = (not re.search(r"<p[ >]", content, re.I)
                       and not re.search(r"<a[ >]", content, re.I) and not has_table)
            if plain_len < 5 or no_para:
                print("  skip empty/placeholder: %s" % dtitle[:40], flush=True)
                skip_count += 1
                continue
            item = {
                "site_name": SITE_NAME, "source_url": abs_url, "url": abs_url,
                "title": dtitle, "pub_date": ddate, "summary": "",
                "content": content, "category": CATEGORY, "tags": "",
                "group_name": GROUP_NAME,
                "attachments": json.dumps(
                    [{"title": t, "url": u} for u, t in atts], ensure_ascii=False) if atts else "",
            }
            if JSONL_PATH:
                jsonl_rows.append(item)
            else:
                batch.append(item)
            new_count += 1
            print("  + %s (%s, %d字%s%s)" % (dtitle[:36], ddate, plain_len,
                                             ", 表格" if has_table else "",
                                             ", 附件%d" % len(atts) if atts else ""), flush=True)
            time.sleep(0.2)
        if not JSONL_PATH and batch:
            push_to_searchdb(batch, batch_label=SCRIPT_NAME)
            batch = []

        nx = next_page_url(html_text, url)
        if not nx or nx in seen_urls:
            print("  → 无下页，停止", flush=True)
            break
        url = nx
        time.sleep(0.3)

    if JSONL_PATH:
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(json.dumps(row, ensure_ascii=False) + "\n")
        print("JSONL written: %s (%d rows)" % (JSONL_PATH, len(jsonl_rows)), flush=True)
    else:
        if batch:
            push_to_searchdb(batch, batch_label=SCRIPT_NAME)

    print("新增: %d | 跳过: %d | CUTOFF截断: %d" % (new_count, skip_count, cutoff_count), flush=True)
    print("=== %s done ===" % SITE_NAME, flush=True)


if __name__ == "__main__":
    main()
