#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_shanggao.py - 上高县-建设项目环境影响评价 (江西宜春)
https://www.shanggao.gov.cn/sgxrmzf/jsxmhjyxpj/pc/list.html

CMS: UCAP/政务公开站群 + JSON 检索接口
列表+正文: POST https://www.shanggao.gov.cn/searchManuscript
   body: {"current":N,"pageSize":20,"channelTreeIds":["<cid>"]}
   resp: {"data":{"total":387,"results":[{title,pubDate,id,urls,content:{content:HTML}}]}}
   => 正文 HTML 内联，**无需抓详情页**
⚠️ http→https 301：旧脚本用 http + curl -s 不跟跳转 → JSON 解析失败 → 每天静默 0 条、
   DB 停更到 2026-08-19（2026-09-11 重写修复）
⚠️ channelTreeIds 传多个是【无去重并集】(767 = 380 + 387)，且排序按日期混排 →
   URL 会落在旧栏目。故改为**按栏目分别抓取**：用户栏目 jsxmhjyxpj 优先，
   旧镜像栏目 jsxmhjyjpj2 补其独有的 16 条，按标题全局去重。
   jsxmhjyxpj   1996848032669868032 (387) ← 用户指定
   jsxmhjyjpj2  1996849911248297984 (380) ← 镜像旧栏目
"""
import sys, os, re, json, time, html as html_lib, urllib.parse
from datetime import datetime, timedelta

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

JSONL_PATH = os.environ.get("JSONL_PATH", "")

SITE_NAME = "上高环评公示"
GROUP_NAME = "江西"
SCRIPT_NAME = "crawl_shanggao.py"
CATEGORY = "环评公示"
BASE_URL = "https://www.shanggao.gov.cn"
API_URL = BASE_URL + "/searchManuscript"
LIST_URL = BASE_URL + "/sgxrmzf/jsxmhjyxpj/pc/list.html"
# 按顺序抓取；靠前的栏目 URL 优先（标题去重时先到先得）
CHANNELS = [
    ("1996848032669868032", "jsxmhjyxpj"),    # 用户指定栏目
    ("1996849911248297984", "jsxmhjyjpj2"),   # 镜像旧栏目（补独有条目）
]
CUTOFF = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Content-Type": "application/json",
    "Accept": "application/json, text/javascript, */*; q=0.01",
    "Referer": LIST_URL,
}

_MAX_PAGES = 5
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 5


def clean_title(t):
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"^\s*(?:&middot;|·|\u00b7)?\s*(?:&nbsp;|\u00a0)?\s*", "", t)
    t = re.sub(r"^[•·]\s*", "", t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch(url, timeout=30):
    h = dict(HEADERS)
    last_err = None
    for attempt in range(3):
        try:
            import requests, urllib3
            urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
            r = requests.get(url, headers=h, timeout=timeout, verify=False)
            if r.status_code == 200:
                r.encoding = ENC
                return r.text
            last_err = "HTTP %s" % r.status_code
        except Exception as e:
            last_err = str(e)[:80]
        time.sleep(2)
    print("  [fetch fail] %s -> %s" % (url[:90], last_err), flush=True)
    return None


def html_to_text(content, page_url):
    """正文 HTML 清洗：保留 <p>/<table>，附件转 <p><a>绝对URL</a></p> 段落"""
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<meta[^>]*>", "", content, flags=re.I)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)

    attachments = []
    link_protect = {}

    content = re.sub(r"【字号：[^】]*】", "", content)

    # ① 附件/图片 <a> 保护（在表格保护之前，防表格正则吞掉紧跟的附件链接）
    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for a in _soup.find_all("a", href=True):
                href = a["href"]
                if (a.get("appendix") or a.get("data-appendix")
                        or re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps|et|ofd)", href, re.I)
                        or "/module/download/" in href):
                    real_href = a.get("oldsrc") or href
                    txt = (a.get_text(strip=True) or a.get("_title") or a.get("title")
                           or os.path.basename(real_href.split("?")[0]) or "附件")
                    abs_url = urllib.parse.urljoin(page_url, real_href)
                    attachments.append((abs_url, txt))
                    key = "__ATTACH__%d__" % len(link_protect)
                    link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
                    a.replace_with(key)
            # 附件图标 img（acrobat.png / fileTypeImages/icon_*.gif）清理
            for img in _soup.find_all("img"):
                src = (img.get("src") or img.get("oldsrc") or "")
                if re.search(r"(icons?/|fileTypeImages/icon_)", src, re.I):
                    img.decompose()
            # 纯图片段落 → 图片链接段
            for p in _soup.find_all("p"):
                imgs = p.find_all("img")
                if imgs and not p.get_text(strip=True):
                    srcs = []
                    for img in imgs:
                        src = img.get("src") or img.get("oldsrc") or ""
                        if not src:
                            continue
                        abs_url = urllib.parse.urljoin(page_url, src)
                        txt = (img.get("alt") or img.get("title") or "").strip() \
                            or os.path.basename(src.split("?")[0]) or "图片"
                        srcs.append('<a href="%s" target="_blank">%s</a>' % (abs_url, txt))
                    if srcs:
                        key = "__IMG__%d__" % len(link_protect)
                        link_protect[key] = "<p>" + "<br/>".join(srcs) + "</p>"
                        p.replace_with(key)
            # 其余裸 img（正文插图）→ <p><a>图</a></p>
            for img in _soup.find_all("img"):
                src = img.get("src") or img.get("oldsrc") or ""
                if not src:
                    continue
                abs_url = urllib.parse.urljoin(page_url, src)
                txt = (img.get("alt") or img.get("title") or "").strip() \
                    or os.path.basename(src.split("?")[0]) or "图片"
                key = "__IMG__%d__" % len(link_protect)
                link_protect[key] = '<p><a href="%s" target="_blank">%s</a></p>' % (abs_url, txt)
                img.replace_with(key)
            content = str(_soup)
        except Exception:
            pass

    # ② 残余 <a>（相对路径/无属性）统一绝对化并入 attachments
    #    ⚠️ mailto:/tel:/javascript:/#锚点 **不是附件** — 直接原样保留（正文里「邮箱：xxx@163.com」
    #    这类链接若当附件处理会污染 attachments 列 + 生成假附件段，122/248 行中招）
    def _link_repl(m):
        href = m.group(1)
        if re.match(r"^(?:javascript:|mailto:|tel:|s?ftp:|#)", href, re.I):
            return "" if href.lower().startswith("javascript:") else m.group(0)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        attachments.append((abs_url, txt))
        key = "__LINK__%d__" % len(link_protect)
        link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
        return key
    content = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl,
                     content, flags=re.S | re.I)

    # ③ 清属性 + 解 div / 删空 div
    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for tag in _soup.find_all(True):
                for attr in ("style", "class", "lang", "dir", "align", "valign", "width",
                             "height", "border", "cellpadding", "cellspacing",
                             "data-mce-style", "data-mce-href", "data-mce-src"):
                    tag.attrs.pop(attr, None)
                for _a in [k for k in list(tag.attrs) if k.startswith("data-")]:
                    tag.attrs.pop(_a, None)
            for d in _soup.find_all("div"):
                if not d.get_text(strip=True) and not d.find("table") \
                        and not d.find("img") and not d.find("a"):
                    d.decompose()
                else:
                    d.unwrap()
            content = str(_soup)
        except Exception:
            pass

    # ④ 表格保护（支持嵌套：由内向外剥，占位符还原时天然嵌套正确）
    table_protect = []

    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return "__TBL__%d__" % (len(table_protect) - 1)

    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0
    prev = None
    while prev != content:
        prev = content
        content = re.sub(r"<table(?:(?!<table).)*?</table>", _tbl_repl,
                         content, flags=re.S | re.I)

    # ⑤ 分段
    content = re.sub(r"</p>", "</p>\n\n", content, flags=re.I)
    content = re.sub(r"<br\s*/?>", "\n", content, flags=re.I)
    content = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+|strong|em|b|i|u)\b[^>]*>", "",
                     content, flags=re.I)

    parts = []
    for block in content.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        b = re.sub(r"(?<=>)\s*[\r\n\t]+\s*", "", b)
        b = re.sub(r"\s*[\r\n\t]+\s*(?=<)", "", b)
        b = b.replace("\r", "").replace("\t", " ")
        b = re.sub(r"[ \t]{2,}", " ", b)
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace("__TBL__%d__" % i, tbl)
        # 裸链接块 → 包 <p>
        if not re.search(r"<p[ >]", b, re.I) and re.search(r"<a[ >]", b, re.I):
            b = "<p>%s</p>" % b
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace("__TBL__%d__" % i, tbl)

    out = re.sub(r"__(?:LINK|TBL|IMG|ATTACH)__\d+__", "", out)
    out = html_lib.unescape(out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    # 段落去重
    seen, final = set(), []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key or key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    # 附件去重（同 URL 只留一次）
    dedup, seen_u = [], set()
    for u, t in attachments:
        if u in seen_u:
            continue
        seen_u.add(u)
        dedup.append((u, t))

    return out, has_table, dedup

def fetch_list(cid, page):
    body = json.dumps({"current": page, "pageSize": 20, "channelTreeIds": [cid]})
    last = None
    for attempt in range(3):
        try:
            import requests, urllib3
            urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
            r = requests.post(API_URL, data=body.encode("utf-8"), headers=HEADERS,
                              timeout=40, verify=False, allow_redirects=True)
            if r.status_code == 200:
                d = r.json().get("data", {})
                return d.get("results") or [], d.get("total")
            last = "HTTP %s" % r.status_code
        except Exception as e:
            last = str(e)[:80]
        time.sleep(2)
    print("  [list fail] cid=%s page %d -> %s" % (cid, page, last), flush=True)
    return [], None


def parse_detail(it, page_url):
    """正文内联在 content.content，无需抓详情页"""
    title = clean_title(it.get("title", ""))
    date = ""
    dm = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", str(it.get("pubDate") or ""))
    if dm:
        date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
    content_html = ""
    c = it.get("content")
    if isinstance(c, dict):
        content_html = c.get("content") or ""
    elif isinstance(c, str):
        try:
            content_html = (json.loads(c) or {}).get("content") or ""
        except Exception:
            content_html = c
    return title, date, content_html


def article_url(it, col_code):
    urls = it.get("urls")
    if isinstance(urls, str):
        try:
            urls = json.loads(urls)
        except Exception:
            urls = {}
    if isinstance(urls, dict) and urls.get("pc"):
        return urllib.parse.urljoin(BASE_URL, urls["pc"])
    aid = it.get("id", "")
    return "%s/sgxrmzf/%s/pc/content/%s/content_%s.html" % (BASE_URL, col_code, aid, aid)


def main():
    sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
    from crawler_lib import push_to_searchdb

    print("=== %s (%s) pages=%d/channel cutoff=%s ===" % (SITE_NAME, SCRIPT_NAME, _PAGES, CUTOFF),
          flush=True)

    seen_titles, seen_urls = set(), set()
    new_count = skip_count = cutoff_count = dup_count = 0
    jsonl_rows, batch = [], []

    for cid, col in CHANNELS:
        for page in range(1, _PAGES + 1):
            try:
                results, tot = fetch_list(cid, page)
            except Exception as e:
                print("  fetch error cid=%s p%d: %s" % (cid, page, e), flush=True)
                break
            if not results:
                print("  %s p%d: empty, stop" % (col, page), flush=True)
                break
            print("  %s p%d: %d items (total=%s)" % (col, page, len(results), tot), flush=True)

            for it in results:
                title = clean_title(it.get("title", ""))
                key = re.sub(r"\s+", "", title)
                if not key or len(title) < 2:
                    skip_count += 1
                    continue
                if key in seen_titles:
                    dup_count += 1
                    continue
                seen_titles.add(key)

                pubdate = str(it.get("pubDate") or "")[:10]
                if pubdate and pubdate < CUTOFF:
                    cutoff_count += 1
                    continue

                abs_url = article_url(it, col)
                if abs_url in seen_urls:
                    dup_count += 1
                    continue
                seen_urls.add(abs_url)

                dtitle, ddate, content_html = parse_detail(it, abs_url)
                if not dtitle:
                    dtitle = title
                if not ddate:
                    ddate = pubdate
                content, has_table, atts = html_to_text(content_html, abs_url)
                plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
                no_para = (not re.search(r"<p[ >]", content, re.I)
                           and not re.search(r"<a[ >]", content, re.I) and not has_table)
                if plain_len < 5 or no_para:
                    print("  skip empty/placeholder: %s" % dtitle[:40], flush=True)
                    skip_count += 1
                    continue

                item = {
                    "site_name": SITE_NAME, "source_url": abs_url, "url": abs_url,
                    "title": dtitle, "pub_date": ddate, "summary": "",
                    "content": content, "category": CATEGORY, "tags": "",
                    "group_name": GROUP_NAME,
                    "attachments": json.dumps([{"title": t, "url": u} for u, t in atts],
                                              ensure_ascii=False) if atts else "",
                }
                if JSONL_PATH:
                    jsonl_rows.append(item)
                else:
                    batch.append(item)
                new_count += 1
                print("  + %s (%s, %d字%s)" % (dtitle[:38], ddate, plain_len,
                                               ", 表格" if has_table else ""), flush=True)
                time.sleep(0.15)
            if not JSONL_PATH and batch:
                push_to_searchdb(batch, batch_label=SCRIPT_NAME)
                batch = []
            time.sleep(0.25)
        print("  -- 栏目 %s 完成, 累计新增 %d --" % (col, new_count), flush=True)

    if JSONL_PATH:
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(json.dumps(row, ensure_ascii=False) + "\n")
        print("JSONL written: %s (%d rows)" % (JSONL_PATH, len(jsonl_rows)), flush=True)
    else:
        if batch:
            push_to_searchdb(batch, batch_label=SCRIPT_NAME)

    print("新增: %d | 跳过: %d | CUTOFF截断: %d | 标题重复(含跨栏目): %d"
          % (new_count, skip_count, cutoff_count, dup_count), flush=True)
    print("=== %s done ===" % SITE_NAME, flush=True)


if __name__ == "__main__":
    main()
