#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_dongtai_hjsp.py - 东台市 (www.dongtai.gov.cn)
https://www.dongtai.gov.cn/col/col7952/index.html (环评公示)
https://www.dongtai.gov.cn/col/col33593/index.html (重大建设项目)
CMS: Hanweb jpage dataproxy (GET)
列表: GET /module/web/jpage/dataproxy.jsp?page=N&col=1&appid=1&webid=49&path=/&columnid={cid}&sourceContentType=1&unitid=27566&webname=东台市人民政府&permissiontype=0
  响应 XML <datastore><totalrecord>236</totalrecord>...<record><![CDATA[<li><a href="/art/2026/8/5/art_7952_ID.html">标题</a><span>08-05</span></li>]]></record>
详情: /art/YYYY/M/D/art_{cid}_ID.html, meta ArticleTitle/PubDate, 正文 div.detail_texts div.content
坑: 加速乐(365cyd) WAF — 详情请求带 Referer 必 403, 必须无 Referer 访问

栏目 (--col):
  hjsp : 建设项目环评公示 columnid=7952 (total=236)
  zdjs : 重大建设项目 columnid=33593 (total=45)
"""
import sys, os, re, time, json, html as html_lib, urllib.parse

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

DB_PATH = os.environ.get("DB_PATH", "/mnt/data/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")
ENC = "utf-8"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "X-Requested-With": "XMLHttpRequest",
}

COL_MAP = {
    "hjsp": {"columnid": "7952", "name": "东台市-建设项目环评公示", "cat": "环评公示", "url": "https://www.dongtai.gov.cn/col/col7952/index.html"},
    "zdjs": {"columnid": "33593", "name": "东台市-重大建设项目", "cat": "重大建设项目", "url": "https://www.dongtai.gov.cn/col/col33593/index.html"},
}

_ACTIVE_COL = "hjsp"
for i, a in enumerate(sys.argv):
    if a == "--col" and i + 1 < len(sys.argv):
        _ACTIVE_COL = sys.argv[i + 1]
    elif a.startswith("--col="):
        _ACTIVE_COL = a.split("=", 1)[1]
if _ACTIVE_COL not in COL_MAP:
    print("未知栏目 %s, 可选: %s" % (_ACTIVE_COL, list(COL_MAP.keys())), flush=True)
    sys.exit(1)

SITE_NAME = COL_MAP[_ACTIVE_COL]["name"]
GROUP_NAME = "江苏"
SCRIPT_NAME = "crawl_dongtai_hjsp.py"
CATEGORY = COL_MAP[_ACTIVE_COL]["cat"]
BASE_URL = "https://www.dongtai.gov.cn"
LIST_URL = COL_MAP[_ACTIVE_COL]["url"]
PROXY_URL = "https://www.dongtai.gov.cn/module/web/jpage/dataproxy.jsp"
JQ_PARAMS = {
    "col": "1",
    "appid": "1",
    "webid": "49",
    "path": "/",
    "columnid": COL_MAP[_ACTIVE_COL]["columnid"],
    "sourceContentType": "1",
    "unitid": "27566",
    "webname": "东台市人民政府",
    "permissiontype": "0",
}
PERPAGE = 15

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass

_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1


def clean_title(t):
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"^\s*(?:&middot;|·|\u00b7)?\s*(?:&nbsp;|\u00a0)?\s*", "", t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch(url, referer=None, timeout=30):
    """referer=None 时不带 Referer 头（东台加速乐要求详情页无 Referer）"""
    h = dict(HEADERS)
    if referer:
        h["Referer"] = referer
    last_err = None
    for attempt in range(3):
        try:
            import requests, urllib3
            urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
            r = requests.get(url, headers=h, timeout=timeout, verify=False)
            if r.status_code == 200:
                r.encoding = ENC
                txt = r.text
                # 加速乐挑战页: notice-jiasule / 访问过于频繁 特征 (200 但无正文) → 重试
                if ("notice-jiasule" in txt or "访问过于频繁" in txt or "请稍候再试" in txt) and "detail_texts" not in txt:
                    last_err = "jiasule challenge (%dB)" % len(txt)
                    time.sleep(3 + attempt * 3)
                    continue
                return txt
            last_err = "HTTP %s" % r.status_code
        except Exception as e:
            last_err = str(e)[:80]
        time.sleep(2)
    print("  [fetch fail] %s -> %s" % (url[:90], last_err), flush=True)
    return None


def html_to_text(content, page_url):
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<meta[^>]*>", "", content, flags=re.I)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)

    # 清理详情页头部残留: <!--xxgktypeinfo--> 注释 + div.head (h1 标题 + 发布日期/来源) + 裸文本
    content = re.sub(r"<!--\s*xxgktypeinfo\s*-->", "", content, flags=re.I)
    content = re.sub(r"<div[^>]*class=\"?head\"?[^>]*>.*?</div>", "", content, flags=re.S | re.I)
    content = re.sub(r"(?m)^\s*xxgktypeinfo\s*$", "", content)
    content = re.sub(r"xxgktypeinfo\s*", "", content, count=1)

    attachments = []
    link_protect = {}

    content = re.sub(r"【字号：[^】]*】", "", content)
    content = re.sub(r"发布(?:时间|日期)[:：][^<]{0,30}", "", content)
    # 删除关联阅读/相关文章尾部
    content = re.sub(r"<h3[^>]*>\s*关联阅读[:：]?\s*</h3>", "", content, flags=re.I)
    content = re.sub(r"(?m)^\s*关联阅读[:：]?\s*$", "", content)

    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for a in _soup.find_all("a", href=True):
                if a.get("appendix") or a.get("data-appendix") or re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps|et|ofd)", a.get("href", ""), re.I):
                    href = a["href"]
                    real_href = a.get("oldsrc") or href
                    txt = a.get_text(strip=True) or a.get("_title") or a.get("title") or os.path.basename(real_href.split("?")[0])
                    abs_url = urllib.parse.urljoin(page_url, real_href)
                    attachments.append((abs_url, txt))
                    key = "__ATTACH__%d__" % len(link_protect)
                    link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
                    a.replace_with(key)
            for img in _soup.find_all("img"):
                src = (img.get("src") or img.get("oldsrc") or "")
                if "fileTypeImages/icon_" in src.lower() or "filetypeimages" in src.lower():
                    img.decompose()
            for p in _soup.find_all("p"):
                imgs = p.find_all("img")
                if imgs and not p.get_text(strip=True):
                    srcs = []
                    for img in imgs:
                        src = img.get("src") or img.get("oldsrc") or ""
                        if src:
                            abs_url = urllib.parse.urljoin(page_url, src)
                            txt = (img.get("alt") or img.get("title") or "").strip() or os.path.basename(src.split("?")[0]) or "图片"
                            srcs.append('<a href="%s" target="_blank">%s</a>' % (abs_url, txt))
                    if srcs:
                        key = "__IMG__%d__" % len(link_protect)
                        link_protect[key] = "<p>" + "<br/>".join(srcs) + "</p>"
                        p.replace_with(key)
            for img in _soup.find_all("img"):
                src = img.get("src") or img.get("oldsrc") or ""
                if not src or "fileTypeImages/icon_" in src.lower():
                    continue
                abs_url = urllib.parse.urljoin(page_url, src)
                txt = (img.get("alt") or img.get("title") or "").strip() or os.path.basename(src.split("?")[0]) or "图片"
                key = "__IMG__%d__" % len(link_protect)
                link_protect[key] = '<p><a href="%s" target="_blank">%s</a></p>' % (abs_url, txt)
                img.replace_with(key)
            content = str(_soup)
        except Exception:
            pass

    def _link_repl(m):
        href = m.group(1)
        if href.startswith("javascript:"):
            return ""
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        attachments.append((abs_url, txt))
        key = "__LINK__%d__" % len(link_protect)
        link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
        return key
    content = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content, flags=re.S | re.I)

    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for tag in _soup.find_all(True):
                for attr in ("style", "class", "lang", "dir", "align", "valign", "width", "height", "border", "cellpadding", "cellspacing"):
                    tag.attrs.pop(attr, None)
            # 空 div 删除, 普通 div unwrap (保留内容)
            for d in _soup.find_all("div"):
                if not d.get_text(strip=True) and not d.find("table") and not d.find("img") and not d.find("a"):
                    d.decompose()
                else:
                    d.unwrap()
            content = str(_soup)
        except Exception:
            pass

    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return "__TBL__%d__" % (len(table_protect) - 1)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0
    content = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content, flags=re.S | re.I)

    content = re.sub(r"</p>", "</p>\n\n", content, flags=re.I)
    content = re.sub(r"<br\s*/?>", "\n", content, flags=re.I)
    content = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content, flags=re.I)

    parts = []
    for block in content.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        b = re.sub(r"(?<=>)\s*[\r\n\t]+\s*", "", b)
        b = re.sub(r"\s*[\r\n\t]+\s*(?=<)", "", b)
        b = b.replace("\r", "").replace("\t", " ")
        b = re.sub(r"[ \t]{2,}", " ", b)
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace("__TBL__%d__" % i, tbl)
        # 裸链接块 (附件/下载 无 p 包裹) → 包 <p> 段落
        if not re.search(r"<p[ >]", b, re.I) and re.search(r"<a[ >]", b, re.I):
            b = "<p>%s</p>" % b
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace("__TBL__%d__" % i, tbl)

    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)
    out = re.sub(r"__IMG__\d+__", "", out)
    out = re.sub(r"__ATTACH__\d+__", "", out)

    out = html_lib.unescape(out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    return out, has_table, attachments


def main():
    import requests
    sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
    from crawler_lib import push_to_searchdb
    s = requests.Session()
    s.headers.update(HEADERS)

    new_count = 0
    skip_count = 0
    seen_urls = set()
    jsonl_rows = []
    batch = []

    for page in range(1, _PAGES + 1):
        try:
            xml_text = fetch_list(s, page)
            items = parse_list(xml_text)
        except Exception as e:
            print("Page %d fetch error: %s" % (page, e), flush=True)
            break
        if not items:
            print("Page %d: empty, stop" % page, flush=True)
            break
        print("Page %d: found %d items" % (page, len(items)), flush=True)
        for abs_url, title, date in items:
            if abs_url in seen_urls:
                continue
            seen_urls.add(abs_url)
            # 详情请求必须无 Referer（加速乐）
            dhtml = fetch(abs_url)
            if not dhtml:
                skip_count += 1
                continue
            dtitle, ddate, content_html = parse_detail(dhtml, abs_url)
            if not dtitle:
                dtitle = title
            if not ddate:
                ddate = date
            content, has_table, atts = html_to_text(content_html, abs_url)
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            no_para = not re.search(r"<p[ >]", content, re.I) and not re.search(r"<a[ >]", content, re.I) and not has_table
            # 纯图片附件页 (如 佳默电气2.jpg 公示): 保留图片链接, plain_len>=5 即入库
            if plain_len < 5 or no_para:
                print("  skip empty/placeholder: %s" % dtitle[:40], flush=True)
                skip_count += 1
                continue
            item = {
                "site_name": SITE_NAME, "source_url": abs_url, "url": abs_url,
                "title": dtitle, "pub_date": ddate, "summary": "",
                "content": content, "category": CATEGORY, "tags": "",
                "group_name": GROUP_NAME,
                "attachments": "; ".join("%s|%s" % (u, t) for u, t in atts) if atts else "",
            }
            if JSONL_PATH:
                jsonl_rows.append(item)
                print("  JSONL + %s (%s)" % (dtitle[:40], ddate), flush=True)
                new_count += 1
                continue
            batch.append(item)
            new_count += 1
            print("  + %s (%s)" % (dtitle[:40], ddate), flush=True)
            time.sleep(0.25)
        time.sleep(0.5)
        if not JSONL_PATH and batch:
            push_to_searchdb(batch, batch_label=SCRIPT_NAME)
            batch = []

    if JSONL_PATH:
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(json.dumps(row, ensure_ascii=False) + "\n")
        print("JSONL written: %s (%d rows)" % (JSONL_PATH, len(jsonl_rows)), flush=True)
    else:
        if batch:
            push_to_searchdb(batch, batch_label=SCRIPT_NAME)

    print("新增: %d" % new_count, flush=True)
    print("跳过: %d" % skip_count, flush=True)
    print("=== %s done ===" % SITE_NAME, flush=True)


def fetch_list(session, page):
    params = dict(JQ_PARAMS)
    params["page"] = str(page)
    r = session.get(PROXY_URL, params=params, timeout=30, verify=False)
    r.encoding = ENC
    return r.text


def parse_list(xml_text):
    items = []
    # XML: <record><![CDATA[<li><a href="/art/2026/8/5/art_7952_ID.html">标题</a><span>08-05</span></li>]]></record>
    for m in re.finditer(r"<!\[CDATA\[(.*?)\]\]>", xml_text, re.S):
        frag = m.group(1)
        for lm in re.finditer(r'<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>', frag, re.S | re.I):
            href = html_lib.unescape(lm.group(1)).strip()
            inner = lm.group(2)
            txt = clean_title(re.sub(r"<[^>]+>", "", inner))
            if not txt or len(txt) < 4:
                continue
            if not re.search(r"/art/\d{4}/\d+/\d+/art_\d+_\d+\.html", href):
                continue
            if href.startswith("/"):
                href = BASE_URL + href
            elif not href.startswith("http"):
                href = urllib.parse.urljoin(BASE_URL + "/", href)
            # 日期: URL 带年份 /art/2026/8/5/, span 带 MM-DD
            date = ""
            ym = re.search(r"/art/(\d{4})/\d+/\d+/", href)
            dm = re.search(r"(\d{1,2})-(\d{1,2})", frag)
            if ym:
                if dm:
                    date = "%s-%s-%s" % (ym.group(1), dm.group(1).zfill(2), dm.group(2).zfill(2))
                else:
                    date = ym.group(1) + "-01-01"
            items.append((href, txt, date))
    return items


def parse_detail(html_text, page_url):
    title = ""
    date = ""
    content_html = ""
    if BeautifulSoup:
        soup = BeautifulSoup(html_text, "html.parser")
        mt = soup.find("meta", attrs={"name": re.compile(r"ArticleTitle", re.I)})
        if mt and mt.get("content"):
            title = clean_title(mt["content"])
        if not title:
            h1 = soup.find("h1")
            if h1:
                title = clean_title(h1.get_text(strip=True))
        md = soup.find("meta", attrs={"name": re.compile(r"PubDate", re.I)})
        if md and md.get("content"):
            dm = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", md["content"])
            if dm:
                date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
        # 正文: div.detail_texts div.content
        el = (soup.select_one("div.detail_texts div.content")
              or soup.select_one("div.detail_texts")
              or soup.select_one("div.content")
              or soup.find(id="zoom") or soup.select_one("div.zoom"))
        if el:
            content_html = "".join(str(c) for c in el.contents)
    if not date:
        dm = re.search(r"(\d{4})[-年/](\d{1,2})[-月/](\d{1,2})", html_text)
        if dm:
            date = "%s-%s-%s" % (dm.group(1), dm.group(2).zfill(2), dm.group(3).zfill(2))
    return title, date, content_html


if __name__ == "__main__":
    main()
