#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
常州市生态环境局 — 建设项目环境影响评价审批 (sthjj.changzhou.gov.cn)
列表: /class/ALFJELKI (第1页) + /class/ALFJELKI/N (N=2..6, 30条/页)
      <a href='/html/hbj/YYYY/ALFJELKI_MMDD/NNNN.html' title='完整标题'>标题</a> + td 日期
详情: /html/hbj/YYYY/ALFJELKI_MMDD/NNNN.html
      标题: meta name='ArticleTitle' | 日期: meta name='PubDate'(取前10)
      正文: td.NewsText (含表格)
入库: crawler_lib.push_to_searchdb → /root/search.db (FTS触发器自动维护)
注意: 服务器 curl 直连可用, 无需 playwright; meta 用单引号
"""
import sys, os, re, json, time, urllib.parse, random, urllib.request, ssl
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "常州市生态环境局-建设项目环评审批"
BASE = "https://sthjj.changzhou.gov.cn"
COLUMN = "ALFJELKI"
CUTOFF = "2018-01-01"
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
CTX = ssl._create_unverified_context()

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1


def http_get(url, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers={"User-Agent": UA})
            resp = urllib.request.urlopen(req, timeout=30, context=CTX)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [HTTP-ERR] {url[:70]} : {str(e)[:50]}")
                return None
            time.sleep(2)
    return None


def list_url(page):
    if page <= 1:
        return f"{BASE}/class/{COLUMN}"
    return f"{BASE}/class/{COLUMN}/{page}"


def parse_list(html_text):
    items = []
    # <a href='/html/hbj/2026/ALFJELKI_0819/27923.html' target='_blank' title='完整标题'>
    # 只匹配本栏目 ALFJELKI_ 前缀文章
    for m in re.finditer(
        r"<a[^>]*href='(/html/hbj/\d{4}/ALFJELKI_[^']+\.html)'[^>]*title='([^']*)'",
        html_text, re.S,
    ):
        href, title = m.group(1).strip(), m.group(2).strip()
        if not title or "html" not in href:
            continue
        # 日期: 该 a 所在 tr 内找 td 日期
        tr_start = html_text.rfind("<tr", 0, m.start())
        tr_end = html_text.find("</tr>", m.end())
        seg = html_text[tr_start:tr_end] if tr_start >= 0 and tr_end > 0 else ""
        m_date = re.search(r"(\d{4}-\d{2}-\d{2})", seg)
        date = m_date.group(1) if m_date else ""
        items.append({"url": BASE + href, "title": title, "date": date})
    # 去重 (同一 URL 可能多出现)
    seen = set()
    uniq = []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            uniq.append(it)
    return uniq


def extract_news_text(html_text):
    m = re.search(r"<td[^>]*class=['\"]NewsText['\"][^>]*>(.*?)</td>\s*</tr>", html_text, re.S)
    if m:
        return m.group(1)
    # 兜底: 任意 NewsText
    m2 = re.search(r"<td[^>]*class=['\"]NewsText['\"][^>]*>(.*)", html_text, re.S)
    if m2:
        seg = m2.group(1)
        depth = 1
        for mm in re.finditer(r"<td\b|</td>", seg):
            if mm.group(0).startswith("</td>"):
                depth -= 1
                if depth == 0:
                    return seg[:mm.start()]
            else:
                depth += 1
    return ""


def clean_body(body):
    body = re.sub(r"<script.*?</script>", "", body, flags=re.S | re.I)
    body = re.sub(r"<style.*?</style>", "", body, flags=re.S | re.I)
    body = re.sub(r"<!--.*?-->", "", body, flags=re.S)
    for kw in ("版权所有", "主办单位", "网站标识码", "苏ICP", "联系我们",
               "无障碍", "网站地图", "常州市人民政府"):
        i = body.find(kw)
        if i > 0:
            body = body[:i]
    return body.strip()


def parse_detail(html_text):
    title = ""
    for pat in [r"name=['\"]ArticleTitle['\"]\s+content=['\"]([^'\"]+)['\"]",
                r"content=['\"]([^'\"]+)['\"]\s+name=['\"]ArticleTitle['\"]"]:
        m = re.search(pat, html_text)
        if m:
            title = re.sub(r"\s+", " ", m.group(1)).strip()
            break
    pub_date = ""
    for pat in [r"name=['\"]PubDate['\"]\s+content=['\"]([^'\"]+)['\"]",
                r"content=['\"]([^'\"]+)['\"]\s+name=['\"]PubDate['\"]"]:
        m = re.search(pat, html_text)
        if m:
            pub_date = m.group(1).strip()[:10]
            break
    body = extract_news_text(html_text)
    if not body:
        return None
    body = clean_body(body)
    if not body:
        return None
    return {"title": title, "pub_date": pub_date, "body": body}


def html_to_text(content, page_url):
    """NewsText 正文 → 存储格式: 附件绝对化 + 表格保留 + 段落化"""
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)
    content2 = content

    attachments = []
    link_protect = {}

    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = txt.strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        is_attach = re.search(r"\.(pdf|docx?|xlsx?|pptx?|zip|rar|wps|et|dps|txt|ofd)(\?|$)", abs_url, re.I)
        if is_attach:
            attachments.append((abs_url, txt))
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<p><a href="{abs_url}" target="_blank">{txt}</a></p>'
        return key

    content2 = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content2, flags=re.S | re.I)
    content2 = re.sub(r"<a\s[^>]*href='([^']+)'[^>]*>(.*?)</a>", _link_repl, content2, flags=re.S | re.I)

    def _img_repl(m):
        src = m.group(1)
        abs_src = urllib.parse.urljoin(page_url, src)
        if re.search(r"ueditor|editor|icon|logo|qrcode|ewm|btn", abs_src, re.I):
            return ""
        return f'<p><a href="{abs_src}" target="_blank">查看图片</a></p>'

    content2 = re.sub(r'<img[^>]*src="([^"]+)"[^>]*>', _img_repl, content2, flags=re.S | re.I)
    content2 = re.sub(r"<img[^>]*src='([^']+)'[^>]*>", _img_repl, content2, flags=re.S | re.I)

    table_protect = []

    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"

    content2 = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content2, flags=re.S | re.I)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0

    content2 = re.sub(r"</p>", "</p>\n\n", content2, flags=re.I)
    content2 = re.sub(r"<br\s*/?>", "\n", content2, flags=re.I)
    content2 = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content2, flags=re.I)

    parts = []
    for block in content2.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)
    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key or key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()
    return out, has_table, attachments


def run():
    print(f"🚀 {SITE_NAME} — 爬取 {_PAGES} 页")
    items_all = []
    for pg in range(1, _PAGES + 1):
        url = list_url(pg)
        html_text = http_get(url)
        if not html_text:
            print(f"  [p{pg}] HTTP失败")
            continue
        items = parse_list(html_text)
        print(f"  第{pg}页: {len(items)} 条")
        if not items:
            if pg > 1:
                break
            continue
        # 截断检测
        cutoff_hit = False
        for it in items:
            if it["date"] and it["date"] < CUTOFF:
                cutoff_hit = True
                break
            items_all.append(it)
        if cutoff_hit:
            print(f"  [CUTOFF] 第{pg}页出现 < {CUTOFF}")
            break
        time.sleep(random.uniform(0.5, 1.2))

    print(f"  [LIST] 共 {len(items_all)} 条")
    if not items_all:
        return

    records = []
    for i, it in enumerate(items_all, 1):
        html_text = http_get(it["url"])
        if not html_text:
            print(f"  [{i}/{len(items_all)}] [ERR] {it['title'][:40]}")
            continue
        d = parse_detail(html_text)
        if not d:
            print(f"  [{i}/{len(items_all)}] [NO-BODY] {it['title'][:40]}")
            continue
        title = d["title"] or it["title"]
        pub_date = d["pub_date"] or it["date"]
        content, has_table, attachments = html_to_text(d["body"], it["url"])
        plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
        has_links = "<a href" in content
        if plain_len < 10 and not has_links:
            print(f"  [{i}/{len(items_all)}] [EMPTY] {title[:40]}")
            content = f"<p>{title}</p>"
        records.append({
            "site_name": SITE_NAME, "title": title, "pub_date": pub_date,
            "url": it["url"], "content": content, "attachments": json.dumps(attachments, ensure_ascii=False),
        })
        print(f"  [{i}/{len(items_all)}] 最新: {title[:55]}...")
        time.sleep(random.uniform(0.3, 0.8))

    new_c = push_to_searchdb(records, batch_label="changzhou_sthjj")
    print(f"Done: {new_c} records")


if __name__ == "__main__":
    run()
