#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
江西省生态环境厅 - 拟受理项目公示
http://sthjt.jiangxi.gov.cn/jxssthjt/col/col42169/index.html

站点: TRS jpage 动态列表 (页面 HTML 无文章链接, JS 渲染)
列表: POST /queryList (form: current/unitid=380055/webSiteCode=jxssthjt/channelCode=col42169/perPage/pageSize)
      响应 data.total=338 条, results[].source 含 title/pubDate/content.content(完整HTML正文)/urls.pc(详情页)
      每页 15 条 (perPage=15)
详情: 无需二次请求——列表响应已含完整正文 (content.content)
附件: articleFiles 字段 (本栏目为空 [{}])

⚠️ 服务器环境 HTTPS 访问可能被反代劫持返回莒县人民政府页面! 默认用 HTTP。
   本地 DNS 挂时: JXSTHJT_FORCE_IP=59.63.125.18 强制 IP + Host 头。
"""
import sys
import os
import re
import time
import json
import html as html_lib
import urllib.parse
import sqlite3

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

# ---------------- config ----------------
_BASE_DOMAIN = "http://sthjt.jiangxi.gov.cn"
_FORCE_IP = os.environ.get("JXSTHJT_FORCE_IP", "")
if _FORCE_IP:
    BASE_URL = f"http://{_FORCE_IP}"
    _EXTRA_HEADERS = {"Host": "sthjt.jiangxi.gov.cn"}
else:
    BASE_URL = _BASE_DOMAIN
    _EXTRA_HEADERS = {}
API_URL = BASE_URL + "/queryList"
SITE_NAME = "江西省生态环境厅-拟受理项目公示"
GROUP_NAME = "江西"
SCRIPT_NAME = "crawl_jxsthjt_nslx.py"
DB_PATH = os.environ.get("JXSTHJT_DB", "/mnt/data/search.db")
UNITID = "380055"
WEBSITE_CODE = "jxssthjt"
CHANNEL_CODE = "col42169"
PAGE_SIZE = 15

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "application/json, text/plain, */*",
    "Content-Type": "application/x-www-form-urlencoded",
    "Referer": "http://sthjt.jiangxi.gov.cn/jxssthjt/col/col42169/index.html",
}
HEADERS.update(_EXTRA_HEADERS)

# crawler_lib 所在目录 (classify_industry 行业分类复用)
for _p in ("/root/gov_crawler", "/root/gov_crawler"):
    if os.path.isdir(_p) and _p not in sys.path:
        sys.path.insert(0, _p)

# default: daily incremental = 1 page
_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass

_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1


# ---------------- helpers ----------------
def clean_title(t):
    """unescape entities + strip &middot;&nbsp; prefix + strip zero-width + collapse whitespace."""
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = re.sub(r"^\s*[·•]+\s*", "", t)
    t = t.replace("\xa0", " ").replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def api_post(page):
    """POST /queryList page N, return results list."""
    import requests
    data = {
        "current": str(page),
        "unitid": UNITID,
        "webSiteCode": WEBSITE_CODE,
        "channelCode": CHANNEL_CODE,
        "perPage": str(PAGE_SIZE),
        "pageSize": str(PAGE_SIZE),
    }
    last = None
    for i in range(3):
        try:
            r = requests.post(API_URL, data=data, headers=HEADERS, timeout=30)
            if r.status_code == 200:
                return r.json()
            last = r
        except Exception as e:
            last = e
        time.sleep(1.5 * (i + 1))
    if isinstance(last, Exception):
        raise last
    return None


def html_to_text(content, page_url):
    """Convert article HTML to stored format (same pipeline as gov scripts)."""
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)
    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for tag in _soup.find_all(True):
                for attr in ("style", "class", "lang", "dir", "align", "valign", "width", "height", "border", "cellpadding", "cellspacing"):
                    tag.attrs.pop(attr, None)
            content = str(_soup)
        except Exception:
            pass

    # 1) FIRST protect attachment <a> links
    content2 = content
    attachments = []
    link_protect = {}
    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, html_lib.unescape(href))
        attachments.append((abs_url, txt))
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt}</a>'
        return key
    content2 = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content2, flags=re.S | re.I)

    # 2) THEN protect tables
    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"
    content2 = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content2, flags=re.S | re.I)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0

    content2 = re.sub(r"</p>", "</p>\n\n", content2, flags=re.I)
    content2 = re.sub(r"<br\s*/?>", "\n", content2, flags=re.I)
    content2 = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content2, flags=re.I)

    parts = []
    for block in content2.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)

    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)

    out = html_lib.unescape(out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    # dedupe paragraphs
    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    return out, has_table, attachments


def store_record(conn, url, title, content, date, has_table):
    cur = conn.execute(
        "SELECT id FROM gov_raw WHERE page_url=?",
        (url,),
    )
    if cur.fetchone():
        return False
    try:
        industry = "other"
        try:
            from crawler_lib import classify_industry
            industry = classify_industry(title)
        except Exception:
            pass
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, industry),
        )
        if cur.rowcount > 0:
            summary = re.sub(r"<[^>]+>", "", content)
            summary = html_lib.unescape(summary)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, SITE_NAME, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, has_table)
        raise


def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    conn.execute("PRAGMA journal_mode=WAL")

    new_count = 0
    skip_count = 0
    seen_urls = set()

    # 第 1 页拿 total
    d = api_post(1)
    if not d or "data" not in d:
        print("新增: 0 (api error)")
        return
    total = d["data"].get("total", 0)
    total_pages = max(1, (total + PAGE_SIZE - 1) // PAGE_SIZE)
    print(f"total={total} total_pages={total_pages} 本次抓取页数={min(_PAGES, total_pages)}")

    for page in range(1, min(_PAGES, total_pages) + 1):
        if page > 1:
            d = api_post(page)
            if not d or "data" not in d:
                continue
        results = d["data"].get("results", [])
        if not results:
            continue
        for item in results:
            src = item.get("source", {})
            title = clean_title(src.get("title", ""))
            pub_date = (src.get("pubDate") or "")[:10]
            # 详情 URL
            try:
                urls = json.loads(src.get("urls") or "{}")
                detail_path = urls.get("pc", "") or ""
            except Exception:
                detail_path = ""
            if not detail_path:
                detail_path = f"/jxssthjt/col/col42169/content/content_{src.get('id','')}.html"
            url = urllib.parse.urljoin(BASE_URL, detail_path)
            if url in seen_urls:
                continue
            seen_urls.add(url)
            # 正文
            content_html = ""
            try:
                c = src.get("content") or {}
                content_html = c.get("content", "") if isinstance(c, dict) else ""
            except Exception:
                content_html = ""
            # 附件 articleFiles
            try:
                files = src.get("articleFiles") or "[]"
                if isinstance(files, str):
                    files = json.loads(files)
                if isinstance(files, list):
                    for f in files:
                        if isinstance(f, dict):
                            fname = f.get("name") or f.get("fileName") or ""
                            furl = f.get("url") or f.get("fileUrl") or ""
                            if furl:
                                if not furl.startswith(("http://", "https://")):
                                    furl = urllib.parse.urljoin(BASE_URL, furl)
                                content_html += f'\n<p><a href="{furl}" target="_blank">{html_lib.escape(fname or os.path.basename(furl))}</a></p>'
            except Exception:
                pass
            content, has_table, attachments = html_to_text(content_html, url)
            if not content:
                skip_count += 1
                continue
            if store_record(conn, url, title, content, pub_date, has_table):
                new_count += 1
            else:
                skip_count += 1

    conn.commit()
    conn.close()
    print(f"新增: {new_count}")


if __name__ == "__main__":
    main()
