#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
江苏环评网 js-eia.cn 首页公示专栏抓取 (getIndexData 全类型)
https://www.js-eia.cn/  →  公示专栏 (AJAX: /getIndexData)

策略 (用户确认 2026-08-18):
- 数据源 = 首页公示专栏 getIndexData (聚合所有类型, 带 create_time)
- 只抓当日与前一日发布的公示 (2天窗口)
- 7 个分组: first=信息公开(1) second=公参公示(2) full=全本公示(3)
            completion=竣工公示(4) adjust=调试公示(5) check=验收公示(6)
            pubother=其他环境信息/土壤调查/水土保持等 (pubtype 1-6)
- 详情 URL: project/detail?type=N&proid=xxx 或 pubother/detail?pubid=xxx
- 限速防封: 站点明确警告"异常高频访问会封号+封IP"
  - 列表请求间隔 3-5s, 详情请求间隔 6-10s (随机)
- 登录: 复用扫码登录 cookie (jseia_cookie.txt), 失效时提示重新扫码

用法:
  python3 crawl_jseia.py [--pages=N] [--jsonl] [--dryrun] [--days=1]
  默认: 直接写 /root/search.db
"""
import sys, os, re, json, time, random, sqlite3
import urllib.request, urllib.parse
import ssl
_SSL_CTX = ssl.create_default_context()
_SSL_CTX.check_hostname = False
_SSL_CTX.verify_mode = ssl.CERT_NONE
from datetime import datetime, timedelta, timezone

BASE_URL = "https://www.js-eia.cn"
SITE_NAME = "江苏环评信息公示平台"
GROUP_NAME = "江苏"
SCRIPT_NAME = "crawl_jseia.py"

# 分组 → (类型名, project type 或 pub 标记)
GROUPS = {
    "first":      ("信息公开",   1),
    "second":     ("公参公示",   2),
    "full":       ("全本公示",   3),
    "completion": ("竣工公示",   4),
    "adjust":     ("调试公示",   5),
    "check":      ("验收公示",   6),
}
PUBTYPE_MAP = {1: "企业环境信息", 2: "清洁生产审核", 3: "应急预案公开",
               4: "其他环境信息", 5: "土壤状况调查", 6: "水土保持信息"}

DAYS_BACK = 1          # 日期窗口: 当日 + 前 N 日
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")
COOKIE_FILE = os.environ.get("JSEIA_COOKIE", "/root/gov_crawler/jseia_cookie.txt")

# 限速 (秒)
LIST_DELAY_MIN, LIST_DELAY_MAX = 3, 5
DETAIL_DELAY_MIN, DETAIL_DELAY_MAX = 6, 10
MAX_RETRY = 2

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def _drop_container_parts(parts):
    """剔除「容器段」：find_all(['p','div']) 会同时收下容器 <div> 与它内部的 <p>，
    导致同一内容重复（容器那份常还被 get_text(strip=True) 拍平）。
    判据：先按值去重，再剔除被其它段完全包含的段。保护：剔除后为空则返回去重结果。
    """
    if not parts:
        return parts
    ps = [p for p in parts if isinstance(p, str)]
    if len(ps) != len(parts):
        return parts
    seen, uniq = set(), []
    for p in parts:
        if p not in seen:
            seen.add(p); uniq.append(p)
    keep = [a for a in uniq
            if not (len(a) >= 40 and any(b is not a and b and b in a for b in uniq))]
    return keep if keep else uniq


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def load_cookies():
    cookies = {}
    if not os.path.exists(COOKIE_FILE):
        print(f"[FATAL] cookie 文件不存在: {COOKIE_FILE}", file=sys.stderr)
        sys.exit(1)
    for line in open(COOKIE_FILE, encoding="utf-8"):
        line = line.strip()
        if not line or line.startswith("#"):
            continue
        parts = line.split("\t")
        if len(parts) >= 7:
            cookies[parts[5]] = parts[6]
        elif "=" in line:
            k, v = line.split("=", 1)
            cookies[k] = v
    return cookies


def http_get(url, cookies, referer=None):
    headers = {
        "User-Agent": UA,
        "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
        "Accept-Language": "zh-CN,zh;q=0.9",
        "Cookie": "; ".join(f"{k}={v}" for k, v in cookies.items()),
    }
    if referer:
        headers["Referer"] = referer
    req = urllib.request.Request(url, headers=headers)
    try:
        resp = urllib.request.urlopen(req, timeout=30, context=_SSL_CTX)
        return resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        print(f"  [GET fail] {url[:70]} {e}", file=sys.stderr)
        return None


def http_get_json(url, cookies, referer=None):
    """JSON 接口请求 (getIndexData 需要 XHR 头, 否则 500/空)"""
    headers = {
        "User-Agent": UA,
        "Cookie": "; ".join(f"{k}={v}" for k, v in cookies.items()),
        "Referer": referer or BASE_URL + "/",
        "Accept": "application/json, text/javascript, */*; q=0.01",
        "X-Requested-With": "XMLHttpRequest",
    }
    req = urllib.request.Request(url, headers=headers)
    try:
        resp = urllib.request.urlopen(req, timeout=30, context=_SSL_CTX)
        return json.loads(resp.read().decode("utf-8", errors="replace"))
    except Exception as e:
        print(f"  [GETJSON fail] {url[:70]} {e}", file=sys.stderr)
        return None


def parse_date_cn(s):
    """'2026年08月17日' -> '2026-08-17'"""
    m = re.search(r"(\d{4})年(\d{1,2})月(\d{1,2})日", s or "")
    if m:
        return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return ""


def fetch_index(cookies):
    """GET 首页建立会话 + getIndexData 全量"""
    http_get(BASE_URL + "/", cookies)
    time.sleep(random.uniform(LIST_DELAY_MIN, LIST_DELAY_MAX))
    d = http_get_json(BASE_URL + "/getIndexData", cookies)
    if not d:
        return None
    return d


def collect_window_items(d, cutoff, today):
    """从 getIndexData 收集窗口内条目"""
    items = []
    # project 6 组
    for k, (tn, typeid) in GROUPS.items():
        for it in d.get(k, []):
            ts = int(it.get("create_time", "0"))
            dt = datetime.fromtimestamp(ts, tz=timezone(timedelta(hours=8))).strftime("%Y-%m-%d")
            if cutoff <= dt <= today:
                items.append({
                    "kind": "project", "typename": tn, "typeid": typeid,
                    "proid": it.get("proid"), "title": it.get("title", ""), "date": dt,
                })
    # pubother (pubtype 1-6)
    for it in d.get("pubother", []):
        ts = int(it.get("create_time", "0"))
        dt = datetime.fromtimestamp(ts, tz=timezone(timedelta(hours=8))).strftime("%Y-%m-%d")
        if cutoff <= dt <= today:
            pt = it.get("pubtype", 4)
            try:
                pti = int(pt)
            except (TypeError, ValueError):
                pti = 4
            items.append({
                "kind": "pubother", "typename": PUBTYPE_MAP.get(pti, "其他环境信息"),
                "typeid": pti, "pubid": it.get("id"), "title": it.get("title", ""), "date": dt,
            })
    return items


def fetch_detail(cookies, item):
    """抓详情页, 返回 HTML"""
    if item["kind"] == "project":
        list_url = f"{BASE_URL}/project/list?typeid={item['typeid']}"
        http_get(list_url, cookies, list_url)
        time.sleep(random.uniform(LIST_DELAY_MIN, LIST_DELAY_MAX))
        url = f"{BASE_URL}/project/detail?type={item['typeid']}&proid={item['proid']}"
        html = http_get(url, cookies, list_url)
    else:
        # pubother: 直接访问 (实测无需前置, 但保底 GET publist)
        list_url = f"{BASE_URL}/publist?typeid={item['typeid']}"
        http_get(list_url, cookies, list_url)
        time.sleep(random.uniform(LIST_DELAY_MIN, LIST_DELAY_MAX))
        url = f"{BASE_URL}/pubother/detail?pubid={item['pubid']}"
        html = http_get(url, cookies, list_url)
    if not html or "登录须知" in html:
        return None
    return html


def parse_detail(html, item):
    """从详情页提取标题/日期/正文/附件 (bs4 分段版)"""
    title = ""
    m = re.search(r'<center[^>]*style=["\'](?:[^"\']*24px[^"\']*)["\'][^>]*>(.*?)</center>', html, re.S)
    if m:
        title = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    if not title:
        m = re.search(r"<h5[^>]*>\s*<center[^>]*>(.*?)</center>", html, re.S)
        if m:
            title = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    if not title:
        m = re.search(r"<title>(.*?)</title>", html, re.S)
        if m:
            title = re.sub(r"<[^>]+>", "", m.group(1)).strip().split("_")[0]
    title = re.sub(r"\s+", " ", title).strip()
    if not title:
        title = item["title"]

    date_text = ""
    m = re.search(r"发布日期[：:]\s*(\d{4}年\d{1,2}月\d{1,2}日)", html)
    if m:
        date_text = parse_date_cn(m.group(1))
    if not date_text:
        m = re.search(r"(\d{4}年\d{1,2}月\d{1,2}日)", html)
        if m:
            date_text = parse_date_cn(m.group(1))
    if not date_text:
        date_text = item["date"]

    # 正文容器 (div#article_content)
    content_div = None
    for pat in [r'<div[^>]*id=["\']article_content["\'][^>]*>',
                r'<div[^>]*class=["\'](?:content|article|detail-content|text)[^"\']*["\'][^>]*>',
                r'<div[^>]*id=["\'](?:content|article|zoom|detail)[^"\']*["\'][^>]*>']:
        m = re.search(pat, html)
        if m:
            start = m.end()
            depth = 1
            i = start
            while i < len(html) and depth > 0:
                if html[i:i+4] == "<div":
                    depth += 1
                    i += 4
                elif html[i:i+6] == "</div>":
                    depth -= 1
                    i += 6
                else:
                    i += 1
            content_div = html[start:i]
            break

    # 附件
    attachments = []
    if content_div:
        for m in re.finditer(r'<a[^>]*href=["\']([^"\']+\.(?:docx?|pdf|xlsx?|rar|zip|wps|et|dps))["\'][^>]*>(.*?)</a>', content_div, re.I):
            href, text = m.group(1), re.sub(r"<[^>]+>", "", m.group(2)).strip()
            if href.startswith("/"):
                href = BASE_URL + href
            elif not href.startswith("http"):
                href = BASE_URL + "/" + href
            attachments.append({"name": text or href.split("/")[-1], "url": href})
        seen = set()
        atts = []
        for a in attachments:
            if a["url"] not in seen:
                seen.add(a["url"])
                atts.append(a)
        attachments = atts

    # 正文 (bs4 p/pre 级分段, span 合并)
    if content_div:
        try:
            from bs4 import BeautifulSoup as _BS
            _soup = _BS(html, "html.parser")
            _cd = _soup.find("div", id="article_content")
            if not _cd:
                _cd = _soup.find("div", id=re.compile(r"article_content|content|zoom|detail", re.I))
            if not _cd:
                _cd = _soup.find("div", class_=re.compile(r"content|article|notice", re.I))
            parts = []
            if _cd:
                for el in _cd.find_all(["p", "pre", "table", "div"], recursive=True):
                    if el.name in ("p", "pre") and el.find_parent("table") and el.name == "p":
                        continue
                    if el.name == "table" and el.find_parent("table"):
                        continue
                    if el.name == "table":
                        tbl = str(el)
                        tbl = re.sub(r'href=["\'](/[^"\']+)["\']', lambda m: 'href="%s%s"' % (BASE_URL, m.group(1)), tbl)
                        tbl = re.sub(r'src=["\'](/[^"\']+)["\']', lambda m: 'src="%s%s"' % (BASE_URL, m.group(1)), tbl)
                        parts.append(tbl)
                        continue
                    if el.name == "div":
                        if el.find(["p", "pre", "table"], recursive=False):
                            continue
                        txt = el.get_text("", strip=True)
                        if txt:
                            parts.append(txt)
                        continue
                    txt = el.get_text("", strip=True)
                    txt = txt.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "")
                    if el.name == "pre":
                        txt = body_text(el)
                    if txt and not re.match(r"^(责任编辑|初审|复审|终审|\[纠错\]|浏览次数|字号|分享到)", txt):
                        parts.append(txt)
                # 单字符碎片 (日期被 span 拆开如 "日") 并入最近的非附件名段
                att_names = [a["name"] for a in attachments]
                merged = []
                for p in parts:
                    if not p or (merged and merged[-1] == p):
                        continue
                    if merged and len(p) <= 2 and not p.startswith("<"):
                        j = len(merged) - 1
                        while j >= 0 and any(merged[j] == n or merged[j].endswith(n) for n in att_names):
                            j -= 1
                        if j >= 0:
                            merged[j] += p
                        else:
                            merged.append(p)
                    else:
                        merged.append(p)
                parts = merged
                parts = _drop_container_parts(parts)
                # 统一输出 HTML: 正文段落包 <p>, 与全站 ct-html 渲染一致
                html_parts = []
                for p in parts:
                    if p.startswith("<"):  # 表格/附件链接等已是 HTML
                        html_parts.append(p)
                    else:
                        html_parts.append("<p>" + p + "</p>")
                content = "\n".join(html_parts)
            else:
                content = ""
        except ImportError:
            content = ""
    else:
        content = ""

    # 附件转内嵌段
    for a in attachments:
        link = f'<p><a href="{a["url"]}" target="_blank">{a["name"]}</a></p>'
        if link not in content:
            content += "\n\n" + link

    # 外链兜底
    if len(content.strip()) < 20:
        content = f'<p><a href="{BASE_URL}/pubother/detail?pubid={item["pubid"]}" target="_blank">{title}</a></p>' if item["kind"] == "pubother" else \
                  f'<p><a href="{BASE_URL}/project/detail?type={item["typeid"]}&proid={item["proid"]}" target="_blank">{title}</a></p>'

    return {"title": title, "date": date_text, "content": content, "attachments": attachments}


def item_url(item):
    if item["kind"] == "pubother":
        return f"{BASE_URL}/pubother/detail?pubid={item['pubid']}"
    return f"{BASE_URL}/project/detail?type={item['typeid']}&proid={item['proid']}"


def push_to_searchdb(items):
    if JSONL_PATH:
        with open(JSONL_PATH, "a", encoding="utf-8") as f:
            for it in items:
                f.write(json.dumps(it, ensure_ascii=False) + "\n")
        print(f"  [jsonl] wrote {len(items)}")
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=300000")
    c = conn.cursor()
    new_count = 0
    dup_count = 0
    for it in items:
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (it["url"],))
        if c.fetchone():
            dup_count += 1
            continue
        c.execute(
            "INSERT OR IGNORE INTO gov_raw (title, site_name, group_name, page_url, publish_date, content, summary, attachments, date_rank, script_name) VALUES (?,?,?,?,?,?,?,?,?,?)",
            (it["title"], SITE_NAME, GROUP_NAME, it["url"], it["date"], it["content"], "",
             "; ".join(a["url"] for a in it["attachments"]),
             int(it["date"].replace("-", "")) if re.match(r"\d{4}-\d{2}-\d{2}", it["date"]) else 0,
             SCRIPT_NAME),
        )
        if c.rowcount:
            new_count += 1
    conn.commit()
    conn.close()
    print(f"  新增: {new_count}, 已存在: {dup_count}")


def main():
    for i, a in enumerate(sys.argv):
        if a.startswith("--days="):
            global DAYS_BACK
            try:
                DAYS_BACK = int(a.split("=", 1)[1])
            except ValueError:
                pass
    dryrun = "--dryrun" in sys.argv

    cookies = load_cookies()
    tz = timezone(timedelta(hours=8))
    today = datetime.now(tz).strftime("%Y-%m-%d")
    cutoff = (datetime.now(tz) - timedelta(days=DAYS_BACK)).strftime("%Y-%m-%d")
    print(f"[窗口] 当日+前{DAYS_BACK}日: {cutoff} ~ {today}", file=sys.stderr)

    d = fetch_index(cookies)
    if not d:
        print("[FATAL] getIndexData 获取失败", file=sys.stderr)
        sys.exit(1)

    col_items = collect_window_items(d, cutoff, today)
    col_items.sort(key=lambda x: x["date"], reverse=True)

    # 过滤已入库条目 (日跑时前一日已抓过, 跳过详情避免重复耗时/请求)
    if not dryrun and not JSONL_PATH:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute("PRAGMA busy_timeout=300000")
        c = conn.cursor()
        urls = [item_url(x) for x in col_items]
        known = set()
        for i in range(0, len(urls), 200):
            ph = ",".join("?" * len(urls[i:i+200]))
            c.execute(f"SELECT page_url FROM gov_raw WHERE page_url IN ({ph})", urls[i:i+200])
            known.update(r[0] for r in c.fetchall())
        conn.close()
        before = len(col_items)
        col_items = [x for x in col_items if item_url(x) not in known]
        print(f"[去重] {before} → {len(col_items)} (已存在 {before - len(col_items)})", file=sys.stderr)

    print(f"[列表] 窗口内 {len(col_items)} 条待抓", file=sys.stderr)
    from collections import Counter
    for tn, n in Counter(x["typename"] for x in col_items).most_common():
        print(f"  {tn}: {n}", file=sys.stderr)

    all_items = []
    for idx, item in enumerate(col_items):
        time.sleep(random.uniform(DETAIL_DELAY_MIN, DETAIL_DELAY_MAX))
        html = fetch_detail(cookies, item)
        if not html:
            print(f"  [{idx+1}/{len(col_items)}] 详情失败: {item['title'][:30]}", file=sys.stderr)
            continue
        dd = parse_detail(html, item)
        all_items.append({
            "url": item_url(item),
            "title": dd["title"], "date": dd["date"], "content": dd["content"],
            "attachments": dd["attachments"],
        })
        print(f"  [{idx+1}/{len(col_items)}] [{item['typename']}] {dd['title'][:38]} | {dd['date']}", file=sys.stderr)

    print(f"\n总计: {len(all_items)} 条", file=sys.stderr)
    if dryrun:
        for it in all_items:
            print(json.dumps(it, ensure_ascii=False))
    else:
        push_to_searchdb(all_items)


if __name__ == "__main__":
    main()
