#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_hejian_sthj.py - 河间市-生态环境
https://www.hejian.gov.cn/hejian/c107617/list.shtml
列表: table.zb_tab > tr > td.name_detail > a (标题CSS截断,须详情页取完整)
  每行4td: 标题 | 来源 | 分类 | 日期
分页: list_{N}.shtml (createPageHTML('page_tag',50,1,'list','shtml',1000) → 20页)
详情: meta ArticleTitle(完整标题) + meta PubDate(真实日期) + div.file_detail 正文
  排除: div.location_ breadcrumb / ul.category 元数据 / div.operation 打印按钮
附件: a href 相对路径 {hash}/files/xxx.zip → urljoin 转绝对
"""
import sys
import os
import re
import time
import json
import sqlite3
import html as html_lib
import urllib.parse
import requests
from bs4 import BeautifulSoup

BASE_URL = "https://www.hejian.gov.cn"
LIST_URL_P1 = "https://www.hejian.gov.cn/hejian/c107617/list.shtml"
LIST_URL_PN = "https://www.hejian.gov.cn/hejian/c107617/list_{page}.shtml"
SITE_NAME = "河间市-生态环境"
GROUP_NAME = "河北"
SCRIPT_NAME = "crawl_hejian_sthj.py"
DB_PATH = os.environ.get("DB_PATH", "/mnt/data/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")

try:
    import requests.packages.urllib3
    requests.packages.urllib3.disable_warnings()
except Exception:
    pass

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

# --pages 参数
_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

# --jsonl 参数
for i, a in enumerate(sys.argv):
    if a == "--jsonl" and i + 1 < len(sys.argv):
        JSONL_PATH = sys.argv[i + 1]


def fetch_page(url, retries=5):
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if i == retries - 1:
                print(f"  [WARN] {url}: {e}", file=sys.stderr)
                return ""
            time.sleep(2.5)
    return ""


def clean_title(t):
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"^\s*(?:&middot;|·|\u00b7)?\s*(?:&nbsp;|\u00a0)?\s*", "", t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def get_total_pages(html_text):
    """从 createPageHTML('page_tag', 每页条数, 当前页, 'list', 'shtml', 总数) 提取总页数
    ⚠️ 河间实际每页 20 条(非参数值50), 总数1000 → 50页"""
    m = re.search(r"createPageHTML\s*\(\s*['\"]page_tag['\"]\s*,\s*(\d+)\s*,\s*\d+\s*,\s*['\"][^'\"]*['\"]\s*,\s*['\"][^'\"]*['\"]\s*,\s*(\d+)", html_text)
    if m:
        per_page = int(m.group(1))
        total_items = int(m.group(2))
        # 实测每页 20 条,但参数可能虚标;保守按 20/页 计算,并向上取整
        pages = (total_items + 19) // 20
        return max(1, pages)
    return 1


def parse_list(html_text):
    items = []
    soup = BeautifulSoup(html_text, "html.parser")
    tb = soup.find("table", class_="zb_tab")
    if not tb:
        return items
    for tr in tb.find_all("tr"):
        tds = tr.find_all("td")
        if len(tds) < 4:
            continue
        a = tds[0].find("a", href=True)
        if not a:
            continue
        href = a.get("href", "")
        title = clean_title(a.get_text(strip=True))
        date = clean_title(tds[3].get_text(strip=True))
        if not href or not title:
            continue
        if re.search(r"\.{3,}\s*$", title):
            title = title  # 列表截断标题,详情页再取完整
        abs_url = urllib.parse.urljoin(LIST_URL_P1, href)
        dm = re.search(r"(\d{4}-\d{2}-\d{2})", date)
        date = dm.group(1) if dm else ""
        items.append((abs_url, title, date))
    return items


def parse_detail(html_text, page_url):
    title = ""
    date = ""
    content_html = ""
    soup = BeautifulSoup(html_text, "html.parser")
    mt = soup.find("meta", attrs={"name": re.compile(r"ArticleTitle", re.I)})
    if mt and mt.get("content"):
        title = clean_title(mt["content"])
    md = soup.find("meta", attrs={"name": re.compile(r"PubDate|publishdate", re.I)})
    if md and md.get("content"):
        dm = re.match(r"(\d{4}-\d{2}-\d{2})", md["content"].strip())
        if dm:
            date = dm.group(1)
    # 正文容器 div.file_detail
    el = soup.find("div", class_="file_detail")
    if el:
        content_html = "".join(str(c) for c in el.contents)
        # ⚠️ file_detail 内含 <h1> 标题重复(含空 h1 残留),标题已单独存,正文去掉 h1
        content_html = re.sub(r"<h1[^>]*>.*?</h1>", "", content_html, flags=re.S | re.I)
    return title, date, content_html


def html_to_text(content, page_url):
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<meta[^>]*>", "", content, flags=re.I)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)

    attachments = []
    link_protect = {}

    content = re.sub(r"【字号：[^】]*】", "", content)
    content = re.sub(r"发布(?:时间|日期)[:：][^<]{0,30}", "", content)

    _soup = BeautifulSoup(content, "html.parser")
    for a in _soup.find_all("a", href=True):
        if a.get("appendix") or a.get("data-appendix") or re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps|et|ofd)", a.get("href", ""), re.I):
            href = a["href"]
            real_href = a.get("oldsrc") or href
            txt = a.get_text(strip=True) or a.get("_title") or a.get("title") or os.path.basename(real_href.split("?")[0])
            abs_url = urllib.parse.urljoin(page_url, real_href)
            if not abs_url.startswith(("http://", "https://")):
                abs_url = "http:" + abs_url if abs_url.startswith("//") else abs_url
            attachments.append((abs_url, txt))
            key = f"__ATTACH__{len(link_protect)}__"
            link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt}</a>'
            a.replace_with(key)
    for p in _soup.find_all("p"):
        imgs = p.find_all("img")
        if imgs and not p.get_text(strip=True):
            srcs = []
            for img in imgs:
                src = img.get("src") or img.get("oldsrc") or ""
                if src:
                    abs_url = urllib.parse.urljoin(page_url, src)
                    srcs.append(f'<a href="{abs_url}" target="_blank">{os.path.basename(src.split("?")[0]) or "图片"}</a>')
            if srcs:
                key = f"__IMG__{len(link_protect)}__"
                link_protect[key] = "<p>" + "<br/>".join(srcs) + "</p>"
                p.replace_with(key)
    content = str(_soup)

    def _link_repl(m):
        href = m.group(1)
        if href.startswith("javascript:"):
            return ""
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        if not abs_url.startswith(("http://", "https://")):
            abs_url = "http:" + abs_url if abs_url.startswith("//") else abs_url
        attachments.append((abs_url, txt))
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt}</a>'
        return key
    content = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content, flags=re.S | re.I)

    try:
        _soup = BeautifulSoup(content, "html.parser")
        for tag in _soup.find_all(True):
            for attr in ("style", "class", "lang", "dir", "align", "valign", "width", "height", "border", "cellpadding", "cellspacing"):
                tag.attrs.pop(attr, None)
        content = str(_soup)
    except Exception:
        pass

    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0
    content = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content, flags=re.S | re.I)

    content = re.sub(r"</p>", "</p>\n\n", content, flags=re.I)
    content = re.sub(r"<br\s*/?>", "\n", content, flags=re.I)
    content = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+|ucapcontent)\b[^>]*>", "", content, flags=re.I)

    parts = []
    for block in content.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        b = re.sub(r"(?<=>)\s*[\r\n\t]+\s*", "", b)
        b = re.sub(r"\s*[\r\n\t]+\s*(?=<)", "", b)
        b = b.replace("\r", "").replace("\t", " ")
        b = re.sub(r"[ \t]{2,}", " ", b)
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)

    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)
    out = re.sub(r"__IMG__\d+__", "", out)
    out = re.sub(r"__ATTACH__\d+__", "", out)

    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    return out, has_table, attachments


def store_record(conn, url, title, content, date, has_table):
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (url, SITE_NAME))
    if cur.fetchone():
        return False
    try:
        industry = "other"
        try:
            from crawler_lib import classify_industry
            industry = classify_industry(title)
        except Exception:
            pass
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, industry),
        )
        if cur.rowcount > 0:
            summary = re.sub(r"<[^>]+>", "", content)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, SITE_NAME, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, has_table)
        raise


def main():
    conn = None
    if not JSONL_PATH:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute("PRAGMA busy_timeout=60000")
        conn.execute("PRAGMA journal_mode=WAL")

    new_count = 0
    skip_count = 0
    seen_urls = set()
    jsonl_rows = []

    # 先取首页确定总页数
    first_html = fetch_page(LIST_URL_P1)
    total_pages = get_total_pages(first_html)
    pages_to_run = min(_PAGES, total_pages)
    print(f"Total pages: {total_pages}, running: {pages_to_run}", flush=True)

    for page in range(1, pages_to_run + 1):
        url = LIST_URL_P1 if page == 1 else LIST_URL_PN.format(page=page)
        html_text = first_html if page == 1 else fetch_page(url)
        if not html_text:
            print(f"Page {page} fetch error", flush=True)
            break
        items = parse_list(html_text)
        print(f"Page {page}: found {len(items)} items", flush=True)
        if not items:
            print("Empty page, stop pagination", flush=True)
            break
        for abs_url, title, date in items:
            if abs_url in seen_urls:
                continue
            seen_urls.add(abs_url)
            dhtml = fetch_page(abs_url)
            if not dhtml:
                skip_count += 1
                continue
            dtitle, ddate, content_html = parse_detail(dhtml, abs_url)
            if not dtitle:
                dtitle = title
            # ⚠️ 日期优先级: 列表页日期(真实发文日期) > 详情页 meta PubDate(可能是页面生成时间)
            # 河间 hejian.gov.cn: meta PubDate=2025-09-01 是站点生成时间,列表页 2025-07-02 才是真日期
            if date:
                ddate = date
            content, has_table, atts = html_to_text(content_html, abs_url)
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            no_para = not re.search(r"<p[ >]", content_html, re.I) and not has_table
            if plain_len < 10 or no_para:
                print(f"  skip empty/placeholder: {dtitle[:40]}", flush=True)
                skip_count += 1
                continue
            if JSONL_PATH:
                jsonl_rows.append({
                    "page_url": abs_url, "title": dtitle, "content": content,
                    "publish_date": ddate, "has_table": has_table,
                    "site_name": SITE_NAME, "group_name": GROUP_NAME,
                    "script_name": SCRIPT_NAME,
                })
                print(f"  JSONL + {dtitle[:40]} ({ddate})", flush=True)
                new_count += 1
                continue
            ok = store_record(conn, abs_url, dtitle, content, ddate, has_table)
            if ok:
                new_count += 1
                print(f"  + {dtitle[:40]} ({ddate})", flush=True)
            else:
                skip_count += 1
            time.sleep(0.25)
        time.sleep(0.5)

    if JSONL_PATH:
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(json.dumps(row, ensure_ascii=False) + "\n")
        print(f"JSONL written: {JSONL_PATH} ({len(jsonl_rows)} rows)", flush=True)

    if conn is not None:
        conn.commit()
        conn.close()
    print(f"新增: {new_count}", flush=True)
    print(f"跳过: {skip_count}", flush=True)
    print(f"=== {SITE_NAME} done ===", flush=True)


if __name__ == "__main__":
    main()
