#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
三门峡市生态环境局 - 区县分局栏目爬虫
https://sthj.smx.gov.cn/{col}/0000/zhengfuxinxi-1.html
CMS: 自定义 (siteview 模板)
列表: zhengfuxinxi-N.html, 15条/页
详情: div.articleTitle (标题) / div.publicDate (日期) / div.articleDetails (正文HTML)

注意: 站点 DNS 解析到 IPv6 不可达导致连接超时, 需固定 IPv4 直连 (111.6.94.72)
"""
import re
import sys
import os
import time
import sqlite3
import urllib.parse
import requests

# ---------------- config ----------------
SITE_BASE = "http://sthj.smx.gov.cn"
FIXED_IP = "111.6.94.72"  # DNS IPv6 不可达, 固定 IPv4
COLUMNS = {
    "25012": "三门峡市-陕州区",
    "25006": "三门峡市-义马市",
    "25021": "三门峡市-灵宝市",
    "25018": "三门峡市-湖滨区",
    "25015": "三门峡市-示范区",
    "25024": "三门峡市-卢氏县",
    "25009": "三门峡市-渑池县",
}
GROUP_NAME = "三门峡"
SCRIPT_NAME = "crawl_smx_district.py"
DB_PATH = os.environ.get("SEARCH_DB", "/mnt/data/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")  # 本地降级模式: 输出 JSONL 不写库

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


class FixedIPAdapter(requests.adapters.HTTPAdapter):
    """将 sthj.smx.gov.cn 固定解析到可达 IPv4, 避免 DNS IPv6 超时"""

    def send(self, request, **kwargs):
        if request.url.startswith("http://sthj.smx.gov.cn/"):
            request.url = request.url.replace("http://sthj.smx.gov.cn/", f"http://{FIXED_IP}/", 1)
            request.headers["Host"] = "sthj.smx.gov.cn"
        return super().send(request, **kwargs)


def make_session():
    s = requests.Session()
    s.headers.update(HEADERS)
    s.mount("http://", FixedIPAdapter())
    return s


def clean_title(t):
    if not t:
        return ""
    t = t.replace("\u200b", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch(session, url, timeout=20):
    r = session.get(url, timeout=timeout)
    r.encoding = r.apparent_encoding or "utf-8"
    return r.text


def parse_list(html, col):
    """解析列表页, 返回 [(abs_url, title, date)]"""
    items = []
    # 文章链接: /25012/2025/10/2151751.html 或 /25012/616750560/1545874.html
    for m in re.finditer(
        r'<a[^>]*href="([^"]*?/' + col + r'/[^"]+\.html)"[^>]*>(.*?)</a>',
        html, re.S | re.I,
    ):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = clean_title(txt)
        if not txt or len(txt) < 8:
            continue
        # 跳过导航/分页
        if txt in ("首页", "上一页", "下一页", "尾页") or "无障碍" in txt or "适老" in txt:
            continue
        if not href.startswith("http"):
            href = urllib.parse.urljoin(SITE_BASE, href)
        # 日期: 链接后 200 字符内
        ctx = html[m.end():m.end() + 200]
        dm = re.search(r"(\d{4}-\d{2}-\d{2})", ctx)
        date = dm.group(0) if dm else ""
        items.append((href, txt, date))
    return items


def parse_detail(html, page_url):
    """解析详情页: 标题/日期/正文HTML"""
    title = ""
    date = ""
    content_html = ""
    # 标题: div.articleTitle
    m = re.search(r'<div[^>]*class="articleTitle"[^>]*>(.*?)</div>', html, re.S | re.I)
    if m:
        title = clean_title(re.sub(r"<[^>]+>", "", m.group(1)))
    if not title:
        m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html, re.I)
        if m:
            title = clean_title(m.group(1))
    # 日期: div.publicDate 内第二个 span
    m = re.search(r'<div[^>]*class="publicDate"[^>]*>(.*?)</div>', html, re.S | re.I)
    if m:
        dm = re.search(r"(\d{4}-\d{2}-\d{2})", m.group(1))
        if dm:
            date = dm.group(1)
    if not date:
        m = re.search(r"(\d{4}-\d{2}-\d{2})", html)
        if m:
            date = m.group(0)
    # 正文: div.articleDetails
    m = re.search(r'<div[^>]*class="articleDetails"[^>]*>(.*?)</div>', html, re.S | re.I)
    if m:
        content_html = m.group(1)
    return title, date, content_html


def html_to_text(content, page_url):
    """保留表格/附件链接的正文转换"""
    if not content:
        return ""
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)
    content = re.sub(r'<[^>]+style="[^"]*display:\s*none[^"]*"[^>]*>.*?</[^>]+>', "", content, flags=re.S | re.I)
    content = re.sub(r"<p[^>]*>\s*<p", "<p", content, flags=re.I)
    content = re.sub(r"</p>\s*</p>", "</p>", content, flags=re.I)
    content = re.sub(r"<p[^>]*>\s*(?:&ensp;|&nbsp;|\s)*\s*</p>", "", content, flags=re.I)
    try:
        from bs4 import BeautifulSoup
        _soup = BeautifulSoup(content, "html.parser")
        for tag in _soup.find_all(True):
            for attr in ("style", "class", "lang", "dir", "align", "valign", "width", "height", "border", "cellpadding", "cellspacing"):
                tag.attrs.pop(attr, None)
        content = str(_soup)
    except Exception:
        pass
    link_protect = {}
    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = re.sub(r"\s+", " ", txt).strip()
        if not txt:
            txt = href.split("/")[-1] or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        if abs_url.startswith("//"):
            abs_url = "http:" + abs_url
        key = "__LK%d__" % len(link_protect)
        link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
        return key
    content = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content, flags=re.S | re.I)
    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return "__TB%d__" % (len(table_protect) - 1)
    content = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content, flags=re.S | re.I)
    content = re.sub(r"</p>", "</p>\n\n", content, flags=re.I)
    content = re.sub(r"<br\s*/?>", "\n", content, flags=re.I)
    content = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content, flags=re.I)
    parts = []
    for block in content.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"\s+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace("__TB%d__" % i, tbl)
        parts.append(b.strip())
    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace("__TB%d__" % i, tbl)
    out = re.sub(r"__LK\d+__", "", out)
    out = re.sub(r"__TB\d+__", "", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    return "\n\n".join(final).strip()


def store_record(conn, url, title, content, date, site_name):
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (url, site_name))
    if cur.fetchone():
        return False
    try:
        industry = "other"
        try:
            from crawler_lib import classify_industry
            industry = classify_industry(title)
        except Exception:
            pass
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, site_name, GROUP_NAME, SCRIPT_NAME,
             1 if "<table" in content else 0, industry),
        )
        if cur.rowcount > 0:
            summary = re.sub(r"<[^>]+>", "", content)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, site_name, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, site_name)
        raise


def crawl_column(session, col, max_pages, conn, jsonl_rows=None):
    site_name = COLUMNS[col]
    print(f"=== {site_name} ({col}) ===", flush=True)
    new_count = 0
    skip_count = 0
    seen_urls = set()
    for page in range(1, max_pages + 1):
        url = f"{SITE_BASE}/{col}/0000/zhengfuxinxi-{page}.html"
        try:
            html = fetch(session, url)
        except Exception as e:
            print(f"  Page {page} fetch error: {e}", flush=True)
            break
        items = parse_list(html, col)
        print(f"  Page {page}: {len(items)} items", flush=True)
        if not items:
            print("  Empty page, stop", flush=True)
            break
        for abs_url, title, date in items:
            if abs_url in seen_urls:
                continue
            seen_urls.add(abs_url)
            try:
                dhtml = fetch(session, abs_url)
            except Exception as e:
                print(f"  detail error {abs_url}: {e}", flush=True)
                skip_count += 1
                continue
            dtitle, ddate, content_html = parse_detail(dhtml, abs_url)
            if not dtitle:
                dtitle = title
            if not ddate:
                ddate = date
            content = html_to_text(content_html, abs_url)
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            if plain_len < 10:
                print(f"  skip empty: {dtitle[:30]}", flush=True)
                skip_count += 1
                continue
            if jsonl_rows is not None:
                # 本地降级模式: 收集 JSONL 行
                jsonl_rows.append({
                    "site_name": site_name,
                    "url": abs_url,
                    "title": dtitle,
                    "pub_date": ddate,
                    "content": content,
                    "attachments": [],
                })
                new_count += 1
                print(f"  JSONL + {dtitle[:40]} ({ddate})", flush=True)
                continue
            ok = store_record(conn, abs_url, dtitle, content, ddate, site_name)
            if ok:
                new_count += 1
                print(f"  + {dtitle[:40]} ({ddate})", flush=True)
            else:
                skip_count += 1
            time.sleep(0.2)
        time.sleep(0.3)
    print(f"  {site_name}: 新增 {new_count}, 跳过 {skip_count}", flush=True)
    return new_count


def main():
    max_pages = 1
    only_cols = None
    for i, a in enumerate(sys.argv):
        if a.startswith("--pages="):
            try:
                max_pages = int(a.split("=", 1)[1])
            except ValueError:
                pass
        elif a.startswith("--col="):
            only_cols = [x.strip() for x in a.split("=", 1)[1].split(",") if x.strip()]
    cols = only_cols if only_cols else list(COLUMNS.keys())

    session = make_session()
    jsonl_rows = [] if JSONL_PATH else None
    conn = None
    if jsonl_rows is None:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute("PRAGMA busy_timeout=60000")
        conn.execute("PRAGMA journal_mode=WAL")

    total_new = 0
    for col in cols:
        try:
            total_new += crawl_column(session, col, max_pages, conn, jsonl_rows)
            if conn:
                conn.commit()
        except Exception as e:
            print(f"  COLUMN {col} ERROR: {e}", flush=True)
            if conn:
                conn.rollback()

    if jsonl_rows is not None:
        import json as _json
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(_json.dumps(row, ensure_ascii=False) + "\n")
        print(f"JSONL written: {JSONL_PATH} ({len(jsonl_rows)} rows)", flush=True)
    else:
        conn.close()
    print(f"=== {SCRIPT_NAME} done: 新增 {total_new} ===", flush=True)


if __name__ == "__main__":
    main()
