#!/usr/bin/env python3
"""
温州市生态环境局 - 建设项目环境影响评价信息公示 爬虫
==================================================
CMS: JPAAS 发布服务器（浙江统一平台），列表走 API
栏目: col1229880650（建设项目环境影响评价信息公示，共192条≈8页，25条/页）
站点: https://sthjj.wenzhou.gov.cn/col/col1229880650/index.html

用法:
  python3 crawl_wenzhou_sthjj.py             # 增量（第1页）
  python3 crawl_wenzhou_sthjj.py --pages=5   # 爬前5页
  python3 crawl_wenzhou_sthjj.py --full      # 全量8页
"""
import re, sys, os, time, json, sqlite3, html as html_mod
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")

SITE_NAME = "温州市生态环境局-建设项目环境影响评价信息公示"
DOMAIN = "sthjj.wenzhou.gov.cn"
SITE_URL = "https://sthjj.wenzhou.gov.cn/col/col1229880650/index.html"
CATEGORY_NAME = "建设项目环境影响评价信息公示"
GROUP = "浙江"
INDUSTRY = "环评公示"
MAX_PAGES = 8          # 192条 / 25条每页 = 8页
PAGE_SIZE = 25

API_URL = "https://sthjj.wenzhou.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
LIST_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "2490",
    "tplSetId": "UaizqAb90FxcqrGu9suMg",
    "pageType": "column",
    "tagId": "内页分页列表",
    "editType": "null",
    "pageId": "1229880650",
}
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": SITE_URL,
}

stats = {"new": 0, "skip": 0, "errors": 0}


def clean_title(t):
    """标题清洗：HTML实体反转义 + strip &middot;&nbsp; 前缀"""
    if not t:
        return ""
    t = html_mod.unescape(t)            # &middot; &nbsp; &amp; → 字符
    t = t.replace("\u00a0", " ").strip()
    t = re.sub(r"^[·\u00b7]\s*", "", t)  # 去掉开头的 · 及空白
    return t.strip()


def fetch_list_page(session, page_no):
    """通过API获取列表页"""
    params = dict(LIST_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page_no, "pageSize": PAGE_SIZE}, ensure_ascii=False)
    try:
        resp = session.get(API_URL, params=params, timeout=30)
        data = resp.json()
        if not data.get("success"):
            return None, 0
        html = data["data"]["html"]
        count_m = re.search(r'count="(\d+)"', html)
        total = int(count_m.group(1)) if count_m else 0
        return html, total
    except Exception as e:
        print("  [API ERROR] page %d: %s" % (page_no, e), file=sys.stderr)
        return None, 0


def parse_list_html(html):
    """解析列表HTML：li > a.bt_link[title] + span.bt_time"""
    items = []
    soup = BeautifulSoup(html, "lxml")
    for li in soup.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        if "/art/" not in href:
            continue
        title = a.get("title", "") or a.get_text(strip=True)
        title = clean_title(title)
        span = li.find("span", class_="bt_time")
        date = span.get_text(strip=True) if span else ""
        full_url = href if href.startswith("http") else "https://" + DOMAIN + href
        items.append({"title": title, "url": full_url, "date": date[:10]})
    return items


def extract_content(zoom, detail_url):
    """正文提取：段落取文本、表格保留原始HTML、图片转markdown、\n\n分段、附件嵌入"""
    parts = []
    processed = set()
    for el in zoom.find_all(["p", "table", "img", "h1", "h2", "h3", "h4"], recursive=True):
        if id(el) in processed:
            continue
        processed.add(id(el))
        if el.name == "table":
            # 表格保留原始 HTML（政府公示表格多合并单元格，markdown 会丢信息）
            tbl_html = str(el)
            tbl_html = re.sub(r"\s+style=\"[^\"]*\"", "", tbl_html)
            tbl_html = re.sub(r"\s+class=\"[^\"]*\"", "", tbl_html)
            parts.append(tbl_html)
            # 标记表格内所有后代，避免重复处理
            for sub in el.find_all(["p", "table", "img", "h1", "h2", "h3", "h4"], recursive=True):
                processed.add(id(sub))
        elif el.name == "img":
            src = el.get("src", "")
            if src:
                full = urljoin(detail_url, src) if not src.startswith("http") else src
                alt = el.get("alt", "")
                parts.append("![%s](%s)" % (alt, full) if alt else "![](%s)" % full)
        elif el.name == "p":
            if el.find("table"):
                # 表格嵌套在 p 内：克隆去表格后取剩余文本
                clone = BeautifulSoup(str(el), "lxml")
                for t in clone.find_all("table"):
                    t.decompose()
                txt = clone.get_text(" ", strip=True)
            else:
                txt = el.get_text(" ", strip=True)
            if txt:
                txt = re.sub(r"[^\S\n]+", " ", txt)
                parts.append(txt)
        else:
            txt = el.get_text(" ", strip=True)
            if txt:
                parts.append(txt)

    # 附件链接（#zoom 内 .pdf/.doc 等）嵌入正文
    for a in zoom.find_all("a", href=True):
        href = a["href"].strip()
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et)$", href, re.I):
            full = urljoin(detail_url, href) if not href.startswith("http") else href
            name = a.get_text(strip=True) or href.split("/")[-1]
            att_line = "附件：[%s](%s)" % (name, full)
            if att_line not in parts:
                parts.append(att_line)
    return "\n\n".join(parts)


def fetch_detail(session, url):
    """获取详情页内容"""
    try:
        r = session.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")

        # 标题
        title = ""
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = clean_title(meta["content"])
        if not title:
            h1 = soup.find("h1")
            if h1:
                title = clean_title(h1.get_text(strip=True))

        # 日期
        date = ""
        meta_d = soup.find("meta", attrs={"name": "PubDate"})
        if meta_d and meta_d.get("content"):
            date = meta_d["content"].strip()[:10]

        # 正文
        content = ""
        zoom = soup.select_one("#zoom")
        if zoom:
            for tag in zoom.find_all(["script", "style"]):
                tag.decompose()
            content = extract_content(zoom, url)

        summary = re.sub(r"<[^>]+>", " ", content)
        summary = re.sub(r"\s+", " ", summary).strip()[:200]

        return title, date, content, summary
    except Exception as e:
        print("    detail error %s: %s" % (url[-40:], e), file=sys.stderr)
        return "", "", "", ""


def store_record(title, page_url, publish_date, content, summary):
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(title, page_url, source_url, site_name, publish_date, category, industry, group_name, content, summary) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (title, page_url, SITE_URL, SITE_NAME, publish_date,
             CATEGORY_NAME, INDUSTRY, GROUP, content, summary),
        )
        conn.commit()
        is_new = conn.total_changes > 0

        if not is_new and content:
            row = conn.execute("SELECT id, content FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row and (not row[1] or row[1].strip() == ""):
                conn.execute("UPDATE gov_raw SET content=?, summary=? WHERE id=?", (content, summary, row[0]))
                conn.commit()
                is_new = True

        if is_new:
            row = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row:
                c2 = sqlite3.connect(SEARCH_DB, timeout=60)
                try:
                    c2.execute(
                        "INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                        (row[0], title, SITE_NAME, summary or title),
                    )
                    c2.commit()
                except Exception:
                    pass
                finally:
                    c2.close()
            stats["new"] += 1
        else:
            stats["skip"] += 1
    except Exception as e:
        stats["errors"] += 1
        print("  DB error: %s" % e, file=sys.stderr)
    finally:
        conn.close()


def main():
    import argparse
    parser = argparse.ArgumentParser(description="温州市生态环境局 环评公示爬虫")
    parser.add_argument("--full", action="store_true", help="全量8页")
    parser.add_argument("--pages", type=int, default=1, help="爬取页数")
    args = parser.parse_args()

    os.chdir(BASE_DIR)

    pages_to_crawl = MAX_PAGES if args.full else args.pages
    session = requests.Session()
    session.headers.update(HEADERS)

    print("🔍 %s — 爬取 %d 页" % (SITE_NAME, pages_to_crawl))
    for page_no in range(1, pages_to_crawl + 1):
        list_html, total = fetch_list_page(session, page_no)
        if not list_html:
            print("  [p%d] API 失败，停止" % page_no)
            break
        items = parse_list_html(list_html)
        if not items:
            print("  [p%d] 无条目，停止" % page_no)
            break
        print("  [p%d/%d] %d 条 (total=%s)" % (page_no, pages_to_crawl, len(items), total or "?"))
        for it in items:
            title, date, content, summary = fetch_detail(session, it["url"])
            final_title = title or it["title"]
            final_date = date or it["date"]
            store_record(final_title, it["url"], final_date, content, summary or final_title)
            time.sleep(0.4)
        # 已到最后一页（不足一页）则停
        if len(items) < PAGE_SIZE:
            break

    print("\n✅ 完成! 新%d, 跳过%d, 错误%d" % (stats["new"], stats["skip"], stats["errors"]))


if __name__ == "__main__":
    main()
