#!/usr/bin/env python3
"""
温州市瓯海区人民政府 - 建设项目环境影响评价信息公示 爬虫
==================================================
CMS: JPAAS 发布服务器（浙江统一平台），列表走 API
栏目: col1229865783（建设项目环境影响评价信息公示，55条≈3页，25条/页）
站点: https://www.ouhai.gov.cn/col/col1229865783/index.html

用法:
  python3 crawl_ouhai_sthjj.py             # 增量（第1页）
  python3 crawl_ouhai_sthjj.py --pages=5   # 爬前5页
  python3 crawl_ouhai_sthjj.py --full      # 全量3页
"""
import re, sys, os, time, json, sqlite3, html as html_mod
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")

SITE_NAME = "温州市瓯海区人民政府-建设项目环境影响评价信息公示"
DOMAIN = "www.ouhai.gov.cn"
SITE_URL = "https://www.ouhai.gov.cn/col/col1229865783/index.html"
CATEGORY_NAME = "建设项目环境影响评价信息公示"
GROUP = "浙江"
INDUSTRY = "环评公示"
MAX_PAGES = 3           # 55条 / 25条每页 = 3页
PAGE_SIZE = 25

API_URL = "https://www.ouhai.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
LIST_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "1827",
    "tplSetId": "uOnCZokpGn8yCLzYmbGrN",
    "pageType": "column",
    "tagId": "2022新闻list",
    "editType": "null",
    "pageId": "1229865783",
}
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": SITE_URL,
}

stats = {"new": 0, "skip": 0, "errors": 0}


def clean_title(t):
    """标题清洗：HTML实体反转义 + strip &middot;&nbsp; 前缀"""
    if not t:
        return ""
    t = html_mod.unescape(t)
    t = t.replace("\u00a0", " ").strip()
    t = re.sub(r"^[·\u00b7]\s*", "", t)
    return t.strip()


def fetch_list_page(session, page_no):
    """通过API获取列表页"""
    params = dict(LIST_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page_no, "pageSize": PAGE_SIZE}, ensure_ascii=False)
    try:
        resp = session.get(API_URL, params=params, timeout=30)
        data = resp.json()
        if not data.get("success"):
            return None, 0
        html = data["data"]["html"]
        count_m = re.search(r'count="(\d+)"', html)
        total = int(count_m.group(1)) if count_m else 0
        return html, total
    except Exception as e:
        print("  [API ERROR] page %d: %s" % (page_no, e), file=sys.stderr)
        return None, 0


def parse_list_html(html):
    """解析列表HTML：li > a[href] + span(日期)。瓯海 title 属性为空，标题取 a 文本"""
    items = []
    soup = BeautifulSoup(html, "lxml")
    for li in soup.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        if "/art/" not in href:
            continue
        title = clean_title(a.get("title", "") or a.get_text(strip=True))
        span = li.find("span")
        date = span.get_text(strip=True) if span else ""
        full_url = href if href.startswith("http") else "https://" + DOMAIN + href
        items.append({"title": title, "url": full_url, "date": date[:10]})
    return items


def extract_content(cont, detail_url):
    """正文提取：段落取文本、表格保留原始HTML、图片转markdown、\n\n分段、附件嵌入"""
    parts = []
    processed = set()
    for el in cont.find_all(["p", "table", "img", "h1", "h2", "h3", "h4"], recursive=True):
        if id(el) in processed:
            continue
        processed.add(id(el))
        if el.name == "table":
            tbl_html = str(el)
            tbl_html = re.sub(r"\s+style=\"[^\"]*\"", "", tbl_html)
            tbl_html = re.sub(r"\s+class=\"[^\"]*\"", "", tbl_html)
            parts.append(tbl_html)
            for sub in el.find_all(["p", "table", "img", "h1", "h2", "h3", "h4"], recursive=True):
                processed.add(id(sub))
        elif el.name == "img":
            src = el.get("src", "")
            if src:
                full = urljoin(detail_url, src) if not src.startswith("http") else src
                alt = el.get("alt", "")
                parts.append("![%s](%s)" % (alt, full) if alt else "![](%s)" % full)
        elif el.name == "p":
            if el.find("table"):
                clone = BeautifulSoup(str(el), "lxml")
                for t in clone.find_all("table"):
                    t.decompose()
                txt = clone.get_text(" ", strip=True)
            else:
                txt = el.get_text(" ", strip=True)
            if txt:
                txt = re.sub(r"[^\S\n]+", " ", txt)
                parts.append(txt)
        else:
            txt = el.get_text(" ", strip=True)
            if txt:
                parts.append(txt)

    # 附件链接嵌入
    for a in cont.find_all("a", href=True):
        href = a["href"].strip()
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et)$", href, re.I):
            full = urljoin(detail_url, href) if not href.startswith("http") else href
            name = a.get_text(strip=True) or href.split("/")[-1]
            att_line = "附件：[%s](%s)" % (name, full)
            if att_line not in parts:
                parts.append(att_line)
    return "\n\n".join(parts)


def fetch_detail(session, url):
    """获取详情页内容"""
    try:
        r = session.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "lxml")

        title = ""
        meta = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta and meta.get("content"):
            title = clean_title(meta["content"])
        if not title:
            h1 = soup.find("h1")
            if h1:
                title = clean_title(h1.get_text(strip=True))

        date = ""
        meta_d = soup.find("meta", attrs={"name": "PubDate"})
        if meta_d and meta_d.get("content"):
            date = meta_d["content"].strip()[:10]

        # 正文：瓯海用 .oh_main_cont_flbox_show_cont，备选 #zoom
        content = ""
        cont = soup.select_one(".oh_main_cont_flbox_show_cont") or soup.select_one("#zoom")
        if cont:
            for tag in cont.find_all(["script", "style"]):
                tag.decompose()
            content = extract_content(cont, url)

        summary = re.sub(r"<[^>]+>", " ", content)
        summary = re.sub(r"\s+", " ", summary).strip()[:200]

        return title, date, content, summary
    except Exception as e:
        print("    detail error %s: %s" % (url[-40:], e), file=sys.stderr)
        return "", "", "", ""


def store_record(title, page_url, publish_date, content, summary):
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(title, page_url, source_url, site_name, publish_date, category, industry, group_name, content, summary) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (title, page_url, SITE_URL, SITE_NAME, publish_date,
             CATEGORY_NAME, INDUSTRY, GROUP, content, summary),
        )
        conn.commit()
        is_new = conn.total_changes > 0

        if not is_new and content:
            row = conn.execute("SELECT id, content FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row and (not row[1] or row[1].strip() == ""):
                conn.execute("UPDATE gov_raw SET content=?, summary=? WHERE id=?", (content, summary, row[0]))
                conn.commit()
                is_new = True

        if is_new:
            row = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone()
            if row:
                c2 = sqlite3.connect(SEARCH_DB, timeout=60)
                try:
                    c2.execute(
                        "INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                        (row[0], title, SITE_NAME, summary or title),
                    )
                    c2.commit()
                except Exception:
                    pass
                finally:
                    c2.close()
            stats["new"] += 1
        else:
            stats["skip"] += 1
    except Exception as e:
        stats["errors"] += 1
        print("  DB error: %s" % e, file=sys.stderr)
    finally:
        conn.close()


def main():
    import argparse
    parser = argparse.ArgumentParser(description="瓯海区 环评公示爬虫")
    parser.add_argument("--full", action="store_true", help="全量3页")
    parser.add_argument("--pages", type=int, default=1, help="爬取页数")
    args = parser.parse_args()

    os.chdir(BASE_DIR)

    pages_to_crawl = MAX_PAGES if args.full else args.pages
    session = requests.Session()
    session.headers.update(HEADERS)

    print("🔍 %s — 爬取 %d 页" % (SITE_NAME, pages_to_crawl))
    for page_no in range(1, pages_to_crawl + 1):
        list_html, total = fetch_list_page(session, page_no)
        if not list_html:
            print("  [p%d] API 失败，停止" % page_no)
            break
        items = parse_list_html(list_html)
        if not items:
            print("  [p%d] 无条目，停止" % page_no)
            break
        print("  [p%d/%d] %d 条 (total=%s)" % (page_no, pages_to_crawl, len(items), total or "?"))
        for it in items:
            title, date, content, summary = fetch_detail(session, it["url"])
            final_title = title or it["title"]
            final_date = date or it["date"]
            store_record(final_title, it["url"], final_date, content, summary or final_title)
            time.sleep(0.4)
        if len(items) < PAGE_SIZE:
            break

    print("\n✅ 完成! 新%d, 跳过%d, 错误%d" % (stats["new"], stats["skip"], stats["errors"]))


if __name__ == "__main__":
    main()
