#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
济源产城融合示范区生态环境局 - 环境评价栏目爬虫
列表: https://hbj.jiyuan.gov.cn/hjpj/  (TRS/WCM 静态页, 声明48页, 每页5条)
分页: index_N.html  (第2页=index_1.html ... 第48页=index_47.html)
列表结构: <li><span>2026-07-30</span> <a href="./hpgs/t1006601.html">标题</a><em></em></li>
详情容器: div.view.TRS_UEDITOR (位于 div#BodyLabel 内)
日期:     <SPAN class=zw_time>日期： 2026-08-07 </SPAN>; 列表 span 为 YYYY-MM-DD
用法: python3 crawl_jiyuan_hjpj.py [--pages=N]
"""
import os, sys, re, time, html as html_mod
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = "济源市生态环境局-环境评价"
CATEGORY = "环境评价"
LIST_URL = "https://hbj.jiyuan.gov.cn/hjpj/"
PAGE_URL = "https://hbj.jiyuan.gov.cn/hjpj/index_{}.html"   # 第N页(N>=2) -> index_{N-1}.html
TOTAL_PAGES = 48
CONTAINERS = ["div.view.TRS_UEDITOR", "div#BodyLabel", "div.zwcontent", "div.content"]

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
HEADERS = {"User-Agent": UA, "Accept": "text/html,application/xhtml+xml", "Accept-Language": "zh-CN,zh;q=0.9"}

ATTACH_RE = re.compile(r'\.(pdf|doc|docx|xls|xlsx|ppt|pptx|zip|rar|wps|et|ofd|txt)$', re.I)


def parse_pages_arg():
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            return int(a.split("=")[1])
        if a.isdigit():
            return int(a)
    return 1


def clean_title(t):
    t = html_mod.unescape(t or "")
    t = re.sub(r"[\u200b\u200e\u200f\ufeff\xa0]", "", t)
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"^[•·\-—]\s*", "", t)
    return t.strip()


def fetch(url, session):
    try:
        r = session.get(url, timeout=25)
        r.encoding = "utf-8"
        return r.text if r.status_code == 200 else ""
    except Exception as e:
        print(f"    [WARN] fetch fail {url}: {e}")
        return ""


def parse_list(html, list_url):
    """解析列表页: 提取本栏目文章 (./xxx/tID.html) + 日期 (li>span)"""
    soup = BeautifulSoup(html, "html.parser")
    items, seen = [], set()
    for li in soup.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        # 只取本栏目相对链接 ./xxx/tID.html (排除 ../ 其它栏目与 http 外链)
        if not href.startswith("./"):
            continue
        if not re.search(r't\d+\.html?$', href):
            continue
        title = clean_title(a.get_text(" ", strip=True))
        if len(title) < 4 or any(x in title for x in ["首页", "网站首页", "邮箱", "搜索"]):
            continue
        url = urljoin(list_url, href)
        if url in seen:
            continue
        seen.add(url)
        date = ""
        sp = li.find("span")
        if sp:
            m = re.search(r'(\d{4})[-/.](\d{1,2})[-/.](\d{1,2})', sp.get_text())
            if m:
                date = "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))
        if not date:  # 从 URL tYYYYMMDD 兜底
            dm = re.search(r't(\d{4})(\d{2})(\d{2})\.', href)
            if dm:
                date = "%s-%s-%s" % dm.groups()
        items.append({"title": title, "url": url, "date": date})
    return items


def extract_detail(html, page_url):
    """提取 标题/日期/正文(保留<p>)/附件(绝对化)"""
    soup = BeautifulSoup(html, "html.parser")

    # 标题: H1 (zw_con 内) -> title meta
    title = ""
    h1 = soup.find("h1")
    if h1:
        title = clean_title(h1.get_text(" ", strip=True))
    if not title:
        m = re.search(r'<meta[^>]*name=["\']ArticleTitle["\'][^>]*content=["\']([^"\']*)["\']', html, re.I)
        if m:
            title = clean_title(m.group(1))
    if not title:
        t = soup.find("title")
        if t:
            title = clean_title(t.get_text(strip=True).split("-")[0])

    # 日期: SPAN.zw_time "日期： 2026-08-07"
    date = ""
    tm = re.search(r'日期[：:]\s*(\d{4})[-/.](\d{1,2})[-/.](\d{1,2})', html)
    if tm:
        date = "%s-%02d-%02d" % (tm.group(1), int(tm.group(2)), int(tm.group(3)))
    if not date:
        m = re.search(r'<meta[^>]*name=["\']PubDate["\'][^>]*content=["\'](\d{4}-\d{2}-\d{2})', html, re.I)
        if m:
            date = m.group(1)

    # 正文容器
    body_el = None
    for sel in CONTAINERS:
        body_el = soup.select_one(sel)
        if body_el and body_el.get_text(strip=True):
            break
    if body_el is None:
        return title, date, "", []

    # 附件扫描范围: div#BodyLabel 优先(含 TRS_UEDITOR 之外的附件表格), 否则容器内
    att_scope = soup.select_one("div#BodyLabel") or body_el
    attachments = []
    for a in att_scope.find_all("a", href=True):
        href = a["href"].strip()
        if ATTACH_RE.search(href) or "download" in href.lower() or "/attached/" in href or "/upload" in href.lower():
            full = urljoin(page_url, href)
            name = a.get_text(strip=True) or href.split("/")[-1]
            attachments.append({"name": name, "url": full})

    # 正文: 保留 <p> 文本, table/img 原样, 附件链接绝对化
    parts = []
    for el in body_el.find_all(["p", "table", "img"]):
        if el.name == "p":
            if el.find_parent("table") or el.find("table") or el.find("p"):
                continue
            for a in el.find_all("a", href=True):
                href = a["href"].strip()
                if ATTACH_RE.search(href) or "download" in href.lower():
                    a["href"] = urljoin(page_url, href)
            txt = el.get_text(" ", strip=True)
            txt = re.sub(r"\s+", " ", txt)
            if txt:
                parts.append("<p>%s</p>" % txt)
        elif el.name == "table":
            # 附件表格内链接绝对化
            for a in el.find_all("a", href=True):
                href = a["href"].strip()
                if ATTACH_RE.search(href) or "download" in href.lower():
                    a["href"] = urljoin(page_url, href)
            parts.append(str(el))
        elif el.name == "img":
            src = el.get("src", "")
            if src:
                full = urljoin(page_url, src)
                alt = el.get("alt", "")
                parts.append('<p><img src="%s" alt="%s"></p>' % (full, alt))

    for att in attachments:
        parts.append('<p><a href="%s">%s</a></p>' % (att["url"], att["name"]))

    content = "\n".join(parts)
    if not content:
        txt = body_el.get_text(" ", strip=True)
        if txt:
            content = "<p>%s</p>" % re.sub(r"\s+", " ", txt)
    return title, date, content, attachments


def main():
    max_pages = parse_pages_arg()
    if max_pages > TOTAL_PAGES:
        max_pages = TOTAL_PAGES
    print("[Jiyuan-hjpj] SITE=%s max_pages=%d" % (SITE_NAME, max_pages))

    session = requests.Session()
    session.headers.update(HEADERS)

    all_items, seen = [], set()
    for pg in range(1, max_pages + 1):
        if pg == 1:
            url = LIST_URL
        else:
            url = PAGE_URL.format(pg - 1)  # 第N页 -> index_{N-1}.html
        h = fetch(url, session)
        if not h or len(h) < 200:
            print("  Page %d: 空/失败, 停止" % pg)
            break
        items = parse_list(h, LIST_URL)
        new_items = [x for x in items if x["url"] not in seen]
        for x in new_items:
            seen.add(x["url"])
        all_items.extend(new_items)
        print("  Page %d: %d条(新%d) 累计%d" % (pg, len(items), len(new_items), len(all_items)))
        if pg < max_pages:
            time.sleep(0.5)

    print("[List] 共 %d 条待抓详情" % len(all_items))
    results = []
    n_p = 0
    for i, it in enumerate(all_items, 1):
        h = fetch(it["url"], session)
        if not h:
            print("  [WARN] 详情失败: %s" % it["url"])
            continue
        title, date, content, attachments = extract_detail(h, it["url"])
        if not title:
            title = it["title"]
        if not date:
            date = it["date"]
        # 附件 JSON 存入 attachments 字段
        att_json = ""
        if attachments:
            import json
            att_json = json.dumps(attachments, ensure_ascii=False)
        if "<p>" in content:
            n_p += 1
        item = {
            "site_name": SITE_NAME,
            "source_url": it["url"],
            "url": it["url"],
            "pub_date": date,
            "title": title,
            "content": content,
            "category": CATEGORY,
            "attachments": att_json,
        }
        results.append(item)
        print("  [%d/%d] %s | %s | p数=%d 附件=%d" % (
            i, len(all_items), title[:40], date, content.count("<p>"), len(attachments)))
        time.sleep(0.3)

    push_to_searchdb(results, batch_label="jiyuan_hjpj")
    total_att = 0
    for x in results:
        if x.get("attachments"):
            import json
            try:
                total_att += len(json.loads(x["attachments"]))
            except Exception:
                pass
    print("[Done] 详情抓取 %d 条, 含<p> %d 条, 附件总数 %d" % (len(results), n_p, total_att))


if __name__ == "__main__":
    main()
