#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
海南省生态环境厅 - 环境影响评价 (政务公开 xxgk)
站点: https://hnsthb.hainan.gov.cn
栏目: /xxgk/0200/0202/hjywgl/hjyxpj/xxgkindex.html
列表: TRS 静态页 xxgkindex.html + xxgkindex_{N}.html (20条/页, 声明78页实际77, P78→302/404)
详情: meta ArticleTitle/PubDate + div.view.TRS_UEDITOR 正文 + div.other-word 附件
"""
import re, os, sys, time
import requests, warnings
warnings.filterwarnings('ignore')
from urllib.parse import urljoin
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

_MAX_PG = None
for _a in sys.argv[1:]:
    if _a.startswith("--pages="):
        try:
            _MAX_PG = int(_a.split("=", 1)[1])
        except Exception:
            pass
    elif _a.isdigit():
        _MAX_PG = int(_a)
if _MAX_PG is not None:
    print(f'[AutoPg] max_pages={_MAX_PG}')

SITE_NAME = "海南省生态环境厅-环境影响评价"
DOMAIN = "hnsthb.hainan.gov.cn"
LIST_URL = f"https://{DOMAIN}/xxgk/0200/0202/hjywgl/hjyxpj/xxgkindex.html"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36"}


def fetch(url, label=""):
    for retry in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=25)
            if r.status_code == 200:
                r.encoding = "utf-8"
                return r.text
            print(f"    [WARN] {label} HTTP {r.status_code}, 重试{retry+1}")
        except Exception as e:
            print(f"    [WARN] {label} 异常: {e}, 重试{retry+1}")
        time.sleep(2)
    return ""


def parse_list(html, list_url):
    """解析列表页: tr > a.new_title_a 标题链接 + 独立日期 td"""
    items = []
    soup = BeautifulSoup(html, "lxml")
    for a in soup.select("a.new_title_a"):
        href = (a.get("href") or "").strip()
        if not href:
            continue
        title = re.sub(r"\s+", " ", a.get_text()).strip()
        # 去实体前缀 &middot; &nbsp;
        title = title.replace("&middot;", "").replace("&nbsp;", "").strip()
        if not title:
            continue
        url = urljoin(list_url, href)
        # 外链过滤: 非本站域名的链接跳过 (招标平台等混入)
        if re.match(r"https?://", url) and DOMAIN not in url:
            continue
        # 找该 a 所在 tr 的日期
        pub_date = ""
        tr = a.find_parent("tr")
        if tr:
            m = re.search(r"<td>\s*(20\d{2}-\d{2}-\d{2})\s*</td>", str(tr))
            if m:
                pub_date = m.group(1)
        items.append({"title": title, "url": url, "pub_date": pub_date})
    # URL 去重（列表双链接防御）
    seen, uniq = set(), []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            uniq.append(it)
    return uniq


def parse_detail(html, url, list_title):
    """详情: meta ArticleTitle/PubDate + TRS_UEDITOR 正文 + other-word 附件"""
    # 标题: 详情 meta 优先，回退列表标题
    m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    title = (m.group(1).strip() if m else "") or list_title
    title = title.replace("&middot;", "").replace("&nbsp;", "").strip()
    title = re.sub(r"^[•·]\s*", "", title).strip()
    title = title.replace("\u200b", "").replace("\ufeff", "").strip()
    # 日期: meta PubDate (YYYY-MM-DD HH:MM)
    pub_date = ""
    m = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    if m:
        pub_date = m.group(1)
    # 正文: div.view.TRS_UEDITOR 平衡 div
    content = ""
    m = re.search(r'<div class="view TRS_UEDITOR[^"]*"[^>]*>(.*?)</div>\s*<p class="clear">', html, re.S)
    if not m:
        m = re.search(r'<div class="view TRS_UEDITOR[^"]*"[^>]*>(.*)', html, re.S)
    if m:
        body = m.group(1)
        # 平衡 div 计数截取
        depth = 0
        end = 0
        for mm in re.finditer(r"<div[^>]*>|</div>", body):
            if mm.group(0).startswith("</"):
                depth -= 1
                if depth <= 0:
                    end = mm.start()
                    break
            else:
                depth += 1
        body = body[:end] if end else body
        content = clean_content(body, url)
    # 附件: div.other-word 内 li > a
    atts = []
    m = re.search(r'<div class="other-word">(.*?)</div>', html, re.S)
    if m:
        for a in re.findall(r'<a[^>]+href="([^"]+)"[^>]*>([^<]+)</a>', m.group(1)):
            href, name = a
            if re.search(r"\.(doc|docx|pdf|xls|xlsx|wps|rar|zip|txt)$", href, re.I):
                name = re.sub(r"\s+", " ", name).strip()
                atts.append(f'<p><a href="{urljoin(url, href)}">{name}</a></p>')
    if atts:
        content = content.rstrip() + "\n" + "\n".join(atts)
    return title, pub_date, content


def clean_content(body, page_url):
    """清洗正文: 绝对化图片/链接, 规范化 <p>, 保留段落结构"""
    # 图片/链接相对路径绝对化
    body = re.sub(r'(<img[^>]*src=")([^"]+)"',
                  lambda m: m.group(1) + urljoin(page_url, m.group(2)) + '"', body)
    body = re.sub(r'(<a[^>]*href=")([^"]+)"',
                  lambda m: m.group(1) + urljoin(page_url, m.group(2)) + '"', body)
    # 规范化 <p style=...> -> <p> (search_app 严格匹配闭合 <p>)
    body = re.sub(r"<p[^>]*>", "<p>", body)
    # 去除 script/style
    body = re.sub(r"<script.*?</script>", "", body, flags=re.S)
    body = re.sub(r"<style.*?</style>", "", body, flags=re.S)
    # 去除纯空白段落 (剥标签+实体后无有效文本且无 img; 保留含 img 段落)
    def _drop_blank_para(m):
        seg = m.group(0)
        if "<img" in seg:
            return seg
        txt = re.sub(r"<[^>]+>", "", seg)
        txt = txt.replace("&nbsp;", " ").replace("\u3000", " ").replace("&#160;", " ")
        txt = re.sub(r"\s+", "", txt)
        return "" if not txt else seg
    body = re.sub(r"<p>.*?</p>", _drop_blank_para, body, flags=re.S)
    # 去除多余空段落
    body = re.sub(r"<p>\s*(?:<br\s*/?>)?\s*</p>", "", body)
    return body.strip()


def run(max_pages=200):
    print(f"\n{'='*50}")
    print(f"🚀 {SITE_NAME}")
    print(f"{'='*50}")

    start = datetime.now()
    all_items = []
    pg = 1
    while True:
        if pg == 1:
            url = LIST_URL
        else:
            url = LIST_URL.replace("xxgkindex.html", f"xxgkindex_{pg}.html")
        html = fetch(url, f"第{pg}页")
        if not html:
            print(f"  ⚠ 第{pg}页抓取失败 -> 结束")
            break
        items = parse_list(html, url)
        if not items:
            print(f"  第{pg}页: 空/无条目 -> 结束 (共{pg-1}页)")
            break
        first_d = items[0]["pub_date"] or "?"
        last_d = items[-1]["pub_date"] or "?"
        print(f"  第{pg}页: {len(items)}条 ({first_d} ~ {last_d})")
        all_items.extend(items)
        # 日期推进停止: 末条早于 CUTOFF
        if items[-1]["pub_date"] and items[-1]["pub_date"] < CUTOFF:
            print(f"  [CUTOFF] {items[-1]['pub_date']} < {CUTOFF} -> 停止")
            break
        if _MAX_PG is not None and pg >= _MAX_PG:
            print(f"  [AutoPg] 达到 max_pages={_MAX_PG} -> 停止")
            break
        pg += 1
        time.sleep(0.8)

    # URL 去重
    seen, uniq = set(), []
    for it in all_items:
        if it["url"] not in seen:
            seen.add(it["url"])
            uniq.append(it)
    print(f"  列表共 {len(all_items)} 条, 去重后 {len(uniq)} 条")

    valid = []
    for i, item in enumerate(uniq):
        if item["pub_date"] and item["pub_date"] < CUTOFF:
            continue
        html = fetch(item["url"], f"详情{i+1}")
        if not html:
            continue
        title, pub_date, content = parse_detail(html, item["url"], item["title"])
        if not content:
            # 图片型公告保留
            has_img = "<img" in html
            if not has_img:
                print(f"  [SKIP] 空正文: {title[:40]}")
                continue
        # 摘要
        txt = re.sub(r"<[^>]+>", "", content)
        txt = re.sub(r"\s+", " ", txt).strip()
        summary = txt[:200] or title
        valid.append({
            "site_name": SITE_NAME,
            "title": title,
            "pub_date": pub_date or item["pub_date"],
            "content": content,
            "summary": summary,
            "source_url": item["url"],
            "url": item["url"],
        })
        if (i + 1) % 20 == 0:
            print(f"  [{i+1}/{len(uniq)}] 已解析 {len(valid)} 条")
        time.sleep(0.5)

    print(f"  有效 {len(valid)} 条 -> push_to_searchdb")
    push_to_searchdb(valid, batch_label="hnsthb_xxgk")
    el = (datetime.now() - start).total_seconds()
    print(f"  ✅ 完成, 耗时 {el:.0f}s")


if __name__ == "__main__":
    run(max_pages=_MAX_PG or 200)
