#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""栾川县人民政府-重点领域信息公开-环境保护 爬虫

站点: https://www.luanchuan.gov.cn/zfxxgk/fdzdgknr/zdlyxxgk/hjbh/
列表: div.newsitem > div.newsitemtitle > a (标题含 <br> 需清洗) + div.newsitemtime > b (日期)
分页: 第1页 /, 第N页 index_{N-1}.html (每页20条, 共173页)
详情: 标题 h2.newstitle, 日期 div.htime, 正文 div#contents.newscontents
正文保留 <p> HTML; 附件链接 urljoin 绝对化; 标题清洗(剥实体/零宽/装饰符/内嵌<br>)
"""
import sys, os, re, json, time, html as html_mod
from datetime import datetime, timezone, timedelta
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb, normalize_pub_date

SITE_NAME = "栾川县-环境保护"
CATEGORY = "环境保护"
BASE_URL = "https://www.luanchuan.gov.cn"
LIST_BASE = "https://www.luanchuan.gov.cn/zfxxgk/fdzdgknr/zdlyxxgk/hjbh"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365 * 3)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
}

# 附件扩展名
ATT_RE = re.compile(r"\.(pdf|doc|docx|xls|xlsx|ppt|pptx|rar|zip|wps|et)$", re.I)
# 零宽字符 / BOM / 特殊空格
ZERO_WIDTH_RE = re.compile(
    r"[\u200b\u200c\u200d\u200e\u200f\u2060\ufeff\u00ad\u3000\u00a0]"
)
# 装饰符: 站点名后缀、方括号/书名号包裹的栏目词、常见装饰字符
DECOR_RE = re.compile(
    r"(_栾川县人民政府|_栾川县人民政府$|【[^】]*】|《[^》]*》|\[[^\]]*\]|\s*[-_—|｜·•]\s*栾川县人民政府\s*$)",
    re.I,
)


def clean_title(raw):
    """清洗标题: 剥HTML实体 -> 去零宽/装饰符 -> 内嵌<br>转空格 -> 去残留标签 -> 压缩空白"""
    if not raw:
        return ""
    t = html_mod.unescape(str(raw))
    t = ZERO_WIDTH_RE.sub("", t)
    t = re.sub(r"<br\s*/?>", " ", t, flags=re.I)
    t = re.sub(r"<[^>]+>", "", t)
    t = DECOR_RE.sub("", t)
    t = re.sub(r"\s+", " ", t).strip()
    return t


def fetch_page(page_num):
    """第1页 -> LIST_BASE/, 第N页(N>=2) -> LIST_BASE/index_{N-1}.html"""
    if page_num == 1:
        url = f"{LIST_BASE}/"
    else:
        url = f"{LIST_BASE}/index_{page_num - 1}.html"
    try:
        r = requests.get(url, headers=HEADERS, timeout=25)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  [ERROR] page {page_num} {url}: {e}")
        return None


def parse_items(html):
    """解析列表: div.newsitem > .newsitemtitle > a + .newsitemtime > b"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for item in soup.select("div.newsitem"):
        a = item.select_one("div.newsitemtitle a")
        if not a:
            continue
        href = (a.get("href") or "").strip()
        if not href or not re.search(r"/\d{4}/\d{2}-\d{2}/\d+\.html$", href):
            continue
        url = urljoin(LIST_BASE + "/", href)
        # 标题: 优先 title 属性(与正文一致), 否则取文本; 两者均可能含 <br>
        title = clean_title(a.get("title") or a.get_text(" ", strip=True))
        t = item.select_one("div.newsitemtime b")
        date_text = t.get_text(strip=True) if t else ""
        date_text = normalize_pub_date(date_text)
        items.append((title, url, date_text))
    return items


def fetch_detail(url):
    """抓详情页: 标题/日期/正文(保留 <p> HTML)/附件(绝对化)"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=25)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print(f"  [ERROR] detail {url}: {e}")
        return None, None, None, []

    soup = BeautifulSoup(html, "html.parser")

    # 标题
    title = ""
    h2 = soup.find("h2", class_="newstitle")
    if h2:
        title = clean_title(h2.get_text(" ", strip=True))
    if not title:
        hh = soup.find("div", class_="hh")
        if hh:
            title = clean_title(hh.get_text(" ", strip=True))
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = clean_title(title_tag.get_text(" ", strip=True))
            title = re.sub(r"_环境保护_栾川县人民政府$", "", title).strip()

    # 日期 fallback: div.htime "发布时间：2026-07-31"
    date_str = ""
    htime = soup.find("div", class_="htime")
    if htime:
        m = re.search(r"(\d{4})[-年/.](\d{1,2})[-月/.](\d{1,2})", htime.get_text(" ", strip=True))
        if m:
            date_str = normalize_pub_date(f"{m.group(1)}-{m.group(2)}-{m.group(3)}")

    # 正文容器: div#contents.newscontents
    content_div = soup.find("div", id="contents") or soup.find("div", class_="newscontents")
    content = ""
    attachments = []
    if content_div:
        # 删除脚本/样式
        for tag in content_div.find_all(["script", "style", "iframe"]):
            tag.decompose()
        # 附件链接 & 图片 urljoin 绝对化
        for a in content_div.find_all("a"):
            href = (a.get("href") or "").strip()
            if href and not href.startswith(("javascript:", "mailto:")):
                a["href"] = urljoin(url, href)
                if ATT_RE.search(href):
                    attachments.append({
                        "name": a.get_text(strip=True) or "附件",
                        "url": urljoin(url, href),
                    })
        for img in content_div.find_all("img"):
            src = (img.get("src") or "").strip()
            if src:
                img["src"] = urljoin(url, src)
        # 保留 <p> 等结构化 HTML (取容器内部 HTML)
        content = "".join(str(c) for c in content_div.contents).strip()

    return title, date_str, content, attachments


def main(pages=1, incremental=False):
    print(f"\n{'=' * 56}\n{SITE_NAME} ({CATEGORY})\n{'=' * 56}")
    all_items = []
    seen = set()

    for page in range(1, pages + 1):
        print(f"--- Page {page} ---", end=" ", flush=True)
        html = fetch_page(page)
        if not html:
            print("fetch fail")
            break
        items = parse_items(html)
        if not items:
            print("empty")
            break
        print(f"{len(items)} items")

        exhausted = True
        for title, url, date_text in items:
            if url in seen:
                continue
            seen.add(url)
            if date_text and date_text < THREE_YEARS_AGO:
                continue
            exhausted = False
            if incremental:
                import sqlite3 as s3
                conn = s3.connect("/root/search.db")
                exists = conn.execute("SELECT 1 FROM gov_raw WHERE source_url=?", (url,)).fetchone()
                conn.close()
                if exists:
                    continue
            all_items.append((title, url, date_text))

        if exhausted:
            print("  全部早于3年截断, 停止")
            break
        time.sleep(0.3)

    print(f"📋 待抓详情: {len(all_items)} 条")

    results = []
    stat_empty = 0
    stat_p = 0
    stat_att = 0
    for i, (title, url, list_date) in enumerate(all_items):
        print(f"  [{i + 1}/{len(all_items)}] {title[:46]}...", end=" ", flush=True)
        dt, dd, content, attachments = fetch_detail(url)
        ftitle = dt or title
        if not content or not content.strip():
            stat_empty += 1
        if re.search(r"<p[\s>]", content):
            stat_p += 1
        stat_att += len(attachments)
        summary = re.sub(r"<[^>]+>", " ", content or "").strip()
        summary = re.sub(r"\s+", " ", summary)[:300]
        results.append({
            "site_name": SITE_NAME,
            "source_url": url,
            "url": url,
            "pub_date": dd or list_date,
            "title": ftitle,
            "content": content or "",
            "category": CATEGORY,
            "summary": summary,
            "tags": SITE_NAME,
            "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        })
        print("OK")
        time.sleep(0.3)

    if results:
        push_to_searchdb(results, "luanchuan_hjbh")
    print(f"\n✅ 完成! 入库 {len(results)} 条 | 空正文 {stat_empty} | 含<p> {stat_p} | 附件 {stat_att}")
    return results


if __name__ == "__main__":
    t0 = time.time()
    pages = 1
    incremental = False
    if "--pages" in sys.argv:
        pages = int(sys.argv[sys.argv.index("--pages") + 1])
    if "--incremental" in sys.argv:
        incremental = True
    main(pages=pages, incremental=incremental)
    print(f"Time: {time.time() - t0:.1f}s")
