#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
福州市生态环境局 - 环评审批公示 爬虫
CMS: TRS（拓尔思）+ Avalon.js 动静结合
列表: 静态HTML含全部文章链接（105条/页，单页全量）
详情: div.article_content > div.TRS_Editor (部分嵌套<html><body>)
"""
import re, sys, json, time, requests, sqlite3, argparse, socket
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.fuzhou.gov.cn"
LIST_URL = BASE_URL + "/zgfzzt/shbj/xxgk/spgs/"
DB_PATH = "/root/search.db"
SITE_NAME = "福州市生态环境局-环评审批公示"
GROUP = "福建"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

# DNS: www.fuzhou.gov.cn 部分环境无法解析，预查IP备用
FUZHOU_IPS = ["183.250.188.86", "117.27.88.101", "218.106.155.152"]


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def _patch_dns():
    original_getaddrinfo = socket.getaddrinfo

    def patched_getaddrinfo(host, port, family=0, type=0, proto=0, flags=0):
        if host == "www.fuzhou.gov.cn":
            for ip in FUZHOU_IPS:
                try:
                    return original_getaddrinfo(ip, port, family, type, proto, flags)
                except Exception:
                    continue
        return original_getaddrinfo(host, port, family, type, proto, flags)

    socket.getaddrinfo = patched_getaddrinfo


_patch_dns()


def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def get_soup(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [ERROR] fetch: {e}", file=sys.stderr)
        return None


def extract_list_items(soup):
    """从页面提取所有文章链接（静态HTML含全部条目）"""
    items = []
    for div in soup.find_all("div", class_=re.compile(r"list_base_date_01")):
        for li in div.find_all("li"):
            a = li.find("a", href=True)
            if not a:
                continue
            href = a["href"].strip()
            if not re.search(r't\d{8}_\d+\.htm$', href):
                continue
            title = a.get("title") or a.get_text(strip=True) or ""
            title = title.strip()
            title = re.sub(r'^[\s\xa0·\u00b7]+', '', title).strip()
            if not title or len(title) < 6:
                continue
            full_url = urljoin(LIST_URL, href)
            full_url = re.sub(r'\.\./+', '', full_url)  # clean ./../../
            # 日期
            span = li.find("span")
            pub_date = span.get_text(strip=True) if span else ""
            # 去重
            if not any(item["url"] == full_url for item in items):
                items.append({"title": title, "url": full_url, "date": pub_date})
    return items


def extract_date(soup):
    """从详情页提取发布时间"""
    m = soup.find("meta", attrs={"name": "PubDate"})
    if m and m.get("content"):
        return m["content"].strip()[:10]
    sp = soup.find("span", class_="article_time")
    if sp:
        m2 = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', sp.get_text())
        if m2:
            return m2.group(1)
    return ""


def extract_title(soup):
    """从详情页提取标题"""
    m = soup.find("meta", attrs={"name": "ArticleTitle"})
    if m and m.get("content"):
        return m["content"].strip()
    div = soup.find("div", class_="article_title")
    if div:
        t = div.get_text(strip=True)
        if t:
            return t
    return ""


def extract_content(soup):
    """提取正文HTML，保留表格、段落、附件

    福州市主站使用 TRS_Editor 容器，
    晋安区分站使用 <html><body> 包装。
    同时兼容两种结构。
    """
    div = (soup.find("div", class_="article_content article_content_01 font_family_cn") or
           soup.find("div", class_=re.compile(r"article_content")))
    if not div:
        return ""

    # 确定内容容器
    # 情况A: TRS_Editor (福州市主站)
    trs = div.find("div", class_="TRS_Editor")
    if trs:
        container = trs
    else:
        # 情况B: <html><body> 包装 (晋安区分站)
        html_tag = div.find("html")
        if html_tag:
            body_tag = html_tag.find("body")
            container = body_tag if body_tag else html_tag
        else:
            container = div

    parts = []

    for child in container.children:
        child_str = str(child)
        tag = getattr(child, "name", None)

        if tag == "p":
            text = body_text(child)
            if text:
                inner_html = re.sub(r'\s+', ' ', str(child).strip())
                parts.append(inner_html)
        elif tag == "table":
            parts.append(child_str.strip())
        elif tag in ("div", "section"):
            # 检查是否包含表格
            tbl = child.find("table")
            if tbl:
                parts.append(str(tbl).strip())
            else:
                text = body_text(child)
                if text:
                    parts.append(f"<p>{text}</p>")
        elif tag in ("ul", "ol"):
            parts.append(child_str.strip())

    # 附件区域
    attach = soup.find("div", class_="article_attachment")
    if attach:
        parts.append(attach_str(attach))

    return "\n\n".join(parts)


def attach_str(div):
    """格式化附件区域"""
    lines = []
    for a in div.find_all("a", href=True):
        href = a["href"]
        name = a.get_text(strip=True) or "附件"
        if href.startswith("/") or href.startswith("./"):
            href = urljoin(BASE_URL, href)
        lines.append(f'<a href="{href}">{name}</a>')
    return "<p>附件：" + "；".join(lines) + "</p>" if lines else str(div)


def save_to_db(conn, title, pub_date, content, url):
    """保存到search.db"""
    page_url = url
    source_url = SITE_NAME

    conn.execute("""
        INSERT OR IGNORE INTO gov_raw
            (title, page_url, source_url, content, site_name, publish_date, group_name, script_name)
        VALUES (?, ?, ?, ?, ?, ?, ?, ?)
    """, (title, page_url, source_url, content, SITE_NAME, pub_date, GROUP, "crawl_fuzhou.py"))
    conn.commit()
    rid = conn.execute("SELECT last_insert_rowid()").fetchone()[0]
    conn.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                 (rid, title, SITE_NAME, content[:200] if content else ""))
    conn.commit()
    return rid


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=1, help="爬取页数（单页已含全量）")
    args = parser.parse_args()

    print(f"[{SITE_NAME}] 开始爬取")

    soup = get_soup(LIST_URL)
    if not soup:
        print("[ERROR] 无法获取列表页")
        sys.exit(1)

    items = extract_list_items(soup)
    print(f"  列表页获取到 {len(items)} 篇文章")

    conn = init_db()
    new_count = 0
    skip_count = 0
    err_count = 0

    for idx, item in enumerate(items):
        url = item["url"]
        # 去重检查
        exists = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,)).fetchone()
        if exists:
            skip_count += 1
            continue

        print(f"  [{idx+1}/{len(items)}] {item['title'][:40]}...", end=" ", flush=True)

        detail_soup = get_soup(url)
        if not detail_soup:
            print("[ERROR 获取详情]")
            err_count += 1
            continue

        title = item["title"] or extract_title(detail_soup)
        pub_date = item.get("date") or extract_date(detail_soup)
        content = extract_content(detail_soup)

        if not title or not content:
            print("[跳过 无标题/内容]")
            skip_count += 1
            continue

        rid = save_to_db(conn, title, pub_date, content, url)
        new_count += 1
        print(f"OK id={rid}")

        time.sleep(1)

    conn.close()
    print(f"\n[{SITE_NAME}] 完成: +{new_count} 跳过{skip_count} 错误{err_count}")


if __name__ == "__main__":
    main()
