#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
福州市晋安区人民政府 - 环评审批 爬虫
CMS: TRS（拓尔思）+ Avalon.js 动静结合
列表: 静态HTML含全部文章链接（63条，15条/页 * 5页）
分页: 全部在首页静态HTML中，不需要分页遍历
详情: div.article_content.article_content_01.font_family_cn
"""
import re, sys, json, time, requests, sqlite3, argparse, subprocess, socket
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.fzja.gov.cn"
LIST_URL = BASE_URL + "/xjwz/zwgk/zfxxgkzdgz/hjbh/hjxmsp/"
DB_PATH = "/root/search.db"
SITE_NAME = "福州市晋安区-环评审批"
GROUP = "福建"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
MAX_PAGES = 1  # 全部文章在首页HTML中

# DNS: 部分环境无法解析 www.fzja.gov.cn，预先解析备用
# dig @8.8.8.8 www.fzja.gov.cn +short → 218.106.155.202 / 112.54.42.145 / 121.204.110.21
FZJA_IPS = ["218.106.155.202", "112.54.42.145", "121.204.110.21"]


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def _patch_dns():
    """劫持 socket.getaddrinfo 使 www.fzja.gov.cn 使用已知 IP"""
    original_getaddrinfo = socket.getaddrinfo

    def patched_getaddrinfo(host, port, family=0, type=0, proto=0, flags=0):
        if host == "www.fzja.gov.cn":
            for ip in FZJA_IPS:
                try:
                    return original_getaddrinfo(ip, port, family, type, proto, flags)
                except Exception:
                    continue
        return original_getaddrinfo(host, port, family, type, proto, flags)

    socket.getaddrinfo = patched_getaddrinfo


_patch_dns()


def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def get_soup(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [ERROR] fetch: {e}", file=sys.stderr)
        return None


def extract_list_items(soup):
    """从页面提取所有文章链接（静态HTML包含全部63条）"""
    items = []
    # 找到列表容器
    for div in soup.find_all("div", class_=re.compile(r"list_base")):
        for a in div.find_all("a", href=True):
            href = a["href"].strip()
            # 只抓 t20260727_5350663.htm 格式的文章链接
            if re.search(r't\d{8}_\d+\.htm$', href):
                title = a.get("title") or a.get_text(strip=True) or ""
                title = title.strip()
                # 去掉 &middot;&nbsp; 实体前缀
                title = re.sub(r'^[\s\xa0·\u00b7]+', '', title).strip()
                if not title or len(title) < 6:
                    continue
                full_url = href if href.startswith("http") else urljoin(LIST_URL, href)
                # 去重
                if not any(item["url"] == full_url for item in items):
                    items.append({"title": title, "url": full_url})
    return items


def extract_date(soup):
    """从详情页提取发布时间"""
    # meta PubDate
    m = soup.find("meta", attrs={"name": "PubDate"})
    if m and m.get("content"):
        return m["content"].strip()[:10]
    # span.article_time
    sp = soup.find("span", class_="article_time")
    if sp:
        m2 = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', sp.get_text())
        if m2:
            return m2.group(1)
    return ""


def extract_title(soup):
    """从详情页提取标题"""
    # meta ArticleTitle
    m = soup.find("meta", attrs={"name": "ArticleTitle"})
    if m and m.get("content"):
        return m["content"].strip()
    # div.article_title
    div = soup.find("div", class_="article_title")
    if div:
        t = div.get_text(strip=True)
        if t:
            return t
    return ""


def extract_content(soup):
    """提取正文HTML，保留表格、段落、附件"""
    div = (soup.find("div", class_="article_content article_content_01 font_family_cn") or
           soup.find("div", class_=re.compile(r"article_content")))
    if not div:
        return ""
    
    # 有些TRS页面在article_content内嵌套了<html><body>...</body></html>
    # 需要解除<html>包装，直接提取<body>内的内容
    html_tag = div.find("html")
    if html_tag:
        body_tag = html_tag.find("body")
        if body_tag:
            # 直接用body替换div的内容容器
            container = body_tag
        else:
            container = html_tag
    else:
        container = div
    
    # 收集段落
    parts = []
    
    for child in container.children:
        child_str = str(child)
        tag = getattr(child, "name", None)
        
        if tag == "p":
            text = body_text(child)
            if text:
                inner_html = re.sub(r'\s+', ' ', str(child).strip())
                parts.append(inner_html)
        elif tag == "table":
            parts.append(child_str.strip())
        elif tag in ("div", "section"):
            # 检查是否包含表格
            tbl = child.find("table")
            if tbl:
                parts.append(str(tbl).strip())
            else:
                text = body_text(child)
                if text:
                    parts.append(f"<p>{text}</p>")
        elif tag in ("ul", "ol"):
            parts.append(child_str.strip())
    
    # 附件区域
    attach = soup.find("div", class_="article_attachment")
    if attach:
        parts.append(attach_str(attach))
    
    return "\n\n".join(parts)


def attach_str(div):
    """格式化附件区域"""
    lines = []
    for a in div.find_all("a", href=True):
        href = a["href"]
        name = a.get_text(strip=True) or "附件"
        if href.startswith("/") or href.startswith("./"):
            href = urljoin(BASE_URL, href)
        lines.append(f'<a href="{href}">{name}</a>')
    return "<p>附件：" + "；".join(lines) + "</p>" if lines else str(div)


def save_to_db(conn, title, pub_date, content, url):
    """保存到search.db"""
    page_url = url
    source_url = SITE_NAME
    
    conn.execute("""
        INSERT OR IGNORE INTO gov_raw
            (title, page_url, source_url, content, site_name, publish_date, group_name, script_name)
        VALUES (?, ?, ?, ?, ?, ?, ?, ?)
    """, (title, page_url, source_url, content, SITE_NAME, pub_date, GROUP, "crawl_fzja.py"))
    conn.commit()
    rid = conn.execute("SELECT last_insert_rowid()").fetchone()[0]
    # Insert into FTS search index
    conn.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                 (rid, title, SITE_NAME, content[:200] if content else ""))
    conn.commit()
    return rid


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help="爬取页数")
    args = parser.parse_args()
    pages = min(args.pages, MAX_PAGES)
    
    print(f"[{SITE_NAME}] 开始爬取，pages={pages}")
    
    soup = get_soup(LIST_URL)
    if not soup:
        print("[ERROR] 无法获取列表页")
        sys.exit(1)
    
    items = extract_list_items(soup)
    print(f"  列表页获取到 {len(items)} 篇文章")
    
    conn = init_db()
    new_count = 0
    skip_count = 0
    err_count = 0
    
    for idx, item in enumerate(items):
        url = item["url"]
        # 去重检查
        exists = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,)).fetchone()
        if exists:
            skip_count += 1
            continue
        
        print(f"  [{idx+1}/{len(items)}] {item['title'][:40]}...", end=" ", flush=True)
        
        detail_soup = get_soup(url)
        if not detail_soup:
            print("[ERROR 获取详情]")
            err_count += 1
            continue
        
        title = item["title"] or extract_title(detail_soup)
        pub_date = extract_date(detail_soup)
        content = extract_content(detail_soup)
        
        if not title or not content:
            print("[跳过 无标题/内容]")
            skip_count += 1
            continue
        
        rid = save_to_db(conn, title, pub_date, content, url)
        new_count += 1
        print(f"OK id={rid}")
        
        time.sleep(1)
    
    conn.close()
    print(f"\n[{SITE_NAME}] 完成: +{new_count} 跳过{skip_count} 错误{err_count}")


if __name__ == "__main__":
    main()
