#!/usr/bin/env python3
"""
岫岩满族自治县-生态环境
URL: http://www.xiuyan.gov.cn/xyxzf/zwgk/hjbh/glist.html
分页: glist.html(第1页) + API: cms.anshan.gov.cn/html/page.xhtml (第2~26页)
共26页/386条，内容含表格+PDF附件
"""

import requests
import sqlite3
import json
import os
import argparse
import re
from bs4 import BeautifulSoup

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "岫岩县-环境保护"
BASE_URL = "http://www.xiuyan.gov.cn"
LIST_URL = BASE_URL + "/xyxzf/zwgk/hjbh/glist.html"
API_TEMPLATE = "http://cms.anshan.gov.cn/html/page.xhtml?s=XYXZF&o={}&p=157404820375520&c=157404820375520"
TOTAL_PAGES = 26

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_soup(url, timeout=20):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = "utf-8"
    return BeautifulSoup(r.text, "html.parser")


def parse_list(soup):
    """提取列表页: ul > li > div.time + div.name > a"""
    items = []
    for li in soup.select("ul li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if "html/XYXZF" not in href:
            continue
        title = a.get_text(strip=True)
        time_div = li.find("div", class_="time")
        pub_date = time_div.get_text(strip=True) if time_div else ""
        items.append((href, title, pub_date))
    return items


def extract_detail(soup, url):
    """提取详情页: div.info.hwq-info-article 内容"""
    title = ""
    pub_date = ""
    source = ""
    content = ""
    attachments = []

    # Meta
    meta_t = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_t and meta_t.get("content"):
        title = meta_t["content"].strip()
    meta_pd = soup.find("meta", attrs={"name": "PubDate"})
    if meta_pd and meta_pd.get("content"):
        pub_date = meta_pd["content"].strip()

    # Content
    center = soup.select_one(".hwq-info-article-center")
    if not center:
        info = soup.select_one(".info.hwq-info-article")
        center = info if info else soup

    # 附件链接
    for a_tag in center.find_all("a"):
        a_href = a_tag.get("href", "")
        if a_href.endswith((".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar")):
            if not a_href.startswith("http"):
                a_href = BASE_URL + a_href
            attachments.append({
                "name": a_tag.get_text(strip=True) or os.path.basename(a_href),
                "url": a_href,
            })

    # 组装正文：保留段落和表格的自然结构
    parts = []
    for child in center.children:
        if child.name == "table":
            parts.append(str(child))
            parts.append("")
        elif child.name == "p":
            txt = child.get_text(strip=True)
            # 只跳过纯"点击：""次"这些噪声
            if txt and txt not in ("点击：", "次"):
                parts.append(txt)
        elif child.name == "div":
            txt = child.get_text(strip=True)
            if txt:
                parts.append(txt)

    content = "\n".join(parts)
    if not content.strip():
        content = body_text(center)

    if not title:
        # 从正文第一行取标题
        lines = content.split("\n")
        if lines:
            title = lines[0]

    return title, pub_date, content, attachments


def crawl(max_pages=None):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()

    pages = max_pages if max_pages else TOTAL_PAGES
    all_items = []

    for page in range(1, pages + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = API_TEMPLATE.format(page)

        print(f"[列表] 第{page}/{pages}页: ...{url[-60:]}")
        try:
            soup = get_soup(url)
            items = parse_list(soup)
            print(f"  -> {len(items)} 条")
            if not items:
                print("  -> 空页，停止")
                break
            all_items.extend(items)
        except Exception as e:
            print(f"  -> 失败: {e}")
            break

    new_count = 0
    error_count = 0

    for idx, (page_url, list_title, list_date) in enumerate(all_items, 1):
        cur.execute(
            "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
            (page_url, SITE_NAME),
        )
        if cur.fetchone():
            print(f"  [{idx}/{len(all_items)}] 跳过: {list_title[:30]}...")
            continue

        print(f"  [{idx}/{len(all_items)}] {list_title[:40]}...")
        try:
            soup = get_soup(page_url)
            title, pub_date, content, attachments = extract_detail(soup, page_url)
            if not title:
                title = list_title
            if not pub_date:
                pub_date = list_date

            att_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""

            cur.execute(
                """INSERT OR REPLACE INTO gov_raw (page_url, site_name, title, publish_date, source_url, content, attachments, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, 'crawl_xiuyan.py')""",
                (page_url, SITE_NAME, title, pub_date, page_url, content, att_json),
            )
            conn.commit()
            new_count += 1
            print(f"    -> 新增")
        except Exception as e:
            error_count += 1
            print(f"    -> 异常: {e}")

    conn.close()
    return new_count, len(all_items), error_count


if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=None)
    args = parser.parse_args()

    new, total, errors = crawl(args.max_pages)
    print(f"\n完成: 新增 {new}, 列表 {total}, 异常 {errors}")
