#!/usr/bin/env python3
"""
安义县-生态环境局-主动回应（环评公示）
http://anyi.nc.gov.cn/ayxzf/xsthjjgggs2021/just_list.shtml
适用: 服务器端运行 (使用 push_to_searchdb)
运行: python3 crawl_anyi_v2.py --pages 5   (初始5页)
      python3 crawl_anyi_v2.py --pages 1   (日跑增量)
"""

import sys, re, json
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup, Tag

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb

BASE_URL = "http://anyi.nc.gov.cn"
LIST_PATH = "/ayxzf/xsthjjgggs2021/just_list.shtml"
SITE_NAME = "安义县生态环境局-主动回应"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "http://anyi.nc.gov.cn/ayxzf/xsthjjgggs2021/just_list.shtml",
}
session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_page_url(n):
    if n <= 1:
        return urljoin(BASE_URL, LIST_PATH)
    base = LIST_PATH.replace(".shtml", "")
    return urljoin(BASE_URL, f"{base}_{n}.shtml")


def parse_list(html, base_url):
    soup = BeautifulSoup(html, "html.parser")
    res = []
    for li in soup.select("ul.pageList > li"):
        a = li.find("a")
        if not a:
            continue
        t = a.get("title", "").strip() or a.get_text(strip=True)
        if not t:
            continue
        u = urljoin(base_url, a.get("href", "").strip())
        s = li.find("span", class_="time")
        d = s.get_text(strip=True) if s else ""
        res.append((t, u, d))
    return res


def extract_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    h1 = soup.find("h1", class_="article-title")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        mt = soup.find("meta", attrs={"name": "ArticleTitle"})
        if mt and mt.get("content"):
            title = mt["content"].strip()
    date_str = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        date_str = md["content"].strip()

    cp = []
    cd = soup.find("div", class_="article-content-body")
    if not cd:
        cd = soup.find(id="zoomcon")
    if cd:
        ucap = cd.find("ucapcontent")
        if ucap:
            cd = ucap

    if cd:
        for child in list(cd.children):
            if isinstance(child, str):
                t = child.strip()
                if t:
                    cp.append(t)
            elif isinstance(child, Tag):
                n = child.name.lower() if child.name else ""
                if n == "p":
                    txt = ""
                    for ch in child.children:
                        if isinstance(ch, Tag):
                            cn = ch.name.lower()
                            if cn == "span":
                                txt += ch.get_text(strip=True)
                            elif cn == "br":
                                txt += "\n"
                            elif cn == "a":
                                lt = ch.get_text(strip=True)
                                hr = ch.get("href", "")
                                txt += f"[{lt}]({hr})" if hr else lt
                            elif cn == "o:p":
                                continue
                            else:
                                txt += ch.get_text(strip=True)
                        elif isinstance(ch, str):
                            t = ch.strip()
                            if t:
                                txt += t
                    txt = re.sub(r"[ \t]+", " ", txt).strip()
                    if txt:
                        cp.append(txt)
                elif n == "table":
                    cp.append(str(child))
                    rows = child.find_all('tr')
                    if rows:
                        hc = rows[0].find_all(["td", "th"])
                        cc = len(hc)
                        if cc > 0:
                            md_lines = []
                            md_lines.append("| " + " | ".join(
                                x.get_text(strip=True) or " " for x in hc) + " |")
                            md_lines.append("| " + " | ".join(["---"] * cc) + " |")
                            for r in rows[1:]:
                                cells = r.find_all(["td", "th"])
                                rd = [x.get_text(strip=True) or " " for x in cells]
                                while len(rd) < cc:
                                    rd.append(" ")
                                md_lines.append("| " + " | ".join(rd[:cc]) + " |")
                            cp.append("\n".join(md_lines))
                elif n in ("ul", "ol", "div"):
                    t = body_text(child).strip()
                    if t:
                        cp.append(t)

    attachments = []
    if cd:
        for a_tag in cd.find_all("a"):
            href = a_tag.get("href", "").strip()
            if not href:
                continue
            ext = re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar|txt)(\?|$)', href, re.I)
            if ext or "/files/" in href:
                nm = a_tag.get_text(strip=True) or "附件"
                fu = urljoin(url, href)
                attachments.append({"name": nm, "url": fu})
                cp.append(f"[{nm}]({fu})")

    return title, date_str, "\n\n".join(cp), attachments


def crawl(max_pages=5):
    all_items = []
    for pn in range(1, max_pages + 1):
        pu = get_page_url(pn)
        print(f"\n{'='*50}")
        print(f"[PAGE {pn}] {pu}")
        try:
            r = session.get(pu, timeout=30)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print(f"  HTTP {r.status_code}")
                continue
        except Exception as e:
            print(f"  ERROR {e}")
            continue
        soup = BeautifulSoup(r.text, "html.parser")
        if not soup.select("ul.pageList > li"):
            print(f"  No more items, stop at page {pn}")
            break
        items = parse_list(r.text, pu)
        print(f"  {len(items)} items")

        for idx, (t, du, ds) in enumerate(items, 1):
            print(f"\n  [{idx}/{len(items)}] {t[:55]}...")
            try:
                dr = session.get(du, timeout=30)
                dr.encoding = "utf-8"
                if dr.status_code != 200:
                    print(f"    HTTP {dr.status_code}")
                    continue
                dt, dd, dc, da = extract_detail(dr.text, du)
                if not dt:
                    dt = t
                if not dd:
                    dd = ds
            except Exception as e:
                print(f"    ERROR {e}")
                continue
            if not dc:
                print(f"    WARNING: empty content (WAF?)")
                continue

            summary = re.sub(r"\s+", "", dc)[:200] if dc else ""
            print(f"    -> 标题: {dt[:35]}... 日期: {dd} 正文: {len(dc)} 附件: {len(da)}")
            all_items.append({
                "site_name": SITE_NAME, "source_url": du, "url": du,
                "title": dt, "pub_date": dd, "summary": summary,
                "content": dc,
                "attachments": json.dumps(da, ensure_ascii=False) if da else "",
            })

    if all_items:
        print(f"\n[*] 入库 {len(all_items)} 条...")
        push_to_searchdb(all_items, SITE_NAME)
        print(f"  DONE: {len(all_items)}")
    else:
        print("No items to push")


if __name__ == "__main__":
    mp = 5
    if len(sys.argv) > 2 and sys.argv[1] == "--pages":
        mp = int(sys.argv[2])
    crawl(max_pages=mp)
