#!/usr/bin/env python3
"""
武威市环境保护产业协会 - 信息公开
http://wwhjbh.com/plus/list.php?tid=36
CMS: Dedecms (织梦)

列表: list.php?tid=36, 分页: &TotalResult=321&PageNo=N, 10条/页
详情: /plus/view.php?aid=NNN
  - 标题: div.xwbiao (完整无省略)
  - 日期: <center>YYYY-MM-DD HH:MM</center>
  - 正文: td#neir (含<br/>分段)

运行: python3 /root/crawl_wwhjbh.py --pages 5   (初始)
      python3 /root/crawl_wwhjbh.py --pages 1   (日跑)
"""

import sys, os, re, json
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb

BASE_URL = "http://wwhjbh.com"
LIST_URL = "/plus/list.php?tid=36"
SITE_NAME = "武威市环境保护产业协会-信息公开"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_page_url(n):
    if n <= 1:
        return urljoin(BASE_URL, LIST_URL)
    return urljoin(BASE_URL, f"/plus/list.php?tid=36&TotalResult=321&PageNo={n}")


def parse_list(html):
    """Extract (title, url, date_str) from list page."""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    seen_urls = set()
    for tr in soup.find_all("tr"):
        tds = tr.find_all("td")
        if len(tds) < 2:
            continue
        a_tag = tds[0].find("a", href=True)
        if not a_tag:
            continue
        href = a_tag["href"].strip()
        if "view.php?aid=" not in href:
            continue
        title = a_tag.get_text(strip=True)
        if not title:
            continue
        full_url = urljoin(BASE_URL, href)
        if full_url in seen_urls:
            continue
        date_td = tds[-1]
        date_str = date_td.get_text(strip=True)
        # Only accept if date_td contains a valid date
        if not re.search(r'\d{4}-\d{2}-\d{2}', date_str):
            continue
        seen_urls.add(full_url)
        items.append((title, full_url, date_str))
    return items


def extract_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")

    # Title from div.xwbiao
    title = ""
    xwbiao = soup.find("div", class_="xwbiao")
    if xwbiao:
        title = xwbiao.get_text(strip=True)

    # Date from <center>YYYY-MM-DD HH:MM</center>
    date_str = ""
    m = re.search(r'(\d{4}-\d{2}-\d{2}\s*\d{2}:\d{2})', html)
    if m:
        date_str = m.group(1).strip()

    # Content from td#neir
    content_parts = []
    neir = soup.find("td", id="neir")
    if neir:
        # Get text with <br/> as paragraph separators
        texts = []
        for child in neir.children:
            if child.name == "br":
                continue
            if isinstance(child, str):
                t = child.strip()
                if t:
                    t = re.sub(r'&nbsp;', ' ', t)
                    t = re.sub(r'[\xa0]+', ' ', t)
                    texts.append(t)
            elif hasattr(child, "get_text"):
                t = body_text(child).strip()
                if t:
                    texts.append(t)

        # Also check for tables inside neir
        for table in neir.find_all("table"):
            md = _table_to_markdown(table)
            if md:
                texts.append(md)

        # Check for attachments
        for a_tag in neir.find_all("a"):
            href = a_tag.get("href", "").strip()
            if not href:
                continue
            ext = re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar|txt)(\?|$)', href, re.I)
            if ext or "/uploads/" in href:
                name = a_tag.get_text(strip=True) or "附件"
                full_url = urljoin(url, href)
                texts.append(f"[{name}]({full_url})")

        content = "\n\n".join(texts)
        # Clean up excessive whitespace
        content = re.sub(r'[ \t]+', ' ', content)
        content = re.sub(r'\n{3,}', '\n\n', content)
    else:
        content = ""

    return title, date_str, content.strip()


def _table_to_markdown(table):
    rows = table.find_all("tr")
    if not rows:
        return ""
    md = []
    hc = rows[0].find_all(["td", "th"])
    cc = len(hc)
    if cc == 0:
        return ""
    md.append("| " + " | ".join(x.get_text(strip=True) or " " for x in hc) + " |")
    md.append("| " + " | ".join(["---"] * cc) + " |")
    for row in rows[1:]:
        cells = row.find_all(["td", "th"])
        rd = [x.get_text(strip=True) or " " for x in cells]
        while len(rd) < cc:
            rd.append(" ")
        md.append("| " + " | ".join(rd[:cc]) + " |")
    return "\n".join(md)


def crawl(max_pages=5):
    all_items = []
    for pn in range(1, max_pages + 1):
        pu = get_page_url(pn)
        print(f"\n{'='*50}")
        print(f"[PAGE {pn}] {pu}")
        try:
            r = session.get(pu, timeout=30)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print(f"  HTTP {r.status_code}")
                continue
        except Exception as e:
            print(f"  ERROR {e}")
            continue

        items = parse_list(r.text)
        if not items:
            print(f"  No items, stop at page {pn}")
            break
        print(f"  {len(items)} items")

        for idx, (list_title, du, list_date) in enumerate(items, 1):
            print(f"\n  [{idx}/{len(items)}] {list_title[:55]}...")
            try:
                dr = session.get(du, timeout=30)
                dr.encoding = "utf-8"
                if dr.status_code != 200:
                    print(f"    HTTP {dr.status_code}")
                    continue
                dt, dd, dc = extract_detail(dr.text, du)
                if not dt:
                    dt = list_title
                if not dd:
                    dd = list_date
            except Exception as e:
                print(f"    ERROR {e}")
                continue

            if not dc:
                print(f"    WARNING: empty content")
                continue

            summary = re.sub(r"\s+", "", dc)[:200] if dc else ""
            print(f"    -> 标题: {dt[:40]}... 日期: {dd} 正文: {len(dc)}")
            all_items.append({
                "site_name": SITE_NAME, "source_url": du, "url": du,
                "title": dt, "pub_date": dd, "summary": summary,
                "content": dc or "(无正文内容)",
            })

    if all_items:
        print(f"\n[*] 入库 {len(all_items)} 条...")
        push_to_searchdb(all_items, SITE_NAME)
        print(f"  DONE: {len(all_items)}")
    else:
        print("No items to push")


if __name__ == "__main__":
    mp = 5
    if len(sys.argv) > 2 and sys.argv[1] == "--pages":
        mp = int(sys.argv[2])
    crawl(max_pages=mp)
