#!/usr/bin/env python3
"""
张家港市人民政府 - 建设项目环评
https://www.zjg.gov.cn/zjg/jsxmv/zdlylist.shtml
CMS: 苏州政府CMS (AJAX静态页面)

注意: 用户提供的 /zjg/jsxm/moreinfozdly_list.shtml 已失效(302重定向)
      实际页面为 /zjg/jsxmv/zdlylist.shtml (建设项目环评)

列表: zdlylist.shtml -> zdlylist_N.shtml, 28条/页
详情: /zjgszwz/jsxmhp01n/{YYYYmm}/{uuid}.shtml
  - 标题: <UCAPTITLE>...</UCAPTITLE> inside h1.article-title
  - 日期: <PUBLISHTIME>...</PUBLISHTIME>
  - 正文: <UCAPCONTENT> 内含 <p>段落 + 图片 + 表格

运行: python3 /root/crawl_zjg_jsxmv.py --pages 5
      python3 /root/crawl_zjg_jsxmv.py --pages 1
"""

import sys, os, re, json
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup, Tag

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb

BASE_URL = "https://www.zjg.gov.cn"
LIST_PATH = "/zjg/jsxmv/zdlylist.shtml"
SITE_NAME = "张家港市-建设项目环评"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "https://www.zjg.gov.cn/zjg/sbje/zdlylist.shtml",
}
session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_page_url(n):
    if n <= 1:
        return urljoin(BASE_URL, LIST_PATH)
    base = LIST_PATH.replace(".shtml", "")
    return urljoin(BASE_URL, f"{base}_{n}.shtml")


def parse_list(html):
    """Extract (title, url, date_str) from list page (my-news-items3)."""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    # Find the static article list in my-news-items3
    list_div = soup.find("div", class_="my-news-items3")
    if not list_div:
        # Fallback: search all divs with this content pattern
        list_div = soup
    for li in list_div.find_all("li"):
        h4 = li.find("h4")
        if not h4:
            continue
        a_tag = h4.find("a", href=True)
        if not a_tag:
            continue
        href = a_tag["href"].strip()
        if "jsxmhp01n" not in href:
            continue
        title = a_tag.get_text(strip=True)
        if not title:
            continue
        full_url = urljoin(BASE_URL, href)
        span_time = li.find("span", class_="time")
        date_str = span_time.get_text(strip=True) if span_time else ""
        items.append((title, full_url, date_str))
    return items


def extract_detail(html, url):
    """Extract title, date, content from detail page."""
    soup = BeautifulSoup(html, "html.parser")

    # Title from UCAPTITLE (uppercase in raw HTML)
    title = ""
    ucap_title = soup.find("ucaptitle")
    if ucap_title:
        title = ucap_title.get_text(strip=True)
    if not title:
        h1 = soup.find("h1", class_="article-title")
        if h1:
            title = h1.get_text(strip=True)
            # Strip UCAPTITLE tags if present
            title = re.sub(r'<[^>]+>', '', title).strip()

    # Date from PUBLISHTIME
    date_str = ""
    pub_time = soup.find("publishtime")
    if pub_time:
        date_str = pub_time.get_text(strip=True)

    # Content from UCAPCONTENT
    cp = []
    content_div = soup.find("div", class_="article-content-body")
    ucap_content = None
    if content_div:
        # Try both uppercase and lowercase
        ucap_content = content_div.find("ucapcontent")
        if not ucap_content:
            # Search all tags (BeautifulSoup may lowercase)
            for child in content_div.children:
                if isinstance(child, Tag) and child.name and child.name.lower() == "ucapcontent":
                    ucap_content = child
                    break

    if not ucap_content and content_div:
        ucap_content = content_div

    if ucap_content:
        for child in ucap_content.children:
            if isinstance(child, str):
                t = child.strip()
                if t:
                    cp.append(t)
            elif isinstance(child, Tag):
                tn = child.name.lower()
                if tn == "p":
                    # Check if paragraph has only images
                    imgs = child.find_all("img")
                    if imgs and not child.get_text(strip=True):
                        for img in imgs:
                            src = img.get("src", "")
                            if src:
                                full_src = urljoin(url, src)
                                cp.append(f"![图片]({full_src})")
                    else:
                        texts = []
                        for ch in child.children:
                            if isinstance(ch, Tag):
                                cn = ch.name.lower()
                                if cn == "img":
                                    src = ch.get("src", "")
                                    if src:
                                        texts.append(f"![图片]({urljoin(url, src)})")
                                elif cn == "span":
                                    texts.append(ch.get_text(strip=True))
                                elif cn == "a":
                                    lt = ch.get_text(strip=True)
                                    hr = ch.get("href", "")
                                    if hr:
                                        texts.append(f"[{lt}]({urljoin(url, hr)})")
                                    else:
                                        texts.append(lt)
                                elif cn == "br":
                                    texts.append("\n")
                                else:
                                    texts.append(ch.get_text(strip=True))
                            elif isinstance(ch, str):
                                t = ch.strip()
                                if t:
                                    texts.append(t)
                        txt = "".join(texts)
                        txt = re.sub(r"[ \t]+", " ", txt).strip()
                        if txt:
                            cp.append(txt)
                elif tn == "div":
                    # Nested div - check for text
                    txt = body_text(child).strip()
                    if txt:
                        cp.append(txt)
                    # Also check for images
                    for img in child.find_all("img"):
                        src = img.get("src", "")
                        if src:
                            cp.append(f"![图片]({urljoin(url, src)})")
                elif tn == "table":
                    md = _table_to_markdown(child)
                    if md:
                        cp.append(md)

    # Attachments
    appendix_div = soup.find("div", class_="article-appendixs")
    if appendix_div:
        for a_tag in appendix_div.find_all("a"):
            href = a_tag.get("href", "")
            txt = a_tag.get_text(strip=True)
            if href and txt:
                cp.append(f"[{txt}]({urljoin(url, href)})")

    content = "\n\n".join(cp)
    return title, date_str, content


def _table_to_markdown(table):
    rows = table.find_all("tr")
    if not rows:
        return ""
    md = []
    hc = rows[0].find_all(["td", "th"])
    cc = len(hc)
    if cc == 0:
        return ""
    md.append("| " + " | ".join(x.get_text(strip=True) or " " for x in hc) + " |")
    md.append("| " + " | ".join(["---"] * cc) + " |")
    for row in rows[1:]:
        cells = row.find_all(["td", "th"])
        rd = [c.get_text(strip=True) or " " for c in cells]
        while len(rd) < cc:
            rd.append(" ")
        md.append("| " + " | ".join(rd[:cc]) + " |")
    return "\n".join(md)


def crawl(max_pages=5):
    all_items = []
    for pn in range(1, max_pages + 1):
        pu = get_page_url(pn)
        print(f"\n{'='*50}")
        print(f"[PAGE {pn}] {pu}")
        try:
            r = session.get(pu, timeout=30)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print(f"  HTTP {r.status_code}")
                continue
        except Exception as e:
            print(f"  ERROR {e}")
            continue

        items = parse_list(r.text)
        if not items:
            print(f"  No items, break")
            break
        print(f"  {len(items)} items")

        for idx, (lt, du, ds) in enumerate(items, 1):
            print(f"\n  [{idx}/{len(items)}] {lt[:55]}...")
            try:
                dr = session.get(du, timeout=30)
                dr.encoding = "utf-8"
                if dr.status_code != 200:
                    print(f"    HTTP {dr.status_code}")
                    continue
                dt, dd, dc = extract_detail(dr.text, du)
                if not dt:
                    dt = lt
                if not dd:
                    dd = ds
            except Exception as e:
                print(f"    ERROR {e}")
                continue

            if not dc:
                print(f"    WARNING: empty content")
                continue

            summary = re.sub(r"\s+", "", dc)[:200] if dc else ""
            print(f"    -> 标题: {dt[:40]}... 日期: {dd} 正文: {len(dc)}")
            all_items.append({
                "site_name": SITE_NAME, "source_url": du, "url": du,
                "title": dt, "pub_date": dd, "summary": summary,
                "content": dc,
            })

    if all_items:
        print(f"\n[*] 入库 {len(all_items)} 条...")
        push_to_searchdb(all_items, SITE_NAME)
        print(f"  DONE: {len(all_items)}")
    else:
        print("No items to push")


if __name__ == "__main__":
    mp = 5
    if len(sys.argv) > 2 and sys.argv[1] == "--pages":
        mp = int(sys.argv[2])
    crawl(max_pages=mp)
