#!/usr/bin/env python3
"""crawl_szhbgz_huan.py - 苏州环保公众网 环评公示"""
import requests
from bs4 import BeautifulSoup
import sqlite3
import os
import re
import sys
import time

BASE_URL = "https://www.szhbgz.org"
LIST_URL = BASE_URL + "/huan.aspx?Page={page}"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 20
DELAY = 0.5

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "苏州环保公众网"
COLUMN_NAME = "环评公示"

session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_soup(url):
    try:
        r = session.get(url, timeout=TIMEOUT)
        r.encoding = "utf-8"
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [WARN] 请求失败: {e}")
        return None


def extract_detail(url):
    soup = get_soup(url)
    if not soup:
        return "", "", ""

    desc = soup.find("div", class_="desc")
    if not desc:
        return "", "", ""

    # Title from h4
    title = ""
    h4 = desc.find("h4")
    if h4:
        title = h4.get_text(strip=True)

    # Date from p.date
    pub_date = ""
    date_p = desc.find("p", class_="date")
    if date_p:
        d = date_p.get_text(strip=True)
        # Format: 2026/07/09 -> 2026-07-09
        d = d.replace("/", "-")
        if re.match(r"\d{4}-\d{1,2}-\d{1,2}", d):
            pub_date = d[:10]

    # Remove scripts/styles
    for tag in desc.find_all(["script", "style"]):
        tag.decompose()

    # Remove p.date and h4 from extraction
    for tag in desc.find_all(["h4", "p", "table"]):
        if tag == h4 or tag == date_p:
            tag.extract()

    # Process images
    for img in desc.find_all("img"):
        src = img.get("src", "")
        alt = img.get("alt", "")
        if src and not src.startswith("http"):
            src = BASE_URL + src if src.startswith("/") else BASE_URL + "/" + src
        img_tag = f"![{alt}]({src})"
        new_tag = BeautifulSoup(f"<span>{img_tag}</span>", "html.parser")
        img.replace_with(new_tag)

    # Process links
    for a in desc.find_all("a"):
        href = a.get("href", "")
        if not href or href.startswith("#") or href.startswith("javascript"):
            a.unwrap()
            continue
        if not href.startswith("http"):
            href = BASE_URL + href if href.startswith("/") else BASE_URL + "/" + href
        a_text = a.get_text(strip=True)
        if not a_text:
            a_text = "附件"
        markdown_link = f"[{a_text}]({href})"
        new_tag = BeautifulSoup(f"<span>{markdown_link}</span>", "html.parser")
        a.replace_with(new_tag)

    # Extract content - p and table, skip p inside table
    paragraphs = []
    for tag in desc.find_all(["p", "table"]):
        if tag.name == "table":
            paragraphs.append(str(tag))
        elif tag.name == "p":
            if tag.find_parent("table"):
                continue
            text = body_text(tag)
            if not text or text in ("\xa0", ""):
                continue
            text = text.replace("\xa0", " ").strip()
            text = re.sub(r"\n+", "", text)
            if text:
                paragraphs.append(text)

    content = "\n\n".join(paragraphs)
    return title, pub_date, content


def crawl():
    page = 1
    total_added = 0
    MAX_PAGES = 123

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    while page <= MAX_PAGES:
        url = LIST_URL.format(page=page)
        print(f"  第{page}页...", end=" ", flush=True)
        soup = get_soup(url)
        if not soup:
            print("FAIL")
            break

        # Find list items in the right column
        items_container = soup.find("div", class_="itemright")
        if not items_container:
            print("no itemright div")
            break

        items = items_container.select("li.clearfix")
        if not items:
            print("空列表，结束")
            break

        print(f"{len(items)}条")
        for item in items:
            a_tag = item.find("a", href=re.compile(r"huan_detail\.aspx"))
            if not a_tag:
                continue
            link = a_tag.get("href", "").strip()
            if not link:
                continue
            if not link.startswith("http"):
                link = BASE_URL + "/" + link if link.startswith("/") else BASE_URL + "/" + link

            list_title = a_tag.get_text(strip=True)

            c.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (link,))
            if c.fetchone():
                continue

            time.sleep(DELAY)
            ext_title, pub_date, content = extract_detail(link)
            if not ext_title:
                ext_title = list_title
            if not content or len(content) < 20:
                content = f'<p><a href="{link}">{ext_title}</a></p>'
                print(f"    ⚠ {list_title[:40]}... - 正文为空，回退为链接")

            try:
                c.execute(
                    """INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, date_rank, summary, script_name) VALUES (?,?,?,?,?,?,?, 'crawl_szhbgz_huan.py')""",
                    (link, ext_title, SITE_NAME, pub_date, content,
                     pub_date.replace("-", "") if pub_date else "0", ext_title[:200]),
                )
                total_added += 1
            except Exception as e:
                print(f"    ERROR: {e}")

        if page % 10 == 0:
            conn.commit()
            print(f"  [已提交 {total_added} 条]")

        page += 1
        time.sleep(DELAY)

    conn.commit()
    conn.close()
    print(f"\n完成: 新增 {total_added} 条")
    return total_added


if __name__ == "__main__":
    print(f"站点: {SITE_NAME} - {COLUMN_NAME}")
    added = crawl()
    print(f"总计新增: {added}")
