#!/usr/bin/env python3
"""crawl_szwz.py - 苏州市吴中区人民政府 公告公示"""
import requests
from bs4 import BeautifulSoup
import sqlite3
import os
import re
import sys
import time

BASE_URL = "http://www.szwz.gov.cn"
LIST_URL = BASE_URL + "/szwz/gggs/list{page}.shtml"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 20
DELAY = 0.5

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "苏州市吴中区人民政府"
COLUMN_NAME = "公告公示"

session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_soup(url):
    try:
        r = session.get(url, timeout=TIMEOUT)
        r.encoding = "utf-8"
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [WARN] 请求失败: {e}")
        return None


def resolve_url(base_url, detail_soup, src):
    """Resolve a potentially relative URL to absolute, using detail page URL."""
    if src.startswith("http"):
        return src
    if src.startswith("/"):
        return BASE_URL + src
    # Relative to the current shtml file's directory
    # e.g., base_url = /szwz/gggs/202607/xxx.shtml
    # src = xxx/images/yyy.jpg
    # result = /szwz/gggs/202607/xxx/images/yyy.jpg
    base_dir = base_url.rsplit("/", 1)[0]
    return base_dir + "/" + src


def extract_detail(url):
    soup = get_soup(url)
    if not soup:
        return "", "", ""

    # Title from meta
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()

    # Publish date from meta
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    # Content from infoContent_info > UCAPCONTENT
    info_div = soup.find("div", class_="infoContent_info")
    if not info_div:
        return title, pub_date, ""

    content_div = info_div.find("UCAPCONTENT")
    if not content_div:
        content_div = info_div

    # Remove scripts/styles
    for tag in content_div.find_all(["script", "style"]):
        tag.decompose()

    # Process images - resolve relative paths
    for img in content_div.find_all("img"):
        src = img.get("src", "")
        alt = img.get("alt", "")
        if src:
            src = resolve_url(url, soup, src)
            img_tag = f"![{alt}]({src})"
            new_tag = BeautifulSoup(f"<span>{img_tag}</span>", "html.parser")
            img.replace_with(new_tag)

    # Process links (attachments, external)
    for a in content_div.find_all("a"):
        href = a.get("href", "")
        if not href or href.startswith("#"):
            a.unwrap()
            continue
        href = resolve_url(url, soup, href)
        a_text = a.get_text(strip=True)
        if a_text:
            markdown_link = f"[{a_text}]({href})"
            new_tag = BeautifulSoup(f"<span>{markdown_link}</span>", "html.parser")
            a.replace_with(new_tag)
        else:
            a.unwrap()

    # Remove <o:p> tags (Word formatting leftovers)
    for tag in content_div.find_all("o:p"):
        tag.decompose()
    # Remove <span> tags but keep their text
    for tag in content_div.find_all(["span", "font"]):
        tag.unwrap()

    # Extract content in document order, skip p inside tables
    paragraphs = []
    for tag in content_div.find_all(["p", "table"]):
        if tag.name == "table":
            paragraphs.append(str(tag))
        elif tag.name == "p":
            if tag.find_parent("table"):
                continue
            text = body_text(tag)
            if not text or text in ("\xa0", ""):
                continue
            # Clean &nbsp; and collapse internal newlines (broken CMS formatting)
            text = text.replace("\xa0", " ").strip()
            text = re.sub(r"\n+", "", text)
            if text:
                paragraphs.append(text)

    content = "\n\n".join(paragraphs)

    # Also check for attachment section
    ext_dl = soup.find("dl", class_="article-extended")
    if ext_dl:
        file_links = []
        for a in ext_dl.find_all("a", href=True):
            href = a["href"]
            a_text = a.get_text(strip=True)
            if href and a_text:
                href = resolve_url(url, soup, href)
                file_links.append(f"[{a_text}]({href})")
        if file_links:
            if content:
                content += "\n\n**附件：**\n" + "\n".join(file_links)
            else:
                content = "\n".join(file_links)

    return title, pub_date, content


def crawl():
    page = 1
    total_added = 0

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    while True:
        if page == 1:
            url = LIST_URL.format(page="")
        else:
            url = LIST_URL.format(page=f"_{page}")
        print(f"  第{page}页...", end=" ", flush=True)
        soup = get_soup(url)
        if not soup:
            print("FAIL")
            break

        items = soup.select("li.ewb-info-item")
        if not items:
            print("空列表，结束")
            break

        print(f"{len(items)}条")
        for item in items:
            a_tag = item.find("a")
            if not a_tag:
                continue
            link = a_tag.get("href", "").strip()
            if not link:
                continue
            link = resolve_url(url, soup, link)

            list_title = a_tag.get_text(strip=True)

            # Check if already exists
            c.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (link,))
            if c.fetchone():
                continue

            # Extract detail
            time.sleep(DELAY)
            ext_title, pub_date, content = extract_detail(link)
            if not ext_title:
                ext_title = list_title
            if not content or len(content) < 20:
                content = f'<p><a href="{link}">{ext_title}</a></p>'
                print(f"    ⚠ {list_title[:40]}... - 正文为空，回退为链接")

            # Insert
            try:
                c.execute(
                    """INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, date_rank, summary, script_name) VALUES (?,?,?,?,?,?,?, 'crawl_szwz.py')""",
                    (link, ext_title, SITE_NAME, pub_date, content,
                     pub_date.replace("-", "") if pub_date else "0", ext_title[:200]),
                )
                total_added += 1
            except Exception as e:
                print(f"    ERROR: {e}")

        # Commit every 10 pages to avoid timeout data loss
        if page % 10 == 0:
            conn.commit()
            print(f"  [已提交 {total_added} 条]")

        # Check if last page (0 items = truly empty)
        if len(items) == 0:
            break
        # Max 117 pages (from JS total=1872 records)
        if page >= 117:
            break
        page += 1
        time.sleep(DELAY)

    conn.commit()
    conn.close()
    print(f"\n完成: 新增 {total_added} 条")
    return total_added


if __name__ == "__main__":
    print(f"站点: {SITE_NAME} - {COLUMN_NAME}")
    added = crawl()
    print(f"总计新增: {added}")
