#!/usr/bin/env python3
"""crawl_szyq.py - 宿州市埇桥区人民政府 公示公告"""
import requests
from bs4 import BeautifulSoup
import sqlite3
import os
import re
import sys
import time

BASE_URL = "https://www.szyq.gov.cn"
COLUMN_ID = "22954469"
LIST_URL = f"{BASE_URL}/content/column/{COLUMN_ID}?pageIndex={{page}}&pageSize=20"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 20
DELAY = 1.5  # WAF rate limit

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "宿州市埇桥区人民政府"
COLUMN_NAME = "公示公告"

session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_soup(url):
    try:
        r = session.get(url, timeout=TIMEOUT)
        r.encoding = "utf-8"
        return BeautifulSoup(r.text, "html.parser")
    except Exception as e:
        print(f"  [WARN] 请求失败: {e}")
        return None


def extract_detail(url):
    soup = get_soup(url)
    if not soup:
        return "", "", ""

    # Title
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()

    # Publish date
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    # Content
    content_div = soup.find("div", class_="j-fontContent")
    if not content_div:
        content_div = soup.find("div", class_="newscontnet")
    if not content_div:
        content_div = soup.find("div", class_="minh500")

    if content_div:
        # Remove scripts/styles
        for tag in content_div.find_all(["script", "style"]):
            tag.decompose()

        # Process attachments first - find file links
        attachments_list = []
        for a in content_div.find_all("a"):
            href = a.get("href", "")
            if not href:
                continue
            has_file_icon = bool(a.find_previous("img", src=re.compile(r"/assets/images/files2/")))
            if has_file_icon:
                if not href.startswith("http"):
                    href = BASE_URL + href if href.startswith("/") else BASE_URL + "/" + href
                a_text = a.get_text(strip=True)
                if not a_text:
                    a_text = f"附件.{href.split('.')[-1]}"
                markdown_link = f"[{a_text}]({href})"
                new_tag = BeautifulSoup(f"<span>{markdown_link}</span>", "html.parser")
                a.replace_with(new_tag)
                attachments_list.append(markdown_link)

        # Process remaining <a> tags for external links
        for a in content_div.find_all("a"):
            href = a.get("href", "")
            if not href or href.startswith("#"):
                a.unwrap()
                continue
            if not href.startswith("http"):
                if href.startswith("/"):
                    href = BASE_URL + href
                else:
                    href = BASE_URL + "/" + href
            a_text = a.get_text(strip=True)
            if a_text:
                markdown_link = f"[{a_text}]({href})"
                new_tag = BeautifulSoup(f"<span>{markdown_link}</span>", "html.parser")
                a.replace_with(new_tag)
            else:
                a.unwrap()

        # Process images
        for img in content_div.find_all("img"):
            src = img.get("src", "")
            alt = img.get("alt", "")
            if src and not src.startswith("http"):
                src = BASE_URL + src if src.startswith("/") else BASE_URL + "/" + src
            img_tag = f"![{alt}]({src})"
            new_tag = BeautifulSoup(f"<span>{img_tag}</span>", "html.parser")
            img.replace_with(new_tag)

        # Extract content in document order, skip p tags inside tables
        paragraphs = []
        for tag in content_div.find_all(["p", "table"]):
            if tag.name == "table":
                paragraphs.append(str(tag))
            elif tag.name == "p":
                if tag.find_parent("table"):
                    continue
                text = body_text(tag)
                if not text or text in ("\xa0", ""):
                    continue
                text = text.replace("\xa0", " ").strip()
                if text:
                    paragraphs.append(text)

        content = "\n\n".join(paragraphs)
    else:
        content = ""

    return title, pub_date, content


def crawl():
    page = 1
    total_added = 0

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    while True:
        url = LIST_URL.format(page=page)
        print(f"  第{page}页...", end=" ", flush=True)
        soup = get_soup(url)
        if not soup:
            print("FAIL")
            break

        items = soup.find_all("li", class_=re.compile(r"^(odd|even)$"))
        if not items:
            print("空列表，结束")
            break

        print(f"{len(items)}条")
        for item in items:
            a_tag = item.find("a")
            if not a_tag:
                continue
            link = a_tag.get("href", "").strip()
            if not link:
                continue
            if not link.startswith("http"):
                link = BASE_URL + link if link.startswith("/") else BASE_URL + "/" + link
            # Skip external links
            if BASE_URL not in link:
                continue
            list_title = a_tag.get("title", "").strip()
            if not list_title:
                list_title = a_tag.get_text(strip=True)

            # Check if already exists
            c.execute("SELECT 1 FROM gov_raw WHERE page_url=?", (link,))
            if c.fetchone():
                continue

            # Extract detail
            time.sleep(DELAY)
            title, pub_date, content = extract_detail(link)
            if not title:
                title = list_title
            if not content:
                content = f'<p><a href="{link}">{title}</a></p>'
                print(f"    ⚠ {title} - 正文为空，回退为链接")

            # Insert
            try:
                c.execute(
                    """INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, date_rank, summary, script_name) VALUES (?,?,?,?,?,?,?, 'crawl_szyq.py')""",
                    (link, title, SITE_NAME, pub_date, content, pub_date.replace("-", "") if pub_date else "0", title[:200]),
                )
                total_added += 1
            except Exception as e:
                print(f"    ERROR: {e}")

        # Check if last page
        if len(items) < 20:
            break
        page += 1
        time.sleep(DELAY)

    conn.commit()
    conn.close()
    print(f"\n完成: 新增 {total_added} 条")
    return total_added


if __name__ == "__main__":
    print(f"站点: {SITE_NAME} - {COLUMN_NAME}")
    added = crawl()
    print(f"总计新增: {added}")
