#!/usr/bin/env python3
"""
沙湾市人民政府 - 生态环境
URL: https://www.xjsw.gov.cn/xxgk/fdzdgknr/zdly/sthj2
CMS: 综合CMS (PowerCMS)
List: Static HTML, pagination: sthj2, sthj2_2, sthj2_3, sthj2_4 (4页共67条)
Detail: div.mainContent > article.articleCon > p(正文) + a(附件) + table(元数据)
"""
import sys, os, re, json, time, sqlite3
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

BASE = "https://www.xjsw.gov.cn"
LIST_BASE = BASE + "/xxgk/fdzdgknr/zdly/sthj2"
SITE = "沙湾市人民政府-生态环境"
COLUMN = "生态环境"
PROVINCE = "新疆"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": BASE + "/xxgk/fdzdgknr/zdly/sthj2",
}
session = requests.Session()
session.headers.update(HEADERS)
session.verify = False


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def fetch(url, retries=3):
    for attempt in range(retries):
        try:
            r = session.get(url, timeout=30)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            log(f"  [WARN] 请求失败 (尝试 {attempt+1}/{retries}): {e}")
            time.sleep(2)
    return ""


def parse_list_page(html):
    """解析列表页"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    # 列表在 <ul class="xxgk-list">
    ul = soup.find("ul", class_="xxgk-list")
    if not ul:
        # fallback: 找包含 content_ 链接的 ul
        for u in soup.find_all("ul"):
            if u.find("a", href=re.compile(r"/content_\d+")):
                ul = u
                break
    if not ul:
        return items

    for li in ul.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a.get("href", "")
        if not re.search(r"/content_\d+", href):
            continue
        # 标题：从 a 的 title 属性提取，或从 a 标签文本
        title_attr = a.get("title", "")
        title = ""
        if title_attr:
            m = re.search(r"标题：(.+)", title_attr)
            if m:
                title = m.group(1).strip()
        if not title:
            title = a.get_text(strip=True)

        # 日期：从 li 中的 span.date 或 title 属性
        date_str = ""
        span_date = li.find("span", class_="date")
        if span_date:
            date_str = span_date.get_text(strip=True)
        if not date_str:
            m = re.search(r"发表时间：(\d{4}-\d{2}-\d{2})", title_attr)
            if m:
                date_str = m.group(1)

        items.append({
            "url": urljoin(BASE, href),
            "title": title.strip(),
            "date": date_str,
        })
    return items


def fetch_detail(url):
    """提取详情页: 标题、正文、附件"""
    html = fetch(url)
    if not html:
        return "", "", [], ""

    soup = BeautifulSoup(html, "html.parser")
    for t in soup(["script", "style"]):
        t.decompose()

    # 正文区域
    main_content = soup.find("div", class_="mainContent")
    if not main_content:
        main_content = soup

    article = main_content.find("article", class_="articleCon")
    if not article:
        # fallback: 任何有 <p> 文本的容器
        article = main_content

    # 标题：从 articleCon 内的 h2 获取
    title = ""
    if article:
        h2 = article.find("h2")
        if h2:
            title = h2.get_text(strip=True)
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        # 从 title 属性或 meta 提取
        for m in soup.find_all("meta"):
            if m.get("name", "").lower() == "articletitle":
                title = m.get("content", "")
                break

    # 日期: meta 或搜索
    date_str = ""
    for m in soup.find_all("meta"):
        nm = m.get("name", "").lower()
        if "pubdate" in nm or "publishdate" in nm or "issuedate" in nm:
            c = m.get("content", "")
            if c:
                date_str = c[:10]
                break
    if not date_str:
        m = re.search(r"发布日期：(\d{4}-\d{2}-\d{2})", html)
        if m:
            date_str = m.group(1)
    if not date_str:
        m = re.search(r"发布时间：(\d{4}-\d{2}-\d{2})", html)
        if m:
            date_str = m.group(1)

    # 正文内容 (段落 + 表格)
    content_parts = []
    attachments = []

    if article:
        for child in article.find_all(["p", "div"], recursive=True):
            if child.name == "p":
                t = child.get_text("", strip=True)
                # 跳过元数据行和空行
                if t and len(t) > 5 and "索引号" not in t and "发布机构" not in t and "所属主题" not in t and "发布日期" not in t and "浏览次数" not in t and "字体" not in t and "打印正文" not in t and "分享到" not in t and "【" not in t:
                    # 跳过纯附件行
                    has_attach = any(
                        a_href.get("href","").lower().endswith(ext)
                        for a_href in child.find_all("a", href=True)
                        for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx"]
                    )
                    if not has_attach or len(t) > 20:
                        content_parts.append(t)

            elif child.name == "table" and "govDetailTable" not in child.get("class", []):
                # 非元数据表格保留HTML
                t_html = str(child)
                content_parts.append(f"\n[表格]\n{t_html}\n[/表格]")

        # 附件
        for a in article.find_all("a", href=True):
            href = a["href"]
            fname = a.get_text(strip=True)
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|txt)$', href, re.I):
                full_url = urljoin(BASE, href)
                attachments.append({"name": fname or "附件", "url": full_url})

    # 如果没有<p>正文但有关联附件
    if not content_parts and attachments:
        content_parts.append(f"[本公告为附件文件格式，详见附件]")
        for att in attachments:
            content_parts.append(f"[附件: {att['name']}]({att['url']})")

    content = "\n\n".join(content_parts) if content_parts else ""
    return title, content, attachments, date_str


def import_to_db(record):
    try:
        db = sqlite3.connect(DB_PATH, timeout=10)
        title = (record.get("title") or "")[:500]
        page_url = (record.get("page_url") or "")[:1000]
        content = record.get("content") or ""
        publish_date = (record.get("publish_date") or "")[:20]
        site_name = (record.get("site_name") or "unknown")[:100]
        attachments_str = json.dumps(record.get("attachments") or [], ensure_ascii=False)

        old = db.execute("SELECT rowid FROM gov_raw WHERE page_url = ?", (page_url,)).fetchone()
        if old:
            db.execute("DELETE FROM gov_search WHERE rowid = ?", (old[0],))
            db.execute("DELETE FROM gov_raw WHERE page_url = ?", (page_url,))

        db.execute(
            "INSERT INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?, ?, ?, ?, ?, ?, 'synced', ?)",
            (title, page_url, content, publish_date, site_name, page_url, attachments_str)
        )
        new_rowid = db.execute("SELECT last_insert_rowid()").fetchone()[0]
        summary = content[:500] if content else title[:500]
        # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
        #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
        db.commit()
        db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                   (new_rowid, title, site_name, summary))
        db.commit()
        db.close()
        log(f"  ✅ {title[:30]}")
        return True
    except Exception as e:
        log(f"  [ERR] DB import failed: {e}")
        return False


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=5,
                        help="爬取前N页（共4页，每页~17条，共67条）")
    args = parser.parse_args()

    total_pages = min(args.pages, 4)  # 该栏目只有4页
    count = 0

    for page in range(1, total_pages + 1):
        page_url = LIST_BASE if page == 1 else f"{LIST_BASE}_{page}"
        log(f"📄 列表[page={page}]: {page_url}")
        html = fetch(page_url)
        if not html:
            break
        items = parse_list_page(html)
        if not items:
            log(f"  → 无数据")
            break
        log(f"  → {len(items)} 条")

        for idx, item in enumerate(items, 1):
            url = item["url"]
            list_title = item["title"]
            log(f"  ({idx}/{len(items)}) {list_title[:40]}")
            detail_title, content, attachments, detail_date = fetch_detail(url)

            record = {
                "title": detail_title or list_title,
                "page_url": url,
                "publish_date": detail_date or item["date"],
                "content": content,
                "attachments": attachments,
                "site_name": SITE,
                "column": COLUMN,
                "province": PROVINCE,
            }
            ok = import_to_db(record)
            if ok:
                count += 1
            time.sleep(0.5)

    log(f"\n✅ {SITE} 爬取完成，共入库 {count} 条")


if __name__ == "__main__":
    main()
