#!/usr/bin/env python3
"""
沧源佤族自治县 - 环境保护
https://cangyuan.gov.cn/zfxxgk_cyx_sthjj/hjbh.html
CMS: 自定义, 静态HTML分页
列表: <dl class="xxgk-textList1 line1"><dd><a title="完整标题" href="/zfxxgk_cyx_sthjj/artview/263/{ID}.html">标题</a><span>日期</span>
分页: hjbh.html(1) + hjbh/index_N.html(2-8), 106条
详情: div.xxgk-articleBox
SSL修复: --curves prime256v1
"""

import sys, os, re, json, subprocess, time
from bs4 import BeautifulSoup

BASE_URL = "https://cangyuan.gov.cn"
LIST_URL = "https://cangyuan.gov.cn/zfxxgk_cyx_sthjj/hjbh.html"
SITE_NAME = "沧源佤族自治县-环境保护"
MAX_PAGES = 5  # 前5页

HEADERS = ["User-Agent: Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"]


def curl_get(url):
    """使用curl带--curves prime256v1获取页面"""
    try:
        result = subprocess.run(
            ["curl", "-sL", "--max-time", "20", "--curves", "prime256v1", url],
            capture_output=True, text=True, timeout=25
        )
        return result.stdout
    except:
        return ""


def extract_table_tags(html_str):
    """从HTML字符串中提取所有<table>标签，用唯一标记替换"""
    markers = []
    def replace_table(m):
        idx = len(markers)
        key = f"___TBL_{idx}___"
        markers.append((key, m.group(0)))
        return key
    modified = re.sub(
        r"<table[^>]*>.*?</table>",
        replace_table,
        html_str,
        flags=re.DOTALL | re.IGNORECASE
    )
    return modified, markers


def fetch_list_page(page_url):
    """获取列表页"""
    html = curl_get(page_url)
    if not html or len(html) < 1000:
        print(f"  EMPTY page: {page_url}", file=sys.stderr)
        return []

    items = []
    soup = BeautifulSoup(html, "html.parser")
    dl = soup.select_one("dl.xxgk-textList1")
    if not dl:
        print(f"  NO list dl found", file=sys.stderr)
        return items

    for dd in dl.find_all("dd"):
        a = dd.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        title = a.get("title", "").strip() or a.get_text(strip=True)
        if not title:
            continue
        full_url = href if href.startswith("http") else f"{BASE_URL}{href}"
        date_span = dd.find("span")
        date_str = date_span.get_text(strip=True) if date_span else ""
        items.append({"title": title, "url": full_url, "date": date_str})

    return items


def get_content(url):
    """获取详情页正文"""
    html = curl_get(url)
    if not html or len(html) < 1000:
        return "", "", [], ""

    soup = BeautifulSoup(html, "html.parser")
    full_title = ""
    pub_date = ""

    # 标题
    title_tag = soup.find("title")
    if title_tag:
        full_title = title_tag.get_text(strip=True)

    # 日期
    dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", html[:3000])
    if dm:
        pub_date = dm.group(1)

    # 正文
    content_html = ""
    body = soup.select_one("div.xxgk-articleBox")
    if not body:
        body = soup.select_one("div.xxgk-singleArticle")

    attachments = []
    if body:
        # 替换附件链接
        for a_tag in body.find_all("a", href=True):
            h = a_tag["href"].strip()
            if re.search(r"\.(docx?|pdf|pptx?|xlsx?)($|\?)", h, re.IGNORECASE):
                fname = a_tag.get_text(strip=True) or h.split("/")[-1]
                attach_url = h if h.startswith("http") else f"{BASE_URL}{h}"
                attachments.append({"url": attach_url, "name": fname})
                md_link = f"[{fname}]({attach_url})"
                a_tag.replace_with(md_link)

        # 保存表格
        body_html = str(body)
        body_html, table_markers = extract_table_tags(body_html)

        body_soup = BeautifulSoup(body_html, "html.parser")
        for tag in body_soup.find_all(["script", "style"]):
            tag.decompose()
        for tag in body_soup.find_all(["span", "font", "b", "strong", "i", "em", "u"]):
            tag.unwrap()
        for p in body_soup.find_all("p"):
            p.insert(0, "\n\n")
            p.unwrap()
        for d in body_soup.find_all("div"):
            d.insert(0, "\n")
            d.unwrap()
        for br in body_soup.find_all("br"):
            br.replace_with("\n")

        content_html = body_soup.get_text(separator="", strip=False)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html).strip()

        for key, tbl_html in table_markers:
            content_html = content_html.replace(key, f"\n\n{tbl_html}\n\n")

    if not content_html or len(content_html.strip()) < 20:
        content_html = f"[{full_title or '无正文'}]({url})"

    return content_html, pub_date, attachments, full_title


def main():
    import sqlite3

    # 收集所有列表页URL
    page_urls = [LIST_URL]
    for pi in range(2, MAX_PAGES + 1):
        page_urls.append(f"https://cangyuan.gov.cn/zfxxgk_cyx_sthjj/hjbh/index_{pi}.html")

    all_items = []
    for pi, page_url in enumerate(page_urls, 1):
        items = fetch_list_page(page_url)
        if not items:
            continue
        print(f"第{pi}页: {len(items)} 条", file=sys.stderr)

        for i, item in enumerate(items):
            print(f"  [{pi}.{i+1}] {item['title'][:40]}...", end=" ", file=sys.stderr)
            content, pub_date, attachments, full_title = get_content(item["url"])
            date = item["date"] or pub_date
            use_title = full_title or item["title"]
            all_items.append({
                "title": use_title[:500],
                "page_url": item["url"],
                "content": content,
                "publish_date": date[:20] if date else "",
                "site_name": SITE_NAME,
                "source_url": item["url"],
                "status": "synced",
                "attachments": json.dumps(attachments, ensure_ascii=False),
            })
            print(f"OK (body:{len(content)}B, attach:{len(attachments)})", file=sys.stderr)
            time.sleep(0.5)

    if not all_items:
        print("no data", file=sys.stderr)
        return

    # 写入search.db
    DB_PATH = "/root/search.db"
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=OFF")

    added = 0
    for item in all_items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?,?,?,?,?,?,?,?)",
                (item["title"][:500], item["page_url"][:1000], item["content"],
                 item["publish_date"][:20], item["site_name"], item["page_url"],
                 "synced", item["attachments"])
            )
            if db.total_changes > 0:
                added += 1
        except Exception as e:
            print(f"  DB ERROR: {e}", file=sys.stderr)
    db.commit()

    # 增量FTS
    db.execute("""
        INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary)
        SELECT r.rowid, r.title, r.site_name, substr(r.content, 1, 500)
        FROM gov_raw r
        WHERE r.site_name = ?
        AND r.content IS NOT NULL AND r.content != ''
        AND r.rowid NOT IN (SELECT rowid FROM gov_search)
    """, (SITE_NAME,))
    db.commit()
    db.close()

    print(f"DB: +{added} / {len(all_items)} items", file=sys.stderr)

    for item in all_items:
        item["attachments"] = json.loads(item["attachments"])
        print(json.dumps(item, ensure_ascii=False))


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"time: {time.time()-t0:.1f}s", file=sys.stderr)
