#!/usr/bin/env python3
"""
华坪县人民政府 - 公示公告
https://www.huaping.gov.cn/xljhpx/c101699/zfxxgk_nrz.shtml
CMS: 丽江市网站群, 静态分页
列表: ul>li><a>标题<b>日期
分页: zfxxgk_nrz.shtml(1) + zfxxgk_nrz_N.shtml(2-20), 426条
详情: div.TRS_Editor 正文
"""

import sys, os, re, json, requests, time
from bs4 import BeautifulSoup

BASE_URL = "https://www.huaping.gov.cn"
LIST_URL = "https://www.huaping.gov.cn/xljhpx/c101699/zfxxgk_nrz.shtml"
SITE_NAME = "华坪县人民政府-公示公告"
MAX_PAGES = 5

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}


def extract_table_tags(html_str):
    markers = []
    def replace_table(m):
        idx = len(markers)
        key = f"___TBL_{idx}___"
        markers.append((key, m.group(0)))
        return key
    modified = re.sub(
        r"<table[^>]*>.*?</table>",
        replace_table,
        html_str,
        flags=re.DOTALL | re.IGNORECASE
    )
    return modified, markers


def fetch_list_page(page_url):
    """获取列表页"""
    try:
        r = requests.get(page_url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        r.raise_for_status()
    except:
        return []

    items = []
    soup = BeautifulSoup(r.text, "html.parser")
    # Find the content ul
    content_div = soup.find("div", class_="zfxxgk_zdgkc")
    if not content_div:
        # Try alternate container
        content_div = soup.find("div", class_="scroll_main")
    if not content_div:
        return items

    ul = content_div.find("ul")
    if not ul:
        return items

    for li in ul.find_all("li", recursive=False):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        title = a.get_text(strip=True)
        if not title or not href:
            continue
        full_url = href if href.startswith("http") else f"{BASE_URL}{href}"
        b = li.find("b")
        date_str = b.get_text(strip=True) if b else ""
        items.append({"title": title, "url": full_url, "date": date_str})

    return items


def get_content(url):
    """获取详情页正文"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        r.raise_for_status()
    except:
        return "", "", [], ""

    soup = BeautifulSoup(r.text, "html.parser")
    full_title = ""
    if soup.title:
        full_title = soup.title.get_text(strip=True)

    pub_date = ""
    dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", r.text[:3000])
    if dm:
        pub_date = dm.group(1)

    content_html = ""
    body = soup.select_one("div.TRS_Editor") or soup.find(id="zoom") or soup.select_one(".newscontnet, .j-fontContent")

    attachments = []
    if body:
        # 替换附件链接
        for a_tag in body.find_all("a", href=True):
            h = a_tag["href"].strip()
            if re.search(r"\.(docx?|pdf|pptx?|xlsx?)($|\?)", h, re.IGNORECASE):
                fname = a_tag.get_text(strip=True) or h.split("/")[-1]
                if h.startswith("http"):
                    attach_url = h
                elif h.startswith("/"):
                    attach_url = f"{BASE_URL}{h}"
                else:
                    # 相对路径: 基于文章URL的目录
                    base_dir = url.rsplit("/", 1)[0]
                    attach_url = f"{base_dir}/{h}"
                attachments.append({"url": attach_url, "name": fname})
                md_link = f"[{fname}]({attach_url})"
                a_tag.replace_with(md_link)

        # 保存表格
        body_html = str(body)
        body_html, table_markers = extract_table_tags(body_html)

        body_soup = BeautifulSoup(body_html, "html.parser")
        for tag in body_soup.find_all(["script", "style"]):
            tag.decompose()
        for tag in body_soup.find_all(["span", "font", "b", "strong", "i", "em", "u"]):
            tag.unwrap()
        for p in body_soup.find_all("p"):
            p.insert(0, "\n\n")
            p.unwrap()
        for d in body_soup.find_all("div"):
            d.insert(0, "\n")
            d.unwrap()
        for br in body_soup.find_all("br"):
            br.replace_with("\n")

        content_html = body_soup.get_text(separator="", strip=False)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html).strip()

        for key, tbl_html in table_markers:
            content_html = content_html.replace(key, f"\n\n{tbl_html}\n\n")

    if not content_html or len(content_html.strip()) < 20:
        content_html = f'<p><a href="{url}">{full_title or "无正文"}</a></p>'

    return content_html, pub_date, attachments, full_title


def main():
    import sqlite3

    # 收集列表页URL
    page_urls = [LIST_URL]
    for pi in range(2, MAX_PAGES + 1):
        # URL: zfxxgk_nrz.shtml -> zfxxgk_nrz_2.shtml
        filename = f"zfxxgk_nrz_{pi}.shtml"
        page_urls.append(f"https://www.huaping.gov.cn/xljhpx/c101699/{filename}")

    all_items = []
    for pi, page_url in enumerate(page_urls, 1):
        items = fetch_list_page(page_url)
        if not items:
            continue
        print(f"第{pi}页: {len(items)} 条", file=sys.stderr)

        for i, item in enumerate(items):
            print(f"  [{pi}.{i+1}] {item['title'][:40]}...", end=" ", file=sys.stderr)
            content, pub_date, attachments, full_title = get_content(item["url"])
            date = item["date"] or pub_date
            use_title = full_title or item["title"]
            all_items.append({
                "title": use_title[:500],
                "page_url": item["url"],
                "content": content,
                "publish_date": date[:20] if date else "",
                "site_name": SITE_NAME,
                "source_url": item["url"],
                "status": "synced",
                "attachments": json.dumps(attachments, ensure_ascii=False),
            })
            print(f"OK (body:{len(content)}B, attach:{len(attachments)})", file=sys.stderr)
            time.sleep(0.3)

    if not all_items:
        print("no data", file=sys.stderr)
        return

    DB_PATH = "/root/search.db"
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=OFF")

    added = 0
    for item in all_items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?,?,?,?,?,?,?,?)",
                (item["title"][:500], item["page_url"][:1000], item["content"],
                 item["publish_date"][:20], item["site_name"], item["page_url"],
                 "synced", item["attachments"])
            )
            if db.total_changes > 0:
                added += 1
        except Exception as e:
            print(f"  DB ERROR: {e}", file=sys.stderr)
    db.commit()

    db.execute("""
        INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary)
        SELECT r.rowid, r.title, r.site_name, substr(r.content, 1, 500)
        FROM gov_raw r
        WHERE r.site_name = ?
        AND r.content IS NOT NULL AND r.content != ''
        AND r.rowid NOT IN (SELECT rowid FROM gov_search)
    """, (SITE_NAME,))
    db.commit()
    db.close()

    print(f"DB: +{added} / {len(all_items)} items", file=sys.stderr)

    for item in all_items:
        item["attachments"] = json.loads(item["attachments"])
        print(json.dumps(item, ensure_ascii=False))


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"time: {time.time()-t0:.1f}s", file=sys.stderr)
