#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
第二师铁门关市 - 公示公告
https://www.tmg.gov.cn/info/iIndex.jsp?catalog_id=13&cat_id=13&cur_page=1

CMS: 新疆兵团信息公开平台 (同btnsss)
JS分页: pageCount=181, 16条/页
catalog_id=13 公示公告
"""
import sys, os, re, time, json, requests
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "https://www.tmg.gov.cn/",
}
DOMAIN = "https://www.tmg.gov.cn"
LIST_URL = "https://www.tmg.gov.cn/info/iIndex.jsp?catalog_id=13&cat_id=13"
SITE_NAME = "tmg_tzgg"
MAX_PAGES = 5


def extract_table_tags(html_str):
    """Replace <table> with placeholders to preserve HTML tables"""
    tbl_soup = BeautifulSoup(html_str, "html.parser")
    tables = tbl_soup.find_all("table")
    replacements = []
    for i, tb in enumerate(tables):
        key = "___TBL_%d___" % i
        replacements.append((key, str(tb)))
        tb.replace_with(key)
    return str(tbl_soup), replacements


def fetch(url, encoding="utf-8"):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = encoding
        return r.text
    except Exception as e:
        print("  FETCH ERROR %s: %s" % (url, e), file=sys.stderr)
        return None


def parse_list_items(html):
    items = []
    soup = BeautifulSoup(html, "lxml")
    gk = soup.find("div", class_="gk-nr-ty")
    if not gk:
        return items
    for li in gk.find_all("li"):
        a = li.find("a", class_="gk-br")
        sp = li.find("span")
        if not a or not sp or not a.get("href"):
            continue
        href = a["href"].strip()
        title = a.get_text(strip=True)
        # Remove bullet prefix
        title = re.sub(r"^[•·\s]+", "", title).strip()
        if not title or len(title) < 5:
            continue
        date = sp.get_text(strip=True)[:10]
        full_url = href if href.startswith("http") else DOMAIN + href
        items.append({
            "url": full_url,
            "title": title,
            "date": date,
        })
    return items


def get_content(url, title_from_list):
    html = fetch(url)
    if not html:
        return "", "", [], title_from_list

    soup = BeautifulSoup(html, "lxml")

    # 完整标题：meta ArticleTitle > div.con-tt > 回退列表标题
    full_title = title_from_list
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content", "").strip():
        full_title = meta_title["content"].strip()
    else:
        con_tt = soup.select_one("div.con-tt")
        if con_tt:
            t = con_tt.get_text(strip=True)
            if t:
                full_title = t

    # 日期：meta PubDate > 文件元数据表 > HTML全文正则
    pub_date = ""
    meta_pub = soup.find("meta", attrs={"name": "PubDate"})
    if meta_pub and meta_pub.get("content", "").strip():
        m = re.search(r"(\d{4}[-/]\d{2}[-/]\d{2})", meta_pub["content"])
        if m:
            pub_date = m.group(1).replace("/", "-")
    if not pub_date:
        file_table = soup.select_one("div.file-table")
        if file_table:
            m = re.search(r"发布日期.*?<td[^>]*>(\d{4}[-/]\d{2}[-/]\d{2})", str(file_table), re.DOTALL)
            if m:
                pub_date = m.group(1).replace("/", "-")
    if not pub_date:
        m = re.search(r"(\d{4}[-/]\d{2}[-/]\d{2})\s*\d{2}:\d{2}", html)
        if m:
            pub_date = m.group(1).replace("/", "-")

    # 正文
    body = soup.select_one("div.con-nr")
    content_html = ""
    attachments = []

    if body:
        # 附件链接
        for a_tag in body.find_all("a", href=True):
            h = a_tag["href"].strip()
            if re.search(r"\.(docx?|pdf|pptx?|xlsx?|jpg|png|gif)($|\?)", h, re.IGNORECASE):
                fname = a_tag.get_text(strip=True) or h.split("/")[-1]
                attach_url = h if h.startswith("http") else DOMAIN + h
                attachments.append({"url": attach_url, "name": fname})
                md_link = "[%s](%s)" % (fname, attach_url)
                a_tag.replace_with(md_link)

        # 提取表格
        body_html = str(body)
        body_html, table_markers = extract_table_tags(body_html)

        body_soup = BeautifulSoup(body_html, "html.parser")
        for tag in body_soup.find_all(["script", "style"]):
            tag.decompose()
        for tag in body_soup.find_all(["span", "font", "b", "strong", "i", "em", "u"]):
            tag.unwrap()
        for p in body_soup.find_all("p"):
            p.insert(0, "\n\n")
            p.unwrap()
        for d in body_soup.find_all("div"):
            d.insert(0, "\n")
            d.unwrap()
        for br in body_soup.find_all("br"):
            br.replace_with("\n")

        content_html = body_soup.get_text(separator="", strip=False)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html).strip()

        for key, tbl_html in table_markers:
            content_html = content_html.replace(key, "\n\n" + tbl_html + "\n\n")

    # 全文也扫描附件
    if not attachments:
        for a_tag in soup.find_all("a", href=True):
            h = a_tag["href"].strip().lower()
            if re.search(r"\.(docx?|pdf|pptx?|xlsx?|zip|rar)($|\?)", h, re.IGNORECASE):
                fname = a_tag.get_text(strip=True) or h.split("/")[-1]
                attach_url = a_tag["href"].strip()
                if not attach_url.startswith("http"):
                    attach_url = DOMAIN + attach_url
                attachments.append({"url": attach_url, "name": fname})

    if not content_html or len(content_html.strip()) < 20:
        content_html = '<p><a href="%s">%s</a></p>' % (url, full_title)

    return content_html, pub_date, attachments, full_title


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=0, help="Number of pages to crawl")
    args = parser.parse_args()

    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    c = conn.cursor()

    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    existing = c.fetchone()[0]
    incremental = existing > 0

    max_pages = args.pages if args.pages > 0 else (1 if incremental else MAX_PAGES)
    print("站点 %s: 已有 %d 条, incremental=%s, 爬取 %d 页" % (SITE_NAME, existing, incremental, max_pages), file=sys.stderr)

    all_new = 0
    for pg in range(1, max_pages + 1):
        page_url = LIST_URL if pg == 1 else LIST_URL + "&cur_page=%d" % pg
        html = fetch(page_url)
        if not html:
            print("  第%d页 获取失败" % pg, file=sys.stderr)
            continue

        items = parse_list_items(html)
        if not items:
            print("  第%d页 无条目" % pg, file=sys.stderr)
            break
        print("  第%d页: %d 条" % (pg, len(items)), file=sys.stderr)

        for it in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (it["url"], SITE_NAME))
            if c.fetchone():
                continue

            content, pub_date, attachments, full_title = get_content(it["url"], it["title"])
            if not content.strip() or len(content.strip()) < 20:
                print("    ! 跳过空正文: %s" % full_title[:50], file=sys.stderr)
                continue
            date = it["date"] or pub_date

            try:
                c.execute(
                    """INSERT INTO gov_raw (site_name, page_url, title, content, publish_date, source_url, status, attachments)
                       VALUES (?,?,?,?,?,?,?,?)""",
                    (SITE_NAME, it["url"], full_title[:500], content, date[:20],
                     it["url"], "synced", json.dumps(attachments, ensure_ascii=False))
                )
                new_id = c.lastrowid
                # Sync FTS (gov_search by rowid)
                summary = content[:200].replace("\n", " ")
                # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                conn.commit()
                c.execute(
                    "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?,?,?,?)",
                    (new_id, full_title[:200], SITE_NAME, summary[:200])
                )
                all_new += 1
                print("    + %s (%s) body:%dB attach:%d" % (full_title[:50], date, len(content), len(attachments)), file=sys.stderr)
            except Exception as e:
                print("    ! 入库失败: %s" % e, file=sys.stderr)

        time.sleep(0.3)

    conn.commit()
    conn.close()
    print("\n完成! 新增 %d 条" % all_new, file=sys.stderr)


if __name__ == "__main__":
    main()
