#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""安宁市人民政府 - 公示公告 (gsgg) 爬虫
站点: www.kman.gov.cn (安宁市, 云南昆明)
列表: /zfxxgk/fdzdgknr/gsgg/ (index.shtml) 
  分页: 2-10页 index_{N}.shtml, >10页 zcms/catalog/36648/pc/index_{N}.shtml, 总133页×25条≈3325条
  条目: div.data-table-item > p.w659 > a[href=/c/{date}/{id}.shtml] 标题 + p.w80 > a[title=日期]
详情: /c/{YYYY-MM-DD}/{id}.shtml
  meta ArticleTitle/PubDate + div.content > h1 + div.activity 正文
"""
import sys, re, time, sqlite3, ssl, urllib.request
from urllib.parse import urljoin
from bs4 import BeautifulSoup

SITE_NAME = "安宁市-公示公告"
SCRIPT_NAME = "crawl_kman_gsgg.py"
GROUP_NAME = "云南"
BASE_URL = "http://www.kman.gov.cn"
LIST_FIRST = BASE_URL + "/zfxxgk/fdzdgknr/gsgg/"
CATALOG_URL = BASE_URL + "/zcms/catalog/36648/pc/index_{N}.shtml"
TOTAL_PAGES = 133
DB_PATH = "/mnt/data/search.db"
OUTPUT_FILE = "/root/gov_crawler/kman_gsgg_output.jsonl"

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def http_get(url, timeout=25, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [WARN] 请求失败 {url}: {e}", file=sys.stderr)
                return ""
            time.sleep(1.5)
    return ""


def clean_title(title):
    """strip &middot;&nbsp;&#32; 实体前缀 和省略号截断后缀"""
    title = title.replace("&middot;", "").replace("&nbsp;", "").replace("&#32;", "").replace("\u00b7", "")
    title = re.sub(r"^[\s\xa0·\u00b7]+", "", title)
    title = re.sub(r"\s*\.{3,}\s*$", "", title)
    return title.strip()


def parse_list(html):
    """从列表页提取 (url, title, date)"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for item in soup.find_all("div", class_="data-table-item"):
        a = item.find("a", href=re.compile(r"/c/\d{4}-\d{2}-\d{2}/\d+\.shtml"))
        if not a:
            continue
        href = a["href"]
        url = urljoin(BASE_URL, href)
        title = clean_title(a.get_text(" ", strip=True))
        # 日期: p.w80 > a[title=完整日期]
        date = ""
        dm = re.search(r"/(\d{4}-\d{2}-\d{2})/\d+\.shtml$", href)
        if dm:
            date = dm.group(1)
        if not date:
            for sp in item.find_all("a", title=True):
                tm = re.search(r"(\d{4}-\d{2}-\d{2})", sp["title"])
                if tm:
                    date = tm.group(1)
                    break
        if title:
            items.append((url, title, date))
    return items


def extract_detail(html, page_url):
    """提取 (title, date, content_html, attachments)"""
    soup = BeautifulSoup(html, "html.parser")

    # 标题 (meta ArticleTitle 优先)
    title = ""
    mt = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"', html)
    if mt:
        title = clean_title(mt.group(1))
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = clean_title(h1.get_text(strip=True))

    # 日期
    date = ""
    mp = re.search(r'<meta\s+name="PubDate"\s+content="(\d{4}-\d{2}-\d{2})', html)
    if mp:
        date = mp.group(1)
    if not date:
        dm = re.search(r"发布(?:时间|日期)[:：]\s*(\d{4}-\d{2}-\d{2})", html)
        if dm:
            date = dm.group(1)

    # 正文容器: div.activity
    cm = soup.find("div", class_="activity")
    if not cm:
        return title, date, "", []

    # 去掉噪声
    for s in cm.find_all("script"):
        s.decompose()
    for st in cm.find_all("style"):
        st.decompose()
    for h in cm.find_all(style=re.compile(r"display\s*:\s*none", re.I)):
        h.decompose()

    # 附件收集
    attachments = []
    seen_att = set()
    for a in cm.find_all("a", href=True):
        href = a["href"]
        txt = a.get_text(strip=True)
        if re.search(r"(?i)\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|ofd)(\?|$)", href) or "download" in href.lower() or "attach" in href.lower() or txt.startswith("附件"):
            abs_url = urljoin(page_url, href)
            if abs_url in seen_att:
                continue
            seen_att.add(abs_url)
            attachments.append((a, txt, abs_url))
    for a, txt, abs_url in attachments:
        new_a = soup.new_tag("a", href=abs_url, target="_blank")
        new_a.string = txt if txt else abs_url.split("/")[-1]
        if a.parent is not None:
            for sib in a.parent.find_all("img", src=re.compile(r"(?i)\.(gif|png|jpg|jpeg)")):
                if sib in list(a.parent.contents)[:list(a.parent.contents).index(a)]:
                    sib.decompose()
        a.replace_with(new_a)
    attached_hrefs = {abs_url for _, _, abs_url in attachments}

    # 正文段落提取 (activity 直接 p 子元素, 保留表格)
    parts = []
    for el in cm.find_all(recursive=False):
        if el.name == "table":
            parts.append(str(el))
        elif el.name in ("p", "div", "h1", "h2", "h3", "ul", "ol", "li"):
            txt = el.get_text("", strip=True)
            if not txt:
                continue
            inner_tables = el.find_all("table")
            el_has_attach = any(
                a for a in el.find_all("a", href=True)
                if urljoin(page_url, a["href"]) in attached_hrefs
                or a["href"] in attached_hrefs
            )
            if inner_tables or el_has_attach:
                parts.append(str(el))
                continue
            parts.append(txt)
        else:
            txt = el.get_text("", strip=True)
            if txt:
                parts.append(txt)

    # 段落去重
    seen = set()
    final_parts = []
    for p in parts:
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if key and key not in seen:
            seen.add(key)
            final_parts.append(p)

    # \n\n 分段; 未入流的附件追加
    content = "\n\n".join(final_parts)
    for a, txt, abs_url in attachments:
        if abs_url not in content:
            content += f'\n\n<p><a href="{abs_url}" target="_blank">{txt}</a></p>'

    # 清理
    content = re.sub(r"打印本页[^\n]*", "", content)
    content = re.sub(r"\n{3,}", "\n\n", content).strip()
    return title, date, content, attachments


def main():
    max_pages = 1
    args = sys.argv[1:]
    i = 0
    while i < len(args):
        a = args[i]
        if a.startswith("--pages="):
            max_pages = int(a.split("=", 1)[1])
        elif a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
            i += 1
        i += 1

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()
    total_new = 0
    total_dup = 0
    total_skip = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = LIST_FIRST
        elif page <= 10:
            list_url = BASE_URL + f"/zfxxgk/fdzdgknr/gsgg/index_{page}.shtml"
        else:
            list_url = CATALOG_URL.replace("{N}", str(page))
        html = http_get(list_url)
        if not html:
            print(f"  [WARN] 第{page}页获取失败, 跳过", file=sys.stderr)
            continue
        items = parse_list(html)
        print(f"  第{page}页: 找到 {len(items)} 条")
        for url, title, date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                total_dup += 1
                continue
            dhtml = http_get(url)
            if not dhtml:
                total_skip += 1
                continue
            d_title, d_date, content, attachments = extract_detail(dhtml, url)
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                total_skip += 1
                continue
            if not d_title:
                d_title = title
            if not d_date:
                d_date = date
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if "<table" in content else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (d_title, summary, content, url, url, d_date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                rid = cur.lastrowid
                c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, d_title, SITE_NAME, summary))
                conn.commit()
                total_new += 1
                print(f"    [{d_date}] {d_title[:45]}")
            except sqlite3.OperationalError as e:
                if "locked" in str(e).lower():
                    # 调度器并发写库 → 重试
                    for attempt in range(3):
                        time.sleep(8 + attempt * 5)
                        try:
                            conn.rollback()
                            cur = c.execute(
                                "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                                (d_title, summary, content, url, url, d_date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                            rid = cur.lastrowid
                            c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                                      (rid, d_title, SITE_NAME, summary))
                            conn.commit()
                            total_new += 1
                            print(f"    [{d_date}] {d_title[:45]} (重试{attempt+1})")
                            break
                        except sqlite3.OperationalError as e2:
                            if attempt == 2:
                                total_skip += 1
                                print(f"    [SKIP-锁] {d_title[:40]}: {e2}")
                else:
                    total_skip += 1
                    print(f"    [SKIP] {d_title[:40]}: {e}")
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.4)

    conn.close()
    print(f"\n新增: {total_new}  重复: {total_dup}  过短/404: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
