#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""昆山市人民政府-公示公告 (crawl_ksgsgg.py) — 升级修复版
CMS: UCAP
列表: /kss/gsgg/common_list2.shtml (第1页) + common_list2_{N}.shtml (N>=2)
  分页: createPageHTML('page_div',1000, N,'common_list2','shtml',5000) → 1000页 × 5条 = 5000条
  条目: <h4><a href="/kss/gsgg/{YYYYMM}/{uuid}.shtml" title="title">title</a><span class="time">YYYY-MM-DD</span></h4>
详情: /kss/gsgg/{YYYYMM}/{uuid}.shtml
  meta ArticleTitle/PubDate + <UCAPCONTENT> (Word转换段落)
  附件: 正文内 <a href="{uuid}/files/xxx.pdf"> (相对详情页URL)
"""
import sys, re, time, sqlite3, ssl, urllib.request
from urllib.parse import urljoin
from bs4 import BeautifulSoup

SITE_NAME = "昆山市-公示公告"
SCRIPT_NAME = "crawl_ksgsgg.py"
GROUP_NAME = "江苏"
BASE_URL = "https://www.ks.gov.cn"
LIST_URL = f"{BASE_URL}/kss/gsgg/common_list2.shtml"
TOTAL_PAGES = 1000
DB_PATH = "/mnt/data/search.db"
OUTPUT_FILE = "/root/gov_crawler/ksgsgg_output.jsonl"

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def http_get(url, timeout=25, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [WARN] 请求失败 {url}: {e}", file=sys.stderr)
                return ""
            time.sleep(1.5)
    return ""


def clean_title(title):
    """strip &middot;&nbsp; 实体前缀 和省略号截断后缀"""
    title = title.replace("&middot;", "").replace("&nbsp;", "").replace("\u00b7", "")
    title = re.sub(r"^[\s\xa0·\u00b7]+", "", title)
    title = re.sub(r"\s*\.{3,}\s*$", "", title)
    return title.strip()


def parse_list(html):
    """从列表页提取 (url, title, date) — 只抓本站 /kss/gsgg/ 详情链接"""
    items = []
    for m in re.finditer(
        r'<h4><a\s+href="([^"]+)"\s+title="([^"]*)"[^>]*>(.*?)</a>\s*<span\s+class="time">\s*(\d{4}-\d{2}-\d{2})\s*</span>',
        html, re.DOTALL):
        href, t, disp, date = m.group(1), m.group(2), m.group(3), m.group(4)
        # 只抓本站详情 (过滤外链)
        if not re.search(r"/kss/gsgg/\d+[^\"<>]*\.shtml$", href):
            continue
        url = urljoin(BASE_URL, href)
        title = clean_title((t or disp).strip())
        if title:
            items.append((url, title, date))
    return items


def extract_detail(html, page_url):
    """提取 (title, date, content_html, attachments)"""
    soup = BeautifulSoup(html, "html.parser")

    # 标题 (meta ArticleTitle 优先)
    title = ""
    mt = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"', html)
    if mt:
        title = clean_title(mt.group(1))
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = clean_title(h1.get_text(strip=True))

    # 日期
    date = ""
    mp = re.search(r'<meta\s+name="PubDate"\s+content="(\d{4}-\d{2}-\d{2})', html)
    if mp:
        date = mp.group(1)
    if not date:
        dm = re.search(r"时间[:：]\s*(\d{4}-\d{2}-\d{2})", html)
        if dm:
            date = dm.group(1)

    # 正文容器: div.article-content.article-content-body (含 UCAPCONTENT)
    cm = soup.find("div", class_="article-content")
    if cm:
        uc = cm.find("ucapcontent")
        if uc:
            cm = uc
    else:
        m = re.search(r"<UCAPCONTENT>(.*?)</UCAPCONTENT>", html, re.DOTALL)
        if m:
            cm = BeautifulSoup(m.group(1), "html.parser")
    if not cm:
        return title, date, "", []

    # 去掉噪声
    for s in cm.find_all("script"):
        s.decompose()
    for st in cm.find_all("style"):
        st.decompose()
    for h in cm.find_all(style=re.compile(r"display\s*:\s*none", re.I)):
        h.decompose()

    # 附件收集 (正文内 a + 附件区)
    attachments = []
    seen_att = set()
    for a in cm.find_all("a", href=True):
        href = a["href"]
        txt = a.get_text(strip=True)
        if re.search(r"(?i)\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|ofd)(\?|$)", href) or "download" in href.lower() or "file" in href.lower() or txt.startswith("附件"):
            abs_url = urljoin(page_url, href)
            if abs_url in seen_att:
                continue
            seen_att.add(abs_url)
            attachments.append((a, txt, abs_url))
    for a, txt, abs_url in attachments:
        new_a = soup.new_tag("a", href=abs_url, target="_blank")
        new_a.string = txt if txt else abs_url.split("/")[-1]
        if a.parent is not None:
            for sib in a.parent.find_all("img", src=re.compile(r"(?i)\.(gif|png|jpg|jpeg)")):
                if sib in list(a.parent.contents)[:list(a.parent.contents).index(a)]:
                    sib.decompose()
        a.replace_with(new_a)
    attached_hrefs = {abs_url for _, _, abs_url in attachments}

    # 正文段落提取 (Word 转换 p 段落, 保留表格)
    parts = []
    for el in cm.find_all(recursive=False):
        if el.name == "table":
            parts.append(str(el))
        elif el.name in ("p", "div", "h1", "h2", "h3", "ul", "ol", "li"):
            txt = el.get_text("", strip=True)
            if not txt:
                continue
            inner_tables = el.find_all("table")
            el_has_attach = any(
                a for a in el.find_all("a", href=True)
                if urljoin(page_url, a["href"]) in attached_hrefs
                or a["href"] in attached_hrefs
            )
            if inner_tables or el_has_attach:
                parts.append(str(el))
                continue
            parts.append(txt)
        else:
            txt = el.get_text("", strip=True)
            if txt:
                parts.append(txt)

    # 段落去重
    seen = set()
    final_parts = []
    for p in parts:
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if key and key not in seen:
            seen.add(key)
            final_parts.append(p)

    # \n\n 分段; 未入流的附件追加
    content = "\n\n".join(final_parts)
    for a, txt, abs_url in attachments:
        if abs_url not in content:
            content += f'\n\n<p><a href="{abs_url}" target="_blank">{txt}</a></p>'

    # 清理
    content = re.sub(r"打印本页[^\n]*", "", content)
    content = re.sub(r"\n{3,}", "\n\n", content).strip()
    return title, date, content, attachments


def main():
    max_pages = 1
    args = sys.argv[1:]
    i = 0
    while i < len(args):
        a = args[i]
        if a.startswith("--pages="):
            max_pages = int(a.split("=", 1)[1])
        elif a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
            i += 1
        i += 1

    conn = sqlite3.connect(DB_PATH, timeout=30)
    c = conn.cursor()
    total_new = 0
    total_dup = 0
    total_skip = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = LIST_URL
        else:
            list_url = f"{BASE_URL}/kss/gsgg/common_list2_{page}.shtml"
        html = http_get(list_url)
        if not html:
            print(f"  [WARN] 第{page}页获取失败, 跳过", file=sys.stderr)
            continue
        items = parse_list(html)
        print(f"  第{page}页: 找到 {len(items)} 条")
        for url, title, date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                total_dup += 1
                continue
            dhtml = http_get(url)
            if not dhtml:
                total_skip += 1
                continue
            d_title, d_date, content, attachments = extract_detail(dhtml, url)
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                total_skip += 1
                continue
            if not d_title:
                d_title = title
            if not d_date:
                d_date = date
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if "<table" in content else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (d_title, summary, content, url, url, d_date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                rid = cur.lastrowid
                c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, d_title, SITE_NAME, summary))
                conn.commit()
                total_new += 1
                print(f"    [{d_date}] {d_title[:45]}")
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.4)

    conn.close()
    print(f"\n新增: {total_new}  重复: {total_dup}  过短/404: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
