#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""明光市人民政府 - 政务动态/通知公告 (zwdt/tzgg) 爬虫
CMS: Lonsun (LS)
列表: ul.doc_list > li > a[href=/zwdt/tzgg/{id}.html] + span.date
分页: /content/column/160685244?pageIndex=N  (total 1254 / 63页, 20/页)
详情: meta ArticleTitle/PubDate + div.con_main (保留表格HTML)
"""
import sys, re, time, sqlite3, ssl, urllib.request
from urllib.parse import urljoin
from bs4 import BeautifulSoup

SITE_NAME = "明光市-通知公告"
SCRIPT_NAME = "crawl_mingguang_tzgg.py"
GROUP_NAME = "安徽"
BASE_URL = "https://www.mingguang.gov.cn"
LIST_FIRST = BASE_URL + "/zwdt/tzgg/index.html"
COLUMN_ID = "160685244"
TOTAL_PAGES = 63
DB_PATH = "/mnt/data/search.db"
OUTPUT_FILE = "/root/gov_crawler/mingguang_tzgg_output.jsonl"

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def http_get(url, timeout=25, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [WARN] 请求失败 {url}: {e}", file=sys.stderr)
                return ""
            time.sleep(1.5)
    return ""


def clean_title(title):
    """strip &middot;&nbsp; 实体前缀 和省略号截断后缀"""
    title = title.replace("&middot;", "").replace("&nbsp;", "").replace("\u00b7", "")
    title = re.sub(r"^[\s\xa0·\u00b7]+", "", title)
    title = re.sub(r"\s*\.{3,}\s*$", "", title)
    return title.strip()


def parse_list(html):
    """从列表页提取 (url, title, date)"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_="doc_list")
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        if not re.search(r"/zwdt/tzgg/\d+\.html$", href):
            continue
        url = urljoin(BASE_URL, href)
        title = clean_title(a.get_text(" ", strip=True))
        if not title and a.get("title"):
            title = clean_title(a["title"])
        date = ""
        sp = li.find("span", class_="date")
        if sp:
            date = sp.get_text(strip=True)
        if title:
            items.append((url, title, date))
    return items


def extract_detail(html, page_url):
    """提取 (title, date, content_html, attachments)"""
    soup = BeautifulSoup(html, "html.parser")

    # 标题
    title = ""
    mt = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"', html)
    if mt:
        title = clean_title(mt.group(1))
    if not title:
        h1 = soup.find("h1", class_="newstitle")
        if h1:
            title = clean_title(h1.get_text(strip=True))

    # 日期
    date = ""
    mp = re.search(r'<meta\s+name="PubDate"\s+content="(\d{4}-\d{2}-\d{2})', html)
    if mp:
        date = mp.group(1)
    if not date:
        dm = re.search(r"发布时间[:：]\s*(\d{4}-\d{2}-\d{2})", html)
        if dm:
            date = dm.group(1)

    # 正文容器
    cm = soup.find("div", class_="con_main")
    if not cm:
        wenzhang = soup.find("div", class_="wenzhang")
        cm = wenzhang.find("div", id="color_printsssss") if wenzhang else None
    if not cm:
        return title, date, "", []

    # 去掉标题/信息/打印关闭/样式噪声
    h1 = cm.find("h1", class_="newstitle")
    if h1:
        h1.decompose()
    ni = cm.find("div", class_="newsinfo")
    if ni:
        ni.decompose()
    # 打印/关闭按钮
    for d in cm.find_all("div"):
        cls = " ".join(d.get("class", []))
        if "print" in cls or "back" in cls or "bfuns" in cls or "dtcode" in cls:
            d.decompose()
    # style
    for s in cm.find_all("style"):
        s.decompose()
    # 隐藏元素
    for h in cm.find_all(style=re.compile(r"display\s*:\s*none", re.I)):
        h.decompose()

    # 定位最内层内容容器（con_main > div.minh500 > div.j-fontContent > p/table）
    content_box = cm
    for cand in cm.find_all("div", class_=re.compile(r"j-fontContent|newscontnet|minh500")):
        content_box = cand
    # 若容器仍是纯 div 包装（内部全是块级 p/table），下探到最像正文的那个
    while True:
        children = content_box.find_all(recursive=False)
        if not children:
            break
        if all(ch.name == "div" for ch in children) and any(ch.find("p") or ch.find("table") for ch in children):
            content_box = max(children, key=lambda d: len(d.get_text("", strip=True)))
        else:
            break

    # 附件收集（先收集后统一处理，替换为完整URL）
    attachments = []
    for a in cm.find_all("a", href=True):
        href = a["href"]
        txt = a.get_text(strip=True)
        if re.search(r"(?i)\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|ofd)(\?|$)", href) or "download" in href.lower() or "file" in href.lower() or txt.startswith("附件"):
            abs_url = urljoin(page_url, href)
            attachments.append((a, txt, abs_url))
    for a, txt, abs_url in attachments:
        new_a = soup.new_tag("a", href=abs_url, target="_blank")
        new_a.string = txt if txt else abs_url.split("/")[-1]
        # 移除附件前的文件类型小图标（xls.gif/pdf.gif 等）：同一父级下紧邻 a 之前的 img
        if a.parent is not None:
            for sib in a.parent.find_all("img", src=re.compile(r"(?i)\.(gif|png|jpg|jpeg)")):
                if sib in list(a.parent.contents)[:list(a.parent.contents).index(a)]:
                    sib.decompose()
        a.replace_with(new_a)
    attached_hrefs = {abs_url for _, _, abs_url in attachments}

    # 提取正文段落（保留表格HTML；含附件链接的段落保留HTML使URL内嵌）
    parts = []
    for el in content_box.find_all(recursive=False):
        if el.name == "table":
            parts.append(str(el))
        elif el.name in ("p", "div", "h1", "h2", "h3", "ul", "ol", "li"):
            # 跳过纯空段落（剥标签后判空）
            txt = el.get_text("", strip=True)
            if not txt:
                continue
            inner_tables = el.find_all("table")
            el_has_attach = any(
                a for a in el.find_all("a", href=True)
                if urljoin(page_url, a["href"]) in attached_hrefs
            )
            if inner_tables or el_has_attach:
                parts.append(str(el))
                continue
            parts.append(txt)
        else:
            txt = el.get_text("", strip=True)
            if txt:
                parts.append(txt)

    # 段落去重（完全相同只保留第一条）
    seen = set()
    final_parts = []
    for p in parts:
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if key and key not in seen:
            seen.add(key)
            final_parts.append(p)

    # \n\n 分段；仅追加未被段落流包含的附件（去重兜底）
    content = "\n\n".join(final_parts)
    for a, txt, abs_url in attachments:
        if abs_url not in content:
            content += f'\n\n<p><a href="{abs_url}" target="_blank">{txt}</a></p>'

    # 清理遗留打印噪声
    content = re.sub(r"打印本页[^\n]*", "", content)
    content = re.sub(r"\n{3,}", "\n\n", content).strip()
    return title, date, content, attachments


def main():
    max_pages = 1
    args = sys.argv[1:]
    i = 0
    while i < len(args):
        a = args[i]
        if a.startswith("--pages="):
            max_pages = int(a.split("=", 1)[1])
        elif a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
            i += 1
        i += 1

    conn = sqlite3.connect(DB_PATH, timeout=30)
    c = conn.cursor()
    total_new = 0
    total_dup = 0
    total_skip = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = LIST_FIRST
        else:
            list_url = f"{BASE_URL}/content/column/{COLUMN_ID}?pageIndex={page}"
        html = http_get(list_url)
        if not html:
            print(f"  [WARN] 第{page}页获取失败, 跳过", file=sys.stderr)
            continue
        items = parse_list(html)
        print(f"  第{page}页: 找到 {len(items)} 条")
        for url, title, date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                total_dup += 1
                continue
            dhtml = http_get(url)
            if not dhtml:
                total_skip += 1
                continue
            d_title, d_date, content, attachments = extract_detail(dhtml, url)
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                total_skip += 1
                continue
            if not d_title:
                d_title = title
            if not d_date:
                d_date = date
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if "<table" in content else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (d_title, summary, content, url, url, d_date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                rid = cur.lastrowid
                c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, d_title, SITE_NAME, summary))
                conn.commit()
                total_new += 1
                print(f"    [{d_date}] {d_title[:45]}")
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.4)

    conn.close()
    print(f"\n新增: {total_new}  重复: {total_dup}  过短/404: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
