#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""渌口区人民政府 - 通知公告 (c18458) 爬虫
CMS: zzcms/DFS (株洲政府网站群)
列表: /c18458/index.html + /c18458/pages/{N}.html  15条/页, 共149页≈2235条
详情: /c18458/{YYYYMMDD}/i{id}.html
详情页: meta ArticleTitle/PubDate/ContentSource + div.details-content.article-content-body 正文
附件: 正文外, 协议相对 //img.zhuzhou.gov.cn/zzcms/DFS/file/... (urljoin 处理)
"""
import sys, re, time, sqlite3, ssl, urllib.request
from urllib.parse import urljoin
from bs4 import BeautifulSoup

SITE_NAME = "渌口区-通知公告"
SCRIPT_NAME = "crawl_lukou_tzgg.py"
GROUP_NAME = "湖南"
BASE_URL = "http://www.lukou.gov.cn"
LIST_FIRST = BASE_URL + "/c18458/index.html"
TOTAL_PAGES = 149
DB_PATH = "/mnt/data/search.db"
OUTPUT_FILE = "/root/gov_crawler/lukou_tzgg_output.jsonl"

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def http_get(url, timeout=25, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [WARN] 请求失败 {url}: {e}", file=sys.stderr)
                return ""
            time.sleep(1.5)
    return ""


def clean_title(title):
    """strip &middot;&nbsp; 实体前缀 和省略号截断后缀"""
    title = title.replace("&middot;", "").replace("&nbsp;", "").replace("\u00b7", "")
    title = re.sub(r"^[\s\xa0·\u00b7]+", "", title)
    title = re.sub(r"\s*\.{3,}\s*$", "", title)
    return title.strip()


def parse_list(html):
    """从列表页提取 (url, title, date)"""
    items = []
    # 列表条目: href="/c18458/20260730/i2517805.html" 文本
    for m in re.finditer(r'href="([^"]*?/c18458/\d{8}/i\d+\.html)"[^>]*>\s*([^<]{2,80})<', html):
        href, t = m.group(1), m.group(2)
        url = urljoin(BASE_URL, href)
        title = clean_title(t.strip())
        # 日期从 URL 提取 (20260730 -> 2026-07-30)
        dm = re.search(r"/(\d{4})(\d{2})(\d{2})/i\d+\.html$", href)
        date = f"{dm.group(1)}-{dm.group(2)}-{dm.group(3)}" if dm else ""
        if title:
            items.append((url, title, date))
    return items


def extract_detail(html, page_url):
    """提取 (title, date, content_html, attachments)"""
    soup = BeautifulSoup(html, "html.parser")

    # 标题 (meta ArticleTitle 优先)
    title = ""
    mt = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"', html)
    if mt:
        title = clean_title(mt.group(1))
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = clean_title(h1.get_text(strip=True))

    # 日期
    date = ""
    mp = re.search(r'<meta\s+name="PubDate"\s+content="(\d{4}-\d{2}-\d{2})', html)
    if mp:
        date = mp.group(1)
    if not date:
        dm = re.search(r"发布时间[:：]\s*(\d{4}-\d{2}-\d{2})", html)
        if dm:
            date = dm.group(1)

    # 正文容器
    cm = soup.find("div", class_="details-content") or soup.find("div", class_="article-content-body") or soup.find("div", class_="details-wrap")
    if not cm:
        return title, date, "", []

    # 去掉标题/信息/打印关闭/样式噪声
    for h in cm.find_all(["h1", "h2", "h3"]):
        h.decompose()
    for d in cm.find_all("div"):
        cls = " ".join(d.get("class", []))
        if "print" in cls or "back" in cls or "bfuns" in cls or "dtcode" in cls or "ewm" in cls:
            d.decompose()
    for s in cm.find_all("style"):
        s.decompose()
    for h in cm.find_all(style=re.compile(r"display\s*:\s*none", re.I)):
        h.decompose()

    # 附件收集：正文内 + 整个页面 (正文外附件区也要)
    attachments = []
    for a in soup.find_all("a", href=True):
        href = a["href"]
        txt = a.get_text(strip=True)
        if re.search(r"(?i)\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|ofd)(\?|$)", href) or "download" in href.lower() or "attach" in href.lower() or txt.startswith("附件"):
            abs_url = urljoin(page_url, href)
            if not abs_url.startswith(("http://", "https://")):
                abs_url = "http:" + abs_url if abs_url.startswith("//") else abs_url
            attachments.append((a, txt, abs_url))
    for a, txt, abs_url in attachments:
        new_a = soup.new_tag("a", href=abs_url, target="_blank")
        new_a.string = txt if txt else abs_url.split("/")[-1]
        if a.parent is not None:
            for sib in a.parent.find_all("img", src=re.compile(r"(?i)\.(gif|png|jpg|jpeg)")):
                if sib in list(a.parent.contents)[:list(a.parent.contents).index(a)]:
                    sib.decompose()
        a.replace_with(new_a)
    attached_hrefs = {abs_url for _, _, abs_url in attachments}

    # 正文段落提取 (details-content 直接 p 子元素)
    parts = []
    for el in cm.find_all(recursive=False):
        if el.name == "table":
            parts.append(str(el))
        elif el.name in ("p", "div", "h1", "h2", "h3", "ul", "ol", "li"):
            txt = el.get_text("", strip=True)
            if not txt:
                continue
            inner_tables = el.find_all("table")
            el_has_attach = any(
                a for a in el.find_all("a", href=True)
                if urljoin(page_url, a["href"]) in attached_hrefs
                or a["href"] in attached_hrefs
            )
            if inner_tables or el_has_attach:
                parts.append(str(el))
                continue
            parts.append(txt)
        else:
            txt = el.get_text("", strip=True)
            if txt:
                parts.append(txt)

    # 段落去重
    seen = set()
    final_parts = []
    for p in parts:
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if key and key not in seen:
            seen.add(key)
            final_parts.append(p)

    # \n\n 分段; 未入流的附件追加
    content = "\n\n".join(final_parts)
    for a, txt, abs_url in attachments:
        if abs_url not in content:
            content += f'\n\n<p><a href="{abs_url}" target="_blank">{txt}</a></p>'

    # 清理
    content = re.sub(r"打印本页[^\n]*", "", content)
    content = re.sub(r"\n{3,}", "\n\n", content).strip()
    return title, date, content, attachments


def main():
    max_pages = 1
    args = sys.argv[1:]
    i = 0
    while i < len(args):
        a = args[i]
        if a.startswith("--pages="):
            max_pages = int(a.split("=", 1)[1])
        elif a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
            i += 1
        i += 1

    conn = sqlite3.connect(DB_PATH, timeout=30)
    c = conn.cursor()
    total_new = 0
    total_dup = 0
    total_skip = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = LIST_FIRST
        else:
            list_url = f"{BASE_URL}/c18458/pages/{page}.html"
        html = http_get(list_url)
        if not html:
            print(f"  [WARN] 第{page}页获取失败, 跳过", file=sys.stderr)
            continue
        items = parse_list(html)
        print(f"  第{page}页: 找到 {len(items)} 条")
        for url, title, date in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                total_dup += 1
                continue
            dhtml = http_get(url)
            if not dhtml:
                total_skip += 1
                continue
            d_title, d_date, content, attachments = extract_detail(dhtml, url)
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                total_skip += 1
                continue
            if not d_title:
                d_title = title
            if not d_date:
                d_date = date
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if "<table" in content else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (d_title, summary, content, url, url, d_date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                rid = cur.lastrowid
                c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, d_title, SITE_NAME, summary))
                conn.commit()
                total_new += 1
                print(f"    [{d_date}] {d_title[:45]}")
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.4)

    conn.close()
    print(f"\n新增: {total_new}  重复: {total_dup}  过短/404: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
