#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
绵竹市人民政府 - 通用爬虫
https://www.mz.gov.cn
自定义CMS，标准JSP分页，支持多栏目
Usage:
  python3 crawl_mz_gsgg.py                        # 默认公示公告 (cat_id=15948)
  python3 crawl_mz_gsgg.py --cat-id 24186 --name mz_zgzl --max-pages 2
  python3 crawl_mz_gsgg.py --cat-id 15954 --name mz_hbdchjbh --max-pages 20 --is-index
  python3 crawl_mz_gsgg.py --cat-id 24186 --full
"""
import sys, os, re, time, json, requests
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DOMAIN = "https://www.mz.gov.cn"
LIST_URL_TEMPLATE = DOMAIN + "/%s?cat_id=%s&cur_page=%d"
SITE_NAME = "mz_gsgg"
MAX_PAGES = 19
DEFAULT_PAGES = 5


def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        raw = r.content
        m = re.search(rb'charset=["\']?([a-zA-Z0-9_-]+)', raw[:4096], re.I)
        enc = "utf-8"
        if m:
            e = m.group(1).decode().lower().replace("-", "").replace("_", "")
            if e in ("gbk", "gb2312"):
                enc = "gbk"
        r.encoding = enc
        return r.text
    except Exception as e:
        print("  FETCH ERROR %s: %s" % (url, e), file=sys.stderr)
        return None


def parse_list_items(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.select("div.list_div ul li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if not href or not href.endswith(".htm"):
            continue
        full_url = href if href.startswith("http") else DOMAIN + href
        title = a.get("title", "") or a.get_text(strip=True)
        if not title or len(title) < 3:
            continue
        date_str = ""
        tb = li.select_one("time")
        if tb:
            dt = tb.get_text(strip=True)
            m = re.search(r"(\d{4})-(\d{1,2})-(\d{1,2})", dt)
            if m:
                date_str = "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))
        items.append({"url": full_url, "title": title.strip(), "date": date_str})
    return items


def _clean_span_spaces(text):
    """Strip span-fragment whitespace between Chinese chars, digits, punctuation"""
    if not text:
        return text
    text = re.sub(r'(?<=[\u4e00-\u9fff])\s+(?=[\u4e00-\u9fff])', '', text)
    text = re.sub(r'(?<=[\u4e00-\u9fff])\s+(?=\d)', '', text)
    text = re.sub(r'(?<=\d)\s+(?=[\u4e00-\u9fff])', '', text)
    text = re.sub(r'\s+([，。、；：？！)】」》])', r'\1', text)
    text = re.sub(r'([（【「《])\s+', r'\1', text)
    text = re.sub(r'(?<=\d)\s+(?=\d)', '', text)
    text = re.sub(r'&#\d+;', '', text)
    return re.sub(r'\s+', ' ', text).strip()


def get_detail(url, title_from_list):
    html = fetch(url)
    if not html:
        return "", "", [], title_from_list

    soup = BeautifulSoup(html, "html.parser")

    full_title = title_from_list
    h1 = soup.select_one("h1")
    if h1:
        t = h1.get_text(strip=True)
        if t:
            full_title = t

    pub_date = ""
    attr = soup.select_one("p#attribute")
    if attr:
        attr_text = attr.get_text()
        m = re.search(r"发布时间[：:]\s*(\d{4})-(\d{1,2})-(\d{1,2})", attr_text)
        if m:
            pub_date = "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))

    attachments = []
    fj_span = soup.select_one("span.fj")
    if fj_span:
        file_tag = fj_span.find("file")
        if file_tag:
            for a in file_tag.find_all("a"):
                href = a.get("href", "")
                if href:
                    f_url = href if href.startswith("http") else DOMAIN + href
                    f_title = a.get_text(strip=True) or os.path.basename(href)
                    attachments.append({"url": f_url, "title": f_title})

    content_parts = []
    article = soup.select_one("article.content")
    if article:
        for el in article.children:
            tag = el.name
            if tag is None:
                text = str(el).strip()
                if text and not re.match(r'^\$\{.*\}$', text.strip()):
                    content_parts.append(text)
            elif tag == "table":
                content_parts.append(str(el))
            elif tag == "p":
                txt = el.get_text(" ", strip=True)
                txt = _clean_span_spaces(txt)
                if txt and not re.match(r'^\$\{.*\}$', txt):
                    content_parts.append(txt)
            elif tag == "div":
                # 处理 div 内的 p 和 table（保持段落分段）
                for child in el.children:
                    if child.name == "p":
                        txt = child.get_text(" ", strip=True)
                        txt = _clean_span_spaces(txt)
                        if txt and not re.match(r'^\$\{.*\}$', txt):
                            content_parts.append(txt)
                    elif child.name == "table":
                        content_parts.append(str(child))
            elif tag == "br":
                pass
            elif tag == "img":
                src = el.get("src", "")
                alt = el.get("alt", "")
                if src:
                    img_url = src if src.startswith("http") else DOMAIN + src
                    content_parts.append("![%s](%s)" % (alt, img_url))

    content = "\n\n".join(content_parts)
    if not content or len(content.strip()) < 20:
        content = '<p><a href="%s">%s</a></p>' % (url, full_title)

    if article:
        for a in article.find_all("a", href=True):
            href = a["href"]
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar)$', href, re.I):
                f_url = href if href.startswith("http") else DOMAIN + href
                f_title = a.get_text(strip=True) or os.path.basename(href)
                if not any(att["url"] == f_url for att in attachments):
                    attachments.append({"url": f_url, "title": f_title})
                if content:
                    a_href_escaped = re.escape(href)
                    content = re.sub(
                        r'<a[^>]*href=["\']' + a_href_escaped + r'["\'][^>]*>.*?</a>',
                        "[%s](%s)" % (f_title, f_url),
                        content, flags=re.DOTALL
                    )

    content = re.sub(r'\$\{[^}]+\}', '', content)
    content = re.sub(r'\n{3,}', '\n\n', content)
    full_title = re.sub(r'\s+', ' ', full_title).strip()

    return content, pub_date, attachments, full_title


def main():
    # Parse common args
    cat_id = "15948"
    site_name = "mz_gsgg"
    max_pages = 19
    list_page = "info/iList.jsp"
    pages = DEFAULT_PAGES

    i = 1
    while i < len(sys.argv):
        a = sys.argv[i]
        if a == "--full":
            pages = MAX_PAGES
        elif a == "--pages" and i + 1 < len(sys.argv):
            try:
                pages = int(sys.argv[i + 1])
            except ValueError:
                pass
            i += 1
        elif a == "--cat-id" and i + 1 < len(sys.argv):
            cat_id = sys.argv[i + 1]
            i += 1
        elif a == "--name" and i + 1 < len(sys.argv):
            site_name = sys.argv[i + 1]
            i += 1
        elif a == "--max-pages" and i + 1 < len(sys.argv):
            try:
                max_pages = int(sys.argv[i + 1])
            except ValueError:
                pass
            i += 1
        elif a == "--is-index":
            list_page = "info/iIndex.jsp"
        i += 1

    pages = min(pages, max_pages)

    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()

    all_new = 0
    for pg in range(1, pages + 1):
        url = LIST_URL_TEMPLATE % (list_page, cat_id, pg)
        print("第%d页: %s" % (pg, url), file=sys.stderr)
        html = fetch(url)
        if not html:
            print("  第%d页 获取失败" % pg, file=sys.stderr)
            continue

        items = parse_list_items(html)
        if not items:
            print("  第%d页 无条目" % pg, file=sys.stderr)
            if pg == 1:
                break
            continue

        print("  第%d页: %d 条" % (pg, len(items)), file=sys.stderr)

        for it in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (it["url"], site_name))
            if c.fetchone():
                continue

            content, pub_date, attachments, full_title = get_detail(it["url"], it["title"])
            date = it["date"] or pub_date

            try:
                c.execute(
                    """INSERT INTO gov_raw (site_name, page_url, title, content, publish_date, source_url, status, attachments)
                       VALUES (?,?,?,?,?,?,?,?)""",
                    (site_name, it["url"], full_title[:500], content, date[:20],
                     it["url"], "synced", json.dumps(attachments, ensure_ascii=False))
                )
                all_new += 1
                print("    + %s (%s) body:%dB attach:%d" % (full_title[:50], date, len(content), len(attachments)), file=sys.stderr)
            except Exception as e:
                print("    ! 入库失败: %s" % e, file=sys.stderr)

        time.sleep(0.3)
        conn.commit()

    print("\n[%s] 完成! 新增 %d 条" % (site_name, all_new), file=sys.stderr)

    conn.close()


if __name__ == "__main__":
    main()
