#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""喀左县人民政府 - 环评信息 (民意征集/环评信息栏目) 爬虫
CMS: 朝阳市政府网站群 (cms.chaoyang.gov.cn)
列表: /kzxzf/hdjl/myzj/glist.html  (15条/页, 共45页, 当前第 1/45 页)
     第1页静态HTML, 第2+页走 API:
     GET https://cms.chaoyang.gov.cn/html/page.xhtml?s=KZXZF&o={页-1}&p=162737070000524&c=162737070000524
     ⚠️ 必须 https + Referer=https://www.kazuo.gov.cn/ (http 或 无 Referer 返回朝阳主站内容)
详情: /html/KZXZF/{YYYYMM}/{objectId}.html
     meta ArticleTitle/PubDate + div.center-info 正文 (p>span 段落)
"""
import sys, re, time, sqlite3, ssl, urllib.request
from urllib.parse import urljoin
from bs4 import BeautifulSoup

SITE_NAME = "喀左县人民政府-环评信息"
SCRIPT_NAME = "crawl_kazuo_hjxx.py"
GROUP_NAME = "辽宁"
LIST_URL = "https://www.kazuo.gov.cn/kzxzf/hdjl/myzj/glist.html"
API_URL = "https://cms.chaoyang.gov.cn/html/page.xhtml"
SITE_CODE = "KZXZF"
PAGE_ID = "162737070000524"
COLUMN_ID = "162737070000524"
REFERER = "https://www.kazuo.gov.cn/"
DB_PATH = "/mnt/data/search.db"
PER_PAGE = 15

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def http_get(url, timeout=30, retries=3, referer=REFERER):
    for i in range(retries):
        try:
            hdrs = dict(HEADERS)
            if referer:
                hdrs["Referer"] = referer
            req = urllib.request.Request(url, headers=hdrs)
            resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [WARN] 请求失败 {url}: {e}", file=sys.stderr)
                return ""
            time.sleep(1.2)
    return ""


def clean_title(title):
    """strip &middot;&nbsp; 实体前缀 和省略号截断后缀"""
    title = title.replace("&middot;", "").replace("&nbsp;", "")
    title = title.replace("\u200b", "").replace("\ufeff", "")
    title = re.sub(r"^[\s\xa0·\u00b7]+", "", title)
    title = re.sub(r"\s*\.{3,}\s*$", "", title)
    return title.strip()


def parse_list(html):
    """从列表页/API响应提取 (url, title, date)"""
    items = []
    for m in re.finditer(r'<li><a\s+[^>]*href="([^"]+)"[^>]*>([^<]*)</a><span>([^<]*)</span>', html):
        href, t, d = m.group(1), m.group(2), m.group(3)
        url = urljoin("https://www.kazuo.gov.cn/", href)
        title = clean_title(t.strip())
        if title and "html/KZXZF" in url:
            items.append((url, title, d.strip()))
    return items


def fetch_page(page):
    """第page页 (1-based) 列表条目"""
    if page == 1:
        html = http_get(LIST_URL, referer=None)
        return parse_list(html)
    api = f"{API_URL}?s={SITE_CODE}&o={page - 1}&p={PAGE_ID}&c={COLUMN_ID}"
    html = http_get(api)
    return parse_list(html)


def extract_detail(html, page_url):
    """提取 (title, date, content_html, attachments)"""
    title = ""
    mt = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"', html)
    if mt:
        title = clean_title(mt.group(1))
    date = ""
    mp = re.search(r'<meta\s+name="PubDate"\s+content="(\d{4}-\d{2}-\d{2})', html)
    if mp:
        date = mp.group(1)

    soup = BeautifulSoup(html, "html.parser")
    cm = soup.find("div", class_="center-info")
    if not cm:
        return title, date, "", []

    # 附件收集 (center-info 内)
    attachments = []
    for a in cm.find_all("a", href=True):
        href = a["href"]
        txt = a.get_text(strip=True)
        if (re.search(r"(?i)\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|ofd)(\?|$)", href)
                or "download" in href.lower() or "attach" in href.lower() or txt.startswith("附件")):
            abs_url = urljoin(page_url, href)
            if not abs_url.startswith(("http://", "https://")):
                abs_url = "http:" + abs_url if abs_url.startswith("//") else abs_url
            name = txt if txt else abs_url.split("/")[-1]
            attachments.append((a, name, abs_url))
    for a, name, abs_url in attachments:
        new_a = soup.new_tag("a", href=abs_url, target="_blank")
        new_a.string = name
        if a.parent is not None:
            p = a.parent
            idx_a = None
            for ci, child in enumerate(p.contents):
                if child is a:
                    idx_a = ci
                    break
            if idx_a is not None:
                for child in list(p.contents[:idx_a]):
                    if getattr(child, "name", None) == "img":
                        child.decompose()
        a.replace_with(new_a)
    attached_hrefs = {abs_url for _, _, abs_url in attachments}

    # 正文段落提取 (直接子元素, 表格保留HTML, 附件段保留HTML)
    parts = []
    for el in cm.find_all(recursive=False):
        if el.name == "table":
            parts.append(str(el))
            continue
        if el.name in ("p", "div", "h1", "h2", "h3", "ul", "ol", "li", "blockquote"):
            txt = el.get_text("", strip=True)
            if not txt and not el.find("img"):
                continue
            inner_tables = el.find_all("table")
            el_has_attach = any(
                urljoin(page_url, x["href"]) in attached_hrefs or x["href"] in attached_hrefs
                for x in el.find_all("a", href=True))
            if inner_tables or el_has_attach:
                parts.append(str(el))
                continue
            parts.append(txt)
        else:
            txt = el.get_text("", strip=True)
            if txt:
                parts.append(txt)

    # 段落去重
    seen = set()
    final_parts = []
    for p in parts:
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if key and key not in seen:
            seen.add(key)
            final_parts.append(p)

    content = "\n\n".join(final_parts)
    content_unescaped = content.replace("&amp;", "&")
    for _, name, abs_url in attachments:
        if abs_url not in content and abs_url not in content_unescaped:
            content += f'\n\n<p><a href="{abs_url}" target="_blank">{name}</a></p>'

    content = re.sub(r"打印本页[^\n]*", "", content)
    content = re.sub(r"\n{3,}", "\n\n", content).strip()
    return title, date, content, attachments


def main():
    max_pages = 1
    args = sys.argv[1:]
    i = 0
    while i < len(args):
        a = args[i]
        if a.startswith("--pages="):
            max_pages = int(a.split("=", 1)[1])
        elif a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
            i += 1
        i += 1

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()
    total_new = 0
    total_dup = 0
    total_skip = 0

    fail_streak = 0
    for page in range(1, max_pages + 1):
        items = fetch_page(page)
        if not items:
            fail_streak += 1
            print(f"  [WARN] 第{page}页无数据 (fail_streak={fail_streak})", file=sys.stderr)
            if fail_streak >= 2 and page > 1:
                break
            continue
        fail_streak = 0
        print(f"  第{page}页: 找到 {len(items)} 条")
        for url, title, date in items:
            try:
                c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
                if c.fetchone():
                    total_dup += 1
                    continue
            except sqlite3.OperationalError:
                time.sleep(3)
                continue
            dhtml = http_get(url, referer=REFERER)
            if not dhtml:
                total_skip += 1
                continue
            d_title, d_date, content, attachments = extract_detail(dhtml, url)
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                total_skip += 1
                continue
            if not d_title:
                d_title = title
            if not d_date:
                d_date = date
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if "<table" in content else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (d_title, summary, content, url, url, d_date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                rid = cur.lastrowid
                c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, d_title, SITE_NAME, summary))
                conn.commit()
                total_new += 1
                print(f"    [{d_date}] {d_title[:45]}")
            except sqlite3.OperationalError as e:
                if "locked" in str(e):
                    conn.rollback()
                    time.sleep(8)
                    try:
                        cur = c.execute(
                            "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                            (d_title, summary, content, url, url, d_date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                        rid = cur.lastrowid
                        c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                                  (rid, d_title, SITE_NAME, summary))
                        conn.commit()
                        total_new += 1
                        print(f"    [{d_date}] {d_title[:45]} (重试成功)")
                    except Exception:
                        conn.rollback()
                        total_skip += 1
                else:
                    total_skip += 1
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.4)

    conn.close()
    print(f"\n新增: {total_new}  重复: {total_dup}  过短/404: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
