#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""锦州经济技术开发区（滨海新区）管委会 - 开发建设 (zsyz/kfjs) 爬虫
CMS: VSB (Visual SiteBuilder) 反向分页
列表: /zsyz/kfjs.htm (第1页, 20条/页) + /zsyz/kfjs/{TOTAL+1-N}.htm (第N页, 反向!)
     分页条 p_no/p_no_d: 当前页1 显示 2->kfjs/3.htm 3->kfjs/2.htm 尾页->kfjs/1.htm => TOTAL=4页
详情: /info/1075/{id}.htm
     老模板无 meta ArticleTitle/PubDate: 标题 div.title + 日期 div.tit span '发布时间：' + 正文 div.v_news_content (删 vsbcontent_start/end)
"""
import sys, re, time, sqlite3, ssl, urllib.request
from urllib.parse import urljoin
from bs4 import BeautifulSoup

SITE_NAME = "锦州经开区-开发建设"
SCRIPT_NAME = "crawl_jzbhxq_kfjs.py"
GROUP_NAME = "辽宁"
BASE_URL = "http://www.jzbhxq.gov.cn"
LIST_URL = BASE_URL + "/zsyz/kfjs.htm"
DB_PATH = "/mnt/data/search.db"

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}


def http_get(url, timeout=30, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            resp = urllib.request.urlopen(req, timeout=timeout, context=ctx)
            return resp.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i == retries - 1:
                print(f"  [WARN] 请求失败 {url}: {e}", file=sys.stderr)
                return ""
            time.sleep(1.2)
    return ""


def clean_title(title):
    """strip &middot;&nbsp; 实体前缀 和省略号截断后缀"""
    title = title.replace("&middot;", "").replace("&nbsp;", "")
    title = title.replace("\u200b", "").replace("\ufeff", "")
    title = re.sub(r"^[\s\xa0·\u00b7]+", "", title)
    title = re.sub(r"\s*\.{3,}\s*$", "", title)
    return title.strip()


def get_total_pages(html):
    """从分页条 p_no/p_no_d 收集页码, 返回总页数 (反向分页: URL数字=TOTAL+1-页码)"""
    nums = set()
    m = re.search(r'p_no_d[^>]*>\s*(\d+)', html)
    if m:
        nums.add(int(m.group(1)))
    for m in re.finditer(r'class="p_no"[^>]*>\s*<a[^>]*>(\d+)</a>', html):
        nums.add(int(m.group(1)))
    return max(nums) if nums else 1


def parse_list(html):
    """从列表页提取 (url, title, date)。条目 div.height38#line_uN_M, title 属性在 href 之前!"""
    items = []
    # 前瞻分隔: 每个条目到下一个 height38 或脚本结束
    pat = r'<div class="height38" id="line_u\d+_\d+">(.*?)(?=<div class="height38" id="line_u\d+_\d+">|<script>_showDynClickBatch)'
    for seg in re.findall(pat, html, re.S):
        a = re.search(r'<a[^>]*href="([^"]+)"', seg)
        t = re.search(r'<a[^>]*title="([^"]*)"', seg)
        if not a:
            continue
        href = a.group(1)
        title = clean_title(t.group(1)) if t and t.group(1) else ""
        if not title:
            tm = re.search(r'<a[^>]*>([^<]+)</a>', seg)
            if tm:
                title = clean_title(tm.group(1))
        dm = re.search(r'(\d{4}-\d{2}-\d{2})', seg)
        date = dm.group(1) if dm else ""
        url = urljoin(LIST_URL, href)
        if title and "info/" in url:
            items.append((url, title, date))
    return items


_total_pages = 0


def fetch_page(page):
    """第page页 (1-based) 列表条目。反向分页: 第N页 URL = kfjs/{TOTAL+1-N}.htm"""
    global _total_pages
    if page == 1:
        html = http_get(LIST_URL)
        if not html:
            return []
        _total_pages = get_total_pages(html)
        return parse_list(html)
    if _total_pages == 0:
        html = http_get(LIST_URL)
        if html:
            _total_pages = get_total_pages(html)
    if _total_pages == 0:
        return []
    url = f"{BASE_URL}/zsyz/kfjs/{_total_pages + 1 - page}.htm"
    html = http_get(url)
    return parse_list(html)


def extract_detail(html, page_url, list_title):
    """提取 (title, date, content_html)"""
    soup = BeautifulSoup(html, "html.parser")

    title = ""
    td = soup.find("div", class_="title")
    if td:
        title = clean_title(td.get_text(strip=True))
    if not title:
        tt = soup.find("title")
        if tt:
            title = clean_title(tt.get_text(strip=True).split("-")[0])
    if not title:
        title = list_title

    date = ""
    dm = re.search(r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", html)
    if dm:
        date = dm.group(1)
    if not date:
        dm2 = re.search(r"(\d{4}-\d{2}-\d{2})", html)
        if dm2:
            date = dm2.group(1)

    cm = soup.find("div", class_="v_news_content")
    if not cm:
        cm = soup.find("div", id=re.compile(r"vsb_content"))
    if not cm:
        return title, date, ""

    # 装饰标记清理: 仅删无文本的空标记 (vsbcontent_start/end 在此站是真实内容段落, 不能 decompose!)
    for tag in cm.find_all(class_=re.compile(r"vsbcontent_start|vsbcontent_end")):
        if not tag.get_text("", strip=True):
            tag.decompose()
    for tag in cm.find_all(["script", "style"]):
        tag.decompose()

    # 附件链接绝对化
    attachments = []
    for a in cm.find_all("a", href=True):
        href = a["href"]
        txt = a.get_text(strip=True) or a.get("title", "")
        if (re.search(r"(?i)\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|ofd)(\?|$)", href)
                or "download" in href.lower() or "attach" in href.lower() or txt.startswith("附件")):
            abs_url = urljoin(page_url, href)
            if not abs_url.startswith(("http://", "https://")):
                abs_url = "http:" + abs_url if abs_url.startswith("//") else abs_url
            name = txt if txt else abs_url.split("/")[-1]
            attachments.append((a, name, abs_url))
    for a, name, abs_url in attachments:
        new_a = soup.new_tag("a", href=abs_url, target="_blank")
        new_a.string = name
        if a.parent is not None:
            p = a.parent
            idx_a = None
            for ci, child in enumerate(p.contents):
                if child is a:
                    idx_a = ci
                    break
            if idx_a is not None:
                for child in list(p.contents[:idx_a]):
                    if getattr(child, "name", None) == "img":
                        child.decompose()
        a.replace_with(new_a)
    attached_hrefs = {abs_url for _, _, abs_url in attachments}

    # 正文段落提取
    parts = []
    for el in cm.find_all(recursive=False):
        if el.name == "table":
            parts.append(str(el))
            continue
        if el.name in ("p", "div", "h1", "h2", "h3", "ul", "ol", "li", "blockquote"):
            txt = el.get_text("", strip=True)
            if not txt and not el.find("img"):
                continue
            inner_tables = el.find_all("table")
            el_has_attach = any(
                urljoin(page_url, x["href"]) in attached_hrefs or x["href"] in attached_hrefs
                for x in el.find_all("a", href=True))
            if inner_tables or el_has_attach:
                parts.append(str(el))
                continue
            parts.append(txt)
        else:
            txt = el.get_text("", strip=True)
            if txt:
                parts.append(txt)

    # 段落去重
    seen = set()
    final_parts = []
    for p in parts:
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if key and key not in seen:
            seen.add(key)
            final_parts.append(p)

    content = "\n\n".join(final_parts)
    content_unescaped = content.replace("&amp;", "&")
    for _, name, abs_url in attachments:
        if abs_url not in content and abs_url not in content_unescaped:
            content += f'\n\n<p><a href="{abs_url}" target="_blank">{name}</a></p>'

    content = re.sub(r"打印本页[^\n]*", "", content)
    content = re.sub(r"\n{3,}", "\n\n", content).strip()
    return title, date, content


def main():
    max_pages = 1
    args = sys.argv[1:]
    i = 0
    while i < len(args):
        a = args[i]
        if a.startswith("--pages="):
            max_pages = int(a.split("=", 1)[1])
        elif a == "--pages" and i + 1 < len(args):
            max_pages = int(args[i + 1])
            i += 1
        i += 1

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()
    total_new = 0
    total_dup = 0
    total_skip = 0

    fail_streak = 0
    for page in range(1, max_pages + 1):
        items = fetch_page(page)
        if not items:
            fail_streak += 1
            print(f"  [WARN] 第{page}页无数据 (fail_streak={fail_streak})", file=sys.stderr)
            if fail_streak >= 2 and page > 1:
                break
            continue
        fail_streak = 0
        print(f"  第{page}页: 找到 {len(items)} 条")
        for url, title, date in items:
            try:
                c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
                if c.fetchone():
                    total_dup += 1
                    continue
            except sqlite3.OperationalError:
                time.sleep(3)
                continue
            dhtml = http_get(url)
            if not dhtml:
                total_skip += 1
                continue
            d_title, d_date, content = extract_detail(dhtml, url, title)
            if not content or len(re.sub(r"<[^>]+>", "", content).strip()) < 10:
                total_skip += 1
                continue
            if not d_title:
                d_title = title
            if not d_date:
                d_date = date
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            has_table = 1 if "<table" in content else 0
            try:
                cur = c.execute(
                    "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                    (d_title, summary, content, url, url, d_date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                rid = cur.lastrowid
                c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                          (rid, d_title, SITE_NAME, summary))
                conn.commit()
                total_new += 1
                print(f"    [{d_date}] {d_title[:45]}")
            except sqlite3.OperationalError as e:
                if "locked" in str(e):
                    conn.rollback()
                    time.sleep(8)
                    try:
                        cur = c.execute(
                            "INSERT INTO gov_raw (title, summary, content, page_url, source_url, publish_date, site_name, script_name, group_name, has_table, date_rank) VALUES (?,?,?,?,?,?,?,?,?,?,?)",
                            (d_title, summary, content, url, url, d_date, SITE_NAME, SCRIPT_NAME, GROUP_NAME, has_table, 0))
                        rid = cur.lastrowid
                        c.execute("INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                                  (rid, d_title, SITE_NAME, summary))
                        conn.commit()
                        total_new += 1
                        print(f"    [{d_date}] {d_title[:45]} (重试成功)")
                    except Exception:
                        conn.rollback()
                        total_skip += 1
                else:
                    total_skip += 1
            except sqlite3.IntegrityError:
                total_dup += 1
            time.sleep(0.3)

    conn.close()
    print(f"\n新增: {total_new}  重复: {total_dup}  过短/404: {total_skip}  总计: {total_new+total_dup+total_skip}")


if __name__ == "__main__":
    main()
