#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
临川区人民政府 - col1432 政府信息公开(公共资源配置 D00004D00005D00004)
URL: http://www.jxlc.gov.cn/col/col1432/
CMS: Hanweb 大汉 xxgk/search.jsp (需先访问列表页拿 cookie, 否则 {"msg":"closed"} 503)
列表: POST /module/xxgk/search.jsp  infotypeId=D00004D00005D00004 jdid=5 currpage=N  18条/页
详情: /art/YYYY/M/D/art_4780_xxx.html  meta ArticleTitle/PubDate + div#zoom

用法:
  python3 crawl_jxlc_hjbh.py --pages=1     # 增量: 前1页(约18条)
  python3 crawl_jxlc_hjbh.py --pages=5     # 首批前5页
  python3 crawl_jxlc_hjbh.py --pages=200   # 全量(933条≈52页)
"""
import os, re, sys, sqlite3, time, html as html_mod
from datetime import datetime
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

# ── 参数解析 ──────────────────────────────────────────────
def parse_pages(argv):
    """兼容 --pages=N / --pages N / 裸数字 / --full"""
    if '--full' in argv:
        return 9999
    for i, a in enumerate(argv):
        if a == '--pages' and i + 1 < len(argv) and argv[i+1].isdigit():
            return int(argv[i+1])
        if a.startswith('--pages='):
            return int(a.split('=')[1])
        if a.isdigit():
            return int(a)
    return 1  # 默认前1页

MAX_PAGES = parse_pages(sys.argv[1:])
print(f"[jxlc-hjbh] max_pages={MAX_PAGES}")

DB_PATH = os.getenv("SEARCH_DB", "/mnt/data/search.db")
BASE = "https://www.jxlc.gov.cn"
LIST_URL = "https://www.jxlc.gov.cn/col/col1432/"
SEARCH_URL = BASE + "/module/xxgk/search.jsp"
PER_PAGE = 18
CUTOFF_DATE = (datetime.now().strftime("%Y-%m-%d"))

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "X-Requested-With": "XMLHttpRequest",
    "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8",
}

SITE_NAME = "临川区人民政府-环境保护"
CATEGORY = "环境保护"
GROUP_NAME = "江西"

# ── DB helpers ─────────────────────────────────────────────
def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    conn.row_factory = sqlite3.Row
    return conn

def clean_title(t):
    """清理标题: 实体前缀 &middot;&nbsp; + 零宽空格 + 空白"""
    if not t:
        return t
    t = t.replace("&middot;", "").replace("&nbsp;", "").replace("&ensp;", "").replace("&emsp;", "")
    t = re.sub(r"[\u200b\u200c\u200d\ufeff]", "", t)
    return t.strip()

def is_dup(conn, page_url):
    return conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone() is not None

def insert_item(conn, title, page_url, publish_date, content):
    date_rank = int(publish_date.replace("-", "")) if publish_date and "-" in publish_date else 0
    plain = ""
    if content:
        plain = BeautifulSoup(content, "html.parser").get_text(strip=True)[:200]
    try:
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary, script_name, group_name, has_table) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, title, page_url, publish_date, content, date_rank, CATEGORY, plain,
             "crawl_jxlc_hjbh.py", GROUP_NAME, 1 if "<table" in (content or "") else 0),
        )
        if cur.rowcount > 0:
            try:
                # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                conn.commit()
                conn.execute(
                    "INSERT OR REPLACE INTO gov_search(title, site_name, summary) VALUES (?, ?, ?)",
                    (title, SITE_NAME, plain),
                )
            except Exception as e:
                print(f"    [FTS WARN] {e}")
            return True
        return False
    except Exception as e:
        print(f"    [ERR] insert: {e}")
        return False

# ── 列表页 ────────────────────────────────────────────────
def fetch_list(session, infotype_id, page_num):
    data = {
        "infotypeId": infotype_id,
        "jdid": "5",
        "area": "",
        "divid": "div1432",
        "vc_title": "",
        "vc_number": "",
        "currpage": str(page_num),
        "vc_filenumber": "",
        "vc_all": "",
        "texttype": "",
        "fbtime": "",
    }
    resp = session.post(SEARCH_URL, data=data, headers=HEADERS, timeout=30)
    resp.encoding = "utf-8"
    return resp.text

def parse_list(html_text):
    items = []
    lis = re.findall(r"<li>(.*?)</li>", html_text, re.DOTALL)
    for li in lis:
        m = re.search(r"href='([^']+)'\s+title=\"([^\"]+)\"", li)
        d = re.search(r"<b>\s*(\d{4}-\d{2}-\d{2})\s*</b>", li)
        if m and d:
            href = m.group(1)
            title = clean_title(m.group(2))
            url = href if href.startswith("http") else urljoin(SEARCH_URL, href)
            items.append((title, url, d.group(1).strip()))
    return items

def get_total(session, infotype_id):
    html_text = fetch_list(session, infotype_id, 1)
    items = parse_list(html_text)
    tr = re.search(r"nTotalCount\s*[^0-9]*(\d+)", html_text)
    total = int(tr.group(1)) if tr else len(items)
    return total, items, html_text

# ── 详情页 ────────────────────────────────────────────────
def parse_detail(html_text, fallback_title):
    soup = BeautifulSoup(html_text, "html.parser")

    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = clean_title(meta["content"])

    pub_date = ""
    meta_d = soup.find("meta", attrs={"name": "PubDate"})
    if meta_d and meta_d.get("content"):
        m2 = re.match(r"(\d{4}-\d{1,2}-\d{1,2})", meta_d["content"].strip())
        if m2:
            pub_date = m2.group(1)

    content = ""
    zoom = soup.find("div", id="zoom")
    if zoom:
        content = extract_content(zoom)
    if not content:
        for cid in ["content", "article", "text", "mainText", "zoomcon"]:
            div = soup.find("div", id=cid)
            if div:
                content = extract_content(div)
                break

    if not title:
        title = fallback_title
    return title, content, pub_date

def extract_content(node):
    """提取正文: 保留表格HTML、附件链接绝对化、\\n\\n分段、去重复内容"""
    soup = BeautifulSoup(str(node), "html.parser")

    # 移除脚本/样式/隐藏元素
    for tag in soup.find_all(["script", "style", "noscript"]):
        tag.decompose()
    for tag in soup.find_all(attrs={"style": re.compile(r"display\s*:\s*none", re.I)}):
        tag.decompose()

    # 附件链接绝对化 + 内嵌图片下载链接保留
    for a in soup.find_all("a"):
        href = a.get("href")
        if href:
            a["href"] = urljoin(BASE, href)
        # 附件链接去掉 img 图标, 只留文字
        img = a.find("img")
        if img:
            img.decompose()

    # 正文: 保留 p/table/div 结构, 段落间 \n\n
    parts = []
    for el in soup.find_all(["p", "table", "div", "li", "h1", "h2", "h3", "h4", "tr"]):
        tag = el.name
        if tag == "tr":
            # 表格行由表格整体处理, 跳过
            continue
        # 跳过嵌套在表格内的 p
        if tag == "p" and el.find_parent("table"):
            continue
        # 跳过空元素
        text = el.get_text(strip=True)
        if not text:
            continue
        if tag == "table":
            parts.append(str(el))
        elif tag in ("p", "div", "li", "h1", "h2", "h3", "h4"):
            inner = str(el)
            # div 内如果已经包含 p/table, 由子元素处理, 避免重复
            if tag == "div" and el.find(["p", "table"]):
                continue
            parts.append(inner)
    content = "\n\n".join(parts)

    # 去重: 相同段落只保留一次(连续重复)
    segs = content.split("\n\n")
    deduped = []
    for s in segs:
        if not deduped or s != deduped[-1]:
            deduped.append(s)
    return "\n\n".join(deduped).strip()

# ── 主流程 ────────────────────────────────────────────────
def main():
    conn = get_conn()
    session = requests.Session()

    # 关键: 先访问列表页获取 cookie 会话
    r0 = session.get(LIST_URL, headers={"User-Agent": HEADERS["User-Agent"]}, timeout=30)
    print(f"[session] list page HTTP {r0.status_code}, cookies={list(session.cookies.keys())}")

    INFOTYPE = "D00004D00005D00004"
    total, page1_items, _ = get_total(session, INFOTYPE)
    total_pages = (total + PER_PAGE - 1) // PER_PAGE
    pages = min(MAX_PAGES, total_pages)
    print(f"[jxlc-hjbh] total={total} records, {total_pages} pages, scanning {pages}")

    # 收集列表条目
    all_items = list(page1_items)
    if pages > 1:
        for pg in range(2, pages + 1):
            try:
                html_text = fetch_list(session, INFOTYPE, pg)
                items = parse_list(html_text)
                all_items.extend(items)
            except Exception as e:
                print(f"  [WARN] page {pg}: {e}")
                time.sleep(3)
            time.sleep(0.4)

    print(f"[jxlc-hjbh] collected {len(all_items)} list items")

    # 抓详情
    new_count = dup_count = err_count = 0
    for idx, (title, url, date_str) in enumerate(all_items, 1):
        try:
            # 只抓本域详情, 跨站链接跳过(如 jxsggzy 公共资源交易网)
            if "jxlc.gov.cn" not in url:
                dup_count += 1
                continue
            if is_dup(conn, url):
                dup_count += 1
                continue
            resp = session.get(url, headers={"User-Agent": HEADERS["User-Agent"]}, timeout=30)
            resp.encoding = "utf-8"
            d_title, content, pub_date = parse_detail(resp.text, title)
            if not pub_date:
                pub_date = date_str
            if insert_item(conn, d_title, url, pub_date, content):
                new_count += 1
            else:
                dup_count += 1
        except Exception as e:
            err_count += 1
            print(f"  [ERR] {title[:40]}: {e}")
        conn.commit()
        time.sleep(0.3)
        if idx % 20 == 0:
            print(f"  [{idx}/{len(all_items)}] new={new_count} dup={dup_count} err={err_count}")

    conn.close()
    print(f"[jxlc-hjbh] DONE: +{new_count} new, {dup_count} dup, {err_count} err")
    print(f"新增: {new_count}")

if __name__ == "__main__":
    main()
