#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_jiangjin_bmjz.py - 江津区-部门街镇
http://www.jiangjin.gov.cn/zwxx_180/bmjz/
列表: div.list > a[href=./YYYYMM/t...html] > span(标题) + span(日期)
分页: index_N.html (24 页 ~480 条, 每页 20 条; pageCount 写死25但P25=404)
详情: /zwxx_180/bmjz/YYYYMM/t...html
  → div.title(标题) + div.info(发布日期) + div.trs_editor_view.TRS_UEDITOR(正文)
附件: ⚠️ JS变量 hasFJ 内 document.write 动态生成, 正则提取 <script>var hasFJ = '...'</script>
"""
import sys
import os
import re
import time
import html as html_lib
import urllib.parse
import sqlite3
import json

import requests
from bs4 import BeautifulSoup, Comment

# ---------------- config ----------------
BASE_URL = "http://www.jiangjin.gov.cn"
LIST_URL = BASE_URL + "/zwxx_180/bmjz/"
SITE_NAME = "江津区-部门街镇"
GROUP_NAME = "重庆"
SCRIPT_NAME = "crawl_jiangjin_bmjz.py"
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")
ENC = "utf-8"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

REQUEST_INTERVAL = 1.2
_last_req = 0.0

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1


def http_get(url, retries=3, timeout=30):
    global _last_req
    dt = time.time() - _last_req
    if dt < REQUEST_INTERVAL:
        time.sleep(REQUEST_INTERVAL - dt)
    _last_req = time.time()
    for i in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=timeout, verify=False)
            if r.status_code == 200:
                r.encoding = "utf-8"
                return r
        except Exception:
            pass
        time.sleep(2 + i * 2)
    return None


def clean_title(t):
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch_list(page):
    """抓一页列表, 返回 [(abs_url, title, date)]"""
    if page == 1:
        url = LIST_URL
    else:
        url = LIST_URL + f"index_{page}.html"
    r = http_get(url)
    if r is None:
        return []
    html_text = r.text
    items = []
    # <a href="./202608/t20260811_15917602.html"><span>标题</span><span>日期</span></a>
    for m in re.finditer(r'<a href="\./(\d{6})/t\d+_\d+\.html">\s*<span>\s*(.*?)\s*</span>\s*<span>\s*(\d{4}-\d{2}-\d{2})\s*</span>\s*</a>', html_text, re.S):
        href, title, date = m.group(1), clean_title(m.group(2)), m.group(3)
        abs_url = urllib.parse.urljoin(LIST_URL, f"./{href}/t{m.group(0).split('/t')[1][:1]}")
        # 重新构造: 从原始匹配里拿完整路径
        raw = m.group(0)
        hm = re.search(r'href="(\./[^"]+)"', raw)
        if hm:
            abs_url = urllib.parse.urljoin(LIST_URL, hm.group(1))
        items.append((abs_url, title, date))
    return items


def extract_hasfj(html_text):
    """从 JS 变量 hasFJ 提取附件 [(abs_url, title)]"""
    atts = []
    m = re.search(r"var hasFJ = '(.*?)';", html_text, re.S)
    if not m:
        return atts
    block = m.group(1)
    # 转义还原
    block = block.replace("\\'", "'").replace('\\"', '"')
    for am in re.finditer(r'<a href="([^"]+)"[^>]*>(.*?)</a>', block, re.S):
        href, txt = am.group(1), clean_title(am.group(2))
        if not txt:
            continue
        if href.startswith("./"):
            abs_url = urllib.parse.urljoin(LIST_URL, href)
        elif href.startswith("/"):
            abs_url = urllib.parse.urljoin(BASE_URL, href)
        else:
            abs_url = href
        atts.append((abs_url, txt))
    return atts


def parse_detail(url):
    """详情页: 返回 (title, pubdate, zoom_soup, fj_atts)"""
    r = http_get(url)
    if r is None:
        return None, None, None, []
    html_text = r.text
    soup = BeautifulSoup(html_text, "html.parser")
    tm = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html_text)
    title = clean_title(tm.group(1)) if tm else ""
    if not title:
        tdiv = soup.find("div", class_="title")
        if tdiv:
            title = clean_title(tdiv.get_text())
    dm = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html_text)
    pubdate = dm.group(1) if dm else ""
    if not pubdate:
        dm2 = re.search(r"发布日期：\s*(\d{4}-\d{2}-\d{2})", html_text)
        pubdate = dm2.group(1) if dm2 else ""
    zoom = soup.find("div", class_="trs_editor_view")
    if not zoom:
        # 另一模板: div.view.TRS_UEDITOR
        for d in soup.find_all("div"):
            cls = d.get("class") or []
            if "TRS_UEDITOR" in cls and ("view" in cls or "trs_editor_view" in cls):
                zoom = d
                break
    if not zoom:
        zoom = soup.find("div", id="zoom")
    fj_atts = extract_hasfj(html_text)
    if zoom:
        for c in zoom.find_all(string=lambda s: isinstance(s, Comment)):
            c.extract()
    else:
        # 附件型公告: 无正文容器, 正文只有 JS 动态附件 → 用空容器 + hasFJ 附件
        zoom = soup.new_tag("div")
    return title, pubdate, zoom, fj_atts


def html_to_text(zoom_soup, abs_url):
    if zoom_soup is None:
        return "", False, []
    for tag in zoom_soup.find_all(["script", "style", "iframe", "object"]):
        tag.decompose()
    has_table = bool(zoom_soup.find("table"))
    attachments = []
    for a in zoom_soup.find_all("a", href=True):
        h = a["href"].strip()
        if h.startswith("./"):
            a["href"] = urllib.parse.urljoin(LIST_URL, h)
        elif h.startswith("/"):
            a["href"] = urllib.parse.urljoin(BASE_URL, h)
        if h.lower().endswith((".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar")):
            txt = clean_title(a.get_text())
            if txt:
                attachments.append((a["href"], txt))
    for img in zoom_soup.find_all("img", src=True):
        s = img["src"].strip()
        if s.startswith("./"):
            img["src"] = urllib.parse.urljoin(LIST_URL, s)
        elif s.startswith("/"):
            img["src"] = urllib.parse.urljoin(BASE_URL, s)
    content = str(zoom_soup)
    return content, has_table, attachments


def append_attachments(content, attachments):
    if not attachments:
        return content
    parts = [content] if content.strip() else []
    for abs_url, txt in attachments:
        parts.append(f'<p><a href="{abs_url}" target="_blank">{txt}</a></p>')
    return "\n\n".join(parts)


def load_existing_urls(conn):
    try:
        cur = conn.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        return {row[0] for row in cur.fetchall()}
    except Exception:
        return set()


def store_record(conn, url, title, content, date, has_table):
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (url, SITE_NAME))
    if cur.fetchone():
        return False
    try:
        industry = "other"
        try:
            from crawler_lib import classify_industry
            industry = classify_industry(title)
        except Exception:
            pass
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, industry),
        )
        if cur.rowcount > 0:
            summary = re.sub(r"<[^>]+>", "", content)
            summary = html_lib.unescape(summary)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, SITE_NAME, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, has_table)
        raise


def main():
    conn = None
    if not JSONL_PATH:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute("PRAGMA busy_timeout=60000")
        conn.execute("PRAGMA journal_mode=WAL")
        existing = load_existing_urls(conn)
        print(f"已入库 {len(existing)} 条, 用于列表级预查重", flush=True)

    new_count = 0
    skip_count = 0
    jsonl_rows = []

    for page in range(1, _PAGES + 1):
        items = fetch_list(page)
        print(f"[list] page={page} items={len(items)}", flush=True)
        if not items:
            print("Empty page, stop pagination", flush=True)
            break
        for abs_url, title, date in items:
            if not JSONL_PATH and abs_url in existing:
                skip_count += 1
                continue
            dtitle, ddate, zoom, fj = parse_detail(abs_url)
            if zoom is None:
                print(f"  skip empty: {title[:30]} | {abs_url}", flush=True)
                skip_count += 1
                continue
            if not dtitle:
                dtitle = title
            if not ddate:
                ddate = date
            content, has_table, atts = html_to_text(zoom, abs_url)
            all_atts = atts + fj
            seen = set()
            uniq = []
            for u, t in all_atts:
                if u in seen:
                    continue
                seen.add(u)
                uniq.append((u, t))
            content = append_attachments(content, uniq)
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            if plain_len < 10 and "<img" not in content and not uniq:
                print(f"  skip empty: {dtitle[:30]} | {abs_url}", flush=True)
                skip_count += 1
                continue
            if JSONL_PATH:
                jsonl_rows.append({
                    "page_url": abs_url, "title": dtitle, "content": content,
                    "publish_date": ddate, "has_table": has_table,
                    "site_name": SITE_NAME, "group_name": GROUP_NAME,
                    "script_name": SCRIPT_NAME,
                })
                print(f"  JSONL + {dtitle[:40]} ({ddate})", flush=True)
                new_count += 1
                continue
            ok = store_record(conn, abs_url, dtitle, content, ddate, has_table)
            if ok:
                new_count += 1
                existing.add(abs_url)
                print(f"  + {dtitle[:40]} ({ddate})", flush=True)
            else:
                skip_count += 1

    if JSONL_PATH:
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(json.dumps(row, ensure_ascii=False) + "\n")
        print(f"JSONL written: {JSONL_PATH} ({len(jsonl_rows)} rows)", flush=True)

    if conn is not None:
        conn.commit()
        conn.close()
    print(f"[done] new={new_count} skip={skip_count}", flush=True)


if __name__ == "__main__":
    main()
