#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
crawl_hyx_tzgg.py - 衡阳县-西渡高新区通知公告
https://www.hyx.gov.cn/zwgk/bmxxgkml/xdgxq/tzgg/
防护: 帝联CDN(403) + 瑞数4代JS注入(80端口挑战, 正文需浏览器解密) -> 必须 Playwright
列表: li > a, 文本 "YYYY-MM-DD 标题", 详情 /tzgg/YYYYMMDD/iNNNNNNN.html
分页: index.html / pages/N.html (共3页 41条, 每页15/15/11)
详情: 博山CMS(/bcms), meta ArticleTitle/PubDate + div#div_content 正文
  - 普通文章: div_content 内 HTML (p/table/img)
  - PDF型文章: div_content 内 span.edui-pdf[data-pdf=/DFS//file/...] -> 下载PDF -> fitz提取文本
"""
import sys
import os
import re
import time
import html as html_lib
import urllib.parse
import sqlite3
import json

from bs4 import BeautifulSoup, Comment

# ---------------- config ----------------
BASE_URL = "http://www.hyx.gov.cn"
LIST_URL = BASE_URL + "/zwgk/bmxxgkml/xdgxq/tzgg/index.html"
SITE_NAME = "衡阳县-西渡高新区通知公告"
GROUP_NAME = "湖南"
SCRIPT_NAME = "crawl_hyx_tzgg.py"
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
PAGE_INTERVAL = 1.5  # 详情页间间隔

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1


def clean_title(t):
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t)
    return t.strip()


# ---------------- playwright helpers ----------------
def make_browser(p):
    b = p.chromium.launch(headless=True, args=["--no-sandbox"])
    ctx = b.new_context(
        user_agent=UA,
        locale="zh-CN",
        ignore_https_errors=True,
        viewport={"width": 1366, "height": 900},
    )
    return b, ctx


def robust_goto(page, url, timeout=40):
    """瑞数挑战: commit + 轮询 title 含 衡阳"""
    for attempt in range(3):
        try:
            page.goto(url, timeout=timeout * 1000, wait_until="commit")
            for i in range(timeout // 4):
                page.wait_for_timeout(4000)
                t = page.title()
                if "衡阳" in t:
                    page.wait_for_timeout(1500)
                    return True
        except Exception:
            pass
    return False


def fetch_list(page, page_no):
    """Playwright 抓一页列表, 返回 [(abs_url, title, date)]"""
    if page_no == 1:
        url = LIST_URL
    else:
        url = f"{BASE_URL}/zwgk/bmxxgkml/xdgxq/tzgg/pages/{page_no}.html"
    if not robust_goto(page, url):
        print(f"[list] page={page_no} FAIL", flush=True)
        return []
    items = page.eval_on_selector_all("li", """els => els.map(e => {
        const a = e.querySelector('a');
        const t = e.textContent.replace(/\\s+/g,' ').trim();
        return a && t ? {t: t.slice(0, 120), h: a.getAttribute('href')} : null;
    }).filter(Boolean)""")
    out = []
    for it in items:
        m = re.match(r"^(\d{4}-\d{2}-\d{2})\s*(.*)$", it["t"])
        if not m:
            continue
        date, title = m.group(1), clean_title(m.group(2))
        h = it["h"] or ""
        if not h.startswith("http"):
            h = urllib.parse.urljoin(BASE_URL, h)
        if re.search(r"i\d+\.html$", h):
            out.append((h, title, date))
    return out


def fetch_pdf_text(page, pdf_rel):
    """用 Playwright request(带瑞数cookie) 下载 PDF 并提取文本"""
    pdf_url = urllib.parse.urljoin(BASE_URL, pdf_rel)
    try:
        resp = page.request.get(pdf_url, timeout=45000)
        if resp.status != 200:
            print(f"    PDF status {resp.status}: {pdf_url[:90]}", flush=True)
            return ""
        data = resp.body()
        if len(data) < 500:
            return ""
        import fitz
        import io
        doc = fitz.open(stream=data, filetype="pdf")
        parts = []
        for p in doc:
            t = p.get_text().strip()
            if t:
                parts.append(t)
        doc.close()
        return "\n".join(parts)
    except Exception as e:
        print(f"    PDF err: {type(e).__name__} {str(e)[:100]}", flush=True)
        return ""


def parse_detail(page, url):
    """返回 (title, pubdate, content, has_table, attachments)"""
    if not robust_goto(page, url):
        return None, None, None, False, []
    try:
        info = page.eval_on_selector("#div_content", """e => {
            const pdf = e.querySelector('.edui-pdf');
            return {
                pdf: pdf ? pdf.getAttribute('data-pdf') : null,
                html: e.innerHTML,
            };
        }""")
    except Exception:
        return None, None, None, False, []
    if not info:
        return None, None, None, False, []
    # meta 标题/日期
    hd = page.content()
    title = ""
    tm = re.search(r'name="ArticleTitle" content="([^"]*)"', hd)
    if tm:
        title = clean_title(tm.group(1))
    pubdate = ""
    pm = re.search(r'name="PubDate" content="(\d{4}-\d{2}-\d{2})', hd)
    if pm:
        pubdate = pm.group(1)

    # PDF 型正文
    if info.get("pdf"):
        txt = fetch_pdf_text(page, info["pdf"])
        if len(txt) > 200:
            # PDF 提取文本作为正文 (纯文本, FTS 可搜)
            return title, pubdate, txt, False, []
        # PDF 提取为空(扫描件) -> 回退: 标题 + PDF链接段
        content = f"<p>{html_lib.escape(title)}</p>\n<p><a href=\"{urllib.parse.urljoin(BASE_URL, info['pdf'])}\" target=\"_blank\">查看原文PDF</a></p>"
        return title, pubdate, content, False, [(urllib.parse.urljoin(BASE_URL, info["pdf"]), "查看原文PDF")]

    # 普通 HTML 正文
    soup = BeautifulSoup(info["html"], "html.parser")
    for tag in soup.find_all(["script", "style", "iframe", "object", "meta", "link", "title"]):
        tag.decompose()
    for c in soup.find_all(string=lambda s: isinstance(s, Comment)):
        c.extract()
    # 移除相关阅读/二维码等尾部
    for sel in ["#xgwdP", ".xgwdP", ".ewm", "#qrcode"]:
        for t in soup.select(sel):
            t.decompose()
    has_table = bool(soup.find("table"))
    attachments = []
    for a in soup.find_all("a", href=True):
        h = a["href"].strip()
        if h.startswith("/") or h.startswith("http"):
            a["href"] = urllib.parse.urljoin(BASE_URL, h)
        if h.lower().endswith((".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar", ".wps", ".jpg", ".png")) or "file" in h.lower():
            txt = clean_title(a.get_text())
            if txt:
                attachments.append((a["href"], txt))
    for img in soup.find_all("img", src=True):
        s = img["src"].strip()
        if s.startswith("/") or s.startswith("http"):
            img["src"] = urllib.parse.urljoin(BASE_URL, s)
    content = str(soup)
    return title, pubdate, content, has_table, attachments


def append_attachments(content, attachments):
    if not attachments:
        return content
    parts = [content] if content.strip() else []
    for abs_url, txt in attachments:
        parts.append(f'<p><a href="{abs_url}" target="_blank">{txt}</a></p>')
    return "\n\n".join(parts)


def load_existing_urls(conn):
    try:
        cur = conn.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        return {row[0] for row in cur.fetchall()}
    except Exception:
        return set()


def store_record(conn, url, title, content, date, has_table):
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (url, SITE_NAME))
    if cur.fetchone():
        return False
    try:
        industry = "other"
        try:
            from crawler_lib import classify_industry
            industry = classify_industry(title)
        except Exception:
            pass
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, industry),
        )
        if cur.rowcount > 0:
            summary = re.sub(r"<[^>]+>", "", content)
            summary = html_lib.unescape(summary)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, SITE_NAME, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, has_table)
        raise


def main():
    conn = None
    if not JSONL_PATH:
        conn = sqlite3.connect(DB_PATH, timeout=60)
        conn.execute("PRAGMA busy_timeout=60000")
        conn.execute("PRAGMA journal_mode=WAL")
        existing = load_existing_urls(conn)
        print(f"已入库 {len(existing)} 条, 用于列表级预查重", flush=True)

    new_count = 0
    skip_count = 0
    jsonl_rows = []

    from playwright.sync_api import sync_playwright
    with sync_playwright() as p:
        b, ctx = make_browser(p)
        page = ctx.new_page()
        for page_no in range(1, _PAGES + 1):
            items = fetch_list(page, page_no)
            print(f"[list] page={page_no} items={len(items)}", flush=True)
            if not items:
                print("Empty page, stop pagination", flush=True)
                break
            for abs_url, title, date in items:
                if not JSONL_PATH and abs_url in existing:
                    skip_count += 1
                    continue
                dtitle, ddate, content, has_table, atts = parse_detail(page, abs_url)
                if dtitle is None:
                    print(f"  skip empty: {title[:30]} | {abs_url}", flush=True)
                    skip_count += 1
                    continue
                if not dtitle:
                    dtitle = title
                if not ddate:
                    ddate = date
                content = append_attachments(content, atts)
                plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
                if plain_len < 10 and "<img" not in content and not atts:
                    print(f"  skip empty: {dtitle[:30]} | {abs_url}", flush=True)
                    skip_count += 1
                    continue
                if JSONL_PATH:
                    jsonl_rows.append({
                        "page_url": abs_url, "title": dtitle, "content": content,
                        "publish_date": ddate, "has_table": has_table,
                        "site_name": SITE_NAME, "group_name": GROUP_NAME,
                        "script_name": SCRIPT_NAME,
                    })
                    print(f"  JSONL + {dtitle[:40]} ({ddate})", flush=True)
                    new_count += 1
                    continue
                ok = store_record(conn, abs_url, dtitle, content, ddate, has_table)
                if ok:
                    new_count += 1
                    existing.add(abs_url)
                    print(f"  + {dtitle[:40]} ({ddate})", flush=True)
                else:
                    skip_count += 1
                time.sleep(PAGE_INTERVAL)
        b.close()

    if JSONL_PATH:
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(json.dumps(row, ensure_ascii=False) + "\n")
        print(f"JSONL written: {JSONL_PATH} ({len(jsonl_rows)} rows)", flush=True)

    if conn is not None:
        conn.commit()
        conn.close()
    print(f"[done] new={new_count} skip={skip_count}", flush=True)


if __name__ == "__main__":
    main()
