#!/usr/bin/env python3
import os
# -*- coding: utf-8 -*-
"""
香山网(www.x3cn.com) - 今日珠海板块(forum-52) 环评/验收公示爬虫
Discuz! X 论坛 - 使用curl绕过SSL/TLS兼容性问题
Usage:
  python3 crawl_x3cn.py                    # 默认5页初始
  python3 crawl_x3cn.py --pages 10         # 指定页数
  python3 crawl_x3cn.py --full             # 全量128页
"""
import sys, os, re, time, json, subprocess
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
DOMAIN = "https://www.x3cn.com"
FORUM_URL = DOMAIN + "/forum-52-%d.html"
MOBILE_URL = DOMAIN + "/forum.php?mod=viewthread&tid=%s&mobile=2"
SITE_NAME = "x3cn_xrzh"
MAX_PAGES = 128
DEFAULT_PAGES = 5


def fetch(url):
    try:
        r = subprocess.run(
            ["curl", "-sL", "--max-time", "15",
             "-H", "User-Agent: " + UA,
             url],
            capture_output=True, timeout=20
        )
        raw = r.stdout
        # Detect charset from HTML meta
        m = re.search(rb'charset=["\']?([a-zA-Z0-9_-]+)', raw[:4096], re.I)
        enc = "utf-8"
        if m:
            e = m.group(1).decode().lower().replace("-", "").replace("_", "")
            if e in ("gbk", "gb2312"):
                enc = "gbk"
        return raw.decode(enc, errors="replace")
    except Exception as e:
        print("  FETCH ERROR %s: %s" % (url, e), file=sys.stderr)
        return None


def extract_thread_id(url):
    m = re.search(r"thread-(\d+)", url)
    return m.group(1) if m else None


def parse_list_items(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for a in soup.select("a.xst"):
        href = a.get("href", "")
        if not href or not href.startswith("thread-"):
            continue
        title = a.get_text(strip=True)
        if not title or len(title) < 5:
            continue
        full_url = href if href.startswith("http") else DOMAIN + "/" + href
        tid = extract_thread_id(full_url)
        items.append({"url": full_url, "title": title, "tid": tid})
    return items


def get_detail(url, title_from_list):
    tid = extract_thread_id(url)
    if not tid:
        return "", "", [], title_from_list

    mobile_url = MOBILE_URL % tid
    html = fetch(mobile_url)
    if not html:
        return "", "", [], title_from_list

    soup = BeautifulSoup(html, "html.parser")

    full_title = title_from_list
    pt = soup.select_one("title")
    if pt:
        t = pt.get_text(strip=True)
        t = re.sub(r"^\[.*?\]\s*", "", t)
        t = re.sub(r"\s*[-–—]\s*香山网.*$", "", t)
        if t:
            full_title = t

    pub_date = ""
    m = re.search(r"发表于\s*(\d{1,2})-(\d{1,2})", html)
    if m:
        import datetime
        now = datetime.datetime.now()
        pub_date = "%d-%02d-%02d" % (now.year, int(m.group(1)), int(m.group(2)))
    elif re.search(r"(\d{4}[-/]\d{1,2}[-/]\d{1,2})", html):
        m2 = re.search(r"(\d{4}[-/]\d{1,2}[-/]\d{1,2})", html)
        if m2:
            pub_date = m2.group(1).replace("/", "-")
            parts = pub_date.split("-")
            if len(parts) == 3:
                pub_date = "%s-%02d-%02d" % (parts[0], int(parts[1]), int(parts[2]))

    body = soup.select_one("div[class*=content]")
    if not body:
        body = soup.select_one("td.t_f")
    content_html = ""
    attachments = []

    if body:
        for a_tag in body.find_all("a", href=True):
            h = a_tag["href"].strip()
            txt = a_tag.get_text(strip=True)
            if re.search(r"\.(docx?|pdf|pptx?|xlsx?|jpg|png|gif)($|\?)", h, re.IGNORECASE) or "/download/" in h:
                fname = txt or h.split("/")[-1]
                attach_url = h if h.startswith("http") else DOMAIN + h
                attachments.append({"url": attach_url, "name": fname})
                a_tag.replace_with("[%s](%s)" % (fname, attach_url))

        tbl_soup = BeautifulSoup(str(body), "html.parser")
        tables = tbl_soup.find_all("table")
        table_markers = []
        for i, tb in enumerate(tables):
            key = "___TBL_%d___" % i
            table_markers.append((key, str(tb)))
            tb.replace_with(key)

        body_html = str(tbl_soup)
        body_soup = BeautifulSoup(body_html, "html.parser")

        for tag in body_soup.find_all(["script", "style"]):
            tag.decompose()
        for tag in body_soup.find_all(["span", "font", "b", "strong", "i", "em", "u"]):
            tag.unwrap()
        for p in body_soup.find_all("p"):
            p.insert(0, "\n\n")
            p.unwrap()
        for d in body_soup.find_all("div"):
            d.insert(0, "\n")
            d.unwrap()
        for br in body_soup.find_all("br"):
            br.replace_with("\n")

        content_html = body_soup.get_text(separator="", strip=False)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html).strip()

        for key, tbl_html in table_markers:
            content_html = content_html.replace(key, "\n\n" + tbl_html + "\n\n")

    if not content_html or len(content_html.strip()) < 20:
        content_html = '<p><a href="%s">%s</a></p>' % (url, full_title)

    return content_html, pub_date, attachments, full_title


def main():
    import argparse
    parser = argparse.ArgumentParser(description="香山网-今日珠海爬虫")
    parser.add_argument("--pages", type=int, default=None, help="爬取页数")
    parser.add_argument("--full", action="store_true", help="全量爬取全部页")
    args = parser.parse_args()

    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()

    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    existing = c.fetchone()[0]

    if args.full:
        max_pages = MAX_PAGES
    elif args.pages is not None:
        max_pages = args.pages
    else:
        max_pages = DEFAULT_PAGES

    print("站点=%s: 已有 %d 条, 爬取 %d 页" % (SITE_NAME, existing, max_pages), file=sys.stderr)

    all_new = 0
    for pg in range(1, max_pages + 1):
        page_url = FORUM_URL % pg
        html = fetch(page_url)
        if not html:
            print("  第%d页 获取失败" % pg, file=sys.stderr)
            continue

        items = parse_list_items(html)
        if not items:
            print("  第%d页 无条目" % pg, file=sys.stderr)
            break

        print("  第%d页: %d 条" % (pg, len(items)), file=sys.stderr)

        for it in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (it["url"], SITE_NAME))
            if c.fetchone():
                continue

            content, pub_date, attachments, full_title = get_detail(it["url"], it["title"])
            date = pub_date or ""

            try:
                c.execute(
                    """INSERT INTO gov_raw (site_name, page_url, title, content, publish_date, source_url, status, attachments)
                       VALUES (?,?,?,?,?,?,?,?)""",
                    (SITE_NAME, it["url"], full_title[:500], content, date[:20],
                     it["url"], "synced", json.dumps(attachments, ensure_ascii=False))
                )
                all_new += 1
                print("    + %s (%s) body:%dB attach:%d" % (full_title[:50], date, len(content), len(attachments)), file=sys.stderr)
            except Exception as e:
                print("    ! 入库失败: %s" % e, file=sys.stderr)

        time.sleep(0.3)

    conn.commit()
    print("\n完成! 新增 %d 条" % all_new, file=sys.stderr)

    if all_new > 0:
        c.execute("""INSERT OR IGNORE INTO gov_raw_fts(rowid, title, content, publish_date, site_name, source_url)
                      SELECT rowid, title, content, publish_date, site_name, source_url FROM gov_raw
                      WHERE site_name=? """, (SITE_NAME,))
        fts_added = c.execute("SELECT changes()").fetchone()[0]
        conn.commit()
        print("  FTS 增量更新: %d 条" % fts_added, file=sys.stderr)

    conn.close()


if __name__ == "__main__":
    main()
