#!/usr/bin/env python3
"""
广昌县人民政府 - 部门公告
URL: http://www.jxgc.gov.cn/col/col6875/index.html
CMS: TRS WCM (jpage dataproxy POST 分页)
List: HTML第1页内嵌XML datastore (45条/页)
      补充数据从 dataproxy.jsp?page=1 POST 获取 (301条, 去重后约可覆盖5页)
Detail: div#zoom (表格HTML保留，段落提取文本)
"""
import sys, os, re, json, time, sqlite3, html
from datetime import datetime, timedelta
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

# ── 配置 ──
BASE = "http://www.jxgc.gov.cn"
LIST_URL = BASE + "/col/col6875/index.html"
PROXY_URL = BASE + "/module/web/jpage/dataproxy.jsp"
SITE = "广昌县人民政府"
COLUMN = "部门公告"
PROVINCE = "江西"
TOTAL_PAGES = 5  # 最多5页，约225条

# DB
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

PROXY_PARAMS = {
    "page": 1, "appid": 1, "webid": 9, "path": "/",
    "columnid": 6875, "unitid": 69431,
}

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
session = requests.Session()
session.headers.update(HEADERS)


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def strip_html(text):
    """Convert HTML to clean plain text for FTS summary."""
    if not text:
        return ""
    text = html.unescape(text)
    text = re.sub(r'</?(?:p|div|h[1-6]|li|tr|blockquote|section|article|table|br\s*/?)[^>]*>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'<[^>]+>', '', text)
    text = re.sub(r'[ \t]+', ' ', text)
    text = re.sub(r'\n{3,}', '\n\n', text)
    return text.strip()


def import_to_db(record):
    """将单条记录写入 search.db (gov_raw + gov_search FTS)"""
    try:
        db = sqlite3.connect(DB_PATH, timeout=10)
        title = (record.get("title") or "")[:500]
        page_url = (record.get("page_url") or "")[:1000]
        content = strip_html(record.get("content") or "")
        publish_date = (record.get("publish_date") or "")[:20]
        site_name = (record.get("site_name") or "unknown")[:100]
        attachments_str = json.dumps(record.get("attachments") or [], ensure_ascii=False)

        # 删除旧FTS关联（如果page_url已存在）
        old_rowids = db.execute(
            "SELECT rowid FROM gov_raw WHERE page_url = ?", (page_url,)
        ).fetchall()
        for (rid,) in old_rowids:
            db.execute("DELETE FROM gov_search WHERE rowid = ?", (rid,))

        # INSERT OR REPLACE
        db.execute(
            "INSERT OR REPLACE INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?, ?, ?, ?, ?, ?, 'synced', ?)",
            (title, page_url, content, publish_date, site_name, page_url, attachments_str)
        )

        # FTS增量同步
        new_rowid = db.execute("SELECT last_insert_rowid()").fetchone()[0]
        summary = content[:500] if content else title[:500]
        db.execute(
            "INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
            (new_rowid, title, site_name, summary)
        )
        db.commit()
        db.close()
        return True
    except Exception as e:
        log(f"  [ERR] DB import failed: {e}")
        return False


def fetch(url, post_data=None):
    """请求政府站（政府站响应可能较慢，使用较长超时）"""
    try:
        if post_data:
            r = session.post(url, data=post_data, timeout=(10, 30))
        else:
            r = session.get(url, timeout=(10, 30))
        r.encoding = "utf-8"
        return r.text if r.status_code == 200 else None
    except Exception as e:
        log(f"  [ERR] fetch failed: {e}")
        return None


def parse_list_items(cdata_text):
    """从CDATA片段解析列表项：url, title(date完整不截断), date"""
    items = []
    records = re.findall(r"<record><!\[CDATA\[(.*?)\]\]></record>", cdata_text, re.DOTALL)
    for cdata in records:
        a_match = re.search(r"<a\s+href='([^']+)'\s*title='([^']*)'", cdata)
        span_match = re.search(r'class="bt-data-time"[^>]*>([^<]+)</span>', cdata)
        if a_match and span_match:
            href = a_match.group(1).strip()
            title = a_match.group(2).strip()
            date_str = span_match.group(1).strip()
            full_url = urljoin(BASE, href)
            items.append({"url": full_url, "title": title, "date": date_str})
    return items


def get_page1_items():
    """从HTML第1页内嵌datastore获取列表"""
    html = fetch(LIST_URL)
    if not html:
        return []
    m = re.search(r"<datastore>(.*?)</datastore>", html, re.DOTALL)
    if not m:
        return []
    return parse_list_items(m.group(1))


def get_proxy_items():
    """从dataproxy page=1获取补充数据"""
    xml_text = fetch(PROXY_URL, post_data=dict(PROXY_PARAMS))
    if not xml_text:
        return []
    return parse_list_items(xml_text)


def extract_detail(detail_url, list_title):
    """提取详情页：完整标题(meta ArticleTitle)、正文(表格HTML保留)、附件、日期"""
    html = fetch(detail_url)
    if not html:
        return None, [], list_title

    soup = BeautifulSoup(html, "html.parser")

    # 标题：优先 meta ArticleTitle（源站的完整标题，非列表页截断版）
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        full_title = meta_title["content"].strip()
    else:
        title_tag = soup.find("title")
        full_title = title_tag.get_text(strip=True) if title_tag else list_title
        full_title = re.sub(r'^广昌县人民政府\s*\S*\s*', '', full_title)

    if not full_title or full_title == list_title:
        full_title = list_title

    # 日期：meta PubDate
    date_str = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        date_str = meta_date["content"].strip()[:10]

    # 附件（含doc/docx/pdf等链接）
    attachments = []
    for a_tag in soup.find_all("a", href=re.compile(r'\.(doc|docx|pdf|xls|xlsx|zip|rar)$', re.I)):
        href = a_tag.get("href", "").strip()
        if href:
            name = a_tag.get_text(strip=True) or "附件"
            attachments.append({"name": name, "url": urljoin(detail_url, href)})

    # 正文：div#zoom
    zoom = soup.find("div", id="zoom")
    if not zoom:
        zoom = soup.find("div", class_=re.compile(r"bt-article-02|bfr_article_content"))
    if not zoom:
        return None, attachments, full_title

    # 移除无用元素
    for tag in zoom.find_all(["script", "style", "meta"]):
        tag.decompose()

    parts = []
    for child in zoom.children:
        if child.name is None:
            continue
        tag_name = child.name.lower()

        if tag_name == "table":
            # 表格HTML原样保留，仅unwrap span/font（保留表格结构）
            tbl_soup = BeautifulSoup(str(child), "html.parser")
            for sp in tbl_soup.find_all(["span", "font"]):
                sp.unwrap()
            parts.append(str(tbl_soup))
        elif tag_name == "p":
            inner_tables = child.find_all("table")
            if inner_tables:
                for t in inner_tables:
                    tbl_soup = BeautifulSoup(str(t), "html.parser")
                    for sp in tbl_soup.find_all(["span", "font"]):
                        sp.unwrap()
                    parts.append(str(tbl_soup))
            else:
                text = child.get_text(separator="", strip=True)
                if text:
                    parts.append(text)
        elif tag_name == "div":
            inner_tables = child.find_all("table", recursive=False)
            if inner_tables:
                for t in inner_tables:
                    tbl_soup = BeautifulSoup(str(t), "html.parser")
                    for sp in tbl_soup.find_all(["span", "font"]):
                        sp.unwrap()
                    parts.append(str(tbl_soup))
            else:
                text = child.get_text(separator="", strip=True)
                if text and len(text) > 20:
                    parts.append(text)
        elif tag_name in ("h1", "h2", "h3", "h4", "h5"):
            text = child.get_text(separator="", strip=True)
            if text:
                parts.append(text)

    content = "\n\n".join(parts).strip()

    # 空内容回退
    if len(content) < 20 and attachments:
        content = f"[{full_title}]({detail_url})"
        for att in attachments:
            content += f"\n[{att['name']}]({att['url']})"
    elif content and attachments:
        content += "\n\n**附件：**"
        for att in attachments:
            content += f"\n[{att['name']}]({att['url']})"

    return content, attachments, full_title


def output_item(item, idx):
    record = {
        "title": item.get("title", ""),
        "page_url": item.get("url", ""),
        "publish_date": item.get("date", ""),
        "content": item.get("content", ""),
        "attachments": item.get("attachments", []),
        "site_name": f"{SITE}-{COLUMN}",
        "column": COLUMN,
        "province": PROVINCE,
    }
    # 输出 JSONL（兼容管道）
    print(json.dumps(record, ensure_ascii=False))
    # 直写 search.db + FTS
    ok = import_to_db(record)
    return ok


def crawl_all(label="incremental"):
    """爬取：
    - incremental: 只爬第1页（45条）
    - full: 爬5页（从第1页+补充数据，去重）
    """
    if label == "incremental":
        log(f"\n{'='*50}")
        log(f"🏠 {SITE} - {COLUMN}")
        log(f"📄 增量（第1页）")
        log(f"{'='*50}")
        cutoff = datetime.now() - timedelta(hours=48)

        items = get_page1_items()
        log(f"第1页: {len(items)}条")
    else:
        log(f"\n{'='*50}")
        log(f"🏠 {SITE} - {COLUMN}")
        log(f"📄 全量前{TOTAL_PAGES}页")
        log(f"{'='*50}")
        cutoff = datetime.now() - timedelta(days=365 * 3)

        # 第1页：HTML datastore
        page1 = get_page1_items()
        log(f"第1页 (HTML): {len(page1)}条")

        # 补充数据：从dataproxy page=1获取，去重
        proxy_items = get_proxy_items()
        proxy_urls = {i["url"] for i in proxy_items}
        page1_urls = {i["url"] for i in page1}
        extra = [i for i in proxy_items if i["url"] not in page1_urls]
        # 取前4页的量（按时间倒序，extra已经是时间倒序）
        extra_per_page = len(page1)  # 每页约45条
        extra_limit = extra_per_page * (TOTAL_PAGES - 1)
        extra = extra[:extra_limit]
        log(f"补充 (dataproxy去重, {TOTAL_PAGES-1}页): {len(extra)}条")

        items = page1 + extra
        log(f"合计: {len(items)}条")

    total = len(items)
    ok_count = 0
    for idx, item in enumerate(items, 1):
        # 时间过滤
        try:
            item_date = datetime.strptime(item["date"], "%Y-%m-%d")
            if item_date < cutoff:
                continue
        except (ValueError, KeyError):
            pass

        content, attachments, full_title = extract_detail(item["url"], item["title"])
        if content is None:
            log(f"  [{idx}/{total}] ⏭️ 详情空: {item['title'][:40]}")
            continue
        item["title"] = full_title
        item["content"] = content
        item["attachments"] = attachments
        if output_item(item, idx):
            ok_count += 1
        if idx % 20 == 0:
            log(f"  [PROGRESS] {idx}/{total}")
        time.sleep(0.3)

    log(f"\n[DONE] 共处理 {ok_count} 条")
    return ok_count


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    crawl_all(mode)
