#!/usr/bin/env python3
"""月湖区-公告公示爬虫
站点: http://www.yuehu.gov.cn
栏目: 公告公示 (col3794)
分页: JPage POST /module/web/jpage/dataproxy.jsp, 45条/页
CMS: Hanweb JPage
"""

import os, re, sys, time, subprocess
from bs4 import BeautifulSoup
import requests

DB_PATH = os.getenv("SEARCH_DB", "/mnt/data/search.db")
BASE_URL = "http://www.yuehu.gov.cn"
LIST_URL = f"{BASE_URL}/col/col3794/index.html"
API_URL = f"{BASE_URL}/module/web/jpage/dataproxy.jsp"
JPPARAMS = {
    "webid": "34",
    "path": "http://www.yuehu.gov.cn/",
    "columnid": "3794",
    "unitid": "35403",
    "webname": "月湖区人民政府网站",
    "permissiontype": "0",
}
SITE_NAME = "月湖区-公告公示"
INDUSTRY = "政府公告"
GROUP = "江西"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def fetch(url, retries=3, post_data=None):
    for i in range(retries):
        try:
            if post_data:
                r = requests.post(url, data=post_data, headers=HEADERS, timeout=30)
            else:
                r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if i < retries - 1:
                time.sleep(2)
            else:
                print(f"  [ERROR] {url}: {e}", file=sys.stderr)
                return None


def parse_records(html):
    """从HTML/API响应中提取记录"""
    items = []
    # Find <record> CDATA sections
    records = re.findall(r"<record>(.*?)</record>", html, re.DOTALL)
    for rec in records:
        # Extract href and title
        m = re.search(r'href="([^"]*)"[^>]*>([^<]*)</a>', rec)
        if not m:
            continue
        href = m.group(1).strip()
        title = m.group(2).strip()
        if not href or not title or "art_" not in href:
            continue
        if not href.startswith("http"):
            href = BASE_URL + href
        # Fix trailing dot
        href = href.rstrip(".")
        if not href.endswith(".html"):
            href += ".html"

        # Extract date from record
        date_str = ""
        dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", rec)
        if dm:
            date_str = dm.group(1)

        items.append({
            "page_url": href,
            "title": title,
            "date": date_str,
        })
    return items


def parse_detail(html, url):
    """解析详情页"""
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    publish_date = ""
    content = ""

    # Title
    title_tag = soup.find("title")
    if title_tag:
        t = title_tag.get_text(strip=True)
        # Format: 月湖区人民政府网站 公告公示 TITLE
        for prefix in ["月湖区人民政府网站 公告公示 ", "月湖区人民政府网站 公告公示"]:
            if t.startswith(prefix):
                t = t[len(prefix):]
                break
        if t:
            title = t

    # Publish date
    body_text = soup.get_text()
    dm = re.search(r"发布时间[：:]?\s*(\d{4}-\d{1,2}-\d{1,2})", body_text)
    if dm:
        publish_date = dm.group(1)

    # Content - 按<p>分段，避免get_text(strip=True)丢失换行
    content_div = soup.select_one("div.bfr_article_content")
    caption = content_div.select_one("div.caption") if content_div else None
    if not caption:
        caption = soup.select_one("div.caption")
    if caption:
        paragraphs = []
        for p in caption.find_all("p", recursive=True):
            if p.find_parent("table"):
                continue
            text = p.get_text(strip=True)
            if text:
                paragraphs.append(text)
        content = "\n\n".join(paragraphs) if paragraphs else ""

    return title, publish_date, content


def save_to_db(page_url, title, publish_date, content):
    def esc(s):
        return s.replace("'", "''") if s else ""
    check_sql = f"SELECT rowid FROM gov_raw WHERE page_url = '{esc(page_url)}'"
    result = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, check_sql], capture_output=True, text=True, timeout=10)
    if result.stdout.strip():
        return "dup"
    summary = content[:200] if content else ""
    sql = f"""INSERT INTO gov_raw (page_url, title, publish_date, content, site_name, industry, summary)
VALUES ('{esc(page_url)}', '{esc(title)}', '{esc(publish_date)}', '{esc(content)}', '{esc(SITE_NAME)}', '{INDUSTRY}', '{esc(summary)}')"""
    result = subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, text=True, timeout=10)
    if result.returncode != 0 and "UNIQUE" not in result.stderr:
        print(f"  [DB ERROR] {result.stderr}", file=sys.stderr)
        return "error"
    sync_sql = f"""INSERT INTO gov_search (rowid, title, site_name, summary)
SELECT rowid, title, site_name, summary FROM gov_raw
WHERE page_url = '{esc(page_url)}' AND rowid NOT IN (SELECT rowid FROM gov_search)"""
    subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sync_sql], capture_output=True, text=True, timeout=10)
    return "new"


def crawl():
    max_pages_str = sys.argv[1] if len(sys.argv) > 1 else "3"
    try:
        max_pages = int(max_pages_str)
    except ValueError:
        max_pages = 3

    print(f"[{SITE_NAME}] Starting crawl, max_pages={max_pages}", flush=True)
    all_items = []

    # Page 1 from HTML
    html = fetch(LIST_URL)
    if html:
        items = parse_records(html)
        print(f"  Page 1 (HTML): {len(items)} items", flush=True)
        all_items.extend(items)

    # Page 2+ via JPage API (returns all remaining)
    for page in range(2, max_pages + 1):
        post_data = dict(JPPARAMS, page=str(page))
        api_html = fetch(API_URL, post_data=post_data)
        if not api_html:
            break
        items = parse_records(api_html)
        if not items:
            break
        # Dedup: check if items already in all_items (by URL)
        existing_urls = {i["page_url"] for i in all_items}
        new_items = [i for i in items if i["page_url"] not in existing_urls]
        print(f"  Page {page} (API): {len(items)} items, {len(new_items)} new", flush=True)
        if not new_items:
            break
        all_items.extend(new_items)
        if len(items) < 30:
            break

    print(f"Total list items: {len(all_items)}", flush=True)

    new_count = dup_count = error_count = 0
    for item in all_items:
        print(f"  Fetching: {item['title'][:40]}...", flush=True)
        detail_html = fetch(item["page_url"])
        if not detail_html:
            error_count += 1
            continue
        title, publish_date, content = parse_detail(detail_html, item["page_url"])
        if not title:
            title = item.get("title", "")
        if not publish_date:
            publish_date = item.get("date", "")
        if not title:
            error_count += 1
            continue
        try:
            result = save_to_db(item["page_url"], title, publish_date, content)
            if result == "new":
                new_count += 1
            elif result == "dup":
                dup_count += 1
            else:
                error_count += 1
        except Exception as e:
            print(f"  [ERROR] DB: {e}", file=sys.stderr)
            error_count += 1

    print(f"\n=== {SITE_NAME} Done ===", flush=True)
    print(f"New: {new_count}, Dup: {dup_count}, Error: {error_count}", flush=True)


if __name__ == "__main__":
    crawl()
